From 22afe53066be28265f04030cc51f5d19d5f07fba Mon Sep 17 00:00:00 2001 From: Mikhail Filimonov Date: Thu, 1 Oct 2026 21:47:26 +0200 Subject: [PATCH] CAS GC: batch the namespace janitor's deletes under a soft time budget The namespace janitor (GC phase 16) deleted dead `_log`/`_snap` keys one page per folding round: a dropped table with 20k parts took 16 rounds and 16k DELETE requests to drain. It now deletes a page's dead keys in batches sized by the store's batch-delete limit (`objects_chunk_size_to_delete` on S3, 1 elsewhere), runs the batches as jobs on the GC I/O pool (the pool's `cas_gc_io_concurrency` threads bound them, the same pattern as the pending-deletes fan-out) while the next page is listed, and keeps taking pages until a soft 20 s budget: no new page after it, in-flight jobs complete. A pass that began mid-stream wraps once to the stream start. Rules: - Dead `_log`/`_snap` keys go in write-once cohorts with no per-key precondition and no per-key fallback; a batch that still fails after retries is leaked and the cursor advances. - A page is held (cursor not advanced, nothing published) only on lost authority, a capability refusal (`NOT_IMPLEMENTED`), a schedule refusal or a local failure. The capability is remembered per disk. - An all-live page ends the pass, so a quiet pool still costs one LIST per round. - The CAS path honours the storage's batch limit; it used to ignore `objects_chunk_size_to_delete`. Also: `IObjectStorage::batchDeleteKeyLimit`, the S3 adapter keeps the error name on a refused bulk delete, `ThreadName::CAS_GC_JANITOR`, user docs for phases 15-17 and `system.cas_gc_log`. Testing: `CAS*` gate 2605; new integration test `tests/integration/test_cas_janitor_drain` (RustFS; batch, 100-key chunks, no batch delete) fails on the base commit and passes here; A/B on the local RustFS stand (docs/superpowers/reports/2026-10-01-cas-29-9-janitor-ab): 20k keys drain in 1 janitor round instead of 16, 31 s instead of 1307 s, 52 DELETE requests instead of 16k, 160 LIST instead of 3134; on a store without batch delete at 20 ms RTT: 2 rounds instead of 1000 keys per round. Co-Authored-By: Claude Fable 5.1 Signed-off-by: Mikhail Filimonov --- .../cas/architecture/garbage-collection.md | 108 ++- docs/en/antalya/cas/configuration.md | 4 +- .../en/operations/system-tables/cas_gc_log.md | 6 +- src/Common/ProfileEvents.cpp | 2 +- src/Common/setThreadName.h | 1 + .../ContentAddressed/Backend/CasBackend.h | 4 + .../Backend/CasInMemoryBackend.h | 5 + .../Backend/CasInstrumentedBackend.h | 1 + .../Backend/CasObjectStorageBackend.cpp | 11 +- .../Backend/CasObjectStorageBackend.h | 4 + .../Backend/CasThrottlingBackend.h | 1 + .../ContentAddressedSettings.cpp | 4 +- .../ContentAddressed/Formats/CasLayout.h | 22 +- .../ContentAddressed/Gc/CasGc.cpp | 245 +++--- .../ContentAddressed/Gc/CasGc.h | 54 +- .../Gc/CasNamespaceJanitor.cpp | 498 +++++++++--- .../ContentAddressed/Gc/CasNamespaceJanitor.h | 67 +- .../ContentAddressed/Tools/CasFsck.cpp | 2 +- .../ObjectStorages/IObjectStorage.h | 5 + .../ObjectStorages/S3/S3ObjectStorage.cpp | 33 +- .../ObjectStorages/S3/S3ObjectStorage.h | 6 +- .../cas_namespace_janitor_test_helpers.h | 247 ++++++ src/Disks/tests/cas_scripted_s3_server.h | 296 ++++++++ src/Disks/tests/cas_test_helpers.h | 101 +++ .../tests/gtest_cas_bulk_delete_backend.cpp | 46 ++ .../gtest_cas_gc_bulk_delete_fallback.cpp | 161 ++-- .../tests/gtest_cas_gc_frontier_gate.cpp | 19 +- .../gtest_cas_gc_manifest_bulk_delete.cpp | 103 ++- src/Disks/tests/gtest_cas_gc_round_defer.cpp | 10 +- .../tests/gtest_cas_namespace_janitor.cpp | 86 +-- .../gtest_cas_namespace_janitor_batches.cpp | 707 ++++++++++++++++++ .../gtest_cas_namespace_janitor_pipeline.cpp | 321 ++++++++ .../tests/gtest_cas_namespace_janitor_s3.cpp | 203 +++++ src/Disks/tests/gtest_cas_ref_gc.cpp | 144 ++-- .../gtest_cas_s3_bulk_delete_fallback.cpp | 314 ++++---- src/Disks/tests/gtest_cas_write_once_key.cpp | 22 + src/IO/S3/deleteFileFromS3.cpp | 6 +- .../test_cas_janitor_drain/__init__.py | 0 .../configs/storage_conf.xml | 50 ++ .../test_cas_janitor_drain/test.py | 106 +++ 40 files changed, 3296 insertions(+), 729 deletions(-) create mode 100644 src/Disks/tests/cas_namespace_janitor_test_helpers.h create mode 100644 src/Disks/tests/cas_scripted_s3_server.h create mode 100644 src/Disks/tests/gtest_cas_namespace_janitor_batches.cpp create mode 100644 src/Disks/tests/gtest_cas_namespace_janitor_pipeline.cpp create mode 100644 src/Disks/tests/gtest_cas_namespace_janitor_s3.cpp create mode 100644 tests/integration/test_cas_janitor_drain/__init__.py create mode 100644 tests/integration/test_cas_janitor_drain/configs/storage_conf.xml create mode 100644 tests/integration/test_cas_janitor_drain/test.py diff --git a/docs/en/antalya/cas/architecture/garbage-collection.md b/docs/en/antalya/cas/architecture/garbage-collection.md index 21953307f9d1..1d575ba2f17b 100644 --- a/docs/en/antalya/cas/architecture/garbage-collection.md +++ b/docs/en/antalya/cas/architecture/garbage-collection.md @@ -57,15 +57,15 @@ follower or a deferred round execution returns before that commit. | 13 | `round_commit` | fold | Retention-prune old generations, then publish the single `gc/state` `CAS` that adopts the whole round | | 14 | `handoff_reclaim` | post-`CAS` | Reclaim a generation a ref moved off during this round, which the ordinary retention prune already skipped and will not revisit | | 15 | `manifest_deletes` | post-`CAS` | Delete manifest bodies whose owner-removal minus-one edge the `CAS` in phase 13 just adopted | -| 16 | `namespace_cleanup` | leader; suppressed on `DEFER` | One bounded page of the perpetual namespace janitor, reclaiming dead-life debris | +| 16 | `namespace_cleanup` | leader; one suppressed page on `DEFER` | The perpetual namespace janitor: pages of dead-life debris under a 20 s soft budget | | 17 | `ref_object_cleanup` | post-`CAS` | Prune ref logs and snapshots once both fold coverage and a live snapshot make them safe to delete | | 18 | `orphan_sweep` | post-`CAS` | Exact-token deletion for the [orphan-manifest sweep](/antalya/cas/architecture/manifests-and-refs#orphan-sweep), after phase 13 adopted each candidate's blob-source retirements and the cursor | Phases 2 through 4 run on every leader round; a follower returns after phase 1. Phases 5–15 and 17–18 run only when phase 4 decides to fold; phase 16 runs after phase 15 on a fold and right after -phase 4 on a `DEFER`. A `DEFER` verdict is therefore not a bare no-op: it still runs one bounded -namespace-janitor page with `suppress_destructive = true` — listing and classification only, no -deletes and no cursor advance — and then returns, publishing no fold artifact and no commit `CAS`. +phase 4 on a `DEFER`. A folding round runs janitor pages until the phase budget ends; a `DEFER` +verdict is therefore not a bare no-op: it still runs one namespace-janitor page with `suppress_destructive = true` — it lists and parses the keys but never +classifies them against the catalog, deletes nothing and does not advance the cursor — and then returns, publishing no fold artifact and no commit `CAS`. Its lease `CAS` may already have created or renewed the lease in phase 1: ```mermaid @@ -229,7 +229,7 @@ if any of: changed rows ≥ `gc_fold_threshold` (default 1); an adopted shard ha delete; an adopted shard has a condemned blob due to graduate (`oldest_nonpending_condemn_round < round + 1`); or `gc_fold_max_defer_rounds` (default 8) consecutive defers were reached. Both thresholds are internal `PoolConfig` fields, not disk -settings. On `defer`, one suppressed namespace-janitor page runs (phase 16's work) and the round +settings. On `defer`, one suppressed namespace-janitor page runs (phase 16's work, a single page where a folding round runs several) and the round returns without a commit. On `fold`, phase 6 reuses this plan and the same `LIST`. ## Phase 5 — parent seal read {#phase-5-parent-seal-read} @@ -500,7 +500,8 @@ Deletes owner-removed manifest bodies, now that phase 13's `CAS` adopted their m - **Reads / writes:** batch `DELETE` of the manifest keys collected by phase 8's fold of `-1` owner edges, in chunks of `cas_gc_bulk_delete_chunk_keys` (default 1000, the backend maximum). A manifest key is write-once, so the delete carries no per-key precondition; an absent key is simply gone. A - backend without a batch-delete verb (GCS) falls back to one admitted `DELETE` per key. + storage that rejects a batch delete (`NOT_IMPLEMENTED`) stops the family for the round; the remaining + bodies are left to the orphan-manifest sweep (phase 18) and later requests carry one key. - **Safety:** each body is unreachable from any live ref (its owner-removal was folded and committed) and is never re-derived — the intake cursor that found the `-1` edge is now committed, so a folded log is never revisited. Hence the phase is unbudgeted by design and drains the whole @@ -510,34 +511,56 @@ Deletes owner-removed manifest bodies, now that phase 13's `CAS` adopted their m all-or-nothing per request: the chunks before the failing one are recorded, the failing chunk's keys are not, and a key one of its attempts did delete shows up as already gone in the next fold - **Observability:** phase row `manifest_deletes`; metrics `attempted`, `accepted` (keys recorded - as deleted or absent), `requests` (one per chunk, or the failed bulk call plus one per key on the - fallback), `suppressed`; one `ManifestDelete` row per key in `system.cas_log` + as deleted or absent), `requests` (one per chunk), `unsent` (bodies not deleted this round), + `capability_learned` (1 when the storage rejected a batch delete), `suppressed`; one `ManifestDelete` row per key in `system.cas_log` -Only a crash, `suppress_destructive` or a chunk that exhausted its retries leaves an entry — it is -then picked up by the orphan-manifest sweep (phase 18). +Only a crash, `suppress_destructive`, a rejected batch delete or a chunk that exhausted its retries +leaves an entry — it is then picked up by the orphan-manifest sweep (phase 18). ## Phase 16 — namespace cleanup {#phase-16-namespace-cleanup} -One bounded page of the perpetual namespace janitor: deletes the physical objects of namespace lives -no longer in the catalog (dead-life debris). - -- **Runs on:** fold path here; also on the deferred path right after phase 4 with - `suppress_destructive` forced on -- **Reads:** the durable `janitor_cursor`; one `LIST` page (≤ 1000 keys) of `cas/ns/`; a fresh - ref-catalog snapshot; `gc/state` per fence re-check -- **Writes / deletes:** exact-token `DELETE` per dead-life `_log` / `_snap` / `_ckpt` / `_files` - object; one `CAS` on the maintenance state when the page is decided -- **Safety:** each delete is under a GC fence re-check (`lease.owner` / `lease.seq`) before it and - once at the end; the incarnation segment in every key makes an old life's objects structurally - unreachable from a reborn same-name namespace, so a missed key can only leak storage, never expose - it -- **Fails the round if:** nothing — the whole page is wrapped in a catch-all ("namespace janitor - skipped this round") -- **Observability:** phase row `namespace_cleanup`; metrics `janitor_pages`, `janitor_keys`, - `janitor_deleted`, `leaked` - -The cursor advances only when the whole page was decided under a held fence and an unambiguous -catalog; under suppression it lists and classifies but deletes nothing and does not advance. +The perpetual namespace janitor deletes the physical objects of namespace lives no longer in the catalog +(dead-life debris), page by page from the durable `janitor_cursor`. + +- **Runs on:** the fold path, pages until the budget ends; on the deferred path, one page right after + phase 4 with `suppress_destructive` forced on +- **Per page (round thread):** a `gc/state` read for authority, one `LIST` page (≤ 1000 keys) of `cas/ns/`, + one ref-catalog read after the `LIST`, exact-token `DELETE`s of dead `_ckpt` and `_files` objects +- **Dead `_log` / `_snap` (GC I/O pool):** one batch delete per job of up to `cas_gc_bulk_delete_chunk_keys` + keys, capped by the storage's batch-delete limit (the disk key `objects_chunk_size_to_delete` on S3, 1 on other native object storages, 1000 in the emulated mode). + With a limit of 1, jobs are per-key. A page's jobs are all enqueued on the GC I/O pool, whose + `cas_gc_io_concurrency` threads bound how many run at once; the next page is listed while they run, and + they finish before that page is used +- **Budget:** 20 s, soft. No page starts after it; the page in progress and the jobs in flight finish +- **Wrap:** a pass that began mid-stream continues once from the start of the stream after the last page. It + skips keys past the start cursor, which the pass already handled, and ends at the first page that reaches + the start cursor +- **Stops early when:** a page had no dead-life debris, the pass ended, authority was lost, a job held + its page, a later page's read failed, or a page was left undecided (ambiguous catalog, suppression, or + an exact delete that lost admission) +- **Writes:** one `CAS` on the maintenance state at the end, with the cursor after the last complete page + (empty once the pass has wrapped); nothing under suppression +- **Failed jobs:** a batch that fails after its retries is leaked: its keys are retried on the next pass + and the page advances. Lost authority, a refused batch delete (`NOT_IMPLEMENTED`) or a local failure + holds the page: the cursor stops before it and nothing is published for it. The batch size is fixed + per phase, so a refusal ends the phase; the capability is remembered per disk and later rounds use + one-key jobs +- **Safety:** authority is re-read once per page and each job samples the result before its request, so no job + sends after the round has observed the loss of authority (a job already past its first request completes); + every deleted key belongs to a life absent from a catalog cut taken after its page's `LIST`. The incarnation segment in + every key makes an old life's objects unreachable from a re-created same-name table, so a missed key + can only leak storage, never expose it +- **Fails the round if:** nothing; the phase is wrapped in a catch-all ("namespace janitor stopped this + round") +- **Observability:** phase row `namespace_cleanup`, see + [`system.cas_gc_log`](/operations/system-tables/cas_gc_log#per-phase-rows); event + `CASGCNamespaceCleanupLeaks` + +### Unversioned buckets {#phase-16-unversioned-buckets} + +A batch delete without a token assumes an unversioned bucket, which the mount probe checks. If versioning +is turned on after mount, the batch reports success while noncurrent versions remain: an empty `LIST` +then proves the namespace drained, not that storage was reclaimed. ## Phase 17 — ref object cleanup {#phase-17-ref-object-cleanup} @@ -551,8 +574,9 @@ from the catalog. `gc/state` (authority re-validation). No `HEAD`: `_log` / `_snap` keys are write-once, there is nothing to re-observe - **Writes / deletes:** batch `DELETE` of the planned `_log` / `_snap` keys in chunks of - `cas_gc_bulk_delete_chunk_keys` (one admitted `DELETE` per key on a backend without batch - delete); the checkpoint-named snapshot is always retained + `cas_gc_bulk_delete_chunk_keys`; a storage that rejects a batch + delete stops the pass for the round, and the same candidates are recomputed next round. The + checkpoint-named snapshot is always retained - **Safety:** before each chunk, re-validates: ref-catalog token still equals the fold's catalog cut, same row and life, unchanged GC fence. The first failure stops the whole pass. The per-round `cas_gc_round_ref_cleanup_budget` cap counts objects and cuts a chunk to what remains; @@ -563,7 +587,7 @@ from the catalog. namespace, but a chunk delete that exhausts its retry policy propagates (see [post-commit failures](#post-commit-failures)) - **Observability:** phase row `ref_object_cleanup`; metrics `namespaces_planned`, `suppressed`, - `trim_enabled`; `ProfileEvent` `CASRefCleanupObjectsDeleted` + `trim_enabled`, `capability_learned`; `ProfileEvent` `CASRefCleanupObjectsDeleted` ## Phase 18 — orphan sweep {#phase-18-orphan-sweep} @@ -753,7 +777,7 @@ folding round is one `LIST` of `cas/ns/stream/`, the heartbeat floor (`LIST` plu seal, catalog and `gc/state` reads of phases 2, 4, 5 and 7, one successful lease `CAS`, and one commit `CAS`. A deferred round execution is cheaper: the same `LIST`, the heartbeat floor, phase 2's seal / catalog / `gc/state` reads, phase 4's two seal reads and catalog read, the lease `GET`/`CAS`, -and one suppressed namespace-janitor page (its own `LIST` page and reads, no deletes) — no commit +and one suppressed namespace-janitor page (its own `LIST` page and reads, no deletes; a folding round runs pages until the 20 s budget ends) — no commit `CAS` at all. The round's work is self-regulated: what a pass cannot finish within its budgets is carried and @@ -924,8 +948,9 @@ the hand-off's own budget (`cas_gc_round_handoff_prefix_wholesale_budget`). ### Phase 15 — manifest deletes {#cost-phase-15} -One batch `DELETE` request per `cas_gc_bulk_delete_chunk_keys` entries of `mf_cleanup` (on a -backend without batch delete: the refused bulk call plus one `DELETE` per key). No writes under +One batch `DELETE` request per `min(cas_gc_bulk_delete_chunk_keys, storage limit)` entries of a cohort of +`mf_cleanup` (a cohort of 1000 keys with a storage limit of 100 is 10 requests; a storage that +rejects a batch delete ends the family for the round). No writes under `suppress_destructive`. ### Phase 16 — namespace cleanup {#cost-phase-16} @@ -933,11 +958,12 @@ backend without batch delete: the refused bulk call plus one `DELETE` per key). | Key | Operation | Requests | |---|---|---:| | `/gc/maintenance_state` | `GET` | 1 (durable `janitor_cursor`) | -| `/cas/ns/` | `LIST` | one page | -| `/cas/ref_catalog` | `GET` | 1 | +| `/cas/ns/` | `LIST` | one per page | +| `/cas/ref_catalog` | `GET` | one per page | | `/gc/state` | `GET` | one per fence check | -| dead-life object | `DELETE` | one per object (plus one `HEAD` per object whose `LIST` entry carried no token) | -| `/gc/maintenance_state` | `CAS` | 1 when the page is decided | +| dead `_ckpt` / `_files` object | `DELETE` | one per object (plus one `HEAD` per object whose `LIST` entry carried no token) | +| dead `_log` / `_snap` keys | batch `DELETE` | one per job of up to `cas_gc_bulk_delete_chunk_keys` keys, capped by the storage's limit; one per key when the limit is 1 | +| `/gc/maintenance_state` | `CAS` | one per phase | ### Phase 17 — ref object cleanup {#cost-phase-17} @@ -945,7 +971,7 @@ backend without batch delete: the refused bulk call plus one `DELETE` per key). |---|---|---:| | checkpoint-named `_log`, predecessor seal, `_snap` | `GET` | per planned namespace (recovery-triple validation before any delete) | | `/cas/ref_catalog` and `/gc/state` | `GET` | one each per chunk (authority re-validation) | -| `_log` / `_snap` keys | batch `DELETE` | one request per chunk of ≤ `cas_gc_bulk_delete_chunk_keys` keys (the refused bulk call plus one per key on a backend without batch delete) | +| `_log` / `_snap` keys | batch `DELETE` | one request per ≤ `min(cas_gc_bulk_delete_chunk_keys, storage limit)` keys (a chunk of 1000 keys with a storage limit of 100 is 10 requests; a rejected batch delete ends the pass for the round) | ### Phase 18 — orphan sweep {#cost-phase-18} diff --git a/docs/en/antalya/cas/configuration.md b/docs/en/antalya/cas/configuration.md index 17b1e3482768..c3b9130970b9 100644 --- a/docs/en/antalya/cas/configuration.md +++ b/docs/en/antalya/cas/configuration.md @@ -106,7 +106,7 @@ entirely before release. Treat this table as a snapshot of the current build, no | `cas_part_folder_cache_max_entry_bytes` | 16 MiB | Oversized part-folder views bypass retention above this size | | `cas_manifest_decode_cache_bytes` | 128 MiB | Manifest decode cache byte budget (`0` disables) | | `cas_gc_meta_pool_size` | `16` | Bounded pool size for GC per-hash freshness-meta writes | -| `cas_gc_io_concurrency` | `16` | Bounded pool size for GC object-storage requests that run in parallel: the fold's read-ahead (checkpoints, ref logs, manifests, zero-candidate HEADs), the orphan-manifest sweep planning reads, the `SYSTEM CAS GC REBUILD` read-ahead, and the `pending_deletes` blob `HEAD` + conditional `DELETE` fan-out. Not covered: meta writes (`cas_gc_meta_pool_size`) and all other GC requests, which run on the round thread. `1` runs the covered requests sequentially. `cas_gc_read_concurrency` is rejected without an alias; use `cas_gc_io_concurrency` instead | +| `cas_gc_io_concurrency` | `16` | Bounded pool size for GC object-storage requests that run in parallel: the fold's read-ahead (checkpoints, ref logs, manifests, zero-candidate HEADs), the orphan-manifest sweep planning reads, the `SYSTEM CAS GC REBUILD` read-ahead, the `pending_deletes` blob `HEAD` + conditional `DELETE` fan-out, and the namespace janitor's dead-life deletes. Not covered: meta writes (`cas_gc_meta_pool_size`) and all other GC requests, which run on the round thread. `1` runs the covered requests on a one-thread pool, one at a time; the janitor's jobs still overlap the round thread's authority refresh and next page `LIST`. `cas_gc_read_concurrency` is rejected without an alias; use `cas_gc_io_concurrency` instead | | `cas_attempt_timeout_ms` | `5000` | Budget for one HTTP attempt of a writable Native mount's control-plane requests (read, head, list, remove, conditional write), at least 1. Together with the connect cap it forms the attempt envelope (`cas_attempt_timeout_ms + 2 × cap`; the cap is `cas_attempt_timeout_ms` itself when the disk's `connect_timeout_ms` is `0`, else `min(connect_timeout_ms, cas_attempt_timeout_ms)`) that the lease arithmetic reserves: one TCP connect and one TLS handshake under the cap each, send/receive bounded per socket operation by `cas_attempt_timeout_ms`. With background renewal the cadence check requires `cas_mount_renew_period_ms + 2 × envelope + cas_lease_safety_margin_ms < cas_mount_lease_ttl_ms`, which puts an effective ceiling on the frozen connect cap: under the defaults (TTL 30000, period 10000, margin 2000) the envelope must stay under 9000, so a disk `connect_timeout_ms` of 2000 ms or more refuses to open writable — lower the connect timeout or raise the TTL if you hit this | | `cas_lease_safety_margin_ms` | `2000` | Startup-only margin validated against the mount lease TTL: the attempt envelope + `cas_lease_safety_margin_ms` must be strictly less than the mount lease TTL, and `cas_mount_renew_period_ms` + 2 × envelope + `cas_lease_safety_margin_ms` too, or the disk refuses to open writable | | `cas_unsafe_remount_no_delay` | `0` | Reclaim a mount slot that carries this server's own uuid at once after a hard restart, without observing the slot's token for the lease TTL. Unsafe whenever two processes can hold the same `server_uuid` (a copied uuid file, a stalled predecessor). After such a reclaim the predecessor can still start conditional writes until its own cutoff (`confirmed deadline − cas_lease_safety_margin_ms − 2 × envelope`) or until its next renewal meets the token guard, and a request it already sent may still materialize later. That is not a data hazard: ref-log keys carry `(writer_epoch, sequence)` and creates are conditional, so two writers can never commit different bodies to one key, and recovery's epoch seal settles any straggler (recovery fails closed after 64 successive seal-create attempts displaced by newly materializing old-epoch transactions). The exposure is availability, not data. Intended for test stands and deployments that guarantee one process per uuid | @@ -179,7 +179,7 @@ for the remaining caps, `0` means unbounded. |---|---|---|---| | `cas_manifest_sweep_list_budget_keys` | `1000` | `UInt64` | Orphan-manifest sweep `LIST` budget per round | | `cas_manifest_sweep_delete_budget_keys` | `100` | `UInt64` | Orphan-manifest sweep `DELETE` budget per round | -| `cas_gc_bulk_delete_chunk_keys` | `1000` | `1`–`1000` | Keys per batch delete request in GC's write-once families (owner-removed manifest bodies, covered ref logs and snapshots) | +| `cas_gc_bulk_delete_chunk_keys` | `1000` | `1`–`1000` | Keys per batch delete request in GC's write-once families: owner-removed manifest bodies, covered ref logs and snapshots, and dead-life ref logs and snapshots of the namespace janitor. Capped by the storage's batch-delete limit (the disk key `objects_chunk_size_to_delete` on S3, where 0 acts as 1); a storage without batch delete gets one-key requests | | `cas_gc_round_graduation_budget` | `5000` | `0` = unbounded | Blob-graduation (`condemned` → `delete_pending`) cohort cap per round | | `cas_gc_round_redelete_budget` | `5000` | `0` = unbounded | Exact-token re-delete cohort cap for prior `delete_pending` rows per round | | `cas_gc_round_sweep_namespace_budget` | `20` | `0` = unbounded | Distinct namespaces per orphan-manifest sweep page whose protection view may be built | diff --git a/docs/en/operations/system-tables/cas_gc_log.md b/docs/en/operations/system-tables/cas_gc_log.md index 46fcd559cfb0..d79a61bb7b72 100644 --- a/docs/en/operations/system-tables/cas_gc_log.md +++ b/docs/en/operations/system-tables/cas_gc_log.md @@ -85,9 +85,9 @@ The phases, in execution order: | `meta_pool_wait` | Drain the round's per-hash freshness-meta writes. | none on this thread — see the caveat below | | `round_commit` | The generation-retention prune and the round's single `gc/state` compare-and-swap. | prune `LIST`s and deletes, one compare-and-swap | | `handoff_reclaim` | Wholesale-reclaim generations a moved run ref stranded below the retention cursor. | prefix `LIST`s and deletes | -| `manifest_deletes` | Exact-token deletes of owner-removed manifest bodies, after their decrements were adopted. | one `DELETE` per body | -| `namespace_cleanup` | Run one bounded `cas/ns/` page across the stream and state subtrees for the perpetual dead-life janitor. This phase is physical reclamation, not a lifecycle gate. | one namespace-root page `LIST`, catalog cut, exact-token deletes | -| `ref_object_cleanup` | Delete ref logs covered by both the durable fold cursor and a durable snapshot, plus superseded snapshots. | one `HEAD` + one `DELETE` per deletable object | +| `manifest_deletes` | Batch deletes of owner-removed manifest bodies, after their decrements were adopted. `capability_learned` is 1 when the storage rejected a batch delete; the family stops for the round. `unsent` counts bodies not deleted this round. | one batch `DELETE` per `cas_gc_bulk_delete_chunk_keys` bodies, capped by the storage's batch-delete limit | +| `namespace_cleanup` | The perpetual dead-life janitor: pages of `cas/ns/` under a 20 s soft budget. `phase_metrics`: `janitor_pages`, `janitor_keys`, `janitor_deleted` (keys of succeeded jobs, absent ones included, plus exact `_ckpt` and `_files` removals), `leaked`, `batches` (jobs that reached the store call, including those the request gate refused), `batches_leaked`, `batches_held`, `delete_jobs`, `delete_jobs_skipped`, `batch_keys`, `budget_exhausted`, `cursor_advanced`. Delete jobs run on the GC I/O pool, so this row's `ProfileEvents` do not include their requests. | per page: one `gc/state` `GET`, one namespace-root page `LIST`, one catalog `GET`, exact-token deletes; one batch or one-key `DELETE` per job; one maintenance-state `CAS` per phase | +| `ref_object_cleanup` | Delete ref logs covered by both the durable fold cursor and a durable snapshot, plus superseded snapshots. `capability_learned` is 1 when the storage rejected a batch delete; the family stops for the round. | one batch `DELETE` per `cas_gc_bulk_delete_chunk_keys` keys, capped by the storage's batch-delete limit | | `orphan_sweep` | The budgeted, cursor-paced orphan part-manifest backstop. | budgeted `LIST` and deletes | Which phase dominates a round: diff --git a/src/Common/ProfileEvents.cpp b/src/Common/ProfileEvents.cpp index a9986b5aa485..bf93cd312911 100644 --- a/src/Common/ProfileEvents.cpp +++ b/src/Common/ProfileEvents.cpp @@ -927,7 +927,7 @@ The server successfully detected this situation and will download merged part fr M(CASGCRefWalkPlansBuilt, "Number of complete catalog-authoritative CAS ref walk plans constructed by ordinary GC and rebuild. A regular or rebuilding invocation that reaches the post-LIST catalog cut increments this exactly once, including a round that later defers.", ValueType::Number) \ M(CASGCUnmatchedAdoptedParentLives, "Number of adopted-parent CAS ref-life rows dropped because the post-LIST catalog cut has no matching physical life. Each occurrence is inert for planning and suppression and is logged with its exact physical life id; a persistent nonzero rate indicates old generation state is outliving catalog removal.", ValueType::Number) \ M(CASGCStuckRemovals, "Number of adopted CAS GC rounds that observed a Removing namespace at or beyond the diagnostic age threshold without terminal cleanup evidence. Incremented and warned every such round; diagnostic only, with no effect on folding, suppression, appends, or deletion.", ValueType::Number) \ - M(CASGCNamespaceCleanupLeaks, "Number of dead-life namespace objects whose reclamation the perpetual janitor could not confirm because HEAD or exact-delete failed. Each occurrence is logged with the exact key and remains leak-only: it neither suppresses destructive GC nor blocks catalog lifecycle progress.", ValueType::Number) \ + M(CASGCNamespaceCleanupLeaks, "Number of dead-life namespace objects whose reclamation the perpetual janitor could not confirm because HEAD, exact delete or a batch delete failed. Each occurrence is logged with the exact key and remains leak-only: it neither suppresses destructive GC nor blocks catalog lifecycle progress.", ValueType::Number) \ M(CASDetachedWorkDrainTimeouts, "Counts CAS storage teardowns whose bounded wait for detached background work expired with work still in flight. The teardown proceeds, but for that teardown it could not be established that no tracked task still holds the pool. Expected to stay at zero.", ValueType::Number) \ M(CASEventDroppedContextExpired, "Number of CAS system-log events dropped because the storage's `Context` reference expired before delivery. A non-zero value indicates event production outlived the owning server context.", ValueType::Number) \ M(CASGCUnappliedFoldedTransactions, "Number of ref transactions a GC round folded and merged but whose blob deltas never reached a shard reducer. Always 0 on a healthy round; a nonzero value fails the round closed, because the round would otherwise advance its fold cursor past a transaction it never applied.", ValueType::Number) \ diff --git a/src/Common/setThreadName.h b/src/Common/setThreadName.h index 8a0222dc9639..b34b48cc5171 100644 --- a/src/Common/setThreadName.h +++ b/src/Common/setThreadName.h @@ -35,6 +35,7 @@ namespace DB M(CAS_ANOMALY_DIAG, "CasAnomalyDiag") \ M(CAS_GC_HEARTBEAT, "CasGcHeartbeat") \ M(CAS_GC_REDELETE, "CasGcRedelete") \ + M(CAS_GC_JANITOR, "CasGcJanitor") \ M(CAS_GC_SCHEDULER, "CasGcSched") \ M(CAS_LEASE_RENEWER, "CasLeaseRenewer") \ M(CAS_REF_SNAPSHOT_PUBLISH, "CasRefSnapPub") \ diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasBackend.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasBackend.h index 96911defc15f..eb502f9724c2 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasBackend.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasBackend.h @@ -201,6 +201,10 @@ class Backend /// absence is success. virtual void removeManyWriteOnce(const std::vector & keys, TransportAccess &) = 0; + /// The most keys one `removeManyWriteOnce` sends to the storage as one request; at least 1. A decorator + /// must forward it: the default of 1 would turn every batch below it into one-key requests. + virtual size_t bulkDeleteKeyLimit() const { return 1; } + /// Authoritative, cache-bypassing probe of one key -- see `ProbeOutcome`. DEFAULT (used by every /// backend without sharper raw-error evidence, e.g. `InMemoryBackend`): derived from `head`/`read` /// alone, so it can only distinguish `Present` from `KeyAbsent`, and ANY exception from either diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasInMemoryBackend.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasInMemoryBackend.h index 2074750e3ef6..89d06e9965f1 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasInMemoryBackend.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasInMemoryBackend.h @@ -1,5 +1,6 @@ #pragma once #include +#include #include #include #include @@ -59,6 +60,9 @@ class InMemoryBackend : public Backend /// How many `removeManyWriteOnce` calls reached the store, armed failures included. size_t bulkRemoveCalls() const; + /// The emulation deletes keys one by one under its lock, so the CAS maximum is the only bound. + size_t bulkDeleteKeyLimit() const override { return kBulkDeleteMaxKeys; } + /// Creates the key when `expected_value` is empty, or replaces the incarnation it names. A /// refused precondition leaves the store unchanged. Value enforcement can be disabled with /// `setEnforceTokens` to model a backend that incorrectly ignores the condition. @@ -156,6 +160,7 @@ class InMemoryBackend : public Backend void onWriteCommitted(const String & key, std::function hook); private: + /// Complete in-memory incarnation state for one key. All fields are read or modified while /// `mutex_` is held; replacing `value` marks a new incarnation even when the bytes are unchanged. struct Object diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasInstrumentedBackend.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasInstrumentedBackend.h index ed0c25dab861..5d1f3f36e953 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasInstrumentedBackend.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasInstrumentedBackend.h @@ -161,6 +161,7 @@ class InstrumentedBackend final : public Backend bool supportsListTokens() const override { return inner->supportsListTokens(); } uint64_t attemptTimeoutMs() const override { return inner->attemptTimeoutMs(); } uint64_t attemptEnvelopeMs() const override { return inner->attemptEnvelopeMs(); } + size_t bulkDeleteKeyLimit() const override { return inner->bulkDeleteKeyLimit(); } bool refreshCredentials() override { return inner->refreshCredentials(); } private: diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasObjectStorageBackend.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasObjectStorageBackend.cpp index 0589929d2912..0843f0120a9f 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasObjectStorageBackend.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasObjectStorageBackend.cpp @@ -1,6 +1,7 @@ #include #include +#include #include #include #include @@ -1003,6 +1004,11 @@ void ObjectStorageBackend::emuForgetDeletedToken(const String & key) } } +size_t ObjectStorageBackend::bulkDeleteKeyLimit() const +{ + return mode == Mode::Native ? object_storage->batchDeleteKeyLimit() : kBulkDeleteMaxKeys; +} + void ObjectStorageBackend::removeManyWriteOnce(const std::vector & keys, TransportAccess & access) { if (keys.empty()) @@ -1013,9 +1019,8 @@ void ObjectStorageBackend::removeManyWriteOnce(const std::vector & objects.reserve(keys.size()); for (const WriteOnceKey & key : keys) objects.emplace_back(key.str()); - /// `NOT_IMPLEMENTED` from a storage without a batch delete propagates -- this layer never - /// substitutes a per-key loop of its own (that would run under a single admission for up to - /// 1000 keys). The caller in CasGc.cpp catches it and retries one key per admitted request. + /// `NOT_IMPLEMENTED` from a storage without a batch delete propagates; this layer never loops per + /// key, and the GC caller stops the family for the round. object_storage->removeObjectsIfExistUnderProfile(objects, controlRequest(access.attemptNo())); return; } diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasObjectStorageBackend.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasObjectStorageBackend.h index f9f51396efbb..cc3c83f4f143 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasObjectStorageBackend.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasObjectStorageBackend.h @@ -101,6 +101,10 @@ class ObjectStorageBackend final : public Backend /// under the control-plane profile; `EmulatedSingleProcess` deletes each present key under the /// emulation lock with the same token bookkeeping as the single-key delete. void removeManyWriteOnce(const std::vector & keys, TransportAccess & access) override; + + /// Native: the object storage's own limit. Emulated: `kBulkDeleteMaxKeys`, since this backend deletes + /// the keys one by one under its emulation lock. + size_t bulkDeleteKeyLimit() const override; /// Native mints its store's own dialect (ETag or GCS generation); the emulated adapter mints its /// own values. Dialect dialect() const override { return mode == Mode::Native ? native_token_type : Dialect::Emulated; } diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasThrottlingBackend.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasThrottlingBackend.h index e620195a31c3..8f69d4464347 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasThrottlingBackend.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasThrottlingBackend.h @@ -125,6 +125,7 @@ class ThrottlingBackend final : public Backend bool supportsListTokens() const override { return inner->supportsListTokens(); } uint64_t attemptTimeoutMs() const override { return inner->attemptTimeoutMs(); } uint64_t attemptEnvelopeMs() const override { return inner->attemptEnvelopeMs(); } + size_t bulkDeleteKeyLimit() const override { return inner->bulkDeleteKeyLimit(); } bool refreshCredentials() override { return inner->refreshCredentials(); } void checkPoolPreconditions() override { inner->checkPoolPreconditions(); } void checkSkipAccessCheckSupport() override { inner->checkSkipAccessCheckSupport(); } diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/ContentAddressedSettings.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/ContentAddressedSettings.cpp index 7331d2bad29a..1189a7c9aa23 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/ContentAddressedSettings.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/ContentAddressedSettings.cpp @@ -79,8 +79,8 @@ constexpr std::string_view CAS_KEY_PREFIX = "cas_"; DECLARE(UInt64, part_folder_cache_max_entry_bytes, 16ULL << 20, "Oversized part-folder views bypass retention above this size", 0) \ DECLARE(UInt64, manifest_decode_cache_bytes, 128ULL << 20, "Manifest DECODE cache byte budget (0 disables)", 0) \ DECLARE(UInt64, gc_meta_pool_size, 16, "Bounded pool size for GC per-hash freshness-meta writes", 0) \ - DECLARE(UInt64, gc_io_concurrency, 16, "Maximum number of threads in the GC I/O pool. Used for fold and rebuild read-ahead, orphan-manifest sweep planning reads, and pending_deletes HEAD plus conditional DELETE. Per-hash meta writes use gc_meta_pool_size; other GC requests run on the round thread. 1 disables parallel GC I/O", 0) \ - DECLARE(UInt64, gc_bulk_delete_chunk_keys, 1000, "Keys per batch delete request in GC's write-once families (owner-removed manifest bodies, covered ref logs and snapshots); 1 to 1000", 0) \ + DECLARE(UInt64, gc_io_concurrency, 16, "Maximum number of threads in the GC I/O pool. Used for fold and rebuild read-ahead, orphan-manifest sweep planning reads, pending_deletes HEAD plus conditional DELETE, and namespace janitor deletes of dead ref logs and snapshots. Per-hash meta writes use gc_meta_pool_size; other GC requests run on the round thread. At 1 the jobs run on a one-thread pool: they stay off the round thread but still overlap its authority refresh and next LIST", 0) \ + DECLARE(UInt64, gc_bulk_delete_chunk_keys, 1000, "Keys per batch delete request in GC's write-once families (owner-removed manifest bodies, covered ref logs and snapshots, dead-life ref logs and snapshots of the namespace janitor), capped by the storage's batch-delete limit; 1 to 1000", 0) \ DECLARE(UInt64, attempt_timeout_ms, 5000, "Budget for one HTTP attempt of a writable Native mount's control-plane requests (read, head, list, remove, conditional write), at least 1. With the connect cap it forms the attempt envelope the lease arithmetic reserves", 0) \ DECLARE(UInt64, lease_safety_margin_ms, 2000, "Startup-only margin validated against the mount lease TTL: attempt envelope + this must be strictly less than the TTL, and renew period + 2 × envelope + this too", 0) \ DECLARE(String, staging_backend, "local", "Blob staging backend (local | s3); s3 is opt-in", 0) \ diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasLayout.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasLayout.h index 5ab57bb7854b..b54dab0ea5f3 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasLayout.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Formats/CasLayout.h @@ -158,7 +158,7 @@ class Layout /// not try an uncompressed variant. String refLogKey(const NamespaceLifeId & ns_id, const RefTxnId & id) const { - return namespaceStreamPrefix(ns_id) + "_log/" + renderRefTxnId(id) + String(storedSuffix(FormatId::RefLog)); + return refStreamObjectKey(ns_id.incarnation, RefObjectKind::Log, id); } /// Writer-published table snapshot at `.../_snap/.zst`. The snapshot @@ -166,7 +166,7 @@ class Layout /// `X` reuses the `RefTxnId` of the last log it covers. String refSnapshotKey(const NamespaceLifeId & ns_id, const RefTxnId & id) const { - return namespaceStreamPrefix(ns_id) + "_snap/" + renderRefTxnId(id) + String(storedSuffix(FormatId::RefSnapshot)); + return refStreamObjectKey(ns_id.incarnation, RefObjectKind::Snap, id); } /// The write-once forms of `refLogKey` and `refSnapshotKey`: the same strings, typed as keys a @@ -181,6 +181,17 @@ class Layout return WriteOnceKey(refSnapshotKey(ns_id, id)); } + /// The write-once form of a LISTED `_log`/`_snap` key, minted from the identity `parseRefObjectKey` + /// recovered from it. `std::nullopt` when the canonical key for that identity is not `listed_key`: a + /// token-free delete must name only keys this layout writes. + std::optional writeOnceStreamKey(const ParsedRefObjectKey & parsed, std::string_view listed_key) const + { + String canonical = refStreamObjectKey(parsed.life_id, parsed.kind, parsed.txn_id); + if (canonical != listed_key) + return std::nullopt; + return WriteOnceKey(std::move(canonical)); + } + /// The life's checkpoint object (spec INV-4) at `/cas/ns/state//_ckpt`. Unlike /// immutable stream objects it is mutable (token-CAS), carries no transaction id, and therefore lives /// in the point/path-addressed state tree rather than a `_log`/`_snap` directory -- @@ -486,6 +497,13 @@ class Layout /// Parses the one physical-id segment after the rest of a life-owned key identified its family. NamespaceLifePhysicalId namespaceLifePhysicalIdOf(std::string_view key, std::string_view segment) const; + String refStreamObjectKey(NamespaceLifePhysicalId life_id, RefObjectKind kind, const RefTxnId & id) const + { + const bool log = kind == RefObjectKind::Log; + return namespaceStreamRootPrefix() + renderIncarnation(life_id) + (log ? "/_log/" : "/_snap/") + renderRefTxnId(id) + + String(storedSuffix(log ? FormatId::RefLog : FormatId::RefSnapshot)); + } + /// Build ///. /// Throws BAD_ARGUMENTS if id is shorter than 2 characters. String shardedKey(const String & ns, const String & id) const diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasGc.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasGc.cpp index f871182c8edb..a069184eb5c6 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasGc.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasGc.cpp @@ -85,6 +85,10 @@ namespace DB::Cas namespace { +/// Soft limit on one janitor phase: no page starts after it. +constexpr uint64_t kJanitorPhaseBudgetMs = 20'000; +constexpr size_t kJanitorPageKeys = 1000; + /// The `on_page_fetched` hook GC passes to every `forEachListedKey`/`recoverRefTable` /// call it owns (never passed by fsck/offline-repair callers of those shared helpers) -- one increment /// per physical LIST page, never per listed key. @@ -360,51 +364,68 @@ Gc::Gc(PoolPtr store_, UInt128 gc_id_, std::function now_ms_fn_, io_pool_refuse_at_for_test = store->poolConfig().gc_io_pool_refuse_at_for_test; } -void Gc::runNamespaceJanitorPage( +size_t Gc::bulkDeleteChunkKeys() const +{ + const uint64_t configured = store->poolConfig().gc_bulk_delete_chunk_keys; + const uint64_t storage_limit = store->poolBackendPtr()->bulkDeleteKeyLimit(); + return std::clamp(std::min(configured, storage_limit), 1, kBulkDeleteMaxKeys); +} + +void Gc::runNamespaceJanitor( const GcState & leased_state, bool suppress_destructive, uint64_t cleanup_evidence_rows) { GcPhaseTimer t(phase_sink, "namespace_cleanup"); t.metric("evidence_rows", cleanup_evidence_rows); - NamespaceJanitorResult janitor_result; + NamespaceJanitorResult result; try { - CasRequests & requests = store->openRequests(); - const Layout & layout = store->layout(); - NamespaceJanitor janitor(requests, layout, 1000); - /// ONE authority read per page, made here rather than from the predicate: the janitor's - /// operation samples its liveness before every request, and a page walks up to a thousand keys. - refreshAuthority(leased_state.lease.seq); - janitor_result = janitor.runOnePage(suppress_destructive, [this] { return authority_held; }); - for (const String & anomaly : janitor_result.anomalies) - LOG_WARNING(logger, "CAS namespace janitor: {}", anomaly); - if (janitor_result.leaked) - ProfileEvents::increment(ProfileEvents::CASGCNamespaceCleanupLeaks, janitor_result.leaked); + const uint64_t lease_seq = leased_state.lease.seq; + JanitorRunContext context; + /// One authority read per page, from the refresh; the predicate is sampled before every request. + context.liveness = [this] { return authority_held.load(); }; + context.refresh_authority = [this, lease_seq] { refreshAuthority(lease_seq); }; + context.batch_keys = bulkDeleteChunkKeys(); + context.budget_ms = kJanitorPhaseBudgetMs; + context.now_ms = mono_ms_fn; + context.io_pool = io_pool.get(); + context.schedule_refuse_at_for_test = &io_pool_refuse_at_for_test; + NamespaceJanitor(store->openRequests(), store->layout(), kJanitorPageKeys).run(suppress_destructive, context, result); } catch (const std::exception & e) { - LOG_WARNING(logger, "CAS namespace janitor skipped this round: {}", e.what()); + LOG_WARNING(logger, "CAS namespace janitor stopped this round: {}", e.what()); } - t.metric("janitor_pages", janitor_result.pages); - t.metric("janitor_keys", janitor_result.keys); - t.metric("janitor_deleted", janitor_result.deleted); - t.metric("leaked", janitor_result.leaked); + for (const String & anomaly : result.anomalies) + LOG_WARNING(logger, "CAS namespace janitor: {}", anomaly); + if (result.leaked) + ProfileEvents::increment(ProfileEvents::CASGCNamespaceCleanupLeaks, result.leaked); + t.metric("janitor_pages", result.pages); + t.metric("janitor_keys", result.keys); + t.metric("janitor_deleted", result.deleted); + t.metric("leaked", result.leaked); + t.metric("batches", result.batches); + t.metric("batches_leaked", result.batches_leaked); + t.metric("batches_held", result.batches_held); + t.metric("delete_jobs", result.delete_jobs); + t.metric("delete_jobs_skipped", result.delete_jobs_skipped); + t.metric("batch_keys", result.batch_keys); + t.metric("budget_exhausted", result.budget_exhausted ? 1 : 0); + t.metric("cursor_advanced", result.cursor_advanced ? 1 : 0); } -uint64_t removeChunkWriteOnceOrOneByOne(CasOperation & op, const std::vector & chunk, const Retry & policy) +void removeCohortWriteOnce( + CasOperation & op, + const std::vector & cohort, + size_t request_keys, + const Retry & policy, + const std::function & on_request_done) { - try + const size_t step = std::clamp(request_keys, 1, kBulkDeleteMaxKeys); + for (size_t begin = 0; begin < cohort.size(); begin += step) { - op.removeManyWriteOnce(chunk, policy); - return 1; - } - catch (const Exception & e) - { - if (e.code() != ErrorCodes::NOT_IMPLEMENTED) - throw; - for (const WriteOnceKey & key : chunk) - op.removeManyWriteOnce({key}, policy); - /// +1: the failed bulk attempt above is itself a call this helper made. - return 1 + chunk.size(); + const size_t end = std::min(cohort.size(), begin + step); + op.removeManyWriteOnce(std::vector(cohort.begin() + begin, cohort.begin() + end), policy); + on_request_done(begin, end); } } @@ -820,7 +841,7 @@ RoundReport Gc::runRegularRound(std::function on_lease_acquired, bool al /// DEFER has no `FoldResult`, hence no complete global destructive verdict. The janitor still /// takes its bounded page and catalog cut, but suppression keeps both deletes and valid-page /// cursor progress at the same position for the bounded forced fold to retry. - runNamespaceJanitorPage(state, /*suppress_destructive=*/true, /*cleanup_evidence_rows=*/0); + runNamespaceJanitor(state, /*suppress_destructive=*/true, /*cleanup_evidence_rows=*/0); return report; /// no fold, no pre-CAS deletes, no gc/state CAS — sealed generation stays pinned } @@ -1271,57 +1292,79 @@ RoundReport Gc::runRegularRound(std::function on_lease_acquired, bool al const std::map & mf_cleanup_now = suppress_destructive ? kNoManifestCleanup : folded.mf_cleanup; - /// Chunks of write-once keys, one request each, with no per-key precondition: a manifest key is - /// never written twice, so the body at it is the one the fold observed or nothing. The engine - /// reissues a failed chunk whole; a chunk that exhausts its policy throws here, and the chunks - /// before it are already recorded below. A key that one of the exhausted chunk's own attempts - /// did delete is not recorded either: deletion and recording are all-or-nothing per request, - /// never per key, so the next round's fold sees that key as already gone. The etag the fold - /// observed rides the event as information only. - const size_t chunk_keys = std::clamp(store->poolConfig().gc_bulk_delete_chunk_keys, 1, kBulkDeleteMaxKeys); + /// Cohorts of write-once keys with no per-key precondition: a manifest key is never written twice, + /// so the body at it is the one the fold observed or nothing. Each cohort is cut into storage + /// requests; the engine reissues a failed request whole, and a request that exhausts its policy + /// throws here with the requests before it already recorded below. A key that one of the exhausted + /// request's own attempts did delete is not recorded either: deletion and recording are + /// all-or-nothing per request, never per key, so the next round's fold sees that key as already + /// gone. The etag the fold observed rides the event as information only. + const size_t cohort_keys = std::clamp(store->poolConfig().gc_bulk_delete_chunk_keys, 1, kBulkDeleteMaxKeys); + const size_t request_keys = bulkDeleteChunkKeys(); uint64_t attempted = 0; uint64_t requests = 0; + bool capability_learned = false; std::vector chunk; std::vector *> chunk_entries; const auto flush = [&] { if (chunk.empty()) return; - /// A backend without `DeleteObjects` (GCS) falls back to one admitted delete per key here; - /// `chunk_entries`' per-key bookkeeping below is unaffected either way -- it counts objects - /// that are gone after this call returns, not how many requests it took to get them there. - requests += removeChunkWriteOnceOrOneByOne(op, chunk, Retry::standard()); - for (const auto * entry : chunk_entries) + removeCohortWriteOnce(op, chunk, request_keys, Retry::standard(), [&](size_t begin, size_t end) { - ++report.manifests_deleted; - EventEmitter{*store}.emit([&](CasEvent & e) + ++requests; + for (size_t i = begin; i < end; ++i) { - e.type = CasEventType::ManifestDelete; - e.namespace_ = entry->first.root_namespace.string(); - e.object_kind = CasEventObjectKind::Manifest; - e.object_hash = manifestRefDebugString(entry->first.ref); - e.token = entry->second.render(); - e.round = new_round; - e.gen = generation; - e.outcome = "deleted_or_absent"; - e.reason = "owner-removed manifest body; batch delete of a write-once key after decrements adopted"; - }); - } + const auto * entry = chunk_entries[i]; + ++report.manifests_deleted; + EventEmitter{*store}.emit([&](CasEvent & e) + { + e.type = CasEventType::ManifestDelete; + e.namespace_ = entry->first.root_namespace.string(); + e.object_kind = CasEventObjectKind::Manifest; + e.object_hash = manifestRefDebugString(entry->first.ref); + e.token = entry->second.render(); + e.round = new_round; + e.gen = generation; + e.outcome = "deleted_or_absent"; + e.reason = "owner-removed manifest body; batch delete of a write-once key after decrements adopted"; + }); + } + }); chunk.clear(); chunk_entries.clear(); }; - for (const auto & entry : mf_cleanup_now) + try + { + for (const auto & entry : mf_cleanup_now) + { + ++attempted; + chunk.push_back(layout.writeOnceManifestKey(entry.first)); + chunk_entries.push_back(&entry); + if (chunk.size() >= cohort_keys) + flush(); + } + flush(); + } + catch (const Exception & e) { - ++attempted; - chunk.push_back(layout.writeOnceManifestKey(entry.first)); - chunk_entries.push_back(&entry); - if (chunk.size() >= chunk_keys) - flush(); + if (e.code() != ErrorCodes::NOT_IMPLEMENTED) + throw; + /// The intake cursor that found these bodies is committed. The orphan-manifest sweep reclaims a + /// skipped body once the namespace's sealed cursor is past the body's writer epoch. + /// Later requests carry one key. + capability_learned = true; + ++requests; + LOG_WARNING(logger, "CAS GC manifest_deletes: the storage rejected a batch delete; the remaining " + "owner-removed manifest bodies are left to the orphan-manifest sweep, which reclaims each once the " + "namespace's sealed cursor is past its writer epoch: {}", e.message()); } - flush(); + const uint64_t accepted = report.manifests_deleted - manifests_deleted_before; t.metric("attempted", attempted); - t.metric("accepted", report.manifests_deleted - manifests_deleted_before); + t.metric("accepted", accepted); t.metric("requests", requests); + t.metric("unsent", mf_cleanup_now.size() - accepted); + t.metric("capability_learned", capability_learned ? 1 : 0); t.metric("suppressed", suppress_destructive ? 1 : 0); } @@ -1331,17 +1374,19 @@ RoundReport Gc::runRegularRound(std::function on_lease_acquired, bool al uint64_t cleanup_evidence_rows = 0; for (const auto & [life_id, ref_life_state] : folded.fold_seal.ref_lives) cleanup_evidence_rows += ref_life_state.cleanup_evidence ? 1 : 0; - runNamespaceJanitorPage(state, suppress_destructive, cleanup_evidence_rows); + runNamespaceJanitor(state, suppress_destructive, cleanup_evidence_rows); /// PHASE 17/18 `ref_object_cleanup`. Emitted even when the whole pass is skipped (`trim_enabled` is /// a test seam, `suppressed` gates the deletes), because "this phase did nothing and why" is exactly /// what a reader of a round that reclaimed nothing needs to see. { GcPhaseTimer t(phase_sink, "ref_object_cleanup"); + bool capability_learned = false; if (trim_enabled) - cleanupRefObjects(folded, state.lease, suppress_destructive, round_work_budget); + capability_learned = cleanupRefObjects(folded, state.lease, suppress_destructive, round_work_budget); t.metric("suppressed", suppress_destructive ? 1 : 0); t.metric("trim_enabled", trim_enabled ? 1 : 0); t.metric("namespaces_planned", folded.ref_tables.size()); + t.metric("capability_learned", capability_learned ? 1 : 0); } /// Bounded orphan-manifest backstop. The fold already exact-read each candidate, retired its exact @@ -3703,14 +3748,14 @@ void Gc::reportSweepRetention(const ManifestSweepResult & result) retain_rollup_passes_since_report = 0; } -void Gc::cleanupRefObjects( +bool Gc::cleanupRefObjects( const FoldResult & folded, const GcLease & adopted_lease, bool suppress_destructive, GcRoundWorkBudget & work_budget) { /// A clamp / ref-folding abort this round may leave landed-before-cut edges unfolded behind the clamp, /// so a covered-log cleanup could delete a log whose delta is not yet durable -- defer to a clean pass. if (suppress_destructive) - return; + return false; CasOperation op = store->openRequests().admit(); const Layout & layout = store->layout(); @@ -3833,32 +3878,42 @@ void Gc::cleanupRefObjects( cohort.push_back(layout.writeOnceRefSnapshotKey(life, snap_id)); } - const size_t chunk_keys = std::clamp(store->poolConfig().gc_bulk_delete_chunk_keys, 1, kBulkDeleteMaxKeys); + const size_t cohort_keys = std::clamp(store->poolConfig().gc_bulk_delete_chunk_keys, 1, kBulkDeleteMaxKeys); + const size_t request_keys = bulkDeleteChunkKeys(); for (size_t begin = 0; begin < cohort.size(); ) { - /// Cumulative per-round cap in KEYS, exactly as before; a chunk is cut to what remains. The - /// plan recomputes the same remaining candidates from durable state next round, so nothing - /// here needs its own cursor. + /// Cumulative per-round cap in KEYS; a cohort is cut to what remains. The plan recomputes the same + /// remaining candidates from durable state next round, so nothing here needs its own cursor. if (!work_budget.refCleanupAvailable()) - return; - size_t end = std::min(cohort.size(), begin + chunk_keys); + return false; + size_t end = std::min(cohort.size(), begin + cohort_keys); if (work_budget.max_ref_cleanup_objects != 0) end = std::min(end, begin + (work_budget.max_ref_cleanup_objects - work_budget.ref_cleanup_objects_used)); std::vector chunk(cohort.begin() + begin, cohort.begin() + end); + /// One authority read per cohort, not per request: on a store without batch delete a per-request + /// read would add two GETs per key. if (!authorityHolds(chunk.front().str())) - return; - /// A backend without `DeleteObjects` (GCS) falls back to one admitted delete per key here. - /// The budget and the profile event below count OBJECTS in `chunk`, which is the same - /// `chunk.size()` whichever way `removeChunkWriteOnceOrOneByOne` actually sent them. - removeChunkWriteOnceOrOneByOne(op, chunk, Retry::standard()); - work_budget.ref_cleanup_objects_used += chunk.size(); - ProfileEvents::increment(ProfileEvents::CASRefCleanupObjectsDeleted, chunk.size()); /// cleanup object deletion - /// Advance by what was actually sent, not the nominal chunk size: the budget cap above can - /// truncate a chunk short of `chunk_keys`, and advancing by the full stride would skip the - /// untried remainder instead of retrying it next iteration. + return false; + try + { + removeCohortWriteOnce(op, chunk, request_keys, Retry::standard(), [&](size_t request_begin, size_t request_end) + { + work_budget.ref_cleanup_objects_used += request_end - request_begin; + ProfileEvents::increment(ProfileEvents::CASRefCleanupObjectsDeleted, request_end - request_begin); + }); + } + catch (const Exception & e) + { + if (e.code() != ErrorCodes::NOT_IMPLEMENTED) + throw; + LOG_WARNING(logger, "CAS GC ref cleanup: the storage rejected a batch delete; the remaining candidates " + "wait for the next round: {}", e.message()); + return true; + } begin = end; } } + return false; } namespace @@ -4849,16 +4904,17 @@ void Gc::rememberObservation(const GcLease & lease) void Gc::refreshAuthority(uint64_t admitted_generation) { - /// Fail-closed first, so every early exit below leaves this leader deposed. - authority_held = false; + /// Janitor jobs read the flag during this read, so it keeps the confirmed value until one verdict is + /// stored at the end; every failure still stores `false`. + bool held = false; try { CasOperation op = store->openRequests().admit(); - const auto got = op.read(store->layout().gcStateKey(), Retry::standard()); - if (!got) - return; - const GcState current = decodeGcState(got->bytes); - authority_held = current.lease.owner == gc_id && current.lease.seq == admitted_generation; + if (const auto got = op.read(store->layout().gcStateKey(), Retry::standard())) + { + const GcState current = decodeGcState(got->bytes); + held = current.lease.owner == gc_id && current.lease.seq == admitted_generation; + } } catch (...) { @@ -4866,6 +4922,7 @@ void Gc::refreshAuthority(uint64_t admitted_generation) "CAS gc: the leader-authority probe failed; this round's destructive operations treat the " "lease as lost"); } + authority_held.store(held); } void Gc::pulseHeartbeat(Pool & store, UInt128 gc_id) @@ -5011,7 +5068,7 @@ CatalogLifecycleReconcileResult Gc::drainCompletedRemoving(const GcState & lease /// second -- the single reading taken here would otherwise authorise all of them. const uint64_t admitted_generation = leased_state.lease.seq; refreshAuthority(admitted_generation); - CasOperation op = store->openRequests().admit([this] { return authority_held; }); + CasOperation op = store->openRequests().admit([this] { return authority_held.load(); }); return CatalogLifecycleReconciler(op, store->layout(), *parent) .reconcile([this, admitted_generation] { refreshAuthority(admitted_generation); }); } diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasGc.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasGc.h index 33f2bdcf09a8..e00f50b0e77d 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasGc.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasGc.h @@ -73,27 +73,16 @@ enum class UniversePolicy : uint8_t /// decision ever reads them. uint64_t retiredLogicalSize(ObjectKind kind, uint64_t object_size, uint64_t blob_header_len); -/// Deletes `chunk` as one bulk `removeManyWriteOnce` request, falling back to one admitted request per -/// key when the object storage answers with `NOT_IMPLEMENTED` -- the signal -/// `S3ObjectStorage::removeObjectsIfExistImpl` gives (without sending anything else itself) once -/// `DeleteObjects` is known unsupported (a configured GCS backend, or one that just failed a batch -/// attempt this same call). The fallback is not merely "the same deletes issued more slowly": each -/// `op.removeManyWriteOnce({key}, policy)` is its OWN admission (fence, budget, deadline checked -/// afresh), which one bulk call covering up to `kBulkDeleteMaxKeys` physical deletes under a SINGLE -/// admission cannot be -- exactly the gap a storage-side per-key loop would have left open. Every other -/// failure propagates unchanged: retry/reissue for it is the engine's own policy, applied to each -/// admitted attempt -- bulk or single -- the same way it always was. -/// -/// Returns the number of `op.removeManyWriteOnce` calls THIS HELPER issued: 1 for the bulk path, or -/// 1 + `chunk.size()` for the fallback -- the failed bulk attempt counted alongside the one call per key -/// that followed it, since that attempt is a call this helper made whether or not it reached the network -/// (there is no signal available here to tell "sent and rejected" apart from "refused locally, unsent"; -/// `S3ObjectStorage::removeObjectsIfExistImpl` reports both as the same NOT_IMPLEMENTED). This is call -/// COUNT, not a distinct network-request count -- the same granularity `CountingBackend::bulkRemoveCalls` -/// and the `CASBulkDeleteRequests` profile event already use elsewhere for "request". -/// Declared here (not file-local) so a unit test can drive it directly against a scripted backend, -/// rather than only through a full GC round. -uint64_t removeChunkWriteOnceOrOneByOne(CasOperation & op, const std::vector & chunk, const Retry & policy); +/// Deletes `cohort` as consecutive `removeManyWriteOnce` requests of at most `request_keys` keys and calls +/// `on_request_done(begin, end)` after each one returns. Every failure propagates, `NOT_IMPLEMENTED` included: +/// re-sending a rejected batch key by key would delete under a decision nobody made for those keys, and the +/// capability the rejection teaches makes the next round cut one-key requests anyway. +void removeCohortWriteOnce( + CasOperation & op, + const std::vector & cohort, + size_t request_keys, + const Retry & policy, + const std::function & on_request_done); /// Pure skip-unchanged decision. Returns true iff the current round may be /// DEFERRED (re-adopt the sealed generation, no fold/delete). A round MUST fold when: enough shards @@ -428,7 +417,7 @@ class Gc /// process, injected for tests, defaults to `Pool::bootMs()`, and the ONLY clock the heartbeat /// gate's own fence-out threshold is measured against (mirrors `claimMountAwaitingExpiry`'s /// `mono_ms_fn`, but at heartbeat-gate granularity — one GC round is one observation tick). - /// Everything else in the round stays deterministic/clock-free. + /// The namespace janitor also reads `mono_ms_fn`, for its phase budget. /// /// `log_` is the logger every round-engine log line is emitted through; pass a disk/srid-scoped /// logger (e.g. `CasGcScheduler`'s own `log`, built from a `fmt::format("{}::...", storage_path)` @@ -505,6 +494,10 @@ class Gc /// instance, and the scheduler holds `gc_round_mutex` across both the install and the round. void setPhaseSink(GcPhaseSink sink) { phase_sink = std::move(sink); } + /// Keys per GC write-once delete request: `gc_bulk_delete_chunk_keys` capped by the storage's batch + /// limit, clamped to [1, `kBulkDeleteMaxKeys`] because an empty request would never advance. + size_t bulkDeleteChunkKeys() const; + void setRebuildEdgeBudgetForTest(uint64_t n) { rebuild_edge_budget_override = n; } /// TEST SEAM: disable the round's journal trim so a folded event stays in the journal @@ -548,11 +541,10 @@ class Gc /// It performs no physical LIST or delete. CatalogLifecycleReconcileResult drainCompletedRemoving(const GcState & leased_state); - /// Run exactly one independently paced physical namespace-maintenance page. The caller supplies - /// the one round-wide destructive verdict when it exists; DEFER passes suppression because it has - /// no folded frontier verdict. This helper owns only janitor I/O and phase metrics, never lifecycle - /// transitions or the hot stream walk plan. - void runNamespaceJanitorPage( + /// One namespace-janitor phase: pages from the persisted cursor until the soft budget, a hold, or a page + /// without dead-life debris. DEFER passes suppression, which lists one page and deletes nothing. Owns only + /// janitor I/O and phase metrics, never lifecycle transitions. + void runNamespaceJanitor( const GcState & leased_state, bool suppress_destructive, uint64_t cleanup_evidence_rows); void reportStuckRemovals(const RefPlan & plan, uint64_t current_round); @@ -871,7 +863,9 @@ class Gc /// same-id `_log` and `_snap` validate through `readCheckpointSnapshotBase` do. Logs and snapshots /// are then deletable only strictly BELOW that checkpoint (and logs also at or below the durable /// cursor). A namespace with no checkpoint base, or an invalid triple, is leak-only this pass. - void cleanupRefObjects( + /// Returns true when the storage rejected a batch delete: the family stops for this round and the + /// next round's plan recomputes the same candidates. + bool cleanupRefObjects( const FoldResult & folded, const GcLease & adopted_lease, bool suppress_destructive, GcRoundWorkBudget & work_budget); @@ -987,8 +981,8 @@ class Gc /// The read that sets this flag is therefore made by the round, at the granularity the old /// hand-written fence check had (once per drain, once per janitor page), never from inside the /// predicate. The staleness that buys is bounded by that granularity and stated where each caller - /// refreshes it. - bool authority_held = false; + /// refreshes it. Atomic because namespace-janitor jobs read it while the round thread refreshes it. + std::atomic authority_held{false}; /// the contender's observation window (steal protocol) bool has_observation = false; diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasNamespaceJanitor.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasNamespaceJanitor.cpp index f8c153ed0f20..ef79bc933e9f 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasNamespaceJanitor.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasNamespaceJanitor.cpp @@ -2,6 +2,21 @@ #include #include #include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +namespace DB::ErrorCodes +{ + extern const int CANNOT_SCHEDULE_TASK; +} namespace DB::Cas { @@ -19,147 +34,448 @@ void throwOnRefusedOrGaveUp(WriteResult && result, std::string_view what) (void)orThrow(std::move(result), what); } +enum class JobOutcome : uint8_t +{ + Deleted, + Leaked, + Held, + Skipped, + /// The enqueue was refused, so the job never ran; held like a lost authority. + EnqueueRefused, +}; + +/// One delete job: written by whoever runs it, read by the round thread after every job finished. +struct JobSlot +{ + std::vector keys; + JobOutcome outcome = JobOutcome::Skipped; + std::exception_ptr error; +}; + +/// The page whose jobs are in flight. It joins the published prefix once its jobs settle without a hold. +struct PendingPage +{ + String next_cursor; + bool decided = false; +}; + +String describeKeys(const std::vector & keys) +{ + if (keys.size() == 1) + return "'" + keys.front().str() + "'"; + return fmt::format("'{}' .. '{}' ({} keys)", keys.front().str(), keys.back().str(), keys.size()); } -NamespaceJanitorResult NamespaceJanitor::runOnePage(bool suppress_deletes, Liveness liveness) +/// Holding the cursor on an ordinary store failure would let one key that always fails starve every page +/// behind it; leaking costs one pass and never deletes wrongly. So only lost authority, `NOT_IMPLEMENTED` +/// and a deterministic local failure hold the page. Every other failure leaks after its last attempt, whether the +/// policy refused it at once or ran out of retries. +JobOutcome classifyFailure(const std::exception_ptr & error, const CasOperation & job_op) { - NamespaceJanitorResult result; - CasOperation op = requests.admit(std::move(liveness)); - const GcMaintenanceReadResult progress = readGcMaintenanceState(op, layout); - if (progress.status == GcMaintenanceReadStatus::Corrupt) + /// A lost-authority refusal reads false when `admitted` re-samples, whatever the exception says. + /// A budget refusal can read true again after a renewal, so it falls through to the exception. + if (!job_op.admitted()) + return JobOutcome::Held; + try { - result.anomalies.push_back(progress.diagnostic); - throwOnRefusedOrGaveUp( - casGcMaintenanceState(op, layout, progress.etag, GcMaintenanceState{}, Retry::standard()), - "CAS namespace janitor: corrupt maintenance-state reset"); - return result; + std::rethrow_exception(error); } + catch (const DB::Exception & e) + { + return isDeterministicLocalFailure(e.code()) ? JobOutcome::Held : JobOutcome::Leaked; + } + catch (const Poco::Exception &) + { + return JobOutcome::Leaked; + } + catch (...) + { + return JobOutcome::Held; + } +} - const String cursor = progress.state ? progress.state->janitor_cursor : String{}; - ListPage page; +/// Runs one job on its own operation, resumed from the phase's generation, so the mount fence and the +/// liveness predicate admit it on its own. A failed job leaks or holds by `classifyFailure`. +void runJob(JobSlot & slot, CasRequests & requests, uint64_t generation, const Liveness & liveness, std::atomic & stop) +{ + if (stop.load()) + return; + CasOperation job_op = requests.resume(generation, liveness); try { - page = op.list(layout.namespaceRootPrefix(), cursor, page_budget, Retry::standard()); + job_op.removeManyWriteOnce(slot.keys, Retry::standard()); + slot.outcome = JobOutcome::Deleted; } catch (...) { - (void)casGcMaintenanceState(op, layout, progress.etag, GcMaintenanceState{}, Retry::once()); - throw; + slot.error = std::current_exception(); + slot.outcome = classifyFailure(slot.error, job_op); + if (slot.outcome == JobOutcome::Held) + stop.store(true); } - result.pages = 1; - result.keys = page.keys.size(); +} - const CasRefCatalog::Snapshot catalog_cut = CasRefCatalog::read(op, layout); - bool ambiguous = false; +/// The exact-token delete of one listed `_ckpt`/`_files` key; false when admission was lost. +bool removeExactly(CasOperation & op, const ListedKey & listed, NamespaceJanitorResult & out) +{ + std::optional etag = listed.etag; + if (!etag) + { + try + { + const std::optional current = op.head(listed.key, Retry::standard()); + if (!current) + return true; + etag = current->etag; + } + catch (const std::exception & e) + { + ++out.leaked; + out.anomalies.push_back("leaked dead-life object '" + listed.key + "': exact HEAD failed: " + e.what()); + return true; + } + } + if (!op.admitted()) + return false; try { - catalog_cut.life_index.throwIfAmbiguous("CAS namespace janitor"); + if (op.remove(listed.key, *etag, Retry::standard()) == Removal::Removed) + ++out.deleted; } - catch (const DB::Exception & e) + catch (const std::exception & e) { - result.anomalies.push_back(e.message()); - ambiguous = true; + ++out.leaked; + out.anomalies.push_back("leaked dead-life object '" + listed.key + "': exact delete failed: " + e.what()); } - /// A valid page is complete only when the round had deletion authority for every dead-life - /// candidate on it. Advancing while the global gate is closed can phase-lock a dead page onto - /// every suppressed round and a different page onto every bounded forced fold. An ambiguous cut - /// retains the old cursor so an authoritative round retries the exact page; a lost liveness sample - /// only reaches this retained-cursor path when it is caught between the two `op.admitted()` checks - /// below -- a sample lost earlier throws out of a read verb (the maintenance read, the list, or a - /// HEAD) before this line is ever reached, ending the page by exception instead. Malformed keys, - /// absent objects and token mismatches are final per-key outcomes and therefore do not by - /// themselves prevent progress. - bool page_decided = !ambiguous && !suppress_deletes; - - for (const ListedKey & listed : page.keys) - { - std::optional life_id; - try + return true; +} + +} + +void NamespaceJanitor::run(bool suppress_deletes, const JanitorRunContext & context, NamespaceJanitorResult & out) +{ + const auto now = [&] { return context.now_ms ? context.now_ms() : uint64_t{0}; }; + const uint64_t phase_start = now(); + /// Saturates: a clock read below the start is no time spent, never a wrapped huge one. + const auto elapsed = [&] + { + const uint64_t current = now(); + return current > phase_start ? current - phase_start : uint64_t{0}; + }; + const size_t batch_keys = std::clamp(context.batch_keys, 1, kBulkDeleteMaxKeys); + out.batch_keys = batch_keys; + + if (context.refresh_authority) + context.refresh_authority(); + CasOperation op = requests.admit(context.liveness); + const GcMaintenanceReadResult progress = readGcMaintenanceState(op, layout); + if (progress.status == GcMaintenanceReadStatus::Corrupt) + { + out.anomalies.push_back(progress.diagnostic); + throwOnRefusedOrGaveUp( + casGcMaintenanceState(op, layout, progress.etag, GcMaintenanceState{}, Retry::standard()), + "CAS namespace janitor: corrupt maintenance-state reset"); + return; + } + const String start_cursor = progress.state ? progress.state->janitor_cursor : String{}; + + /// The jobs of the page in progress. A deque, because running jobs hold pointers to their slots. + std::deque slots; + std::atomic stop{false}; + const uint64_t generation = op.generation(); + + std::optional> runner; + if (context.io_pool) + runner.emplace(*context.io_pool, ThreadName::CAS_GC_JANITOR); + /// Handle `i` belongs to `slots[i]`: a refused enqueue is the last slot of its page and has no handle. + std::vector::Task>> handles; + /// Every exit waits for every job: jobs reference `slots` and `stop`. + SCOPE_EXIT_SAFE({ ThreadPoolCallbackRunnerLocal::waitForAllToFinish(handles); }); + size_t enqueued = 0; + + /// An incomplete page (undecided, or with a held or skipped job) always ends the loop, so the complete + /// pages are a prefix and the last complete page holds the cursor to publish. + std::optional pending; + std::optional publish_cursor; + uint64_t publish_leaked = 0; + + /// Waits for the pending page's jobs and folds them into `out`. + const auto settlePage = [&] + { + ThreadPoolCallbackRunnerLocal::waitForAllToFinish(handles); + for (size_t i = 0; i < handles.size(); ++i) { - if (listed.key.starts_with(layout.namespaceStreamRootPrefix())) + try { - if (const auto parsed = layout.parseRefObjectKey(listed.key)) - life_id = parsed->life_id; + handles[i]->future.get(); } - else if (listed.key.starts_with(layout.namespaceStateRootPrefix())) + catch (...) { - if (const auto parsed = layout.parseRefCkptKey(listed.key)) - life_id = *parsed; - else if (const auto file_parsed = layout.parseNamespaceFileKey(listed.key)) - life_id = file_parsed->life_id; + slots[i].outcome = JobOutcome::Held; + slots[i].error = std::current_exception(); + stop.store(true); } } - catch (const DB::Exception & e) + handles.clear(); + + bool page_held = pending && !pending->decided; + uint64_t page_leaked = 0; + for (const JobSlot & slot : slots) { - result.anomalies.push_back(listed.key + ": " + e.message()); - continue; + if (slot.outcome != JobOutcome::Skipped && slot.outcome != JobOutcome::EnqueueRefused) + ++out.batches; + switch (slot.outcome) + { + case JobOutcome::Deleted: + out.deleted += slot.keys.size(); + break; + case JobOutcome::Leaked: + page_leaked += slot.keys.size(); + ++out.batches_leaked; + out.anomalies.push_back("leaked dead-life objects " + describeKeys(slot.keys) + ": " + getExceptionMessage(slot.error, false)); + break; + case JobOutcome::Held: + case JobOutcome::EnqueueRefused: + ++out.batches_held; + page_held = true; + out.anomalies.push_back("held dead-life objects " + describeKeys(slot.keys) + ": " + getExceptionMessage(slot.error, false)); + break; + case JobOutcome::Skipped: + ++out.delete_jobs_skipped; + page_held = true; + break; + } } - - if (!life_id) + slots.clear(); + if (pending && !page_held) { - result.anomalies.push_back(listed.key + ": unrecognized namespace object key"); - continue; + publish_cursor = std::move(pending->next_cursor); + publish_leaked += page_leaked; } - if (ambiguous || suppress_deletes || catalog_cut.life_index.resolve(*life_id)) - continue; + pending.reset(); + }; - std::optional etag = listed.etag; - if (!etag) + const auto submit = [&](size_t page_index, size_t job_index, std::vector keys) + { + if (stop.load()) + return false; + JobSlot & slot = slots.emplace_back(); + slot.keys = std::move(keys); + const auto job = [&context, &job_requests = requests, &stop, generation, slot_ptr = &slot, page_index, job_index] { - try + if (context.on_job_start_for_test) + context.on_job_start_for_test(page_index, job_index); + runJob(*slot_ptr, job_requests, generation, context.liveness, stop); + }; + if (!runner) + { + ++out.delete_jobs; + job(); + return true; + } + try + { + if (context.schedule_refuse_at_for_test && *context.schedule_refuse_at_for_test == enqueued) { - const std::optional current = op.head(listed.key, Retry::standard()); - if (!current) - continue; - etag = current->etag; + context.schedule_refuse_at_for_test->reset(); + throw Exception(ErrorCodes::CANNOT_SCHEDULE_TASK, "Injected CAS namespace janitor enqueue refusal at job {}", enqueued); } - catch (const std::exception & e) + handles.emplace_back(runner->enqueueAndGiveOwnership(job)); + } + catch (...) + { + slot.outcome = JobOutcome::EnqueueRefused; + slot.error = std::current_exception(); + stop.store(true); + return false; + } + ++enqueued; + ++out.delete_jobs; + return true; + }; + + String cursor = start_cursor; + bool wrapped = false; + for (size_t page_index = 0;; ++page_index) + { + if (page_index > 0) + { + /// Authority is monotone within the phase: the loop breaks on the first loss a refresh reports and never + /// refreshes after it, so a request that saw `false` is never contradicted by a later `true`. + /// A gate-refused exact `HEAD` also sees a loss, but it records a leak and the page continues. + if (context.refresh_authority) + context.refresh_authority(); + if (!op.admitted()) + break; + } + + ListPage page; + try + { + page = op.list(layout.namespaceRootPrefix(), cursor, page_keys, Retry::standard()); + } + catch (...) + { + if (page_index > 0) { - ++result.leaked; - result.anomalies.push_back( - "leaked dead-life object '" + listed.key + "': exact HEAD failed: " + e.what()); - continue; + out.anomalies.push_back("namespace page LIST failed: " + getCurrentExceptionMessage(false)); + break; } + /// Only the persisted cursor is reset: a cursor the store rejects would fail every round. + (void)casGcMaintenanceState(op, layout, progress.etag, GcMaintenanceState{}, Retry::once()); + throw; + } + /// The previous page's jobs ran during the LIST; a hold among them ends the phase before this page is used. + settlePage(); + if (stop.load()) + break; + ++out.pages; + /// The pass already handled everything past the start cursor; do not classify or delete it twice. + if (wrapped) + std::erase_if(page.keys, [&](const ListedKey & listed) { return listed.key > start_cursor; }); + out.keys += page.keys.size(); + + std::optional catalog_cut; + try + { + catalog_cut.emplace(CasRefCatalog::read(op, layout)); } - if (!op.admitted()) + catch (...) { - page_decided = false; + if (page_index == 0) + throw; + out.anomalies.push_back("namespace page catalog read failed: " + getCurrentExceptionMessage(false)); break; } + + bool ambiguous = false; try { - if (op.remove(listed.key, *etag, Retry::standard()) == Removal::Removed) - ++result.deleted; + catalog_cut->life_index.throwIfAmbiguous("CAS namespace janitor"); } - catch (const std::exception & e) + catch (const DB::Exception & e) { - ++result.leaked; - result.anomalies.push_back( - "leaked dead-life object '" + listed.key + "': exact delete failed: " + e.what()); + out.anomalies.push_back(e.message()); + ambiguous = true; + } + /// A page is decided only when the round had deletion authority for every dead-life candidate on it. + /// Advancing while the global gate is closed could phase-lock a dead page onto every suppressed round. + /// Malformed keys, absent objects and token mismatches are final per-key outcomes. + bool decided = !ambiguous && !suppress_deletes; + bool dead_candidate = false; + std::vector exact_keys; + std::vector stream_keys; + for (const ListedKey & listed : page.keys) + { + std::optional life_id; + std::optional stream_key; + try + { + if (listed.key.starts_with(layout.namespaceStreamRootPrefix())) + { + stream_key = layout.parseRefObjectKey(listed.key); + if (stream_key) + life_id = stream_key->life_id; + } + else if (listed.key.starts_with(layout.namespaceStateRootPrefix())) + { + if (const auto parsed = layout.parseRefCkptKey(listed.key)) + life_id = *parsed; + else if (const auto file_parsed = layout.parseNamespaceFileKey(listed.key)) + life_id = file_parsed->life_id; + } + } + catch (const DB::Exception & e) + { + out.anomalies.push_back(listed.key + ": " + e.message()); + continue; + } + if (!life_id) + { + out.anomalies.push_back(listed.key + ": unrecognized namespace object key"); + continue; + } + if (ambiguous || suppress_deletes || catalog_cut->life_index.resolve(*life_id)) + continue; + dead_candidate = true; + /// A dead life's `_log`/`_snap` are write-once and never reborn under its prefix, so they need no token. + if (stream_key) + { + if (std::optional write_once = layout.writeOnceStreamKey(*stream_key, listed.key)) + { + stream_keys.push_back(std::move(*write_once)); + continue; + } + } + exact_keys.push_back(&listed); } - } - /// Recheck even when the page had no dead candidate. A tenure that observes fence loss after LIST - /// or after the last exact delete must not publish progress. Loss after this check may still race - /// with the leak-only maintenance CAS; already completed exact deletes remain safe to repeat. - if (page_decided && !op.admitted()) - page_decided = false; + for (const ListedKey * listed : exact_keys) + { + if (!removeExactly(op, *listed, out)) + { + decided = false; + break; + } + } + size_t job_index = 0; + /// Reserved up front so no push after an enqueue can throw and orphan the scheduled job. + if (runner && decided) + handles.reserve((stream_keys.size() + batch_keys - 1) / batch_keys); + for (size_t begin = 0; decided && begin < stream_keys.size(); begin += batch_keys) + { + if (!op.admitted()) + { + decided = false; + break; + } + const size_t end = std::min(stream_keys.size(), begin + batch_keys); + if (!submit(page_index, job_index++, std::vector(stream_keys.begin() + begin, stream_keys.begin() + end))) + { + break; + } + } + /// A wrapped page publishes an empty cursor: the pass has covered the whole stream. + pending = PendingPage{.next_cursor = wrapped ? String{} : page.next_cursor, .decided = decided}; - if (page_decided) - { - const GcMaintenanceState next{.janitor_cursor = page.next_cursor}; - try + if (!decided || !dead_candidate || stop.load()) + break; + const bool wrap_here = page.next_cursor.empty() && !wrapped && !start_cursor.empty(); + /// A wrapped pass ends at the first page that reaches the start cursor. + if ((page.next_cursor.empty() && !wrap_here) || (wrapped && page.next_cursor >= start_cursor)) + break; + if (elapsed() >= context.budget_ms) { - const WriteResult published = casGcMaintenanceState(op, layout, progress.etag, next, Retry::standard()); - if (std::holds_alternative(published) || std::holds_alternative(published)) - result.anomalies.push_back("cursor publication did not commit"); + out.budget_exhausted = true; + break; } - catch (const std::exception & e) + /// Wrap once: without it a dropped table costs an extra round, because the fold round's pass + /// advances one live page past the debris and the keys before that page wait for the next round. + wrapped = wrapped || wrap_here; + cursor = wrap_here ? String{} : page.next_cursor; + } + + settlePage(); + /// Re-checked even when no page had a dead candidate: a tenure that lost its fence must not publish. + if (!publish_cursor || !op.admitted()) + return; + try + { + const WriteResult published = casGcMaintenanceState( + op, layout, progress.etag, GcMaintenanceState{.janitor_cursor = *publish_cursor}, Retry::standard()); + if (std::holds_alternative(published) || std::holds_alternative(published)) + out.anomalies.push_back("cursor publication did not commit"); + else if (std::holds_alternative(published)) { - result.anomalies.push_back("cursor publication failed: " + String(e.what())); + out.cursor_advanced = true; + /// Only a published page leaves its failed keys behind; an unpublished one is listed again next round. + out.leaked += publish_leaked; } } - return result; + catch (const std::exception & e) + { + out.anomalies.push_back("cursor publication failed: " + String(e.what())); + } } } diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasNamespaceJanitor.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasNamespaceJanitor.h index d23d21402c73..4c85ec574bcf 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasNamespaceJanitor.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasNamespaceJanitor.h @@ -1,42 +1,83 @@ #pragma once #include #include +#include #include +#include +#include #include namespace DB::Cas { +/// Counts of one janitor phase, summed over its pages. struct NamespaceJanitorResult { uint64_t pages = 0; uint64_t keys = 0; uint64_t deleted = 0; uint64_t leaked = 0; + uint64_t batches = 0; + uint64_t batches_leaked = 0; + uint64_t batches_held = 0; + uint64_t delete_jobs = 0; + uint64_t delete_jobs_skipped = 0; + uint64_t batch_keys = 0; + bool budget_exhausted = false; + bool cursor_advanced = false; std::vector anomalies; }; -/// Runs one bounded, leak-only page over the physical namespace ownership tree. +/// What one janitor phase runs with. +struct JanitorRunContext +{ + /// Sampled before every request; cheap and never throws. + Liveness liveness; + /// Re-reads authority before each page; `liveness` answers its verdict. Empty: no refresh. + std::function refresh_authority; + /// Keys per delete job, fixed for the phase; clamped to [1, `kBulkDeleteMaxKeys`]. + size_t batch_keys = kBulkDeleteMaxKeys; + /// No page starts once this much time has passed since the phase began; the page in progress finishes. + uint64_t budget_ms = std::numeric_limits::max(); + /// Phase clock; empty reads as a clock that never moves. + std::function now_ms; + /// Delete jobs run here, so its thread count bounds them; null runs them inline on the caller's thread, + /// in submit order, which keeps unit tests deterministic. A page's jobs are all enqueued before the next + /// page is listed. + ThreadPool * io_pool = nullptr; + /// TEST SEAM: the enqueue at this index within the phase throws; reset once it fires. + std::optional * schedule_refuse_at_for_test = nullptr; + /// TEST SEAM: runs in each job before its stop-flag check and its admission. + std::function on_job_start_for_test; +}; + +/// Reclaims the objects of namespace lives absent from the catalog, page by page from the persisted cursor. class NamespaceJanitor { public: - NamespaceJanitor(CasRequests & requests_, const Layout & layout_, size_t page_budget_) - : requests(requests_), layout(layout_), page_budget(page_budget_) {} + NamespaceJanitor(CasRequests & requests_, const Layout & layout_, size_t page_keys_) + : requests(requests_), layout(layout_), page_keys(page_keys_) {} - /// `liveness` is admitted once for the whole page (one `CasOperation` covers the read, the list, - /// every delete and the cursor publication): a fact the fence cannot see, such as "this tenure - /// still holds the GC round's own lease" -- see `CasRequests::admit`. It is SAMPLED BEFORE EVERY - /// REQUEST the page makes (and before every reissue of one), not just at the two points this - /// function itself checks `op.admitted()` -- so it must be cheap and must never throw. A sample - /// that returns false ends whichever request was about to be sent: a read verb (the maintenance - /// read, the list, a HEAD) throws out of this call, and a write verb (a delete, the cursor - /// publication) reports it as `GaveUp` rather than sending anything. - NamespaceJanitorResult runOnePage(bool suppress_deletes, Liveness liveness); + /// One phase. Each page: refresh authority, LIST, one catalog cut after the LIST, exact-token deletes of + /// dead `_ckpt`/`_files`, then jobs of `batch_keys` token-free deletes of dead canonical `_log`/`_snap`. + /// Another page starts only within the budget, while no job has held, after a decided page that had a + /// dead-life candidate and a next cursor. A pass that began mid-stream wraps once: after the last page it + /// continues from the stream start, skips keys past the start cursor and ends at the page that reaches it; + /// a wrapped page publishes an empty cursor. At the end one publication of the cursor after the longest + /// prefix of complete pages. `suppress_deletes` lists one page and parses its keys without classifying them, + /// deletes nothing and publishes no cursor, but still writes a cursor reset when the state is corrupt or the LIST fails. + /// A page's jobs run while the next page is listed and are settled before that page is used. + /// Jobs share a phase-wide stop flag: a job that holds sets it, a job that finds it set sends nothing, and the + /// caller stops submitting. `liveness` must be monotone within the phase: once it returns `false` it stays false. + /// Throws when the maintenance read, the first page's LIST (after resetting the cursor) or its catalog + /// read fails; `out` keeps the counts gathered so far (`pages` and `keys` are added before a page's catalog read, + /// so a page whose read throws is still counted) and no cursor is published. + void run(bool suppress_deletes, const JanitorRunContext & context, NamespaceJanitorResult & out); private: CasRequests & requests; const Layout & layout; - size_t page_budget; + size_t page_keys; }; } diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Tools/CasFsck.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Tools/CasFsck.cpp index 8b67d00251a1..b4105290318d 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Tools/CasFsck.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Tools/CasFsck.cpp @@ -510,7 +510,7 @@ void runFsckImpl(Pool & store, bool detail, const FsckProgress & on_progress, co /// Physical life-owned keys carry no logical name. Classify each COMPLETE, canonical key against a /// catalog cut taken AFTER this physical listing finishes (observe-then-cut), not the earlier - /// `catalog_cut` above: `NamespaceJanitor::runOnePage` (the only real deleter of this debris) uses + /// `catalog_cut` above: `NamespaceJanitor::run` (the only real deleter of this debris) uses /// the identical ordering, and it is what makes "life absent from a LATER cut" sound -- creation /// always admits a `Creating` catalog row before writing any life-owned object (spec §2), so a life /// that is absent from a cut taken after the listing cannot be a concurrent birth this listing raced. diff --git a/src/Disks/DiskObjectStorage/ObjectStorages/IObjectStorage.h b/src/Disks/DiskObjectStorage/ObjectStorages/IObjectStorage.h index 436f8c05bd51..00a2a0d21c44 100644 --- a/src/Disks/DiskObjectStorage/ObjectStorages/IObjectStorage.h +++ b/src/Disks/DiskObjectStorage/ObjectStorages/IObjectStorage.h @@ -380,6 +380,11 @@ class IObjectStorage virtual void removeObjectsIfExistUnderProfile( const StoredObjects & objects, const ObjectStorageControlRequest & request); + /// The most objects one `removeObjectsIfExistUnderProfile` call sends as one request; at least 1. + /// Callers cut their batches to this size, so one call stays one storage request. The default + /// suits a storage without a batch delete. + virtual size_t batchDeleteKeyLimit() const { return 1; } + /// Copy object with different attributes if required virtual void copyObject( /// NOLINT const StoredObject & object_from, diff --git a/src/Disks/DiskObjectStorage/ObjectStorages/S3/S3ObjectStorage.cpp b/src/Disks/DiskObjectStorage/ObjectStorages/S3/S3ObjectStorage.cpp index 6eb155dd4c44..b3350868eb05 100644 --- a/src/Disks/DiskObjectStorage/ObjectStorages/S3/S3ObjectStorage.cpp +++ b/src/Disks/DiskObjectStorage/ObjectStorages/S3/S3ObjectStorage.cpp @@ -675,9 +675,8 @@ void S3ObjectStorage::removeObjectsIfExistImpl( return; /// A batch of exactly one object is a plain `DeleteObject`, never `DeleteObjects` -- the same rule - /// `deleteFilesFromS3` applies to a single key. This is what makes the CAS-side per-key fallback - /// work on a backend with no `DeleteObjects` at all (GCS): that backend rejects the verb itself, not - /// a key count, so a "batch" of one object sent as `DeleteObjects` would fail there too. + /// `deleteFilesFromS3` applies to a single key. A storage without `DeleteObjects` (GCS) + /// rejects the verb itself, so a one-object `DeleteObjects` would fail there too. if (objects.size() == 1) { const StoredObject & object = objects.front(); @@ -688,11 +687,9 @@ void S3ObjectStorage::removeObjectsIfExistImpl( } /// GCS has no `DeleteObjects`: a capability the config declared false, or that an earlier batch - /// attempt on this same storage already learned false, must not be retried here. This storage never - /// loops over `objects` itself to work around it -- a CAS caller admits one request per physical - /// delete (see `ObjectStorageBackend::removeManyWriteOnce` and its own caller in CasGc.cpp), which an - /// internal loop over more than one object, running under a SINGLE admission, cannot be. Report the - /// absence of the capability instead, and let that caller decide how to retry. + /// attempt on this storage already learned false, is reported rather than retried. This storage never + /// loops over `objects` itself: a loop under one admission would delete without the per-request checks + /// the caller applies to each request. if (auto support_batch_delete = s3_capabilities.isBatchDeleteSupported(); support_batch_delete.has_value() && !support_batch_delete.value()) throw Exception(ErrorCodes::NOT_IMPLEMENTED, "{} does not support DeleteObjects", getName()); @@ -748,12 +745,15 @@ void S3ObjectStorage::removeObjectsIfExistImpl( throw Exception(ErrorCodes::NOT_IMPLEMENTED, "{} does not support DeleteObjects", getName()); } - throw S3Exception(err.GetErrorType(), "{} (Code: {}) while removing {} objects from S3 in one request", - err.GetMessage(), static_cast(err.GetErrorType()), objects.size()); + throw S3Exception( + PreformattedMessage::create("{} (Code: {}) while removing {} objects from S3 in one request", + err.GetMessage(), static_cast(err.GetErrorType()), objects.size()), + err.GetErrorType(), err.GetExceptionName()); } String failed_keys; std::optional first_error_type; + String first_error_name; for (const auto & err : outcome.GetResult().GetErrors()) { const auto error_type = classifyDeleteObjectsErrorCode(err.GetCode()); @@ -763,10 +763,21 @@ void S3ObjectStorage::removeObjectsIfExistImpl( failed_keys += ", "; failed_keys += err.GetKey() + " (" + err.GetCode() + ": " + err.GetMessage() + ")"; if (!first_error_type) + { first_error_type = error_type; + first_error_name = err.GetCode(); + } } if (first_error_type) - throw S3Exception(*first_error_type, "batch removal left objects behind: [{}]", failed_keys); + throw S3Exception( + PreformattedMessage::create("batch removal left objects behind: [{}]", failed_keys), *first_error_type, first_error_name); +} + +size_t S3ObjectStorage::batchDeleteKeyLimit() const +{ + if (const auto supported = s3_capabilities.isBatchDeleteSupported(); supported.has_value() && !*supported) + return 1; + return std::max(1, s3_settings.get()->request_settings[S3RequestSetting::objects_chunk_size_to_delete]); } bool S3ObjectStorage::conditionalOpsUseGenerationTokens() const diff --git a/src/Disks/DiskObjectStorage/ObjectStorages/S3/S3ObjectStorage.h b/src/Disks/DiskObjectStorage/ObjectStorages/S3/S3ObjectStorage.h index 5f5903f2c6aa..74a675cc1dac 100644 --- a/src/Disks/DiskObjectStorage/ObjectStorages/S3/S3ObjectStorage.h +++ b/src/Disks/DiskObjectStorage/ObjectStorages/S3/S3ObjectStorage.h @@ -132,11 +132,13 @@ class S3ObjectStorage : public IObjectStorage /// single physical request per call, so there is nothing here for that capability to say no to). /// For more than one object, throws `NOT_IMPLEMENTED` without sending anything once `DeleteObjects` /// is known unsupported (a configured or a just-learned `S3Capabilities::isBatchDeleteSupported() == - /// false`) -- this storage never substitutes a per-key loop of its own, since the caller is the one - /// that can admit each physical delete as its own request. + /// false`) -- this storage never substitutes a per-key loop of its own. void removeObjectsIfExistUnderProfile( const StoredObjects & objects, const ObjectStorageControlRequest & request) override; + /// `objects_chunk_size_to_delete`, at least 1, while `DeleteObjects` is not known unsupported; 1 after. + size_t batchDeleteKeyLimit() const override; + void tagObjects(const StoredObjects & objects, const std::string & tag_key, const std::string & tag_value) override; ObjectMetadata getObjectMetadata(const std::string & path, bool with_tags) const override; diff --git a/src/Disks/tests/cas_namespace_janitor_test_helpers.h b/src/Disks/tests/cas_namespace_janitor_test_helpers.h new file mode 100644 index 000000000000..72ead092b658 --- /dev/null +++ b/src/Disks/tests/cas_namespace_janitor_test_helpers.h @@ -0,0 +1,247 @@ +#pragma once + +#include "cas_test_helpers.h" +#include +#include +#include +#include +#include +#include +#include +#include +#include + +namespace CurrentMetrics +{ + extern const Metric LocalThread; + extern const Metric LocalThreadActive; + extern const Metric LocalThreadScheduled; +} + +namespace DB::Cas::tests::janitor +{ + +inline std::unique_ptr makeJobPool(size_t threads) +{ + return std::make_unique( + CurrentMetrics::LocalThread, CurrentMetrics::LocalThreadActive, CurrentMetrics::LocalThreadScheduled, + /*max_threads*/ threads, /*max_free_threads*/ threads, /*queue_size*/ 0); +} + +inline void createObj(Backend & backend, const String & key, const String & bytes) +{ + OperationForTest op(backend); + ASSERT_TRUE(std::holds_alternative((*op).create(key, bytes, Retry::once()))) << key; +} + +inline std::optional readObj(Backend & backend, const String & key) +{ + OperationForTest op(backend); + return (*op).read(key, Retry::standard()); +} + +inline bool present(Backend & backend, const String & key) +{ + return readObj(backend, key).has_value(); +} + +inline void seedCatalog(Backend & backend, const Layout & layout, RefCatalog catalog = {}) +{ + createObj(backend, layout.refCatalogKey(), encodeRefCatalog(catalog)); +} + +inline NamespaceLifeId life(const char * name, uint64_t id) +{ + return NamespaceLifeId::fromCatalogEntry(RootNamespace{name}, UInt128{id}); +} + +/// `readGcMaintenanceState` takes an admitted operation, which cannot bind to an rvalue. +inline GcMaintenanceReadResult readState(CasRequests & requests, const Layout & layout) +{ + auto op = requests.admit(); + return readGcMaintenanceState(op, layout); +} + +/// The published janitor cursor; `std::nullopt` while nothing was ever published. +inline std::optional publishedCursor(CasRequests & requests, const Layout & layout) +{ + const GcMaintenanceReadResult state = readState(requests, layout); + if (state.status != GcMaintenanceReadStatus::Valid || !state.state) + return std::nullopt; + return state.state->janitor_cursor; +} + +/// `count` `_log` keys of `owner`, in listing order. +inline std::vector seedLogs(Backend & backend, const Layout & layout, const NamespaceLifeId & owner, uint64_t count) +{ + std::vector keys; + keys.reserve(count); + for (uint64_t i = 1; i <= count; ++i) + keys.push_back(layout.refLogKey(owner, RefTxnId{1, i})); + std::sort(keys.begin(), keys.end()); + for (const String & key : keys) + createObj(backend, key, "log"); + return keys; +} + +/// A live namespace with a checkpoint, the shape a GC round's fold expects; returns its life. +inline NamespaceLifeId seedLiveNamespace(Backend & backend, const Layout & layout) +{ + const RootNamespace live_namespace{"00/live@cas@"}; + fixture::admitLive(backend, layout, live_namespace); + const NamespaceLifeId live = fixture::fixtureLife(live_namespace); + createObj(backend, layout.refCkptKey(live), + encodeRefCkpt(RefCkpt{.life_epoch = std::optional{1}, .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = std::nullopt})); + return live; +} + +inline NamespaceJanitorResult runPhase(CasRequests & requests, const Layout & layout, const JanitorRunContext & context, + bool suppress_deletes = false, size_t page_keys = 1000) +{ + NamespaceJanitorResult result; + NamespaceJanitor(requests, layout, page_keys).run(suppress_deletes, context, result); + return result; +} + +/// One page, no refresh, sequential: the contract the single-page tests were written against. +inline NamespaceJanitorResult runOnePage(NamespaceJanitor janitor, bool suppress_deletes, Liveness liveness) +{ + NamespaceJanitorResult result; + JanitorRunContext context; + context.liveness = std::move(liveness); + context.budget_ms = 0; + janitor.run(suppress_deletes, context, result); + return result; +} + +/// The batch-capability store with the janitor's requests counted apart and hooks at fixed points. Hooks +/// run with no backend lock held; their own store access goes through the uncounted primitives. +class JanitorBackend : public BatchCapabilityBackend +{ +public: + using Access = TransportAccess; + using Keys = std::vector; + + struct BulkCall + { + std::vector keys; + std::thread::id thread; + }; + + RawListPage list(const String & prefix, const String & cursor, size_t limit, Access & access) override + { + RawListPage page = BatchCapabilityBackend::list(prefix, cursor, limit, access); + if (!prefix.ends_with("/cas/ns/")) + return page; + if (tokenless) + for (auto & key : page.keys) + key.value.reset(); + const size_t index = namespace_lists.fetch_add(1); + if (after_namespace_list) + after_namespace_list(index, access); + return page; + } + + bool supportsListTokens() const override { return !tokenless; } + + RawRemoval remove(const String & key, const String & expected_value, Access & access) override + { + { + std::lock_guard lock(calls_mutex); + exact_removes.push_back(key); + } + if (before_exact_remove) + before_exact_remove(key, access); + return BatchCapabilityBackend::remove(key, expected_value, access); + } + + void removeManyWriteOnce(const Keys & keys, Access & access) override + { + { + std::lock_guard lock(calls_mutex); + BulkCall call{.keys = {}, .thread = std::this_thread::get_id()}; + for (const WriteOnceKey & key : keys) + call.keys.push_back(key.str()); + bulk_calls.push_back(std::move(call)); + } + if (before_bulk) + before_bulk(keys, access); + BatchCapabilityBackend::removeManyWriteOnce(keys, access); + if (after_bulk) + after_bulk(keys); + } + + std::vector bulkCalls() const + { + std::lock_guard lock(calls_mutex); + return bulk_calls; + } + + std::vector exactRemoves() const + { + std::lock_guard lock(calls_mutex); + return exact_removes; + } + + size_t namespaceLists() const { return namespace_lists.load(); } + + /// A concurrent actor's delete of whatever `key` holds now. + void removeUncounted(const String & key, Access & access) + { + if (const auto current = InMemoryBackend::read(key, access)) // NOLINT(bugprone-parent-virtual-call) + (void)InMemoryBackend::remove(key, current->value, access); // NOLINT(bugprone-parent-virtual-call) + } + + /// A concurrent actor's write: a create when `key` is absent, else a replacement. + void writeUncounted(const String & key, const String & bytes, Access & access) + { + const auto current = InMemoryBackend::read(key, access); // NOLINT(bugprone-parent-virtual-call) + const std::optional expected = current ? std::optional(current->value) : std::nullopt; + ASSERT_TRUE(InMemoryBackend::write(key, bytes, expected, access).has_value()); // NOLINT(bugprone-parent-virtual-call) + } + + bool tokenless = false; + std::function after_namespace_list; + std::function before_exact_remove; + std::function before_bulk; + std::function after_bulk; + +private: + mutable std::mutex calls_mutex; + std::vector bulk_calls; + std::vector exact_removes; + std::atomic namespace_lists{0}; +}; + +inline const Layout & janitorLayout() +{ + static const Layout layout("p"); + return layout; +} + +/// A pool-less janitor: an open fence, a catalog, and a request and phase clock the test drives. +struct JanitorFixture +{ + std::shared_ptr backend = std::make_shared(); + FakeClock clock; + CasRequests requests{backend, Fence::open(), clock.nowFn(), clock.sleepFn()}; + + explicit JanitorFixture(RefCatalog catalog = {}) { seedCatalog(*backend, janitorLayout(), std::move(catalog)); } + + JanitorRunContext context(size_t batch_keys = kBulkDeleteMaxKeys) + { + JanitorRunContext result; + result.batch_keys = batch_keys; + result.now_ms = clock.nowFn(); + return result; + } + + NamespaceJanitorResult run(const JanitorRunContext & context, bool suppress_deletes = false) + { + return runPhase(requests, janitorLayout(), context, suppress_deletes); + } + + std::optional cursor() { return publishedCursor(requests, janitorLayout()); } +}; + +} diff --git a/src/Disks/tests/cas_scripted_s3_server.h b/src/Disks/tests/cas_scripted_s3_server.h new file mode 100644 index 000000000000..72fb4f731b36 --- /dev/null +++ b/src/Disks/tests/cas_scripted_s3_server.h @@ -0,0 +1,296 @@ +#pragma once + +#include "config.h" + +#if USE_AWS_S3 + +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#include +#include +#include +#include +#include +#include +#include +#include +#include + +namespace DB::S3RequestSetting +{ + extern const S3RequestSettingsUInt64 objects_chunk_size_to_delete; +} + +namespace DB::Cas::tests::s3 +{ + +/// A local HTTP server standing in for S3. `DeleteObjects` arrives as a POST to the bucket root; a one-key +/// `DeleteObject` arrives as an HTTP DELETE of the key's path. +class ScriptedS3Server +{ +public: + using Responder = std::function; + +private: + class Handler : public Poco::Net::HTTPRequestHandler + { + ScriptedS3Server & owner; + + public: + explicit Handler(ScriptedS3Server & owner_) : owner(owner_) { } + + void handleRequest(Poco::Net::HTTPServerRequest & request, Poco::Net::HTTPServerResponse & response) override + { + { + std::lock_guard lock(owner.mutex); + owner.methods_seen.push_back(request.getMethod()); + } + /// An unread body on a keep-alive connection would be parsed as the start of the next request. + std::string body; + Poco::StreamCopier::copyToString(request.stream(), body); + owner.responder(request, body, response); + } + }; + + class Factory : public Poco::Net::HTTPRequestHandlerFactory + { + ScriptedS3Server & owner; + + Poco::Net::HTTPRequestHandler * createRequestHandler(const Poco::Net::HTTPServerRequest &) override + { + return new Handler(owner); + } + + public: + explicit Factory(ScriptedS3Server & owner_) : owner(owner_) { } + }; + + std::unique_ptr server_socket; + Poco::SharedPtr handler_factory; + Poco::AutoPtr server_params; + std::unique_ptr server; + Responder responder; + mutable std::mutex mutex; + std::vector methods_seen; + +public: + explicit ScriptedS3Server(Responder responder_) + : server_socket(std::make_unique(0)) + , handler_factory(new Factory(*this)) + , server_params(new Poco::Net::HTTPServerParams()) + , server(std::make_unique(handler_factory, *server_socket, server_params)) + , responder(std::move(responder_)) + { + server->start(); + } + + ~ScriptedS3Server() { server->stopAll(true); } + + std::string getUrl() const { return "http://" + server_socket->address().toString(); } + + size_t countMethod(const std::string & method) const + { + std::lock_guard lock(mutex); + return static_cast(std::count(methods_seen.begin(), methods_seen.end(), method)); + } +}; + +/// The keys a scripted server removed, so a test double can mirror them into its own store. +class ScriptedDeletes +{ +public: + void add(const std::string & key) + { + std::lock_guard lock(mutex); + keys.insert(key); + } + + bool contains(const std::string & key) const + { + std::lock_guard lock(mutex); + return keys.contains(key); + } + +private: + mutable std::mutex mutex; + std::set keys; +}; + +/// The object key of a one-key `DeleteObject`: its path after the bucket, without the query string. +inline std::string keyOfDeleteObject(const Poco::Net::HTTPServerRequest & request) +{ + const std::string prefix = "/test-bucket/"; + const std::string & uri = request.getURI(); + const std::string path = uri.substr(0, uri.find('?')); + return path.starts_with(prefix) ? path.substr(prefix.size()) : path; +} + +/// The keys a `DeleteObjects` body names, in body order. +inline std::vector keysOfDeleteObjects(const std::string & body) +{ + std::vector keys; + for (size_t open = body.find(""); open != std::string::npos; open = body.find("", open)) + { + open += 5; + const size_t close = body.find("", open); + keys.push_back(body.substr(open, close - open)); + } + return keys; +} + +inline void sendXml(Poco::Net::HTTPServerResponse & response, Poco::Net::HTTPResponse::HTTPStatus status, const std::string & body) +{ + response.setContentType("application/xml"); + response.setContentLength(body.size()); + response.setStatus(status); + auto & out = response.send(); + out << body; + out.flush(); +} + +/// A quiet-mode `DeleteObjects` answer (HTTP 200) listing only `errors`, as `(key, code)`, in order. +inline void sendDeleteObjectsResult( + Poco::Net::HTTPServerResponse & response, const std::vector> & errors) +{ + std::string body = "" + ""; + for (const auto & [key, code] : errors) + body += "" + key + "" + code + "" + code + ""; + body += ""; + sendXml(response, Poco::Net::HTTPResponse::HTTP_OK, body); +} + +/// A quiet-mode `DeleteObjects` success (HTTP 200) whose body lists only the failed keys, exactly as a +/// real S3 backend would report a mixed outcome. +inline void sendBatchSuccessWithErrors(Poco::Net::HTTPServerResponse & response, const std::string & not_found_key, const std::string & denied_key) +{ + const std::string body = + "" + "" + "" + not_found_key + "NoSuchKeyThe specified key does not exist." + "" + denied_key + "AccessDeniedAccess Denied" + ""; + sendXml(response, Poco::Net::HTTPResponse::HTTP_OK, body); +} + +/// A request-level `DeleteObjects` failure in the "batch delete is not implemented" class that +/// `deleteFileFromS3.cpp`'s `deleteFilesFromS3` also treats as "fall back to plain `DeleteObject`". +inline void sendBatchNotImplemented(Poco::Net::HTTPServerResponse & response) +{ + const std::string body = + "" + "NotImplementedA header you provided implies functionality that is not implemented"; + sendXml(response, Poco::Net::HTTPResponse::HTTP_BAD_REQUEST, body); +} + +/// A request-level `DeleteObjects` failure in an ordinary (not "unsupported") class: this must keep +/// today's fail-close behaviour and never fall back to per-key deletes. +inline void sendBatchInternalError(Poco::Net::HTTPServerResponse & response) +{ + const std::string body = + "" + "InternalErrorWe encountered an internal error, please try again."; + sendXml(response, Poco::Net::HTTPResponse::HTTP_INTERNAL_SERVER_ERROR, body); +} + +inline void sendDeleteObjectSuccess(Poco::Net::HTTPServerResponse & response) +{ + response.setContentLength(0); + response.setStatus(Poco::Net::HTTPResponse::HTTP_NO_CONTENT); + response.send(); +} + +/// A single-key `DeleteObject` failure -- used to script the size-one path's own error handling, as +/// distinct from the batch response's per-key `` elements covered by the test above. +inline void sendSingleDeleteError(Poco::Net::HTTPServerResponse & response, Poco::Net::HTTPResponse::HTTPStatus status, const std::string & code, const std::string & message) +{ + const std::string body = + "" + "" + code + "" + message + ""; + sendXml(response, status, body); +} + +struct StorageOptions +{ + std::optional objects_chunk_size_to_delete{}; + /// Empty: the storage keeps its default callback, which returns no client, so nothing is reissued. + DB::S3ObjectStorage::S3CredentialsRefreshCallback credentials_refresh_callback{}; +}; + +inline std::unique_ptr makeClientForTest(const std::string & endpoint) +{ + DB::RemoteHostFilter remote_host_filter; + DB::S3::PocoHTTPClientConfiguration cfg = DB::S3::ClientFactory::instance().createClientConfiguration( + "us-east-1", + remote_host_filter, + /* s3_max_redirects = */ 100, + DB::S3::PocoHTTPClientConfiguration::RetryStrategy{.max_retries = 0}, + /* s3_slow_all_threads_after_network_error = */ false, + /* s3_slow_all_threads_after_retryable_error = */ false, + /* enable_s3_requests_logging = */ false, + /* for_disk_s3 = */ true, + /* opt_disk_name = */ {}, + /* request_throttler = */ {}); + cfg.endpointOverride = endpoint; + cfg.connectTimeoutMs = 10000; + cfg.requestTimeoutMs = 10000; + cfg.s3_use_adaptive_timeouts = false; + /// Every test here starts its own server on an ephemeral port; with keep-alive on, the process-wide + /// HTTP connection pool can hand a later test a connection to a port whose server is already gone + /// (`Connection reset by peer` under `--gtest_repeat`). One connection per request is what a + /// short-lived test server should get. + cfg.http_keep_alive_timeout = 0; + return DB::S3::ClientFactory::instance().create( + cfg, + DB::S3::ClientSettings{ + .use_virtual_addressing = false, + .disable_checksum = false, + .gcs_issue_compose_request = false, + .is_s3express_bucket = false, + }, + "ACCESS_KEY_ID", "SECRET_ACCESS_KEY", "", {}, {}, DB::S3::CredentialsConfiguration{}); + +} + +inline std::shared_ptr makeStorageForTest( + const std::string & endpoint, const DB::S3Capabilities & capabilities, const StorageOptions & options = {}) +{ + auto settings = std::make_unique(); + if (options.objects_chunk_size_to_delete) + settings->request_settings[DB::S3RequestSetting::objects_chunk_size_to_delete] = *options.objects_chunk_size_to_delete; + const DB::S3::URI uri(endpoint + "/test-bucket/"); + if (options.credentials_refresh_callback) + return std::make_shared( + makeClientForTest(endpoint), std::move(settings), uri, capabilities, DB::ObjectStorageKeyGeneratorPtr{}, + "disk", /*for_disk_s3=*/ true, options.credentials_refresh_callback); + return std::make_shared( + makeClientForTest(endpoint), std::move(settings), uri, capabilities, DB::ObjectStorageKeyGeneratorPtr{}, "disk"); +} + +inline DB::ContextPtr contextForTest() +{ + return getContext().context; +} + +} + +#endif diff --git a/src/Disks/tests/cas_test_helpers.h b/src/Disks/tests/cas_test_helpers.h index ae87f129fce3..b23b416ab1c2 100644 --- a/src/Disks/tests/cas_test_helpers.h +++ b/src/Disks/tests/cas_test_helpers.h @@ -74,6 +74,7 @@ namespace DB::ContentAddressedSetting namespace DB::ErrorCodes { extern const int CORRUPTED_DATA; + extern const int NOT_IMPLEMENTED; } namespace DB::Cas::tests @@ -1972,6 +1973,106 @@ class CountingBackend : public DB::Cas::InMemoryBackend uint64_t publish_total = 0; }; +/// The in-memory store with `S3ObjectStorage`'s batch-delete capability. A `removeManyWriteOnce` of more +/// than one key is refused locally with `NOT_IMPLEMENTED` once the capability is false; a store that rejects +/// `DeleteObjects` turns an unknown capability false on its first multi-key request. One key always goes. +class BatchCapabilityBackend : public CountingBackend +{ +public: + size_t bulkDeleteKeyLimit() const override + { + std::lock_guard lock(capability_mutex); + return batch_supported == std::optional{false} ? 1 : storage_limit; + } + + void removeManyWriteOnce(const std::vector & keys, DB::Cas::TransportAccess & access) override + { + { + std::lock_guard lock(capability_mutex); + call_sizes.push_back(keys.size()); + if (keys.size() > 1 && batch_supported == std::optional{false}) + throw DB::Exception(DB::ErrorCodes::NOT_IMPLEMENTED, "batch delete is known unsupported"); + request_sizes.push_back(keys.size()); + if (keys.size() > 1 && !batch_supported && rejects_batches) + { + batch_supported = false; + throw DB::Exception(DB::ErrorCodes::NOT_IMPLEMENTED, "the store rejected DeleteObjects"); + } + } + CountingBackend::removeManyWriteOnce(keys, access); + } + + void setBatchDeleteSupported(std::optional value) + { + std::lock_guard lock(capability_mutex); + batch_supported = value; + } + + void setStoreRejectsBatches(bool value) + { + std::lock_guard lock(capability_mutex); + rejects_batches = value; + } + + void setStorageLimit(size_t value) + { + std::lock_guard lock(capability_mutex); + storage_limit = value; + } + + /// Every `removeManyWriteOnce` that reached this backend, locally refused ones included. + std::vector callSizes() const + { + std::lock_guard lock(capability_mutex); + return call_sizes; + } + + /// Requests the modeled store received: calls minus local refusals. + std::vector requestSizes() const + { + std::lock_guard lock(capability_mutex); + return request_sizes; + } + +private: + mutable std::mutex capability_mutex; + std::optional batch_supported; + bool rejects_batches = false; + size_t storage_limit = DB::Cas::kBulkDeleteMaxKeys; + std::vector call_sizes; + std::vector request_sizes; +}; + +/// Keeps `row` = the metrics of the last `phase` row, and the catalog and `gc/state` reads made between +/// the end of `before_phase` and the end of `phase`. +struct PhaseReads +{ + std::map row; + uint64_t catalog_before = 0; + uint64_t state_before = 0; + uint64_t catalog_in_phase = 0; + uint64_t state_in_phase = 0; +}; + +inline DB::Cas::GcPhaseSink phaseReadsSink(PhaseReads & reads, const CountingBackend & backend, const DB::Cas::Layout & layout, + const String & before_phase, const String & phase) +{ + return [&reads, &backend, layout, before_phase, phase](const DB::Cas::GcPhaseRecord & record) + { + if (record.phase == before_phase) + { + reads.catalog_before = backend.getCount(layout.refCatalogKey()); + reads.state_before = backend.getCount(layout.gcStateKey()); + } + else if (record.phase == phase) + { + reads.row = record.metrics; + reads.catalog_in_phase = backend.getCount(layout.refCatalogKey()) - reads.catalog_before; + reads.state_in_phase = backend.getCount(layout.gcStateKey()) - reads.state_before; + } + }; +} + /// Records the ORDER of writes (so a test can compare indices) and lets a test refuse or fail chosen /// writes by key. Delegates every request to `CountingBackend` unchanged, so the per-key counters /// remain available as the positive control. diff --git a/src/Disks/tests/gtest_cas_bulk_delete_backend.cpp b/src/Disks/tests/gtest_cas_bulk_delete_backend.cpp index 68aeabeb9523..b41851c8aa30 100644 --- a/src/Disks/tests/gtest_cas_bulk_delete_backend.cpp +++ b/src/Disks/tests/gtest_cas_bulk_delete_backend.cpp @@ -8,6 +8,8 @@ #include #include #include +#include +#include #include #include #include @@ -140,6 +142,14 @@ TEST(CASBulkDeleteBackend, InstrumentedCountsOneRequestAndOneDeletePerKeyClass) } #if USE_AWS_S3 +TEST(CASBulkDeleteBackend, ThrottlingBackendForwardsStorageLimit) +{ + auto inner = std::make_shared(); + inner->setStorageLimit(250); + ThrottlingBackend backend(inner, ThrottlingBackend::Mode::FirstPerKey, 1, 429); + EXPECT_EQ(backend.bulkDeleteKeyLimit(), 250u); +} + TEST(CASBulkDeleteBackend, ThrottlingRefusesTheChunkOnceAndTheEngineReissuesIt) { auto inner = std::make_shared(); @@ -195,3 +205,39 @@ TEST(CASBulkDeleteBackend, LocalObjectStorageRefusesTheProfileOverload) }); } #endif + +TEST(CASBulkDeleteBackend, InMemoryStorageLimitDefaultsToTheCasMaximum) +{ + InMemoryBackend backend; + EXPECT_EQ(backend.bulkDeleteKeyLimit(), kBulkDeleteMaxKeys); +} + +TEST(CASBulkDeleteBackend, PoolWrappedBackendForwardsStorageLimit) +{ + auto backend = std::make_shared(); + backend->setStorageLimit(250); + auto store = DB::Cas::tests::openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/ 0); + ASSERT_NE(dynamic_cast(store->poolBackendPtr().get()), nullptr) + << "the pool must wrap its backend, or this test proves nothing about the decorator"; + EXPECT_EQ(store->poolBackendPtr()->bulkDeleteKeyLimit(), 250u); + Gc gc(store, DB::UInt128{1}); + EXPECT_EQ(gc.bulkDeleteChunkKeys(), 250u); +} + +TEST(CASBulkDeleteBackend, GcChunkIsTheSmallerOfSettingAndStorageLimit) +{ + auto backend = std::make_shared(); + backend->setStorageLimit(700); + auto store = Pool::open(backend, PoolConfig{.pool_prefix = "p", .server_root_id = "test", .gc_bulk_delete_chunk_keys = 300}); + EXPECT_EQ(Gc(store, DB::UInt128{1}).bulkDeleteChunkKeys(), 300u); + backend->setBatchDeleteSupported(false); + EXPECT_EQ(Gc(store, DB::UInt128{1}).bulkDeleteChunkKeys(), 1u); +} + +TEST(CASBulkDeleteBackend, ZeroStorageLimitClampsGcChunkToOneKey) +{ + auto backend = std::make_shared(); + backend->setStorageLimit(0); + auto store = DB::Cas::tests::openPoolForTest(backend, 0); + EXPECT_EQ(Gc(store, DB::UInt128{1}).bulkDeleteChunkKeys(), 1u); +} diff --git a/src/Disks/tests/gtest_cas_gc_bulk_delete_fallback.cpp b/src/Disks/tests/gtest_cas_gc_bulk_delete_fallback.cpp index c490ada2c3ae..8ac94924b291 100644 --- a/src/Disks/tests/gtest_cas_gc_bulk_delete_fallback.cpp +++ b/src/Disks/tests/gtest_cas_gc_bulk_delete_fallback.cpp @@ -11,15 +11,12 @@ #include #include -/// `removeChunkWriteOnceOrOneByOne` (CasGc.h) is what both GC bulk-delete call sites (manifest_deletes' -/// flush() and cleanupRefObjects' chunk loop) use to survive a backend without `DeleteObjects`. Tested -/// here in isolation, directly against the engine, rather than only through the much larger machinery of -/// a full GC round. +/// `removeCohortWriteOnce` (CasGc.h) cuts a cohort into requests of the storage's batch size and never +/// re-sends a rejected request in smaller pieces. namespace DB::ErrorCodes { extern const int CORRUPTED_DATA; -extern const int NETWORK_ERROR; extern const int NOT_IMPLEMENTED; } @@ -54,131 +51,91 @@ PoolPtr openPlainPool(const std::shared_ptr & backend) return Pool::open(backend, config); } -} - -TEST(CASGCBulkDeleteFallback, HappyPathIsOneRequest) +/// Creates `keys`, one committed body each. +void seedBodies(CasOperation & op, const std::vector & keys) { - auto backend = std::make_shared(); - auto store = openPlainPool(backend); - CasOperation op = store->openRequests().admit(); - const std::vector keys = manifestKeys(3); for (const WriteOnceKey & key : keys) ASSERT_TRUE(std::holds_alternative(op.create(key.str(), "b", Retry::once()))); +} - const uint64_t requests_issued = removeChunkWriteOnceOrOneByOne(op, keys, Retry::once()); - - EXPECT_EQ(requests_issued, 1u); - EXPECT_EQ(backend->bulkRemoveCalls(), 1u); - for (const WriteOnceKey & key : keys) - EXPECT_FALSE(op.head(key.str(), Retry::once()).has_value()) << key.str(); } -TEST(CASGCBulkDeleteFallback, NotImplementedFallsBackToOneRequestPerKeyEachDeleted) +TEST(CASGCCohortDelete, RequestsAreCutAtTheRequestSize) { auto backend = std::make_shared(); auto store = openPlainPool(backend); CasOperation op = store->openRequests().admit(); - const std::vector keys = manifestKeys(3); - for (const WriteOnceKey & key : keys) - ASSERT_TRUE(std::holds_alternative(op.create(key.str(), "b", Retry::once()))); - - backend->failNextBulkRemoveWith(std::make_exception_ptr(DB::Exception(DB::ErrorCodes::NOT_IMPLEMENTED, "no batch delete"))); + const std::vector keys = manifestKeys(5); + seedBodies(op, keys); - const uint64_t requests_issued = removeChunkWriteOnceOrOneByOne(op, keys, Retry::once()); + std::vector> requests; + removeCohortWriteOnce(op, keys, 2, Retry::once(), [&](size_t begin, size_t end) { requests.emplace_back(begin, end); }); - EXPECT_EQ(requests_issued, 4u) << "the failed bulk attempt is itself a call, counted alongside the 3 that followed it"; - EXPECT_EQ(backend->bulkRemoveCalls(), 4u) << "1 failed bulk attempt + 3 single-key fallback requests"; + EXPECT_EQ(requests, (std::vector>{{0, 2}, {2, 4}, {4, 5}})); + EXPECT_EQ(backend->bulkRemoveCalls(), 3u); for (const WriteOnceKey & key : keys) EXPECT_FALSE(op.head(key.str(), Retry::once()).has_value()) << key.str(); } -/// A teardown begun WHILE the fallback is mid-loop stops the remainder at admission, exactly as any -/// other CAS request would be: `removeChunkWriteOnceOrOneByOne`'s per-key loop is not a special path -/// around the engine's own fence, it is ordinary calls through it. -TEST(CASGCBulkDeleteFallback, TeardownBegunBetweenTwoFallbackKeysStopsTheRemainderAtAdmission) +TEST(CASGCCohortDelete, CohortHelperPropagatesNotImplemented) { - auto backend = std::make_shared(); - auto store = openPlainPool(backend); - CasOperation op = store->openRequests().admit(); - const std::vector keys = manifestKeys(4); - for (const WriteOnceKey & key : keys) - ASSERT_TRUE(std::holds_alternative(op.create(key.str(), "b", Retry::once()))); - - backend->failNextBulkRemoveWith(std::make_exception_ptr(DB::Exception(DB::ErrorCodes::NOT_IMPLEMENTED, "no batch delete"))); - - /// The hook does not run on the armed (failing) bulk attempt (it is rethrown before the hook would - /// fire), so this counts only the fallback's own per-key calls that actually reached the backend. - /// Teardown is armed once the SECOND such call has been served, so it is the THIRD key's own - /// admission -- checked at the start of its own `removeManyWriteOnce`, before this hook could run - /// again -- that is refused; the fourth key is never attempted at all. - size_t backend_calls_served = 0; - backend->onBeforeBulkRemove([&] { - if (++backend_calls_served == 2) - store->beginTeardown(); - }); - - expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { (void)removeChunkWriteOnceOrOneByOne(op, keys, Retry::standard()); }); - - /// `store->beginTeardown()` is irreversible here (this test never re-opens the pool), so `op` itself - /// -- the open plane -- refuses every further request, verification reads included. Read through the - /// mount plane instead: a different fence over the SAME backend, unaffected by open-plane teardown - /// (see `CASGCTeardownStop.OpenPlaneRefusesAfterTeardownBeganAndTheMountPlaneDoesNot`). - CasOperation verify = store->mountRequests().admit(); - EXPECT_FALSE(verify.head(keys[0].str(), Retry::once()).has_value()) << "deleted before teardown began"; - EXPECT_FALSE(verify.head(keys[1].str(), Retry::once()).has_value()) << "deleted before teardown began"; - EXPECT_TRUE(verify.head(keys[2].str(), Retry::once()).has_value()) << "refused at admission, never reached the backend"; - EXPECT_TRUE(verify.head(keys[3].str(), Retry::once()).has_value()) << "never attempted"; - EXPECT_EQ(backend->bulkRemoveCalls(), 3u) << "1 failed bulk attempt + 2 single-key fallback requests that landed"; + SCOPED_TRACE("request size 1000, the first request is rejected"); + auto backend = std::make_shared(); + auto store = openPlainPool(backend); + CasOperation op = store->openRequests().admit(); + const std::vector keys = manifestKeys(1000); + seedBodies(op, keys); + backend->failNextBulkRemoveWith(std::make_exception_ptr(DB::Exception(DB::ErrorCodes::NOT_IMPLEMENTED, "no batch delete"))); + size_t callbacks = 0; + + expectThrowsCode(DB::ErrorCodes::NOT_IMPLEMENTED, + [&] { removeCohortWriteOnce(op, keys, 1000, Retry::once(), [&](size_t, size_t) { ++callbacks; }); }); + + EXPECT_EQ(backend->bulkRemoveCalls(), 1u) << "a rejected request is never re-sent key by key"; + EXPECT_EQ(callbacks, 0u); + for (const WriteOnceKey & key : keys) + EXPECT_TRUE(op.head(key.str(), Retry::once()).has_value()) << key.str(); + } + { + SCOPED_TRACE("request size 250, the second request is rejected"); + auto backend = std::make_shared(); + auto store = openPlainPool(backend); + CasOperation op = store->openRequests().admit(); + const std::vector keys = manifestKeys(1000); + seedBodies(op, keys); + /// The hook runs only on a call that applies, so it arms the rejection for the call after the first. + backend->onBeforeBulkRemove([&] + { + backend->failNextBulkRemoveWith(std::make_exception_ptr(DB::Exception(DB::ErrorCodes::NOT_IMPLEMENTED, "no batch delete"))); + }); + size_t callbacks = 0; + + expectThrowsCode(DB::ErrorCodes::NOT_IMPLEMENTED, + [&] { removeCohortWriteOnce(op, keys, 250, Retry::once(), [&](size_t, size_t) { ++callbacks; }); }); + + EXPECT_EQ(backend->bulkRemoveCalls(), 2u); + EXPECT_EQ(callbacks, 1u); + for (size_t i = 0; i < keys.size(); ++i) + EXPECT_EQ(op.head(keys[i].str(), Retry::once()).has_value(), i >= 250) << keys[i].str(); + } } -/// A REAL error on one of the fallback's per-key deletes (not "batch not supported", so not caught and -/// retried again) stops the loop exactly where it happened: the keys before it are deleted, the ones -/// from it on are never attempted, and the error itself propagates out of the helper. -TEST(CASGCBulkDeleteFallback, ARealErrorOnAFallbackKeyStopsTheRemainderAndPropagates) +TEST(CASGCCohortDelete, ARealErrorStopsTheRemainingRequestsAndPropagates) { auto backend = std::make_shared(); auto store = openPlainPool(backend); CasOperation op = store->openRequests().admit(); - const std::vector keys = manifestKeys(4); - for (const WriteOnceKey & key : keys) - ASSERT_TRUE(std::holds_alternative(op.create(key.str(), "b", Retry::once()))); - - backend->failNextBulkRemoveWith(std::make_exception_ptr(DB::Exception(DB::ErrorCodes::NOT_IMPLEMENTED, "no batch delete"))); - - /// The hook does not run on an armed (failing) call, so this fires only on the fallback's own - /// per-key calls that actually reached the backend -- the FIRST of which (key[0]'s own delete) arms - /// a real, non-capability failure for the call right after it, i.e. key[1]'s. + const std::vector keys = manifestKeys(6); + seedBodies(op, keys); backend->onBeforeBulkRemove([&] { backend->failNextBulkRemoveWith(std::make_exception_ptr(DB::Exception(DB::ErrorCodes::CORRUPTED_DATA, "not a capability problem"))); }); - expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { (void)removeChunkWriteOnceOrOneByOne(op, keys, Retry::once()); }); + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { removeCohortWriteOnce(op, keys, 2, Retry::once(), [](size_t, size_t) {}); }); - EXPECT_FALSE(op.head(keys[0].str(), Retry::once()).has_value()) << "deleted before the real error"; - EXPECT_TRUE(op.head(keys[1].str(), Retry::once()).has_value()) << "this delete is the one that failed"; - EXPECT_TRUE(op.head(keys[2].str(), Retry::once()).has_value()) << "never attempted"; - EXPECT_TRUE(op.head(keys[3].str(), Retry::once()).has_value()) << "never attempted"; - EXPECT_EQ(backend->bulkRemoveCalls(), 3u) << "1 failed bulk attempt + key[0]'s delete + key[1]'s failed attempt"; -} - -/// A failure outside the "batch delete not supported" class must propagate as-is, with no fallback: -/// the helper does not treat every `removeManyWriteOnce` failure as "try one key at a time". -TEST(CASGCBulkDeleteFallback, OtherFailureClassPropagatesWithNoFallback) -{ - auto backend = std::make_shared(); - auto store = openPlainPool(backend); - CasOperation op = store->openRequests().admit(); - const std::vector keys = manifestKeys(3); - for (const WriteOnceKey & key : keys) - ASSERT_TRUE(std::holds_alternative(op.create(key.str(), "b", Retry::once()))); - - backend->failNextBulkRemoveWith(std::make_exception_ptr(DB::Exception(DB::ErrorCodes::CORRUPTED_DATA, "not a capability problem"))); - - expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { (void)removeChunkWriteOnceOrOneByOne(op, keys, Retry::once()); }); - - EXPECT_EQ(backend->bulkRemoveCalls(), 1u) << "no per-key fallback for a non-capability failure"; - for (const WriteOnceKey & key : keys) - EXPECT_TRUE(op.head(key.str(), Retry::once()).has_value()) << "nothing was deleted"; + EXPECT_EQ(backend->bulkRemoveCalls(), 2u) << "request 1 applied, request 2 failed, request 3 never sent"; + for (size_t i = 0; i < keys.size(); ++i) + EXPECT_EQ(op.head(keys[i].str(), Retry::once()).has_value(), i >= 2) << keys[i].str(); } diff --git a/src/Disks/tests/gtest_cas_gc_frontier_gate.cpp b/src/Disks/tests/gtest_cas_gc_frontier_gate.cpp index d8c90fa7dc1c..a35f8c50e6a9 100644 --- a/src/Disks/tests/gtest_cas_gc_frontier_gate.cpp +++ b/src/Disks/tests/gtest_cas_gc_frontier_gate.cpp @@ -241,7 +241,7 @@ class DrainRaceBackend final : public CountingBackend std::function after_read_hook; }; -class PostFoldUnreadableTerminalBackend final : public CountingBackend +class DeadCheckpointHeadFailureBackend final : public CountingBackend { public: /// Unhide the names the primitive overrides below would otherwise shadow. @@ -260,7 +260,7 @@ class PostFoldUnreadableTerminalBackend final : public CountingBackend std::optional head(const String & key, TransportAccess & access) override { if (!bypass_fault && key == unreadable_key) - throw std::runtime_error("injected post-fold terminal read failure for " + key); + throw std::runtime_error("injected dead checkpoint HEAD failure for " + key); return CountingBackend::head(key, access); } @@ -2821,10 +2821,11 @@ TEST(CASGCFrontierGate, CleanupEvidenceLeavesRemovedNamespaceCheckpointForJanito /// Once a terminal has folded, a later physical read failure is janitor debt, not lifecycle evidence /// loss. Removing this per-key leak handling would either make the signal disappear or let one dead -/// object prevent the janitor from considering the rest of its page. -TEST(CASGCFrontierGate, PostFoldUnreadableTerminalIsCountedWithoutSuppressingProgress) +/// object prevent the janitor from considering the rest of its page. The unreadable object is the dead +/// checkpoint: dead `_log`/`_snap` keys are deleted token-free and never read. +TEST(CASGCFrontierGate, PostFoldUnreadableDeadCheckpointIsCountedWithoutSuppressingProgress) { - auto backend = std::make_shared(); + auto backend = std::make_shared(); auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/0); CasRequests requests = openRequestsForTest(backend); CasOperation op = requests.admit(); @@ -2876,10 +2877,11 @@ TEST(CASGCFrontierGate, PostFoldUnreadableTerminalIsCountedWithoutSuppressingPro dropRefTransition(*backend, layout, progressing, "victim", manifest); const String terminal_key = layout.refLogKey(removed_life, RefTxnId{1, 2}); + const String checkpoint_key = layout.refCkptKey(removed_life); const String later_dead_residue = layout.refLogKey(removed_life, RefTxnId{1, 3}); ASSERT_TRUE(std::holds_alternative( op.create(later_dead_residue, "dead residue after the folded terminal", Retry::once()))); - backend->makeUnreadable(terminal_key); + backend->makeUnreadable(checkpoint_key); std::map namespace_cleanup; const uint64_t leaks_before @@ -2900,7 +2902,8 @@ TEST(CASGCFrontierGate, PostFoldUnreadableTerminalIsCountedWithoutSuppressingPro EXPECT_EQ(report.manifests_deleted, 1u) << "the janitor leak cannot promote itself into pool-wide destructive suppression"; EXPECT_FALSE(op.head(layout.manifestKey(manifest_id), Retry::once()).has_value()); - EXPECT_TRUE(backend->existsIgnoringFault(terminal_key)); + EXPECT_TRUE(backend->existsIgnoringFault(checkpoint_key)); + EXPECT_FALSE(backend->existsIgnoringFault(terminal_key)); EXPECT_FALSE(backend->existsIgnoringFault(later_dead_residue)) << "one unreadable key cannot stop the perpetual janitor from deciding the rest of its page"; ASSERT_FALSE(namespace_cleanup.empty()); @@ -2909,7 +2912,7 @@ TEST(CASGCFrontierGate, PostFoldUnreadableTerminalIsCountedWithoutSuppressingPro ProfileEvents::global_counters[ProfileEvents::CASGCNamespaceCleanupLeaks].load() - leaks_before, 1u); const String captured = log_capture.captured(); - EXPECT_NE(captured.find(terminal_key), String::npos); + EXPECT_NE(captured.find(checkpoint_key), String::npos); EXPECT_NE(captured.find("leak"), String::npos); } diff --git a/src/Disks/tests/gtest_cas_gc_manifest_bulk_delete.cpp b/src/Disks/tests/gtest_cas_gc_manifest_bulk_delete.cpp index 3a7c2ca3ba7f..fdb55f3b2d61 100644 --- a/src/Disks/tests/gtest_cas_gc_manifest_bulk_delete.cpp +++ b/src/Disks/tests/gtest_cas_gc_manifest_bulk_delete.cpp @@ -95,31 +95,96 @@ TEST(CASGCManifestBulkDelete, FiveBodiesInChunksOfTwoAreThreeRequests) EXPECT_FALSE((*op).head(store->layout().manifestKey(id), Retry::once()).has_value()); } -/// The object storage rejects the chunk's one bulk `removeManyWriteOnce` as NOT_IMPLEMENTED (a -/// GCS-backed pool): the phase's `flush()` falls back to one admitted request per key -/// (`removeChunkWriteOnceOrOneByOne`, CasGc.h), and every manifest in the chunk is still recorded -/// deleted -- the per-key event emission this phase does is unaffected by how the deletes were sent. -TEST(CASGCManifestBulkDelete, NotImplementedFallsBackToOneRequestPerKeyAndStillRecordsAllOfThem) +TEST(CASGCManifestBulkDelete, ManifestCleanupSendsCohortAsStorageRequests) { - auto backend = std::make_shared(); - auto store = Pool::open(backend, PoolConfig{.pool_prefix = "p", .server_root_id = "test", - .gc_fold_max_defer_rounds = 0}); - const auto ids = seedDroppedManifests(*backend, store->layout(), 5); - - /// One armed failure: the chunk's own bulk attempt (all 5 land in one chunk under the default - /// chunk size) fails as "batch delete not supported"; the 5 single-key fallback calls that follow - /// are not armed and succeed. - backend->failNextBulkRemoveWith(std::make_exception_ptr( - DB::Exception(DB::ErrorCodes::NOT_IMPLEMENTED, "no batch delete"))); + PhaseReads reads; + auto backend = std::make_shared(); + backend->setBatchDeleteSupported(false); + PoolConfig config{.pool_prefix = "p", .server_root_id = "test", .gc_fold_max_defer_rounds = 0}; + config.gc_bulk_delete_chunk_keys = 4; + auto store = Pool::open(backend, config); + const auto ids = seedDroppedManifests(*backend, store->layout(), 10); Gc gc(store, kGc); + gc.setPhaseSink(phaseReadsSink(reads, *backend, store->layout(), "handoff_reclaim", "manifest_deletes")); const uint64_t deleted = reclaim(gc, store, *backend, ids, 16); + gc.setPhaseSink({}); + + EXPECT_EQ(deleted, 10u); + const std::vector calls = backend->callSizes(); + EXPECT_EQ(calls.size(), 10u); + EXPECT_TRUE(std::all_of(calls.begin(), calls.end(), [](size_t n) { return n == 1; })); + EXPECT_EQ(reads.row["requests"], 10u); + EXPECT_EQ(reads.row["accepted"], 10u); + EXPECT_EQ(reads.row["unsent"], 0u); + EXPECT_EQ(reads.catalog_in_phase, 0u) << "manifest cleanup reads no authority, as before"; + EXPECT_EQ(reads.state_in_phase, 0u); +} - EXPECT_EQ(deleted, 5u) << "the per-key fallback must still record every manifest as deleted"; - EXPECT_EQ(backend->bulkRemoveCalls(), 6u) << "1 failed bulk attempt + 5 single-key fallback requests"; +TEST(CASGCManifestBulkDelete, ManifestCleanupContainsCapabilityRejection) +{ + std::vector> rows; + auto backend = std::make_shared(); + PoolConfig config{.pool_prefix = "p", .server_root_id = "test", .gc_fold_max_defer_rounds = 0}; + config.gc_bulk_delete_chunk_keys = 2; + auto store = Pool::open(backend, config); + const auto ids = seedDroppedManifests(*backend, store->layout(), 5); + /// The first cohort goes through; the store then rejects `DeleteObjects` on the next one. + backend->onBeforeBulkRemove([&] { backend->setStoreRejectsBatches(true); }); + + Gc gc(store, kGc); + gc.setPhaseSink([&](const GcPhaseRecord & record) + { + if (record.phase == "manifest_deletes" && record.metrics.at("attempted") > 0) + rows.push_back(record.metrics); + }); + for (size_t round = 0; round < 16 && rows.empty(); ++round) + { + ASSERT_NO_THROW((void)runRegularRoundReclaiming(gc)) << "a rejected batch must not escape the round"; + store->renewWatermarkOnce(); + } + ASSERT_EQ(rows.size(), 1u); + EXPECT_EQ(rows[0]["capability_learned"], 1u); + EXPECT_EQ(rows[0]["accepted"], 2u) << "the cohort before the rejection stays recorded"; + EXPECT_EQ(rows[0]["unsent"], 3u); + EXPECT_EQ(rows[0]["requests"], 2u) << "the rejected request counts as sent"; + EXPECT_EQ(backend->requestSizes(), (std::vector{2, 2})) << "no one-key request for the rejected cohort this round"; + + /// The intake cursor that found the bodies is committed, so only the orphan-manifest sweep reclaims + /// them. It does so once the namespace cursor is past their writer epoch, and once the mount floor of + /// the namespace's server root ("00") has retired their builds. + uint64_t swept = 0; + uint64_t later_attempted = 0; + gc.setPhaseSink([&](const GcPhaseRecord & record) + { + if (record.phase == "orphan_sweep") + swept += record.metrics.at("deleted"); + else if (record.phase == "manifest_deletes") + later_attempted += record.metrics.at("attempted"); + }); + const Layout & layout = store->layout(); + const uint64_t seal_sequence = appendRefLogSeed(*backend, layout, kNs, {epochSealOp()}); + publishAt(*backend, layout, kNs, RefTxnId{2, 1}, "next", /*build_sequence*/ 50, UInt128(0x7002), /*birth*/ false, + /*prev_epoch_seal*/ RefTxnId{1, seal_sequence}); + replaceRecoverableCkptForRawFixture(*backend, layout, kNs, RefCkpt{ + .life_epoch = 1, + .committed_through = RefTxnId{2, 1}, + .checkpoint_snapshot_id = std::nullopt, + .last_epoch_seal = RefTxnId{1, seal_sequence}, + }); + setWatermarkMinActive(*backend, layout, "00", /*writer_epoch*/ 1, /*min_active_build_sequence*/ 100); OperationForTest op(*backend); - for (const ManifestId & id : ids) - EXPECT_FALSE((*op).head(store->layout().manifestKey(id), Retry::once()).has_value()); + bool all_gone = false; + for (size_t round = 0; round < 64 && !all_gone; ++round) + { + (void)runRegularRoundReclaiming(gc); + store->renewWatermarkOnce(); + all_gone = std::none_of(ids.begin(), ids.end(), + [&](const ManifestId & id) { return (*op).head(store->layout().manifestKey(id), Retry::once()).has_value(); }); + } + EXPECT_TRUE(all_gone) << "the orphan-manifest sweep deletes the bodies the rejected round left"; + EXPECT_EQ(swept, 3u) << "the three unsent bodies are deleted by the orphan sweep"; + EXPECT_EQ(later_attempted, 0u) << "no later fold re-nominates them into manifest_deletes"; } TEST(CASGCManifestBulkDelete, AThrowInTheSecondChunkKeepsTheFirstChunksAuditAndAbortsTheRound) diff --git a/src/Disks/tests/gtest_cas_gc_round_defer.cpp b/src/Disks/tests/gtest_cas_gc_round_defer.cpp index 43a6a909e86a..cf27dbb8393e 100644 --- a/src/Disks/tests/gtest_cas_gc_round_defer.cpp +++ b/src/Disks/tests/gtest_cas_gc_round_defer.cpp @@ -11,7 +11,7 @@ #include #include #include -#include "cas_test_helpers.h" +#include "cas_namespace_janitor_test_helpers.h" using namespace DB::Cas; using namespace DB::Cas::tests; @@ -403,7 +403,7 @@ TEST(CASGCRoundDefer, DeferredRoundRetriesPartialJanitorPageAtForcedFoldWithoutP /// Establish real opaque backend progress rather than fabricating a cursor value. One key remains /// after this page and the durable cursor must be non-empty. const NamespaceJanitorResult first_page - = NamespaceJanitor(requests, layout, 1).runOnePage(false, [] { return true; }); + = DB::Cas::tests::janitor::runOnePage(NamespaceJanitor(requests, layout, 1), false, [] { return true; }); ASSERT_EQ(first_page.pages, 1u); ASSERT_EQ(first_page.deleted, 1u); const GcMaintenanceReadResult partial = readGcMaintenanceState(op, layout); @@ -472,14 +472,14 @@ TEST(CASGCRoundDefer, DeferredRoundRetriesPartialJanitorPageAtForcedFoldWithoutP ASSERT_TRUE(folded.acquired_lease); ASSERT_FALSE(folded.deferred) << "gc_fold_max_defer_rounds=1 forces the round immediately following one DEFER to fold"; - EXPECT_EQ(backend->listCount(layout.namespaceRootPrefix()), 1u) - << "the authoritative fold must run the janitor exactly once, not once per call site"; + EXPECT_EQ(backend->listCount(layout.namespaceRootPrefix()), 2u) + << "one janitor run, not one per call site: the rest of the stream, then the wrap to the start"; const auto folded_cleanup = std::find_if(phases.begin(), phases.end(), [](const GcPhaseRecord & phase) { return phase.phase == "namespace_cleanup"; }); ASSERT_NE(folded_cleanup, phases.end()); - EXPECT_EQ(folded_cleanup->metrics.at("janitor_pages"), 1u); + EXPECT_EQ(folded_cleanup->metrics.at("janitor_pages"), 2u); EXPECT_GE(folded_cleanup->metrics.at("janitor_keys"), 1u); EXPECT_EQ(folded_cleanup->metrics.at("janitor_deleted"), 1u); EXPECT_EQ(static_cast(op.head(key_a, Retry::once()).has_value()) + static_cast(op.head(key_b, Retry::once()).has_value()), 0u) diff --git a/src/Disks/tests/gtest_cas_namespace_janitor.cpp b/src/Disks/tests/gtest_cas_namespace_janitor.cpp index 0ada8882e7b0..a88d6cbb4d64 100644 --- a/src/Disks/tests/gtest_cas_namespace_janitor.cpp +++ b/src/Disks/tests/gtest_cas_namespace_janitor.cpp @@ -1,9 +1,8 @@ -#include "cas_test_helpers.h" -#include -#include +#include "cas_namespace_janitor_test_helpers.h" using namespace DB::Cas; using namespace DB::Cas::tests; +using namespace DB::Cas::tests::janitor; namespace DB::ErrorCodes { @@ -13,14 +12,6 @@ namespace DB::ErrorCodes namespace { -/// `readGcMaintenanceState` now takes an admitted `CasOperation`, which cannot bind to an rvalue: every -/// call site below goes through this helper rather than materializing its own throwaway operation. -GcMaintenanceReadResult readState(CasRequests & requests, const Layout & layout) -{ - auto op = requests.admit(); - return readGcMaintenanceState(op, layout); -} - class OrderedJanitorBackend : public CountingBackend { public: @@ -206,7 +197,7 @@ class FailMaintenancePublicationBackend : public CountingBackend bool fail_publication = false; }; -/// The catch-path reset in `NamespaceJanitor::runOnePage` fires on any LIST failure. Both faults here +/// The catch-path reset in `NamespaceJanitor::run` fires on any LIST failure. Both faults here /// throw a `DB::Exception` classified `NETWORK_ERROR`: a `std::runtime_error` is not a `Poco::Exception`, /// so `CasOperation`'s engine treats it as an unmodeled local bug and surfaces it immediately on every /// path (read or write) without ever reaching the ambiguity-resolving machinery this test needs -- a @@ -231,31 +222,6 @@ class ThrowingListAndAmbiguousWriteBackend : public CountingBackend uint64_t write_attempts = 0; }; -/// A one-shot `create`, asserting it committed (mirrors the retired `backend->putIfAbsent(key, bytes)`). -void createObj(Backend & backend, const String & key, const String & bytes) -{ - OperationForTest op(backend); - ASSERT_TRUE(std::holds_alternative((*op).create(key, bytes, Retry::once()))); -} - -/// An exact read (mirrors the retired `backend->get(key)`). -std::optional readObj(Backend & backend, const String & key) -{ - OperationForTest op(backend); - return (*op).read(key, Retry::standard()); -} - -void seedCatalog(Backend & backend, const Layout & layout, RefCatalog catalog = {}) -{ - createObj(backend, layout.refCatalogKey(), encodeRefCatalog(catalog)); -} - -NamespaceLifeId life(const char * name, uint64_t id) -{ - const RootNamespace ns{name}; - return NamespaceLifeId::fromCatalogEntry(ns, UInt128{id}); -} - } TEST(CASNamespaceJanitor, DeletesDeadFilesAndCheckpointFromOnePostListCatalogCut) @@ -272,7 +238,7 @@ TEST(CASNamespaceJanitor, DeletesDeadFilesAndCheckpointFromOnePostListCatalogCut backend->resetCounts(); NamespaceJanitor janitor(requests, layout, 100); - const NamespaceJanitorResult result = janitor.runOnePage(false, [] { return true; }); + const NamespaceJanitorResult result = runOnePage(janitor, false, [] { return true; }); EXPECT_EQ(result.pages, 1u); EXPECT_EQ(result.keys, 2u); @@ -302,7 +268,7 @@ TEST(CASNamespaceJanitor, RetainsEveryCurrentLifecycleAndSuppressesAmbiguousCut) NamespaceLifeId::fromCatalogEntry(entry.ns, entry.incarnation)), "keep"); NamespaceJanitor janitor(requests, layout, 100); - const auto result = janitor.runOnePage(false, [] { return true; }); + const auto result = runOnePage(janitor, false, [] { return true; }); EXPECT_EQ(result.deleted, 0u); EXPECT_EQ(backend->deleteTotal(), 0u); } @@ -329,7 +295,7 @@ TEST(CASNamespaceJanitor, CatalogFirstCreatingRetainsEveryObjectOfTheNewLife) backend->resetCounts(); const NamespaceJanitorResult result - = NamespaceJanitor(requests, layout, 100).runOnePage(false, [] { return true; }); + = runOnePage(NamespaceJanitor(requests, layout, 100), false, [] { return true; }); EXPECT_EQ(result.deleted, 0u); EXPECT_EQ(backend->deleteTotal(), 0u); @@ -361,7 +327,7 @@ TEST(CASNamespaceJanitor, CancelledCreatingCheckpointIsReclaimedThroughPublicLif EXPECT_TRUE(CasRefCatalog::read(read_op, layout).catalog.entries.empty()); const NamespaceJanitorResult result - = NamespaceJanitor(requests, layout, 100).runOnePage(false, [] { return true; }); + = runOnePage(NamespaceJanitor(requests, layout, 100), false, [] { return true; }); EXPECT_EQ(result.deleted, 1u); EXPECT_FALSE(readObj(*backend, ckpt).has_value()); } @@ -382,7 +348,7 @@ TEST(CASNamespaceJanitor, SuppressionAndFenceLossDeleteNothing) backend->resetCounts(); NamespaceJanitor janitor(requests, layout, 1); - EXPECT_EQ(janitor.runOnePage(true, [] { return true; }).deleted, 0u); + EXPECT_EQ(runOnePage(janitor, true, [] { return true; }).deleted, 0u); EXPECT_EQ(readState(requests, layout).status, GcMaintenanceReadStatus::Absent) << "a globally suppressed page is undecided and must not mint cleanup progress"; EXPECT_EQ(backend->writeTotal(), 0u); @@ -391,7 +357,7 @@ TEST(CASNamespaceJanitor, SuppressionAndFenceLossDeleteNothing) /// itself -- a sample false from the start therefore ends the call by exception rather than by a /// quiet no-op result. DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, - [&] { (void)janitor.runOnePage(false, [] { return false; }); }); + [&] { (void)runOnePage(janitor, false, [] { return false; }); }); EXPECT_EQ(readState(requests, layout).status, GcMaintenanceReadStatus::Absent) << "fence loss must not mint progress past a page whose deletion was not authorized"; EXPECT_TRUE(readObj(*backend, first).has_value()); @@ -417,7 +383,7 @@ TEST(CASNamespaceJanitor, FenceLossOnRetainedOnlyPageDoesNotAdvanceCursor) /// A liveness sample false from the start is refused at the maintenance read, before the page ever /// gets to examine an object -- retained-only or not; the page ends by exception. DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, - [&] { (void)NamespaceJanitor(requests, layout, 1).runOnePage(false, [] { return false; }); }); + [&] { (void)runOnePage(NamespaceJanitor(requests, layout, 1), false, [] { return false; }); }); EXPECT_EQ(backend->deleteTotal(), 0u); EXPECT_TRUE(readObj(*backend, ckpt).has_value()); @@ -435,7 +401,7 @@ TEST(CASNamespaceJanitor, FenceLossAfterLastDeleteRetainsCursorWithoutRollingBac const String dead = layout.refCkptKey(life("dead-after-delete", 64)); createObj(*backend, dead, "dead"); - const NamespaceJanitorResult result = NamespaceJanitor(requests, layout, 1).runOnePage( + const NamespaceJanitorResult result = runOnePage(NamespaceJanitor(requests, layout, 1), false, [&] { return !backend->delete_done; }); EXPECT_EQ(result.deleted, 1u); @@ -456,13 +422,13 @@ TEST(CASNamespaceJanitor, CursorResumesThenResetsAtEnd) createObj(*backend, layout.namespaceFilesPrefix(dead) + "b", "b"); NamespaceJanitor first_process(requests, layout, 1); - EXPECT_EQ(first_process.runOnePage(false, [] { return true; }).deleted, 1u); + EXPECT_EQ(runOnePage(first_process, false, [] { return true; }).deleted, 1u); const auto mid = readState(requests, layout); ASSERT_EQ(mid.status, GcMaintenanceReadStatus::Valid); ASSERT_TRUE(mid.state); EXPECT_FALSE(mid.state->janitor_cursor.empty()); NamespaceJanitor restarted_process(requests, layout, 1); - EXPECT_EQ(restarted_process.runOnePage(false, [] { return true; }).deleted, 1u); + EXPECT_EQ(runOnePage(restarted_process, false, [] { return true; }).deleted, 1u); EXPECT_TRUE(readState(requests, layout).state->janitor_cursor.empty()); } @@ -482,7 +448,7 @@ TEST(CASNamespaceJanitor, TakesOneCatalogCutAfterListingAndContinuesPastMalforme backend->resetCounts(); backend->events.clear(); - const auto result = NamespaceJanitor(requests, layout, 100).runOnePage(false, [] { return true; }); + const auto result = runOnePage(NamespaceJanitor(requests, layout, 100), false, [] { return true; }); EXPECT_EQ(result.deleted, 1u); EXPECT_FALSE(result.anomalies.empty()); EXPECT_TRUE(readObj(*backend, malformed).has_value()); @@ -505,7 +471,7 @@ TEST(CASNamespaceJanitor, MalformedKeyIsFinalAndAdvancesCursor) createObj(*backend, second, "second"); const NamespaceJanitorResult result - = NamespaceJanitor(requests, layout, 1).runOnePage(false, [] { return true; }); + = runOnePage(NamespaceJanitor(requests, layout, 1), false, [] { return true; }); EXPECT_EQ(result.deleted, 0u); EXPECT_FALSE(result.anomalies.empty()); @@ -532,7 +498,7 @@ TEST(CASNamespaceJanitor, DuplicateCurrentLifeSuppressesWholePage) const String dead_b = layout.refCkptKey(life("dead-b", 93)); createObj(*backend, dead_a, "a"); createObj(*backend, dead_b, "b"); - const auto result = NamespaceJanitor(requests, layout, 1).runOnePage(false, [] { return true; }); + const auto result = runOnePage(NamespaceJanitor(requests, layout, 1), false, [] { return true; }); EXPECT_EQ(result.deleted, 0u); EXPECT_EQ(backend->deleteTotal(), 0u); EXPECT_TRUE(readObj(*backend, dead_a).has_value()); @@ -550,12 +516,12 @@ TEST(CASNamespaceJanitor, CorruptProgressResetsWithoutDeletingAndFilesOnlyOmitte const String dead = layout.namespaceFilesPrefix(life("dead", 101)) + "only-residue"; createObj(*backend, dead, "bytes"); createObj(*backend, layout.gcMaintenanceStateKey(), "corrupt"); - EXPECT_EQ(NamespaceJanitor(requests, layout, 100).runOnePage(false, [] { return true; }).deleted, 0u); + EXPECT_EQ(runOnePage(NamespaceJanitor(requests, layout, 100), false, [] { return true; }).deleted, 0u); EXPECT_TRUE(readObj(*backend, dead).has_value()); EXPECT_EQ(readState(requests, layout).status, GcMaintenanceReadStatus::Valid); - EXPECT_EQ(NamespaceJanitor(requests, layout, 100).runOnePage(false, [] { return true; }).deleted, 0u); + EXPECT_EQ(runOnePage(NamespaceJanitor(requests, layout, 100), false, [] { return true; }).deleted, 0u); EXPECT_TRUE(readObj(*backend, dead).has_value()); - EXPECT_EQ(NamespaceJanitor(requests, layout, 100).runOnePage(false, [] { return true; }).deleted, 1u); + EXPECT_EQ(runOnePage(NamespaceJanitor(requests, layout, 100), false, [] { return true; }).deleted, 1u); EXPECT_FALSE(readObj(*backend, dead).has_value()); } @@ -569,7 +535,7 @@ TEST(CASNamespaceJanitor, ExactTokenMismatchRetainsConcurrentReplacement) const String later = layout.refCkptKey(life("dead-b", 112)); createObj(*backend, dead, "old"); createObj(*backend, later, "later"); - const auto result = NamespaceJanitor(requests, layout, 1).runOnePage(false, [] { return true; }); + const auto result = runOnePage(NamespaceJanitor(requests, layout, 1), false, [] { return true; }); EXPECT_EQ(result.deleted, 0u); ASSERT_TRUE(readObj(*backend, dead).has_value()); EXPECT_EQ(readObj(*backend, dead)->bytes, "winner"); @@ -598,7 +564,7 @@ TEST(CASNamespaceJanitor, TokenlessListHeadsDeadKeysAndRetainsConcurrentReplacem backend->replace_on_head = raced_key; backend->resetCounts(); - const auto result = NamespaceJanitor(requests, layout, 100).runOnePage(false, [] { return true; }); + const auto result = runOnePage(NamespaceJanitor(requests, layout, 100), false, [] { return true; }); EXPECT_EQ(result.deleted, 1u); EXPECT_TRUE(result.anomalies.empty()); @@ -623,7 +589,7 @@ TEST(CASNamespaceJanitor, TokenlessListRechecksFenceAfterHeadBeforeDelete) createObj(*backend, dead_key, "dead"); backend->resetCounts(); - const auto result = NamespaceJanitor(requests, layout, 100).runOnePage( + const auto result = runOnePage(NamespaceJanitor(requests, layout, 100), false, [&] { return backend->fence_held; }); EXPECT_EQ(result.deleted, 0u); @@ -644,7 +610,7 @@ TEST(CASNamespaceJanitor, PostListCatalogCutProtectsConcurrentCreationWithOneGet createObj(*backend, first, "ckpt"); createObj(*backend, second, "file"); backend->resetCounts(); - const auto result = NamespaceJanitor(requests, layout, 100).runOnePage(false, [] { return true; }); + const auto result = runOnePage(NamespaceJanitor(requests, layout, 100), false, [] { return true; }); EXPECT_EQ(result.deleted, 0u); EXPECT_EQ(backend->deleteTotal(), 0u); EXPECT_EQ(backend->getCount(layout.refCatalogKey()), 1u); @@ -662,7 +628,7 @@ TEST(CASNamespaceJanitor, BackendRejectedCursorResetsExactlyAndDeletesNothing) createObj(*backend, dead, "bytes"); createObj(*backend, layout.gcMaintenanceStateKey(), encodeGcMaintenanceState({.janitor_cursor = "rejected"})); - EXPECT_THROW(NamespaceJanitor(requests, layout, 100).runOnePage(false, [] { return true; }), std::runtime_error); + EXPECT_THROW(runOnePage(NamespaceJanitor(requests, layout, 100), false, [] { return true; }), std::runtime_error); EXPECT_EQ(backend->deleteTotal(), 0u); EXPECT_TRUE(readObj(*backend, dead).has_value()); EXPECT_TRUE(readState(requests, layout).state->janitor_cursor.empty()); @@ -677,7 +643,7 @@ TEST(CASNamespaceJanitor, CursorPublicationFailureIsLeakOnly) const String dead = layout.refCkptKey(life("dead", 141)); createObj(*backend, dead, "bytes"); backend->fail_publication = true; - const auto result = NamespaceJanitor(requests, layout, 100).runOnePage(false, [] { return true; }); + const auto result = runOnePage(NamespaceJanitor(requests, layout, 100), false, [] { return true; }); EXPECT_EQ(result.deleted, 1u); EXPECT_FALSE(result.anomalies.empty()); EXPECT_FALSE(readObj(*backend, dead).has_value()); @@ -699,7 +665,7 @@ TEST(CASGcMaintenanceState, CatchPathWriteIsOnce) const Layout layout("p"); DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, - [&] { (void)NamespaceJanitor(requests, layout, 100).runOnePage(false, [] { return true; }); }); + [&] { (void)runOnePage(NamespaceJanitor(requests, layout, 100), false, [] { return true; }); }); EXPECT_EQ(backend->write_attempts, 1u) << "the catch-path reset settles by its one resolve read and gives up rather than reissuing"; } diff --git a/src/Disks/tests/gtest_cas_namespace_janitor_batches.cpp b/src/Disks/tests/gtest_cas_namespace_janitor_batches.cpp new file mode 100644 index 000000000000..ee0b097388c1 --- /dev/null +++ b/src/Disks/tests/gtest_cas_namespace_janitor_batches.cpp @@ -0,0 +1,707 @@ +#include "cas_namespace_janitor_test_helpers.h" +#include +#include +#include +#include + +/// The janitor phase with no pool, so jobs run inline in submit order: exact request orders and counts. + +using namespace DB::Cas; +using namespace DB::Cas::tests; +using namespace DB::Cas::tests::janitor; + +namespace DB::ErrorCodes +{ + extern const int CORRUPTED_DATA; + extern const int NETWORK_ERROR; +} + +namespace +{ + +const Layout & kLayout = janitorLayout(); + +bool allAbsent(Backend & backend, const std::vector & keys, size_t begin, size_t end) +{ + for (size_t i = begin; i < end; ++i) + if (present(backend, keys[i])) + return false; + return true; +} + +bool allPresent(Backend & backend, const std::vector & keys, size_t begin, size_t end) +{ + for (size_t i = begin; i < end; ++i) + if (!present(backend, keys[i])) + return false; + return true; +} + +} + +TEST(CASNamespaceJanitor, DeadLifeStreamKeysDrainInBatches) +{ + JanitorFixture f; + const std::vector keys = seedLogs(*f.backend, kLayout, life("dead", 201), 2500); + + const NamespaceJanitorResult result = f.run(f.context()); + + const auto calls = f.backend->bulkCalls(); + ASSERT_EQ(calls.size(), 3u); + EXPECT_EQ(calls[0].keys.size(), 1000u); + EXPECT_EQ(calls[1].keys.size(), 1000u); + EXPECT_EQ(calls[2].keys.size(), 500u); + EXPECT_TRUE(f.backend->exactRemoves().empty()); + EXPECT_EQ(result.pages, 3u); + EXPECT_EQ(result.batches, 3u); + EXPECT_EQ(result.delete_jobs, 3u); + EXPECT_EQ(result.deleted, 2500u); + EXPECT_EQ(result.batch_keys, 1000u); + EXPECT_TRUE(allAbsent(*f.backend, keys, 0, keys.size())); + EXPECT_EQ(f.cursor(), String{}); +} + +TEST(CASNamespaceJanitor, StateKeysNeverEnterTheBatchPath) +{ + for (const bool tokenless : {false, true}) + { + SCOPED_TRACE(tokenless ? "tokenless listing" : "token listing"); + JanitorFixture f; + f.backend->tokenless = tokenless; + const NamespaceLifeId dead = life("dead", 203); + const std::vector stream = seedLogs(*f.backend, kLayout, dead, 1); + const String ckpt = kLayout.refCkptKey(dead); + const String file = kLayout.namespaceFilesPrefix(dead) + "data"; + createObj(*f.backend, ckpt, "ckpt"); + createObj(*f.backend, file, "file"); + /// Both state objects are rewritten between observation and delete: only an exact-token delete keeps the rewrite. + f.backend->before_exact_remove = [&](const String & key, JanitorBackend::Access & access) + { + f.backend->writeUncounted(key, "winner", access); + }; + + const NamespaceJanitorResult result = f.run(f.context()); + + const auto calls = f.backend->bulkCalls(); + ASSERT_EQ(calls.size(), 1u); + EXPECT_EQ(calls[0].keys, stream); + EXPECT_EQ(f.backend->exactRemoves(), (std::vector{ckpt, file})); + EXPECT_EQ(readObj(*f.backend, ckpt)->bytes, "winner"); + EXPECT_EQ(readObj(*f.backend, file)->bytes, "winner"); + EXPECT_FALSE(present(*f.backend, stream.front())); + EXPECT_EQ(result.deleted, 1u); + } +} + +TEST(CASNamespaceJanitor, RecreatedLifeBetweenListAndDeleteKeepsItsKeys) +{ + JanitorFixture f; + const NamespaceLifeId dropped = life("t", 211); + const NamespaceLifeId recreated = life("t", 212); + const std::vector old_keys = seedLogs(*f.backend, kLayout, dropped, 3); + const std::vector new_keys{kLayout.refLogKey(recreated, RefTxnId{1, 1}), kLayout.refSnapshotKey(recreated, RefTxnId{1, 1})}; + f.backend->after_namespace_list = [&](size_t, JanitorBackend::Access & access) + { + RefCatalog live; + live.entries.push_back(CatalogEntry{.ns = recreated.ns, .state = NsState::Live, .incarnation = recreated.incarnation}); + f.backend->writeUncounted(kLayout.refCatalogKey(), encodeRefCatalog(live), access); + for (const String & key : new_keys) + f.backend->writeUncounted(key, "new", access); + }; + + const NamespaceJanitorResult result = f.run(f.context()); + + ASSERT_EQ(f.backend->bulkCalls().size(), 1u); + EXPECT_EQ(f.backend->bulkCalls()[0].keys, old_keys); + EXPECT_EQ(result.deleted, old_keys.size()); + for (const String & key : new_keys) + EXPECT_EQ(readObj(*f.backend, key)->bytes, "new") << key; +} + +TEST(CASNamespaceJanitor, LeaseReplacedBeforeFinalBatch) +{ + JanitorFixture f; + const NamespaceLifeId dropped = life("t", 215); + const NamespaceLifeId recreated = life("t", 216); + const std::vector old_keys = seedLogs(*f.backend, kLayout, dropped, 10); + const std::vector new_keys{kLayout.refLogKey(recreated, RefTxnId{1, 1}), kLayout.refSnapshotKey(recreated, RefTxnId{1, 1})}; + std::atomic durable_lease_is_ours{true}; + std::atomic authority_held{true}; + f.backend->before_bulk = [&](const JanitorBackend::Keys &, JanitorBackend::Access & access) + { + if (!durable_lease_is_ours.exchange(false)) + return; + RefCatalog live; + live.entries.push_back(CatalogEntry{.ns = recreated.ns, .state = NsState::Live, .incarnation = recreated.incarnation}); + f.backend->writeUncounted(kLayout.refCatalogKey(), encodeRefCatalog(live), access); + for (const String & key : new_keys) + f.backend->writeUncounted(key, "new", access); + f.backend->writeUncounted(kLayout.gcMaintenanceStateKey(), encodeGcMaintenanceState({.janitor_cursor = "second-leader"}), access); + }; + JanitorRunContext context = f.context(); + context.liveness = [&] { return authority_held.load(); }; + context.refresh_authority = [&] { authority_held = durable_lease_is_ours.load(); }; + + const NamespaceJanitorResult result = f.run(context); + + ASSERT_EQ(f.backend->bulkCalls().size(), 1u); + EXPECT_EQ(f.backend->bulkCalls()[0].keys, old_keys); + EXPECT_TRUE(allAbsent(*f.backend, old_keys, 0, old_keys.size())); + for (const String & key : new_keys) + EXPECT_EQ(readObj(*f.backend, key)->bytes, "new") << key; + EXPECT_EQ(f.cursor(), String("second-leader")) << "the competing leader's cursor stands"; + EXPECT_FALSE(result.cursor_advanced); + for (const String & anomaly : result.anomalies) + EXPECT_EQ(anomaly.find("cursor publication"), String::npos) << "a Conflict is silent: " << anomaly; + + const size_t lists_before = f.backend->namespaceLists(); + expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { (void)f.run(context); }); + EXPECT_EQ(f.backend->namespaceLists(), lists_before) << "the next refresh stops the deposed leader before its first LIST"; +} + +TEST(CASNamespaceJanitor, PhaseBudgetCheckedBetweenPages) +{ + JanitorFixture f; + const std::vector keys = seedLogs(*f.backend, kLayout, life("dead", 221), 5000); + f.backend->after_namespace_list = [&](size_t, JanitorBackend::Access &) { f.clock.now += 8'000; }; + JanitorRunContext context = f.context(); + context.budget_ms = 20'000; + + const NamespaceJanitorResult result = f.run(context); + + EXPECT_EQ(f.backend->namespaceLists(), 3u) << "pages start at 0, 8 and 16 s; none at 24 s"; + EXPECT_EQ(result.pages, 3u); + EXPECT_TRUE(result.budget_exhausted); + EXPECT_TRUE(allAbsent(*f.backend, keys, 0, 3000)) << "the page in progress at the deadline completes"; + EXPECT_TRUE(allPresent(*f.backend, keys, 3000, 5000)); + EXPECT_EQ(f.cursor(), keys[2999]); +} + +TEST(CASNamespaceJanitor, LaterPageReadFailurePublishesPrefix) +{ + JanitorFixture f; + const std::vector keys = seedLogs(*f.backend, kLayout, life("dead", 223), 4000); + f.backend->after_namespace_list = [&](size_t index, JanitorBackend::Access &) + { + if (index == 2) + f.backend->failNextReadWith(kLayout.refCatalogKey(), + std::make_exception_ptr(DB::Exception(DB::ErrorCodes::CORRUPTED_DATA, "injected catalog read failure"))); + }; + + const NamespaceJanitorResult result = f.run(f.context()); + + EXPECT_EQ(result.deleted, 2000u); + EXPECT_FALSE(result.anomalies.empty()); + EXPECT_EQ(f.cursor(), keys[1999]) << "the complete prefix is published; no reset"; + EXPECT_TRUE(allPresent(*f.backend, keys, 2000, 4000)); +} + +TEST(CASNamespaceJanitor, ObservedAuthorityLossPublishesNothing) +{ + JanitorFixture f; + const std::vector keys = seedLogs(*f.backend, kLayout, life("dead", 224), 4000); + size_t refreshes = 0; + bool held = true; + JanitorRunContext context = f.context(); + context.liveness = [&] { return held; }; + context.refresh_authority = [&] + { + if (++refreshes == 3) + held = false; + }; + + (void)f.run(context); + + EXPECT_EQ(f.backend->namespaceLists(), 2u); + EXPECT_TRUE(allAbsent(*f.backend, keys, 0, 2000)) << "earlier pages' deletes stand"; + EXPECT_TRUE(allPresent(*f.backend, keys, 2000, 4000)); + EXPECT_FALSE(f.cursor().has_value()); +} + +TEST(CASNamespaceJanitor, SteadyStateTakesOnePage) +{ + const CatalogEntry live{.ns = RootNamespace{"live"}, .state = NsState::Live, .incarnation = UInt128{225}}; + JanitorFixture f(RefCatalog{.entries = {live}}); + (void)seedLogs(*f.backend, kLayout, NamespaceLifeId::fromCatalogEntry(live.ns, live.incarnation), 5000); + f.backend->resetCounts(); + + const NamespaceJanitorResult result = f.run(f.context()); + + EXPECT_EQ(result.pages, 1u); + EXPECT_EQ(result.batches, 0u); + EXPECT_EQ(f.backend->namespaceLists(), 1u); + EXPECT_EQ(f.backend->writeCount(kLayout.gcMaintenanceStateKey()), 1u); +} + +TEST(CASNamespaceJanitor, DeferredRoundTakesOneSuppressedPage) +{ + JanitorFixture f; + const std::vector keys = seedLogs(*f.backend, kLayout, life("dead", 226), 3000); + f.backend->resetCounts(); + + const NamespaceJanitorResult result = f.run(f.context(), /*suppress_deletes=*/ true); + + EXPECT_EQ(result.pages, 1u); + EXPECT_EQ(result.deleted, 0u); + EXPECT_TRUE(f.backend->bulkCalls().empty()); + EXPECT_FALSE(f.cursor().has_value()); + EXPECT_EQ(f.backend->writeCount(kLayout.gcMaintenanceStateKey()), 0u); +} + +TEST(CASNamespaceJanitor, MidStreamPassWrapsToTheStart) +{ + JanitorFixture f; + const std::vector keys = seedLogs(*f.backend, kLayout, life("dead", 228), 5000); + createObj(*f.backend, kLayout.gcMaintenanceStateKey(), encodeGcMaintenanceState({.janitor_cursor = keys[1999]})); + f.backend->resetCounts(); + + const NamespaceJanitorResult result = f.run(f.context()); + + EXPECT_EQ(f.backend->namespaceLists(), 5u) << "pages 3, 4, 5, then 1, 2; the page that began the pass is not listed again"; + EXPECT_EQ(result.pages, 5u); + EXPECT_EQ(result.deleted, 5000u); + EXPECT_TRUE(allAbsent(*f.backend, keys, 0, 5000)); + EXPECT_EQ(f.cursor(), String{}); + EXPECT_EQ(f.backend->writeCount(kLayout.gcMaintenanceStateKey()), 1u) << "one cursor publication per phase"; +} + +TEST(CASNamespaceJanitor, WrapStopsAtAnAllLivePage) +{ + const CatalogEntry live{.ns = RootNamespace{"live"}, .state = NsState::Live, .incarnation = UInt128{229}}; + JanitorFixture f(RefCatalog{.entries = {live}}); + const NamespaceLifeId live_life = NamespaceLifeId::fromCatalogEntry(live.ns, live.incarnation); + const NamespaceLifeId dead_life = life("dead", 230); + ASSERT_LT(kLayout.refLogKey(live_life, RefTxnId{1, 1}), kLayout.refLogKey(dead_life, RefTxnId{1, 1})); + std::vector keys = seedLogs(*f.backend, kLayout, live_life, 1000); + const std::vector dead_keys = seedLogs(*f.backend, kLayout, dead_life, 4000); + keys.insert(keys.end(), dead_keys.begin(), dead_keys.end()); + createObj(*f.backend, kLayout.gcMaintenanceStateKey(), encodeGcMaintenanceState({.janitor_cursor = keys[1999]})); + f.backend->resetCounts(); + + (void)f.run(f.context()); + + EXPECT_EQ(f.backend->namespaceLists(), 4u) << "pages 3, 4, 5, then the all-live page 1; page 2 is not listed"; + EXPECT_TRUE(allPresent(*f.backend, keys, 0, 2000)); + EXPECT_TRUE(allAbsent(*f.backend, keys, 2000, 5000)); + EXPECT_EQ(f.cursor(), String{}); +} + +TEST(CASNamespaceJanitor, WrapDoesNotRevisitKeysPastTheStartCursor) +{ + JanitorFixture f; + f.backend->setBatchDeleteSupported(false); + const std::vector keys = seedLogs(*f.backend, kLayout, life("dead", 231), 5000); + /// A key that stays behind after the first part: the wrapped part must not list it a second time. + f.backend->before_bulk = [bad = keys[3000]](const JanitorBackend::Keys & batch, JanitorBackend::Access &) + { + if (batch.front().str() == bad) + throw Poco::TimeoutException("injected throttle, forever"); + }; + createObj(*f.backend, kLayout.gcMaintenanceStateKey(), encodeGcMaintenanceState({.janitor_cursor = keys[1499]})); + f.backend->resetCounts(); + JanitorRunContext context = f.context(f.backend->bulkDeleteKeyLimit()); + context.budget_ms = 10'000'000; + + const NamespaceJanitorResult result = f.run(context); + + EXPECT_EQ(result.leaked, 1u) << "the stuck key is attempted once per phase"; + EXPECT_EQ(result.deleted, 4999u); + EXPECT_EQ(f.backend->namespaceLists(), 6u) << "4 pages from the cursor to the end, then 2 up to the cursor"; + EXPECT_TRUE(present(*f.backend, keys[3000])); + EXPECT_EQ(f.cursor(), String{}); +} + +TEST(CASNamespaceJanitor, MidStreamPassOverLivePagesTakesOnePage) +{ + const CatalogEntry live{.ns = RootNamespace{"live"}, .state = NsState::Live, .incarnation = UInt128{232}}; + JanitorFixture f(RefCatalog{.entries = {live}}); + const std::vector keys = seedLogs(*f.backend, kLayout, NamespaceLifeId::fromCatalogEntry(live.ns, live.incarnation), 5000); + createObj(*f.backend, kLayout.gcMaintenanceStateKey(), encodeGcMaintenanceState({.janitor_cursor = keys[3999]})); + f.backend->resetCounts(); + + (void)f.run(f.context()); + + EXPECT_EQ(f.backend->namespaceLists(), 1u) << "an all-live page that reaches the end stops the pass before any wrap"; + EXPECT_TRUE(allPresent(*f.backend, keys, 0, 5000)); + EXPECT_EQ(f.cursor(), String{}); +} + +TEST(CASNamespaceJanitor, HeldPageKeepsTheCursorUntilItClears) +{ + JanitorFixture f; + (void)seedLogs(*f.backend, kLayout, life("dead", 228), 4000); + JanitorRunContext one_page = f.context(); + one_page.budget_ms = 0; + + const NamespaceJanitorResult a = f.run(one_page); + EXPECT_TRUE(a.cursor_advanced); + const std::optional after_a = f.cursor(); + + std::atomic hold{true}; + f.backend->before_bulk = [&](const JanitorBackend::Keys &, JanitorBackend::Access &) + { + if (hold) + throw DB::Exception(DB::ErrorCodes::CORRUPTED_DATA, "injected local failure"); + }; + const NamespaceJanitorResult b = f.run(one_page); + const NamespaceJanitorResult c = f.run(one_page); + EXPECT_FALSE(b.cursor_advanced); + EXPECT_FALSE(c.cursor_advanced); + EXPECT_EQ(f.cursor(), after_a); + + hold = false; + const NamespaceJanitorResult d = f.run(one_page); + const NamespaceJanitorResult e = f.run(one_page); + EXPECT_TRUE(d.cursor_advanced); + EXPECT_TRUE(e.cursor_advanced); + EXPECT_NE(f.cursor(), after_a); +} + +TEST(CASNamespaceJanitor, NonPocoExceptionHoldsThePage) +{ + JanitorFixture f; + const std::vector keys = seedLogs(*f.backend, kLayout, life("dead", 230), 2000); + f.backend->before_bulk = [&](const JanitorBackend::Keys &, JanitorBackend::Access &) + { + throw std::runtime_error("injected non-Poco failure"); + }; + + const NamespaceJanitorResult result = f.run(f.context()); + + EXPECT_EQ(result.batches_held, 1u); + EXPECT_EQ(result.batches_leaked, 0u); + EXPECT_EQ(result.leaked, 0u); + EXPECT_FALSE(f.cursor().has_value()); + EXPECT_TRUE(allPresent(*f.backend, keys, 0, 2000)); +} + +TEST(CASNamespaceJanitor, AuthorityLostBeforeRetryHolds) +{ + JanitorFixture f; + const std::vector keys = seedLogs(*f.backend, kLayout, life("dead", 230), 3000); + f.backend->failNextBulkRemoveWith(std::make_exception_ptr(Poco::TimeoutException("injected throttle"))); + JanitorRunContext context = f.context(500); + context.liveness = [&] { return f.backend->bulkRemoveCalls() == 0; }; + + const NamespaceJanitorResult result = f.run(context); + + EXPECT_EQ(f.backend->bulkRemoveCalls(), 1u); + EXPECT_TRUE(allPresent(*f.backend, keys, 0, 1000)); + EXPECT_EQ(result.batches_held, 1u); + EXPECT_EQ(result.leaked, 0u); + EXPECT_EQ(result.batches, 1u); + EXPECT_EQ(f.backend->bulkCalls().size(), 1u) << "no second batch"; + EXPECT_EQ(f.backend->namespaceLists(), 1u); + EXPECT_FALSE(f.cursor().has_value()); +} + +TEST(CASNamespaceJanitor, AuthorityLostDuringSuccessfulBatchCounts) +{ + JanitorFixture f; + const std::vector keys = seedLogs(*f.backend, kLayout, life("dead", 231), 3000); + std::atomic held{true}; + f.backend->onBeforeBulkRemove([&] { held = false; }); + JanitorRunContext context = f.context(500); + context.liveness = [&] { return held.load(); }; + + const NamespaceJanitorResult result = f.run(context); + + EXPECT_EQ(f.backend->bulkRemoveCalls(), 1u); + EXPECT_TRUE(allAbsent(*f.backend, keys, 0, 500)); + EXPECT_TRUE(allPresent(*f.backend, keys, 500, 1000)); + EXPECT_EQ(result.batches, 1u); + EXPECT_EQ(result.deleted, 500u); + EXPECT_EQ(result.batches_held, 0u); + EXPECT_EQ(result.leaked, 0u); + EXPECT_EQ(f.backend->namespaceLists(), 1u); + EXPECT_FALSE(f.cursor().has_value()); +} + +TEST(CASNamespaceJanitor, EmptyNamespaceTakesOnePage) +{ + JanitorFixture f; + const NamespaceJanitorResult result = f.run(f.context()); + EXPECT_EQ(result.pages, 1u); + EXPECT_EQ(result.keys, 0u); + EXPECT_EQ(result.delete_jobs, 0u); + EXPECT_EQ(f.cursor(), String{}); +} + +TEST(CASNamespaceJanitor, FirstPageCatalogReadFailureThrowsAndPublishesNothing) +{ + JanitorFixture f; + const std::vector keys = seedLogs(*f.backend, kLayout, life("dead", 232), 10); + f.backend->failNextReadWith(kLayout.refCatalogKey(), + std::make_exception_ptr(DB::Exception(DB::ErrorCodes::CORRUPTED_DATA, "injected catalog read failure"))); + + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { (void)f.run(f.context()); }); + + EXPECT_TRUE(f.backend->bulkCalls().empty()); + EXPECT_FALSE(f.cursor().has_value()); + EXPECT_TRUE(allPresent(*f.backend, keys, 0, keys.size())); +} + +TEST(CASNamespaceJanitor, ClockGoingBackwardsDoesNotExhaustTheBudget) +{ + JanitorFixture f; + const std::vector keys = seedLogs(*f.backend, kLayout, life("dead", 233), 2000); + f.backend->after_namespace_list = [&](size_t index, JanitorBackend::Access &) + { + if (index == 0) + f.clock.now -= 5'000; + }; + JanitorRunContext context = f.context(); + context.budget_ms = 20'000; + + const NamespaceJanitorResult result = f.run(context); + + EXPECT_EQ(f.backend->namespaceLists(), 2u); + EXPECT_FALSE(result.budget_exhausted); + EXPECT_TRUE(allAbsent(*f.backend, keys, 0, 2000)); +} + +TEST(CASNamespaceJanitorIntegration, PoolWrappedStorageLimitCutsJanitorBatches) +{ + auto backend = std::make_shared(); + backend->setStorageLimit(250); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds=*/ 0); + const Layout & layout = store->layout(); + (void)seedLiveNamespace(*backend, layout); + /// With the live checkpoint, 999 dead keys fill exactly one page. + const std::vector keys = seedLogs(*backend, layout, life("dead", 1), 999); + std::map row; + Gc gc(store, UInt128{252}); + gc.setPhaseSink([&](const GcPhaseRecord & record) { if (record.phase == "namespace_cleanup") row = record.metrics; }); + + ASSERT_TRUE(runRegularRoundReclaiming(gc).acquired_lease); + gc.setPhaseSink({}); + + EXPECT_EQ(row["batch_keys"], 250u); + EXPECT_EQ(row["batches"], 4u) << "a decorator that dropped the forwarding would send 999 one-key requests"; + for (const String & key : keys) + EXPECT_FALSE(present(*backend, key)) << key; +} + +TEST(CASNamespaceJanitorIntegration, FoldingRoundDrainsWithinBudget) +{ + auto backend = std::make_shared(); + auto store = openPoolForTest(backend, 0); + const Layout & layout = store->layout(); + const NamespaceLifeId live = seedLiveNamespace(*backend, layout); + const NamespaceLifeId dead = life("dead", 1); + /// Dead `_files` list before the live life's, so the pass reaches live keys after 3,500 dead ones. + ASSERT_LT(layout.namespaceFilesPrefix(dead), layout.namespaceFilesPrefix(live)); + std::vector dead_keys; + for (size_t i = 0; i < 3500; ++i) + { + dead_keys.push_back(layout.namespaceFilesPrefix(dead) + fmt::format("f{:05}", i)); + createObj(*backend, dead_keys.back(), "dead"); + } + for (size_t i = 0; i < 1500; ++i) + createObj(*backend, layout.namespaceFilesPrefix(live) + fmt::format("f{:05}", i), "live"); + backend->resetCounts(); + std::map row; + Gc gc(store, UInt128{254}); + gc.setPhaseSink([&](const GcPhaseRecord & record) { if (record.phase == "namespace_cleanup") row = record.metrics; }); + + ASSERT_TRUE(runRegularRoundReclaiming(gc).acquired_lease); + gc.setPhaseSink({}); + + EXPECT_EQ(row["janitor_pages"], 5u) << "four pages with dead keys, then one without"; + EXPECT_EQ(row["janitor_deleted"], 3500u); + EXPECT_EQ(row["budget_exhausted"], 0u); + EXPECT_EQ(backend->writeCount(layout.gcMaintenanceStateKey()), 1u) << "one cursor publication per phase"; + for (const String & key : dead_keys) + EXPECT_FALSE(present(*backend, key)) << key; +} + +TEST(CASNamespaceJanitorIntegration, GcRoundStopsAtTheBudgetOnItsMonotonicClockAndTheNextRoundResumes) +{ + auto backend = std::make_shared(); + auto store = Pool::open( + backend, PoolConfig{.pool_prefix = "p", .server_root_id = "test", .gc_fold_max_defer_rounds = 0}); + const Layout & layout = store->layout(); + const NamespaceLifeId live = seedLiveNamespace(*backend, layout); + const NamespaceLifeId dead = life("dead", 1); + ASSERT_LT(layout.namespaceFilesPrefix(dead), layout.namespaceFilesPrefix(live)); + std::vector dead_keys; + for (size_t i = 0; i < 3500; ++i) + { + dead_keys.push_back(layout.namespaceFilesPrefix(dead) + fmt::format("f{:05}", i)); + createObj(*backend, dead_keys.back(), "dead"); + } + for (size_t i = 0; i < 1500; ++i) + createObj(*backend, layout.namespaceFilesPrefix(live) + fmt::format("f{:05}", i), "live"); + uint64_t mono_ms = 1'000; + /// Each clock read moves past the 20 s budget, so the phase stops after its first page. + uint64_t mono_step_ms = 30'000; + std::map row; + Gc gc(store, UInt128{255}, {}, [&] { return mono_ms += mono_step_ms; }); + gc.setPhaseSink([&](const GcPhaseRecord & record) { if (record.phase == "namespace_cleanup") row = record.metrics; }); + + ASSERT_TRUE(runRegularRoundReclaiming(gc).acquired_lease); + + EXPECT_EQ(row["budget_exhausted"], 1u); + EXPECT_EQ(row["janitor_pages"], 1u) << "the first page runs; the clock has passed the budget before a second starts"; + EXPECT_EQ(row["cursor_advanced"], 1u); + EXPECT_EQ(row["janitor_deleted"], 1000u); + EXPECT_TRUE(allAbsent(*backend, dead_keys, 0, 1000)); + EXPECT_TRUE(allPresent(*backend, dead_keys, 1000, 3500)); + + mono_step_ms = 0; + ASSERT_TRUE(runRegularRoundReclaiming(gc).acquired_lease); + gc.setPhaseSink({}); + + EXPECT_EQ(row["budget_exhausted"], 0u); + EXPECT_EQ(row["janitor_deleted"], 2500u) << "the second round resumes at the persisted cursor"; + EXPECT_TRUE(allAbsent(*backend, dead_keys, 0, 3500)); +} + +TEST(CASNamespaceJanitor, HeldPageReportsNoLeaks) +{ + JanitorFixture f; + const std::vector keys = seedLogs(*f.backend, kLayout, life("dead", 302), 1000); + f.backend->before_bulk = [&](const JanitorBackend::Keys & batch, JanitorBackend::Access &) + { + if (batch.front().str() == keys[0]) + throw Poco::TimeoutException("injected throttle, every attempt"); + throw DB::Exception(DB::ErrorCodes::CORRUPTED_DATA, "injected local failure"); + }; + + const NamespaceJanitorResult result = f.run(f.context(500)); + + EXPECT_EQ(result.batches_held, 1u); + EXPECT_EQ(result.deleted, 0u); + EXPECT_FALSE(f.cursor().has_value()) << "a held page publishes nothing"; + EXPECT_EQ(result.leaked, 0u) << "its keys are listed again next round, so none is leaked yet"; +} + +TEST(CASNamespaceJanitor, ExhaustedTransientBatchLeaksAndAdvances) +{ + JanitorFixture f; + const std::vector keys = seedLogs(*f.backend, kLayout, life("dead", 301), 1500); + f.backend->before_bulk = [&](const JanitorBackend::Keys &, JanitorBackend::Access &) + { + throw Poco::TimeoutException("injected throttle, every attempt"); + }; + JanitorRunContext context = f.context(); + context.budget_ms = 20'000; + + const NamespaceJanitorResult result = f.run(context); + + const auto calls = f.backend->bulkCalls(); + ASSERT_GT(calls.size(), 1u) << "the policy reissues the batch before giving up"; + for (const auto & call : calls) + EXPECT_EQ(call.keys, calls.front().keys) << "no per-key or partial re-cut"; + EXPECT_EQ(result.leaked, 1000u); + EXPECT_EQ(result.batches_leaked, 1u); + EXPECT_EQ(result.batches_held, 0u); + EXPECT_TRUE(f.backend->exactRemoves().empty()); + EXPECT_TRUE(result.budget_exhausted) << "the 90 s retry window on the phase clock exceeds the budget"; + EXPECT_EQ(f.cursor(), keys[999]); +} + +TEST(CASNamespaceJanitor, SlowDownForeverDoesNotStarveLaterDebris) +{ + const auto seeded = [](JanitorFixture & f) + { + f.backend->setBatchDeleteSupported(false); + const std::vector keys = seedLogs(*f.backend, kLayout, life("dead", 302), 3000); + f.backend->before_bulk = [bad = keys[499]](const JanitorBackend::Keys & batch, JanitorBackend::Access &) + { + if (batch.front().str() == bad) + throw Poco::TimeoutException("injected throttle, forever"); + }; + return keys; + }; + { + SCOPED_TRACE("budget above the retry window"); + JanitorFixture f; + const std::vector keys = seeded(f); + JanitorRunContext context = f.context(f.backend->bulkDeleteKeyLimit()); + context.budget_ms = 10'000'000; + const NamespaceJanitorResult result = f.run(context); + EXPECT_EQ(result.leaked, 1u); + EXPECT_EQ(result.batches_leaked, 1u); + EXPECT_EQ(result.deleted, 2999u); + EXPECT_TRUE(present(*f.backend, keys[499])); + EXPECT_EQ(f.cursor(), String{}); + } + { + SCOPED_TRACE("default budget"); + JanitorFixture f; + const std::vector keys = seeded(f); + JanitorRunContext context = f.context(f.backend->bulkDeleteKeyLimit()); + context.budget_ms = 20'000; + const NamespaceJanitorResult first = f.run(context); + EXPECT_TRUE(first.budget_exhausted); + EXPECT_EQ(f.cursor(), keys[999]); + const NamespaceJanitorResult second = f.run(context); + EXPECT_EQ(second.deleted, 2000u); + EXPECT_EQ(f.cursor(), String{}); + const NamespaceJanitorResult wrapped = f.run(context); + EXPECT_EQ(wrapped.leaked, 1u) << "the next pass leaks the same key again"; + EXPECT_EQ(wrapped.keys, 1u); + } +} + +TEST(CASNamespaceJanitor, PerKeyErrorInOkResponseLeaksTheBatch) +{ + JanitorFixture f; + const std::vector keys = seedLogs(*f.backend, kLayout, life("dead", 303), 1000); + std::atomic fail{true}; + f.backend->before_bulk = [&](const JanitorBackend::Keys & batch, JanitorBackend::Access & access) + { + if (!fail) + return; + for (const WriteOnceKey & key : batch) + if (key.str() != keys[500]) + f.backend->removeUncounted(key.str(), access); + throw Poco::TimeoutException("the survivor answers a throttle on every attempt"); + }; + + const NamespaceJanitorResult result = f.run(f.context()); + + EXPECT_EQ(result.deleted, 0u) << "a failed batch counts nothing as deleted"; + EXPECT_EQ(result.leaked, 1000u); + EXPECT_EQ(result.batches_leaked, 1u); + EXPECT_EQ(f.cursor(), String{}); + fail = false; + const NamespaceJanitorResult next = f.run(f.context()); + EXPECT_EQ(next.keys, 1u) << "the next pass lists only the survivor"; + EXPECT_EQ(next.deleted, 1u); +} + +TEST(CASNamespaceJanitor, UnknownCapabilityRejectionStopsThePhase) +{ + JanitorFixture f; + f.backend->setStoreRejectsBatches(true); + const std::vector keys = seedLogs(*f.backend, kLayout, life("dead", 304), 3000); + + const NamespaceJanitorResult result = f.run(f.context(f.backend->bulkDeleteKeyLimit())); + + EXPECT_EQ(f.backend->requestSizes(), (std::vector{1000})); + EXPECT_EQ(f.backend->callSizes().size(), 1u) << "no one-key call for the rejected keys"; + EXPECT_EQ(f.backend->namespaceLists(), 1u); + EXPECT_FALSE(f.cursor().has_value()); + EXPECT_EQ(result.batches_held, 1u); + + const NamespaceJanitorResult next = f.run(f.context(f.backend->bulkDeleteKeyLimit())); + EXPECT_EQ(next.batch_keys, 1u); + EXPECT_EQ(next.deleted, 3000u); + EXPECT_EQ(f.cursor(), String{}); +} + +TEST(CASNamespaceJanitor, CapabilityFlippedByAnotherUserStopsThePhase) +{ + JanitorFixture f; + const std::vector keys = seedLogs(*f.backend, kLayout, life("dead", 305), 3000); + f.backend->after_bulk = [&](const JanitorBackend::Keys &) { f.backend->setBatchDeleteSupported(false); }; + + const NamespaceJanitorResult result = f.run(f.context()); + + EXPECT_EQ(f.backend->requestSizes(), (std::vector{1000})); + EXPECT_EQ(f.backend->callSizes().size(), 2u) << "page 2's batch is refused without a request"; + EXPECT_EQ(f.backend->namespaceLists(), 2u); + EXPECT_EQ(f.cursor(), keys[999]); +} diff --git a/src/Disks/tests/gtest_cas_namespace_janitor_pipeline.cpp b/src/Disks/tests/gtest_cas_namespace_janitor_pipeline.cpp new file mode 100644 index 000000000000..a6fb51c89f67 --- /dev/null +++ b/src/Disks/tests/gtest_cas_namespace_janitor_pipeline.cpp @@ -0,0 +1,321 @@ +#include "cas_namespace_janitor_test_helpers.h" +#include +#include +#include +#include +#include +#include +#include +#include + +/// The janitor phase with jobs on a real pool of 4 threads. Assertions hold for any interleaving. + +namespace DB::ErrorCodes +{ + extern const int CORRUPTED_DATA; +} + +using namespace DB::Cas; +using namespace DB::Cas::tests; +using namespace DB::Cas::tests::janitor; + +namespace +{ + +const Layout & kLayout = janitorLayout(); + +JanitorRunContext parallel(JanitorFixture & f, ThreadPool & pool, size_t batch_keys) +{ + JanitorRunContext context = f.context(batch_keys); + context.io_pool = &pool; + return context; +} + +size_t countPresent(Backend & backend, const std::vector & keys, size_t begin, size_t end) +{ + size_t count = 0; + for (size_t i = begin; i < end; ++i) + count += present(backend, keys[i]) ? 1 : 0; + return count; +} + +} + +TEST(CASNamespaceJanitorPipeline, PipelineListsNextPageWhileJobsRun) +{ + JanitorFixture f; + ManualBarrier page_two_listed; + const std::vector keys = seedLogs(*f.backend, kLayout, life("dead", 401), 3000); + f.backend->before_bulk = [&](const JanitorBackend::Keys & batch, JanitorBackend::Access &) + { + if (batch.front().str() == keys[0]) + page_two_listed.arriveAndWait(); + }; + f.backend->after_namespace_list = [&](size_t index, JanitorBackend::Access &) + { + if (index == 1) + page_two_listed.release(); + }; + auto pool = makeJobPool(4); + + const NamespaceJanitorResult result = f.run(parallel(f, *pool, 1000)); + + EXPECT_EQ(result.batches_held, 0u) << "a round thread that waited for the batch would time the barrier out"; + EXPECT_EQ(countPresent(*f.backend, keys, 0, 3000), 0u); + EXPECT_EQ(f.cursor(), String{}); +} + +TEST(CASNamespaceJanitorPipeline, NoBatchCapabilitySendsOneKeyRequestsWithoutProbe) +{ + JanitorFixture f; + f.backend->setBatchDeleteSupported(false); + const std::vector keys = seedLogs(*f.backend, kLayout, life("dead", 403), 1000); + auto pool = makeJobPool(4); + + const NamespaceJanitorResult result = f.run(parallel(f, *pool, f.backend->bulkDeleteKeyLimit())); + + const auto calls = f.backend->bulkCalls(); + ASSERT_EQ(calls.size(), 1000u); + for (const auto & call : calls) + { + EXPECT_EQ(call.keys.size(), 1u); + EXPECT_NE(call.thread, std::this_thread::get_id()); + } + EXPECT_TRUE(f.backend->exactRemoves().empty()); + EXPECT_EQ(result.batch_keys, 1u); + EXPECT_EQ(result.delete_jobs, 1000u); + EXPECT_EQ(result.batches_held, 0u); + EXPECT_EQ(f.backend->callSizes(), f.backend->requestSizes()) << "no locally refused multi-key call"; +} + +TEST(CASNamespaceJanitorPipeline, HeldPageStopsCursorAndLaterPagesAreNotDeleted) +{ + JanitorFixture f; + const std::vector keys = seedLogs(*f.backend, kLayout, life("dead", 404), 3000); + f.backend->before_bulk = [&](const JanitorBackend::Keys & batch, JanitorBackend::Access &) + { + if (batch.front().str() == keys[1000]) + throw DB::Exception(DB::ErrorCodes::CORRUPTED_DATA, "injected local failure"); + }; + auto pool = makeJobPool(4); + + const NamespaceJanitorResult result = f.run(parallel(f, *pool, 1000)); + + EXPECT_EQ(countPresent(*f.backend, keys, 0, 1000), 0u); + EXPECT_EQ(countPresent(*f.backend, keys, 1000, 2000), 1000u); + EXPECT_EQ(countPresent(*f.backend, keys, 2000, 3000), 3000u - 2000u) << "the page behind a held one is never deleted"; + EXPECT_EQ(f.cursor(), keys[999]); + EXPECT_EQ(result.batches_held, 1u); + const auto calls = f.backend->bulkCalls(); + EXPECT_EQ(std::count_if(calls.begin(), calls.end(), [&](const auto & call) { return call.keys.front() == keys[1000]; }), 1) + << "a local failure is not retried"; + EXPECT_GE(f.backend->namespaceLists(), 2u); + EXPECT_LE(f.backend->namespaceLists(), 3u) << "the next page is listed at most once, while the held page's job runs"; +} + +TEST(CASNamespaceJanitorPipeline, HeldJobSkipsItsPageSiblingsAndHoldsTheCursor) +{ + JanitorFixture f; + ManualBarrier first_job; + const std::vector keys = seedLogs(*f.backend, kLayout, life("dead", 411), 3000); + f.backend->before_bulk = [&](const JanitorBackend::Keys & batch, JanitorBackend::Access &) + { + if (batch.front().str() == keys[0]) + throw DB::Exception(DB::ErrorCodes::CORRUPTED_DATA, "injected local failure"); + }; + f.backend->after_namespace_list = [&](size_t index, JanitorBackend::Access &) + { + if (index == 1) + first_job.release(); + }; + /// One thread runs page 0's two jobs in order. The first waits until page 1 is listed, so both are enqueued + /// before the first one fails and the second finds the stop flag. + auto pool = makeJobPool(1); + JanitorRunContext context = parallel(f, *pool, 500); + context.on_job_start_for_test = [&](size_t page, size_t job) + { + if (page == 0 && job == 0) + first_job.arriveAndWait(); + }; + + const NamespaceJanitorResult result = f.run(context); + + EXPECT_EQ(result.batches_held, 1u); + EXPECT_EQ(result.delete_jobs_skipped, 1u); + EXPECT_EQ(countPresent(*f.backend, keys, 0, 1000), 1000u) << "the failed job and the skipped one sent nothing"; + EXPECT_FALSE(f.cursor().has_value()) << "the held page is not passed"; + EXPECT_EQ(result.leaked, 0u); + EXPECT_EQ(result.pages, 1u) << "the page listed behind the held one is dropped"; + EXPECT_EQ(f.backend->namespaceLists(), 2u); +} + +TEST(CASNamespaceJanitorPipeline, ExceptionalExitWaitsForOutstandingJobs) +{ + JanitorFixture f; + ManualBarrier job; + std::atomic job_finished{false}; + (void)seedLogs(*f.backend, kLayout, life("dead", 412), 2000); + auto pool = makeJobPool(4); + JanitorRunContext context = parallel(f, *pool, 1000); + size_t refreshes = 0; + context.refresh_authority = [&] + { + if (++refreshes == 2) + throw std::runtime_error("injected authority refresh failure"); + }; + context.on_job_start_for_test = [&](size_t, size_t) + { + job.arriveAndWait(); + job_finished = true; + }; + + auto run = std::async(std::launch::async, [&] { f.run(context); }); + job.waitUntilArrived(); + /// With the guard `run` cannot return while the job is blocked, so this wait times out; without it `run` + /// has already thrown, with the job still holding references into its frame. + const bool returned_early = run.wait_for(std::chrono::seconds(1)) == std::future_status::ready; + job.release(); + EXPECT_THROW(run.get(), std::runtime_error); + + EXPECT_FALSE(returned_early) << "run returned while a job was still outstanding"; + EXPECT_TRUE(job_finished.load()); +} + +TEST(CASNamespaceJanitorPipeline, ParallelPathScheduleRefusalHolds) +{ + JanitorFixture f; + f.backend->setBatchDeleteSupported(false); + const std::vector keys = seedLogs(*f.backend, kLayout, life("dead", 408), 2000); + auto pool = makeJobPool(4); + std::optional refuse_at = 10; + JanitorRunContext context = parallel(f, *pool, 1); + context.schedule_refuse_at_for_test = &refuse_at; + + const NamespaceJanitorResult result = f.run(context); + + EXPECT_FALSE(refuse_at.has_value()) << "the seam resets once it fires"; + EXPECT_EQ(result.delete_jobs, 10u); + EXPECT_EQ(result.batches + result.delete_jobs_skipped, 10u); + EXPECT_EQ(f.backend->bulkCalls().size(), result.batches); + EXPECT_EQ(result.deleted, result.batches); + EXPECT_EQ(2000 - countPresent(*f.backend, keys, 0, 2000), result.batches); + EXPECT_EQ(result.leaked, 0u); + EXPECT_EQ(result.batches_held, 1u) << "the refusal"; + EXPECT_EQ(f.backend->namespaceLists(), 1u) << "a refusal sets the stop flag before the next page is listed"; + EXPECT_FALSE(f.cursor().has_value()); +} + +namespace +{ + +/// Blocks the `gc/state` read of the janitor's second refresh -- the first one after its first LIST -- +/// until the test releases it. +class RefreshGateBackend : public JanitorBackend +{ +public: + RefreshGateBackend(ManualBarrier & gate_, String gc_state_key_) : gate(gate_), gc_state_key(std::move(gc_state_key_)) {} + + std::optional read(const String & key, TransportAccess & access) override + { + if (key == gc_state_key && namespaceLists() == 1 && !gated.exchange(true)) + gate.arriveAndWait(); + return JanitorBackend::read(key, access); + } + +private: + ManualBarrier & gate; + const String gc_state_key; + std::atomic gated{false}; +}; + +} + +TEST(CASNamespaceJanitorIntegration, RefreshInFlightKeepsAuthority) +{ + ManualBarrier refresh_read; + std::atomic page_one_applied{0}; + std::array, 2> first_attempt_failed{}; + std::vector keys; + std::map row; + auto backend = std::make_shared(refresh_read, Layout("p").gcStateKey()); + backend->setStorageLimit(500); + PoolConfig config{.pool_prefix = "p", .server_root_id = "test", .gc_fold_max_defer_rounds = 0}; + config.gc_io_concurrency = 4; + auto store = Pool::open(backend, config); + const Layout & layout = store->layout(); + (void)seedLiveNamespace(*backend, layout); + /// With the live checkpoint, 2,999 dead keys fill exactly three pages: page 1 has 999, cut 500 + 499. + keys = seedLogs(*backend, layout, life("dead", 1), 2999); + backend->before_bulk = [&](const JanitorBackend::Keys & batch, JanitorBackend::Access &) + { + /// Each page-1 job fails once after the refresh read has begun, so its reissue samples authority + /// while the refresh is in flight. + for (size_t job = 0; job < first_attempt_failed.size(); ++job) + { + if (batch.front().str() == keys[500 * job] && !first_attempt_failed[job].exchange(true)) + { + refresh_read.waitUntilArrived(); + throw Poco::TimeoutException("injected throttle; the reissue is gated inside the refresh"); + } + } + }; + backend->after_bulk = [&](const JanitorBackend::Keys & batch) + { + const String & first = batch.front().str(); + if ((first == keys[0] || first == keys[500]) && ++page_one_applied == 2) + refresh_read.release(); + }; + Gc gc(store, UInt128{411}); + gc.setPhaseSink([&](const GcPhaseRecord & record) { if (record.phase == "namespace_cleanup") row = record.metrics; }); + + ASSERT_TRUE(runRegularRoundReclaiming(gc).acquired_lease); + gc.setPhaseSink({}); + + EXPECT_EQ(row["batches_held"], 0u) << "a refresh that stores false first would hold both page-1 jobs"; + EXPECT_EQ(row["delete_jobs_skipped"], 0u); + EXPECT_EQ(row["janitor_deleted"], 2999u); + for (const String & key : keys) + EXPECT_FALSE(present(*backend, key)) << key; + EXPECT_EQ(publishedCursor(store->openRequests(), layout), String{}); +} + +TEST(CASNamespaceJanitorIntegration, TeardownDuringPhaseHoldsWithoutLeaking) +{ + Pool * store_ptr = nullptr; + std::map row; + auto backend = std::make_shared(); + backend->setStorageLimit(500); + PoolConfig config{.pool_prefix = "p", .server_root_id = "test", .gc_fold_max_defer_rounds = 0}; + config.gc_io_concurrency = 1; + auto store = Pool::open(backend, config); + store_ptr = store.get(); + const Layout & layout = store->layout(); + (void)seedLiveNamespace(*backend, layout); + const std::vector keys = seedLogs(*backend, layout, life("dead", 1), 999); + /// One pool thread runs the two jobs in order: the teardown begins in the first job's request, so the + /// second is refused at admission. + std::atomic torn_down{false}; + backend->before_bulk = [&](const JanitorBackend::Keys &, JanitorBackend::Access &) + { + if (!torn_down.exchange(true)) + store_ptr->beginTeardown(); + }; + Gc gc(store, UInt128{412}); + gc.setPhaseSink([&](const GcPhaseRecord & record) { if (record.phase == "namespace_cleanup") row = record.metrics; }); + + /// The round's later phases fail on the closed open plane; only the janitor's accounting is asserted. + try + { + (void)runRegularRoundReclaiming(gc); + } + catch (const DB::Exception &) + { + } + gc.setPhaseSink({}); + + EXPECT_GE(row["batches_held"] + row["delete_jobs_skipped"], 1u); + EXPECT_EQ(row["leaked"], 0u); + CasOperation verify = store->mountRequests().admit(); + for (size_t i = 500; i < 999; ++i) + EXPECT_TRUE(verify.head(keys[i], Retry::once()).has_value()) << "refused at admission: " << keys[i]; +} diff --git a/src/Disks/tests/gtest_cas_namespace_janitor_s3.cpp b/src/Disks/tests/gtest_cas_namespace_janitor_s3.cpp new file mode 100644 index 000000000000..370e8b85cb9a --- /dev/null +++ b/src/Disks/tests/gtest_cas_namespace_janitor_s3.cpp @@ -0,0 +1,203 @@ +#include "config.h" + +#if USE_AWS_S3 + +#include "cas_namespace_janitor_test_helpers.h" +#include +#include +#include + +/// Janitor tests that need `S3Exception` classification or the real S3 delete adapter. + +using namespace DB::Cas; +using namespace DB::Cas::tests; +using namespace DB::Cas::tests::janitor; +using namespace DB::Cas::tests::s3; + +namespace +{ + +const Layout & kLayout = janitorLayout(); + +std::exception_ptr accessDenied() +{ + return std::make_exception_ptr(DB::S3Exception("Access Denied", Aws::S3::S3Errors::ACCESS_DENIED, "AccessDenied")); +} + +/// The in-memory store whose `removeManyWriteOnce` goes through a real `S3ObjectStorage` against a scripted +/// server: the adapter's request shapes, error names and capability learning, every other verb in memory. +/// After each call it mirrors the keys the server reports removed. +class S3DeletePathBackend : public CountingBackend +{ +public: + S3DeletePathBackend(std::shared_ptr storage_, std::shared_ptr deleted_) + : storage(std::move(storage_)), deleted(std::move(deleted_)) {} + + size_t bulkDeleteKeyLimit() const override { return storage->batchDeleteKeyLimit(); } + + void removeManyWriteOnce(const std::vector & keys, TransportAccess & access) override + { + DB::StoredObjects objects; + for (const WriteOnceKey & key : keys) + objects.emplace_back(key.str()); + std::exception_ptr failure; + try + { + storage->removeObjectsIfExistUnderProfile(objects, DB::ObjectStorageControlRequest{}); + } + catch (...) + { + failure = std::current_exception(); + } + std::vector gone; + for (const WriteOnceKey & key : keys) + { + if (deleted->contains(key.str())) + gone.push_back(key); + } + InMemoryBackend::removeManyWriteOnce(gone, access); // NOLINT(bugprone-parent-virtual-call) + if (failure) + std::rethrow_exception(failure); + } + +private: + std::shared_ptr storage; + std::shared_ptr deleted; +}; + +/// Accepts every delete: a one-key DELETE and every key of a `DeleteObjects` are recorded removed. +void acceptAll(ScriptedDeletes & deleted, const Poco::Net::HTTPServerRequest & request, const std::string & body, + Poco::Net::HTTPServerResponse & response) +{ + if (request.getMethod() == "DELETE") + { + deleted.add(keyOfDeleteObject(request)); + sendDeleteObjectSuccess(response); + return; + } + for (const std::string & key : keysOfDeleteObjects(body)) + deleted.add(key); + sendDeleteObjectsResult(response, {}); +} + +/// The janitor over `S3DeletePathBackend`. `respond` must be set before the first request. +struct S3PathFixture +{ + std::shared_ptr deleted = std::make_shared(); + std::function respond + = [this](const auto & request, const std::string & body, auto & response) { acceptAll(*deleted, request, body, response); }; + ScriptedS3Server server{[this](const auto & request, const std::string & body, auto & response) { respond(request, body, response); }}; + FakeClock clock; + std::shared_ptr backend; + std::unique_ptr requests; + + explicit S3PathFixture(const DB::S3Capabilities & capabilities) + { + (void)contextForTest(); + backend = std::make_shared(makeStorageForTest(server.getUrl(), capabilities), deleted); + requests = std::make_unique(backend, Fence::open(), clock.nowFn(), clock.sleepFn()); + seedCatalog(*backend, kLayout); + } + + JanitorRunContext context() + { + JanitorRunContext result; + result.batch_keys = backend->bulkDeleteKeyLimit(); + result.now_ms = clock.nowFn(); + return result; + } + + NamespaceJanitorResult run(const JanitorRunContext & context) { return runPhase(*requests, kLayout, context); } + std::optional cursor() { return publishedCursor(*requests, kLayout); } +}; + +} + +TEST(CASNamespaceJanitorS3, PermanentRefusalLeaksAndAdvances) +{ + JanitorFixture f; + const std::vector keys = seedLogs(*f.backend, kLayout, life("dead", 311), 2000); + std::atomic refuse{true}; + f.backend->before_bulk = [&](const JanitorBackend::Keys &, JanitorBackend::Access &) + { + if (refuse.exchange(false)) + std::rethrow_exception(accessDenied()); + }; + + const NamespaceJanitorResult result = f.run(f.context()); + + EXPECT_EQ(f.backend->bulkCalls().size(), 2u) << "one attempt for the refused batch, one for page 2"; + EXPECT_EQ(result.leaked, 1000u); + EXPECT_EQ(result.batches_leaked, 1u); + for (size_t i = 1000; i < 2000; ++i) + EXPECT_FALSE(present(*f.backend, keys[i])) << keys[i]; + EXPECT_EQ(f.cursor(), String{}); +} + +TEST(CASNamespaceJanitorS3, NameOnlyRefusalLeaksAndAdvances) +{ + S3PathFixture f(DB::S3Capabilities{}); + const std::vector keys = seedLogs(*f.backend, kLayout, life("dead", 314), 1500); + f.respond = [&](const Poco::Net::HTTPServerRequest &, const std::string & body, Poco::Net::HTTPServerResponse & response) + { + bool refuse = false; + for (const std::string & key : keysOfDeleteObjects(body)) + { + if (key == keys[3]) + refuse = true; + else + f.deleted->add(key); + } + if (refuse) + sendDeleteObjectsResult(response, {{keys[3], "EntityTooLarge"}}); + else + sendDeleteObjectsResult(response, {}); + }; + + const NamespaceJanitorResult result = f.run(f.context()); + + EXPECT_EQ(f.server.countMethod("POST"), 2u) << "one attempt for the refused batch, one for page 2"; + EXPECT_EQ(result.leaked, 1000u); + EXPECT_EQ(result.batches_leaked, 1u); + EXPECT_TRUE(present(*f.backend, keys[3])); + for (size_t i = 1000; i < 1500; ++i) + EXPECT_FALSE(present(*f.backend, keys[i])) << keys[i]; + EXPECT_EQ(f.cursor(), String{}); +} + +TEST(CASNamespaceJanitorS3, UnknownCapabilityConcurrentProbesAllStop) +{ + S3PathFixture f(DB::S3Capabilities{}); + const std::vector keys = seedLogs(*f.backend, kLayout, life("dead", 422), 3000); + std::atomic rejected{0}; + f.respond = [&](const Poco::Net::HTTPServerRequest & request, const std::string & body, Poco::Net::HTTPServerResponse & response) + { + if (request.getMethod() == "POST") + { + ++rejected; + sendBatchNotImplemented(response); + return; + } + acceptAll(*f.deleted, request, body, response); + }; + auto pool = makeJobPool(4); + JanitorRunContext context = f.context(); + context.io_pool = pool.get(); + + (void)f.run(context); + + EXPECT_GE(rejected.load(), 1u); + EXPECT_LE(rejected.load(), 4u); + EXPECT_EQ(f.server.countMethod("DELETE"), 0u); + EXPECT_FALSE(f.cursor().has_value()); + EXPECT_EQ(f.backend->bulkDeleteKeyLimit(), 1u) << "the storage remembers false"; + + JanitorRunContext next = f.context(); + next.io_pool = pool.get(); + const NamespaceJanitorResult drained = f.run(next); + EXPECT_EQ(drained.batch_keys, 1u); + EXPECT_EQ(drained.deleted, 3000u); + EXPECT_EQ(f.server.countMethod("DELETE"), 3000u); +} + +#endif diff --git a/src/Disks/tests/gtest_cas_ref_gc.cpp b/src/Disks/tests/gtest_cas_ref_gc.cpp index 835ddb197cbe..75be760bce26 100644 --- a/src/Disks/tests/gtest_cas_ref_gc.cpp +++ b/src/Disks/tests/gtest_cas_ref_gc.cpp @@ -1001,67 +1001,115 @@ TEST(CASRefGc, RefObjectCleanupDeletesExactlyThePlannedSet) << "snapshot " << renderRefTxnId(id) << " not in the plan must survive"; } -/// The same planned set as above, but the object storage rejects the cohort's one bulk -/// `removeManyWriteOnce` as NOT_IMPLEMENTED (a GCS-backed pool): `cleanupRefObjects`' call site falls -/// back to one admitted request per key (`removeChunkWriteOnceOrOneByOne`, CasGc.h), and the outcome -- -/// which keys are gone, and the budget/profile-event accounting -- must be identical to the plain -/// bulk-request path above. -TEST(CASRefGc, RefObjectCleanupFallsBackToOnePerKeyWhenBatchDeleteIsUnsupported) +namespace { - auto backend = std::make_shared(); - auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds*/ 0); - const Layout & layout = store->layout(); - const RootNamespace ns{"00/aa@cas@"}; - fixture::admitLive(*backend, store->layout(), ns); - const ManifestRef r1 = mref(1); - const ManifestRef r2 = mref(2); - writeManifestRaw(*backend, layout, ns, r1, {blobEntryFor("a", DB::UInt128(1))}); - writeManifestRaw(*backend, layout, ns, r2, {blobEntryFor("b", DB::UInt128(2))}); - const uint64_t v1 = publishCommittedTransition(*backend, layout, ns, "t1", std::nullopt, r1); - const uint64_t v2 = publishCommittedTransition(*backend, layout, ns, "t2", std::nullopt, r2); +struct RefCleanupSeed +{ + NamespaceLifeId life; + RefCleanupPlan plan; +}; - RefTableSnapshot old_snap = minimalLiveSnapshot(ns.string(), RefTxnId{1, v1}, - {committedRow("t1", r1)}); - RefTableSnapshot new_snap = minimalLiveSnapshot(ns.string(), RefTxnId{1, v2}, - {committedRow("t1", r1), committedRow("t2", r2)}); - writeRefSnapshotRaw(*backend, layout, old_snap); - writeRefSnapshotRaw(*backend, layout, new_snap); - replaceRecoverableCkptForRawFixture(*backend, layout, ns, RefCkpt{ +/// Publishes `logs` committed transitions (`t1`..`tN`), writes the first and the last snapshot, points the +/// checkpoint at the last one and returns the cleanup plan the GC computes for it. +RefCleanupSeed seedRefCleanup(Backend & backend, const Layout & layout, const RootNamespace & ns, uint64_t logs) +{ + std::vector refs; + std::vector versions; + for (uint64_t i = 1; i <= logs; ++i) + { + refs.push_back(mref(i)); + writeManifestRaw(backend, layout, ns, refs.back(), {blobEntryFor("a" + std::to_string(i), DB::UInt128(i))}); + } + for (uint64_t i = 1; i <= logs; ++i) + versions.push_back(publishCommittedTransition(backend, layout, ns, "t" + std::to_string(i), std::nullopt, refs[i - 1])); + + const auto rowsUpTo = [&](uint64_t n) + { + std::vector result; + for (uint64_t i = 1; i <= n; ++i) + result.push_back(committedRow("t" + std::to_string(i), refs[i - 1])); + /// A snapshot orders its rows by name, which "t10" breaks against "t9". + std::sort(result.begin(), result.end(), [](const RefCommittedRow & lhs, const RefCommittedRow & rhs) { return lhs.ref_name < rhs.ref_name; }); + return result; + }; + writeRefSnapshotRaw(backend, layout, minimalLiveSnapshot(ns.string(), RefTxnId{1, versions.front()}, rowsUpTo(1))); + writeRefSnapshotRaw(backend, layout, minimalLiveSnapshot(ns.string(), RefTxnId{1, versions.back()}, rowsUpTo(logs))); + replaceRecoverableCkptForRawFixture(backend, layout, ns, RefCkpt{ .life_epoch = 1, - .committed_through = RefTxnId{1, v2}, - .checkpoint_snapshot_id = RefTxnId{1, v2}, + .committed_through = RefTxnId{1, versions.back()}, + .checkpoint_snapshot_id = RefTxnId{1, versions.back()}, .last_epoch_seal = std::nullopt, }); - const NamespaceLifeId life = fixture::fixtureLife(ns); - const RefTableListing listing{ - .logs = {RefTxnId{1, v1}, RefTxnId{1, v2}}, - .snapshots = {RefTxnId{1, v1}, RefTxnId{1, v2}}}; - const RefTxnId durable_cursor{1, v2}; - const RefTxnId checkpoint_snapshot_id{1, v2}; - const RefCleanupPlan plan = planRefCleanup(listing, durable_cursor, checkpoint_snapshot_id, std::nullopt); - const uint64_t cohort_size = plan.deletable_logs.size() + plan.deletable_snapshots.size(); - ASSERT_GT(cohort_size, 0u) << "the fixture must actually have something to delete for this test to prove anything"; + RefTableListing listing; + for (const uint64_t version : versions) + listing.logs.push_back(RefTxnId{1, version}); + listing.snapshots = {RefTxnId{1, versions.front()}, RefTxnId{1, versions.back()}}; + const RefTxnId last{1, versions.back()}; + return RefCleanupSeed{fixture::fixtureLife(ns), planRefCleanup(listing, last, last, std::nullopt)}; +} - /// One armed failure: the cohort's own bulk `removeManyWriteOnce` call fails as "batch delete not - /// supported"; the fallback's per-key calls that follow are not armed and succeed. - backend->failNextBulkRemoveWith(std::make_exception_ptr( - DB::Exception(DB::ErrorCodes::NOT_IMPLEMENTED, "no batch delete"))); - const auto cleaned_before = ProfileEvents::global_counters[ProfileEvents::CASRefCleanupObjectsDeleted].load(); +} +TEST(CASRefGc, RefCleanupContainsCapabilityRejection) +{ + std::map row; + auto backend = std::make_shared(); + backend->setStoreRejectsBatches(true); + auto store = openPoolForTest(backend, /*gc_fold_max_defer_rounds*/ 0); + const Layout & layout = store->layout(); + const RootNamespace ns{"00/aa@cas@"}; + fixture::admitLive(*backend, layout, ns); + const RefCleanupSeed seeded = seedRefCleanup(*backend, layout, ns, 3); + const size_t cohort_size = seeded.plan.deletable_logs.size() + seeded.plan.deletable_snapshots.size(); + ASSERT_GE(cohort_size, 2u) << "a one-key cohort would not reach the batch rejection"; + + Gc gc(store, kGc); + gc.setPhaseSink([&](const GcPhaseRecord & record) { if (record.phase == "ref_object_cleanup") row = record.metrics; }); + ASSERT_NO_THROW(ASSERT_TRUE(runRegularRoundReclaiming(gc).acquired_lease)); + EXPECT_EQ(row["capability_learned"], 1u); + EXPECT_EQ(backend->requestSizes(), (std::vector{cohort_size})) << "no one-key request that round"; OperationForTest op(*backend); + for (const RefTxnId & id : seeded.plan.deletable_logs) + EXPECT_TRUE((*op).head(layout.refLogKey(seeded.life, id), Retry::once()).has_value()); + + store->renewWatermarkOnce(); + ASSERT_TRUE(runRegularRoundReclaiming(gc).acquired_lease); + for (const RefTxnId & id : seeded.plan.deletable_logs) + EXPECT_FALSE((*op).head(layout.refLogKey(seeded.life, id), Retry::once()).has_value()); + for (const RefTxnId & id : seeded.plan.deletable_snapshots) + EXPECT_FALSE((*op).head(layout.refSnapshotKey(seeded.life, id), Retry::once()).has_value()); + const std::vector requests = backend->requestSizes(); + EXPECT_TRUE(std::all_of(requests.begin() + 1, requests.end(), [](size_t n) { return n == 1; })) + << "the next round cuts one-key requests"; +} + +TEST(CASRefGc, RefCleanupReadsAuthorityOncePerCohort) +{ + /// Scaled from 2,500 keys in cohorts of 1,000: 11 keys in cohorts of 4 give the same three cohorts. + PhaseReads reads; + auto backend = std::make_shared(); + backend->setBatchDeleteSupported(false); + PoolConfig config{.pool_prefix = "p", .server_root_id = "test", .gc_fold_max_defer_rounds = 0}; + config.gc_bulk_delete_chunk_keys = 4; + auto store = Pool::open(backend, config); + const Layout & layout = store->layout(); + const RootNamespace ns{"00/aa@cas@"}; + fixture::admitLive(*backend, layout, ns); + const RefCleanupSeed seeded = seedRefCleanup(*backend, layout, ns, 11); + const size_t cohort_size = seeded.plan.deletable_logs.size() + seeded.plan.deletable_snapshots.size(); + ASSERT_GT(cohort_size, 8u); + ASSERT_LE(cohort_size, 12u) << "the assertions below expect exactly three cohorts of at most 4"; + Gc gc(store, kGc); + gc.setPhaseSink(phaseReadsSink(reads, *backend, layout, "namespace_cleanup", "ref_object_cleanup")); ASSERT_TRUE(runRegularRoundReclaiming(gc).acquired_lease); - for (const RefTxnId & id : plan.deletable_logs) - EXPECT_FALSE((*op).head(layout.refLogKey(life, id), Retry::once()).has_value()); - for (const RefTxnId & id : plan.deletable_snapshots) - EXPECT_FALSE((*op).head(layout.refSnapshotKey(life, id), Retry::once()).has_value()); - EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRefCleanupObjectsDeleted].load() - cleaned_before, cohort_size) - << "the budget/profile-event accounting counts objects, unaffected by the fallback"; - /// 1 failed bulk attempt + one request per key in the cohort. - EXPECT_EQ(backend->bulkRemoveCalls(), 1 + cohort_size); + EXPECT_EQ(reads.catalog_in_phase, 3u); + EXPECT_EQ(reads.state_in_phase, 3u); + const std::vector calls = backend->callSizes(); + EXPECT_EQ(calls, std::vector(cohort_size, 1)) << "a store without batch delete takes one one-key request per object, the authority read once per cohort of 4 keys"; } /// Task 13 (spec §implementation-impact / §GC Budget): one fold+clean round increments every ref-intake diff --git a/src/Disks/tests/gtest_cas_s3_bulk_delete_fallback.cpp b/src/Disks/tests/gtest_cas_s3_bulk_delete_fallback.cpp index c0e4ee5f5306..41143b39a231 100644 --- a/src/Disks/tests/gtest_cas_s3_bulk_delete_fallback.cpp +++ b/src/Disks/tests/gtest_cas_s3_bulk_delete_fallback.cpp @@ -13,6 +13,8 @@ #include #include #include +#include +#include #include #include @@ -38,198 +40,27 @@ namespace DB::ErrorCodes extern const int NOT_IMPLEMENTED; } +using namespace DB::Cas::tests::s3; + /// `S3ObjectStorage::removeObjectsIfExistImpl` (the CAS bulk-delete path, reached through /// `removeObjectsIfExistUnderProfile`) must honour `S3Capabilities::isBatchDeleteSupported()` the same /// way the generic `deleteFilesFromS3` does, but WITHOUT looping over the objects itself: once the /// capability is known false (a configured `false`, or one just learned from a `DeleteObjects` reply in /// the "batch delete not implemented" error class), it throws `NOT_IMPLEMENTED` without sending anything -/// else, and leaves per-key retry to the caller (the CAS engine admits each such retry as its own -/// request -- see CasGc.cpp's `removeChunkWriteOnceOrOneByOne`). A request failure of any other class +/// else, and leaves the decision to the caller (the CAS GC stops the family for the round -- see +/// `removeCohortWriteOnce`, CasGc.h). A request failure of any other class /// must keep today's fail-close behaviour. The one exception to all of this is a batch of exactly one /// object, which is always a plain `DeleteObject` -- never gated on the capability at all, since a /// single physical request is never something the capability check exists to rule out. -namespace -{ - -/// A real local HTTP server standing in for S3. `DeleteObjects` arrives as a POST to the bucket root; -/// a per-key `DeleteObject` arrives as a plain HTTP DELETE to the key's path -- the two are -/// distinguished by HTTP method alone, with no need to parse the request body or query string. -class ScriptedS3Server -{ -public: - using Responder = std::function; - -private: - class Handler : public Poco::Net::HTTPRequestHandler - { - ScriptedS3Server & owner; - - public: - explicit Handler(ScriptedS3Server & owner_) : owner(owner_) { } - - void handleRequest(Poco::Net::HTTPServerRequest & request, Poco::Net::HTTPServerResponse & response) override - { - { - std::lock_guard lock(owner.mutex); - owner.methods_seen.push_back(request.getMethod()); - } - /// `DeleteObjects` carries a request body (the XML `` payload); leaving it unread on a - /// keep-alive connection makes Poco parse those leftover bytes as the start of the NEXT - /// request once this handler returns, corrupting the very next `DeleteObject` this test expects. - request.stream().ignore(std::numeric_limits::max()); - owner.responder(request, response); - } - }; - class Factory : public Poco::Net::HTTPRequestHandlerFactory - { - ScriptedS3Server & owner; - - Poco::Net::HTTPRequestHandler * createRequestHandler(const Poco::Net::HTTPServerRequest &) override - { - return new Handler(owner); - } - - public: - explicit Factory(ScriptedS3Server & owner_) : owner(owner_) { } - }; - - std::unique_ptr server_socket; - Poco::SharedPtr handler_factory; - Poco::AutoPtr server_params; - std::unique_ptr server; - Responder responder; - mutable std::mutex mutex; - std::vector methods_seen; - -public: - explicit ScriptedS3Server(Responder responder_) - : server_socket(std::make_unique(0)) - , handler_factory(new Factory(*this)) - , server_params(new Poco::Net::HTTPServerParams()) - , server(std::make_unique(handler_factory, *server_socket, server_params)) - , responder(std::move(responder_)) - { - server->start(); - } - - std::string getUrl() const { return "http://" + server_socket->address().toString(); } - - size_t countMethod(const std::string & method) const - { - std::lock_guard lock(mutex); - return static_cast(std::count(methods_seen.begin(), methods_seen.end(), method)); - } -}; - -void sendXml(Poco::Net::HTTPServerResponse & response, Poco::Net::HTTPResponse::HTTPStatus status, const std::string & body) -{ - response.setContentType("application/xml"); - response.setContentLength(body.size()); - response.setStatus(status); - auto & out = response.send(); - out << body; - out.flush(); -} - -/// A quiet-mode `DeleteObjects` success (HTTP 200) whose body lists only the failed keys, exactly as a -/// real S3 backend would report a mixed outcome. -void sendBatchSuccessWithErrors(Poco::Net::HTTPServerResponse & response, const std::string & not_found_key, const std::string & denied_key) -{ - const std::string body = - "" - "" - "" + not_found_key + "NoSuchKeyThe specified key does not exist." - "" + denied_key + "AccessDeniedAccess Denied" - ""; - sendXml(response, Poco::Net::HTTPResponse::HTTP_OK, body); -} - -/// A request-level `DeleteObjects` failure in the "batch delete is not implemented" class that -/// `deleteFileFromS3.cpp`'s `deleteFilesFromS3` also treats as "fall back to plain `DeleteObject`". -void sendBatchNotImplemented(Poco::Net::HTTPServerResponse & response) -{ - const std::string body = - "" - "NotImplementedA header you provided implies functionality that is not implemented"; - sendXml(response, Poco::Net::HTTPResponse::HTTP_BAD_REQUEST, body); -} - -/// A request-level `DeleteObjects` failure in an ordinary (not "unsupported") class: this must keep -/// today's fail-close behaviour and never fall back to per-key deletes. -void sendBatchInternalError(Poco::Net::HTTPServerResponse & response) -{ - const std::string body = - "" - "InternalErrorWe encountered an internal error, please try again."; - sendXml(response, Poco::Net::HTTPResponse::HTTP_INTERNAL_SERVER_ERROR, body); -} - -void sendDeleteObjectSuccess(Poco::Net::HTTPServerResponse & response) -{ - response.setContentLength(0); - response.setStatus(Poco::Net::HTTPResponse::HTTP_NO_CONTENT); - response.send(); -} - -/// A single-key `DeleteObject` failure -- used to script the size-one path's own error handling, as -/// distinct from the batch response's per-key `` elements covered by the test above. -void sendSingleDeleteError(Poco::Net::HTTPServerResponse & response, Poco::Net::HTTPResponse::HTTPStatus status, const std::string & code, const std::string & message) -{ - const std::string body = - "" - "" + code + "" + message + ""; - sendXml(response, status, body); -} - -std::shared_ptr makeStorageForTest(const std::string & endpoint, const DB::S3Capabilities & capabilities) -{ - DB::RemoteHostFilter remote_host_filter; - DB::S3::PocoHTTPClientConfiguration cfg = DB::S3::ClientFactory::instance().createClientConfiguration( - "us-east-1", - remote_host_filter, - /* s3_max_redirects = */ 100, - DB::S3::PocoHTTPClientConfiguration::RetryStrategy{.max_retries = 0}, - /* s3_slow_all_threads_after_network_error = */ false, - /* s3_slow_all_threads_after_retryable_error = */ false, - /* enable_s3_requests_logging = */ false, - /* for_disk_s3 = */ true, - /* opt_disk_name = */ {}, - /* request_throttler = */ {}); - cfg.endpointOverride = endpoint; - cfg.connectTimeoutMs = 10000; - cfg.requestTimeoutMs = 10000; - cfg.s3_use_adaptive_timeouts = false; - /// Every test here starts its own server on an ephemeral port; with keep-alive on, the process-wide - /// HTTP connection pool can hand a later test a connection to a port whose server is already gone - /// (`Connection reset by peer` under `--gtest_repeat`). One connection per request is what a - /// short-lived test server should get. - cfg.http_keep_alive_timeout = 0; - auto client = DB::S3::ClientFactory::instance().create( - cfg, - DB::S3::ClientSettings{ - .use_virtual_addressing = false, - .disable_checksum = false, - .gcs_issue_compose_request = false, - .is_s3express_bucket = false, - }, - "ACCESS_KEY_ID", "SECRET_ACCESS_KEY", "", {}, {}, DB::S3::CredentialsConfiguration{}); - return std::make_shared( - std::move(client), std::make_unique(), - DB::S3::URI(endpoint + "/test-bucket/"), capabilities, - DB::ObjectStorageKeyGeneratorPtr{}, "disk"); -} - -DB::ContextPtr contextForTest() +namespace { - return getContext().context; -} -/// The CAS-side fallback (CasGc.cpp's `removeChunkWriteOnceOrOneByOne`) keys specifically on +/// The two callers of `removeCohortWriteOnce` (CasGc.h) key specifically on /// `NOT_IMPLEMENTED`; a capability-rejection test that only checks "threw a DB::Exception" would still -/// pass if this storage started throwing, say, BAD_ARGUMENTS instead -- which would silently break that -/// fallback while every assertion here kept passing. +/// pass if this storage started throwing, say, BAD_ARGUMENTS instead -- which would make those callers +/// treat a capability rejection as a real error while every assertion here kept passing. void expectNotImplemented(const std::function & fn) { try @@ -249,7 +80,7 @@ TEST(S3BulkDeleteFallback, PerKeyErrorsWithinASuccessfulBatchAreUnchanged) { (void)contextForTest(); - ScriptedS3Server server([](const Poco::Net::HTTPServerRequest &, Poco::Net::HTTPServerResponse & response) + ScriptedS3Server server([](const Poco::Net::HTTPServerRequest &, const std::string &, Poco::Net::HTTPServerResponse & response) { sendBatchSuccessWithErrors(response, "notfound-key", "denied-key"); }); @@ -279,7 +110,7 @@ TEST(S3BulkDeleteFallback, UnsupportedBatchReplyRecordsCapabilityFalseAndThrowsN (void)contextForTest(); std::atomic batch_attempts{0}; - ScriptedS3Server server([&](const Poco::Net::HTTPServerRequest & request, Poco::Net::HTTPServerResponse & response) + ScriptedS3Server server([&](const Poco::Net::HTTPServerRequest & request, const std::string &, Poco::Net::HTTPServerResponse & response) { ASSERT_EQ(request.getMethod(), "POST") << "capability false must never send anything, batch or per-key"; ++batch_attempts; @@ -303,16 +134,15 @@ TEST(S3BulkDeleteFallback, UnsupportedBatchReplyRecordsCapabilityFalseAndThrowsN /// A batch of exactly one object is always a plain `DeleteObject`: never sent as `DeleteObjects`, and /// never gated on `s3_capabilities` at all -- proven here with the capability both explicitly false AND /// left unknown (the default), since a single physical request is never something that check exists to -/// refuse. This is what makes the CAS engine's per-key fallback (CasGc.cpp) actually delete anything on -/// a backend that rejects `DeleteObjects` outright (GCS): a "batch" of one sent as `DeleteObjects` would -/// fail there identically to a bigger one. +/// refuse. A storage that rejects `DeleteObjects` outright (GCS) would fail a "batch" of one sent as +/// `DeleteObjects` identically to a bigger one. TEST(S3BulkDeleteFallback, ExactlyOneObjectIsAlwaysAPlainDeleteObjectRegardlessOfCapability) { (void)contextForTest(); for (const bool explicit_false : {false, true}) { - ScriptedS3Server server([](const Poco::Net::HTTPServerRequest & request, Poco::Net::HTTPServerResponse & response) + ScriptedS3Server server([](const Poco::Net::HTTPServerRequest & request, const std::string &, Poco::Net::HTTPServerResponse & response) { ASSERT_EQ(request.getMethod(), "DELETE"); sendDeleteObjectSuccess(response); @@ -333,7 +163,7 @@ TEST(S3BulkDeleteFallback, ExactlyOneObjectIgnoresAbsenceAndThrowsOnARealError) (void)contextForTest(); { - ScriptedS3Server server([](const Poco::Net::HTTPServerRequest &, Poco::Net::HTTPServerResponse & response) + ScriptedS3Server server([](const Poco::Net::HTTPServerRequest &, const std::string &, Poco::Net::HTTPServerResponse & response) { sendSingleDeleteError(response, Poco::Net::HTTPResponse::HTTP_NOT_FOUND, "NoSuchKey", "The specified key does not exist."); }); @@ -341,7 +171,7 @@ TEST(S3BulkDeleteFallback, ExactlyOneObjectIgnoresAbsenceAndThrowsOnARealError) EXPECT_NO_THROW(storage->removeObjectsIfExistUnderProfile({DB::StoredObject("absent-key")}, DB::ObjectStorageControlRequest{})); } { - ScriptedS3Server server([](const Poco::Net::HTTPServerRequest &, Poco::Net::HTTPServerResponse & response) + ScriptedS3Server server([](const Poco::Net::HTTPServerRequest &, const std::string &, Poco::Net::HTTPServerResponse & response) { sendSingleDeleteError(response, Poco::Net::HTTPResponse::HTTP_FORBIDDEN, "AccessDenied", "Access Denied"); }); @@ -363,7 +193,7 @@ TEST(S3BulkDeleteFallback, OtherFailureClassesKeepFailingClosedWithNoFallback) { (void)contextForTest(); - ScriptedS3Server server([](const Poco::Net::HTTPServerRequest &, Poco::Net::HTTPServerResponse & response) + ScriptedS3Server server([](const Poco::Net::HTTPServerRequest &, const std::string &, Poco::Net::HTTPServerResponse & response) { sendBatchInternalError(response); }); @@ -382,7 +212,7 @@ TEST(S3BulkDeleteFallback, ExplicitlyDisabledCapabilityThrowsNotImplementedWitho { (void)contextForTest(); - ScriptedS3Server server([](const Poco::Net::HTTPServerRequest &, Poco::Net::HTTPServerResponse &) + ScriptedS3Server server([](const Poco::Net::HTTPServerRequest &, const std::string &, Poco::Net::HTTPServerResponse &) { FAIL() << "an explicit false capability must never send anything, batch or per-key"; }); @@ -399,4 +229,110 @@ TEST(S3BulkDeleteFallback, ExplicitlyDisabledCapabilityThrowsNotImplementedWitho EXPECT_EQ(server.countMethod("DELETE"), 0u); } +TEST(CASS3BatchDelete, S3BatchDeleteKeyLimitFollowsChunkSetting) +{ + (void)contextForTest(); + ScriptedS3Server server([](const Poco::Net::HTTPServerRequest &, const std::string &, Poco::Net::HTTPServerResponse & response) + { + sendBatchNotImplemented(response); + }); + + EXPECT_EQ(makeStorageForTest(server.getUrl(), DB::S3Capabilities{})->batchDeleteKeyLimit(), 1000u); + EXPECT_EQ(makeStorageForTest(server.getUrl(), DB::S3Capabilities{}, {.objects_chunk_size_to_delete = 500})->batchDeleteKeyLimit(), 500u); + EXPECT_EQ(makeStorageForTest(server.getUrl(), DB::S3Capabilities{}, {.objects_chunk_size_to_delete = 0})->batchDeleteKeyLimit(), 1u); + EXPECT_EQ(makeStorageForTest(server.getUrl(), DB::S3Capabilities{false})->batchDeleteKeyLimit(), 1u); + + auto learning = makeStorageForTest(server.getUrl(), DB::S3Capabilities{}, {.objects_chunk_size_to_delete = 500}); + expectNotImplemented([&] + { + learning->removeObjectsIfExistUnderProfile({DB::StoredObject("key-a"), DB::StoredObject("key-b")}, DB::ObjectStorageControlRequest{}); + }); + EXPECT_EQ(learning->batchDeleteKeyLimit(), 1u) << "a learned false capability caps every later GC request at one key"; +} + +namespace +{ + +void expectRefusalNamed(const std::function & remove, const std::string & name) +{ + try + { + remove(); + FAIL() << "expected an S3Exception named " << name; + } + catch (const DB::S3Exception & e) + { + EXPECT_EQ(e.getExceptionName(), name) << e.message(); + EXPECT_TRUE(DB::Cas::isDefinitelyRefusedWrite(e)) << e.message(); + } +} + +const DB::StoredObjects kTwoObjects{DB::StoredObject("key-a"), DB::StoredObject("key-b")}; + +} + +TEST(CASS3BatchDelete, S3BatchErrorKeepsCanonicalName) +{ + (void)contextForTest(); + for (const std::string code : {"EntityTooLarge", "MalformedXML"}) + { + SCOPED_TRACE("whole-response " + code); + ScriptedS3Server server([code](const Poco::Net::HTTPServerRequest &, const std::string &, Poco::Net::HTTPServerResponse & response) + { + sendSingleDeleteError(response, Poco::Net::HTTPResponse::HTTP_BAD_REQUEST, code, "refused"); + }); + auto storage = makeStorageForTest(server.getUrl(), DB::S3Capabilities{}); + expectRefusalNamed([&] { storage->removeObjectsIfExistUnderProfile(kTwoObjects, DB::ObjectStorageControlRequest{}); }, code); + } + { + SCOPED_TRACE("per-key EntityTooLarge"); + ScriptedS3Server server([](const Poco::Net::HTTPServerRequest &, const std::string &, Poco::Net::HTTPServerResponse & response) + { + sendDeleteObjectsResult(response, {{"key-b", "EntityTooLarge"}}); + }); + auto storage = makeStorageForTest(server.getUrl(), DB::S3Capabilities{}); + expectRefusalNamed([&] { storage->removeObjectsIfExistUnderProfile(kTwoObjects, DB::ObjectStorageControlRequest{}); }, "EntityTooLarge"); + } + { + /// A NoSuchKey ahead of the real errors is skipped; the first real error names the exception. + SCOPED_TRACE("per-key order: NoSuchKey, AccessDenied, EntityTooLarge"); + ScriptedS3Server server([](const Poco::Net::HTTPServerRequest &, const std::string &, Poco::Net::HTTPServerResponse & response) + { + sendDeleteObjectsResult(response, {{"key-a", "NoSuchKey"}, {"key-b", "AccessDenied"}, {"key-c", "EntityTooLarge"}}); + }); + auto storage = makeStorageForTest(server.getUrl(), DB::S3Capabilities{}); + expectRefusalNamed([&] { storage->removeObjectsIfExistUnderProfile(kTwoObjects, DB::ObjectStorageControlRequest{}); }, "AccessDenied"); + } + { + SCOPED_TRACE("one-key EntityTooLarge through deleteFileFromS3"); + ScriptedS3Server server([](const Poco::Net::HTTPServerRequest &, const std::string &, Poco::Net::HTTPServerResponse & response) + { + sendSingleDeleteError(response, Poco::Net::HTTPResponse::HTTP_BAD_REQUEST, "EntityTooLarge", "refused"); + }); + auto storage = makeStorageForTest(server.getUrl(), DB::S3Capabilities{false}); + expectRefusalNamed([&] { storage->removeObjectsIfExistUnderProfile({DB::StoredObject("solo")}, DB::ObjectStorageControlRequest{}); }, "EntityTooLarge"); + EXPECT_EQ(server.countMethod("DELETE"), 1u); + } +} + +TEST(CASS3BatchDelete, NameOnlyRefusalWithRefreshCallbackCostsAtMostOneExtraRequest) +{ + (void)contextForTest(); + ScriptedS3Server server([](const Poco::Net::HTTPServerRequest &, const std::string &, Poco::Net::HTTPServerResponse & response) + { + sendDeleteObjectsResult(response, {{"key-b", "EntityTooLarge"}}); + }); + const std::string url = server.getUrl(); + std::atomic callbacks{0}; + auto storage = makeStorageForTest(url, DB::S3Capabilities{}, {.credentials_refresh_callback = [&]() -> std::unique_ptr + { + ++callbacks; + return makeClientForTest(url); + }}); + + EXPECT_THROW(storage->removeObjectsIfExistUnderProfile(kTwoObjects, DB::ObjectStorageControlRequest{}), DB::S3Exception); + EXPECT_EQ(server.countMethod("POST"), 2u); + EXPECT_EQ(callbacks.load(), 1u); +} + #endif diff --git a/src/Disks/tests/gtest_cas_write_once_key.cpp b/src/Disks/tests/gtest_cas_write_once_key.cpp index c05eb6b1e345..d9fd69c6756d 100644 --- a/src/Disks/tests/gtest_cas_write_once_key.cpp +++ b/src/Disks/tests/gtest_cas_write_once_key.cpp @@ -30,3 +30,25 @@ TEST(CASWriteOnceKey, FactoriesMintTheSameStringsAsThePlainKeyFunctions) EXPECT_TRUE(layout.parseManifestKey(layout.writeOnceManifestKey(manifest).str()).has_value()); EXPECT_TRUE(layout.parseRefObjectKey(layout.writeOnceRefLogKey(life, id).str()).has_value()); } + +TEST(CASWriteOnceKey, StreamKeyMintedFromAListedKeyMustBeThatKey) +{ + const Layout layout{"p"}; + const NamespaceLifeId life = NamespaceLifeId::fromCatalogEntry(RootNamespace{"test/aa@cas@"}, DB::UInt128(0x1234)); + const RefTxnId id{5, 7}; + + for (const String & listed : {layout.refLogKey(life, id), layout.refSnapshotKey(life, id)}) + { + const auto parsed = layout.parseRefObjectKey(listed); + ASSERT_TRUE(parsed.has_value()) << listed; + const auto minted = layout.writeOnceStreamKey(*parsed, listed); + ASSERT_TRUE(minted.has_value()) << listed; + EXPECT_EQ(minted->str(), listed); + } + + const auto parsed_log = layout.parseRefObjectKey(layout.refLogKey(life, id)); + ASSERT_TRUE(parsed_log.has_value()); + EXPECT_FALSE(layout.writeOnceStreamKey(*parsed_log, layout.refSnapshotKey(life, id)).has_value()); + EXPECT_FALSE(layout.writeOnceStreamKey(*parsed_log, layout.refLogKey(life, RefTxnId{5, 8})).has_value()); + EXPECT_FALSE(layout.writeOnceStreamKey(*parsed_log, "q" + layout.refLogKey(life, id).substr(1)).has_value()); +} diff --git a/src/IO/S3/deleteFileFromS3.cpp b/src/IO/S3/deleteFileFromS3.cpp index 0111e00534a8..900f2c8cd4c3 100644 --- a/src/IO/S3/deleteFileFromS3.cpp +++ b/src/IO/S3/deleteFileFromS3.cpp @@ -74,8 +74,10 @@ void deleteFileFromS3( else { const auto & err = outcome.GetError(); - throw S3Exception(err.GetErrorType(), "{} (Code: {}) while removing object with path {} from S3", - err.GetMessage(), static_cast(err.GetErrorType()), key); + throw S3Exception( + PreformattedMessage::create("{} (Code: {}) while removing object with path {} from S3", + err.GetMessage(), static_cast(err.GetErrorType()), key), + err.GetErrorType(), err.GetExceptionName()); } } diff --git a/tests/integration/test_cas_janitor_drain/__init__.py b/tests/integration/test_cas_janitor_drain/__init__.py new file mode 100644 index 000000000000..e69de29bb2d1 diff --git a/tests/integration/test_cas_janitor_drain/configs/storage_conf.xml b/tests/integration/test_cas_janitor_drain/configs/storage_conf.xml new file mode 100644 index 000000000000..eefb95801286 --- /dev/null +++ b/tests/integration/test_cas_janitor_drain/configs/storage_conf.xml @@ -0,0 +1,50 @@ + + + + + object_storage + s3 + cas + itest-janitor-batch + http://rustfs1:11121/test/cas_janitor_batch/ + clickhouse + clickhouse + 1 + 86400 + + 1 + + + object_storage + s3 + cas + itest-janitor-small-chunk + http://rustfs1:11121/test/cas_janitor_small_chunk/ + clickhouse + clickhouse + 100 + 1 + 86400 + 1 + + + object_storage + s3 + cas + itest-janitor-no-batch + http://rustfs1:11121/test/cas_janitor_no_batch/ + clickhouse + clickhouse + false + 1 + 86400 + 1 + + + +
cas_batch
+
cas_small_chunk
+
cas_no_batch
+
+
+
diff --git a/tests/integration/test_cas_janitor_drain/test.py b/tests/integration/test_cas_janitor_drain/test.py new file mode 100644 index 000000000000..d906ac7cf2a7 --- /dev/null +++ b/tests/integration/test_cas_janitor_drain/test.py @@ -0,0 +1,106 @@ +""" +A dropped table's ref stream drains in one deleting round of the namespace janitor on RustFS: batches of +the storage's size on a batch store, one-key jobs on a store without batch delete. +""" +import pytest + +from helpers.cluster import ClickHouseCluster + +cluster = ClickHouseCluster(__file__) + +INSERTS = 2500 +MAX_ROUNDS = 3 + + +@pytest.fixture(scope="module", autouse=True) +def start_cluster(): + cluster.add_instance( + "node", + main_configs=["configs/storage_conf.xml"], + with_rustfs=True, + stay_alive=True, + ) + try: + cluster.start() + yield cluster + finally: + cluster.shutdown() + + +def stream_keys(disk): + prefix = f"cas_janitor_{disk.removeprefix('cas_')}/cas/ns/stream/" + return [ + o.object_name + for o in cluster.rustfs_client.list_objects(cluster.rustfs_bucket, prefix, recursive=True) + if "/_log/" in o.object_name or "/_snap/" in o.object_name + ] + + +def seed_and_drop(node, disk): + node.query(f"DROP TABLE IF EXISTS t_{disk} SYNC") + node.query( + f"CREATE TABLE t_{disk} (k UInt64) ENGINE = MergeTree ORDER BY k " + f"SETTINGS storage_policy = '{disk}', parts_to_delay_insert = 100000, parts_to_throw_insert = 100000" + ) + node.query(f"SYSTEM STOP MERGES t_{disk}") + # One row per block gives one part per row without INSERTS client round trips. + node.query( + f"INSERT INTO t_{disk} SELECT number FROM numbers({INSERTS}) SETTINGS max_block_size = 1, " + "max_insert_block_size = 1, min_insert_block_size_rows = 1, min_insert_block_size_bytes = 0, " + "max_insert_threads = 1" + ) + assert len(stream_keys(disk)) >= INSERTS + node.query(f"DROP TABLE t_{disk} SYNC") + return len(stream_keys(disk)) + + +def janitor_rows(node, disk, since): + node.query("SYSTEM FLUSH LOGS") + columns = [ + "janitor_deleted", "janitor_pages", "budget_exhausted", "batch_keys", + "batches", "delete_jobs", + ] + select = ", ".join(f"phase_metrics['{c}']" for c in columns) + out = node.query( + f"SELECT {select} FROM system.cas_gc_log WHERE event_type = 'Phase' AND phase = 'namespace_cleanup' " + f"AND disk_name = '{disk}' AND event_time_microseconds >= toDateTime64('{since}', 6) " + "ORDER BY event_time_microseconds FORMAT TSV" + ) + return [dict(zip(columns, map(int, line.split("\t")))) for line in out.strip().splitlines()] + + +def drain(node, disk): + since = node.query("SELECT now64(6)").strip() + dead_keys = seed_and_drop(node, disk) + for _ in range(MAX_ROUNDS): + node.query(f"SYSTEM CAS GC RUN {disk}") + if not stream_keys(disk): + break + assert not stream_keys(disk), f"{disk}: dead ref stream keys left after {MAX_ROUNDS} rounds" + rows = janitor_rows(node, disk, since) + deleting = [i for i, r in enumerate(rows) if r["janitor_deleted"] > 0] + assert len(deleting) == 1, f"{disk}: expected one deleting round, got {rows}" + # Round 1 folds the drop and deletes nothing; the next round drains the whole stream. + assert deleting[0] == 1, f"{disk}: the deleting round is not the one right after the fold round: {rows}" + row = rows[deleting[0]] + assert row["janitor_deleted"] >= dead_keys, f"{disk}: the round did not drain all {dead_keys} dead keys: {rows}" + return row + + +def test_batch_store_drains_bulk_in_one_round(): + row = drain(cluster.instances["node"], "cas_batch") + assert row["janitor_pages"] >= 3 + assert row["budget_exhausted"] == 0 + + +def test_storage_chunk_setting_sizes_batches(): + row = drain(cluster.instances["node"], "cas_small_chunk") + assert row["batch_keys"] == 100 + assert row["batches"] >= 25 + + +def test_store_without_batch_delete_drains_with_one_key_jobs(): + row = drain(cluster.instances["node"], "cas_no_batch") + assert row["batch_keys"] == 1 + assert row["delete_jobs"] >= INSERTS + assert row["batches"] == row["delete_jobs"], f"a one-key job was skipped or held: {row}"