From 75da612cbad9711d6ce7b6c880b37dffec9b734b Mon Sep 17 00:00:00 2001 From: Arthur Passos Date: Mon, 28 Sep 2026 09:19:36 -0300 Subject: [PATCH 01/15] wild and secret idea --- docs/en/antalya/partition_export.md | 70 +- docs/en/antalya/ttl_export.md | 136 ++++ src/Common/CurrentMetrics.cpp | 1 + src/Common/ProfileEvents.cpp | 23 +- src/Core/Settings.cpp | 33 +- src/Core/SettingsChangesHistory.cpp | 13 + .../enableAllExperimentalSettings.cpp | 1 + .../InterpreterKillQueryQuery.cpp | 10 +- src/Parsers/ASTKillQueryQuery.cpp | 4 +- src/Parsers/ASTKillQueryQuery.h | 2 +- src/Parsers/ASTTTLElement.cpp | 7 + src/Parsers/ASTTTLElement.h | 3 + src/Parsers/CommonParsers.h | 1 + src/Parsers/ExpressionElementParsers.cpp | 28 + src/Parsers/ParserKillQueryQuery.cpp | 6 +- src/Processors/TTL/TTLUpdateInfoAlgorithm.cpp | 4 + src/Processors/TTL/TTLUpdateInfoAlgorithm.h | 1 + .../Transforms/TTLCalcTransform.cpp | 5 + src/Processors/Transforms/TTLTransform.cpp | 5 + ...mitInfoEntry.h => ExportCommitInfoEntry.h} | 10 +- ...h => ExportReplicatedMergeTreeTaskEntry.h} | 42 +- ...> ExportReplicatedMergeTreeTaskManifest.h} | 63 +- src/Storages/ExportTaskSource.h | 17 + src/Storages/IStorage.cpp | 2 +- src/Storages/IStorage.h | 28 +- .../DistributedMergePredicate.h | 10 + .../MergeTreeMergePredicate.cpp | 11 + .../MergePredicates/MergeTreeMergePredicate.h | 5 + .../ReplicatedMergeTreeMergePredicate.cpp | 16 + .../ReplicatedMergeTreeMergePredicate.h | 15 + src/Storages/MergeTree/ExportFence.cpp | 135 ++++ src/Storages/MergeTree/ExportFence.h | 76 ++ src/Storages/MergeTree/ExportPartTask.cpp | 20 +- src/Storages/MergeTree/ExportPartitionKey.h | 22 - src/Storages/MergeTree/ExportTTLDeleteGate.h | 39 + src/Storages/MergeTree/ExportTTLIndex.cpp | 263 +++++++ src/Storages/MergeTree/ExportTTLIndex.h | 126 ++++ src/Storages/MergeTree/ExportTTLScheduler.cpp | 707 ++++++++++++++++++ src/Storages/MergeTree/ExportTTLScheduler.h | 225 ++++++ ...PartitionExportInfo.h => ExportTaskInfo.h} | 7 +- ...PartitionUtils.cpp => ExportTaskUtils.cpp} | 462 ++++++++---- ...portPartitionUtils.h => ExportTaskUtils.h} | 55 +- src/Storages/MergeTree/IMergeTreeDataPart.cpp | 6 + src/Storages/MergeTree/MergeTreeData.cpp | 114 ++- src/Storages/MergeTree/MergeTreeData.h | 38 +- .../MergeTree/MergeTreeDataPartTTLInfo.cpp | 17 + .../MergeTree/MergeTreeDataPartTTLInfo.h | 7 +- .../MergeTree/MergeTreeDataWriter.cpp | 3 + .../MergeTree/MergeTreeExportTTLScheduler.cpp | 243 ++++++ .../MergeTree/MergeTreeExportTTLScheduler.h | 80 ++ ...tionExportTask.h => MergeTreeExportTask.h} | 60 +- ...r.cpp => MergeTreeExportTaskScheduler.cpp} | 344 +++++---- ...duler.h => MergeTreeExportTaskScheduler.h} | 49 +- src/Storages/MergeTree/MergeTreeSettings.cpp | 31 + src/Storages/MergeTree/MutateTask.cpp | 13 + .../MergeTree/ReplicatedExportTTLIndex.cpp | 204 +++++ .../MergeTree/ReplicatedExportTTLIndex.h | 83 ++ .../ReplicatedExportTTLScheduler.cpp | 270 +++++++ .../MergeTree/ReplicatedExportTTLScheduler.h | 49 ++ ....cpp => ReplicatedExportTaskScheduler.cpp} | 228 +++--- ...uler.h => ReplicatedExportTaskScheduler.h} | 20 +- ...sk.cpp => ReplicatedExportTaskUpdater.cpp} | 350 +++++---- ...ngTask.h => ReplicatedExportTaskUpdater.h} | 18 +- .../MergeTree/ReplicatedMergeTreeQueue.cpp | 10 + .../MergeTree/ReplicatedMergeTreeQueue.h | 4 + .../ReplicatedMergeTreeRestartingThread.cpp | 12 +- .../MergeTree/registerStorageMergeTree.cpp | 5 + .../MergeTree/tests/gtest_export_fence.cpp | 91 +++ ...ing.cpp => gtest_export_task_ordering.cpp} | 133 ++-- .../tests/gtest_export_ttl_index.cpp | 127 ++++ .../DataLakes/IDataLakeMetadata.h | 9 +- .../DataLakes/Iceberg/IcebergMetadata.cpp | 45 +- .../DataLakes/Iceberg/IcebergMetadata.h | 12 +- .../ObjectStorage/StorageObjectStorage.cpp | 44 +- .../ObjectStorage/StorageObjectStorage.h | 8 +- .../StorageObjectStorageCluster.cpp | 21 +- .../StorageObjectStorageCluster.h | 8 +- src/Storages/StorageInMemoryMetadata.cpp | 18 +- src/Storages/StorageInMemoryMetadata.h | 4 + src/Storages/StorageMergeTree.cpp | 155 ++-- src/Storages/StorageMergeTree.h | 48 +- src/Storages/StorageReplicatedMergeTree.cpp | 454 +++++------ src/Storages/StorageReplicatedMergeTree.h | 70 +- ...pp => StorageSystemDistributedExports.cpp} | 24 +- .../System/StorageSystemDistributedExports.h | 28 + .../System/StorageSystemPartitionExports.h | 30 - .../System/StorageSystemTTLExports.cpp | 84 +++ src/Storages/System/StorageSystemTTLExports.h | 27 + src/Storages/System/attachSystemTables.cpp | 8 +- src/Storages/TTLDescription.cpp | 16 +- src/Storages/TTLDescription.h | 9 +- src/Storages/TTLMode.h | 2 + .../helpers/export_partition_helpers.py | 108 ++- .../test.py | 33 + .../test.py | 6 +- .../common.py | 2 +- .../test_failures.py | 52 +- .../test_schema_match.py | 33 +- .../test_failures.py | 75 +- .../test_lifecycle.py | 183 +++-- .../test_validation.py | 10 +- tests/integration/test_export_ttl/__init__.py | 0 tests/integration/test_export_ttl/common.py | 364 +++++++++ .../allow_experimental_export_partition.xml | 3 + .../configs/config.d/metadata_log.xml | 7 + .../configs/named_collections.xml | 9 + .../configs/users.d/profile.xml | 26 + tests/integration/test_export_ttl/conftest.py | 65 ++ .../test_export_ttl/test_delete_gate.py | 144 ++++ .../test_export_ttl/test_failures.py | 236 ++++++ .../test_export_ttl/test_iceberg_commits.py | 141 ++++ .../test_export_ttl/test_lifecycle.py | 168 +++++ .../test_export_ttl/test_merge_fence.py | 237 ++++++ .../test_partition_keys_iceberg.py | 179 +++++ .../test_partition_keys_object_storage.py | 225 ++++++ .../test_export_ttl/test_replication.py | 237 ++++++ .../test_export_ttl/test_scheduling.py | 309 ++++++++ .../test_export_partition_iceberg.py | 4 +- .../test_export_partition_iceberg_catalog.py | 2 +- .../02995_settings_26_3_13_20001_antalya.tsv | 2 +- .../03745_system_background_schedule_pool.sql | 4 +- ...5027_export_partition_merge_tree.reference | 17 +- ..._partition_replicated_merge_tree.reference | 17 +- ...rtition_key_collision_merge_tree.reference | 5 +- ..._collision_replicated_merge_tree.reference | 5 +- .../05053_export_ttl_syntax.reference | 9 + .../0_stateless/05053_export_ttl_syntax.sql | 109 +++ .../05054_export_ttl_merge_tree.reference | 18 + .../05054_export_ttl_merge_tree.sh | 13 + ...export_ttl_replicated_merge_tree.reference | 18 + .../05055_export_ttl_replicated_merge_tree.sh | 12 + ...6_export_ttl_partition_key_check.reference | 17 + .../05056_export_ttl_partition_key_check.sql | 149 ++++ .../queries/0_stateless/export_partition.lib | 50 +- tests/queries/0_stateless/export_ttl.lib | 102 +++ 135 files changed, 8367 insertions(+), 1480 deletions(-) create mode 100644 docs/en/antalya/ttl_export.md rename src/Storages/{ExportPartitionCommitInfoEntry.h => ExportCommitInfoEntry.h} (87%) rename src/Storages/{ExportReplicatedMergeTreePartitionTaskEntry.h => ExportReplicatedMergeTreeTaskEntry.h} (59%) rename src/Storages/{ExportReplicatedMergeTreePartitionManifest.h => ExportReplicatedMergeTreeTaskManifest.h} (83%) create mode 100644 src/Storages/ExportTaskSource.h create mode 100644 src/Storages/MergeTree/ExportFence.cpp create mode 100644 src/Storages/MergeTree/ExportFence.h delete mode 100644 src/Storages/MergeTree/ExportPartitionKey.h create mode 100644 src/Storages/MergeTree/ExportTTLDeleteGate.h create mode 100644 src/Storages/MergeTree/ExportTTLIndex.cpp create mode 100644 src/Storages/MergeTree/ExportTTLIndex.h create mode 100644 src/Storages/MergeTree/ExportTTLScheduler.cpp create mode 100644 src/Storages/MergeTree/ExportTTLScheduler.h rename src/Storages/MergeTree/{PartitionExportInfo.h => ExportTaskInfo.h} (89%) rename src/Storages/MergeTree/{ExportPartitionUtils.cpp => ExportTaskUtils.cpp} (74%) rename src/Storages/MergeTree/{ExportPartitionUtils.h => ExportTaskUtils.h} (70%) create mode 100644 src/Storages/MergeTree/MergeTreeExportTTLScheduler.cpp create mode 100644 src/Storages/MergeTree/MergeTreeExportTTLScheduler.h rename src/Storages/MergeTree/{MergeTreePartitionExportTask.h => MergeTreeExportTask.h} (84%) rename src/Storages/MergeTree/{MergeTreePartitionExportScheduler.cpp => MergeTreeExportTaskScheduler.cpp} (63%) rename src/Storages/MergeTree/{MergeTreePartitionExportScheduler.h => MergeTreeExportTaskScheduler.h} (60%) create mode 100644 src/Storages/MergeTree/ReplicatedExportTTLIndex.cpp create mode 100644 src/Storages/MergeTree/ReplicatedExportTTLIndex.h create mode 100644 src/Storages/MergeTree/ReplicatedExportTTLScheduler.cpp create mode 100644 src/Storages/MergeTree/ReplicatedExportTTLScheduler.h rename src/Storages/MergeTree/{ExportPartitionTaskScheduler.cpp => ReplicatedExportTaskScheduler.cpp} (60%) rename src/Storages/MergeTree/{ExportPartitionTaskScheduler.h => ReplicatedExportTaskScheduler.h} (85%) rename src/Storages/MergeTree/{ExportPartitionManifestUpdatingTask.cpp => ReplicatedExportTaskUpdater.cpp} (67%) rename src/Storages/MergeTree/{ExportPartitionManifestUpdatingTask.h => ReplicatedExportTaskUpdater.h} (69%) create mode 100644 src/Storages/MergeTree/tests/gtest_export_fence.cpp rename src/Storages/MergeTree/tests/{gtest_export_partition_ordering.cpp => gtest_export_task_ordering.cpp} (50%) create mode 100644 src/Storages/MergeTree/tests/gtest_export_ttl_index.cpp rename src/Storages/System/{StorageSystemPartitionExports.cpp => StorageSystemDistributedExports.cpp} (89%) create mode 100644 src/Storages/System/StorageSystemDistributedExports.h delete mode 100644 src/Storages/System/StorageSystemPartitionExports.h create mode 100644 src/Storages/System/StorageSystemTTLExports.cpp create mode 100644 src/Storages/System/StorageSystemTTLExports.h create mode 100644 tests/integration/test_export_ttl/__init__.py create mode 100644 tests/integration/test_export_ttl/common.py create mode 100644 tests/integration/test_export_ttl/configs/allow_experimental_export_partition.xml create mode 100644 tests/integration/test_export_ttl/configs/config.d/metadata_log.xml create mode 100644 tests/integration/test_export_ttl/configs/named_collections.xml create mode 100644 tests/integration/test_export_ttl/configs/users.d/profile.xml create mode 100644 tests/integration/test_export_ttl/conftest.py create mode 100644 tests/integration/test_export_ttl/test_delete_gate.py create mode 100644 tests/integration/test_export_ttl/test_failures.py create mode 100644 tests/integration/test_export_ttl/test_iceberg_commits.py create mode 100644 tests/integration/test_export_ttl/test_lifecycle.py create mode 100644 tests/integration/test_export_ttl/test_merge_fence.py create mode 100644 tests/integration/test_export_ttl/test_partition_keys_iceberg.py create mode 100644 tests/integration/test_export_ttl/test_partition_keys_object_storage.py create mode 100644 tests/integration/test_export_ttl/test_replication.py create mode 100644 tests/integration/test_export_ttl/test_scheduling.py create mode 100644 tests/queries/0_stateless/05053_export_ttl_syntax.reference create mode 100644 tests/queries/0_stateless/05053_export_ttl_syntax.sql create mode 100644 tests/queries/0_stateless/05054_export_ttl_merge_tree.reference create mode 100755 tests/queries/0_stateless/05054_export_ttl_merge_tree.sh create mode 100644 tests/queries/0_stateless/05055_export_ttl_replicated_merge_tree.reference create mode 100755 tests/queries/0_stateless/05055_export_ttl_replicated_merge_tree.sh create mode 100644 tests/queries/0_stateless/05056_export_ttl_partition_key_check.reference create mode 100644 tests/queries/0_stateless/05056_export_ttl_partition_key_check.sql create mode 100644 tests/queries/0_stateless/export_ttl.lib diff --git a/docs/en/antalya/partition_export.md b/docs/en/antalya/partition_export.md index 83b57a5b5c95..b40162c75d64 100644 --- a/docs/en/antalya/partition_export.md +++ b/docs/en/antalya/partition_export.md @@ -9,17 +9,17 @@ The `ALTER TABLE EXPORT PARTITION` command exports entire partitions from `Merge The set of parts that are exported is based on the list of parts the replica that received the export command sees. On `Replicated*MergeTree`, the other replicas will assist in the export process if they have those parts locally. Otherwise they will ignore it. -The partition export tasks of both engines can be observed through `system.partition_exports`. +Every `EXPORT PARTITION` creates a new export task, identified by its transaction id, that exports the parts the partition has at that moment. Exporting a partition that was exported before is allowed and exports its parts again; whether that duplicates rows depends on the destination and on `export_merge_tree_part_file_already_exists_policy` (with the default `skip` and file name pattern, files already written to a plain object storage destination are reused). Preventing duplicates is up to the user. The setting `export_merge_tree_partition_force_export` is obsolete and has no effect. -`system.replicated_partition_exports` is kept as an alias of `system.partition_exports` for backwards compatibility. It returns exactly the same rows, including exports of plain `MergeTree` tables. +Export tasks of both engines, including those of a [`TTL ... EXPORT TO TABLE`](/docs/en/antalya/ttl_export.md) expression, can be observed through `system.distributed_exports`, one row per task. Manual exports are independent of the `EXPORT` TTL: they neither read nor update what the TTL has exported. -The same partition can not be exported to the same destination more than once. This behavior can be overriden with `export_merge_tree_partition_force_export`. +The export task can be killed by issuing the kill command: `KILL EXPORT `. -The export task can be killed by issuing the kill command: `KILL EXPORT PARTITION `. +Tasks are stored under `/exports/` for a `Replicated*MergeTree` table, and in `exports/.json` in the data directory of a plain `MergeTree` table. Tasks stored by earlier versions of this experimental feature, keyed by partition and destination, are neither read nor migrated. The task is persistent - it should be resumed after crashes, failures and etc. -A part with no surviving rows writes no file. This happens when every row of the part was removed by a lightweight delete: the part is still exported, and counts as done, but it contributes nothing to the destination. If that is true of every part of the partition, the export produces no files at all and there is nothing to commit, so the task reaches `COMPLETED` without touching the destination. Such a part therefore has no entry in the `destination_file_paths` column of `system.partition_exports`. +A part with no surviving rows writes no file. This happens when every row of the part was removed by a lightweight delete: the part is still exported, and counts as done, but it contributes nothing to the destination. If that is true of every part of the partition, the export produces no files at all and there is nothing to commit, so the task reaches `COMPLETED` without touching the destination. Such a part therefore has no entry in the `destination_file_paths` column of `system.distributed_exports`. ### On Apache Iceberg storage exports: @@ -27,7 +27,7 @@ Each MergeTree part that has surviving rows will become a separate file (or more The manifest file produced by the commit contains a summary field `clickhouse.export-partition-transaction-id` that stores the transaction id. This field is used to implement idempotency and avoid data duplication. Some Apache Iceberg storage managers employ old manifests cleanup, ClickHouse does not. -**IMPORTANT**: In case the storage is managed by a 3rd party application that cleans up old manifest files, it is important that the TTL of such files are greater than the timeout of export partition tasks. If it is not configured in such a way, it is possible to accidentally duplicate data in the extremely rare case a ClickHouse node is the only node working on a given export task, commits the data to Iceberg, crashes before marking the task as done and only boots up after the manifest cleanup has deleted the commit manifest. In such scenario, ClickHouse would attempt to commit those files again producing duplicates. The task timeout on ClickHouse side is controlled by the setting `export_merge_tree_partition_task_timeout_seconds`. +**IMPORTANT**: In case the storage is managed by a 3rd party application that cleans up old manifest files, it is important that the TTL of such files are greater than the timeout of export partition tasks. If it is not configured in such a way, it is possible to accidentally duplicate data in the extremely rare case a ClickHouse node is the only node working on a given export task, commits the data to Iceberg, crashes before marking the task as done and only boots up after the manifest cleanup has deleted the commit manifest. In such scenario, ClickHouse would attempt to commit those files again producing duplicates. The task timeout on ClickHouse side is controlled by the setting `export_merge_tree_task_timeout_seconds`. The Iceberg manifest files contain statistics about the data. Exporting a merge tree partition is a non ephemeral long running task, in which nodes can be turned off and turned on. This means the stats of individual files need to be persisted somewhere in order to produce the final manifest. This is implemented through sidecars. Each data file exported will contain a "sibling" sidecar file named `_clickhouse_export_part_sidecar.avro`. ClickHouse does not clean up these files, and they can be safely deleted once the data is comitted. @@ -40,7 +40,7 @@ The source partition must not be split in the destination. This is validated at ### On plain object storage exports: -Each MergeTree part will become a separate file with the following name convention: `//_.`. To ensure atomicity, a commit file containing the relative paths of all exported parts is also shipped. A data file should only be considered part of the dataset if a commit file references it. The commit file will be named using the following convention: `/commit__`. +Each MergeTree part will become a separate file with the following name convention: `//_.`. To ensure atomicity, a commit file containing the relative paths of all exported parts is also shipped. A data file should only be considered part of the dataset if a commit file references it. The commit file will be named using the following convention: `/commit_`. ## Plain (non-replicated) MergeTree {#plain-non-replicated-mergetree} @@ -85,23 +85,26 @@ TO TABLE [destination_database.]destination_table ### Query Settings -#### `export_merge_tree_partition_force_export` (Optional) - -- **Type**: `Bool` -- **Default**: `false` -- **Description**: Ignore existing partition export and overwrite the ZooKeeper entry. Allows re-exporting a partition that was already exported to the same destination. **IMPORTANT:** this is dangerous because it can lead to duplicated data, use it with caution. - -#### `export_merge_tree_partition_retry_initial_backoff_seconds` (Optional) +#### `export_merge_tree_retry_initial_backoff_seconds` (Optional) - **Type**: `UInt64` - **Default**: `5` -- **Description**: Initial delay (in seconds) before retrying a failed part export. The delay grows exponentially with the per-replica retry count (`delay = min(initial << (attempts - 1), max)`). The back-off is per-replica in-memory state: it only spaces this replica's retries out in time and never prevents another replica from attempting the same part. Retryable failures (transient memory/network/object-storage/Keeper errors) are retried until the task succeeds or `export_merge_tree_partition_task_timeout_seconds` elapses, while non-retryable failures (e.g. schema/type incompatibilities) fail the task immediately. +- **Description**: Initial delay (in seconds) before retrying a failed part export. The delay grows exponentially with the per-replica retry count (`delay = min(initial << (attempts - 1), max)`). The back-off is per-replica in-memory state: it only spaces this replica's retries out in time and never prevents another replica from attempting the same part. Retryable failures (transient memory/network/object-storage/Keeper errors) are retried until the task succeeds or `export_merge_tree_task_timeout_seconds` elapses, while non-retryable failures (e.g. schema/type incompatibilities) fail the task immediately. -#### `export_merge_tree_partition_retry_max_backoff_seconds` (Optional) +#### `export_merge_tree_retry_max_backoff_seconds` (Optional) - **Type**: `UInt64` - **Default**: `300` -- **Description**: Maximum delay (in seconds) between retries of a failed part export. Caps the exponential growth controlled by `export_merge_tree_partition_retry_initial_backoff_seconds`. +- **Description**: Maximum delay (in seconds) between retries of a failed part export. Caps the exponential growth controlled by `export_merge_tree_retry_initial_backoff_seconds`. + +#### `export_merge_tree_partition_all_on_error` (Optional) {#export-merge-tree-partition-all-on-error} + +- **Type**: `ExportPartitionAllOnError` +- **Default**: `throw_first` +- **Description**: How `EXPORT PARTITION ALL` handles a partition that cannot be exported. Possible values: + - `throw_first` - stop at the first failing partition and throw + - `collect` - try every partition, then throw one error listing the failing ones + - `skip_conflicts` - behaves like `throw_first`: re-exporting a partition is no longer refused, so there are no conflicts to skip #### `export_merge_tree_part_file_already_exists_policy` (Optional) @@ -130,12 +133,12 @@ TO TABLE [destination_database.]destination_table - **Default**: `{part_name}_{checksum}` - **Description**: Pattern for the filename of the exported merge tree part. The `part_name` and `checksum` are calculated and replaced on the fly. Additional macros are supported. -### `export_merge_tree_partition_task_timeout_seconds` (Optional) +### `export_merge_tree_task_timeout_seconds` (Optional) - **Type**: `UInt64` -- **Default**: `3600` +- **Default**: `86400` - **Description**: The timeout is measured from the manifest's create_time. Set to 0 to disable the timeout. -When the timeout is exceeded the task transitions to KILLED (same terminal state as `KILL QUERY ... EXPORT PARTITION`), and a `last_exception_per_replica` entry on the replica that fires the timeout is populated with a timeout reason. +When the timeout is exceeded the task transitions to KILLED (same terminal state as `KILL EXPORT`), and a `last_exception_per_replica` entry on the replica that fires the timeout is populated with a timeout reason. Notes: - Enforcement is best-effort: actual kill latency is bounded by one manifest-updater poll cycle (~30s) plus ZooKeeper watch propagation. @@ -186,13 +189,14 @@ PARTITION BY year; INSERT INTO rmt_table VALUES (1, 2020), (2, 2020), (3, 2020), (4, 2021); ALTER TABLE rmt_table EXPORT PARTITION ID '2020' TO TABLE s3_table; +``` ## Killing Exports -You can cancel in-progress partition exports using the `KILL EXPORT PARTITION` command: +You can cancel in-progress exports using the `KILL EXPORT` command: ```sql -KILL EXPORT PARTITION +KILL EXPORT WHERE partition_id = '2020' AND source_table = 'rmt_table' AND destination_table = 's3_table' @@ -202,13 +206,13 @@ WHERE partition_id = '2020' ### Active and Completed Exports -Monitor partition exports using the `system.partition_exports` table: +Monitor exports using the `system.distributed_exports` table: ```sql -arthur :) select * from system.partition_exports Format Vertical; +arthur :) select * from system.distributed_exports Format Vertical; SELECT * -FROM system.partition_exports +FROM system.distributed_exports FORMAT Vertical Query id: 9efc271a-a501-44d1-834f-bc4d20156164 @@ -234,7 +238,10 @@ destination_file_paths: {'2022_0_0_0':['data/year=2022/2022_0_0_0_.par committed_metadata_file: committed_manifest_list: committed_manifest_file: -committed_marker_file: data/commit_2022_9b2c1e5a-3f47-4c8e-8a1d-6f0b2d4e7c31 +committed_marker_file: data/commit_9b2c1e5a-3f47-4c8e-8a1d-6f0b2d4e7c31 +local_backoff_per_part: [] +source: query +retry_of: [] Row 2: ────── @@ -258,6 +265,9 @@ committed_metadata_file: data/metadata/v3.metadata.json committed_manifest_list: data/metadata/snap-4029103741930112856-1-.avro committed_manifest_file: data/metadata/-m0.avro committed_marker_file: +local_backoff_per_part: [] +source: query +retry_of: [] 2 rows in set. Elapsed: 0.019 sec. @@ -286,15 +296,21 @@ Status values include: - `committed_manifest_file` — for Iceberg destinations: path of the manifest file referenced by `committed_manifest_list`. Empty under the same conditions as `committed_metadata_file`. - `committed_marker_file` — for plain object storage destinations: path of the per-transaction commit marker file written by the destination. Empty for Iceberg destinations and for tasks that have not committed yet. +### Source columns {#source-columns} + +- `source` — `query` for a task of `EXPORT PARTITION`, `ttl` for a task of the table's `TTL ... EXPORT TO TABLE` expression. +- `retry_of` — for a task of the `EXPORT` TTL: transaction ids of earlier tasks that failed to export some of its parts. Its commit checks whether any of them landed at the destination after all. + To pick the latest exception across replicas: ```sql SELECT arraySort(x -> -x.time, last_exception_per_replica)[1] AS latest_exception -FROM system.partition_exports +FROM system.distributed_exports WHERE source_table = 'rmt_table' AND destination_table = 's3_table'; ``` ## Related Features - [ALTER TABLE EXPORT PART](/docs/en/antalya/part_export.md) - Export individual parts (non-replicated) +- [TTL ... EXPORT TO TABLE](/docs/en/antalya/ttl_export.md) - Export parts in the background once their TTL is due diff --git a/docs/en/antalya/ttl_export.md b/docs/en/antalya/ttl_export.md new file mode 100644 index 000000000000..18a7593762e0 --- /dev/null +++ b/docs/en/antalya/ttl_export.md @@ -0,0 +1,136 @@ +--- +description: 'Export parts of MergeTree tables to Apache Iceberg or object storage in the background once their TTL is due' +sidebar_label: 'TTL EXPORT TO TABLE' +sidebar_position: 30 +slug: /antalya/ttl_export +title: 'TTL ... EXPORT TO TABLE' +doc_type: 'reference' +--- + +# TTL ... EXPORT TO TABLE {#ttl-export-to-table} + +## Overview {#overview} + +A `TTL EXPORT TO TABLE [database.]table` expression exports the rows of a `MergeTree`-family table to an Apache Iceberg or plain object storage table in the background, once their TTL is due. It is meant for tiering: recent data stays in `MergeTree`, older data is kept in the destination. + +The TTL never exports a row twice, also across failures, restarts and retries: what was exported is recorded per partition, and parts are exported in groups, each committed to the destination in one transaction. The export uses the same machinery as [`EXPORT PARTITION`](/docs/en/antalya/partition_export.md): each group is an export task shown in `system.distributed_exports` with `source = 'ttl'`. + +Both plain `MergeTree` and `Replicated*MergeTree` tables are supported. + +```sql +CREATE TABLE events +( + event_time DateTime, + user_id UInt64, + payload String +) +ENGINE = ReplicatedMergeTree('/clickhouse/tables/{database}/events', '{replica}') +PARTITION BY toYYYYMM(event_time) +ORDER BY (user_id, event_time) +TTL event_time + INTERVAL 30 DAY EXPORT TO TABLE events_archive, + event_time + INTERVAL 90 DAY DELETE +SETTINGS ttl_export_batch_window_seconds = 300; +``` + +Here rows are exported to `events_archive` 30 days after `event_time`, and deleted from `events` after 90 days, but not before they are exported. + +## Requirements {#requirements} + +- The server setting `allow_experimental_export_merge_tree_partition` must be enabled on every replica, and the query setting `allow_experimental_export_ttl` must be enabled for the `CREATE` or `ALTER` that adds the expression. +- The destination must exist when the expression is added. It must be an Apache Iceberg or object storage table that `EXPORT PARTITION` can export to, and its schema must be castable from the source schema, see [`EXPORT PARTITION` requirements](/docs/en/antalya/partition_export.md#requirements). An unqualified table name refers to the database of the source table. +- A table can have at most one `EXPORT` TTL expression, without `WHERE` or `GROUP BY`. The expression must be deterministic and return a `Date` or `DateTime`. +- The rows of a group must land in a single partition of the destination, see [Partition key of the destination](#destination-partition-key). + +## Partition key of the destination {#destination-partition-key} + +Each partition expression of the destination (or each Iceberg partition transform) must either be a function of the source partition key, e.g. the same expression, or be a monotonic function of a single non-`Nullable` column of the source partition key. In the second case, whether the rows of a group land in a single destination partition depends on the rows, so it is checked when the group is exported, from the minimum and maximum values of the column in its parts. + +A destination that can never be compatible is refused by the `CREATE` or `ALTER` that adds the expression, e.g. a destination partitioned by a column that is not in the source partition key, by a function that is not monotonic like `toDayOfWeek(event_time)` or the Iceberg `bucket` transform, or a partitioned destination of an unpartitioned source. + +A destination that is compatible only for some groups is accepted. With a source `PARTITION BY toYYYYMM(event_time)`: + +- `PARTITION BY toYear(event_time)` is compatible for every group, since a month is within a year; +- `PARTITION BY toDate(event_time)` is compatible only for a group whose rows are all on the same day. Any other group is not exported: no export task is created, the error is shown in the `last_error` column of `system.ttl_exports`, and the group is tried again on every check. + +## How parts are exported {#how-parts-are-exported} + +A part becomes eligible once the maximum TTL value of its rows is due, the same rule move TTL uses, so a part is exported as a whole. The TTL does not have to be aligned with the partition key. A part written before the expression was added becomes eligible after `ALTER TABLE ... MATERIALIZE TTL`, which `ALTER TABLE ... MODIFY TTL` runs by default. + +The eligible parts of a partition are collected into a group and exported together, when any of the following holds: + +- no new eligible part appeared in the partition for `ttl_export_batch_window_seconds`; +- the first of them became eligible more than `ttl_export_batch_max_delay_seconds` ago; +- their size reaches `ttl_export_batch_min_bytes`. + +A group has at most `ttl_export_max_parts_per_group` parts and `ttl_export_max_bytes_per_group` bytes, and contains parts of one partition only. Each partition has at most one group being exported at a time, and the table at most `ttl_export_max_concurrent_groups`. Merges of eligible parts continue while a group is being collected; a part that is being merged waits for the merge. + +To keep exported rows apart from the others, parts that are exported, parts that are being exported and parts that are not exported are never merged together. Parts that are being exported are not merged at all until their group commits. On a `Replicated*MergeTree` table, every replica enforces this, which is why a replica advertises that it supports it, and groups are only started while every replica does. + +One replica of a `Replicated*MergeTree` table schedules the groups; the others take part in exporting them like in `EXPORT PARTITION`. The state of the scheduling, i.e. the batch timers and the last errors, is kept in Keeper, so when another replica takes over, it continues where the previous one left off. + +## Failures and retries {#failures-and-retries} + +A group that fails, is killed or times out is retried on the next check as a new task, with a new transaction id. The retry contains every part of the failed group, plus any part that became eligible meanwhile. Its `retry_of` column lists the failed tasks. Throttling comes from the export tasks themselves: per-part retry back-off, `export_merge_tree_task_timeout_seconds` and the commit attempts. + +Before a failed task is retried, the destination is checked for its commit, in case it committed but was not marked as completed. A failed task may also land at the destination after that check, e.g. a request that completes late. The commit of the retry therefore checks each task in `retry_of` again, and leaves out the files of the parts a landed task exported. + +`KILL EXPORT` of a task of the TTL makes it retry. To stop exporting, use `SYSTEM STOP MOVES`, which pauses the TTL export of the table, or remove the expression. On a `Replicated*MergeTree` table, `SYSTEM STOP MOVES` pauses it only on the replica that schedules the groups, shown in the `scheduler_replica` column of `system.ttl_exports`, so run it on every replica, e.g. with `ON CLUSTER`. + +A part that is being merged waits for the merge before it is exported. On a `Replicated*MergeTree` table, `SYSTEM STOP MERGES` does not keep merges from being assigned, so if merges are stopped on the replica that schedules the groups, the parts of a merge assigned meanwhile wait until merges are started again. + +## Rows are not deleted before they are exported {#delete-gate} + +While a table has an `EXPORT` TTL expression, TTL that deletes or rewrites rows (`DELETE`, `WHERE`, `GROUP BY`, column TTL) is not applied to a part that is not exported yet: merges leave such a part alone, and a mutation that would apply TTL to it only recalculates the TTL. Once the part is exported, the TTL is applied by a merge as usual. The number of parts held back is shown in `system.ttl_exports` and by the `ExportTTLPartsHeldByDeleteGate` metric. + +## Changing or removing the expression {#changing-or-removing} + +When the `EXPORT` TTL expression is removed, e.g. by `ALTER TABLE ... REMOVE TTL` or a `MODIFY TTL` without it, or its destination changes, groups being exported to the previous destination are killed, and what was exported to it is forgotten once they finished. Merges are then unrestricted again. Adding the expression back later exports every eligible part again, including parts that were exported before. + +While the destination does not exist, e.g. it was dropped or is not loaded yet, nothing is exported, what was exported to it is kept, and the `last_error` column of `system.ttl_exports` says so. A destination that is dropped and created again under the same name is a new destination for a plain `MergeTree` table, so its eligible parts are exported to it again: an Iceberg table created again over the data of the dropped one then has the rows exported before twice. A `Replicated*MergeTree` table identifies the destination by its name only, because its replicas may have different UUIDs for it, so what was exported to the dropped table is not exported again. + +`ALTER TABLE ... FORGET PARTITION` forgets what was exported from the partition, and is refused while a group of the partition is being exported. + +## Settings {#settings} + +### Query settings {#query-settings} + +- `allow_experimental_export_ttl` — allows adding a `TTL ... EXPORT TO TABLE` expression. + +The settings of the export tasks, e.g. the output format settings, `export_merge_tree_part_file_already_exists_policy` and `export_merge_tree_task_timeout_seconds`, are taken from the settings profile named by `ttl_export_settings_profile`, or the default profile. + +### MergeTree settings {#merge-tree-settings} + +| Setting | Default | Description | +|---|---|---| +| `ttl_export_check_period_seconds` | `10` | How often eligible parts are looked for. | +| `ttl_export_batch_window_seconds` | `60` | A group is exported once no new eligible part appeared for this long. | +| `ttl_export_batch_max_delay_seconds` | `600` | A group is exported at the latest this long after its first part became eligible. | +| `ttl_export_batch_min_bytes` | `256 MiB` | A group is exported as soon as its parts reach this size. `0` disables it. | +| `ttl_export_max_parts_per_group` | `100` | Maximum number of parts of a group. Parts of a failed group are always retried together. | +| `ttl_export_max_bytes_per_group` | `10 GiB` | Maximum size of a group. A bigger part is exported on its own. `0` means unlimited. | +| `ttl_export_max_concurrent_groups` | `4` | Maximum number of groups of the table being exported at the same time. | +| `ttl_export_settings_profile` | `''` | Settings profile of the export tasks. | + +## Monitoring {#monitoring} + +`system.ttl_exports` has one row per partition of a table with an `EXPORT` TTL expression, as of the last check. For a `Replicated*MergeTree` table, every replica has the same rows: the batch timers, the task being exported and the last error come from Keeper, and are those of the replica that schedules the groups, shown in `scheduler_replica`. The part counts are those of the parts of the replica, so they differ while a replica is fetching parts. + +```sql +SELECT partition_id, exported_parts, claimed_parts, eligible_parts, parts_held_by_delete_gate, + first_eligible_time, next_group_time, current_transaction_id, last_error, scheduler_replica +FROM system.ttl_exports +WHERE table = 'events'; +``` + +- `exported_parts`, `claimed_parts` and `eligible_parts` count the active parts that are exported, that are being exported (or waiting to be retried), and that are due for export. +- `next_group_time` is when the eligible parts are exported at the latest. +- `current_transaction_id` is the task exporting the partition now, see `system.distributed_exports`. +- `scheduler_replica` is the replica that schedules the groups of a `Replicated*MergeTree` table, and is empty for a plain `MergeTree` table. + +What was exported is read from Keeper again only when it changes, i.e. when a group is claimed, committed or released: the checks and the merge selection otherwise use a cached copy. The `ExportTTLIndexSnapshotRefreshes` profile event counts how often it is read again, so on an idle table it does not grow. + +## Limitations {#limitations} + +- Export tasks are not removed, so they accumulate in Keeper (or in the data directory of a plain `MergeTree` table), and in `system.distributed_exports`. +- On a plain object storage destination, files that no commit file references may remain, e.g. when a part of a failed group is mutated before its retry, its new name gives a new file, and the file of the failed attempt is left. Readers must follow the commit files, see [`EXPORT PARTITION`](/docs/en/antalya/partition_export.md). +- Parts that are attached again, e.g. by `ALTER TABLE ... ATTACH PARTITION`, get new block numbers and are exported again. diff --git a/src/Common/CurrentMetrics.cpp b/src/Common/CurrentMetrics.cpp index a36be7839914..6526ccce08a6 100644 --- a/src/Common/CurrentMetrics.cpp +++ b/src/Common/CurrentMetrics.cpp @@ -13,6 +13,7 @@ M(MergeParts, "Number of source parts participating in current background merges") \ M(Move, "Number of currently executing moves") \ M(Export, "Number of currently executing exports") \ + M(ExportTTLPartsHeldByDeleteGate, "Number of parts whose TTL is due but that are kept from merges until the EXPORT TTL exports them") \ M(PartMutation, "Number of mutations (ALTER DELETE/UPDATE)") \ M(ReplicatedFetch, "Number of data parts being fetched from replica") \ M(ReplicatedSend, "Number of data parts being sent to replicas") \ diff --git a/src/Common/ProfileEvents.cpp b/src/Common/ProfileEvents.cpp index 184d4941aa3c..7aa8e0ca212c 100644 --- a/src/Common/ProfileEvents.cpp +++ b/src/Common/ProfileEvents.cpp @@ -377,17 +377,18 @@ M(ZooKeeperBytesSent, "Number of bytes send over network while communicating with ZooKeeper.", ValueType::Bytes) \ M(ZooKeeperBytesReceived, "Number of bytes received over network while communicating with ZooKeeper.", ValueType::Bytes) \ \ - M(ExportPartitionZooKeeperRequests, "Total number of ZooKeeper requests made by the export partition feature.", ValueType::Number) \ - M(ExportPartitionZooKeeperGet, "Number of 'get' requests to ZooKeeper made by the export partition feature.", ValueType::Number) \ - M(ExportPartitionZooKeeperGetChildren, "Number of 'getChildren' requests to ZooKeeper made by the export partition feature.", ValueType::Number) \ - M(ExportPartitionZooKeeperGetChildrenWatch, "Number of 'getChildrenWatch' requests to ZooKeeper made by the export partition feature.", ValueType::Number) \ - M(ExportPartitionZooKeeperGetWatch, "Number of 'getWatch' requests to ZooKeeper made by the export partition feature.", ValueType::Number) \ - M(ExportPartitionZooKeeperCreate, "Number of 'create' requests to ZooKeeper made by the export partition feature.", ValueType::Number) \ - M(ExportPartitionZooKeeperSet, "Number of 'set' requests to ZooKeeper made by the export partition feature.", ValueType::Number) \ - M(ExportPartitionZooKeeperRemove, "Number of 'remove' requests to ZooKeeper made by the export partition feature.", ValueType::Number) \ - M(ExportPartitionZooKeeperRemoveRecursive, "Number of 'removeRecursive' requests to ZooKeeper made by the export partition feature.", ValueType::Number) \ - M(ExportPartitionZooKeeperMulti, "Number of 'multi' requests to ZooKeeper made by the export partition feature.", ValueType::Number) \ - M(ExportPartitionZooKeeperExists, "Number of 'exists' requests to ZooKeeper made by the export partition feature.", ValueType::Number) \ + M(ExportTTLIndexSnapshotRefreshes, "Number of times the export index of the EXPORT TTL of a table was read again, because it changed.", ValueType::Number) \ + M(ExportTaskZooKeeperRequests, "Total number of ZooKeeper requests made by the export partition feature.", ValueType::Number) \ + M(ExportTaskZooKeeperGet, "Number of 'get' requests to ZooKeeper made by the export partition feature.", ValueType::Number) \ + M(ExportTaskZooKeeperGetChildren, "Number of 'getChildren' requests to ZooKeeper made by the export partition feature.", ValueType::Number) \ + M(ExportTaskZooKeeperGetChildrenWatch, "Number of 'getChildrenWatch' requests to ZooKeeper made by the export partition feature.", ValueType::Number) \ + M(ExportTaskZooKeeperGetWatch, "Number of 'getWatch' requests to ZooKeeper made by the export partition feature.", ValueType::Number) \ + M(ExportTaskZooKeeperCreate, "Number of 'create' requests to ZooKeeper made by the export partition feature.", ValueType::Number) \ + M(ExportTaskZooKeeperSet, "Number of 'set' requests to ZooKeeper made by the export partition feature.", ValueType::Number) \ + M(ExportTaskZooKeeperRemove, "Number of 'remove' requests to ZooKeeper made by the export partition feature.", ValueType::Number) \ + M(ExportTaskZooKeeperRemoveRecursive, "Number of 'removeRecursive' requests to ZooKeeper made by the export partition feature.", ValueType::Number) \ + M(ExportTaskZooKeeperMulti, "Number of 'multi' requests to ZooKeeper made by the export partition feature.", ValueType::Number) \ + M(ExportTaskZooKeeperExists, "Number of 'exists' requests to ZooKeeper made by the export partition feature.", ValueType::Number) \ M(ExportPartsRejectedByMemoryLimit, "Number of background export part tasks rejected due to background memory limit.", ValueType::Number) \ \ M(DistributedConnectionTries, "Total count of distributed connection attempts.", ValueType::Number) \ diff --git a/src/Core/Settings.cpp b/src/Core/Settings.cpp index d0725772d8aa..dce0ccd53c28 100644 --- a/src/Core/Settings.cpp +++ b/src/Core/Settings.cpp @@ -8084,22 +8084,23 @@ Default value is empty. DECLARE(Bool, export_merge_tree_part_overwrite_file_if_exists, false, R"( Overwrite file if it already exists when exporting a merge tree part )", 0) \ - DECLARE(Bool, export_merge_tree_partition_force_export, false, R"( -Ignore existing partition export and overwrite the zookeeper entry -)", 0) \ - DECLARE(UInt64, export_merge_tree_partition_retry_initial_backoff_seconds, 5, R"( -Initial delay (in seconds) before retrying a failed part export in an export partition task. -The delay grows exponentially with the per-replica retry count (capped doubling): `delay = min(initial << (attempts - 1), max)`, where `max` is `export_merge_tree_partition_retry_max_backoff_seconds`. -The back-off is per-replica in-memory state: it only spaces this replica's retries out in time and never prevents another replica from attempting the same part. Retryable failures are retried until the task succeeds or `export_merge_tree_partition_task_timeout_seconds` elapses. -To survive a long transient outage (e.g. object storage downtime), raise `export_merge_tree_partition_task_timeout_seconds`. + DECLARE(Bool, allow_experimental_export_ttl, false, R"( +Allow creating tables with, or altering tables to have, a `TTL ... EXPORT TO TABLE` expression, which exports parts of a `MergeTree` table to an Iceberg or object storage table in the background once they expire. +Also requires the server setting `allow_experimental_export_merge_tree_partition`. +)", EXPERIMENTAL) \ + DECLARE(UInt64, export_merge_tree_retry_initial_backoff_seconds, 5, R"( +Initial delay (in seconds) before retrying a failed part export in an export task (`EXPORT PARTITION` or `TTL ... EXPORT`). +The delay grows exponentially with the per-replica retry count (capped doubling): `delay = min(initial << (attempts - 1), max)`, where `max` is `export_merge_tree_retry_max_backoff_seconds`. +The back-off is per-replica in-memory state: it only spaces this replica's retries out in time and never prevents another replica from attempting the same part. Retryable failures are retried until the task succeeds or `export_merge_tree_task_timeout_seconds` elapses. +To survive a long transient outage (e.g. object storage downtime), raise `export_merge_tree_task_timeout_seconds`. )", 0) \ - DECLARE(UInt64, export_merge_tree_partition_retry_max_backoff_seconds, 300, R"( -Maximum delay (in seconds) between retries of a failed part export in an export partition task. Caps the exponential growth controlled by `export_merge_tree_partition_retry_initial_backoff_seconds`. + DECLARE(UInt64, export_merge_tree_retry_max_backoff_seconds, 300, R"( +Maximum delay (in seconds) between retries of a failed part export in an export task. Caps the exponential growth controlled by `export_merge_tree_retry_initial_backoff_seconds`. )", 0) \ - DECLARE(UInt64, export_merge_tree_partition_task_timeout_seconds, 86400, R"( -Maximum wall-clock duration (in seconds) an export partition task is allowed to remain in the PENDING state before it is auto-killed by the background cleanup loop. + DECLARE(UInt64, export_merge_tree_task_timeout_seconds, 86400, R"( +Maximum wall-clock duration (in seconds) an export task is allowed to remain in the PENDING state before it is auto-killed by the background cleanup loop. The timeout is measured from the manifest's create_time. Set to 0 to disable the timeout. -When the timeout is exceeded the task transitions to KILLED (same terminal state as `KILL QUERY ... EXPORT PARTITION`), and `last_exception` is populated with a timeout reason. +When the timeout is exceeded the task transitions to KILLED (same terminal state as `KILL EXPORT`), and `last_exception` is populated with a timeout reason. IMPORTANT: In case the storage is managed by a 3rd party application that cleans up old manifest files, it is important that the TTL of such files are greater than the timeout of export partition tasks. If it is not configured in such a way, it is possible to accidentally duplicate data in the extremely rare case a ClickHouse node is the only node working on a given export task, commits the data to Iceberg, crashes before marking the task as done and only boots up after the manifest cleanup has deleted the commit manifest. @@ -8133,7 +8134,7 @@ Failure handling for `ALTER TABLE ... EXPORT PARTITION ALL ...`. Possible values: - `throw_first` (default) - stop at the first failed partition; partitions already scheduled remain scheduled. - `collect` - try every partition and throw a single aggregated exception at the end if any failed; partitions that succeeded remain scheduled. -- `skip_conflicts` - silently skip partitions that are already exported / being exported (errors with code EXPORT_PARTITION_ALREADY_EXPORTED); fail-fast on every other error. +- `skip_conflicts` - silently skip partitions whose export fails with code `EXPORT_PARTITION_ALREADY_EXPORTED`; fail-fast on every other error. `EXPORT PARTITION` no longer refuses re-exports, so nothing is skipped and this behaves like `throw_first`. Has no effect on `EXPORT PARTITION ` (single-partition export). )", 0) \ DECLARE(String, export_merge_tree_part_filename_pattern, "{part_name}_{checksum}", R"( @@ -8598,6 +8599,10 @@ Name of the named collection used by `aiEmbed` when the call does not pass `cred /** Obsolete settings which are kept around for compatibility reasons. They have no effect anymore. */ \ MAKE_OBSOLETE(M, UInt64, export_merge_tree_partition_manifest_ttl, 86400) \ MAKE_OBSOLETE(M, UInt64, export_merge_tree_partition_max_retries, 3) \ + MAKE_OBSOLETE(M, Bool, export_merge_tree_partition_force_export, false) \ + MAKE_OBSOLETE(M, UInt64, export_merge_tree_partition_retry_initial_backoff_seconds, 5) \ + MAKE_OBSOLETE(M, UInt64, export_merge_tree_partition_retry_max_backoff_seconds, 300) \ + MAKE_OBSOLETE(M, UInt64, export_merge_tree_partition_task_timeout_seconds, 86400) \ MAKE_OBSOLETE(M, Bool, allow_experimental_query_deduplication, false) \ MAKE_OBSOLETE(M, Bool, query_condition_cache_store_conditions_as_plaintext, false) \ MAKE_OBSOLETE(M, Bool, update_insert_deduplication_token_in_dependent_materialized_views, 0) \ diff --git a/src/Core/SettingsChangesHistory.cpp b/src/Core/SettingsChangesHistory.cpp index f093c4054cf6..300693a0b234 100644 --- a/src/Core/SettingsChangesHistory.cpp +++ b/src/Core/SettingsChangesHistory.cpp @@ -42,6 +42,11 @@ const VersionToSettingsChangesMap & getSettingsChangesHistory() addSettingsChanges(settings_changes_history, "26.6.2.20001.altinityantalya", { {"use_puffin_files_cache", false, true, "Enables cache of parsed Puffin file content such as deletion vectors."}, + {"allow_experimental_export_ttl", false, false, "New setting to allow `TTL ... EXPORT TO TABLE`."}, + {"export_merge_tree_retry_initial_backoff_seconds", 5, 5, "Renamed from `export_merge_tree_partition_retry_initial_backoff_seconds`, export tasks are no longer tied to a partition."}, + {"export_merge_tree_retry_max_backoff_seconds", 300, 300, "Renamed from `export_merge_tree_partition_retry_max_backoff_seconds`, export tasks are no longer tied to a partition."}, + {"export_merge_tree_task_timeout_seconds", 86400, 86400, "Renamed from `export_merge_tree_partition_task_timeout_seconds`, export tasks are no longer tied to a partition."}, + {"export_merge_tree_partition_force_export", false, false, "Obsolete: `EXPORT PARTITION` no longer refuses re-exports, so there is nothing to force."}, }); addSettingsChanges(settings_changes_history, "26.6", @@ -1348,6 +1353,14 @@ const VersionToSettingsChangesMap & getMergeTreeSettingsChangesHistory() { addSettingsChanges(merge_tree_settings_changes_history, "26.6", { + {"ttl_export_check_period_seconds", 10, 10, "New setting"}, + {"ttl_export_batch_window_seconds", 60, 60, "New setting"}, + {"ttl_export_batch_max_delay_seconds", 600, 600, "New setting"}, + {"ttl_export_batch_min_bytes", 256_MiB, 256_MiB, "New setting"}, + {"ttl_export_max_parts_per_group", 100, 100, "New setting"}, + {"ttl_export_max_bytes_per_group", 10_GiB, 10_GiB, "New setting"}, + {"ttl_export_max_concurrent_groups", 4, 4, "New setting"}, + {"ttl_export_settings_profile", "", "", "New setting"}, {"packed_skip_index_max_bytes", 0, 0, "New setting. Pack any skip-index substream whose serialized on-disk size is at most this many bytes into a single skp_idx.packed archive per part; larger substreams stay in the standalone skp_idx_.idx2 / .mrk2 layout. Decision is made per substream at write time."}, {"allow_tuple_element_aggregation", false, false, "New setting"}, {"shared_merge_tree_enable_keeper_parts_extra_data", false, true, "Enable coordinated merges by default"}, diff --git a/src/Databases/enableAllExperimentalSettings.cpp b/src/Databases/enableAllExperimentalSettings.cpp index ba08b2e552bc..487370d8f819 100644 --- a/src/Databases/enableAllExperimentalSettings.cpp +++ b/src/Databases/enableAllExperimentalSettings.cpp @@ -14,6 +14,7 @@ namespace DB void enableAllExperimentalSettings(ContextMutablePtr context) { context->setSetting("allow_experimental_codecs", 1); + context->setSetting("allow_experimental_export_ttl", 1); context->setSetting("allow_experimental_window_view", 1); context->setSetting("allow_experimental_funnel_functions", 1); context->setSetting("allow_experimental_nlp_functions", 1); diff --git a/src/Interpreters/InterpreterKillQueryQuery.cpp b/src/Interpreters/InterpreterKillQueryQuery.cpp index 6824036e4199..568fe3268198 100644 --- a/src/Interpreters/InterpreterKillQueryQuery.cpp +++ b/src/Interpreters/InterpreterKillQueryQuery.cpp @@ -257,7 +257,7 @@ BlockIO InterpreterKillQueryQuery::execute() break; } - case ASTKillQueryQuery::Type::ExportPartition: + case ASTKillQueryQuery::Type::Export: { if (!getContext()->getServerSettings()[ServerSetting::allow_experimental_export_merge_tree_partition]) { @@ -267,7 +267,7 @@ BlockIO InterpreterKillQueryQuery::execute() Block exports_block = getSelectResult( "source_database, source_table, transaction_id, destination_database, destination_table, partition_id", - "system.partition_exports"); + "system.distributed_exports"); if (exports_block.empty()) return res_io; @@ -319,7 +319,7 @@ BlockIO InterpreterKillQueryQuery::execute() access_denied = true; continue; } - code = storage->killExportPartition(std::string{transaction_id}); + code = storage->killExportTask(std::string{transaction_id}); } } @@ -327,7 +327,7 @@ BlockIO InterpreterKillQueryQuery::execute() } if (res_columns[0]->empty() && access_denied) - throw Exception(ErrorCodes::ACCESS_DENIED, "Not allowed to kill export partition. " + throw Exception(ErrorCodes::ACCESS_DENIED, "Not allowed to kill export. " "To execute this query, it's necessary to have the grant {}", required_access_rights.toString()); res_io.pipeline = QueryPipeline(Pipe(std::make_shared(std::make_shared(header.cloneWithColumns(std::move(res_columns)))))); @@ -559,7 +559,7 @@ AccessRightsElements InterpreterKillQueryQuery::getRequiredAccessForDDLOnCluster | AccessType::ALTER_REWRITE_PARTS ); /// todo arthur think about this - else if (query.type == ASTKillQueryQuery::Type::ExportPartition) + else if (query.type == ASTKillQueryQuery::Type::Export) required_access.emplace_back(AccessType::ALTER_EXPORT_PARTITION); return required_access; } diff --git a/src/Parsers/ASTKillQueryQuery.cpp b/src/Parsers/ASTKillQueryQuery.cpp index 9911e60b5ed9..2e5c909cc62c 100644 --- a/src/Parsers/ASTKillQueryQuery.cpp +++ b/src/Parsers/ASTKillQueryQuery.cpp @@ -27,8 +27,8 @@ void ASTKillQueryQuery::formatQueryImpl(WriteBuffer & ostr, const FormatSettings case Type::Transaction: ostr << "TRANSACTION"; break; - case Type::ExportPartition: - ostr << "EXPORT PARTITION"; + case Type::Export: + ostr << "EXPORT"; break; } diff --git a/src/Parsers/ASTKillQueryQuery.h b/src/Parsers/ASTKillQueryQuery.h index 101c83c566f7..bb3d5109ca8c 100644 --- a/src/Parsers/ASTKillQueryQuery.h +++ b/src/Parsers/ASTKillQueryQuery.h @@ -13,7 +13,7 @@ class ASTKillQueryQuery : public ASTQueryWithOutput, public ASTQueryWithOnCluste { Query, /// KILL QUERY Mutation, /// KILL MUTATION - ExportPartition, /// KILL EXPORT_PARTITION + Export, /// KILL EXPORT PartMoveToShard, /// KILL PART_MOVE_TO_SHARD Transaction, /// KILL TRANSACTION }; diff --git a/src/Parsers/ASTTTLElement.cpp b/src/Parsers/ASTTTLElement.cpp index c6c8d1eb8a08..2af34a109234 100644 --- a/src/Parsers/ASTTTLElement.cpp +++ b/src/Parsers/ASTTTLElement.cpp @@ -79,6 +79,13 @@ void ASTTTLElement::formatImpl(WriteBuffer & ostr, const FormatSettings & settin ostr << " RECOMPRESS "; recompression_codec->format(ostr, settings, state, frame); } + else if (mode == TTLMode::EXPORT) + { + ostr << " EXPORT TO TABLE "; + if (!destination_database.empty()) + ostr << backQuoteIfNeed(destination_database) << "."; + ostr << backQuoteIfNeed(destination_name); + } else if (mode == TTLMode::DELETE) { /// It would be better to output "DELETE" here but that will break compatibility with earlier versions. diff --git a/src/Parsers/ASTTTLElement.h b/src/Parsers/ASTTTLElement.h index 8cef82ed293b..3a3d0b348fa0 100644 --- a/src/Parsers/ASTTTLElement.h +++ b/src/Parsers/ASTTTLElement.h @@ -16,6 +16,9 @@ class ASTTTLElement : public IAST TTLMode mode; DataDestinationType destination_type; String destination_name; + /// `EXPORT TO TABLE` only: database of the destination table `destination_name`. Empty means the + /// database of the table the TTL belongs to. + String destination_database; bool if_exists = false; ASTs group_by_key; diff --git a/src/Parsers/CommonParsers.h b/src/Parsers/CommonParsers.h index 9e42ee533241..221c51408d61 100644 --- a/src/Parsers/CommonParsers.h +++ b/src/Parsers/CommonParsers.h @@ -210,6 +210,7 @@ namespace DB MR_MACROS(EXECUTE_AS, "EXECUTE AS") \ MR_MACROS(EXISTS, "EXISTS") \ MR_MACROS(EXPLAIN, "EXPLAIN") \ + MR_MACROS(EXPORT, "EXPORT") \ MR_MACROS(EXPRESSION, "EXPRESSION") \ MR_MACROS(EXTENDED, "EXTENDED") \ MR_MACROS(EXTERNAL_DDL_FROM, "EXTERNAL DDL FROM") \ diff --git a/src/Parsers/ExpressionElementParsers.cpp b/src/Parsers/ExpressionElementParsers.cpp index e0dd429d1a8f..c91baf8e4377 100644 --- a/src/Parsers/ExpressionElementParsers.cpp +++ b/src/Parsers/ExpressionElementParsers.cpp @@ -2576,6 +2576,9 @@ bool ParserTTLElement::parseImpl(Pos & pos, ASTPtr & node, Expected & expected) ParserKeyword s_materialize_ttl(Keyword::MATERIALIZE_TTL); ParserKeyword s_remove_ttl(Keyword::REMOVE_TTL); ParserKeyword s_modify_ttl(Keyword::MODIFY_TTL); + ParserKeyword s_export(Keyword::EXPORT); + ParserKeyword s_to_table(Keyword::TO_TABLE); + ParserToken s_dot(TokenType::Dot); ParserIdentifier parser_identifier; ParserStringLiteral parser_string_literal; @@ -2620,6 +2623,13 @@ bool ParserTTLElement::parseImpl(Pos & pos, ASTPtr & node, Expected & expected) { mode = TTLMode::RECOMPRESS; } + else if (s_export.ignore(pos, expected)) + { + if (!s_to_table.ignore(pos, expected)) + return false; + mode = TTLMode::EXPORT; + destination_type = DataDestinationType::TABLE; + } else { /// DELETE is the default mode. @@ -2632,6 +2642,7 @@ bool ParserTTLElement::parseImpl(Pos & pos, ASTPtr & node, Expected & expected) ASTPtr recompression_codec; ASTPtr group_by_assignments; bool if_exists = false; + String destination_database; if (mode == TTLMode::MOVE) { @@ -2671,8 +2682,25 @@ bool ParserTTLElement::parseImpl(Pos & pos, ASTPtr & node, Expected & expected) if (!parser_codec.parse(pos, recompression_codec, expected)) return false; } + else if (mode == TTLMode::EXPORT) + { + ASTPtr first_name; + if (!parser_identifier.parse(pos, first_name, expected)) + return false; + destination_name = getIdentifierName(first_name); + + if (s_dot.ignore(pos, expected)) + { + ASTPtr table_name; + if (!parser_identifier.parse(pos, table_name, expected)) + return false; + destination_database = destination_name; + destination_name = getIdentifierName(table_name); + } + } auto ttl_element = make_intrusive(mode, destination_type, destination_name, if_exists); + ttl_element->destination_database = destination_database; ttl_element->setTTL(std::move(ttl_expr)); if (where_expr) ttl_element->setWhere(std::move(where_expr)); diff --git a/src/Parsers/ParserKillQueryQuery.cpp b/src/Parsers/ParserKillQueryQuery.cpp index 99f2d6fd2d64..e5f1e619c3fb 100644 --- a/src/Parsers/ParserKillQueryQuery.cpp +++ b/src/Parsers/ParserKillQueryQuery.cpp @@ -17,7 +17,7 @@ bool ParserKillQueryQuery::parseImpl(Pos & pos, ASTPtr & node, Expected & expect ParserKeyword p_kill{Keyword::KILL}; ParserKeyword p_query{Keyword::QUERY}; ParserKeyword p_mutation{Keyword::MUTATION}; - ParserKeyword p_export_partition{Keyword::EXPORT_PARTITION}; + ParserKeyword p_export{Keyword::EXPORT}; ParserKeyword p_part_move_to_shard{Keyword::PART_MOVE_TO_SHARD}; ParserKeyword p_transaction{Keyword::TRANSACTION}; ParserKeyword p_on{Keyword::ON}; @@ -34,8 +34,8 @@ bool ParserKillQueryQuery::parseImpl(Pos & pos, ASTPtr & node, Expected & expect query->type = ASTKillQueryQuery::Type::Query; else if (p_mutation.ignore(pos, expected)) query->type = ASTKillQueryQuery::Type::Mutation; - else if (p_export_partition.ignore(pos, expected)) - query->type = ASTKillQueryQuery::Type::ExportPartition; + else if (p_export.ignore(pos, expected)) + query->type = ASTKillQueryQuery::Type::Export; else if (p_part_move_to_shard.ignore(pos, expected)) query->type = ASTKillQueryQuery::Type::PartMoveToShard; else if (p_transaction.ignore(pos, expected)) diff --git a/src/Processors/TTL/TTLUpdateInfoAlgorithm.cpp b/src/Processors/TTL/TTLUpdateInfoAlgorithm.cpp index 0ad85213e3bd..97da73e961c6 100644 --- a/src/Processors/TTL/TTLUpdateInfoAlgorithm.cpp +++ b/src/Processors/TTL/TTLUpdateInfoAlgorithm.cpp @@ -40,6 +40,10 @@ void TTLUpdateInfoAlgorithm::finalize(const MutableDataPartPtr & data_part) cons { data_part->ttl_infos.moves_ttl[ttl_update_key] = new_ttl_info; } + else if (ttl_update_field == TTLUpdateField::EXPORT_TTL) + { + data_part->ttl_infos.export_ttl[ttl_update_key] = new_ttl_info; + } else if (ttl_update_field == TTLUpdateField::GROUP_BY_TTL) { data_part->ttl_infos.group_by_ttl[ttl_update_key] = new_ttl_info; diff --git a/src/Processors/TTL/TTLUpdateInfoAlgorithm.h b/src/Processors/TTL/TTLUpdateInfoAlgorithm.h index 52cd15095674..6ae0b537abfa 100644 --- a/src/Processors/TTL/TTLUpdateInfoAlgorithm.h +++ b/src/Processors/TTL/TTLUpdateInfoAlgorithm.h @@ -11,6 +11,7 @@ enum class TTLUpdateField : uint8_t TABLE_TTL, ROWS_WHERE_TTL, MOVES_TTL, + EXPORT_TTL, RECOMPRESSION_TTL, GROUP_BY_TTL, }; diff --git a/src/Processors/Transforms/TTLCalcTransform.cpp b/src/Processors/Transforms/TTLCalcTransform.cpp index b23aef1bc705..bd2931383244 100644 --- a/src/Processors/Transforms/TTLCalcTransform.cpp +++ b/src/Processors/Transforms/TTLCalcTransform.cpp @@ -69,6 +69,11 @@ TTLCalcTransform::TTLCalcTransform( getExpressions(move_ttl, subqueries_for_sets, context), move_ttl, TTLUpdateField::MOVES_TTL, move_ttl.result_column, old_ttl_infos.moves_ttl[move_ttl.result_column], current_time_, force_)); + for (const auto & export_ttl : metadata_snapshot_->getExportTTLs()) + algorithms.emplace_back(std::make_unique( + getExpressions(export_ttl, subqueries_for_sets, context), export_ttl, + TTLUpdateField::EXPORT_TTL, export_ttl.result_column, old_ttl_infos.export_ttl[export_ttl.result_column], current_time_, force_)); + for (const auto & recompression_ttl : metadata_snapshot_->getRecompressionTTLs()) algorithms.emplace_back(std::make_unique( getExpressions(recompression_ttl, subqueries_for_sets, context), recompression_ttl, diff --git a/src/Processors/Transforms/TTLTransform.cpp b/src/Processors/Transforms/TTLTransform.cpp index b2e3696253fc..4161fd30ee22 100644 --- a/src/Processors/Transforms/TTLTransform.cpp +++ b/src/Processors/Transforms/TTLTransform.cpp @@ -142,6 +142,11 @@ TTLTransform::TTLTransform( getExpressions(move_ttl, subqueries_for_sets, context), move_ttl, TTLUpdateField::MOVES_TTL, move_ttl.result_column, old_ttl_infos.moves_ttl[move_ttl.result_column], current_time_, force_)); + for (const auto & export_ttl : metadata_snapshot_->getExportTTLs()) + algorithms.emplace_back(std::make_unique( + getExpressions(export_ttl, subqueries_for_sets, context), export_ttl, + TTLUpdateField::EXPORT_TTL, export_ttl.result_column, old_ttl_infos.export_ttl[export_ttl.result_column], current_time_, force_)); + for (const auto & recompression_ttl : metadata_snapshot_->getRecompressionTTLs()) algorithms.emplace_back(std::make_unique( getExpressions(recompression_ttl, subqueries_for_sets, context), recompression_ttl, diff --git a/src/Storages/ExportPartitionCommitInfoEntry.h b/src/Storages/ExportCommitInfoEntry.h similarity index 87% rename from src/Storages/ExportPartitionCommitInfoEntry.h rename to src/Storages/ExportCommitInfoEntry.h index b69a98ab5753..6b031743aaba 100644 --- a/src/Storages/ExportPartitionCommitInfoEntry.h +++ b/src/Storages/ExportCommitInfoEntry.h @@ -13,7 +13,7 @@ namespace DB /// writing the object-storage files and recording this entry; in that case the /// task still reaches COMPLETED through the recovery path but the commit info /// remains absent. This is best-effort observability and acceptable. -struct ExportPartitionCommitInfoEntry +struct ExportCommitInfoEntry { /// Iceberg: path (in destination object storage) of the new vN.metadata.json /// written by the commit. @@ -28,7 +28,7 @@ struct ExportPartitionCommitInfoEntry String iceberg_manifest_file; /// Plain object storage: path of the commit marker file written by - /// StorageObjectStorage::commitExportPartitionTransaction. Empty for Iceberg. + /// StorageObjectStorage::commitExportTransaction. Empty for Iceberg. String commit_marker_file; Poco::JSON::Object::Ptr toJsonObject() const @@ -41,9 +41,9 @@ struct ExportPartitionCommitInfoEntry return json; } - static ExportPartitionCommitInfoEntry fromJsonObject(const Poco::JSON::Object::Ptr & json) + static ExportCommitInfoEntry fromJsonObject(const Poco::JSON::Object::Ptr & json) { - ExportPartitionCommitInfoEntry entry; + ExportCommitInfoEntry entry; if (json->has("iceberg_metadata_file")) entry.iceberg_metadata_file = json->getValue("iceberg_metadata_file"); @@ -67,7 +67,7 @@ struct ExportPartitionCommitInfoEntry return oss.str(); } - static ExportPartitionCommitInfoEntry fromJsonString(const std::string & json_string) + static ExportCommitInfoEntry fromJsonString(const std::string & json_string) { if (json_string.empty()) return {}; diff --git a/src/Storages/ExportReplicatedMergeTreePartitionTaskEntry.h b/src/Storages/ExportReplicatedMergeTreeTaskEntry.h similarity index 59% rename from src/Storages/ExportReplicatedMergeTreePartitionTaskEntry.h rename to src/Storages/ExportReplicatedMergeTreeTaskEntry.h index dbd12fe0b442..7a23ca333f99 100644 --- a/src/Storages/ExportReplicatedMergeTreePartitionTaskEntry.h +++ b/src/Storages/ExportReplicatedMergeTreeTaskEntry.h @@ -2,8 +2,7 @@ #include #include -#include -#include +#include #include #include #include @@ -12,10 +11,10 @@ namespace DB { -struct ExportReplicatedMergeTreePartitionTaskEntry +struct ExportReplicatedMergeTreeTaskEntry { using DataPartPtr = std::shared_ptr; - ExportReplicatedMergeTreePartitionManifest manifest; + ExportReplicatedMergeTreeTaskManifest manifest; enum class Status { @@ -46,16 +45,10 @@ struct ExportReplicatedMergeTreePartitionTaskEntry mutable std::map> destination_file_paths_per_part; /// In-memory mirror of the /commit_info znode (written atomically - /// with the COMPLETED status transition; see ExportPartitionUtils::commit). + /// with the COMPLETED status transition; see ExportTaskUtils::commit). /// nullopt until commit_info is observed in ZK. Empty fields inside the struct /// for non-Iceberg destinations. - mutable std::optional commit_info; - - std::string getCompositeKey() const - { - return ExportPartitionUtils::compositeKey( - manifest.partition_id, manifest.destination_database, manifest.destination_table); - } + mutable std::optional commit_info; std::string getTransactionId() const { @@ -69,27 +62,22 @@ struct ExportReplicatedMergeTreePartitionTaskEntry } }; -struct ExportPartitionTaskEntryTagByCompositeKey {}; -struct ExportPartitionTaskEntryTagByCreateTime {}; -struct ExportPartitionTaskEntryTagByTransactionId {}; +struct ExportTaskEntryTagByTransactionId {}; +struct ExportTaskEntryTagByCreateTime {}; -// Multi-index container for export partition task entries -// - Index 0 (TagByCompositeKey): hashed_unique on composite key for O(1) lookup +// Multi-index container for export task entries. A task is stored at `/exports/`. +// - Index 0 (TagByTransactionId): hashed_unique on the transaction id, the task's key // - Index 1 (TagByCreateTime): ordered_non_unique on create_time for sorted iteration -using ExportPartitionTaskEntriesContainer = boost::multi_index_container< - ExportReplicatedMergeTreePartitionTaskEntry, +using ExportTaskEntriesContainer = boost::multi_index_container< + ExportReplicatedMergeTreeTaskEntry, boost::multi_index::indexed_by< boost::multi_index::hashed_unique< - boost::multi_index::tag, - boost::multi_index::const_mem_fun + boost::multi_index::tag, + boost::multi_index::const_mem_fun >, boost::multi_index::ordered_non_unique< - boost::multi_index::tag, - boost::multi_index::const_mem_fun - >, - boost::multi_index::hashed_unique< - boost::multi_index::tag, - boost::multi_index::const_mem_fun + boost::multi_index::tag, + boost::multi_index::const_mem_fun > > >; diff --git a/src/Storages/ExportReplicatedMergeTreePartitionManifest.h b/src/Storages/ExportReplicatedMergeTreeTaskManifest.h similarity index 83% rename from src/Storages/ExportReplicatedMergeTreePartitionManifest.h rename to src/Storages/ExportReplicatedMergeTreeTaskManifest.h index dc4de190a980..832fbd83d126 100644 --- a/src/Storages/ExportReplicatedMergeTreePartitionManifest.h +++ b/src/Storages/ExportReplicatedMergeTreeTaskManifest.h @@ -6,14 +6,20 @@ #include #include #include -#include +#include +#include #include #include namespace DB { -struct ExportReplicatedMergeTreePartitionProcessingPartEntry +namespace ErrorCodes +{ + extern const int INCORRECT_DATA; +} + +struct ExportReplicatedMergeTreeProcessingPartEntry { enum class Status @@ -41,13 +47,13 @@ struct ExportReplicatedMergeTreePartitionProcessingPartEntry return oss.str(); } - static ExportReplicatedMergeTreePartitionProcessingPartEntry fromJsonString(const std::string & json_string) + static ExportReplicatedMergeTreeProcessingPartEntry fromJsonString(const std::string & json_string) { Poco::JSON::Parser parser; auto json = parser.parse(json_string).extract(); chassert(json); - ExportReplicatedMergeTreePartitionProcessingPartEntry entry; + ExportReplicatedMergeTreeProcessingPartEntry entry; entry.part_name = json->getValue("part_name"); entry.status = magic_enum::enum_cast(json->getValue("status")).value(); @@ -113,7 +119,7 @@ struct LastExceptionEntry } }; -struct ExportReplicatedMergeTreePartitionProcessedPartEntry +struct ExportReplicatedMergeTreeProcessedPartEntry { String part_name; std::vector paths_in_destination; @@ -131,13 +137,13 @@ struct ExportReplicatedMergeTreePartitionProcessedPartEntry return oss.str(); } - static ExportReplicatedMergeTreePartitionProcessedPartEntry fromJsonString(const std::string & json_string) + static ExportReplicatedMergeTreeProcessedPartEntry fromJsonString(const std::string & json_string) { Poco::JSON::Parser parser; auto json = parser.parse(json_string).extract(); chassert(json); - ExportReplicatedMergeTreePartitionProcessedPartEntry entry; + ExportReplicatedMergeTreeProcessedPartEntry entry; entry.part_name = json->getValue("part_name"); @@ -151,13 +157,21 @@ struct ExportReplicatedMergeTreePartitionProcessedPartEntry } }; -struct ExportReplicatedMergeTreePartitionManifest +/// Descriptor of an export task of a `ReplicatedMergeTree` table, stored at +/// `/exports//metadata.json`. A task exports a group of parts; today +/// they all belong to one partition, which is derived from the part names where needed. +struct ExportReplicatedMergeTreeTaskManifest { String transaction_id; String query_id; - String partition_id; + ExportTaskSource source = ExportTaskSource::query; String destination_database; String destination_table; + /// UUID of the destination table when the task was created, empty if it has none. + String destination_uuid; + /// TTL export only: transaction ids of earlier tasks that failed to export some of these parts. + /// The commit checks whether any of them landed at the destination after all. + std::vector retry_of; String source_replica; size_t number_of_parts; std::vector parts; @@ -194,9 +208,18 @@ struct ExportReplicatedMergeTreePartitionManifest Poco::JSON::Object json; json.set("transaction_id", transaction_id); json.set("query_id", query_id); - json.set("partition_id", partition_id); + json.set("source", String(magic_enum::enum_name(source))); json.set("destination_database", destination_database); json.set("destination_table", destination_table); + if (!destination_uuid.empty()) + json.set("destination_uuid", destination_uuid); + if (!retry_of.empty()) + { + Poco::JSON::Array::Ptr retry_of_array = new Poco::JSON::Array(); + for (const auto & transaction : retry_of) + retry_of_array->add(transaction); + json.set("retry_of", retry_of_array); + } json.set("source_replica", source_replica); json.set("number_of_parts", number_of_parts); @@ -242,18 +265,32 @@ struct ExportReplicatedMergeTreePartitionManifest return oss.str(); } - static ExportReplicatedMergeTreePartitionManifest fromJsonString(const std::string & json_string) + static ExportReplicatedMergeTreeTaskManifest fromJsonString(const std::string & json_string) { Poco::JSON::Parser parser; auto json = parser.parse(json_string).extract(); chassert(json); - ExportReplicatedMergeTreePartitionManifest manifest; + ExportReplicatedMergeTreeTaskManifest manifest; manifest.transaction_id = json->getValue("transaction_id"); manifest.query_id = json->getValue("query_id"); - manifest.partition_id = json->getValue("partition_id"); + if (json->has("source")) + { + const auto source = magic_enum::enum_cast(json->getValue("source")); + if (!source) + throw Exception(ErrorCodes::INCORRECT_DATA, "Unknown source '{}' in export task descriptor", json->getValue("source")); + manifest.source = *source; + } manifest.destination_database = json->getValue("destination_database"); manifest.destination_table = json->getValue("destination_table"); + if (json->has("destination_uuid")) + manifest.destination_uuid = json->getValue("destination_uuid"); + if (json->has("retry_of")) + { + const auto retry_of_array = json->getArray("retry_of"); + for (size_t i = 0; i < retry_of_array->size(); ++i) + manifest.retry_of.push_back(retry_of_array->getElement(static_cast(i))); + } manifest.source_replica = json->getValue("source_replica"); manifest.number_of_parts = json->getValue("number_of_parts"); diff --git a/src/Storages/ExportTaskSource.h b/src/Storages/ExportTaskSource.h new file mode 100644 index 000000000000..b68e0349ae42 --- /dev/null +++ b/src/Storages/ExportTaskSource.h @@ -0,0 +1,17 @@ +#pragma once + +#include + +namespace DB +{ + +/// What created an export task of a `MergeTree` table. +enum class ExportTaskSource : uint8_t +{ + /// `ALTER TABLE ... EXPORT PARTITION`. + query, + /// The table's `TTL ... EXPORT TO TABLE` expression. + ttl, +}; + +} diff --git a/src/Storages/IStorage.cpp b/src/Storages/IStorage.cpp index 81bd9f1148c6..b93a860b0107 100644 --- a/src/Storages/IStorage.cpp +++ b/src/Storages/IStorage.cpp @@ -311,7 +311,7 @@ CancellationCode IStorage::killPartMoveToShard(const UUID & /*task_uuid*/) throw Exception(ErrorCodes::NOT_IMPLEMENTED, "Part moves between shards are not supported by storage {}", getName()); } -CancellationCode IStorage::killExportPartition(const String & /*transaction_id*/) +CancellationCode IStorage::killExportTask(const String & /*transaction_id*/) { throw Exception(ErrorCodes::NOT_IMPLEMENTED, "Export partition is not supported by storage {}", getName()); } diff --git a/src/Storages/IStorage.h b/src/Storages/IStorage.h index 851fe4a87c7b..e2cddb741ef7 100644 --- a/src/Storages/IStorage.h +++ b/src/Storages/IStorage.h @@ -505,7 +505,7 @@ It is currently only implemented in StorageObjectStorage. throw Exception(ErrorCodes::NOT_IMPLEMENTED, "Import is not implemented for storage {}", getName()); } - struct IcebergCommitExportPartitionArguments + struct IcebergCommitExportArguments { std::string metadata_json_string; /// Representative source partition-key columns from one exported part (the part's @@ -516,8 +516,8 @@ It is currently only implemented in StorageObjectStorage. }; /// Paths produced by the destination storage during commit. Surfaced via - /// system.partition_exports for debugging - struct ExportPartitionCommitInfo + /// system.distributed_exports for debugging + struct ExportCommitInfo { /// Iceberg destinations only. String iceberg_metadata_file; @@ -525,20 +525,30 @@ It is currently only implemented in StorageObjectStorage. String iceberg_manifest_file; /// Plain object storage destinations only: path of the commit marker file - /// written/observed by StorageObjectStorage::commitExportPartitionTransaction. + /// written/observed by StorageObjectStorage::commitExportTransaction. String commit_marker_file; }; - virtual ExportPartitionCommitInfo commitExportPartitionTransaction( + /// Makes the files exported by the transaction visible in this storage. All of them belong to + /// one source partition, `partition_id`. + virtual ExportCommitInfo commitExportTransaction( const String & /* transaction_id */, const String & /* partition_id */, const Strings & /* exported_paths */, - const IcebergCommitExportPartitionArguments & /* iceberg_commit_export_partition_arguments */, + const IcebergCommitExportArguments & /* iceberg_commit_export_arguments */, ContextPtr /* local_context */) { - throw Exception(ErrorCodes::NOT_IMPLEMENTED, "commitExportPartitionTransaction is not implemented for storage type {}", getName()); + throw Exception(ErrorCodes::NOT_IMPLEMENTED, "commitExportTransaction is not implemented for storage type {}", getName()); + } + + /// Whether `commitExportTransaction` already made `transaction_id` visible in this storage. + /// Resolves an export whose outcome is unknown, e.g. one killed or timed out after its commit may have landed. + virtual bool isExportTransactionCommitted( + const String & /* transaction_id */, + ContextPtr /* local_context */) + { + throw Exception(ErrorCodes::NOT_IMPLEMENTED, "isExportTransactionCommitted is not implemented for storage type {}", getName()); } - /** Writes the data to a table in distributed manner. * It is supposed that implementation looks into SELECT part of the query and executes distributed @@ -653,7 +663,7 @@ It is currently only implemented in StorageObjectStorage. virtual void setMutationCSN(const String & /*mutation_id*/, UInt64 /*csn*/); /// Cancel a replicated partition export by transaction id. - virtual CancellationCode killExportPartition(const String & /*transaction_id*/); + virtual CancellationCode killExportTask(const String & /*transaction_id*/); /// Cancel a part move to shard. virtual CancellationCode killPartMoveToShard(const UUID & /*task_uuid*/); diff --git a/src/Storages/MergeTree/Compaction/MergePredicates/DistributedMergePredicate.h b/src/Storages/MergeTree/Compaction/MergePredicates/DistributedMergePredicate.h index 79c6ee69b1a7..810280232f12 100644 --- a/src/Storages/MergeTree/Compaction/MergePredicates/DistributedMergePredicate.h +++ b/src/Storages/MergeTree/Compaction/MergePredicates/DistributedMergePredicate.h @@ -3,6 +3,7 @@ #include #include #include +#include #include #include @@ -85,6 +86,12 @@ class DistributedMergePredicate : public IMergePredicate if (left.is_in_volume_where_merges_avoid || right.is_in_volume_where_merges_avoid) return std::unexpected(PreformattedMessage::create("One of parts ({}, {}) lies on volume where merges should be avoided", left.name, right.name)); + if (export_fence_ptr) + { + if (auto reason = export_fence_ptr->checkCanMerge(left.info, right.info)) + return std::unexpected(PreformattedMessage::create("{}", *reason)); + } + int64_t left_max_block = left.info.max_block; int64_t right_min_block = right.info.min_block; chassert(left_max_block < right_min_block); @@ -202,6 +209,9 @@ class DistributedMergePredicate : public IMergePredicate /// Patch parts that should be applied at merges if apply_patches_on_merge is enabled. PatchInfosByPartition patches_by_partition; + + /// Export states of parts of partitions exported by iterations: parts in different states are not merged. + const ExportFence * export_fence_ptr = nullptr; }; } diff --git a/src/Storages/MergeTree/Compaction/MergePredicates/MergeTreeMergePredicate.cpp b/src/Storages/MergeTree/Compaction/MergePredicates/MergeTreeMergePredicate.cpp index c0b71811c073..43dff066d2e5 100644 --- a/src/Storages/MergeTree/Compaction/MergePredicates/MergeTreeMergePredicate.cpp +++ b/src/Storages/MergeTree/Compaction/MergePredicates/MergeTreeMergePredicate.cpp @@ -24,6 +24,8 @@ MergeTreeMergePredicate::MergeTreeMergePredicate(const StorageMergeTree & storag , merge_mutate_lock(merge_mutate_lock_) , committing_blocks(storage.getCommittingBlocks()) , min_update_block(getMinUpdateBlockNumber(committing_blocks)) + , export_fence(storage.getExportFence()) + , delete_gate(storage.getExportTTLDeleteGate()) { auto patches_vector = getPatchPartInfos(storage); patches_by_partition = getPatchPartsByPartition(patches_vector, min_update_block.value_or(std::numeric_limits::max())); @@ -47,6 +49,12 @@ std::expected MergeTreeMergePredicate::canMergeParts( left.projection_names, left.name, right.projection_names, right.name)); } + if (export_fence) + { + if (auto reason = export_fence->checkCanMerge(left.info, right.info)) + return std::unexpected(PreformattedMessage::create("{}", *reason)); + } + { uint64_t left_mutation_version = storage.getCurrentMutationVersion(left.info.getDataVersion(), merge_mutate_lock); uint64_t right_mutation_version = storage.getCurrentMutationVersion(right.info.getDataVersion(), merge_mutate_lock); @@ -74,6 +82,9 @@ std::expected MergeTreeMergePredicate::canUsePartInMe if (storage.currently_merging_mutating_parts.contains(part->info)) return std::unexpected(PreformattedMessage::create("Part {} currently in a merging or mutating process", part->name)); + if (auto reason = delete_gate.check(part->name, part->info, part->ttl_infos, export_fence.get())) + return std::unexpected(PreformattedMessage::create("{}", *reason)); + if (min_update_block && part->info.getDataVersion() >= *min_update_block) { return std::unexpected(PreformattedMessage::create( diff --git a/src/Storages/MergeTree/Compaction/MergePredicates/MergeTreeMergePredicate.h b/src/Storages/MergeTree/Compaction/MergePredicates/MergeTreeMergePredicate.h index fdeb4511b845..45f948d87bba 100644 --- a/src/Storages/MergeTree/Compaction/MergePredicates/MergeTreeMergePredicate.h +++ b/src/Storages/MergeTree/Compaction/MergePredicates/MergeTreeMergePredicate.h @@ -4,6 +4,8 @@ #include #include #include +#include +#include namespace DB { @@ -24,6 +26,9 @@ class MergeTreeMergePredicate final : public IMergePredicate PatchInfosByPartition patches_by_partition; CommittingBlocksSet committing_blocks; std::optional min_update_block; + /// Taken under `merge_mutate_lock`, which also guards claiming parts for the `EXPORT` TTL. + ExportFencePtr export_fence; + ExportTTLDeleteGate delete_gate; }; using MergeTreeMergePredicatePtr = std::shared_ptr; diff --git a/src/Storages/MergeTree/Compaction/MergePredicates/ReplicatedMergeTreeMergePredicate.cpp b/src/Storages/MergeTree/Compaction/MergePredicates/ReplicatedMergeTreeMergePredicate.cpp index 4fae999db306..75cf801840e5 100644 --- a/src/Storages/MergeTree/Compaction/MergePredicates/ReplicatedMergeTreeMergePredicate.cpp +++ b/src/Storages/MergeTree/Compaction/MergePredicates/ReplicatedMergeTreeMergePredicate.cpp @@ -43,6 +43,9 @@ std::expected ReplicatedMergeTreeBaseMergePredicate:: if (inprogress_quorum_part_ptr && *inprogress_quorum_part_ptr == part->name) return std::unexpected(PreformattedMessage::create("Quorum insert for part {} is currently in progress", part->name)); + if (auto reason = delete_gate.check(part->name, part->info, part->ttl_infos, export_fence.get())) + return std::unexpected(PreformattedMessage::create("{}", *reason)); + /// FIXME: remove lock here std::lock_guard lock(queue.state_mutex); return MergeCore::canUsePartInMerges(part->name, part->info); @@ -65,6 +68,9 @@ ReplicatedMergeTreeLocalMergePredicate::ReplicatedMergeTreeLocalMergePredicate(R std::lock_guard lock(queue_.state_mutex); patches_by_partition = getPatchPartsByPartition(virtual_parts_ptr->getPatchPartInfos(), {}); } + + /// Export states are not checked here: a stale copy could keep a partition from ever reaching + /// the ZooKeeper predicate, which checks them against Keeper. } ReplicatedMergeTreeZooKeeperMergePredicate::ReplicatedMergeTreeZooKeeperMergePredicate( @@ -121,6 +127,16 @@ ReplicatedMergeTreeZooKeeperMergePredicate::ReplicatedMergeTreeZooKeeperMergePre inprogress_quorum_part_ptr = inprogress_quorum_part.get(); } +void ReplicatedMergeTreeZooKeeperMergePredicate::loadExportFence(zkutil::ZooKeeperPtr & zookeeper) +{ + if (const auto export_ttl_index = queue.storage.getExportFence()) + { + std::tie(export_fence, export_fence_version) = export_ttl_index->getForMergeAssignment(zookeeper); + export_fence_ptr = export_fence.get(); + } + delete_gate = queue.storage.getExportTTLDeleteGate(); +} + bool ReplicatedMergeTreeZooKeeperMergePredicate::partParticipatesInReplaceRange(const MergeTreeData::DataPartPtr & part, PreformattedMessage & out_reason) const { /// FIXME: remove lock here diff --git a/src/Storages/MergeTree/Compaction/MergePredicates/ReplicatedMergeTreeMergePredicate.h b/src/Storages/MergeTree/Compaction/MergePredicates/ReplicatedMergeTreeMergePredicate.h index e3074fdfcb4d..d29401c2ae41 100644 --- a/src/Storages/MergeTree/Compaction/MergePredicates/ReplicatedMergeTreeMergePredicate.h +++ b/src/Storages/MergeTree/Compaction/MergePredicates/ReplicatedMergeTreeMergePredicate.h @@ -3,6 +3,7 @@ #include #include #include +#include namespace DB { @@ -24,6 +25,10 @@ class ReplicatedMergeTreeBaseMergePredicate : public DistributedMergePredicate inprogress_quorum_part; int32_t merges_version = -1; + int32_t export_fence_version = -1; }; using ReplicatedMergeTreeMergePredicatePtr = std::shared_ptr; diff --git a/src/Storages/MergeTree/ExportFence.cpp b/src/Storages/MergeTree/ExportFence.cpp new file mode 100644 index 000000000000..4aabfd83e28d --- /dev/null +++ b/src/Storages/MergeTree/ExportFence.cpp @@ -0,0 +1,135 @@ +#include + +#include +#include + +#include + +namespace DB +{ + +namespace ExportFenceUtils +{ + +std::vector compactRanges(std::vector ranges) +{ + std::sort(ranges.begin(), ranges.end(), [](const MergeTreePartInfo & lhs, const MergeTreePartInfo & rhs) + { + if (lhs.getPartitionId() != rhs.getPartitionId()) + return lhs.getPartitionId() < rhs.getPartitionId(); + return lhs.min_block < rhs.min_block; + }); + + std::vector result; + result.reserve(ranges.size()); + + for (auto & range : ranges) + { + if (!result.empty() + && result.back().getPartitionId() == range.getPartitionId() + && range.min_block <= result.back().max_block + 1) + { + result.back().max_block = std::max(result.back().max_block, range.max_block); + result.back().level = std::max(result.back().level, range.level); + continue; + } + + result.push_back(std::move(range)); + } + + return result; +} + +bool isCoveredByUnion(const MergeTreePartInfo & part, const std::vector & ranges) +{ + auto next_uncovered = part.min_block; + + for (const auto & range : compactRanges(ranges)) + { + if (range.getPartitionId() != part.getPartitionId() || range.max_block < next_uncovered) + continue; + + if (range.min_block > next_uncovered) + return false; + + next_uncovered = range.max_block + 1; + if (next_uncovered > part.max_block) + return true; + } + + return next_uncovered > part.max_block; +} + +bool intersectsAny(const MergeTreePartInfo & part, const std::vector & ranges) +{ + return std::any_of(ranges.begin(), ranges.end(), [&](const MergeTreePartInfo & range) { return !part.isDisjoint(range); }); +} + +} + +PartExportState ExportFenceEntry::classify(const MergeTreePartInfo & part) const +{ + if (ExportFenceUtils::intersectsAny(part, claimed)) + return PartExportState::CLAIMED; + if (ExportFenceUtils::intersectsAny(part, exported)) + return PartExportState::EXPORTED; + return PartExportState::NONE; +} + +std::optional ExportFence::checkCanMerge(const MergeTreePartInfo & left, const MergeTreePartInfo & right) const +{ + if (left.isPatch() || right.isPatch()) + return std::nullopt; + + const auto it = entries_by_partition.find(left.getPartitionId()); + if (it == entries_by_partition.end()) + return std::nullopt; + + for (const auto & entry : it->second) + { + const auto left_state = entry.classify(left); + const auto right_state = entry.classify(right); + if (left_state != right_state) + return fmt::format( + "Parts {} and {} have different export states for destination {}: {} and {}", + left.getPartNameForLogs(), right.getPartNameForLogs(), entry.destination, + magic_enum::enum_name(left_state), magic_enum::enum_name(right_state)); + + /// A task exports its parts by name, and a claim must consist of the ranges of one task's + /// parts, which is what makes committing a retried task exact. + if (left_state == PartExportState::CLAIMED) + return fmt::format( + "Parts {} and {} are being exported to destination {}", + left.getPartNameForLogs(), right.getPartNameForLogs(), entry.destination); + + /// The ranges of parts that no longer exist (e.g. an exported part removed by a delete TTL) + /// stay in the index. The merged part would span them, so it would be classified by them too. + if (right.min_block > left.max_block + 1) + { + const MergeTreePartInfo gap(left.getPartitionId(), left.max_block + 1, right.min_block - 1, 0, 0); + const auto gap_state = entry.classify(gap); + if (gap_state != PartExportState::NONE && gap_state != left_state) + return fmt::format( + "Parts {} and {} have different export states than blocks between them for destination {}: {} and {}", + left.getPartNameForLogs(), right.getPartNameForLogs(), entry.destination, + magic_enum::enum_name(left_state), magic_enum::enum_name(gap_state)); + } + } + + return std::nullopt; +} + +PartExportState ExportFence::classify(const MergeTreePartInfo & part, const String & destination) const +{ + const auto it = entries_by_partition.find(part.getPartitionId()); + if (it == entries_by_partition.end()) + return PartExportState::NONE; + + for (const auto & entry : it->second) + if (entry.destination == destination) + return entry.classify(part); + + return PartExportState::NONE; +} + +} diff --git a/src/Storages/MergeTree/ExportFence.h b/src/Storages/MergeTree/ExportFence.h new file mode 100644 index 000000000000..069fa37a4f7f --- /dev/null +++ b/src/Storages/MergeTree/ExportFence.h @@ -0,0 +1,76 @@ +#pragma once + +#include +#include + +#include +#include +#include +#include + +namespace DB +{ + +/// Export state of a data part relative to the `TTL ... EXPORT` of its partition to one destination. +enum class PartExportState : UInt8 +{ + /// The part holds rows that were never exported to the destination. + NONE, + /// The part holds rows reserved by a TTL export task that did not commit yet. + CLAIMED, + /// The part holds rows committed to the destination. + EXPORTED, +}; + +/// Block ranges exported to one destination, and claimed for it, in one partition. +/// +/// A part is classified by overlap, not by name: mutations, TTL rewrites and merges among parts of +/// the same state keep block ranges, so a part that overlaps an exported range consists of exported +/// rows only. That holds only because parts in different states are never merged together, which +/// is what `ExportFence::checkCanMerge` enforces. +struct ExportFenceEntry +{ + /// `db.table` of the destination, for messages only. + String destination; + /// Ranges committed to the destination, compacted. + std::vector exported; + /// Ranges claimed by TTL export tasks that did not commit, compacted. + std::vector claimed; + + PartExportState classify(const MergeTreePartInfo & part) const; +}; + +/// Snapshot of the TTL export state of every partition of a table, for every destination. Immutable +/// once published, so a merge predicate can read it without locks. +struct ExportFence +{ + std::unordered_map> entries_by_partition; + + bool empty() const { return entries_by_partition.empty(); } + + /// Returns the reason why the two parts must not be merged, or nothing if they may be: they must + /// have the same state, as must any exported or claimed blocks between them, and claimed parts + /// are not merged at all. + std::optional checkCanMerge(const MergeTreePartInfo & left, const MergeTreePartInfo & right) const; + + /// State of the part relative to `destination`, or NONE if nothing of its partition was exported there. + PartExportState classify(const MergeTreePartInfo & part, const String & destination) const; +}; + +using ExportFencePtr = std::shared_ptr; + +namespace ExportFenceUtils +{ + /// Sorts the ranges and merges the ones that overlap or are adjacent (`max_block + 1 == min_block`). + /// A gap between ranges is preserved: a block number in the gap may still be committed later as + /// a new part, which must not be classified as exported. + std::vector compactRanges(std::vector ranges); + + /// True if every block number of `part` is covered by the union of `ranges`. + bool isCoveredByUnion(const MergeTreePartInfo & part, const std::vector & ranges); + + /// True if `part` overlaps at least one of `ranges`. + bool intersectsAny(const MergeTreePartInfo & part, const std::vector & ranges); +} + +} diff --git a/src/Storages/MergeTree/ExportPartTask.cpp b/src/Storages/MergeTree/ExportPartTask.cpp index 2d1ac9155dbf..8ee7d54f1e9d 100644 --- a/src/Storages/MergeTree/ExportPartTask.cpp +++ b/src/Storages/MergeTree/ExportPartTask.cpp @@ -1,6 +1,6 @@ #include #include -#include +#include #include #include #include @@ -118,13 +118,9 @@ namespace void addExportConvertingActions( QueryPlan & plan_for_part, - const IStorage & destination_storage, + const Block & destination_header, const ContextPtr & local_context) { - FailPointInjection::pauseFailPoint(FailPoints::export_part_pause_before_schema_validation); - - const auto destination_metadata = destination_storage.getInMemoryMetadataPtr(local_context, false); - const auto destination_header = destination_metadata->getSampleBlockNonMaterialized(); const auto & destination_columns = destination_header.getColumnsWithTypeAndName(); const auto schema_match_mode = @@ -135,7 +131,7 @@ namespace auto source_columns = plan_for_part.getCurrentHeader()->getColumnsWithTypeAndName(); const bool src_has_extra_columns = source_columns.size() > destination_columns.size(); - ExportPartitionUtils::checkExportSchemaColumnsCount( + ExportTaskUtils::checkExportSchemaColumnsCount( source_columns.size(), destination_columns.size(), ignore_extra_source_columns); @@ -311,6 +307,8 @@ bool ExportPartTask::executeStep() const auto filename = buildDestinationFilename(manifest, storage.getStorageID(), local_context); + FailPointInjection::pauseFailPoint(FailPoints::export_part_pause_before_schema_validation); + auto import_result = destination_storage->import( filename, block_with_partition_values, @@ -378,7 +376,9 @@ bool ExportPartTask::executeStep() /// This is a hack that materializes the columns before the export so they can be exported to tables that have matching columns materializeSpecialColumns(plan_for_part.getCurrentHeader(), metadata_snapshot, local_context, plan_for_part); - addExportConvertingActions(plan_for_part, *destination_storage, local_context); + /// The destination schema may change after `import` read it, for example when a query reloads it + /// from Iceberg metadata, so convert to the header of the sink rather than to the current schema. + addExportConvertingActions(plan_for_part, sink->getHeader(), local_context); QueryPlanOptimizationSettings optimization_settings(local_context); auto pipeline_settings = BuildQueryPipelineSettings(local_context); @@ -421,11 +421,11 @@ bool ExportPartTask::executeStep() /// Commit the Iceberg metadata inline here so the rows become visible immediately. if (destination_storage->isDataLake() && !manifest.completion_callback) { - IStorage::IcebergCommitExportPartitionArguments iceberg_args; + IStorage::IcebergCommitExportArguments iceberg_args; iceberg_args.metadata_json_string = manifest.iceberg_metadata_json; iceberg_args.partition_source_block = block_with_partition_values; - destination_storage->commitExportPartitionTransaction( + destination_storage->commitExportTransaction( manifest.transaction_id, manifest.data_part->info.getPartitionId(), (*exports_list_entry)->destination_file_paths, diff --git a/src/Storages/MergeTree/ExportPartitionKey.h b/src/Storages/MergeTree/ExportPartitionKey.h deleted file mode 100644 index 504c1f2d7aed..000000000000 --- a/src/Storages/MergeTree/ExportPartitionKey.h +++ /dev/null @@ -1,22 +0,0 @@ -#pragma once - -#include -#include - -namespace DB -{ - -namespace ExportPartitionUtils -{ - -inline String compositeKey( - const String & partition_id, const String & destination_database, const String & destination_table) -{ - return escapeForFileName(partition_id) + "." - + escapeForFileName(destination_database) + "." - + escapeForFileName(destination_table); -} - -} - -} diff --git a/src/Storages/MergeTree/ExportTTLDeleteGate.h b/src/Storages/MergeTree/ExportTTLDeleteGate.h new file mode 100644 index 000000000000..36ddc7f7e00b --- /dev/null +++ b/src/Storages/MergeTree/ExportTTLDeleteGate.h @@ -0,0 +1,39 @@ +#pragma once + +#include +#include +#include + +#include +#include + +#include + +namespace DB +{ + +/// Keeps TTL that deletes or rewrites rows from applying to parts that the `EXPORT` TTL of the +/// table has not exported yet, so no row is lost before it reaches the destination. +struct ExportTTLDeleteGate +{ + bool enabled = false; + /// Key of the destination in the export index. While it is unknown, every part with due TTL is held. + String destination_key; + time_t now = 0; + + /// Returns the reason why the part must not be merged, or nothing if it may be. + std::optional check( + const String & part_name, const MergeTreePartInfo & info, const MergeTreeDataPartTTLInfos & ttl_infos, + const ExportFence * fence) const + { + if (!enabled || !ttl_infos.part_min_ttl || ttl_infos.part_min_ttl > now) + return std::nullopt; + + if (fence && !destination_key.empty() && fence->classify(info, destination_key) == PartExportState::EXPORTED) + return std::nullopt; + + return fmt::format("Part {} has TTL that is due, but the EXPORT TTL has not exported it yet", part_name); + } +}; + +} diff --git a/src/Storages/MergeTree/ExportTTLIndex.cpp b/src/Storages/MergeTree/ExportTTLIndex.cpp new file mode 100644 index 000000000000..9cbb6461b1c2 --- /dev/null +++ b/src/Storages/MergeTree/ExportTTLIndex.cpp @@ -0,0 +1,263 @@ +#include + +#include +#include +#include +#include +#include + +#include +#include + +namespace DB +{ + +namespace ErrorCodes +{ + extern const int INCORRECT_DATA; +} + +namespace +{ + +Poco::JSON::Array::Ptr rangesToJSON(const std::vector & ranges) +{ + Poco::JSON::Array::Ptr array = new Poco::JSON::Array(); + for (const auto & range : ranges) + { + Poco::JSON::Array::Ptr pair = new Poco::JSON::Array(); + pair->add(range.min_block); + pair->add(range.max_block); + array->add(pair); + } + return array; +} + +std::vector rangesFromJSON(const String & partition_id, const Poco::JSON::Array::Ptr & array) +{ + std::vector ranges; + if (!array) + return ranges; + + ranges.reserve(array->size()); + for (size_t i = 0; i < array->size(); ++i) + { + const auto pair = array->getArray(static_cast(i)); + if (!pair || pair->size() != 2) + throw Exception(ErrorCodes::INCORRECT_DATA, "Invalid block range in the export index of partition {}", partition_id); + ranges.emplace_back(partition_id, pair->getElement(0), pair->getElement(1), /* level */ 0, /* mutation */ 0); + } + return ranges; +} + +} + +Int64 ExportTTLIndexEntry::maxBlock() const +{ + Int64 result = 0; + for (const auto & range : exported) + result = std::max(result, range.max_block); + for (const auto & [_, ranges] : claimed) + for (const auto & range : ranges) + result = std::max(result, range.max_block); + return result; +} + +std::vector ExportTTLIndexEntry::allClaimed() const +{ + std::vector result; + for (const auto & [_, ranges] : claimed) + result.insert(result.end(), ranges.begin(), ranges.end()); + return ExportFenceUtils::compactRanges(std::move(result)); +} + +PartExportState ExportTTLIndexEntry::classify(const MergeTreePartInfo & part) const +{ + for (const auto & [_, ranges] : claimed) + if (ExportFenceUtils::intersectsAny(part, ranges)) + return PartExportState::CLAIMED; + if (ExportFenceUtils::intersectsAny(part, exported)) + return PartExportState::EXPORTED; + return PartExportState::NONE; +} + +ExportFenceEntry ExportTTLIndexEntry::toFenceEntry(const String & destination) const +{ + ExportFenceEntry entry; + entry.destination = destination; + entry.exported = exported; + entry.claimed = allClaimed(); + return entry; +} + +void ExportTTLIndexEntry::claim(const String & transaction_id, const std::vector & parts) +{ + auto & ranges = claimed[transaction_id]; + for (const auto & part : parts) + ranges.emplace_back(partition_id, part.min_block, part.max_block, 0, 0); + ranges = ExportFenceUtils::compactRanges(std::move(ranges)); +} + +void ExportTTLIndexEntry::moveClaims(const std::vector & from, const String & to) +{ + auto & target = claimed[to]; + for (const auto & transaction_id : from) + { + const auto it = claimed.find(transaction_id); + if (it == claimed.end() || transaction_id == to) + continue; + target.insert(target.end(), it->second.begin(), it->second.end()); + claimed.erase(it); + } + target = ExportFenceUtils::compactRanges(std::move(target)); +} + +void ExportTTLIndexEntry::commitClaim(const String & transaction_id, const std::vector & parts) +{ + if (const auto it = claimed.find(transaction_id); it != claimed.end()) + { + exported.insert(exported.end(), it->second.begin(), it->second.end()); + claimed.erase(it); + } + for (const auto & part : parts) + exported.emplace_back(partition_id, part.min_block, part.max_block, 0, 0); + exported = ExportFenceUtils::compactRanges(std::move(exported)); +} + +void ExportTTLIndexEntry::releaseClaim(const String & transaction_id) +{ + claimed.erase(transaction_id); +} + +String ExportTTLIndexEntry::toJSONString() const +{ + Poco::JSON::Object json; + json.set("exported", rangesToJSON(exported)); + + Poco::JSON::Object::Ptr claimed_object = new Poco::JSON::Object(); + for (const auto & [transaction_id, ranges] : claimed) + claimed_object->set(transaction_id, rangesToJSON(ranges)); + json.set("claimed", claimed_object); + + std::ostringstream oss; // STYLE_CHECK_ALLOW_STD_STRING_STREAM + oss.exceptions(std::ios::failbit); + Poco::JSON::Stringifier::stringify(json, oss); + return oss.str(); +} + +ExportTTLIndexEntry ExportTTLIndexEntry::fromJSONString(const String & partition_id, const String & json_string) +{ + ExportTTLIndexEntry entry; + entry.partition_id = partition_id; + if (json_string.empty()) + return entry; + + Poco::JSON::Parser parser; + const auto json = parser.parse(json_string).extract(); + if (!json) + throw Exception(ErrorCodes::INCORRECT_DATA, "The export index of partition {} is not a JSON object", partition_id); + + entry.exported = ExportFenceUtils::compactRanges(rangesFromJSON(partition_id, json->getArray("exported"))); + + if (const auto claimed_object = json->getObject("claimed")) + { + for (const auto & transaction_id : claimed_object->getNames()) + entry.claimed[transaction_id] = ExportFenceUtils::compactRanges( + rangesFromJSON(partition_id, claimed_object->getArray(transaction_id))); + } + + return entry; +} + +std::unique_ptr ExportTTLIndexSnapshot::build( + int32_t version, std::map> entries) +{ + auto fence = std::make_shared(); + for (const auto & [destination_key, by_partition] : entries) + for (const auto & [partition_id, versioned] : by_partition) + if (!versioned.entry.empty()) + fence->entries_by_partition[partition_id].push_back(versioned.entry.toFenceEntry(destination_key)); + + auto snapshot = std::make_unique(); + snapshot->version = version; + snapshot->entries = std::move(entries); + snapshot->fence = std::move(fence); + return snapshot; +} + +String ExportTTLSchedulerState::toJSONString() const +{ + Poco::JSON::Object json; + json.set("scheduler_replica", scheduler_replica); + + Poco::JSON::Object::Ptr partitions_object = new Poco::JSON::Object(); + for (const auto & [partition_id, partition] : partitions) + { + Poco::JSON::Object::Ptr partition_object = new Poco::JSON::Object(); + partition_object->set("last_error", partition.last_error); + partition_object->set("first_eligible_time", static_cast(partition.first_eligible_time)); + partition_object->set("last_new_part_time", static_cast(partition.last_new_part_time)); + partitions_object->set(partition_id, partition_object); + } + json.set("partitions", partitions_object); + + std::ostringstream oss; // STYLE_CHECK_ALLOW_STD_STRING_STREAM + oss.exceptions(std::ios::failbit); + Poco::JSON::Stringifier::stringify(json, oss); + return oss.str(); +} + +ExportTTLSchedulerState ExportTTLSchedulerState::fromJSONString(const String & json_string) +{ + ExportTTLSchedulerState state; + if (json_string.empty()) + return state; + + Poco::JSON::Parser parser; + const auto json = parser.parse(json_string).extract(); + if (!json) + throw Exception(ErrorCodes::INCORRECT_DATA, "The state of the TTL export scheduler is not a JSON object"); + + if (json->has("scheduler_replica")) + state.scheduler_replica = json->getValue("scheduler_replica"); + + if (const auto partitions_object = json->getObject("partitions")) + { + for (const auto & partition_id : partitions_object->getNames()) + { + const auto partition_object = partitions_object->getObject(partition_id); + if (!partition_object) + continue; + auto & partition = state.partitions[partition_id]; + partition.last_error = partition_object->optValue("last_error", ""); + partition.first_eligible_time = static_cast(partition_object->optValue("first_eligible_time", 0)); + partition.last_new_part_time = static_cast(partition_object->optValue("last_new_part_time", 0)); + } + } + + return state; +} + +namespace ExportTTLUtils +{ + +String destinationKey(const String & database, const String & table, const String & uuid) +{ + return escapeForFileName(database) + "." + escapeForFileName(table) + "." + (uuid.empty() ? String("none") : escapeForFileName(uuid)); +} + +std::vector rangesOfParts(const std::vector & part_names, MergeTreeDataFormatVersion format_version) +{ + std::vector ranges; + ranges.reserve(part_names.size()); + for (const auto & name : part_names) + { + const auto info = MergeTreePartInfo::fromPartName(name, format_version); + ranges.emplace_back(info.getPartitionId(), info.min_block, info.max_block, 0, 0); + } + return ExportFenceUtils::compactRanges(std::move(ranges)); +} + +} + +} diff --git a/src/Storages/MergeTree/ExportTTLIndex.h b/src/Storages/MergeTree/ExportTTLIndex.h new file mode 100644 index 000000000000..b7c247df3698 --- /dev/null +++ b/src/Storages/MergeTree/ExportTTLIndex.h @@ -0,0 +1,126 @@ +#pragma once + +#include +#include +#include + +#include +#include +#include +#include + +namespace DB +{ + +/// What a table's `TTL ... EXPORT TO TABLE` has exported to one destination from one partition. +/// +/// Rows are identified by block numbers: every insert gets its own block number, a part covers +/// exactly the rows of its block range, and merges and mutations keep the range. Ranges are +/// qualified by the partition because `ReplicatedMergeTree` allocates block numbers per partition. +/// +/// The entry is only updated together with a transition of a TTL export task, in the same Keeper +/// transaction (or under the same lock for a plain `MergeTree`), so it always agrees with what was +/// committed to the destination: +/// - creating a task claims the ranges of its parts; +/// - committing a task moves its claim to `exported`; +/// - retrying failed tasks moves their claims to the new task; +/// - resolving a failed task that does not need a retry releases its claim. +struct ExportTTLIndexEntry +{ + String partition_id; + + /// Ranges committed to the destination, compacted. + std::vector exported; + + /// Ranges owned by TTL export tasks that did not commit, by transaction id, compacted. A task + /// may be in flight, or failed and waiting to be retried. + std::map> claimed; + + bool empty() const { return exported.empty() && claimed.empty(); } + + /// Highest block number of any exported or claimed range, 0 if there is none. + Int64 maxBlock() const; + + std::vector allClaimed() const; + + PartExportState classify(const MergeTreePartInfo & part) const; + + ExportFenceEntry toFenceEntry(const String & destination) const; + + /// Adds the ranges of `parts` to the claim of `transaction_id`. + void claim(const String & transaction_id, const std::vector & parts); + + /// Moves the claims of `from` to `to`. + void moveClaims(const std::vector & from, const String & to); + + /// Moves the claim of `transaction_id` to the exported ranges, adding `parts` as well: a task + /// may commit parts whose claim was lost. + void commitClaim(const String & transaction_id, const std::vector & parts); + + void releaseClaim(const String & transaction_id); + + String toJSONString() const; + static ExportTTLIndexEntry fromJSONString(const String & partition_id, const String & json_string); +}; + +/// An index entry as read, with the version to update it with a check. +struct ExportTTLVersionedEntry +{ + ExportTTLIndexEntry entry; + /// Version of the stored entry, or -1 if it is not stored yet. + int32_t version = -1; +}; + +/// The whole export index of a table as of one version, with the merge fence built from it. +struct ExportTTLIndexSnapshot +{ + /// Version of the `export_fence` node of a `ReplicatedMergeTree` it was read at: every change of + /// the index bumps it, so the snapshot stays valid while it does not change. -1 for a plain `MergeTree`. + int32_t version = -1; + + /// By destination key, then by partition id. + std::map> entries; + + ExportFencePtr fence; + + static std::unique_ptr build( + int32_t version, std::map> entries); +}; + +using ExportTTLIndexSnapshotPtr = std::shared_ptr; + +/// What the replica that schedules the `EXPORT` TTL of a table knows beyond the index: stored with +/// the index of the destination whenever it changes, so every replica shows it, and a replica that +/// takes over the scheduling resumes the batching windows. +struct ExportTTLSchedulerState +{ + struct Partition + { + String last_error; + time_t first_eligible_time = 0; + time_t last_new_part_time = 0; + + bool operator==(const Partition &) const = default; + }; + + String scheduler_replica; + /// By partition id. A partition without an error and waiting for nothing is omitted. + std::map partitions; + + bool operator==(const ExportTTLSchedulerState &) const = default; + + String toJSONString() const; + static ExportTTLSchedulerState fromJSONString(const String & json_string); +}; + +namespace ExportTTLUtils +{ + /// Identifies a destination in the export index. The UUID, if not empty, tells apart a table that + /// was dropped and created again under the same name, which holds none of the exported rows. + String destinationKey(const String & database, const String & table, const String & uuid); + + /// Ranges of the parts named `part_names`, compacted. + std::vector rangesOfParts(const std::vector & part_names, MergeTreeDataFormatVersion format_version); +} + +} diff --git a/src/Storages/MergeTree/ExportTTLScheduler.cpp b/src/Storages/MergeTree/ExportTTLScheduler.cpp new file mode 100644 index 000000000000..82023407c474 --- /dev/null +++ b/src/Storages/MergeTree/ExportTTLScheduler.cpp @@ -0,0 +1,707 @@ +#include + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#include +#include + +#include +#include + +namespace CurrentMetrics +{ + extern const Metric ExportTTLPartsHeldByDeleteGate; +} + +namespace DB +{ + +namespace MergeTreeSetting +{ + extern const MergeTreeSettingsUInt64 ttl_export_check_period_seconds; + extern const MergeTreeSettingsUInt64 ttl_export_batch_window_seconds; + extern const MergeTreeSettingsUInt64 ttl_export_batch_max_delay_seconds; + extern const MergeTreeSettingsUInt64 ttl_export_batch_min_bytes; + extern const MergeTreeSettingsUInt64 ttl_export_max_parts_per_group; + extern const MergeTreeSettingsUInt64 ttl_export_max_bytes_per_group; + extern const MergeTreeSettingsUInt64 ttl_export_max_concurrent_groups; + extern const MergeTreeSettingsString ttl_export_settings_profile; +} + +namespace +{ + +/// The rule of move TTL: a part is eligible once the maximum TTL value of its rows is due. A part +/// without the TTL info (written before the TTL was added and not materialized) is never eligible. +bool isEligible(const IMergeTreeDataPart & part, const TTLDescriptions & export_ttls, time_t now) +{ + return selectTTLDescriptionForTTLInfos(export_ttls, part.ttl_infos.export_ttl, now, /* use_max */ true).has_value(); +} + +} + +ExportTTLScheduler::ExportTTLScheduler(MergeTreeData & storage_) + : storage(storage_) + , log(getLogger(fmt::format("{} (ExportTTLScheduler)", storage_.getLogName()))) + , parts_held_by_delete_gate(CurrentMetrics::ExportTTLPartsHeldByDeleteGate, 0) +{ +} + +String ExportTTLScheduler::getDestinationKey() const +{ + std::lock_guard lock(mutex); + return current_destination_key; +} + +std::vector ExportTTLScheduler::getInfo() const +{ + std::lock_guard lock(mutex); + std::vector result; + result.reserve(info_by_partition.size()); + for (const auto & [_, info] : info_by_partition) + { + result.push_back(info); + if (result.back().last_error.empty()) + result.back().last_error = current_destination_error; + } + return result; +} + +ContextPtr ExportTTLScheduler::makeContext() const +{ + auto context = Context::createCopy(storage.getContext()); + + const String profile = (*storage.getSettings())[MergeTreeSetting::ttl_export_settings_profile]; + if (!profile.empty()) + context->setCurrentProfile(profile); + + /// The destination was validated when the TTL was created, and may be an Iceberg table. + context->setSetting("allow_insert_into_iceberg", true); + + /// A pending mutation must not keep a part from being exported forever. + context->setSetting("export_merge_tree_part_throw_on_pending_mutations", false); + context->setSetting("export_merge_tree_part_throw_on_pending_patch_parts", false); + + return context; +} + +const ExportTTLScheduler::TaskState & ExportTTLScheduler::getCachedTaskState(TaskStates & task_states, const String & transaction_id) +{ + auto it = task_states.find(transaction_id); + if (it == task_states.end()) + it = task_states.emplace(transaction_id, getTaskState(transaction_id)).first; + return it->second; +} + +UInt64 ExportTTLScheduler::run() +{ + const auto settings = storage.getSettings(); + const time_t period = static_cast(std::max(1, (*settings)[MergeTreeSetting::ttl_export_check_period_seconds])); + + const auto metadata = storage.getInMemoryMetadataPtr(nullptr, false); + const auto export_ttls = metadata->getExportTTLs(); + + { + std::lock_guard lock(mutex); + if (!export_ttls.empty()) + may_have_index = true; + if (!may_have_index) + return period * 1000; + } + + const auto context = makeContext(); + + /// Every replica resolves the destination, because the delete gate of its merges depends on it. + StoragePtr destination; + String destination_key; + String destination_error; + if (!export_ttls.empty()) + { + const auto destination_id = storage.getExportTTLDestination(export_ttls.front()); + destination = DatabaseCatalog::instance().tryGetTable(destination_id, context); + if (destination) + { + const auto uuid = destination->getStorageID().uuid; + destination_key = ExportTTLUtils::destinationKey( + destination_id.database_name, + destination_id.table_name, + identifiesDestinationByUUID() && uuid != UUIDHelpers::Nil ? toString(uuid) : ""); + } + else + { + destination_error = fmt::format("The destination table {} of the EXPORT TTL does not exist", destination_id.getNameForLogs()); + } + } + + /// A destination that cannot be resolved may be created again or not loaded yet, so what was + /// exported to it is kept, and its parts stay held from the delete TTL. + const bool destination_missing = !export_ttls.empty() && !destination; + + { + std::lock_guard lock(mutex); + if (!destination_missing && current_destination_key != destination_key) + { + batches.clear(); + last_errors.clear(); + info_by_partition.clear(); + written_state.reset(); + was_scheduler = false; + current_destination_key = destination_key; + } + current_destination_error = destination_error; + } + + const auto snapshot = getIndexSnapshot(); + if (export_ttls.empty() && snapshot->entries.empty()) + { + std::lock_guard lock(mutex); + may_have_index = false; + batches.clear(); + last_errors.clear(); + info_by_partition.clear(); + parts_held_by_delete_gate.changeTo(0); + return period * 1000; + } + + const bool is_scheduler = acquireSchedulerLock(); + if (is_scheduler && !destination_missing) + { + for (const auto & [key, index] : snapshot->entries) + { + if (key == destination_key) + continue; + + try + { + cleanupDestination(key, index); + } + catch (...) + { + tryLogCurrentException(log, fmt::format("While removing the TTL export index of destination {}", key)); + } + } + } + + if (!destination) + { + if (is_scheduler && !destination_error.empty()) + LOG_WARNING(log, "{}, nothing is exported", destination_error); + return period * 1000; + } + + /// The replica that schedules owns the batching windows; the others show what it stored. + ExportTTLSchedulerState stored_state; + bool resume_stored_state = false; + { + bool was = false; + { + std::lock_guard lock(mutex); + was = was_scheduler; + was_scheduler = is_scheduler; + } + + if (!is_scheduler || !was) + { + if (auto state = readSchedulerState(destination_key)) + { + stored_state = std::move(*state); + resume_stored_state = is_scheduler; + } + } + } + + if (resume_stored_state) + { + std::lock_guard lock(mutex); + for (const auto & [partition_id, partition] : stored_state.partitions) + { + auto & batch = batches[partition_id]; + batch.first_eligible_time = partition.first_eligible_time; + batch.last_new_part_time = partition.last_new_part_time; + if (!partition.last_error.empty()) + last_errors[partition_id] = partition.last_error; + } + LOG_INFO(log, "This replica schedules the EXPORT TTL now, resuming the state stored by {}", stored_state.scheduler_replica); + } + + const bool act = is_scheduler && !isPaused(); + const String scheduler_replica = is_scheduler ? getReplicaName() : stored_state.scheduler_replica; + + const time_t now = time(nullptr); + + static const std::map no_entries; + const auto index_it = snapshot->entries.find(destination_key); + const auto & index = index_it == snapshot->entries.end() ? no_entries : index_it->second; + + std::map> parts_by_partition; + for (const auto & part : storage.getDataPartsVectorForInternalUsage()) + if (!part->info.isPatch()) + parts_by_partition[part->info.getPartitionId()].push_back(part); + + TaskStates task_states; + size_t in_flight = 0; + if (act) + { + for (const auto & [_, versioned] : index) + for (const auto & [transaction_id, ranges] : versioned.entry.claimed) + if (getCachedTaskState(task_states, transaction_id).status == TaskStatus::PENDING) + ++in_flight; + } + + std::set partition_ids; + for (const auto & [partition_id, _] : index) + partition_ids.insert(partition_id); + for (const auto & [partition_id, _] : parts_by_partition) + partition_ids.insert(partition_id); + + time_t next_tick = now + period; + const std::vector no_parts; + for (const auto & partition_id : partition_ids) + { + ExportTTLVersionedEntry versioned; + if (const auto it = index.find(partition_id); it != index.end()) + versioned = it->second; + else + versioned.entry.partition_id = partition_id; + + const auto parts_it = parts_by_partition.find(partition_id); + const auto & parts = parts_it == parts_by_partition.end() ? no_parts : parts_it->second; + + PartitionBatch batch; + String last_error; + if (is_scheduler) + { + std::lock_guard lock(mutex); + batch = batches[partition_id]; + if (const auto it = last_errors.find(partition_id); it != last_errors.end()) + last_error = it->second; + } + else if (const auto it = stored_state.partitions.find(partition_id); it != stored_state.partitions.end()) + { + batch.first_eligible_time = it->second.first_eligible_time; + batch.last_new_part_time = it->second.last_new_part_time; + last_error = it->second.last_error; + } + + PartitionView view; + try + { + view = observePartition(versioned.entry, parts, export_ttls, now, std::move(batch), is_scheduler, task_states); + view.info.last_error = last_error; + + if (act) + { + /// Shown as it is after acting, e.g. with the parts of a recorded commit as exported. + if (const auto updated_entry = actOnPartition(destination_key, destination, std::move(versioned), view, now, in_flight, task_states, context)) + view = observePartition(*updated_entry, parts, export_ttls, now, std::move(view.batch), is_scheduler, task_states); + view.info.last_error.clear(); + } + } + catch (...) + { + tryLogCurrentException(log, fmt::format("While exporting partition {} by TTL", partition_id)); + view.info.partition_id = partition_id; + view.info.last_error = getCurrentExceptionMessage(/* with_stacktrace */ false); + } + + view.info.destination_database = destination->getStorageID().database_name; + view.info.destination_table = destination->getStorageID().table_name; + view.info.scheduler_replica = scheduler_replica; + + if (view.info.next_group_time > now) + next_tick = std::min(next_tick, view.info.next_group_time); + if (view.next_eligible_time > now) + next_tick = std::min(next_tick, view.next_eligible_time); + + std::lock_guard lock(mutex); + if (is_scheduler) + { + batches[partition_id] = std::move(view.batch); + if (view.info.last_error.empty()) + last_errors.erase(partition_id); + else + last_errors[partition_id] = view.info.last_error; + } + info_by_partition[partition_id] = std::move(view.info); + } + + ExportTTLSchedulerState state_to_write; + bool write_state = false; + { + std::lock_guard lock(mutex); + std::erase_if(info_by_partition, [&](const auto & item) { return !partition_ids.contains(item.first); }); + std::erase_if(batches, [&](const auto & item) { return !partition_ids.contains(item.first); }); + std::erase_if(last_errors, [&](const auto & item) { return !partition_ids.contains(item.first); }); + + size_t held = 0; + for (const auto & [_, info] : info_by_partition) + held += info.parts_held_by_delete_gate; + parts_held_by_delete_gate.changeTo(held); + + if (is_scheduler) + { + state_to_write.scheduler_replica = scheduler_replica; + for (const auto & partition_id : partition_ids) + { + ExportTTLSchedulerState::Partition partition; + if (const auto it = batches.find(partition_id); it != batches.end()) + { + partition.first_eligible_time = it->second.first_eligible_time; + partition.last_new_part_time = it->second.last_new_part_time; + } + if (const auto it = last_errors.find(partition_id); it != last_errors.end()) + partition.last_error = it->second; + if (partition != ExportTTLSchedulerState::Partition{}) + state_to_write.partitions[partition_id] = std::move(partition); + } + write_state = !written_state || *written_state != state_to_write; + } + } + + if (write_state) + { + try + { + writeSchedulerState(destination_key, state_to_write); + std::lock_guard lock(mutex); + written_state = std::move(state_to_write); + } + catch (...) + { + tryLogCurrentException(log, "While storing the state of the TTL export scheduler"); + } + } + + return static_cast(std::max(1, next_tick - now)) * 1000; +} + +ExportTTLScheduler::ResolvedClaims ExportTTLScheduler::resolveClaims( + ExportTTLIndexEntry & entry, const StoragePtr & destination, TaskStates & task_states, const ContextPtr & context) +{ + ResolvedClaims result; + + std::vector transaction_ids; + for (const auto & [transaction_id, _] : entry.claimed) + transaction_ids.push_back(transaction_id); + + for (const auto & transaction_id : transaction_ids) + { + const auto & state = getCachedTaskState(task_states, transaction_id); + switch (state.status) + { + case TaskStatus::PENDING: + result.in_flight = transaction_id; + break; + + case TaskStatus::COMPLETED: + /// The commit of a plain `MergeTree` records it here, after the task is marked completed. + entry.commitClaim(transaction_id, {}); + result.changed = true; + break; + + case TaskStatus::FAILED: + case TaskStatus::KILLED: + case TaskStatus::MISSING: + /// E.g. a commit that landed and then the task timed out before it was marked completed. + if (destination->isExportTransactionCommitted(transaction_id, context)) + { + LOG_INFO(log, "Export task {} of partition {} did not complete, but it committed to the destination", + transaction_id, entry.partition_id); + entry.commitClaim(transaction_id, {}); + result.changed = true; + break; + } + + result.failed.push_back(transaction_id); + result.retry_of.insert(result.retry_of.end(), state.retry_of.begin(), state.retry_of.end()); + break; + } + } + + return result; +} + +ExportTTLScheduler::PartitionView ExportTTLScheduler::observePartition( + const ExportTTLIndexEntry & entry, + const std::vector & parts, + const TTLDescriptions & export_ttls, + time_t now, + PartitionBatch batch, + bool update_batch, + TaskStates & task_states) +{ + const auto settings = storage.getSettings(); + + PartitionView view; + auto & info = view.info; + info.partition_id = entry.partition_id; + + std::vector eligible; + for (const auto & part : parts) + { + const auto state = entry.classify(part->info); + if (state == PartExportState::EXPORTED) + { + ++info.exported_parts; + continue; + } + + if (part->ttl_infos.part_min_ttl && part->ttl_infos.part_min_ttl <= now) + ++info.parts_held_by_delete_gate; + + if (state == PartExportState::CLAIMED) + { + ++info.claimed_parts; + view.claimed_parts.push_back(part); + continue; + } + + if (part->rows_count == 0) + continue; + + if (!isEligible(*part, export_ttls, now)) + { + for (const auto & [_, ttl_info] : part->ttl_infos.export_ttl) + if (ttl_info.max > now && (!view.next_eligible_time || ttl_info.max < view.next_eligible_time)) + view.next_eligible_time = ttl_info.max; + continue; + } + + eligible.push_back(part); + if (!isPartBeingMerged(part)) + view.shippable.push_back(part); + } + + std::vector eligible_ranges; + for (const auto & part : eligible) + { + eligible_ranges.push_back(part->info); + info.eligible_bytes += part->getBytesOnDisk(); + } + eligible_ranges = ExportFenceUtils::compactRanges(std::move(eligible_ranges)); + + if (update_batch) + { + /// Resumed from the stored state: the parts seen by the previous scheduler are not new. + if (batch.seen_ranges.empty() && batch.first_eligible_time) + batch.seen_ranges = eligible_ranges; + + for (const auto & part : eligible) + { + if (!ExportFenceUtils::isCoveredByUnion(part->info, batch.seen_ranges)) + { + batch.last_new_part_time = now; + if (!batch.first_eligible_time) + batch.first_eligible_time = now; + } + } + + batch.seen_ranges = std::move(eligible_ranges); + if (eligible.empty()) + batch = PartitionBatch{}; + } + + info.eligible_parts = eligible.size(); + info.first_eligible_time = eligible.empty() ? 0 : batch.first_eligible_time; + + std::sort(view.shippable.begin(), view.shippable.end(), [](const auto & lhs, const auto & rhs) { return lhs->info.min_block < rhs->info.min_block; }); + + size_t shippable_bytes = 0; + for (const auto & part : view.shippable) + shippable_bytes += part->getBytesOnDisk(); + + if (!view.shippable.empty() && batch.first_eligible_time) + { + const auto window = static_cast((*settings)[MergeTreeSetting::ttl_export_batch_window_seconds]); + const auto max_delay = static_cast((*settings)[MergeTreeSetting::ttl_export_batch_max_delay_seconds]); + const UInt64 min_bytes = (*settings)[MergeTreeSetting::ttl_export_batch_min_bytes]; + + info.next_group_time = std::min(batch.last_new_part_time + window, batch.first_eligible_time + max_delay); + view.batch_ready = now >= info.next_group_time || (min_bytes && shippable_bytes >= min_bytes); + } + + bool retry_pending = false; + for (const auto & [transaction_id, _] : entry.claimed) + { + const auto status = getCachedTaskState(task_states, transaction_id).status; + if (status == TaskStatus::PENDING) + info.current_transaction_id = transaction_id; + else if (status != TaskStatus::COMPLETED) + retry_pending = true; + } + + if (info.current_transaction_id.empty() && (view.batch_ready || retry_pending)) + info.next_group_time = now; + + view.batch = std::move(batch); + return view; +} + +std::optional ExportTTLScheduler::actOnPartition( + const String & destination_key, + const StoragePtr & destination, + ExportTTLVersionedEntry versioned, + PartitionView & view, + time_t now, + size_t & in_flight, + TaskStates & task_states, + const ContextPtr & context) +{ + auto & entry = versioned.entry; + auto & info = view.info; + const auto & partition_id = entry.partition_id; + const auto settings = storage.getSettings(); + + auto resolved = resolveClaims(entry, destination, task_states, context); + info.current_transaction_id = resolved.in_flight; + + std::vector retry_parts; + std::unordered_set failed_with_parts; + for (const auto & part : view.claimed_parts) + { + for (const auto & transaction_id : resolved.failed) + { + if (ExportFenceUtils::intersectsAny(part->info, entry.claimed.at(transaction_id))) + { + retry_parts.push_back(part); + failed_with_parts.insert(transaction_id); + break; + } + } + } + + /// The parts of a failed task that no longer exist, e.g. were dropped, have nothing left to export. + for (const auto & transaction_id : resolved.failed) + { + if (failed_with_parts.contains(transaction_id)) + continue; + + LOG_INFO(log, "Export task {} of partition {} failed and none of its parts exists anymore, releasing its claim", + transaction_id, partition_id); + entry.releaseClaim(transaction_id); + resolved.changed = true; + } + + const bool ready = !retry_parts.empty() || view.batch_ready; + const UInt64 max_concurrent = (*settings)[MergeTreeSetting::ttl_export_max_concurrent_groups]; + + if (ready && resolved.in_flight.empty() && (!max_concurrent || in_flight < max_concurrent)) + { + info.next_group_time = now; + + const UInt64 max_parts = (*settings)[MergeTreeSetting::ttl_export_max_parts_per_group]; + const UInt64 max_bytes = (*settings)[MergeTreeSetting::ttl_export_max_bytes_per_group]; + + GroupToStart group; + group.transaction_id = toString(UUIDHelpers::generateV4()); + group.destination_key = destination_key; + group.destination = destination; + group.partition_id = partition_id; + + /// Every claimed part of a failed task is retried: a part left out would lose its claim and + /// could be exported again later although the failed task may still land. + group.parts = retry_parts; + size_t group_bytes = 0; + for (const auto & part : group.parts) + group_bytes += part->getBytesOnDisk(); + + for (const auto & transaction_id : resolved.failed) + if (failed_with_parts.contains(transaction_id)) + group.retry_of.push_back(transaction_id); + group.retry_of.insert(group.retry_of.end(), resolved.retry_of.begin(), resolved.retry_of.end()); + std::sort(group.retry_of.begin(), group.retry_of.end()); + group.retry_of.erase(std::unique(group.retry_of.begin(), group.retry_of.end()), group.retry_of.end()); + + for (const auto & part : view.shippable) + { + if (max_parts && group.parts.size() >= max_parts) + break; + if (max_bytes && !group.parts.empty() && group_bytes + part->getBytesOnDisk() > max_bytes) + break; + group.parts.push_back(part); + group_bytes += part->getBytesOnDisk(); + } + + if (!group.parts.empty()) + { + group.entry = versioned; + for (const auto & transaction_id : failed_with_parts) + group.entry.entry.releaseClaim(transaction_id); + + std::vector infos; + infos.reserve(group.parts.size()); + for (const auto & part : group.parts) + infos.push_back(part->info); + group.entry.entry.claim(group.transaction_id, infos); + + if (startGroup(group, context)) + { + ++in_flight; + task_states.insert_or_assign(group.transaction_id, TaskState{TaskStatus::PENDING, group.retry_of}); + LOG_INFO(log, "Started export task {} of {} part(s) of partition {} to {}{}", + group.transaction_id, group.parts.size(), partition_id, destination->getStorageID().getNameForLogs(), + group.retry_of.empty() ? "" : fmt::format(", retrying {}", fmt::join(group.retry_of, ", "))); + return std::move(group.entry.entry); + } + + LOG_DEBUG(log, "Merges or the export index of partition {} changed while starting an export task, will retry", partition_id); + } + } + + if (!resolved.changed) + return std::nullopt; + + if (!updateIndexEntry(destination_key, versioned)) + { + LOG_DEBUG(log, "The export index of partition {} changed concurrently, will retry", partition_id); + return std::nullopt; + } + + return std::move(entry); +} + +void ExportTTLScheduler::cleanupDestination(const String & destination_key, const std::map & index) +{ + TaskStates task_states; + bool claims_left = false; + + for (auto [partition_id, versioned] : index) + { + bool changed = false; + + std::vector transaction_ids; + for (const auto & [transaction_id, _] : versioned.entry.claimed) + transaction_ids.push_back(transaction_id); + + for (const auto & transaction_id : transaction_ids) + { + if (getCachedTaskState(task_states, transaction_id).status == TaskStatus::PENDING) + { + LOG_INFO(log, "Killing export task {}: the EXPORT TTL no longer exports to destination {}", transaction_id, destination_key); + killTask(transaction_id); + claims_left = true; + continue; + } + + versioned.entry.releaseClaim(transaction_id); + changed = true; + } + + if (changed && !updateIndexEntry(destination_key, versioned)) + claims_left = true; + } + + if (!claims_left) + removeDestination(destination_key); +} + +} diff --git a/src/Storages/MergeTree/ExportTTLScheduler.h b/src/Storages/MergeTree/ExportTTLScheduler.h new file mode 100644 index 000000000000..db88a3775819 --- /dev/null +++ b/src/Storages/MergeTree/ExportTTLScheduler.h @@ -0,0 +1,225 @@ +#pragma once + +#include +#include +#include +#include +#include +#include +#include + +#include +#include +#include +#include +#include + +namespace DB +{ + +class MergeTreeData; + +/// What `system.ttl_exports` shows about one partition of a table with a `TTL ... EXPORT` expression. +struct ExportTTLPartitionInfo +{ + String partition_id; + String destination_database; + String destination_table; + size_t exported_parts = 0; + size_t claimed_parts = 0; + size_t eligible_parts = 0; + size_t eligible_bytes = 0; + /// Parts held back from the delete TTL because they are not exported yet. + size_t parts_held_by_delete_gate = 0; + /// When the scheduler first saw an eligible part that is not exported, 0 if there is none. + time_t first_eligible_time = 0; + /// When the next group can start at the latest, 0 if nothing waits. + time_t next_group_time = 0; + /// Transaction id of the task exporting the partition now, empty if there is none. + String current_transaction_id; + String last_error; + /// The replica that schedules the TTL exports of the table, empty for a plain `MergeTree`. + String scheduler_replica; +}; + +/// The background task of a table with a `TTL EXPORT TO TABLE ` expression. +/// +/// On every tick it ships groups of eligible parts (the maximum TTL value of their rows is due) to +/// the destination as export tasks of the table, one partition per group, and never exports a part +/// twice: what was exported is recorded in the export index (see `ExportTTLIndexEntry`), and the +/// merge fence keeps parts in different export states from merging. +/// +/// Eligible parts of a partition are shipped once no new eligible part appeared for the batching +/// window, once the first of them waited for the maximum delay, or once they reach the size +/// threshold. A task that failed without committing keeps its parts claimed, and they are retried +/// first by the next group, which records the failed tasks in `retry_of`. +/// +/// Every replica observes the state of every partition, which `system.ttl_exports` shows. Only +/// the replica holding the scheduler lock acts on it: it resolves finished tasks, starts groups and +/// stores its state (`ExportTTLSchedulerState`) for the other replicas. +/// +/// Engines implement access to the index and to their export tasks. +class ExportTTLScheduler +{ +public: + explicit ExportTTLScheduler(MergeTreeData & storage_); + virtual ~ExportTTLScheduler() = default; + + /// One tick. Returns the number of milliseconds until the next one. + UInt64 run(); + + std::vector getInfo() const; + + /// Key in the export index of the current destination, empty if it is not known yet. + String getDestinationKey() const; + +protected: + enum class TaskStatus : UInt8 + { + PENDING, + COMPLETED, + FAILED, + KILLED, + /// No such task, e.g. a crash between claiming its parts and creating it. + MISSING, + }; + + struct TaskState + { + TaskStatus status = TaskStatus::MISSING; + std::vector retry_of; + }; + + struct GroupToStart + { + String transaction_id; + String destination_key; + StoragePtr destination; + String partition_id; + std::vector parts; + std::vector retry_of; + /// The index entry with the claim of the group, to be stored with a check of its version. + ExportTTLVersionedEntry entry; + }; + + /// Only one replica schedules, which keeps the replicas from conflicting on the index. The lock + /// is kept until it is lost or the table shuts down. Correctness does not depend on it. + virtual bool acquireSchedulerLock() = 0; + + /// E.g. `SYSTEM STOP MOVES`. + virtual bool isPaused() = 0; + + virtual ExportTTLIndexSnapshotPtr getIndexSnapshot() = 0; + + /// Whether the key of the destination in the index contains its UUID, which tells apart a table + /// that was dropped and created again. Replicas have different UUIDs for the same destination + /// unless its database is `Replicated`, so they identify it by name only. + virtual bool identifiesDestinationByUUID() const = 0; + + /// The state stored by the replica that schedules, nothing if there is none. + virtual std::optional readSchedulerState(const String & destination_key) = 0; + virtual void writeSchedulerState(const String & destination_key, const ExportTTLSchedulerState & state) = 0; + virtual String getReplicaName() const = 0; + + virtual TaskState getTaskState(const String & transaction_id) = 0; + + /// Stores `entry` with a check of its version. Returns false on a conflict. + virtual bool updateIndexEntry(const String & destination_key, const ExportTTLVersionedEntry & entry) = 0; + + /// Creates the export task of `group` and stores its index entry. Returns false on a conflict, + /// e.g. a merge was assigned or the index changed meanwhile. + virtual bool startGroup(const GroupToStart & group, const ContextPtr & context) = 0; + + /// Whether the part is a source of an assigned merge that changes its block range. + virtual bool isPartBeingMerged(const MergeTreeDataPartPtr & part) = 0; + + virtual void killTask(const String & transaction_id) = 0; + virtual void removeDestination(const String & destination_key) = 0; + + MergeTreeData & storage; + const LoggerPtr log; + +private: + struct PartitionBatch + { + /// Block ranges of the eligible parts already seen, so a merge of seen parts is not new. + std::vector seen_ranges; + time_t first_eligible_time = 0; + time_t last_new_part_time = 0; + }; + + /// What every replica can tell about a partition without acting on it. + struct PartitionView + { + ExportTTLPartitionInfo info; + PartitionBatch batch; + std::vector claimed_parts; + /// Eligible parts that are neither exported nor claimed nor being merged, by block number. + std::vector shippable; + bool batch_ready = false; + /// When a part that is not due yet becomes eligible, 0 if there is none. + time_t next_eligible_time = 0; + }; + + using TaskStates = std::unordered_map; + + mutable std::mutex mutex; + /// By partition id, for the current destination. Kept by the replica that schedules. + std::map batches; + std::map last_errors; + std::map info_by_partition; + String current_destination_key; + String current_destination_error; + + /// Whether the previous tick scheduled, so a replica that takes over resumes the stored state. + bool was_scheduler = false; + std::optional written_state; + + /// False once there is neither an `EXPORT` TTL nor an index left, so ticks do no Keeper reads. + bool may_have_index = true; + + CurrentMetrics::Increment parts_held_by_delete_gate; + + ContextPtr makeContext() const; + + /// Resolves the claims of the index entry that do not belong to a task in flight. Returns the + /// failed tasks whose parts are still claimed, with their own `retry_of`, and the task in flight. + struct ResolvedClaims + { + std::vector failed; + std::vector retry_of; + String in_flight; + bool changed = false; + }; + ResolvedClaims resolveClaims(ExportTTLIndexEntry & entry, const StoragePtr & destination, TaskStates & task_states, const ContextPtr & context); + + const TaskState & getCachedTaskState(TaskStates & task_states, const String & transaction_id); + + /// Kills the TTL tasks of a destination that is no longer the destination of the TTL, and + /// removes its index once none of them holds a claim. + void cleanupDestination(const String & destination_key, const std::map & index); + + /// Read-only. `update_batch` is set on the replica that schedules, which owns the batching windows. + PartitionView observePartition( + const ExportTTLIndexEntry & entry, + const std::vector & parts, + const TTLDescriptions & export_ttls, + time_t now, + PartitionBatch batch, + bool update_batch, + TaskStates & task_states); + + /// On the replica that schedules: resolves finished tasks and starts a group if one is due. + /// Returns the index entry of the partition if it was changed. + std::optional actOnPartition( + const String & destination_key, + const StoragePtr & destination, + ExportTTLVersionedEntry versioned, + PartitionView & view, + time_t now, + size_t & in_flight, + TaskStates & task_states, + const ContextPtr & context); +}; + +} diff --git a/src/Storages/MergeTree/PartitionExportInfo.h b/src/Storages/MergeTree/ExportTaskInfo.h similarity index 89% rename from src/Storages/MergeTree/PartitionExportInfo.h rename to src/Storages/MergeTree/ExportTaskInfo.h index 29355bf5f163..9310dddb87a3 100644 --- a/src/Storages/MergeTree/PartitionExportInfo.h +++ b/src/Storages/MergeTree/ExportTaskInfo.h @@ -8,7 +8,7 @@ namespace DB { -struct PartitionExportInfo +struct ExportTaskInfo { /// Most recent exception recorded for this task by one replica. A plain `MergeTree` reports a /// single entry with an empty `replica`. `count` is best-effort: concurrent failing writers on @@ -62,6 +62,11 @@ struct PartitionExportInfo /// Parts of this task currently backing off on this node. Empty if none. std::vector backoff_per_part; + + /// What created the task: `query` for `EXPORT PARTITION`, `ttl` for a `TTL ... EXPORT` expression. + String source; + /// TTL export only: earlier tasks that failed to export some of this task's parts. + std::vector retry_of; }; } diff --git a/src/Storages/MergeTree/ExportPartitionUtils.cpp b/src/Storages/MergeTree/ExportTaskUtils.cpp similarity index 74% rename from src/Storages/MergeTree/ExportPartitionUtils.cpp rename to src/Storages/MergeTree/ExportTaskUtils.cpp index 5fe114dc14d1..f7ea51db14f0 100644 --- a/src/Storages/MergeTree/ExportPartitionUtils.cpp +++ b/src/Storages/MergeTree/ExportTaskUtils.cpp @@ -1,13 +1,15 @@ -#include +#include #include #include #include #include #include -#include "Storages/ExportReplicatedMergeTreePartitionManifest.h" -#include "Storages/ExportReplicatedMergeTreePartitionTaskEntry.h" +#include "Storages/ExportReplicatedMergeTreeTaskManifest.h" +#include "Storages/ExportReplicatedMergeTreeTaskEntry.h" #include -#include +#include +#include +#include #if USE_AVRO #include #include @@ -51,12 +53,12 @@ namespace ProfileEvents { - extern const Event ExportPartitionZooKeeperRequests; - extern const Event ExportPartitionZooKeeperGet; - extern const Event ExportPartitionZooKeeperGetChildren; - extern const Event ExportPartitionZooKeeperSet; - extern const Event ExportPartitionZooKeeperCreate; - extern const Event ExportPartitionZooKeeperMulti; + extern const Event ExportTaskZooKeeperRequests; + extern const Event ExportTaskZooKeeperGet; + extern const Event ExportTaskZooKeeperGetChildren; + extern const Event ExportTaskZooKeeperSet; + extern const Event ExportTaskZooKeeperCreate; + extern const Event ExportTaskZooKeeperMulti; } namespace DB @@ -120,7 +122,7 @@ namespace FailPoints namespace fs = std::filesystem; -namespace ExportPartitionUtils +namespace ExportTaskUtils { bool isNonRetryableExportError(int code) { @@ -206,6 +208,13 @@ namespace ExportPartitionUtils return static_cast(now) - static_cast(create_time) > timeout_seconds; } + String getPartitionIdOfParts(const std::vector & part_names, MergeTreeDataFormatVersion format_version) + { + if (part_names.empty()) + throw Exception(ErrorCodes::LOGICAL_ERROR, "An export task has no parts, cannot determine its partition"); + return MergeTreePartInfo::fromPartName(part_names.front(), format_version).getPartitionId(); + } + Block getPartitionSourceBlockForIcebergCommit( MergeTreeData & storage, const String & partition_id, const std::vector & exported_part_names) { @@ -317,25 +326,16 @@ namespace ExportPartitionUtils return context_copy; } - template ContextPtr getContextCopyWithTaskSettings( - const ContextPtr &, const ExportReplicatedMergeTreePartitionManifest &); - template ContextPtr getContextCopyWithTaskSettings( - const ContextPtr &, const MergeTreePartitionExportTask &); + template ContextPtr getContextCopyWithTaskSettings( + const ContextPtr &, const ExportReplicatedMergeTreeTaskManifest &); + template ContextPtr getContextCopyWithTaskSettings( + const ContextPtr &, const MergeTreeExportTask &); #if USE_AVRO - std::string verifyAndExtractDestinationIcebergMetadataJson( - const StorageMetadataPtr & source_metadata, - const StorageMetadataPtr & destination_metadata, - const StoragePtr & dest_storage, - const MergeTreeData::DataPartsVector & parts, - const String & partition_id, - const ContextPtr & context) +namespace +{ + Poco::JSON::Object::Ptr getIcebergMetadataObject(const StoragePtr & dest_storage, const ContextPtr & context) { - if (!context->getSettingsRef()[Setting::allow_insert_into_iceberg]) - throw Exception(ErrorCodes::SUPPORT_IS_DISABLED, - "Iceberg writes are experimental. " - "To allow its usage, enable the setting `allow_insert_into_iceberg` on the initiator (query, session or profile) - replicas inherit it from the scheduled task."); - auto * object_storage = dynamic_cast(dest_storage.get()); auto * object_storage_cluster = dynamic_cast(dest_storage.get()); @@ -351,7 +351,24 @@ namespace ExportPartitionUtils if (!iceberg_metadata) throw Exception(ErrorCodes::BAD_ARGUMENTS, "Destination storage {} is a data lake but not an iceberg table", dest_storage->getName()); - const auto metadata_object = iceberg_metadata->getMetadataJSON(context); + return iceberg_metadata->getMetadataJSON(context); + } +} + + std::string verifyAndExtractDestinationIcebergMetadataJson( + const StorageMetadataPtr & source_metadata, + const StorageMetadataPtr & destination_metadata, + const StoragePtr & dest_storage, + const MergeTreeData::DataPartsVector & parts, + const String & partition_id, + const ContextPtr & context) + { + if (!context->getSettingsRef()[Setting::allow_insert_into_iceberg]) + throw Exception(ErrorCodes::SUPPORT_IS_DISABLED, + "Iceberg writes are experimental. " + "To allow its usage, enable the setting `allow_insert_into_iceberg` on the initiator (query, session or profile) - replicas inherit it from the scheduled task."); + + const auto metadata_object = getIcebergMetadataObject(dest_storage, context); verifyIcebergPartitionCompatibility( metadata_object, source_metadata, destination_metadata, parts, partition_id, context); @@ -363,7 +380,7 @@ namespace ExportPartitionUtils } #endif - IStorage::ExportPartitionCommitInfo commitExportOnDestination( + IStorage::ExportCommitInfo commitExportOnDestination( const String & transaction_id, const String & partition_id, const String & iceberg_metadata_json, @@ -381,7 +398,7 @@ namespace ExportPartitionUtils if (iceberg_partition_timezone) context->setSetting("iceberg_partition_timezone", *iceberg_partition_timezone); - IStorage::IcebergCommitExportPartitionArguments iceberg_args; + IStorage::IcebergCommitExportArguments iceberg_args; if (!iceberg_metadata_json.empty()) { @@ -392,7 +409,7 @@ namespace ExportPartitionUtils getPartitionSourceBlockForIcebergCommit(source_storage, partition_id, exported_part_names); } - return destination_storage->commitExportPartitionTransaction( + return destination_storage->commitExportTransaction( transaction_id, partition_id, exported_paths, iceberg_args, context); } @@ -401,12 +418,12 @@ namespace ExportPartitionUtils /// Otherwise, multiple async requests are sent ExportedPaths getExportedPaths(const LoggerPtr & log, const zkutil::ZooKeeperPtr & zk, const std::string & export_path) { - LOG_DEBUG(log, "ExportPartition: Getting exported paths for {}", export_path); + LOG_DEBUG(log, "Export task: Getting exported paths for {}", export_path); const auto processed_parts_path = fs::path(export_path) / "processed"; - ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRequests); - ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperGetChildren); + ProfileEvents::increment(ProfileEvents::ExportTaskZooKeeperRequests); + ProfileEvents::increment(ProfileEvents::ExportTaskZooKeeperGetChildren); std::vector processed_parts; if (const auto code = zk->tryGetChildren(processed_parts_path, processed_parts); code != Coordination::Error::ZOK) throw Coordination::Exception::fromPath(code, processed_parts_path); @@ -420,8 +437,8 @@ namespace ExportPartitionUtils } auto responses = zk->tryGet(get_paths); - ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRequests); - ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperGet, get_paths.size()); + ProfileEvents::increment(ProfileEvents::ExportTaskZooKeeperRequests); + ProfileEvents::increment(ProfileEvents::ExportTaskZooKeeperGet, get_paths.size()); responses.waitForResponses(); @@ -433,26 +450,89 @@ namespace ExportPartitionUtils if (responses[i].error != Coordination::Error::ZOK) throw Coordination::Exception::fromPath(responses[i].error, get_paths[i]); - const auto processed_part_entry = ExportReplicatedMergeTreePartitionProcessedPartEntry::fromJsonString(responses[i].data); + const auto processed_part_entry = ExportReplicatedMergeTreeProcessedPartEntry::fromJsonString(responses[i].data); for (const auto & path_in_destination : processed_part_entry.paths_in_destination) { result.paths.emplace_back(path_in_destination); } + result.paths_by_part[processed_parts[i]] = processed_part_entry.paths_in_destination; + } + + return result; + } + + void appendCreateExportTaskOps(Coordination::Requests & ops, const std::string & task_path, const ExportReplicatedMergeTreeTaskManifest & manifest) + { + const fs::path path = task_path; + + ops.emplace_back(zkutil::makeCreateRequest(path, "", zkutil::CreateMode::Persistent)); + ops.emplace_back(zkutil::makeCreateRequest(path / "metadata.json", manifest.toJsonString(), zkutil::CreateMode::Persistent)); + + /// Container for per-replica last_exception leaves; children are created lazily by the + /// first writer per replica (see appendExceptionOps). + ops.emplace_back(zkutil::makeCreateRequest(path / "last_exception", "", zkutil::CreateMode::Persistent)); + + ops.emplace_back(zkutil::makeCreateRequest(path / "processing", "", zkutil::CreateMode::Persistent)); + for (const auto & part : manifest.parts) + { + ExportReplicatedMergeTreeProcessingPartEntry entry; + entry.status = ExportReplicatedMergeTreeProcessingPartEntry::Status::PENDING; + entry.part_name = part; + ops.emplace_back(zkutil::makeCreateRequest(path / "processing" / part, entry.toJsonString(), zkutil::CreateMode::Persistent)); } + ops.emplace_back(zkutil::makeCreateRequest(path / "processed", "", zkutil::CreateMode::Persistent)); + ops.emplace_back(zkutil::makeCreateRequest(path / "locks", "", zkutil::CreateMode::Persistent)); + ops.emplace_back(zkutil::makeCreateRequest(path / "status", "PENDING", zkutil::CreateMode::Persistent)); + } + + std::vector getRangesCommittedByRetriedTasks( + const std::vector & retry_of, + const StoragePtr & destination_storage, + const std::function>(const String &)> & get_task_parts, + MergeTreeDataFormatVersion format_version, + const ContextPtr & context) + { + std::vector ranges; + for (const auto & transaction_id : retry_of) + { + if (!destination_storage->isExportTransactionCommitted(transaction_id, context)) + continue; + + const auto parts = get_task_parts(transaction_id); + if (!parts) + throw Exception(ErrorCodes::CORRUPTED_DATA, + "Export task {} committed to the destination, but its description is lost, so it is not known which parts it exported. " + "Refusing to commit the task that retries it, which could export them again", + transaction_id); + + const auto task_ranges = ExportTTLUtils::rangesOfParts(*parts, format_version); + ranges.insert(ranges.end(), task_ranges.begin(), task_ranges.end()); + } + return ExportFenceUtils::compactRanges(std::move(ranges)); + } + + std::vector getPartsNotCommitted( + const std::vector & part_names, const std::vector & committed_ranges, MergeTreeDataFormatVersion format_version) + { + std::vector result; + for (const auto & part_name : part_names) + if (!ExportFenceUtils::isCoveredByUnion(MergeTreePartInfo::fromPartName(part_name, format_version), committed_ranges)) + result.push_back(part_name); return result; } void commit( - const ExportReplicatedMergeTreePartitionManifest & manifest, + const ExportReplicatedMergeTreeTaskManifest & manifest, const StoragePtr & destination_storage, const zkutil::ZooKeeperPtr & zk, const LoggerPtr & log, const std::string & entry_path, const ContextPtr & context_in, MergeTreeData & source_storage, - const String & replica_name) + const String & replica_name, + const ReplicatedExportTTLIndex & export_fence) { /// Failpoint used by integration tests to force persistent commit failure and exercise /// the commit-attempts budget / FAILED state transition. @@ -464,54 +544,90 @@ namespace ExportPartitionUtils /// Per-task ephemeral lock that serializes the commit phase across replicas. /// Without it, `handlePartExportSuccess` (post-last-part path) and `tryCleanup` - /// (poll/recovery path) can drive `commitExportPartitionTransaction` concurrently + /// (poll/recovery path) can drive `commitExportTransaction` concurrently /// for the same task. const auto commit_lock_path = fs::path(entry_path) / "commit_lock"; auto commit_lock = zkutil::EphemeralNodeHolder::tryCreate(commit_lock_path, *zk, replica_name); if (!commit_lock) { - LOG_DEBUG(log, "ExportPartition: commit_lock for {} is held by another replica, skipping commit on this replica", entry_path); + LOG_DEBUG(log, "Export task: commit_lock for {} is held by another replica, skipping commit on this replica", entry_path); return; } - LOG_INFO(log, "ExportPartition: commit_lock for {} acquired by replica {}", entry_path, replica_name); + LOG_INFO(log, "Export task: commit_lock for {} acquired by replica {}", entry_path, replica_name); - /// Honor a concurrent KILL: commit_lock serializes us against killExportPartition, + /// Honor a concurrent KILL: commit_lock serializes us against killExportTask, /// so a non-PENDING status here means cancel won the race. std::string status_str; if (!zk->tryGet(fs::path(entry_path) / "status", status_str)) return; - const auto status = magic_enum::enum_cast(status_str); - if (!status || *status != ExportReplicatedMergeTreePartitionTaskEntry::Status::PENDING) + const auto status = magic_enum::enum_cast(status_str); + if (!status || *status != ExportReplicatedMergeTreeTaskEntry::Status::PENDING) { - LOG_DEBUG(log, "ExportPartition: {} not PENDING, skipping commit", entry_path); + LOG_DEBUG(log, "Export task: {} not PENDING, skipping commit", entry_path); return; } - const auto exported = ExportPartitionUtils::getExportedPaths(log, zk, entry_path); + const auto exported = ExportTaskUtils::getExportedPaths(log, zk, entry_path); if (exported.processed_parts_count < manifest.parts.size()) { throw Exception(ErrorCodes::CORRUPTED_DATA, - "ExportPartition: Reached the commit phase, but only {} of {} parts are marked as processed, " + "Export task: Reached the commit phase, but only {} of {} parts are marked as processed, " "will not commit export. This might be a bug", exported.processed_parts_count, manifest.parts.size()); } - IStorage::ExportPartitionCommitInfo destination_commit_info; + const auto partition_id = getPartitionIdOfParts(manifest.parts, source_storage.format_version); + + /// A task this one retries may have landed after it was considered failed. Its parts are + /// then in the destination already, so only the files of the other parts are committed. + std::vector paths_to_commit = exported.paths; + if (!manifest.retry_of.empty()) + { + const auto exports_path = fs::path(entry_path).parent_path(); + const auto committed_ranges = getRangesCommittedByRetriedTasks( + manifest.retry_of, + destination_storage, + [&](const String & transaction_id) -> std::optional> + { + String metadata_json; + if (!zk->tryGet(exports_path / transaction_id / "metadata.json", metadata_json)) + return std::nullopt; + return ExportReplicatedMergeTreeTaskManifest::fromJsonString(metadata_json).parts; + }, + source_storage.format_version, + context_in); + + if (!committed_ranges.empty()) + { + paths_to_commit.clear(); + for (const auto & part_name : getPartsNotCommitted(manifest.parts, committed_ranges, source_storage.format_version)) + { + const auto it = exported.paths_by_part.find(part_name); + if (it != exported.paths_by_part.end()) + paths_to_commit.insert(paths_to_commit.end(), it->second.begin(), it->second.end()); + } + + LOG_INFO(log, "Export task: a task retried by {} committed some of its parts, committing {} of {} files", + entry_path, paths_to_commit.size(), exported.paths.size()); + } + } + + IStorage::ExportCommitInfo destination_commit_info; - if (exported.paths.empty()) + if (paths_to_commit.empty()) { - LOG_INFO(log, "ExportPartition: {} produced no destination files, nothing to commit", entry_path); + LOG_INFO(log, "Export task: {} has no destination files to commit", entry_path); } else { destination_commit_info = commitExportOnDestination( manifest.transaction_id, - manifest.partition_id, + partition_id, manifest.iceberg_metadata_json, manifest.write_full_path_in_iceberg_metadata, manifest.iceberg_partition_timezone, - exported.paths, + paths_to_commit, manifest.parts, destination_storage, source_storage, @@ -528,42 +644,75 @@ namespace ExportPartitionUtils "Failpoint: simulating crash after Iceberg commit, before ZK COMPLETED"); }); - LOG_INFO(log, "ExportPartition: Committed export, mark as completed"); + LOG_INFO(log, "Export task: Committed export, mark as completed"); const std::string status_path = fs::path(entry_path) / "status"; - const std::string completed_name = String(magic_enum::enum_name(ExportReplicatedMergeTreePartitionTaskEntry::Status::COMPLETED)).data(); + const std::string completed_name = String(magic_enum::enum_name(ExportReplicatedMergeTreeTaskEntry::Status::COMPLETED)).data(); - Coordination::Requests ops; - ops.emplace_back(zkutil::makeSetRequest(status_path, completed_name, -1)); - - ExportPartitionCommitInfoEntry commit_info_entry { + ExportCommitInfoEntry commit_info_entry { destination_commit_info.iceberg_metadata_file, destination_commit_info.iceberg_manifest_list, destination_commit_info.iceberg_manifest_file, destination_commit_info.commit_marker_file}; const std::string commit_info_path = fs::path(entry_path) / "commit_info"; - ops.emplace_back(zkutil::makeCreateRequest(commit_info_path, commit_info_entry.toJsonString(), zkutil::CreateMode::Persistent)); - ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRequests); - ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperMulti); + const bool is_ttl_task = manifest.source == ExportTaskSource::ttl; + /// Replicas identify the destination by name, see `ReplicatedExportTTLScheduler`. + const auto destination_key = ExportTTLUtils::destinationKey(manifest.destination_database, manifest.destination_table, ""); - Coordination::Responses responses; - const auto rc = zk->tryMulti(ops, responses); - - if (rc == Coordination::Error::ZOK) + /// The index entry of a TTL task is stored with a check of its version, because the TTL + /// scheduler may change it meanwhile, e.g. when it resolves another claim of the partition. + static constexpr size_t max_attempts = 10; + for (size_t attempt = 0; attempt < max_attempts; ++attempt) { - LOG_INFO(log, "ExportPartition: Marked export as completed and persisted commit_info"); - return; - } + Coordination::Requests ops; + ops.emplace_back(zkutil::makeSetRequest(status_path, completed_name, -1)); + ops.emplace_back(zkutil::makeCreateRequest(commit_info_path, commit_info_entry.toJsonString(), zkutil::CreateMode::Persistent)); - if (rc == Coordination::Error::ZNODEEXISTS) - { - LOG_INFO(log, "ExportPartition: commit_info already present (peer wrote it first); task already COMPLETED"); - return; + if (is_ttl_task) + { + auto versioned = export_fence.readIndexEntry(zk, destination_key, partition_id); + if (versioned.version < 0) + export_fence.ensureDestination(zk, destination_key, fmt::format("{}.{}", manifest.destination_database, manifest.destination_table)); + + versioned.entry.commitClaim(manifest.transaction_id, ExportTTLUtils::rangesOfParts(manifest.parts, source_storage.format_version)); + for (const auto & transaction_id : manifest.retry_of) + versioned.entry.releaseClaim(transaction_id); + export_fence.appendUpdateEntryOps(ops, destination_key, versioned.entry, versioned.version); + } + + ProfileEvents::increment(ProfileEvents::ExportTaskZooKeeperRequests); + ProfileEvents::increment(ProfileEvents::ExportTaskZooKeeperMulti); + + Coordination::Responses responses; + const auto rc = zk->tryMulti(ops, responses); + + if (rc == Coordination::Error::ZOK) + { + LOG_INFO(log, "Export task: Marked export as completed and persisted commit_info"); + return; + } + + const auto failed_op = zkutil::getFailedOpIndex(rc, responses); + + if (rc == Coordination::Error::ZNODEEXISTS && failed_op == 1) + { + LOG_INFO(log, "Export task: commit_info already present (peer wrote it first); task already COMPLETED"); + return; + } + + if (is_ttl_task && failed_op >= 2 && (rc == Coordination::Error::ZBADVERSION || rc == Coordination::Error::ZNODEEXISTS)) + { + LOG_DEBUG(log, "Export task: the export index changed while completing {}, retrying", entry_path); + continue; + } + + throw Exception(ErrorCodes::NETWORK_ERROR, "Export task: Failed to mark export as completed (rc={}), will not try to fix it", rc); } - throw Exception(ErrorCodes::NETWORK_ERROR, "ExportPartition: Failed to mark export as completed (rc={}), will not try to fix it", rc); + throw Exception(ErrorCodes::NETWORK_ERROR, + "Export task: the export index kept changing while completing {}, will retry the commit", entry_path); } bool handleCommitFailure( @@ -584,28 +733,28 @@ namespace ExportPartitionUtils Coordination::Stat status_stat; std::string current_status; - ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRequests); - ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperGet); + ProfileEvents::increment(ProfileEvents::ExportTaskZooKeeperRequests); + ProfileEvents::increment(ProfileEvents::ExportTaskZooKeeperGet); if (!zk->tryGet(status_path, current_status, &status_stat)) { /// Task was removed (TTL cleanup or force-overwrite). Nothing to do. - LOG_DEBUG(log, "ExportPartition: /status missing for {}, skipping commit-failure bookkeeping", entry_path); + LOG_DEBUG(log, "Export task: /status missing for {}, skipping commit-failure bookkeeping", entry_path); return false; } - const auto status = magic_enum::enum_cast(current_status); + const auto status = magic_enum::enum_cast(current_status); if (!status) { - LOG_WARNING(log, "ExportPartition: Invalid status {} for task {}, skipping commit-failure bookkeeping", current_status, entry_path); + LOG_WARNING(log, "Export task: Invalid status {} for task {}, skipping commit-failure bookkeeping", current_status, entry_path); return false; } - if (status != ExportReplicatedMergeTreePartitionTaskEntry::Status::PENDING) + if (status != ExportReplicatedMergeTreeTaskEntry::Status::PENDING) { /// Another replica already reached a terminal state (COMPLETED or FAILED). /// Do NOT overwrite — a successful commit by a peer must win. LOG_DEBUG(log, - "ExportPartition: /status for {} is {} (not PENDING), skipping commit-failure bookkeeping", + "Export task: /status for {} is {} (not PENDING), skipping commit-failure bookkeeping", entry_path, current_status); return false; } @@ -627,22 +776,22 @@ namespace ExportPartitionUtils /// ZBADVERSION and we safely do nothing — the winning terminal state stands. ops.emplace_back(zkutil::makeSetRequest( status_path, - String(magic_enum::enum_name(ExportReplicatedMergeTreePartitionTaskEntry::Status::FAILED)).data(), + String(magic_enum::enum_name(ExportReplicatedMergeTreeTaskEntry::Status::FAILED)).data(), status_stat.version)); } - ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRequests); - ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperMulti); + ProfileEvents::increment(ProfileEvents::ExportTaskZooKeeperRequests); + ProfileEvents::increment(ProfileEvents::ExportTaskZooKeeperMulti); Coordination::Responses responses; const auto rc = zk->tryMulti(ops, responses); if (rc != Coordination::Error::ZOK) { - LOG_WARNING(log, "ExportPartition: Failed to persist commit failure bookkeeping for {}: {}", entry_path, rc); + LOG_WARNING(log, "Export task: Failed to persist commit failure bookkeeping for {}: {}", entry_path, rc); return false; } LOG_INFO(log, - "ExportPartition: Commit failure recorded for {} (code {}){}", + "Export task: Commit failure recorded for {} (code {}){}", entry_path, exception_code, non_retryable ? ", task transitioned to FAILED (non-retryable)" : ", will retry until task timeout"); @@ -668,8 +817,8 @@ namespace ExportPartitionUtils LastExceptionEntry entry; std::string current_data; - ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRequests); - ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperGet); + ProfileEvents::increment(ProfileEvents::ExportTaskZooKeeperRequests); + ProfileEvents::increment(ProfileEvents::ExportTaskZooKeeperGet); const bool leaf_exists = zk->tryGet(last_exception_path, current_data); if (leaf_exists) { @@ -679,7 +828,7 @@ namespace ExportPartitionUtils } catch (...) { - LOG_WARNING(log, "ExportPartition: last_exception JSON at {} is malformed, resetting", last_exception_path.string()); + LOG_WARNING(log, "Export task: last_exception JSON at {} is malformed, resetting", last_exception_path.string()); entry = LastExceptionEntry{}; } } @@ -698,11 +847,11 @@ namespace ExportPartitionUtils /// enclosing multis would then abort with ZNODEEXISTS and roll back its own part /// lock removal, stranding that part behind its ephemeral lock until session loss /// or task timeout. A peer thread winning this create (ZNODEEXISTS) is benign. - ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRequests); - ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperCreate); + ProfileEvents::increment(ProfileEvents::ExportTaskZooKeeperRequests); + ProfileEvents::increment(ProfileEvents::ExportTaskZooKeeperCreate); const auto create_code = zk->tryCreate(last_exception_path, entry.toJsonString(), zkutil::CreateMode::Persistent); if (create_code != Coordination::Error::ZOK && create_code != Coordination::Error::ZNODEEXISTS) - LOG_INFO(log, "ExportPartition: could not pre-create last_exception leaf {}: {}", last_exception_path.string(), create_code); + LOG_INFO(log, "Export task: could not pre-create last_exception leaf {}: {}", last_exception_path.string(), create_code); } /// Always a version -1 Set: it can neither conflict with a peer's create nor abort the @@ -745,11 +894,12 @@ namespace } /// Dynamically verifies the destination expression maps to a single partition by checking its monotonicity over the source range. + /// Without `minmax`, only checks that it could: the expression is monotonic in a single column of the source partition key. void verifyOutputMapsToSinglePartition( const ActionsDAG::Node * destination_output, const Names & minmax_column_names, const DataTypes & minmax_column_types, - const IMergeTreeDataPart::MinMaxIndex & minmax, + const IMergeTreeDataPart::MinMaxIndex * minmax, const String & partition_id, const ContextPtr & context) { @@ -776,34 +926,42 @@ namespace "Cannot export partition: column '{}' is Nullable, so a NULL forms a separate destination " "partition; partition the source by the matching destination partition expression.", column); - if (!minmax.initialized || slot >= minmax.hyperrectangle.size()) - throw Exception(ErrorCodes::BAD_ARGUMENTS, - "Cannot export partition: no min/max statistics available for column '{}' in partition " - "'{}'; cannot validate partitioning.", column, partition_id); - const auto & min_value = minmax.hyperrectangle[slot].left; - const auto & max_value = minmax.hyperrectangle[slot].right; - const auto & destination_type = chain.input_node->result_type; - - /// If the types are not the same, we need to check if the cast is monotonic - if (!isSameTypeForPartitioning(source_type, destination_type)) + const bool needs_cast = !isSameTypeForPartitioning(source_type, destination_type); + FunctionBasePtr cast_function; + if (needs_cast) { - const auto cast_function - = createInternalCast({source_type, column}, destination_type, CastType::nonAccurate, {}, context); - if (!cast_function->hasInformationAboutMonotonicity() - || !cast_function->getMonotonicityForRange(*source_type, min_value, max_value).is_monotonic) + cast_function = createInternalCast({source_type, column}, destination_type, CastType::nonAccurate, {}, context); + if (!cast_function->hasInformationAboutMonotonicity()) throw Exception(ErrorCodes::BAD_ARGUMENTS, - "Cannot export partition '{}': values of column '{}' cross a non-monotonic cast boundary to " - "the destination type {}, so it spans multiple destination partitions.", - partition_id, column, destination_type->getName()); + "Cannot export partition: the cast of column '{}' to the destination type {} has no known " + "monotonicity, so it cannot be proven that a source partition maps to a single destination partition.", + column, destination_type->getName()); } if (!isMonotonicChain(destination_output, chain)) throw Exception(ErrorCodes::BAD_ARGUMENTS, - "Cannot export partition '{}': the destination partition expression '{}' is not monotonic in " + "Cannot export partition: the destination partition expression '{}' is not monotonic in " "column '{}' (a hash such as icebergBucket never is), so its values at the endpoints of the " "partition do not bound the rows in between.", - partition_id, destination_output->result_name, column); + destination_output->result_name, column); + + if (!minmax) + return; + + if (!minmax->initialized || slot >= minmax->hyperrectangle.size()) + throw Exception(ErrorCodes::BAD_ARGUMENTS, + "Cannot export partition: no min/max statistics available for column '{}' in partition " + "'{}'; cannot validate partitioning.", column, partition_id); + const auto & min_value = minmax->hyperrectangle[slot].left; + const auto & max_value = minmax->hyperrectangle[slot].right; + + /// If the types are not the same, we need to check if the cast is monotonic + if (needs_cast && !cast_function->getMonotonicityForRange(*source_type, min_value, max_value).is_monotonic) + throw Exception(ErrorCodes::BAD_ARGUMENTS, + "Cannot export partition '{}': values of column '{}' cross a non-monotonic cast boundary to " + "the destination type {}, so it spans multiple destination partitions.", + partition_id, column, destination_type->getName()); auto endpoints = source_type->createColumn(); endpoints->insert(min_value); @@ -828,10 +986,12 @@ namespace /// A source partition is not split in the destination when every destination partition expression is /// single-valued over it. That holds structurally when the expression is a deterministic function of the /// source partition key, because rows agreeing on the source key then agree on it as well; the remaining - /// expressions have to be proven from the partition's min/max values. + /// expressions have to be proven from the partition's min/max values. Without `parts`, only checks that + /// such a proof is possible. void verifyPartitionKeyCompatibility( const KeyDescription & source_key, const KeyDescription & destination_key, + const MergeTreeSettingsPtr & source_settings, const MergeTreeData::DataPartsVector & parts, const String & partition_id, const ContextPtr & context) @@ -855,7 +1015,7 @@ namespace const auto matches = matchTrees(source_dag.getOutputs(), destination_dag); const auto minmax_columns = MergeTreeData::getMinMaxColumns( - source_key, parts.front()->storage.getSettings(), MergeTreePartMinMaxIndexColumns::PARTITION_KEY_ONLY); + source_key, source_settings, MergeTreePartMinMaxIndexColumns::PARTITION_KEY_ONLY); const auto minmax_column_names = minmax_columns.getNames(); const auto minmax_column_types = minmax_columns.getTypes(); @@ -863,6 +1023,7 @@ namespace IMergeTreeDataPart::MinMaxIndex minmax; for (const auto & part : parts) minmax.merge(*part->getMinMaxIndex()); + const auto * minmax_of_parts = parts.empty() ? nullptr : &minmax; /* 1. If there is a structural match between the source and destination key, we accept it @@ -876,18 +1037,18 @@ namespace continue; verifyOutputMapsToSinglePartition( - destination_output, minmax_column_names, minmax_column_types, minmax, partition_id, context); + destination_output, minmax_column_names, minmax_column_types, minmax_of_parts, partition_id, context); } } } #if USE_AVRO - void verifyIcebergPartitionCompatibility( +namespace +{ + /// The partition spec of an Iceberg destination as a ClickHouse partition key, nothing if it is unpartitioned. + std::optional getIcebergDestinationPartitionKey( const Poco::JSON::Object::Ptr & metadata_object, - const StorageMetadataPtr & source_metadata, const StorageMetadataPtr & destination_metadata, - const MergeTreeData::DataPartsVector & parts, - const String & partition_id, const ContextPtr & context) { const auto original_schema_id = metadata_object->getValue(Iceberg::f_current_schema_id); @@ -940,7 +1101,7 @@ namespace const auto spec_fields = partition_spec_json->getArray(Iceberg::f_fields); const UInt32 spec_size = spec_fields ? static_cast(spec_fields->size()) : 0; if (spec_size == 0) - return; + return std::nullopt; /// Rebuild the destination spec as a ClickHouse partition key, the way the Iceberg read path does in /// ManifestFileIterator, so the same compatibility rule applies as for a plain object storage @@ -975,10 +1136,21 @@ namespace const auto destination_columns = ColumnsDescription::fromNamesAndTypes( destination_metadata->getSampleBlockNonMaterialized().getNamesAndTypes()); - verifyPartitionKeyCompatibility( - source_metadata->getPartitionKey(), - KeyDescription::getKeyFromAST(partition_key_ast, destination_columns, /*virtuals=*/ {}, context), - parts, partition_id, context); + return KeyDescription::getKeyFromAST(partition_key_ast, destination_columns, /*virtuals=*/ {}, context); + } +} + + void verifyIcebergPartitionCompatibility( + const Poco::JSON::Object::Ptr & metadata_object, + const StorageMetadataPtr & source_metadata, + const StorageMetadataPtr & destination_metadata, + const MergeTreeData::DataPartsVector & parts, + const String & partition_id, + const ContextPtr & context) + { + if (const auto destination_key = getIcebergDestinationPartitionKey(metadata_object, destination_metadata, context)) + verifyPartitionKeyCompatibility( + source_metadata->getPartitionKey(), *destination_key, parts.front()->storage.getSettings(), parts, partition_id, context); } #endif @@ -990,7 +1162,29 @@ namespace const ContextPtr & context) { verifyPartitionKeyCompatibility( - source_metadata->getPartitionKey(), destination_metadata->getPartitionKey(), parts, partition_id, context); + source_metadata->getPartitionKey(), destination_metadata->getPartitionKey(), parts.front()->storage.getSettings(), + parts, partition_id, context); + } + + void verifyPartitionKeyCanBeCompatible( + const StorageMetadataPtr & source_metadata, + const StorageMetadataPtr & destination_metadata, + const StoragePtr & destination_storage, + const MergeTreeSettingsPtr & source_settings, + const ContextPtr & context) + { + if (destination_storage->isDataLake()) + { +#if USE_AVRO + const auto metadata_object = getIcebergMetadataObject(destination_storage, context); + if (const auto destination_key = getIcebergDestinationPartitionKey(metadata_object, destination_metadata, context)) + verifyPartitionKeyCompatibility(source_metadata->getPartitionKey(), *destination_key, source_settings, {}, "", context); +#endif + return; + } + + verifyPartitionKeyCompatibility( + source_metadata->getPartitionKey(), destination_metadata->getPartitionKey(), source_settings, {}, "", context); } namespace @@ -1048,12 +1242,26 @@ namespace return true; } + /// `canBeSafelyCast` only widens numbers, but every `Date` fits in a `Date32` and every `DateTime` in a + /// `DateTime64` of any scale. Iceberg stores them as `date` and `timestamp`, so an Iceberg table declares + /// `Date32` and `DateTime64(6)` columns once it reloads its schema from its metadata. + bool isDateOrTimeWidening(const DataTypePtr & source_type, const DataTypePtr & destination_type) + { + if (isNullableOrLowCardinalityNullable(source_type) && !isNullableOrLowCardinalityNullable(destination_type)) + return false; + + const auto source = removeNullable(removeLowCardinality(source_type)); + const auto destination = removeNullable(removeLowCardinality(destination_type)); + return (isDate(source) && isDate32(destination)) || (isDateTime(source) && isDateTime64(destination)); + } + void verifyExportColumnCastIsSafe( const ColumnWithTypeAndName & source_column, const ColumnWithTypeAndName & destination_column, const StorageID & destination_storage_id) { - if (canBeSafelyCast(source_column.type, destination_column.type)) + if (canBeSafelyCast(source_column.type, destination_column.type) + || isDateOrTimeWidening(source_column.type, destination_column.type)) return; throw Exception(ErrorCodes::INCOMPATIBLE_COLUMNS, @@ -1152,7 +1360,7 @@ namespace if (ActionsDAG::MatchColumnsMode::Position == mode && ignore_extra_source_columns && src_has_extra_columns) { LOG_DEBUG( - getLogger("ExportPartitionUtils"), + getLogger("ExportTaskUtils"), "Source has {} columns while destination has {} columns, " "the {} extra trailing source column(s) will be ignored", source_columns.size(), diff --git a/src/Storages/MergeTree/ExportPartitionUtils.h b/src/Storages/MergeTree/ExportTaskUtils.h similarity index 70% rename from src/Storages/MergeTree/ExportPartitionUtils.h rename to src/Storages/MergeTree/ExportTaskUtils.h index ba347d79b6a4..82f25888dc61 100644 --- a/src/Storages/MergeTree/ExportPartitionUtils.h +++ b/src/Storages/MergeTree/ExportTaskUtils.h @@ -2,6 +2,8 @@ #include #include +#include +#include #include #include #include @@ -24,9 +26,10 @@ namespace DB { class MergeTreeData; -struct ExportReplicatedMergeTreePartitionManifest; +class ReplicatedExportTTLIndex; +struct ExportReplicatedMergeTreeTaskManifest; -namespace ExportPartitionUtils +namespace ExportTaskUtils { bool isNonRetryableExportError(int code); @@ -36,6 +39,9 @@ namespace ExportPartitionUtils bool isExportTaskTimedOut(time_t create_time, size_t timeout_seconds, time_t now); + /// The partition of an export task, derived from the names of its parts, which all belong to one partition. + String getPartitionIdOfParts(const std::vector & part_names, MergeTreeDataFormatVersion format_version); + struct ExportedPaths { /// Number of `/processed` leaves, that is, parts this export has finished. @@ -44,14 +50,34 @@ namespace ExportPartitionUtils /// Destination paths recorded by those leaves, flattened. A leaf may carry none, so this /// can legitimately be shorter than `processed_parts_count`. std::vector paths; + + /// The same paths by the name of the part they were exported from. + std::map> paths_by_part; }; + /// Appends the ops that create the nodes of a new export task at `task_path` (`/exports/`). + void appendCreateExportTaskOps(Coordination::Requests & ops, const std::string & task_path, const ExportReplicatedMergeTreeTaskManifest & manifest); + + /// Block ranges committed to `destination_storage` by the tasks in `retry_of` that landed, e.g. a + /// commit that the destination applied after the task was considered failed. `get_task_parts` + /// returns the parts of a task, or nothing if it is unknown. + std::vector getRangesCommittedByRetriedTasks( + const std::vector & retry_of, + const StoragePtr & destination_storage, + const std::function>(const String &)> & get_task_parts, + MergeTreeDataFormatVersion format_version, + const ContextPtr & context); + + /// Parts of `part_names` whose rows are not in `committed_ranges`. + std::vector getPartsNotCommitted( + const std::vector & part_names, const std::vector & committed_ranges, MergeTreeDataFormatVersion format_version); + /// Reads the destination paths recorded under `/processed`. ExportedPaths getExportedPaths(const LoggerPtr & log, const zkutil::ZooKeeperPtr & zk, const std::string & export_path); /// Build a query context carrying the export task's persisted settings. Templated on the /// descriptor type so it serves both the replicated manifest (backed by ZooKeeper) and the - /// plain `MergeTreePartitionExportTask` (backed by disk); both expose the same setting fields. + /// plain `MergeTreeExportTask` (backed by disk); both expose the same setting fields. template ContextPtr getContextCopyWithTaskSettings(const ContextPtr & context, const ManifestT & manifest); @@ -65,9 +91,9 @@ namespace ExportPartitionUtils const ContextPtr & context); #endif - /// Invokes `commitExportPartitionTransaction` on the destination storage. Does not mark the export + /// Invokes `commitExportTransaction` on the destination storage. Does not mark the export /// task complete; the caller persists the returned commit info. - IStorage::ExportPartitionCommitInfo commitExportOnDestination( + IStorage::ExportCommitInfo commitExportOnDestination( const String & transaction_id, const String & partition_id, const String & iceberg_metadata_json, @@ -83,15 +109,19 @@ namespace ExportPartitionUtils Block getPartitionSourceBlockForIcebergCommit( MergeTreeData & storage, const String & partition_id, const std::vector & exported_part_names); + /// Commits the files of a replicated export task and marks it completed. For a task of the + /// `EXPORT` TTL, the files of parts that a task it retries committed are left out, and the + /// export index records the parts as exported in the same transaction. void commit( - const ExportReplicatedMergeTreePartitionManifest & manifest, + const ExportReplicatedMergeTreeTaskManifest & manifest, const StoragePtr & destination_storage, const zkutil::ZooKeeperPtr & zk, const LoggerPtr & log, const std::string & entry_path, const ContextPtr & context, MergeTreeData & source_storage, - const String & replica_name + const String & replica_name, + const ReplicatedExportTTLIndex & export_fence ); /// Handles a commit-phase failure for a replicated partition export: @@ -158,6 +188,17 @@ namespace ExportPartitionUtils const String & partition_id, const ContextPtr & context); + /// Throws if no group of parts of the source could be exported to the destination without being split + /// across its partitions, checking what does not need the parts: every destination partition expression + /// must match the source partition key structurally, or be monotonic in a single non-Nullable column of + /// it, so that the min/max values of the parts can prove it when they are exported. + void verifyPartitionKeyCanBeCompatible( + const StorageMetadataPtr & source_metadata, + const StorageMetadataPtr & destination_metadata, + const StoragePtr & destination_storage, + const MergeTreeSettingsPtr & source_settings, + const ContextPtr & context); + #if USE_AVRO /// Verifies the source MergeTree partition key is compatible with the destination Iceberg /// partition spec: every destination partition field must be single-valued across the exported diff --git a/src/Storages/MergeTree/IMergeTreeDataPart.cpp b/src/Storages/MergeTree/IMergeTreeDataPart.cpp index 8d7125b8b81f..3e66b5340128 100644 --- a/src/Storages/MergeTree/IMergeTreeDataPart.cpp +++ b/src/Storages/MergeTree/IMergeTreeDataPart.cpp @@ -3021,6 +3021,12 @@ bool IMergeTreeDataPart::checkAllTTLCalculated(const StorageMetadataPtr & metada return false; } + for (const auto & export_desc : metadata_snapshot->getExportTTLs()) + { + if (!ttl_infos.export_ttl.contains(export_desc.result_column)) + return false; + } + for (const auto & group_by_desc : metadata_snapshot->getGroupByTTLs()) { if (!ttl_infos.group_by_ttl.contains(group_by_desc.result_column)) diff --git a/src/Storages/MergeTree/MergeTreeData.cpp b/src/Storages/MergeTree/MergeTreeData.cpp index f0fb344d129c..512c4aba97e0 100644 --- a/src/Storages/MergeTree/MergeTreeData.cpp +++ b/src/Storages/MergeTree/MergeTreeData.cpp @@ -22,7 +22,8 @@ #include #include #include -#include +#include +#include #include #include #include @@ -233,6 +234,7 @@ namespace DB namespace Setting { extern const SettingsBool allow_drop_detached; + extern const SettingsBool allow_experimental_export_ttl; extern const SettingsBool allow_experimental_analyzer; extern const SettingsBool enable_full_text_index; extern const SettingsBool allow_non_metadata_alters; @@ -360,6 +362,7 @@ namespace MergeTreeSetting namespace ServerSetting { + extern const ServerSettingsBool allow_experimental_export_merge_tree_partition; extern const ServerSettingsDouble mark_cache_prewarm_ratio; extern const ServerSettingsDouble primary_index_cache_prewarm_ratio; extern const ServerSettingsDouble index_mark_cache_prewarm_ratio; @@ -1262,6 +1265,109 @@ void MergeTreeData::checkProperties( } checkKeyExpression(*new_sorting_key.expression, new_sorting_key.sample_block, "Sorting", allow_nullable_key_); + + /// A replica applying an ALTER of another replica has no query context, and the ALTER was already validated. + if (!attach && local_context) + checkExportTTL(new_metadata, old_metadata, local_context); +} + +StorageID MergeTreeData::getExportTTLDestination(const TTLDescription & export_ttl) const +{ + return getExportTTLDestination(getStorageID(), export_ttl); +} + +StorageID MergeTreeData::getExportTTLDestination(const StorageID & table_id, const TTLDescription & export_ttl) +{ + return StorageID( + export_ttl.destination_database.empty() ? table_id.database_name : export_ttl.destination_database, + export_ttl.destination_name); +} + +ExportTTLDeleteGate MergeTreeData::getExportTTLDeleteGate() const +{ + ExportTTLDeleteGate gate; + const auto metadata = getInMemoryMetadataPtr(nullptr, false); + gate.enabled = metadata->hasAnyExportTTL(); + if (gate.enabled && export_ttl_scheduler) + gate.destination_key = export_ttl_scheduler->getDestinationKey(); + gate.now = time(nullptr); + return gate; +} + +std::vector MergeTreeData::getExportTTLInfo() const +{ + if (!export_ttl_scheduler) + return {}; + return export_ttl_scheduler->getInfo(); +} + +void MergeTreeData::checkExportTTL( + const StorageInMemoryMetadata & new_metadata, const StorageInMemoryMetadata & old_metadata, ContextPtr local_context) const +{ + const auto export_ttls = new_metadata.getExportTTLs(); + if (export_ttls.empty()) + return; + + const auto describe = [](const TTLDescriptions & ttls) -> String + { + if (ttls.empty()) + return {}; + return ttls.front().expression_ast->formatWithSecretsOneLine() + " " + ttls.front().destination_database + "." + ttls.front().destination_name; + }; + + /// An ALTER that does not change the export TTL does not revalidate it. + if (describe(export_ttls) == describe(old_metadata.getExportTTLs())) + return; + + validateExportTTL(getStorageID(), new_metadata, getSettings(), local_context); +} + +void MergeTreeData::validateExportTTL( + const StorageID & table_id, const StorageInMemoryMetadata & new_metadata, const MergeTreeSettingsPtr & settings, ContextPtr local_context) +{ + const auto export_ttls = new_metadata.getExportTTLs(); + if (export_ttls.empty()) + return; + + if (!local_context->getSettingsRef()[Setting::allow_experimental_export_ttl]) + throw Exception(ErrorCodes::SUPPORT_IS_DISABLED, + "`TTL ... EXPORT TO TABLE` is experimental. Set `allow_experimental_export_ttl` to enable it"); + + if (!local_context->getServerSettings()[ServerSetting::allow_experimental_export_merge_tree_partition]) + throw Exception(ErrorCodes::SUPPORT_IS_DISABLED, + "`TTL ... EXPORT TO TABLE` requires the server setting `allow_experimental_export_merge_tree_partition`"); + + const auto destination_id = getExportTTLDestination(table_id, export_ttls.front()); + if (destination_id.database_name == table_id.database_name && destination_id.table_name == table_id.table_name) + throw Exception(ErrorCodes::BAD_TTL_EXPRESSION, "A table cannot be the destination of its own EXPORT TTL"); + + const auto destination = DatabaseCatalog::instance().tryGetTable(destination_id, local_context); + if (!destination) + throw Exception(ErrorCodes::BAD_TTL_EXPRESSION, "The destination table {} of the EXPORT TTL does not exist", destination_id.getNameForLogs()); + + if (destination->getStorageID().database_name == table_id.database_name && destination->getStorageID().table_name == table_id.table_name) + throw Exception(ErrorCodes::BAD_TTL_EXPRESSION, "A table cannot be the destination of its own EXPORT TTL"); + + if (!destination->supportsImport(local_context)) + throw Exception(ErrorCodes::BAD_TTL_EXPRESSION, + "The destination table {} of the EXPORT TTL must be an Iceberg or object storage table that supports importing MergeTree parts", + destination_id.getNameForLogs()); + + const auto source_metadata = std::make_shared(new_metadata); + const auto destination_metadata = destination->getInMemoryMetadataPtr(local_context, false); + ExportTaskUtils::verifyExportSchemaCastable(source_metadata, destination_metadata, destination->getStorageID(), local_context); + + /// Whether the rows of a group of parts land in a single destination partition can only be proven + /// from the parts, which is done when the group is exported. What can never be proven is refused here. + try + { + ExportTaskUtils::verifyPartitionKeyCanBeCompatible(source_metadata, destination_metadata, destination, settings, local_context); + } + catch (Exception & e) + { + e.addMessage("while checking the partition key of the destination {} of the EXPORT TTL", destination_id.getNameForLogs()); + throw; + } } void MergeTreeData::checkMetadataProperties( @@ -7255,7 +7361,7 @@ void MergeTreeData::exportPartToTable( "To allow its usage, enable the setting `allow_insert_into_iceberg`."); } - ExportPartitionUtils::verifyExportSchemaCastable( + ExportTaskUtils::verifyExportSchemaCastable( source_metadata_ptr, destination_metadata_ptr, dest_storage->getStorageID(), query_context); auto part = getPartIfExists(part_name, {MergeTreeDataPartState::Active, MergeTreeDataPartState::Outdated}); @@ -7300,7 +7406,7 @@ void MergeTreeData::exportPartToTable( metadata_object->stringify(oss); iceberg_metadata_json = oss.str(); - ExportPartitionUtils::verifyIcebergPartitionCompatibility( + ExportTaskUtils::verifyIcebergPartitionCompatibility( metadata_object, source_metadata_ptr, destination_metadata_ptr, @@ -7318,7 +7424,7 @@ void MergeTreeData::exportPartToTable( /// Plain (hive) object storage writes every row of the part to the one directory computed from /// the destination PARTITION BY on the part's min row, so the source partition must map to a /// single destination partition. Equivalent or finer source keys are accepted. - ExportPartitionUtils::verifyPlainPartitionCompatibility( + ExportTaskUtils::verifyPlainPartitionCompatibility( source_metadata_ptr, destination_metadata_ptr, {part}, diff --git a/src/Storages/MergeTree/MergeTreeData.h b/src/Storages/MergeTree/MergeTreeData.h index a9e1b556f591..578ea56fcfbe 100644 --- a/src/Storages/MergeTree/MergeTreeData.h +++ b/src/Storages/MergeTree/MergeTreeData.h @@ -17,6 +17,7 @@ #include #include #include +#include #include #include #include @@ -41,7 +42,7 @@ #include #include #include -#include +#include #include #include @@ -51,6 +52,9 @@ namespace DB { +class ExportTTLScheduler; +struct ExportTTLPartitionInfo; + /// Number of streams is not number parts, but number or parts*files, hence 100. const size_t DEFAULT_DELAYED_STREAMS_FOR_PARALLEL_WRITE = 100; @@ -1112,9 +1116,9 @@ class MergeTreeData : public WithMutableContext, public IStorage, public IBackgr throw Exception(ErrorCodes::NOT_IMPLEMENTED, "EXPORT PARTITION is not implemented for engine {}", getName()); } - /// Snapshot of this table's partition-export tasks for `system.partition_exports`, taken from + /// Snapshot of this table's partition-export tasks for `system.distributed_exports`, taken from /// an in-memory mirror: no disk or ZooKeeper I/O, so it is safe to call from query threads. - virtual std::vector getPartitionExportsInfo() const { return {}; } + virtual std::vector getExportTasksInfo() const { return {}; } /// Checks that Partition could be dropped right now /// Otherwise - throws an exception with detailed information. @@ -1519,6 +1523,7 @@ class MergeTreeData : public WithMutableContext, public IStorage, public IBackgr friend class VersionMetadataOnKeeper; // for access to log friend class MutationsState; // for access to log friend class ExportPartTask; + friend class ExportTTLScheduler; bool require_part_metadata; @@ -1758,6 +1763,33 @@ class MergeTreeData : public WithMutableContext, public IStorage, public IBackgr void checkTTLExpressions(const StorageInMemoryMetadata & new_metadata, const StorageInMemoryMetadata & old_metadata) const; + /// Validates the `TTL ... EXPORT TO TABLE` expression of an ALTER, if it changes it. + void checkExportTTL(const StorageInMemoryMetadata & new_metadata, const StorageInMemoryMetadata & old_metadata, ContextPtr local_context) const; + +public: + /// Validates the `TTL ... EXPORT TO TABLE` expression of the table `table_id` being created or + /// altered. The partition key of the destination is checked against the parts of every group + /// when it is exported. + static void validateExportTTL( + const StorageID & table_id, const StorageInMemoryMetadata & new_metadata, const MergeTreeSettingsPtr & settings, ContextPtr local_context); + + /// The destination of an `EXPORT` TTL: an unqualified table name refers to this table's database. + StorageID getExportTTLDestination(const TTLDescription & export_ttl) const; + static StorageID getExportTTLDestination(const StorageID & table_id, const TTLDescription & export_ttl); + + ExportTTLDeleteGate getExportTTLDeleteGate() const; + + /// The export states of parts as last seen, without reading Keeper. May be older than the + /// index, which only makes the delete gate hold more parts. + virtual ExportFencePtr getLatestExportFence() const { return nullptr; } + + /// For `system.ttl_exports`, empty if the table has no `EXPORT` TTL scheduler. + std::vector getExportTTLInfo() const; + +protected: + /// Created by the engines that support `TTL ... EXPORT`. + std::shared_ptr export_ttl_scheduler; + void checkStoragePolicy(const StoragePolicyPtr & new_storage_policy) const; /// Calculates column and secondary indexes sizes in compressed form for the current state of data_parts. Call with data_parts mutex under lock. diff --git a/src/Storages/MergeTree/MergeTreeDataPartTTLInfo.cpp b/src/Storages/MergeTree/MergeTreeDataPartTTLInfo.cpp index b897ebf5576c..3b694b24baee 100644 --- a/src/Storages/MergeTree/MergeTreeDataPartTTLInfo.cpp +++ b/src/Storages/MergeTree/MergeTreeDataPartTTLInfo.cpp @@ -60,6 +60,9 @@ void MergeTreeDataPartTTLInfos::update(const MergeTreeDataPartTTLInfos & other_i for (const auto & [expression, ttl_info] : other_infos.moves_ttl) moves_ttl[expression].update(ttl_info); + for (const auto & [expression, ttl_info] : other_infos.export_ttl) + export_ttl[expression].update(ttl_info); + table_ttl.update(other_infos.table_ttl); updatePartMinMaxTTL(table_ttl); } @@ -126,6 +129,11 @@ void MergeTreeDataPartTTLInfos::read(ReadBuffer & in) const JSON & moves = json["moves"]; fill_ttl_info_map(moves, moves_ttl, false); } + if (json.has("export")) + { + const JSON & exports = json["export"]; + fill_ttl_info_map(exports, export_ttl, false); + } if (json.has("recompression")) { const JSON & recompressions = json["recompression"]; @@ -213,6 +221,12 @@ void MergeTreeDataPartTTLInfos::write(WriteBuffer & out) const is_first = false; } + if (!export_ttl.empty()) + { + write_infos(export_ttl, "export", is_first); + is_first = false; + } + if (!recompression_ttl.empty()) { write_infos(recompression_ttl, "recompression", is_first); @@ -267,6 +281,9 @@ bool MergeTreeDataPartTTLInfos::hasAnyNonFinishedTTLs() const if (has_non_finished_ttl(moves_ttl)) return true; + if (has_non_finished_ttl(export_ttl)) + return true; + if (has_non_finished_ttl(recompression_ttl)) return true; diff --git a/src/Storages/MergeTree/MergeTreeDataPartTTLInfo.h b/src/Storages/MergeTree/MergeTreeDataPartTTLInfo.h index 06611a9154ed..2a479adb544f 100644 --- a/src/Storages/MergeTree/MergeTreeDataPartTTLInfo.h +++ b/src/Storages/MergeTree/MergeTreeDataPartTTLInfo.h @@ -50,6 +50,10 @@ struct MergeTreeDataPartTTLInfos TTLInfoMap moves_ttl; + /// `TTL ... EXPORT TO TABLE`. A part becomes eligible for export once `max` has passed. + /// Like moves, it does not affect `part_min_ttl`. + TTLInfoMap export_ttl; + TTLInfoMap recompression_ttl; TTLInfoMap group_by_ttl; @@ -79,7 +83,8 @@ struct MergeTreeDataPartTTLInfos bool empty() const { /// part_min_ttl in minimum of rows, rows_where and group_by TTLs - return !part_min_ttl && moves_ttl.empty() && recompression_ttl.empty() && columns_ttl.empty() && rows_where_ttl.empty() && group_by_ttl.empty(); + return !part_min_ttl && moves_ttl.empty() && export_ttl.empty() && recompression_ttl.empty() && columns_ttl.empty() + && rows_where_ttl.empty() && group_by_ttl.empty(); } }; diff --git a/src/Storages/MergeTree/MergeTreeDataWriter.cpp b/src/Storages/MergeTree/MergeTreeDataWriter.cpp index d183cef2ca65..2c11b1ccf985 100644 --- a/src/Storages/MergeTree/MergeTreeDataWriter.cpp +++ b/src/Storages/MergeTree/MergeTreeDataWriter.cpp @@ -953,6 +953,9 @@ MergeTreeTemporaryPartPtr MergeTreeDataWriter::writeTempPartImpl( for (const auto & ttl_entry : recompression_ttl_entries) updateTTL(context, ttl_entry, new_data_part->ttl_infos, new_data_part->ttl_infos.recompression_ttl[ttl_entry.result_column], block, false); + for (const auto & ttl_entry : metadata_snapshot->getExportTTLs()) + updateTTL(context, ttl_entry, new_data_part->ttl_infos, new_data_part->ttl_infos.export_ttl[ttl_entry.result_column], block, false); + new_data_part->ttl_infos.update(move_ttl_infos); /// partition / ttl_infos / minmax (and patch source parts) are built above outside any diff --git a/src/Storages/MergeTree/MergeTreeExportTTLScheduler.cpp b/src/Storages/MergeTree/MergeTreeExportTTLScheduler.cpp new file mode 100644 index 000000000000..ee41e7edaea3 --- /dev/null +++ b/src/Storages/MergeTree/MergeTreeExportTTLScheduler.cpp @@ -0,0 +1,243 @@ +#include + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#include + +namespace fs = std::filesystem; + +namespace ProfileEvents +{ + extern const Event ExportTTLIndexSnapshotRefreshes; +} + +namespace DB +{ + +namespace ErrorCodes +{ + extern const int INCORRECT_DATA; +} + +MergeTreeExportTTLIndex::MergeTreeExportTTLIndex(StorageMergeTree & storage_) + : storage(storage_) + , snapshot(ExportTTLIndexSnapshot::build(/* version */ -1, {})) +{ +} + +String MergeTreeExportTTLIndex::getRootPath() const +{ + return fs::path(storage.getRelativeDataPath()) / "export_ttl"; +} + +String MergeTreeExportTTLIndex::getDestinationPath(const String & destination_key) const +{ + return fs::path(getRootPath()) / destination_key; +} + +String MergeTreeExportTTLIndex::getEntryPath(const String & destination_key, const String & partition_id) const +{ + return fs::path(getDestinationPath(destination_key)) / "partitions" / (escapeForFileName(partition_id) + ".json"); +} + +void MergeTreeExportTTLIndex::load() +{ + auto disk = storage.getDisks().front(); + const auto root_path = getRootPath(); + + std::lock_guard lock(mutex); + entries.clear(); + + if (disk->existsDirectory(root_path)) + { + for (auto destination_it = disk->iterateDirectory(root_path); destination_it->isValid(); destination_it->next()) + { + const auto destination_key = destination_it->name(); + auto & by_partition = entries[destination_key]; + + const auto partitions_path = fs::path(root_path) / destination_key / "partitions"; + if (!disk->existsDirectory(partitions_path)) + continue; + + for (auto it = disk->iterateDirectory(partitions_path); it->isValid(); it->next()) + { + const auto file_name = it->name(); + /// Skip temporary files left by an interrupted write. + if (!file_name.ends_with(".json")) + continue; + + const auto partition_id = unescapeForFileName(file_name.substr(0, file_name.size() - strlen(".json"))); + const auto path = fs::path(partitions_path) / file_name; + + /// Losing an entry would export its parts again, so an unreadable one fails the table load. + String content; + try + { + auto in = disk->readFile(path, getReadSettings()); + readStringUntilEOF(content, *in); + by_partition[partition_id] = ExportTTLVersionedEntry{ + .entry = ExportTTLIndexEntry::fromJSONString(partition_id, content), + .version = 0, + }; + } + catch (Exception & e) + { + e.addMessage("While loading the TTL export index entry {}", path.string()); + throw; + } + catch (...) + { + throw Exception(ErrorCodes::INCORRECT_DATA, "Cannot load the TTL export index entry {}: {}", + path.string(), getCurrentExceptionMessage(false)); + } + } + } + } + + publishSnapshot(); +} + +bool MergeTreeExportTTLIndex::update(const String & destination_key, const ExportTTLVersionedEntry & versioned) +{ + std::lock_guard lock(mutex); + + auto & by_partition = entries[destination_key]; + const auto it = by_partition.find(versioned.entry.partition_id); + const int32_t stored_version = it == by_partition.end() ? -1 : it->second.version; + if (stored_version != versioned.version) + return false; + + auto disk = storage.getDisks().front(); + const auto path = getEntryPath(destination_key, versioned.entry.partition_id); + const auto tmp_path = path + ".tmp"; + disk->createDirectories(fs::path(path).parent_path()); + + { + auto out = disk->writeFile(tmp_path, DBMS_DEFAULT_BUFFER_SIZE, WriteMode::Rewrite, storage.getContext()->getWriteSettings()); + writeString(versioned.entry.toJSONString(), *out); + out->finalize(); + /// The entry must survive a crash before the task it claims for is created or completed. + out->sync(); + } + disk->replaceFile(tmp_path, path); + + by_partition[versioned.entry.partition_id] = ExportTTLVersionedEntry{.entry = versioned.entry, .version = stored_version + 1}; + publishSnapshot(); + return true; +} + +void MergeTreeExportTTLIndex::removeDestination(const String & destination_key) +{ + std::lock_guard lock(mutex); + + auto disk = storage.getDisks().front(); + const auto path = getDestinationPath(destination_key); + if (disk->existsDirectory(path)) + disk->removeRecursive(path); + + entries.erase(destination_key); + publishSnapshot(); + + LOG_INFO(storage.log, "Removed the TTL export index of destination {}", destination_key); +} + +Int64 MergeTreeExportTTLIndex::maxBlock() const +{ + std::lock_guard lock(mutex); + Int64 result = 0; + for (const auto & [_, by_partition] : entries) + for (const auto & partition_entry : by_partition) + result = std::max(result, partition_entry.second.entry.maxBlock()); + return result; +} + +void MergeTreeExportTTLIndex::publishSnapshot() +{ + ProfileEvents::increment(ProfileEvents::ExportTTLIndexSnapshotRefreshes); + snapshot.set(ExportTTLIndexSnapshot::build(/* version */ -1, entries)); +} + +MergeTreeExportTTLScheduler::MergeTreeExportTTLScheduler(StorageMergeTree & storage_, MergeTreeExportTTLIndex & index_) + : ExportTTLScheduler(storage_) + , plain_storage(storage_) + , index(index_) +{ +} + +bool MergeTreeExportTTLScheduler::isPaused() +{ + return plain_storage.parts_mover.moves_blocker.isCancelled(); +} + +ExportTTLScheduler::TaskState MergeTreeExportTTLScheduler::getTaskState(const String & transaction_id) +{ + TaskState state; + const auto task = plain_storage.export_task_scheduler->getTask(transaction_id); + if (!task) + return state; + + switch (task->status) + { + case MergeTreeExportTask::Status::PENDING: state.status = TaskStatus::PENDING; break; + case MergeTreeExportTask::Status::COMPLETED: state.status = TaskStatus::COMPLETED; break; + case MergeTreeExportTask::Status::FAILED: state.status = TaskStatus::FAILED; break; + case MergeTreeExportTask::Status::KILLED: state.status = TaskStatus::KILLED; break; + } + state.retry_of = task->retry_of; + return state; +} + +bool MergeTreeExportTTLScheduler::startGroup(const GroupToStart & group, const ContextPtr & context) +{ + const auto source_metadata = plain_storage.getInMemoryMetadataPtr(context, false); + const auto destination_metadata = group.destination->getInMemoryMetadataPtr(context, false); + ExportTaskUtils::verifyExportSchemaCastable(source_metadata, destination_metadata, group.destination->getStorageID(), context); + + MergeTreeData::DataPartsVector parts(group.parts.begin(), group.parts.end()); + auto descriptor = plain_storage.buildExportTask( + group.destination->getStorageID(), group.destination, source_metadata, destination_metadata, parts, group.partition_id, context); + descriptor.transaction_id = group.transaction_id; + descriptor.source = ExportTaskSource::ttl; + descriptor.retry_of = group.retry_of; + + { + /// Merge selection holds it while it checks the fence and tags the parts it merges. + std::lock_guard lock(plain_storage.currently_processing_in_background_mutex); + for (const auto & part : group.parts) + { + if (part->getState() != MergeTreeDataPartState::Active || plain_storage.currently_merging_mutating_parts.contains(part->info)) + return false; + } + + if (!index.update(group.destination_key, group.entry)) + return false; + } + + plain_storage.export_task_scheduler->addTask(std::move(descriptor), std::vector(group.parts.begin(), group.parts.end())); + return true; +} + +bool MergeTreeExportTTLScheduler::isPartBeingMerged(const MergeTreeDataPartPtr & part) +{ + std::lock_guard lock(plain_storage.currently_processing_in_background_mutex); + return plain_storage.currently_merging_mutating_parts.contains(part->info); +} + +void MergeTreeExportTTLScheduler::killTask(const String & transaction_id) +{ + plain_storage.export_task_scheduler->kill(transaction_id); +} + +} diff --git a/src/Storages/MergeTree/MergeTreeExportTTLScheduler.h b/src/Storages/MergeTree/MergeTreeExportTTLScheduler.h new file mode 100644 index 000000000000..07b3a19cef9d --- /dev/null +++ b/src/Storages/MergeTree/MergeTreeExportTTLScheduler.h @@ -0,0 +1,80 @@ +#pragma once + +#include +#include + +#include +#include + +namespace DB +{ + +class StorageMergeTree; + +/// The export index of the `EXPORT` TTL of a plain `MergeTree`, in the data directory of the table: +/// `export_ttl//partitions/.json`, each file written atomically. +/// Versions are kept in memory only, for the compare-and-set of the scheduler. +class MergeTreeExportTTLIndex +{ +public: + explicit MergeTreeExportTTLIndex(StorageMergeTree & storage_); + + void load(); + + /// Stores `entry` if the stored version is still `entry.version`. Returns false otherwise. + bool update(const String & destination_key, const ExportTTLVersionedEntry & entry); + + void removeDestination(const String & destination_key); + + ExportTTLIndexSnapshotPtr getSnapshot() const { return snapshot.get(); } + ExportFencePtr getFence() const { return snapshot.get()->fence; } + + /// Highest block number in any entry, 0 if there is none. + Int64 maxBlock() const; + +private: + StorageMergeTree & storage; + + mutable std::mutex mutex; + /// By destination key, then by partition id. + std::map> entries; + MultiVersion snapshot; + + String getRootPath() const; + String getDestinationPath(const String & destination_key) const; + String getEntryPath(const String & destination_key, const String & partition_id) const; + + /// Caller must hold `mutex`. + void publishSnapshot(); +}; + +/// `ExportTTLScheduler` of a plain `MergeTree`. A group is claimed in the index before its task is +/// created, under the lock that merge selection holds, so no merge of its parts can be selected +/// meanwhile. A crash in between leaves a claim without a task, which the next tick retries. +class MergeTreeExportTTLScheduler final : public ExportTTLScheduler +{ +public: + MergeTreeExportTTLScheduler(StorageMergeTree & storage_, MergeTreeExportTTLIndex & index_); + +protected: + bool acquireSchedulerLock() override { return true; } + bool isPaused() override; + ExportTTLIndexSnapshotPtr getIndexSnapshot() override { return index.getSnapshot(); } + bool identifiesDestinationByUUID() const override { return true; } + /// A single replica keeps its state in memory. + std::optional readSchedulerState(const String &) override { return std::nullopt; } + void writeSchedulerState(const String &, const ExportTTLSchedulerState &) override {} + String getReplicaName() const override { return {}; } + TaskState getTaskState(const String & transaction_id) override; + bool updateIndexEntry(const String & destination_key, const ExportTTLVersionedEntry & entry) override { return index.update(destination_key, entry); } + bool startGroup(const GroupToStart & group, const ContextPtr & context) override; + bool isPartBeingMerged(const MergeTreeDataPartPtr & part) override; + void killTask(const String & transaction_id) override; + void removeDestination(const String & destination_key) override { index.removeDestination(destination_key); } + +private: + StorageMergeTree & plain_storage; + MergeTreeExportTTLIndex & index; +}; + +} diff --git a/src/Storages/MergeTree/MergeTreePartitionExportTask.h b/src/Storages/MergeTree/MergeTreeExportTask.h similarity index 84% rename from src/Storages/MergeTree/MergeTreePartitionExportTask.h rename to src/Storages/MergeTree/MergeTreeExportTask.h index 8ce9958a0dbe..9967e1a61755 100644 --- a/src/Storages/MergeTree/MergeTreePartitionExportTask.h +++ b/src/Storages/MergeTree/MergeTreeExportTask.h @@ -8,7 +8,8 @@ #include #include #include -#include +#include +#include #include namespace DB @@ -19,8 +20,9 @@ namespace ErrorCodes extern const int INCORRECT_DATA; } -/// On-disk descriptor for a plain (non-replicated) `MergeTree` partition export task. -struct MergeTreePartitionExportTask +/// On-disk descriptor of an export task of a plain (non-replicated) `MergeTree`: a group of parts +/// of one partition, exported by `EXPORT PARTITION` or by the `EXPORT` TTL. +struct MergeTreeExportTask { using FileAlreadyExistsPolicy = MergeTreePartExportManifest::FileAlreadyExistsPolicy; @@ -40,7 +42,7 @@ struct MergeTreePartitionExportTask }; /// Best-effort record of the most recent failure observed for this task. `count` is a running - /// total of failures and matches the semantics documented for `system.partition_exports`. + /// total of failures and matches the semantics documented for `system.distributed_exports`. struct LastException { String message; @@ -52,27 +54,32 @@ struct MergeTreePartitionExportTask /// Identity String transaction_id; String query_id; - String partition_id; String source_database; String source_table; String destination_database; String destination_table; + /// Empty if the destination has no UUID. + String destination_uuid; time_t create_time = 0; + ExportTaskSource source = ExportTaskSource::query; + /// TTL tasks whose parts this task exports again because they failed. Their commits may still + /// land, which the commit of this task checks. + std::vector retry_of; /// Work + progress std::vector parts; Status status = Status::PENDING; LastException last_exception; - std::optional commit_info; + std::optional commit_info; size_t retry_initial_backoff_seconds = 5; size_t retry_max_backoff_seconds = 300; size_t task_timeout_seconds = 86400; /// Export settings. Field names are intentionally identical to - /// `ExportReplicatedMergeTreePartitionManifest` so that - /// `ExportPartitionUtils::getContextCopyWithTaskSettings` works for both descriptors. + /// `ExportReplicatedMergeTreeTaskManifest` so that + /// `ExportTaskUtils::getContextCopyWithTaskSettings` works for both descriptors. size_t max_threads = 0; bool parallel_formatting = false; bool parquet_parallel_encoding = false; @@ -85,7 +92,7 @@ struct MergeTreePartitionExportTask String iceberg_metadata_json; /// Optional for backwards compatibility with descriptors written before these - /// settings were persisted (same shape as ExportReplicatedMergeTreePartitionManifest). + /// settings were persisted (same shape as ExportReplicatedMergeTreeTaskManifest). std::optional parquet_compression_method; std::optional output_format_compression_level; std::optional parquet_row_group_size; @@ -145,12 +152,21 @@ struct MergeTreePartitionExportTask Poco::JSON::Object json; json.set("transaction_id", transaction_id); json.set("query_id", query_id); - json.set("partition_id", partition_id); json.set("source_database", source_database); json.set("source_table", source_table); json.set("destination_database", destination_database); json.set("destination_table", destination_table); + if (!destination_uuid.empty()) + json.set("destination_uuid", destination_uuid); json.set("create_time", create_time); + json.set("source", String(magic_enum::enum_name(source))); + if (!retry_of.empty()) + { + Poco::JSON::Array::Ptr retry_of_array = new Poco::JSON::Array(); + for (const auto & transaction : retry_of) + retry_of_array->add(transaction); + json.set("retry_of", retry_of_array); + } json.set("status", String(magic_enum::enum_name(status))); Poco::JSON::Array::Ptr parts_array = new Poco::JSON::Array(); @@ -212,26 +228,40 @@ struct MergeTreePartitionExportTask return oss.str(); } - static MergeTreePartitionExportTask fromJsonString(const std::string & json_string) + static MergeTreeExportTask fromJsonString(const std::string & json_string) { Poco::JSON::Parser parser; auto json = parser.parse(json_string).extract(); - MergeTreePartitionExportTask task; + MergeTreeExportTask task; task.transaction_id = json->getValue("transaction_id"); task.query_id = json->getValue("query_id"); - task.partition_id = json->getValue("partition_id"); task.source_database = json->getValue("source_database"); task.source_table = json->getValue("source_table"); task.destination_database = json->getValue("destination_database"); task.destination_table = json->getValue("destination_table"); + if (json->has("destination_uuid")) + task.destination_uuid = json->getValue("destination_uuid"); task.create_time = json->getValue("create_time"); + if (json->has("source")) + { + const auto source_str = json->getValue("source"); + if (const auto source = magic_enum::enum_cast(source_str)) + task.source = *source; + else + throw Exception(ErrorCodes::INCORRECT_DATA, "Unknown source '{}' in export task descriptor", source_str); + } + + if (const auto retry_of_array = json->getArray("retry_of")) + for (size_t i = 0; i < retry_of_array->size(); ++i) + task.retry_of.push_back(retry_of_array->getElement(static_cast(i))); + const auto status_str = json->getValue("status"); if (const auto status = magic_enum::enum_cast(status_str)) task.status = status.value(); else - throw Exception(ErrorCodes::INCORRECT_DATA, "Unknown status '{}' in partition export descriptor", status_str); + throw Exception(ErrorCodes::INCORRECT_DATA, "Unknown status '{}' in export task descriptor", status_str); const auto parts_array = json->getArray("parts"); for (size_t i = 0; i < parts_array->size(); ++i) @@ -260,7 +290,7 @@ struct MergeTreePartitionExportTask const auto commit_info_object = json->getObject("commit_info"); if (!commit_info_object) throw Exception(ErrorCodes::INCORRECT_DATA, "Field 'commit_info' of a partition export descriptor is not an object"); - task.commit_info = ExportPartitionCommitInfoEntry::fromJsonObject(commit_info_object); + task.commit_info = ExportCommitInfoEntry::fromJsonObject(commit_info_object); } task.retry_initial_backoff_seconds = json->getValue("retry_initial_backoff_seconds"); diff --git a/src/Storages/MergeTree/MergeTreePartitionExportScheduler.cpp b/src/Storages/MergeTree/MergeTreeExportTaskScheduler.cpp similarity index 63% rename from src/Storages/MergeTree/MergeTreePartitionExportScheduler.cpp rename to src/Storages/MergeTree/MergeTreeExportTaskScheduler.cpp index 275ef1fc05f5..e602bd355435 100644 --- a/src/Storages/MergeTree/MergeTreePartitionExportScheduler.cpp +++ b/src/Storages/MergeTree/MergeTreeExportTaskScheduler.cpp @@ -1,13 +1,12 @@ -#include +#include #include -#include -#include +#include #include #include #include #include #include -#include +#include #include #include #include @@ -21,6 +20,7 @@ #include #include #include +#include #include @@ -31,133 +31,94 @@ namespace DB namespace ErrorCodes { - extern const int EXPORT_PARTITION_ALREADY_EXPORTED; - extern const int CORRUPTED_DATA; extern const int QUERY_WAS_CANCELLED; extern const int UNKNOWN_TABLE; extern const int UNKNOWN_EXCEPTION; + extern const int LOGICAL_ERROR; } -MergeTreePartitionExportScheduler::MergeTreePartitionExportScheduler(StorageMergeTree & storage_) +MergeTreeExportTaskScheduler::MergeTreeExportTaskScheduler(StorageMergeTree & storage_) : storage(storage_) { } -String MergeTreePartitionExportScheduler::describeKey(const MergeTreePartitionExportTask & descriptor) +String MergeTreeExportTaskScheduler::getExportsRelativePath() const { - return fmt::format( - "{} -> {}.{}", - descriptor.partition_id, - backQuoteIfNeed(descriptor.destination_database), - backQuoteIfNeed(descriptor.destination_table)); + return fs::path(storage.getRelativeDataPath()) / "exports"; } -String MergeTreePartitionExportScheduler::getExportsRelativePath() const +String MergeTreeExportTaskScheduler::descriptorRelativePath(const String & transaction_id) const { - return fs::path(storage.getRelativeDataPath()) / "partition_exports"; + return fs::path(getExportsRelativePath()) / (escapeForFileName(transaction_id) + ".json"); } -String MergeTreePartitionExportScheduler::descriptorRelativePath(const String & composite_key) const +void MergeTreeExportTaskScheduler::addTask(MergeTreeExportTask descriptor, std::vector part_references) { - return fs::path(getExportsRelativePath()) / (sipHash128String(composite_key) + ".json"); -} - -std::map::iterator -MergeTreePartitionExportScheduler::findByTransactionId(const String & transaction_id) -{ - for (auto it = tasks.begin(); it != tasks.end(); ++it) - if (it->second.getDescriptor().transaction_id == transaction_id) - return it; - return tasks.end(); -} - -void MergeTreePartitionExportScheduler::addTask( - MergeTreePartitionExportTask descriptor, std::vector part_references, bool force) -{ - const auto composite_key = ExportPartitionUtils::compositeKey(descriptor.partition_id, descriptor.destination_database, descriptor.destination_table); - /// Rendered before `descriptor` is moved into the entry below. - const auto key_description = describeKey(descriptor); const auto transaction_id = descriptor.transaction_id; - - String previous_transaction_id; + const auto destination = fmt::format("{}.{}", backQuoteIfNeed(descriptor.destination_database), backQuoteIfNeed(descriptor.destination_table)); { std::lock_guard lock(mutex); - if (const auto it = tasks.find(composite_key); it != tasks.end()) - { - if (!force) - throw Exception(ErrorCodes::EXPORT_PARTITION_ALREADY_EXPORTED, - "Export with key {} already exported or it is being exported. " - "Set `export_merge_tree_partition_force_export` to overwrite it.", - key_description); - - previous_transaction_id = it->second.getDescriptor().transaction_id; - } - else - { - /// Unreadable entries are not loaded into memory, therefore we need to check the disk as well - const auto path = descriptorRelativePath(composite_key); - if (storage.getDisks().front()->existsFile(path)) - { - if (!force) - throw Exception(ErrorCodes::CORRUPTED_DATA, - "A partition export descriptor for key {} exists on disk at {} but could not be loaded. " - "Inspect the file or set `export_merge_tree_partition_force_export` to overwrite it.", - key_description, path); - - LOG_WARNING(storage.log, - "ExportPartition: overwriting unread descriptor at {} for key {}", path, key_description); - } - } - TaskEntry entry; entry.part_references = std::move(part_references); entry.setDescriptor(std::move(descriptor)); - persist(composite_key, entry.getDescriptor().toJsonString()); - tasks.insert_or_assign(composite_key, std::move(entry)); + persist(transaction_id, entry.getDescriptor().toJsonString()); + if (!tasks.emplace(transaction_id, std::move(entry)).second) + throw Exception(ErrorCodes::LOGICAL_ERROR, "Export task {} already exists", transaction_id); } - if (!previous_transaction_id.empty()) - { - LOG_INFO(storage.log, "ExportPartition: overwriting export with key {}", key_description); - storage.killExportPart(previous_transaction_id); - } + LOG_INFO(storage.log, "Export task: scheduled export task {} to {}", transaction_id, destination); + storage.triggerExportTaskScheduling(); +} - LOG_INFO(storage.log, "ExportPartition: scheduled export task {} (key {})", transaction_id, key_description); - storage.triggerPartitionExportTask(); +void MergeTreeExportTaskScheduler::wakeUpExportTTLIfFinished(const TaskEntry & entry) +{ + const auto & descriptor = entry.getDescriptor(); + if (descriptor.source == ExportTaskSource::ttl && descriptor.status != MergeTreeExportTask::Status::PENDING) + storage.wakeUpExportTTL(); } -CancellationCode MergeTreePartitionExportScheduler::kill(const String & transaction_id) +std::optional MergeTreeExportTaskScheduler::getTask(const String & transaction_id) const +{ + std::lock_guard lock(mutex); + const auto it = tasks.find(transaction_id); + if (it == tasks.end()) + return std::nullopt; + return it->second.getDescriptor(); +} + +CancellationCode MergeTreeExportTaskScheduler::kill(const String & transaction_id) { /// set the status to killed { std::lock_guard lock(mutex); - auto it = findByTransactionId(transaction_id); + auto it = tasks.find(transaction_id); if (it == tasks.end()) return CancellationCode::NotFound; auto & entry = it->second; - if (entry.getDescriptor().status != MergeTreePartitionExportTask::Status::PENDING) + if (entry.getDescriptor().status != MergeTreeExportTask::Status::PENDING) { - LOG_INFO(storage.log, "ExportPartition: export with the transaction id {} is not pending, cannot cancel it", transaction_id); + LOG_INFO(storage.log, "Export task: export with the transaction id {} is not pending, cannot cancel it", transaction_id); return CancellationCode::CancelCannotBeSent; } - + if (entry.committing) { - LOG_INFO(storage.log, "ExportPartition: commit in progress for {}, cannot cancel export partition task", transaction_id); + LOG_INFO(storage.log, "Export task: commit in progress for {}, cannot cancel export partition task", transaction_id); return CancellationCode::CancelCannotBeSent; } auto updated = entry.getDescriptor(); - updated.status = MergeTreePartitionExportTask::Status::KILLED; + updated.status = MergeTreeExportTask::Status::KILLED; persist(it->first, updated.toJsonString()); entry.setDescriptor(std::move(updated)); + wakeUpExportTTLIfFinished(entry); } /// cancel in-flight operations @@ -166,19 +127,23 @@ CancellationCode MergeTreePartitionExportScheduler::kill(const String & transact return CancellationCode::CancelSent; } -std::vector MergeTreePartitionExportScheduler::getInfo() const +std::vector MergeTreeExportTaskScheduler::getInfo() const { - std::vector result; + std::vector result; std::lock_guard lock(mutex); result.reserve(tasks.size()); for (const auto & [key, entry] : tasks) { const auto & descriptor = entry.getDescriptor(); - PartitionExportInfo info; + ExportTaskInfo info; info.destination_database = descriptor.destination_database; info.destination_table = descriptor.destination_table; info.create_time = descriptor.create_time; - info.partition_id = descriptor.partition_id; + info.partition_id = descriptor.parts.empty() + ? "" + : ExportTaskUtils::getPartitionIdOfParts(descriptor.partNames(), storage.format_version); + info.source = String(magic_enum::enum_name(descriptor.source)); + info.retry_of = descriptor.retry_of; info.transaction_id = descriptor.transaction_id; info.query_id = descriptor.query_id; info.parts = descriptor.partNames(); @@ -217,15 +182,15 @@ std::vector MergeTreePartitionExportScheduler::getInfo() co return result; } -bool MergeTreePartitionExportScheduler::tryPersistTimeoutKill(const String & composite_key, TaskEntry & entry, time_t now) +bool MergeTreeExportTaskScheduler::tryPersistTimeoutKill(const String & transaction_id_key, TaskEntry & entry, time_t now) { const auto transaction_id = entry.getDescriptor().transaction_id; const auto timeout_seconds = entry.getDescriptor().task_timeout_seconds; auto updated = entry.getDescriptor(); - updated.status = MergeTreePartitionExportTask::Status::KILLED; + updated.status = MergeTreeExportTask::Status::KILLED; updated.last_exception.message = fmt::format( - "Export partition task timed out: exceeded export_merge_tree_partition_task_timeout_seconds={} (created at {}, now {})", + "Export partition task timed out: exceeded export_merge_tree_task_timeout_seconds={} (created at {}, now {})", timeout_seconds, entry.getDescriptor().create_time, now); updated.last_exception.part = ""; updated.last_exception.time = now; @@ -233,22 +198,23 @@ bool MergeTreePartitionExportScheduler::tryPersistTimeoutKill(const String & com try { - persist(composite_key, updated.toJsonString()); + persist(transaction_id_key, updated.toJsonString()); } catch (...) { - tryLogCurrentException(storage.log, "ExportPartition: failed to persist timeout kill, will retry"); + tryLogCurrentException(storage.log, "Export task: failed to persist timeout kill, will retry"); return false; } entry.setDescriptor(std::move(updated)); + wakeUpExportTTLIfFinished(entry); LOG_WARNING(storage.log, - "ExportPartition: task {} exceeded task_timeout_seconds={}s, transitioned PENDING -> KILLED", + "Export task: task {} exceeded task_timeout_seconds={}s, transitioned PENDING -> KILLED", transaction_id, timeout_seconds); return true; } -bool MergeTreePartitionExportScheduler::enforceTimeouts() +bool MergeTreeExportTaskScheduler::enforceTimeouts() { std::vector timed_out_transactions; bool any_pending = false; @@ -258,10 +224,10 @@ bool MergeTreePartitionExportScheduler::enforceTimeouts() const auto now = time(nullptr); for (auto & [key, entry] : tasks) { - if (entry.getDescriptor().status != MergeTreePartitionExportTask::Status::PENDING) + if (entry.getDescriptor().status != MergeTreeExportTask::Status::PENDING) continue; - if (!ExportPartitionUtils::isExportTaskTimedOut( + if (!ExportTaskUtils::isExportTaskTimedOut( entry.getDescriptor().create_time, entry.getDescriptor().task_timeout_seconds, now)) { any_pending = true; @@ -286,7 +252,7 @@ bool MergeTreePartitionExportScheduler::enforceTimeouts() return any_pending; } -bool MergeTreePartitionExportScheduler::run() +bool MergeTreeExportTaskScheduler::run() { /// Timeouts first, before the "cannot make progress this tick" early returns, so a wedged /// task (no move executors, memory pressure, ...) still expires. @@ -298,14 +264,14 @@ bool MergeTreePartitionExportScheduler::run() const auto available_move_executors = storage.background_moves_assignee.getAvailableMoveExecutors(); if (available_move_executors == 0) { - LOG_INFO(storage.log, "ExportPartition: no move executors available, skipping run on plain"); + LOG_INFO(storage.log, "Export task: no move executors available, skipping run on plain"); return true; } - + if (storage.parts_mover.moves_blocker.isCancelled()) { - LOG_INFO(storage.log, "ExportPartition: moves cancelled, skipping run on plain"); + LOG_INFO(storage.log, "Export task: moves cancelled, skipping run on plain"); return true; } @@ -323,7 +289,7 @@ bool MergeTreePartitionExportScheduler::run() for (auto & [key, entry] : tasks) { const auto & descriptor = entry.getDescriptor(); - if (descriptor.status != MergeTreePartitionExportTask::Status::PENDING) + if (descriptor.status != MergeTreeExportTask::Status::PENDING) continue; if (descriptor.allPartsDone()) @@ -332,7 +298,7 @@ bool MergeTreePartitionExportScheduler::run() /// itself takes the committing lease; skip if a commit is already in flight. if (!entry.committing) { - LOG_DEBUG(storage.log, "ExportPartition: all parts exported for task {}, committing", descriptor.transaction_id); + LOG_DEBUG(storage.log, "Export task: all parts exported for task {}, committing", descriptor.transaction_id); tasks_to_commit.push_back(descriptor.transaction_id); } continue; @@ -342,20 +308,20 @@ bool MergeTreePartitionExportScheduler::run() { if (scheduled >= available_move_executors) { - LOG_DEBUG(storage.log, "ExportPartition: no move executors available, skipping part export for task {}", descriptor.transaction_id); + LOG_DEBUG(storage.log, "Export task: no move executors available, skipping part export for task {}", descriptor.transaction_id); break; } if (part.done || entry.in_flight_parts.contains(part.part_name)) { - LOG_DEBUG(storage.log, "ExportPartition: part {} already exported or in flight for task {}, skipping", part.part_name, descriptor.transaction_id); + LOG_DEBUG(storage.log, "Export task: part {} already exported or in flight for task {}, skipping", part.part_name, descriptor.transaction_id); continue; } if (const auto backoff_it = entry.part_backoff.find(part.part_name); backoff_it != entry.part_backoff.end() && now < backoff_it->second.next_retry_time) { - LOG_DEBUG(storage.log, "ExportPartition: part {} backoff time not reached for task {}, skipping", part.part_name, descriptor.transaction_id); + LOG_DEBUG(storage.log, "Export task: part {} backoff time not reached for task {}, skipping", part.part_name, descriptor.transaction_id); continue; } @@ -380,12 +346,12 @@ bool MergeTreePartitionExportScheduler::run() return true; } -void MergeTreePartitionExportScheduler::scheduleOnePart(const String & transaction_id, const String & part_name) +void MergeTreeExportTaskScheduler::scheduleOnePart(const String & transaction_id, const String & part_name) { - MergeTreePartitionExportTask descriptor_copy; + MergeTreeExportTask descriptor_copy; { std::lock_guard lock(mutex); - auto it = findByTransactionId(transaction_id); + auto it = tasks.find(transaction_id); if (it == tasks.end()) return; @@ -394,7 +360,7 @@ void MergeTreePartitionExportScheduler::scheduleOnePart(const String & transacti /// run() marked the part in flight and released the mutex, so a KILL from a query thread /// may have made the task terminal since then. Dispatching now would write destination /// objects for an export the user already cancelled. - if (entry.getDescriptor().status != MergeTreePartitionExportTask::Status::PENDING) + if (entry.getDescriptor().status != MergeTreeExportTask::Status::PENDING) { entry.in_flight_parts.erase(part_name); return; @@ -407,9 +373,9 @@ void MergeTreePartitionExportScheduler::scheduleOnePart(const String & transacti try { - auto context = ExportPartitionUtils::getContextCopyWithTaskSettings(storage.getContext(), descriptor_copy); + auto context = ExportTaskUtils::getContextCopyWithTaskSettings(storage.getContext(), descriptor_copy); - LOG_INFO(storage.log, "ExportPartition: scheduling part export {} for task {}", part_name, transaction_id); + LOG_INFO(storage.log, "Export task: scheduling part export {} for task {}", part_name, transaction_id); storage.exportPartToTable( part_name, @@ -430,9 +396,9 @@ void MergeTreePartitionExportScheduler::scheduleOnePart(const String & transacti bool still_pending = false; { std::lock_guard lock(mutex); - auto it = findByTransactionId(transaction_id); + auto it = tasks.find(transaction_id); still_pending = it != tasks.end() - && it->second.getDescriptor().status == MergeTreePartitionExportTask::Status::PENDING; + && it->second.getDescriptor().status == MergeTreeExportTask::Status::PENDING; } if (!still_pending) @@ -463,17 +429,17 @@ void MergeTreePartitionExportScheduler::scheduleOnePart(const String & transacti } } -void MergeTreePartitionExportScheduler::handlePartCompletion( +void MergeTreeExportTaskScheduler::handlePartCompletion( const String & transaction_id, const String & part_name, const MergeTreePartExportManifest::CompletionCallbackResult & result) { bool ready_to_commit = false; { std::lock_guard lock(mutex); - auto it = findByTransactionId(transaction_id); + auto it = tasks.find(transaction_id); if (it == tasks.end()) { - LOG_DEBUG(storage.log, "ExportPartition: task {} completed, but manifest not. The task was likely overwritten by a new task or this is a bug", transaction_id); + LOG_DEBUG(storage.log, "Export task: task {} completed, but manifest not. The task was likely overwritten by a new task or this is a bug", transaction_id); return; } @@ -482,7 +448,7 @@ void MergeTreePartitionExportScheduler::handlePartCompletion( entry.in_flight_parts.erase(part_name); /// Task already terminal (KILLED / FAILED / COMPLETED): ignore late completions. - if (entry.getDescriptor().status != MergeTreePartitionExportTask::Status::PENDING) + if (entry.getDescriptor().status != MergeTreeExportTask::Status::PENDING) return; /// A cancelled export (KILL, or SYSTEM STOP MOVES) is not a real failure: leave the part @@ -512,8 +478,8 @@ void MergeTreePartitionExportScheduler::handlePartCompletion( updated.last_exception.time = time(nullptr); updated.last_exception.count += 1; - if (result.exception && ExportPartitionUtils::isNonRetryablePlainExportError(result.exception->code())) - updated.status = MergeTreePartitionExportTask::Status::FAILED; + if (result.exception && ExportTaskUtils::isNonRetryablePlainExportError(result.exception->code())) + updated.status = MergeTreeExportTask::Status::FAILED; } try @@ -522,11 +488,12 @@ void MergeTreePartitionExportScheduler::handlePartCompletion( } catch (...) { - tryLogCurrentException(storage.log, "ExportPartition: failed to persist part-completion state, will retry"); + tryLogCurrentException(storage.log, "Export task: failed to persist part-completion state, will retry"); return; } entry.setDescriptor(std::move(updated)); + wakeUpExportTTLIfFinished(entry); if (result.success) { @@ -534,16 +501,16 @@ void MergeTreePartitionExportScheduler::handlePartCompletion( if (entry.getDescriptor().allPartsDone() && !entry.committing) ready_to_commit = true; } - else if (entry.getDescriptor().status == MergeTreePartitionExportTask::Status::FAILED) + else if (entry.getDescriptor().status == MergeTreeExportTask::Status::FAILED) { - LOG_WARNING(storage.log, "ExportPartition: task {} failed on part {} with non-retryable error", + LOG_WARNING(storage.log, "Export task: task {} failed on part {} with non-retryable error", transaction_id, part_name); } else { auto & backoff = entry.part_backoff[part_name]; ++backoff.attempts; - const auto backoff_seconds = ExportPartitionUtils::computeRetryBackoffSeconds( + const auto backoff_seconds = ExportTaskUtils::computeRetryBackoffSeconds( backoff.attempts, entry.getDescriptor().retry_initial_backoff_seconds, entry.getDescriptor().retry_max_backoff_seconds); @@ -551,7 +518,7 @@ void MergeTreePartitionExportScheduler::handlePartCompletion( const size_t headroom = static_cast(std::numeric_limits::max() - now); backoff.next_retry_time = now + static_cast(std::min(backoff_seconds, headroom)); - LOG_INFO(storage.log, "ExportPartition: task {} part {} failed with retryable error, will retry at {}", + LOG_INFO(storage.log, "Export task: task {} part {} failed with retryable error, will retry at {}", transaction_id, part_name, backoff.next_retry_time); } } @@ -560,10 +527,9 @@ void MergeTreePartitionExportScheduler::handlePartCompletion( tryCommit(transaction_id); } -void MergeTreePartitionExportScheduler::tryCommit(const String & transaction_id) +void MergeTreeExportTaskScheduler::tryCommit(const String & transaction_id) { - MergeTreePartitionExportTask descriptor_copy; - String composite_key; + MergeTreeExportTask descriptor_copy; /// This call is the sole writer of `committing`. Drop the lease on every exit after we claimed /// it, including exceptions that are not `DB::Exception` (otherwise the task is wedged until /// restart: timeout and KILL also refuse to act while the flag is set). @@ -573,24 +539,23 @@ void MergeTreePartitionExportScheduler::tryCommit(const String & transaction_id) if (!claimed) return; std::lock_guard lock(mutex); - auto it = tasks.find(composite_key); - if (it != tasks.end() && it->second.getDescriptor().transaction_id == transaction_id) + auto it = tasks.find(transaction_id); + if (it != tasks.end()) it->second.committing = false; }); { std::lock_guard lock(mutex); - auto it = findByTransactionId(transaction_id); + auto it = tasks.find(transaction_id); if (it == tasks.end()) return; auto & entry = it->second; if (entry.committing) return; - if (entry.getDescriptor().status != MergeTreePartitionExportTask::Status::PENDING || !entry.getDescriptor().allPartsDone()) + if (entry.getDescriptor().status != MergeTreeExportTask::Status::PENDING || !entry.getDescriptor().allPartsDone()) return; - composite_key = it->first; entry.committing = true; claimed = true; descriptor_copy = entry.getDescriptor(); @@ -600,10 +565,56 @@ void MergeTreePartitionExportScheduler::tryCommit(const String & transaction_id) bool success = false; std::optional failure; - IStorage::ExportPartitionCommitInfo destination_commit_info; + IStorage::ExportCommitInfo destination_commit_info; try { - const auto exported_paths = descriptor_copy.collectExportedPaths(); + auto exported_paths = descriptor_copy.collectExportedPaths(); + + std::optional context; + StoragePtr destination_storage; + const auto get_destination = [&] + { + if (destination_storage) + return; + destination_storage = DatabaseCatalog::instance().tryGetTable(destination_storage_id, storage.getContext()); + if (!destination_storage) + throw Exception(ErrorCodes::UNKNOWN_TABLE, "Destination table {} not found for export commit", + destination_storage_id.getNameForLogs()); + context = ExportTaskUtils::getContextCopyWithTaskSettings(storage.getContext(), descriptor_copy); + }; + + /// A task this one retries may have landed after it was considered failed. Its parts are + /// then in the destination already, so only the files of the other parts are committed. + if (!descriptor_copy.retry_of.empty()) + { + get_destination(); + const auto committed_ranges = ExportTaskUtils::getRangesCommittedByRetriedTasks( + descriptor_copy.retry_of, + destination_storage, + [this](const String & retried_transaction_id) -> std::optional> + { + if (const auto task = getTask(retried_transaction_id)) + return task->partNames(); + return std::nullopt; + }, + storage.format_version, + *context); + + if (!committed_ranges.empty()) + { + const auto parts_to_commit = ExportTaskUtils::getPartsNotCommitted( + descriptor_copy.partNames(), committed_ranges, storage.format_version); + const std::unordered_set parts_to_commit_set(parts_to_commit.begin(), parts_to_commit.end()); + + exported_paths.clear(); + for (const auto & part : descriptor_copy.parts) + if (parts_to_commit_set.contains(part.part_name)) + exported_paths.insert(exported_paths.end(), part.paths_in_destination.begin(), part.paths_in_destination.end()); + + LOG_INFO(storage.log, "Export task: a task retried by {} committed some of its parts, committing the files of {} of {} parts", + transaction_id, parts_to_commit.size(), descriptor_copy.parts.size()); + } + } /// Every part is done and its paths were recorded durably at completion, so an empty set /// here is not a lost-data symptom: an Iceberg destination writes no data file for a part @@ -612,22 +623,17 @@ void MergeTreePartitionExportScheduler::tryCommit(const String & transaction_id) if (exported_paths.empty()) { LOG_INFO(storage.log, - "ExportPartition: task {} produced no destination files, nothing to commit", transaction_id); + "Export task: task {} has no destination files to commit", transaction_id); } else { - auto destination_storage = DatabaseCatalog::instance().tryGetTable(destination_storage_id, storage.getContext()); - if (!destination_storage) - throw Exception(ErrorCodes::UNKNOWN_TABLE, "Destination table {} not found for export commit", - destination_storage_id.getNameForLogs()); + get_destination(); - auto context = ExportPartitionUtils::getContextCopyWithTaskSettings(storage.getContext(), descriptor_copy); + LOG_INFO(storage.log, "Export task: all parts exported for task {}, committing", transaction_id); - LOG_INFO(storage.log, "ExportPartition: all parts exported for task {}, committing", transaction_id); - - destination_commit_info = ExportPartitionUtils::commitExportOnDestination( + destination_commit_info = ExportTaskUtils::commitExportOnDestination( descriptor_copy.transaction_id, - descriptor_copy.partition_id, + ExportTaskUtils::getPartitionIdOfParts(descriptor_copy.partNames(), storage.format_version), descriptor_copy.iceberg_metadata_json, descriptor_copy.write_full_path_in_iceberg_metadata, descriptor_copy.iceberg_partition_timezone, @@ -635,7 +641,7 @@ void MergeTreePartitionExportScheduler::tryCommit(const String & transaction_id) descriptor_copy.partNames(), destination_storage, storage, - context); + *context); } success = true; @@ -643,25 +649,25 @@ void MergeTreePartitionExportScheduler::tryCommit(const String & transaction_id) catch (const Exception & e) { failure = e; - LOG_WARNING(storage.log, "ExportPartition: commit for task {} failed: {}", transaction_id, e.message()); + LOG_WARNING(storage.log, "Export task: commit for task {} failed: {}", transaction_id, e.message()); } catch (...) { - tryLogCurrentException(storage.log, "ExportPartition: commit for task " + transaction_id + " failed"); + tryLogCurrentException(storage.log, "Export task: commit for task " + transaction_id + " failed"); failure = Exception::createRuntime(ErrorCodes::UNKNOWN_EXCEPTION, getCurrentExceptionMessage(/*with_stacktrace=*/ false)); } { std::lock_guard lock(mutex); - auto it = tasks.find(composite_key); - if (it == tasks.end() || it->second.getDescriptor().transaction_id != transaction_id) + auto it = tasks.find(transaction_id); + if (it == tasks.end()) return; auto & entry = it->second; /// A concurrent KILL may have won the race while we were committing. Its terminal state is /// already durable, so there is nothing to persist here. `SCOPE_EXIT` still drops the lease. - if (entry.getDescriptor().status != MergeTreePartitionExportTask::Status::PENDING) + if (entry.getDescriptor().status != MergeTreeExportTask::Status::PENDING) return; /// Persist-then-apply under the lock. Note the destination commit above already happened @@ -670,8 +676,8 @@ void MergeTreePartitionExportScheduler::tryCommit(const String & transaction_id) auto updated = entry.getDescriptor(); if (success) { - updated.status = MergeTreePartitionExportTask::Status::COMPLETED; - updated.commit_info = ExportPartitionCommitInfoEntry{ + updated.status = MergeTreeExportTask::Status::COMPLETED; + updated.commit_info = ExportCommitInfoEntry{ destination_commit_info.iceberg_metadata_file, destination_commit_info.iceberg_manifest_list, destination_commit_info.iceberg_manifest_file, @@ -684,36 +690,33 @@ void MergeTreePartitionExportScheduler::tryCommit(const String & transaction_id) updated.last_exception.time = time(nullptr); updated.last_exception.count += 1; - if (failure && ExportPartitionUtils::isNonRetryablePlainExportError(failure->code())) - updated.status = MergeTreePartitionExportTask::Status::FAILED; + if (failure && ExportTaskUtils::isNonRetryablePlainExportError(failure->code())) + updated.status = MergeTreeExportTask::Status::FAILED; /// Otherwise leave PENDING: run() will retry the commit on the next tick. } try { - persist(composite_key, updated.toJsonString()); + persist(transaction_id, updated.toJsonString()); } catch (...) { - tryLogCurrentException(storage.log, "ExportPartition: failed to persist commit result, will retry"); + tryLogCurrentException(storage.log, "Export task: failed to persist commit result, will retry"); return; } entry.setDescriptor(std::move(updated)); + wakeUpExportTTLIfFinished(entry); } } -void MergeTreePartitionExportScheduler::persist(const String & composite_key, const String & descriptor_json) +void MergeTreeExportTaskScheduler::persist(const String & transaction_id, const String & descriptor_json) { - /// Always called while holding the registry `mutex`, so writes are already serialized. The file - /// is named by a 128-bit hash of the composite key (a fixed-length, filesystem-safe, opaque - /// handle), so a force-replace overwrites the previous record in place. The authoritative key is - /// recomputed from the JSON body in loadFromDisk, so the name only needs to be deterministic and - /// collision-free -- sipHash128 (matching FileCacheKey) provides both. + /// Always called while holding the registry `mutex`, so writes are already serialized. auto disk = storage.getDisks().front(); disk->createDirectories(getExportsRelativePath()); - const auto final_path = descriptorRelativePath(composite_key); + const auto final_path = descriptorRelativePath(transaction_id); const auto tmp_path = final_path + ".tmp"; { @@ -725,7 +728,7 @@ void MergeTreePartitionExportScheduler::persist(const String & composite_key, co disk->replaceFile(tmp_path, final_path); } -void MergeTreePartitionExportScheduler::load() +void MergeTreeExportTaskScheduler::load() { auto disk = storage.getDisks().front(); const auto directory = getExportsRelativePath(); @@ -748,17 +751,15 @@ void MergeTreePartitionExportScheduler::load() String content; readStringUntilEOF(content, *buf); - auto descriptor = MergeTreePartitionExportTask::fromJsonString(content); - const auto composite_key = ExportPartitionUtils::compositeKey(descriptor.partition_id, descriptor.destination_database, descriptor.destination_table); - /// Rendered before `descriptor` is moved into the entry below. - const auto key_description = describeKey(descriptor); + auto descriptor = MergeTreeExportTask::fromJsonString(content); + const auto transaction_id = descriptor.transaction_id; TaskEntry entry; /// Re-pin every source part of a resumable task, including already-exported ones. /// Unfinished parts still need to be read; Iceberg commit also derives partition /// values from the original parts, so they must survive until COMPLETED/FAILED/KILLED. - if (descriptor.status == MergeTreePartitionExportTask::Status::PENDING) + if (descriptor.status == MergeTreeExportTask::Status::PENDING) { for (const auto & part : descriptor.parts) { @@ -770,18 +771,15 @@ void MergeTreePartitionExportScheduler::load() entry.setDescriptor(std::move(descriptor)); - /// The file name is a pure function of the composite key, so this code can never write - /// two records for one key. A duplicate means the directory was tampered with or holds - /// records written by an incompatible version; keeping an arbitrary one of them would - /// silently resurrect stale state, so report it instead. - if (!tasks.emplace(composite_key, std::move(entry)).second) + /// The file is named by the transaction id, so a duplicate means the directory was tampered with. + if (!tasks.emplace(transaction_id, std::move(entry)).second) { - LOG_ERROR(storage.log, "ExportPartition: ignoring {}, another record already describes key {}", - file_name, key_description); + LOG_ERROR(storage.log, "Export task: ignoring {}, another record already describes task {}", + file_name, transaction_id); continue; } - LOG_INFO(storage.log, "ExportPartition: loaded export task from disk (key {})", key_description); + LOG_INFO(storage.log, "Export task: loaded export task {} from disk", transaction_id); } catch (...) { diff --git a/src/Storages/MergeTree/MergeTreePartitionExportScheduler.h b/src/Storages/MergeTree/MergeTreeExportTaskScheduler.h similarity index 60% rename from src/Storages/MergeTree/MergeTreePartitionExportScheduler.h rename to src/Storages/MergeTree/MergeTreeExportTaskScheduler.h index fef31599dadf..f1c318db5b76 100644 --- a/src/Storages/MergeTree/MergeTreePartitionExportScheduler.h +++ b/src/Storages/MergeTree/MergeTreeExportTaskScheduler.h @@ -3,13 +3,14 @@ #include #include #include +#include #include #include #include #include #include -#include -#include +#include +#include namespace DB { @@ -17,20 +18,22 @@ namespace DB class StorageMergeTree; -class MergeTreePartitionExportScheduler +class MergeTreeExportTaskScheduler { public: - explicit MergeTreePartitionExportScheduler(StorageMergeTree & storage_); + explicit MergeTreeExportTaskScheduler(StorageMergeTree & storage_); using DataPartPtr = MergeTreePartExportManifest::DataPartPtr; /// Registers and persists a new task, then triggers the scheduler. - /// Throws if a task with the same (partition, destination) key already exists; - void addTask(MergeTreePartitionExportTask descriptor, std::vector part_references, bool force); + void addTask(MergeTreeExportTask descriptor, std::vector part_references); CancellationCode kill(const String & transaction_id); - std::vector getInfo() const; + /// A copy of the descriptor of the task, or nothing if there is no such task. + std::optional getTask(const String & transaction_id) const; + + std::vector getInfo() const; /// Scheduler tick: schedule pending parts of PENDING tasks and commit tasks whose parts are all /// exported. Invoked from the storage's schedule-pool task. Returns true if at least one task is @@ -52,9 +55,8 @@ class MergeTreePartitionExportScheduler /// True while `tryCommit` is in the destination-commit / persist window. Callers must not /// write this flag; overlapping tryCommit calls no-op when it is already set. bool committing = false; - /// In-memory per-part retry back-off. Keyed by part name so a force-replace of the same - /// composite key does not inherit a prior instance's delay. Not persisted: after restart - /// the first retry is immediate, then back-off resumes from subsequent failures. + /// In-memory per-part retry back-off, keyed by part name. Not persisted: after restart the + /// first retry is immediate, then back-off resumes from subsequent failures. struct PartBackoff { size_t attempts = 0; @@ -62,34 +64,32 @@ class MergeTreePartitionExportScheduler }; std::unordered_map part_backoff; - const MergeTreePartitionExportTask & getDescriptor() const { return descriptor; } + const MergeTreeExportTask & getDescriptor() const { return descriptor; } /// Install a (typically just-persisted) snapshot. A terminal status drops the source-part /// pins so outdated parts can be physically removed without a restart. - void setDescriptor(MergeTreePartitionExportTask new_descriptor) + void setDescriptor(MergeTreeExportTask new_descriptor) { descriptor = std::move(new_descriptor); - if (descriptor.status != MergeTreePartitionExportTask::Status::PENDING) + if (descriptor.status != MergeTreeExportTask::Status::PENDING) part_references.clear(); } private: - MergeTreePartitionExportTask descriptor; + MergeTreeExportTask descriptor; }; mutable std::mutex mutex; + /// By transaction id. std::map tasks; - /// Renders a task key for logs and error messages. Unlike `compositeKey` this is not injective, - /// so it must never be used as an identity. - static String describeKey(const MergeTreePartitionExportTask & descriptor); + /// The `EXPORT` TTL records a finished task of its own as exported, or retries it, right away. + void wakeUpExportTTLIfFinished(const TaskEntry & entry); void scheduleOnePart(const String & transaction_id, const String & part_name); void handlePartCompletion(const String & transaction_id, const String & part_name, const MergeTreePartExportManifest::CompletionCallbackResult & result); void tryCommit(const String & transaction_id); - /// Caller must hold `mutex`. `tasks.end()` if no task has this transaction id. - std::map::iterator findByTransactionId(const String & transaction_id); /// Wall-clock timeout pass. Transitions expired PENDING tasks to KILLED (unless a commit is /// already in flight) and cancels their in-flight parts. Returns true if any PENDING work @@ -98,15 +98,14 @@ class MergeTreePartitionExportScheduler /// Caller must hold `mutex`. Persist `KILLED` with a timeout reason. Returns false if the /// local write fails (descriptor is left unchanged so the next tick retries). - bool tryPersistTimeoutKill(const String & composite_key, TaskEntry & entry, time_t now); + bool tryPersistTimeoutKill(const String & transaction_id, TaskEntry & entry, time_t now); - /// Atomically write `descriptor_json` to the task's on-disk file (tmp + replace), named by the - /// composite key so a force-replace naturally overwrites the previous record. Always invoked + /// Atomically write `descriptor_json` to the task's on-disk file (tmp + replace). Always invoked /// while holding the registry `mutex` (write-through), so writes are serialized by that lock. - void persist(const String & composite_key, const String & descriptor_json); + void persist(const String & transaction_id, const String & descriptor_json); - /// Relative path of the descriptor file for `composite_key` (hash + `.json`). - String descriptorRelativePath(const String & composite_key) const; + /// Relative path of the descriptor file of the task, `exports/.json`. + String descriptorRelativePath(const String & transaction_id) const; String getExportsRelativePath() const; }; diff --git a/src/Storages/MergeTree/MergeTreeSettings.cpp b/src/Storages/MergeTree/MergeTreeSettings.cpp index 37e212fcae22..8615a9850873 100644 --- a/src/Storages/MergeTree/MergeTreeSettings.cpp +++ b/src/Storages/MergeTree/MergeTreeSettings.cpp @@ -820,6 +820,37 @@ Possible values: - [load_existing_rows_count_for_old_parts](#load_existing_rows_count_for_old_parts) setting )", 0) \ + DECLARE(UInt64, ttl_export_check_period_seconds, 10, R"( + How often the `TTL ... EXPORT TO TABLE` scheduler of the table looks for parts to export, in seconds. + )", EXPERIMENTAL) \ + DECLARE(UInt64, ttl_export_batch_window_seconds, 60, R"( + The eligible parts of a partition are exported once no new eligible part appeared for this many + seconds, so that parts which become eligible together are exported together. + )", EXPERIMENTAL) \ + DECLARE(UInt64, ttl_export_batch_max_delay_seconds, 600, R"( + The eligible parts of a partition are exported at the latest this many seconds after the first of + them was seen, even if new eligible parts keep appearing. + )", EXPERIMENTAL) \ + DECLARE(UInt64, ttl_export_batch_min_bytes, 256_MiB, R"( + The eligible parts of a partition are exported as soon as their size on disk reaches this many + bytes. 0 disables the threshold. + )", EXPERIMENTAL) \ + DECLARE(UInt64, ttl_export_max_parts_per_group, 100, R"( + Maximum number of parts exported together by one task of the `EXPORT` TTL. Parts of a failed task + are always retried together, even if there are more of them. + )", EXPERIMENTAL) \ + DECLARE(UInt64, ttl_export_max_bytes_per_group, 10_GiB, R"( + Maximum size on disk of the parts exported together by one task of the `EXPORT` TTL. A single part + bigger than this is exported on its own. 0 means unlimited. + )", EXPERIMENTAL) \ + DECLARE(UInt64, ttl_export_max_concurrent_groups, 4, R"( + Maximum number of tasks of the `EXPORT` TTL of the table that run at the same time. There is at + most one per partition. + )", EXPERIMENTAL) \ + DECLARE(String, ttl_export_settings_profile, "", R"( + Settings profile whose settings the tasks of the `EXPORT` TTL run with, e.g. the output format + settings and `export_merge_tree_task_timeout_seconds`. If empty, the default profile is used. + )", EXPERIMENTAL) \ DECLARE(String, merge_workload, "", R"( Used to regulate how resources are utilized and shared between merges and other workloads. Specified value is used as `workload` setting value for diff --git a/src/Storages/MergeTree/MutateTask.cpp b/src/Storages/MergeTree/MutateTask.cpp index 48255dcf4ecb..96cf706b1811 100644 --- a/src/Storages/MergeTree/MutateTask.cpp +++ b/src/Storages/MergeTree/MutateTask.cpp @@ -3585,6 +3585,19 @@ bool MutateTask::prepare() if (ctx->mutating_pipeline_builder.initialized()) ctx->execute_ttl_type = MutationHelpers::shouldExecuteTTL(ctx->metadata_snapshot, ctx->interpreter->getColumnDependencies()); + /// Rows of a part that the `EXPORT` TTL has not exported yet must not be deleted by their TTL: + /// it is only recalculated here, and applied by a merge once the part is exported. + if (ctx->execute_ttl_type == ExecuteTTLType::NORMAL) + { + const auto fence = ctx->data->getLatestExportFence(); + if (auto reason = ctx->data->getExportTTLDeleteGate().check( + ctx->source_part->name, ctx->source_part->info, ctx->source_part->ttl_infos, fence.get())) + { + LOG_DEBUG(ctx->log, "Not applying TTL in the mutation: {}", *reason); + ctx->execute_ttl_type = ExecuteTTLType::RECALCULATE; + } + } + if ((*ctx->data->getSettings())[MergeTreeSetting::exclude_deleted_rows_for_part_size_in_merge] && lightweight_delete_mode) { /// This mutation contains lightweight delete and we need to count the deleted rows, diff --git a/src/Storages/MergeTree/ReplicatedExportTTLIndex.cpp b/src/Storages/MergeTree/ReplicatedExportTTLIndex.cpp new file mode 100644 index 000000000000..62ad5a889c60 --- /dev/null +++ b/src/Storages/MergeTree/ReplicatedExportTTLIndex.cpp @@ -0,0 +1,204 @@ +#include + +#include +#include +#include +#include + +#include + +namespace fs = std::filesystem; + +namespace ProfileEvents +{ + extern const Event ExportTTLIndexSnapshotRefreshes; +} + +namespace DB +{ + +ReplicatedExportTTLIndex::ReplicatedExportTTLIndex(String zookeeper_path_, LoggerPtr log_) + : zookeeper_path(std::move(zookeeper_path_)) + , log(std::move(log_)) + , latest(ExportTTLIndexSnapshot::build(/* version */ -1, {})) +{ +} + +String ReplicatedExportTTLIndex::getFencePath() const +{ + return fs::path(zookeeper_path) / "export_fence"; +} + +String ReplicatedExportTTLIndex::getRootPath() const +{ + return fs::path(zookeeper_path) / "export_ttl"; +} + +String ReplicatedExportTTLIndex::getSchedulerLockPath() const +{ + return fs::path(getRootPath()) / "scheduler_lock"; +} + +String ReplicatedExportTTLIndex::getDestinationPath(const String & destination_key) const +{ + return fs::path(getRootPath()) / destination_key; +} + +String ReplicatedExportTTLIndex::getIndexEntryPath(const String & destination_key, const String & partition_id) const +{ + return fs::path(getDestinationPath(destination_key)) / "partitions" / partition_id; +} + +String ReplicatedExportTTLIndex::getSchedulerStatePath(const String & destination_key) const +{ + return fs::path(getDestinationPath(destination_key)) / "state"; +} + +int32_t ReplicatedExportTTLIndex::readFenceVersion(const zkutil::ZooKeeperPtr & zookeeper) const +{ + const auto path = getFencePath(); + + Coordination::Stat stat; + if (!zookeeper->exists(path, &stat)) + { + const auto code = zookeeper->tryCreate(path, "", zkutil::CreateMode::Persistent); + if (code != Coordination::Error::ZOK && code != Coordination::Error::ZNODEEXISTS) + throw zkutil::KeeperException::fromPath(code, path); + zookeeper->exists(path, &stat); + } + return stat.version; +} + +std::vector ReplicatedExportTTLIndex::listDestinations(const zkutil::ZooKeeperPtr & zookeeper) const +{ + Strings children; + const auto code = zookeeper->tryGetChildren(getRootPath(), children); + if (code == Coordination::Error::ZNONODE) + return {}; + if (code != Coordination::Error::ZOK) + throw zkutil::KeeperException::fromPath(code, getRootPath()); + + std::erase_if(children, [](const String & child) { return child == "scheduler_lock"; }); + return children; +} + +std::map ReplicatedExportTTLIndex::readIndex( + const zkutil::ZooKeeperPtr & zookeeper, const String & destination_key) const +{ + const String partitions_path = fs::path(getDestinationPath(destination_key)) / "partitions"; + + Strings partitions; + const auto code = zookeeper->tryGetChildren(partitions_path, partitions); + if (code == Coordination::Error::ZNONODE) + return {}; + if (code != Coordination::Error::ZOK) + throw zkutil::KeeperException::fromPath(code, partitions_path); + + Strings paths; + paths.reserve(partitions.size()); + for (const auto & partition_id : partitions) + paths.push_back(fs::path(partitions_path) / partition_id); + + auto responses = zookeeper->tryGet(paths); + responses.waitForResponses(); + + std::map result; + for (size_t i = 0; i < partitions.size(); ++i) + { + const auto & response = responses[i]; + if (response.error == Coordination::Error::ZNONODE) + continue; + if (response.error != Coordination::Error::ZOK) + throw zkutil::KeeperException::fromPath(response.error, paths[i]); + + result[partitions[i]] = ExportTTLVersionedEntry{ + .entry = ExportTTLIndexEntry::fromJSONString(partitions[i], response.data), + .version = response.stat.version, + }; + } + return result; +} + +ExportTTLVersionedEntry ReplicatedExportTTLIndex::readIndexEntry( + const zkutil::ZooKeeperPtr & zookeeper, const String & destination_key, const String & partition_id) const +{ + String data; + Coordination::Stat stat; + if (!zookeeper->tryGet(getIndexEntryPath(destination_key, partition_id), data, &stat)) + { + ExportTTLVersionedEntry missing; + missing.entry.partition_id = partition_id; + return missing; + } + return ExportTTLVersionedEntry{.entry = ExportTTLIndexEntry::fromJSONString(partition_id, data), .version = stat.version}; +} + +void ReplicatedExportTTLIndex::ensureDestination( + const zkutil::ZooKeeperPtr & zookeeper, const String & destination_key, const String & description) const +{ + const std::vector> nodes{ + {getRootPath(), ""}, + {getDestinationPath(destination_key), description}, + {fs::path(getDestinationPath(destination_key)) / "partitions", ""}, + }; + + for (const auto & [path, data] : nodes) + { + const auto code = zookeeper->tryCreate(path, data, zkutil::CreateMode::Persistent); + if (code != Coordination::Error::ZOK && code != Coordination::Error::ZNODEEXISTS) + throw zkutil::KeeperException::fromPath(code, path); + } +} + +void ReplicatedExportTTLIndex::appendUpdateEntryOps( + Coordination::Requests & ops, const String & destination_key, const ExportTTLIndexEntry & entry, int32_t version) const +{ + const auto path = getIndexEntryPath(destination_key, entry.partition_id); + if (version < 0) + ops.emplace_back(zkutil::makeCreateRequest(path, entry.toJSONString(), zkutil::CreateMode::Persistent)); + else + ops.emplace_back(zkutil::makeSetRequest(path, entry.toJSONString(), version)); + + ops.emplace_back(zkutil::makeSetRequest(getFencePath(), "", -1)); +} + +void ReplicatedExportTTLIndex::removeDestination(const zkutil::ZooKeeperPtr & zookeeper, const String & destination_key) const +{ + zookeeper->tryRemoveRecursive(getDestinationPath(destination_key)); + zookeeper->trySet(getFencePath(), "", -1); + LOG_INFO(log, "Removed the TTL export index of destination {}", destination_key); +} + +ExportTTLIndexSnapshotPtr ReplicatedExportTTLIndex::getSnapshot(const zkutil::ZooKeeperPtr & zookeeper) +{ + /// Read before the index, so the index is at least as new as the version it is cached for. Every + /// change of the index bumps the version in the same transaction. + const auto version = readFenceVersion(zookeeper); + + if (auto cached = latest.get(); cached->version == version) + return cached; + + std::lock_guard lock(refresh_mutex); + if (auto cached = latest.get(); cached->version == version) + return cached; + + std::map> entries; + for (const auto & destination_key : listDestinations(zookeeper)) + entries[destination_key] = readIndex(zookeeper, destination_key); + + ProfileEvents::increment(ProfileEvents::ExportTTLIndexSnapshotRefreshes); + auto snapshot = ExportTTLIndexSnapshot::build(version, std::move(entries)); + LOG_DEBUG(log, "Read the export index at version {} of the fence: {} partition(s) with exported or claimed parts", + version, snapshot->fence->entries_by_partition.size()); + + latest.set(std::move(snapshot)); + return latest.get(); +} + +std::pair ReplicatedExportTTLIndex::getForMergeAssignment(const zkutil::ZooKeeperPtr & zookeeper) +{ + const auto snapshot = getSnapshot(zookeeper); + return {snapshot->fence, snapshot->version}; +} + +} diff --git a/src/Storages/MergeTree/ReplicatedExportTTLIndex.h b/src/Storages/MergeTree/ReplicatedExportTTLIndex.h new file mode 100644 index 000000000000..a5e5ebd33d31 --- /dev/null +++ b/src/Storages/MergeTree/ReplicatedExportTTLIndex.h @@ -0,0 +1,83 @@ +#pragma once + +#include +#include +#include +#include +#include +#include + +#include +#include +#include +#include + +namespace DB +{ + +/// The `TTL ... EXPORT` state of a `ReplicatedMergeTree` table in Keeper, and the merge fence built from it. +/// +/// Layout under ``: +/// - `export_ttl/`: holds `ExportTTLDestination` of one destination; +/// - `export_ttl//partitions/`: an `ExportTTLIndexEntry`; +/// - `export_ttl//state`: the `ExportTTLSchedulerState` of the replica that schedules; +/// - `export_ttl/scheduler_lock`: ephemeral, held by the replica that schedules TTL exports; +/// - `export_fence`: bumped in every transaction that changes an index entry. A merge is assigned +/// with a check of the version its predicate read the index at, so no merge is assigned from a +/// stale view of the export states. +class ReplicatedExportTTLIndex +{ +public: + ReplicatedExportTTLIndex(String zookeeper_path_, LoggerPtr log_); + + String getFencePath() const; + String getRootPath() const; + String getSchedulerLockPath() const; + String getDestinationPath(const String & destination_key) const; + String getIndexEntryPath(const String & destination_key, const String & partition_id) const; + String getSchedulerStatePath(const String & destination_key) const; + + /// Destination keys that have an index. + std::vector listDestinations(const zkutil::ZooKeeperPtr & zookeeper) const; + + /// Index entries of every partition exported to `destination_key`. + std::map readIndex(const zkutil::ZooKeeperPtr & zookeeper, const String & destination_key) const; + + ExportTTLVersionedEntry readIndexEntry( + const zkutil::ZooKeeperPtr & zookeeper, const String & destination_key, const String & partition_id) const; + + /// Creates the nodes of `destination_key`, so entries can be created in a transaction. + void ensureDestination(const zkutil::ZooKeeperPtr & zookeeper, const String & destination_key, const String & description) const; + + /// Appends the ops that store `entry`, with a check of `version` (see `ExportTTLVersionedEntry`), and bump the fence. + void appendUpdateEntryOps( + Coordination::Requests & ops, const String & destination_key, const ExportTTLIndexEntry & entry, int32_t version) const; + + /// Removes the index of `destination_key` and bumps the fence. Not transactional: an interrupted + /// removal leaves fewer entries, which only lifts more of the fence. + void removeDestination(const zkutil::ZooKeeperPtr & zookeeper, const String & destination_key) const; + + /// The whole index as of the current version of the fence node. It is read again only when that + /// version changed, so while nothing changes this costs one `exists`. + ExportTTLIndexSnapshotPtr getSnapshot(const zkutil::ZooKeeperPtr & zookeeper); + + /// The export states as of the returned version of the fence node. + std::pair getForMergeAssignment(const zkutil::ZooKeeperPtr & zookeeper); + + /// The export states read by the last `getSnapshot`, without reading Keeper. + ExportFencePtr getLatest() const { return latest.get()->fence; } + +private: + const String zookeeper_path; + const LoggerPtr log; + + /// Serializes reading the index again, which readers of `latest` do not wait for. + std::mutex refresh_mutex; + MultiVersion latest; + + int32_t readFenceVersion(const zkutil::ZooKeeperPtr & zookeeper) const; +}; + +using ReplicatedExportTTLIndexPtr = std::shared_ptr; + +} diff --git a/src/Storages/MergeTree/ReplicatedExportTTLScheduler.cpp b/src/Storages/MergeTree/ReplicatedExportTTLScheduler.cpp new file mode 100644 index 000000000000..7a962f171766 --- /dev/null +++ b/src/Storages/MergeTree/ReplicatedExportTTLScheduler.cpp @@ -0,0 +1,270 @@ +#include + +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#include + +namespace fs = std::filesystem; + +namespace ProfileEvents +{ + extern const Event ExportTaskZooKeeperRequests; + extern const Event ExportTaskZooKeeperMulti; +} + +namespace DB +{ + +ReplicatedExportTTLScheduler::ReplicatedExportTTLScheduler(StorageReplicatedMergeTree & storage_) + : ExportTTLScheduler(storage_) + , replicated_storage(storage_) +{ +} + +ReplicatedExportTTLScheduler::~ReplicatedExportTTLScheduler() +{ + releaseSchedulerLock(); +} + +void ReplicatedExportTTLScheduler::releaseSchedulerLock() +{ + std::lock_guard lock(lock_mutex); + lock_holder.reset(); + lock_zookeeper.reset(); +} + +bool ReplicatedExportTTLScheduler::acquireSchedulerLock() +{ + auto zookeeper = replicated_storage.tryGetZooKeeper(); + if (!zookeeper || zookeeper->expired() || replicated_storage.is_readonly) + { + releaseSchedulerLock(); + return false; + } + + std::lock_guard lock(lock_mutex); + if (lock_holder && lock_zookeeper == zookeeper) + return true; + + lock_holder.reset(); + lock_zookeeper.reset(); + + const auto & fence = *replicated_storage.export_fence; + const auto root_path = fence.getRootPath(); + if (const auto code = zookeeper->tryCreate(root_path, "", zkutil::CreateMode::Persistent); + code != Coordination::Error::ZOK && code != Coordination::Error::ZNODEEXISTS) + throw zkutil::KeeperException::fromPath(code, root_path); + + lock_holder = zkutil::EphemeralNodeHolder::tryCreate(fence.getSchedulerLockPath(), *zookeeper, replicated_storage.getReplicaName()); + if (!lock_holder) + return false; + + lock_zookeeper = zookeeper; + LOG_INFO(log, "This replica schedules the EXPORT TTL of the table"); + return true; +} + +bool ReplicatedExportTTLScheduler::isPaused() +{ + return replicated_storage.parts_mover.moves_blocker.isCancelled(); +} + +ExportTTLIndexSnapshotPtr ReplicatedExportTTLScheduler::getIndexSnapshot() +{ + return replicated_storage.export_fence->getSnapshot(replicated_storage.getZooKeeper()); +} + +std::optional ReplicatedExportTTLScheduler::readSchedulerState(const String & destination_key) +{ + String data; + if (!replicated_storage.getZooKeeper()->tryGet(replicated_storage.export_fence->getSchedulerStatePath(destination_key), data)) + return std::nullopt; + return ExportTTLSchedulerState::fromJSONString(data); +} + +void ReplicatedExportTTLScheduler::writeSchedulerState(const String & destination_key, const ExportTTLSchedulerState & state) +{ + const auto zookeeper = replicated_storage.getZooKeeper(); + const auto & fence = *replicated_storage.export_fence; + const auto path = fence.getSchedulerStatePath(destination_key); + const auto data = state.toJSONString(); + + auto code = zookeeper->trySet(path, data); + if (code == Coordination::Error::ZNONODE) + { + fence.ensureDestination(zookeeper, destination_key, destination_key); + code = zookeeper->tryCreate(path, data, zkutil::CreateMode::Persistent); + if (code == Coordination::Error::ZNODEEXISTS) + code = zookeeper->trySet(path, data); + } + + if (code != Coordination::Error::ZOK) + throw zkutil::KeeperException::fromPath(code, path); +} + +String ReplicatedExportTTLScheduler::getReplicaName() const +{ + return replicated_storage.getReplicaName(); +} + +ExportTTLScheduler::TaskState ReplicatedExportTTLScheduler::getTaskState(const String & transaction_id) +{ + using Status = ExportReplicatedMergeTreeTaskEntry::Status; + const auto to_task_status = [](Status status) -> TaskStatus + { + switch (status) + { + case Status::PENDING: return TaskStatus::PENDING; + case Status::COMPLETED: return TaskStatus::COMPLETED; + case Status::FAILED: return TaskStatus::FAILED; + case Status::KILLED: return TaskStatus::KILLED; + } + UNREACHABLE(); + }; + + /// The in-memory mirror of the tasks may lag behind Keeper, which only delays a retry. A task it + /// does not know yet, e.g. just created, is read from Keeper, so it is never taken for missing. + if (const auto tasks = replicated_storage.export_partition_manifests.get()) + { + const auto & by_transaction_id = tasks->get(); + if (const auto it = by_transaction_id.find(transaction_id); it != by_transaction_id.end()) + return TaskState{.status = to_task_status(it->status), .retry_of = it->manifest.retry_of}; + } + + const auto zookeeper = replicated_storage.getZooKeeper(); + const fs::path task_path = fs::path(replicated_storage.zookeeper_path) / "exports" / transaction_id; + const Strings paths{task_path / "status", task_path / "metadata.json"}; + + auto responses = zookeeper->tryGet(paths); + responses.waitForResponses(); + + TaskState state; + if (responses[0].error == Coordination::Error::ZNONODE) + return state; + if (responses[0].error != Coordination::Error::ZOK) + throw zkutil::KeeperException::fromPath(responses[0].error, paths[0]); + + if (const auto status = magic_enum::enum_cast(responses[0].data)) + { + state.status = to_task_status(*status); + } + else + { + /// Treated as failed: the recovery check still decides whether it committed. + LOG_WARNING(log, "Export task {} has an unknown status {}", transaction_id, responses[0].data); + state.status = TaskStatus::FAILED; + } + + if (responses[1].error == Coordination::Error::ZOK) + state.retry_of = ExportReplicatedMergeTreeTaskManifest::fromJsonString(responses[1].data).retry_of; + else if (responses[1].error != Coordination::Error::ZNONODE) + throw zkutil::KeeperException::fromPath(responses[1].error, paths[1]); + + return state; +} + +bool ReplicatedExportTTLScheduler::updateIndexEntry(const String & destination_key, const ExportTTLVersionedEntry & entry) +{ + const auto zookeeper = replicated_storage.getZooKeeper(); + const auto & fence = *replicated_storage.export_fence; + + if (entry.version < 0) + fence.ensureDestination(zookeeper, destination_key, destination_key); + + Coordination::Requests ops; + fence.appendUpdateEntryOps(ops, destination_key, entry.entry, entry.version); + + Coordination::Responses responses; + const auto code = zookeeper->tryMulti(ops, responses); + if (code == Coordination::Error::ZOK) + return true; + if (code == Coordination::Error::ZBADVERSION || code == Coordination::Error::ZNODEEXISTS || code == Coordination::Error::ZNONODE) + return false; + zkutil::KeeperMultiException::check(code, ops, responses); + return false; +} + +bool ReplicatedExportTTLScheduler::startGroup(const GroupToStart & group, const ContextPtr & context) +{ + auto zookeeper = replicated_storage.getZooKeeperAndAssertNotReadonly(); + replicated_storage.checkAllReplicasSupportExportTTL(zookeeper); + + /// Its `/log` version is checked when the task is created, so no merge can be assigned between + /// checking the parts below and claiming them. + const auto merge_predicate = replicated_storage.queue.getMergePredicate(zookeeper, PartitionIdsHint{group.partition_id}); + for (const auto & part : group.parts) + { + const auto covering_part = merge_predicate->getCoveringVirtualPart(part->name); + if (covering_part.empty()) + return false; + + const auto covering_info = MergeTreePartInfo::fromPartName(covering_part, replicated_storage.format_version); + if (covering_info.min_block != part->info.min_block || covering_info.max_block != part->info.max_block) + return false; + } + + const auto source_metadata = replicated_storage.getInMemoryMetadataPtr(context, false); + const auto destination_metadata = group.destination->getInMemoryMetadataPtr(context, false); + ExportTaskUtils::verifyExportSchemaCastable(source_metadata, destination_metadata, group.destination->getStorageID(), context); + + MergeTreeData::DataPartsVector parts(group.parts.begin(), group.parts.end()); + auto manifest = replicated_storage.buildExportTaskManifest( + group.destination->getStorageID(), group.destination, source_metadata, destination_metadata, parts, group.partition_id, context); + manifest.transaction_id = group.transaction_id; + manifest.source = ExportTaskSource::ttl; + manifest.retry_of = group.retry_of; + + const auto & fence = *replicated_storage.export_fence; + if (group.entry.version < 0) + fence.ensureDestination(zookeeper, group.destination_key, group.destination->getStorageID().getNameForLogs()); + + const auto task_path = fs::path(replicated_storage.zookeeper_path) / "exports" / group.transaction_id; + + Coordination::Requests ops; + ops.emplace_back(zkutil::makeCheckRequest(fs::path(replicated_storage.zookeeper_path) / "log", merge_predicate->getVersion())); + ExportTaskUtils::appendCreateExportTaskOps(ops, task_path, manifest); + fence.appendUpdateEntryOps(ops, group.destination_key, group.entry.entry, group.entry.version); + + ProfileEvents::increment(ProfileEvents::ExportTaskZooKeeperRequests); + ProfileEvents::increment(ProfileEvents::ExportTaskZooKeeperMulti); + Coordination::Responses responses; + const auto code = zookeeper->tryMulti(ops, responses); + + if (code == Coordination::Error::ZOK) + { + if (replicated_storage.export_task_updating_task) + replicated_storage.export_task_updating_task->schedule(); + return true; + } + + if (code == Coordination::Error::ZBADVERSION || code == Coordination::Error::ZNODEEXISTS) + return false; + + zkutil::KeeperMultiException::check(code, ops, responses); + return false; +} + +bool ReplicatedExportTTLScheduler::isPartBeingMerged(const MergeTreeDataPartPtr & part) +{ + return replicated_storage.queue.isGoingToBeMergedWithOtherParts(part->info); +} + +void ReplicatedExportTTLScheduler::killTask(const String & transaction_id) +{ + replicated_storage.killExportTask(transaction_id); +} + +void ReplicatedExportTTLScheduler::removeDestination(const String & destination_key) +{ + replicated_storage.export_fence->removeDestination(replicated_storage.getZooKeeper(), destination_key); +} + +} diff --git a/src/Storages/MergeTree/ReplicatedExportTTLScheduler.h b/src/Storages/MergeTree/ReplicatedExportTTLScheduler.h new file mode 100644 index 000000000000..58fe00905e74 --- /dev/null +++ b/src/Storages/MergeTree/ReplicatedExportTTLScheduler.h @@ -0,0 +1,49 @@ +#pragma once + +#include +#include + +#include + +namespace DB +{ + +class StorageReplicatedMergeTree; + +/// `ExportTTLScheduler` of a `ReplicatedMergeTree`: the index is in Keeper (see +/// `ReplicatedExportTTLIndex`), a group is claimed in the transaction that creates its task, +/// with a check that no merge was assigned meanwhile, and one replica schedules at a time. +class ReplicatedExportTTLScheduler final : public ExportTTLScheduler +{ +public: + explicit ReplicatedExportTTLScheduler(StorageReplicatedMergeTree & storage_); + ~ReplicatedExportTTLScheduler() override; + + /// E.g. when the replica goes read-only, so another replica can take over. + void releaseSchedulerLock(); + +protected: + bool acquireSchedulerLock() override; + bool isPaused() override; + ExportTTLIndexSnapshotPtr getIndexSnapshot() override; + bool identifiesDestinationByUUID() const override { return false; } + std::optional readSchedulerState(const String & destination_key) override; + void writeSchedulerState(const String & destination_key, const ExportTTLSchedulerState & state) override; + String getReplicaName() const override; + TaskState getTaskState(const String & transaction_id) override; + bool updateIndexEntry(const String & destination_key, const ExportTTLVersionedEntry & entry) override; + bool startGroup(const GroupToStart & group, const ContextPtr & context) override; + bool isPartBeingMerged(const MergeTreeDataPartPtr & part) override; + void killTask(const String & transaction_id) override; + void removeDestination(const String & destination_key) override; + +private: + StorageReplicatedMergeTree & replicated_storage; + + std::mutex lock_mutex; + /// Keeps the session the lock was created in alive: the holder refers to it. + zkutil::ZooKeeperPtr lock_zookeeper; + zkutil::EphemeralNodeHolderPtr lock_holder; +}; + +} diff --git a/src/Storages/MergeTree/ExportPartitionTaskScheduler.cpp b/src/Storages/MergeTree/ReplicatedExportTaskScheduler.cpp similarity index 60% rename from src/Storages/MergeTree/ExportPartitionTaskScheduler.cpp rename to src/Storages/MergeTree/ReplicatedExportTaskScheduler.cpp index f491c565fad1..f7b055d93215 100644 --- a/src/Storages/MergeTree/ExportPartitionTaskScheduler.cpp +++ b/src/Storages/MergeTree/ReplicatedExportTaskScheduler.cpp @@ -1,4 +1,4 @@ -#include +#include #include #include #include @@ -8,7 +8,7 @@ #include #include #include -#include "Storages/MergeTree/ExportPartitionUtils.h" +#include "Storages/MergeTree/ExportTaskUtils.h" #include "Storages/MergeTree/MergeTreePartExportManifest.h" #include "Formats/FormatFactory.h" #include @@ -17,13 +17,13 @@ namespace ProfileEvents { - extern const Event ExportPartitionZooKeeperRequests; - extern const Event ExportPartitionZooKeeperGet; - extern const Event ExportPartitionZooKeeperGetChildren; - extern const Event ExportPartitionZooKeeperCreate; - extern const Event ExportPartitionZooKeeperSet; - extern const Event ExportPartitionZooKeeperRemove; - extern const Event ExportPartitionZooKeeperMulti; + extern const Event ExportTaskZooKeeperRequests; + extern const Event ExportTaskZooKeeperGet; + extern const Event ExportTaskZooKeeperGetChildren; + extern const Event ExportTaskZooKeeperCreate; + extern const Event ExportTaskZooKeeperSet; + extern const Event ExportTaskZooKeeperRemove; + extern const Event ExportTaskZooKeeperMulti; extern const Event ExportPartsRejectedByMemoryLimit; } @@ -42,12 +42,12 @@ namespace ErrorCodes extern const int LOGICAL_ERROR; } -ExportPartitionTaskScheduler::ExportPartitionTaskScheduler(StorageReplicatedMergeTree & storage_) +ReplicatedExportTaskScheduler::ReplicatedExportTaskScheduler(StorageReplicatedMergeTree & storage_) : storage(storage_) { } -std::optional ExportPartitionTaskScheduler::run() +std::optional ReplicatedExportTaskScheduler::run() { std::optional earliest_backoff_retry; @@ -56,7 +56,7 @@ std::optional ExportPartitionTaskScheduler::run() /// this is subject to TOCTOU - but for now we choose to live with it. if (available_move_executors == 0) { - LOG_DEBUG(storage.log, "ExportPartition scheduler task: No available move executors, skipping"); + LOG_DEBUG(storage.log, "Export task scheduler: No available move executors, skipping"); return earliest_backoff_retry; } @@ -67,14 +67,14 @@ std::optional ExportPartitionTaskScheduler::run() { ProfileEvents::increment(ProfileEvents::ExportPartsRejectedByMemoryLimit); LOG_TRACE(storage.log, - "ExportPartition scheduler task: Reached memory limit for the background tasks ({}), " + "Export task scheduler: Reached memory limit for the background tasks ({}), " "so won't select new parts to export. Current background tasks memory usage: {}.", formatReadableSizeWithBinarySuffix(background_memory_tracker.getSoftLimit()), formatReadableSizeWithBinarySuffix(background_memory_tracker.get())); return earliest_backoff_retry; } - LOG_DEBUG(storage.log, "ExportPartition scheduler task: Available move executors: {}", available_move_executors); + LOG_DEBUG(storage.log, "Export task scheduler: Available move executors: {}", available_move_executors); std::size_t scheduled_exports_count = 0; @@ -90,26 +90,26 @@ std::optional ExportPartitionTaskScheduler::run() auto zk = storage.getZooKeeper(); - pruneLocalBackoff(model->get()); + pruneLocalBackoff(model->get()); // Iterate sorted by create_time - for (const auto & entry : model->get()) + for (const auto & entry : model->get()) { if (scheduled_exports_count >= available_move_executors) { - LOG_DEBUG(storage.log, "ExportPartition scheduler task: Scheduled exports count is greater than available move executors, skipping"); + LOG_DEBUG(storage.log, "Export task scheduler: Scheduled exports count is greater than available move executors, skipping"); break; } /// No need to query zk for status if the local one is not PENDING - if (entry.status != ExportReplicatedMergeTreePartitionTaskEntry::Status::PENDING) + if (entry.status != ExportReplicatedMergeTreeTaskEntry::Status::PENDING) { - LOG_DEBUG(storage.log, "ExportPartition scheduler task: Skipping... Local status is {}", magic_enum::enum_name(entry.status).data()); + LOG_DEBUG(storage.log, "Export task scheduler: Skipping... Local status is {}", magic_enum::enum_name(entry.status).data()); continue; } const auto & manifest = entry.manifest; - const auto key = entry.getCompositeKey(); + const auto key = entry.getTransactionId(); const auto database = storage.getContext()->resolveDatabase(manifest.destination_database); const auto & table = manifest.destination_table; @@ -119,60 +119,60 @@ std::optional ExportPartitionTaskScheduler::run() if (!destination_storage) { - LOG_WARNING(storage.log, "ExportPartition scheduler task: Failed to reconstruct destination storage: {}, skipping", destination_storage_id.getNameForLogs()); + LOG_WARNING(storage.log, "Export task scheduler: Failed to reconstruct destination storage: {}, skipping", destination_storage_id.getNameForLogs()); continue; } - ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRequests); - ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperGet); + ProfileEvents::increment(ProfileEvents::ExportTaskZooKeeperRequests); + ProfileEvents::increment(ProfileEvents::ExportTaskZooKeeperGet); std::string status_in_zk_string; if (!zk->tryGet(fs::path(storage.zookeeper_path) / "exports" / key / "status", status_in_zk_string)) { - LOG_WARNING(storage.log, "ExportPartition scheduler task: Failed to get status, skipping"); + LOG_WARNING(storage.log, "Export task scheduler: Failed to get status, skipping"); continue; } - const auto status_in_zk = magic_enum::enum_cast(status_in_zk_string); + const auto status_in_zk = magic_enum::enum_cast(status_in_zk_string); if (!status_in_zk) { - LOG_WARNING(storage.log, "ExportPartition scheduler task: Failed to get status from zk, skipping"); + LOG_WARNING(storage.log, "Export task scheduler: Failed to get status from zk, skipping"); continue; } - if (status_in_zk.value() != ExportReplicatedMergeTreePartitionTaskEntry::Status::PENDING) + if (status_in_zk.value() != ExportReplicatedMergeTreeTaskEntry::Status::PENDING) { - LOG_DEBUG(storage.log, "ExportPartition scheduler task: Skipping {}... Status from zk is {}", key, magic_enum::enum_name(status_in_zk.value()).data()); + LOG_DEBUG(storage.log, "Export task scheduler: Skipping {}... Status from zk is {}", key, magic_enum::enum_name(status_in_zk.value()).data()); continue; } - ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRequests); - ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperGetChildren); + ProfileEvents::increment(ProfileEvents::ExportTaskZooKeeperRequests); + ProfileEvents::increment(ProfileEvents::ExportTaskZooKeeperGetChildren); std::vector parts_in_processing_or_pending; if (Coordination::Error::ZOK != zk->tryGetChildren(fs::path(storage.zookeeper_path) / "exports" / key / "processing", parts_in_processing_or_pending)) { - LOG_WARNING(storage.log, "ExportPartition scheduler task: Failed to get parts in processing or pending, skipping"); + LOG_WARNING(storage.log, "Export task scheduler: Failed to get parts in processing or pending, skipping"); continue; } if (parts_in_processing_or_pending.empty()) { - LOG_DEBUG(storage.log, "ExportPartition scheduler task: No parts in processing or pending, skipping"); + LOG_DEBUG(storage.log, "Export task scheduler: No parts in processing or pending, skipping"); continue; } /// shuffle the parts to reduce the risk of lock collisions std::shuffle(parts_in_processing_or_pending.begin(), parts_in_processing_or_pending.end(), rng); - ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRequests); - ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperGetChildren); + ProfileEvents::increment(ProfileEvents::ExportTaskZooKeeperRequests); + ProfileEvents::increment(ProfileEvents::ExportTaskZooKeeperGetChildren); std::vector locked_parts; if (Coordination::Error::ZOK != zk->tryGetChildren(fs::path(storage.zookeeper_path) / "exports" / key / "locks", locked_parts)) { - LOG_WARNING(storage.log, "ExportPartition scheduler task: Failed to get locked parts, skipping"); + LOG_WARNING(storage.log, "Export task scheduler: Failed to get locked parts, skipping"); continue; } @@ -184,13 +184,13 @@ std::optional ExportPartitionTaskScheduler::run() { if (scheduled_exports_count >= available_move_executors) { - LOG_DEBUG(storage.log, "ExportPartition scheduler task: Scheduled exports count is greater than available move executors, skipping"); + LOG_DEBUG(storage.log, "Export task scheduler: Scheduled exports count is greater than available move executors, skipping"); break; } if (locked_parts_set.contains(zk_part_name)) { - LOG_DEBUG(storage.log, "ExportPartition scheduler task: Part {} is locked, skipping", zk_part_name); + LOG_DEBUG(storage.log, "Export task scheduler: Part {} is locked, skipping", zk_part_name); continue; } @@ -202,29 +202,29 @@ std::optional ExportPartitionTaskScheduler::run() const auto part = storage.getPartIfExists(zk_part_name, {MergeTreeDataPartState::Active, MergeTreeDataPartState::Outdated}); if (!part) { - LOG_DEBUG(storage.log, "ExportPartition scheduler task: Part {} not found locally, skipping", zk_part_name); + LOG_DEBUG(storage.log, "Export task scheduler: Part {} not found locally, skipping", zk_part_name); continue; } - LOG_INFO(storage.log, "ExportPartition scheduler task: Scheduling part export: {}", zk_part_name); + LOG_INFO(storage.log, "Export task scheduler: Scheduling part export: {}", zk_part_name); - auto context = ExportPartitionUtils::getContextCopyWithTaskSettings(storage.getContext(), manifest); + auto context = ExportTaskUtils::getContextCopyWithTaskSettings(storage.getContext(), manifest); try { - LOG_DEBUG(storage.log, "ExportPartition scheduler task: Exporting part to table"); + LOG_DEBUG(storage.log, "Export task scheduler: Exporting part to table"); - LOG_INFO(storage.log, "ExportPartition scheduler task: Attempting to lock part: {}", zk_part_name); + LOG_INFO(storage.log, "Export task scheduler: Attempting to lock part: {}", zk_part_name); - ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRequests); - ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperCreate); + ProfileEvents::increment(ProfileEvents::ExportTaskZooKeeperRequests); + ProfileEvents::increment(ProfileEvents::ExportTaskZooKeeperCreate); if (Coordination::Error::ZOK != zk->tryCreate(fs::path(storage.zookeeper_path) / "exports" / key / "locks" / zk_part_name, storage.replica_name, zkutil::CreateMode::Ephemeral)) { - LOG_INFO(storage.log, "ExportPartition scheduler task: Failed to lock part {}, skipping", zk_part_name); + LOG_INFO(storage.log, "Export task scheduler: Failed to lock part {}, skipping", zk_part_name); continue; } - LOG_INFO(storage.log, "ExportPartition scheduler task: Locked part: {}", zk_part_name); + LOG_INFO(storage.log, "Export task scheduler: Locked part: {}", zk_part_name); storage.exportPartToTable( part->name, @@ -244,8 +244,8 @@ std::optional ExportPartitionTaskScheduler::run() catch (const Exception &) { tryLogCurrentException(__PRETTY_FUNCTION__); - ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRequests); - ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRemove); + ProfileEvents::increment(ProfileEvents::ExportTaskZooKeeperRequests); + ProfileEvents::increment(ProfileEvents::ExportTaskZooKeeperRemove); zk->tryRemove(fs::path(storage.zookeeper_path) / "exports" / key / "locks" / zk_part_name); /// Dispatch-time failure (e.g. Keeper node full). We do not arm the local /// back-off here: the export never started, so the part stays immediately @@ -257,7 +257,7 @@ std::optional ExportPartitionTaskScheduler::run() return earliest_backoff_retry; } -bool ExportPartitionTaskScheduler::shouldBackOff( +bool ReplicatedExportTaskScheduler::shouldBackOff( const std::string & transaction_id, const std::string & part_name, time_t now, @@ -273,16 +273,16 @@ bool ExportPartitionTaskScheduler::shouldBackOff( return false; const auto next_retry_time = part_it->second.next_retry_time; - LOG_TRACE(storage.log, "ExportPartition scheduler task: Part {} is backing off locally, next retry at {} (now {}), skipping", part_name, next_retry_time, now); + LOG_TRACE(storage.log, "Export task scheduler: Part {} is backing off locally, next retry at {} (now {}), skipping", part_name, next_retry_time, now); earliest_backoff_retry = earliest_backoff_retry ? std::min(*earliest_backoff_retry, next_retry_time) : next_retry_time; return true; } -time_t ExportPartitionTaskScheduler::registerLocalBackoff( +time_t ReplicatedExportTaskScheduler::registerLocalBackoff( const std::string & transaction_id, const std::string & part_name, - const ExportReplicatedMergeTreePartitionManifest & manifest) + const ExportReplicatedMergeTreeTaskManifest & manifest) { std::lock_guard lock(local_backoff_mutex); @@ -291,7 +291,7 @@ time_t ExportPartitionTaskScheduler::registerLocalBackoff( auto & backoff = parts.try_emplace(part_name).first->second; ++backoff.attempts; - const auto backoff_seconds = ExportPartitionUtils::computeRetryBackoffSeconds( + const auto backoff_seconds = ExportTaskUtils::computeRetryBackoffSeconds( backoff.attempts, manifest.retry_initial_backoff_seconds, manifest.retry_max_backoff_seconds); const auto now = time(nullptr); /// Clamp so a huge configured back-off cannot overflow time_t (now is a normal wall-clock value). @@ -300,7 +300,7 @@ time_t ExportPartitionTaskScheduler::registerLocalBackoff( return backoff.next_retry_time; } -void ExportPartitionTaskScheduler::clearLocalBackoff(const std::string & transaction_id, const std::string & part_name) +void ReplicatedExportTaskScheduler::clearLocalBackoff(const std::string & transaction_id, const std::string & part_name) { std::lock_guard lock(local_backoff_mutex); if (const auto task_it = local_backoff.find(transaction_id); task_it != local_backoff.end()) @@ -311,13 +311,13 @@ void ExportPartitionTaskScheduler::clearLocalBackoff(const std::string & transac } } -void ExportPartitionTaskScheduler::pruneLocalBackoff(const ExportPartitionTaskEntriesContainer::index::type & model) +void ReplicatedExportTaskScheduler::pruneLocalBackoff(const ExportTaskEntriesContainer::index::type & model) { std::lock_guard lock(local_backoff_mutex); for (auto it = local_backoff.begin(); it != local_backoff.end();) { const auto found = model.find(it->first); - if (found != model.end() && found->status == ExportReplicatedMergeTreePartitionTaskEntry::Status::PENDING) + if (found != model.end() && found->status == ExportReplicatedMergeTreeTaskEntry::Status::PENDING) { ++it; continue; @@ -327,7 +327,7 @@ void ExportPartitionTaskScheduler::pruneLocalBackoff(const ExportPartitionTaskEn } } -ExportPartitionTaskScheduler::LocalBackoffMap ExportPartitionTaskScheduler::getLocalBackoffSnapshot() const +ReplicatedExportTaskScheduler::LocalBackoffMap ReplicatedExportTaskScheduler::getLocalBackoffSnapshot() const { LocalBackoffMap snapshot; @@ -344,15 +344,15 @@ ExportPartitionTaskScheduler::LocalBackoffMap ExportPartitionTaskScheduler::getL return snapshot; } -void ExportPartitionTaskScheduler::handlePartExportCompletion( +void ReplicatedExportTaskScheduler::handlePartExportCompletion( const std::string & export_key, const std::string & part_name, - const ExportReplicatedMergeTreePartitionManifest & manifest, + const ExportReplicatedMergeTreeTaskManifest & manifest, const StoragePtr & destination_storage, const MergeTreePartExportManifest::CompletionCallbackResult & result) { /// Invoked from MergeTreeBackgroundExecutor threads, so the component is not inherited from selectPartsToExport. - auto component_guard = Coordination::setCurrentComponent("ExportPartitionTaskScheduler::handlePartExportCompletion"); + auto component_guard = Coordination::setCurrentComponent("ReplicatedExportTaskScheduler::handlePartExportCompletion"); const auto export_path = fs::path(storage.zookeeper_path) / "exports" / export_key; const auto processing_parts_path = export_path / "processing"; @@ -369,8 +369,8 @@ void ExportPartitionTaskScheduler::handlePartExportCompletion( } } -void ExportPartitionTaskScheduler::handlePartExportSuccess( - const ExportReplicatedMergeTreePartitionManifest & manifest, +void ReplicatedExportTaskScheduler::handlePartExportSuccess( + const ExportReplicatedMergeTreeTaskManifest & manifest, const StoragePtr & destination_storage, const std::filesystem::path & processing_parts_path, const std::filesystem::path & processed_part_path, @@ -380,39 +380,39 @@ void ExportPartitionTaskScheduler::handlePartExportSuccess( const std::vector & relative_paths_in_destination_storage ) { - LOG_INFO(storage.log, "ExportPartition scheduler task: Part {} exported successfully, paths size: {}", part_name, relative_paths_in_destination_storage.size()); + LOG_INFO(storage.log, "Export task scheduler: Part {} exported successfully, paths size: {}", part_name, relative_paths_in_destination_storage.size()); for (const auto & relative_path_in_destination_storage : relative_paths_in_destination_storage) { - LOG_DEBUG(storage.log, "ExportPartition scheduler task: {}", relative_path_in_destination_storage); + LOG_DEBUG(storage.log, "Export task scheduler: {}", relative_path_in_destination_storage); } if (!tryToMovePartToProcessed(export_path, processing_parts_path, processed_part_path, part_name, relative_paths_in_destination_storage, zk)) { - LOG_WARNING(storage.log, "ExportPartition scheduler task: Failed to move part to processed, will not commit export partition"); + LOG_WARNING(storage.log, "Export task scheduler: Failed to move part to processed, will not commit export partition"); return; } /// Part is done on this replica; drop any local back-off state we held for it. clearLocalBackoff(manifest.transaction_id, part_name); - LOG_INFO(storage.log, "ExportPartition scheduler task: Marked part export {} as completed", part_name); + LOG_INFO(storage.log, "Export task scheduler: Marked part export {} as completed", part_name); if (!areAllPartsProcessed(export_path, zk)) { return; } - LOG_INFO(storage.log, "ExportPartition scheduler task: All parts are processed, will try to commit export partition"); + LOG_INFO(storage.log, "Export task scheduler: All parts are processed, will try to commit export partition"); try { - auto context = ExportPartitionUtils::getContextCopyWithTaskSettings(storage.getContext(), manifest); - ExportPartitionUtils::commit(manifest, destination_storage, zk, storage.log.load(), export_path, context, storage, storage.replica_name); + auto context = ExportTaskUtils::getContextCopyWithTaskSettings(storage.getContext(), manifest); + ExportTaskUtils::commit(manifest, destination_storage, zk, storage.log.load(), export_path, context, storage, storage.replica_name, *storage.export_fence); } catch (const Exception & e) { - LOG_INFO(storage.log, "ExportPartition scheduler task: Caught exception while committing export partition, {}", e.message()); + LOG_INFO(storage.log, "Export task scheduler: Caught exception while committing export partition, {}", e.message()); /// Classify the commit failure: a non-retryable error (e.g. schema/spec mismatch) /// transitions the task to FAILED immediately; a retryable one (transient catalog or @@ -420,7 +420,7 @@ void ExportPartitionTaskScheduler::handlePartExportSuccess( /// commit is retried until the absolute task timeout. /// The exception is recorded in /last_exception via appendExceptionOps /// inside the same multi as the (possible) FAILED set. - const bool became_failed = ExportPartitionUtils::handleCommitFailure( + const bool became_failed = ExportTaskUtils::handleCommitFailure( zk, export_path, e.code(), @@ -431,41 +431,41 @@ void ExportPartitionTaskScheduler::handlePartExportSuccess( if (became_failed) { LOG_WARNING(storage.log, - "ExportPartition scheduler task: Commit for {} transitioned to FAILED due to non-retryable error (code {})", + "Export task scheduler: Commit for {} transitioned to FAILED due to non-retryable error (code {})", export_path.string(), e.code()); } } } -void ExportPartitionTaskScheduler::handlePartExportFailure( +void ReplicatedExportTaskScheduler::handlePartExportFailure( const std::string & part_name, const std::filesystem::path & export_path, const zkutil::ZooKeeperPtr & zk, const std::optional & exception, - const ExportReplicatedMergeTreePartitionManifest & manifest + const ExportReplicatedMergeTreeTaskManifest & manifest ) { - LOG_INFO(storage.log, "ExportPartition scheduler task: Part {} export failed", part_name); + LOG_INFO(storage.log, "Export task scheduler: Part {} export failed", part_name); if (!exception) { - throw Exception(ErrorCodes::LOGICAL_ERROR, "ExportPartition scheduler task: No exception provided for error handling. Sounds like a bug"); + throw Exception(ErrorCodes::LOGICAL_ERROR, "Export task scheduler: No exception provided for error handling. Sounds like a bug"); } Coordination::Stat locked_by_stat; std::string locked_by; - ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRequests); - ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperGet); + ProfileEvents::increment(ProfileEvents::ExportTaskZooKeeperRequests); + ProfileEvents::increment(ProfileEvents::ExportTaskZooKeeperGet); if (!zk->tryGet(export_path / "locks" / part_name, locked_by, &locked_by_stat)) { - LOG_DEBUG(storage.log, "ExportPartition scheduler task: Part {} is not locked by any replica, will not increment error counts", part_name); + LOG_DEBUG(storage.log, "Export task scheduler: Part {} is not locked by any replica, will not increment error counts", part_name); return; } if (locked_by != storage.replica_name) { - LOG_DEBUG(storage.log, "ExportPartition scheduler task: Part {} is locked by another replica, will not increment error counts", part_name); + LOG_DEBUG(storage.log, "Export task scheduler: Part {} is locked by another replica, will not increment error counts", part_name); return; } @@ -480,8 +480,8 @@ void ExportPartitionTaskScheduler::handlePartExportFailure( static constexpr std::size_t max_lock_release_retries = 3; while (retry_count < max_lock_release_retries) { - ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRequests); - ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRemove); + ProfileEvents::increment(ProfileEvents::ExportTaskZooKeeperRequests); + ProfileEvents::increment(ProfileEvents::ExportTaskZooKeeperRemove); const auto removal_code = zk->tryRemove(export_path / "locks" / part_name, locked_by_stat.version); @@ -492,14 +492,14 @@ void ExportPartitionTaskScheduler::handlePartExportFailure( if (Coordination::Error::ZBADVERSION == removal_code) { - LOG_DEBUG(storage.log, "ExportPartition scheduler task: Part {} lock version mismatch, will not increment error counts", part_name); + LOG_DEBUG(storage.log, "Export task scheduler: Part {} lock version mismatch, will not increment error counts", part_name); break; } retry_count++; } - LOG_INFO(storage.log, "ExportPartition scheduler task: Part {} export was cancelled, skipping error handling", part_name); + LOG_INFO(storage.log, "Export task scheduler: Part {} export was cancelled, skipping error handling", part_name); return; } @@ -507,22 +507,22 @@ void ExportPartitionTaskScheduler::handlePartExportFailure( Coordination::Stat status_stat; std::string current_status; - ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRequests); - ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperGet); + ProfileEvents::increment(ProfileEvents::ExportTaskZooKeeperRequests); + ProfileEvents::increment(ProfileEvents::ExportTaskZooKeeperGet); if (!zk->tryGet(status_path, current_status, &status_stat)) { - LOG_DEBUG(storage.log, "ExportPartition scheduler task: /status missing for {}, skipping failure bookkeeping", export_path.string()); + LOG_DEBUG(storage.log, "Export task scheduler: /status missing for {}, skipping failure bookkeeping", export_path.string()); return; } - const auto status = magic_enum::enum_cast(current_status); - if (!status || *status != ExportReplicatedMergeTreePartitionTaskEntry::Status::PENDING) + const auto status = magic_enum::enum_cast(current_status); + if (!status || *status != ExportReplicatedMergeTreeTaskEntry::Status::PENDING) { - LOG_DEBUG(storage.log, "ExportPartition scheduler task: /status for {} is {} (not PENDING), skipping failure bookkeeping", export_path.string(), current_status); + LOG_DEBUG(storage.log, "Export task scheduler: /status for {} is {} (not PENDING), skipping failure bookkeeping", export_path.string(), current_status); return; } - const bool non_retryable = ExportPartitionUtils::isNonRetryableExportError(exception->code()); + const bool non_retryable = ExportTaskUtils::isNonRetryableExportError(exception->code()); Coordination::Requests ops; @@ -534,25 +534,25 @@ void ExportPartitionTaskScheduler::handlePartExportFailure( /// so fail the whole task immediately instead of waiting for the absolute timeout. ops.emplace_back(zkutil::makeSetRequest( status_path, - String(magic_enum::enum_name(ExportReplicatedMergeTreePartitionTaskEntry::Status::FAILED)).data(), + String(magic_enum::enum_name(ExportReplicatedMergeTreeTaskEntry::Status::FAILED)).data(), status_stat.version)); - LOG_WARNING(storage.log, "ExportPartition scheduler task: Part {} failed with non-retryable error (code {}), failing the entire task", part_name, exception->code()); + LOG_WARNING(storage.log, "Export task scheduler: Part {} failed with non-retryable error (code {}), failing the entire task", part_name, exception->code()); } else { - LOG_DEBUG(storage.log, "ExportPartition scheduler task: Part {} failed with retryable error (code {}), will back off and retry until the task timeout", part_name, exception->code()); + LOG_DEBUG(storage.log, "Export task scheduler: Part {} failed with retryable error (code {}), will back off and retry until the task timeout", part_name, exception->code()); } - ExportPartitionUtils::appendExceptionOps( + ExportTaskUtils::appendExceptionOps( ops, zk, export_path, storage.replica_name, part_name, exception->message(), storage.log.load()); - ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRequests); - ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperMulti); + ProfileEvents::increment(ProfileEvents::ExportTaskZooKeeperRequests); + ProfileEvents::increment(ProfileEvents::ExportTaskZooKeeperMulti); Coordination::Responses responses; if (Coordination::Error::ZOK != zk->tryMulti(ops, responses)) { - LOG_WARNING(storage.log, "ExportPartition scheduler task: All failure mechanism failed, will not try to update it"); + LOG_WARNING(storage.log, "Export task scheduler: All failure mechanism failed, will not try to update it"); return; } @@ -561,13 +561,13 @@ void ExportPartitionTaskScheduler::handlePartExportFailure( if (!non_retryable) { const auto next_retry_time = registerLocalBackoff(manifest.transaction_id, part_name, manifest); - LOG_INFO(storage.log, "ExportPartition scheduler task: Part {} backing off locally, next retry at {}", part_name, next_retry_time); + LOG_INFO(storage.log, "Export task scheduler: Part {} backing off locally, next retry at {}", part_name, next_retry_time); } - LOG_INFO(storage.log, "ExportPartition scheduler task: Successfully recorded failure for part {}", part_name); + LOG_INFO(storage.log, "Export task scheduler: Successfully recorded failure for part {}", part_name); } -bool ExportPartitionTaskScheduler::tryToMovePartToProcessed( +bool ReplicatedExportTaskScheduler::tryToMovePartToProcessed( const std::filesystem::path & export_path, const std::filesystem::path & processing_parts_path, const std::filesystem::path & processed_part_path, @@ -579,11 +579,11 @@ bool ExportPartitionTaskScheduler::tryToMovePartToProcessed( Coordination::Stat locked_by_stat; std::string locked_by; - ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRequests); - ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperGet); + ProfileEvents::increment(ProfileEvents::ExportTaskZooKeeperRequests); + ProfileEvents::increment(ProfileEvents::ExportTaskZooKeeperGet); if (!zk->tryGet(export_path / "locks" / part_name, locked_by, &locked_by_stat)) { - LOG_DEBUG(storage.log, "ExportPartition scheduler task: Part {} is not locked by any replica, will not commit or set it as completed", part_name); + LOG_DEBUG(storage.log, "Export task scheduler: Part {} is not locked by any replica, will not commit or set it as completed", part_name); return false; } @@ -591,13 +591,13 @@ bool ExportPartitionTaskScheduler::tryToMovePartToProcessed( /// I guess we should not throw if file already exists for export partition, hard coded. if (locked_by != storage.replica_name) { - LOG_DEBUG(storage.log, "ExportPartition scheduler task: Part {} is locked by another replica, will not commit or set it as completed", part_name); + LOG_DEBUG(storage.log, "Export task scheduler: Part {} is locked by another replica, will not commit or set it as completed", part_name); return false; } Coordination::Requests requests; - ExportReplicatedMergeTreePartitionProcessedPartEntry processed_part_entry; + ExportReplicatedMergeTreeProcessedPartEntry processed_part_entry; processed_part_entry.part_name = part_name; processed_part_entry.paths_in_destination = relative_paths_in_destination_storage; processed_part_entry.finished_by = storage.replica_name; @@ -606,36 +606,36 @@ bool ExportPartitionTaskScheduler::tryToMovePartToProcessed( requests.emplace_back(zkutil::makeCreateRequest(processed_part_path, processed_part_entry.toJsonString(), zkutil::CreateMode::Persistent)); requests.emplace_back(zkutil::makeRemoveRequest(export_path / "locks" / part_name, locked_by_stat.version)); - ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRequests); - ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperMulti); + ProfileEvents::increment(ProfileEvents::ExportTaskZooKeeperRequests); + ProfileEvents::increment(ProfileEvents::ExportTaskZooKeeperMulti); Coordination::Responses responses; if (Coordination::Error::ZOK != zk->tryMulti(requests, responses)) { /// todo arthur remember what to do here - LOG_WARNING(storage.log, "ExportPartition scheduler task: Failed to update export path, skipping"); + LOG_WARNING(storage.log, "Export task scheduler: Failed to update export path, skipping"); return false; } return true; } -bool ExportPartitionTaskScheduler::areAllPartsProcessed( +bool ReplicatedExportTaskScheduler::areAllPartsProcessed( const std::filesystem::path & export_path, const zkutil::ZooKeeperPtr & zk) { - ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRequests); - ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperGetChildren); + ProfileEvents::increment(ProfileEvents::ExportTaskZooKeeperRequests); + ProfileEvents::increment(ProfileEvents::ExportTaskZooKeeperGetChildren); Strings parts_in_processing_or_pending; if (Coordination::Error::ZOK != zk->tryGetChildren(export_path / "processing", parts_in_processing_or_pending)) { - LOG_WARNING(storage.log, "ExportPartition scheduler task: Failed to get parts in processing or pending, will not try to commit export partition"); + LOG_WARNING(storage.log, "Export task scheduler: Failed to get parts in processing or pending, will not try to commit export partition"); return false; } if (!parts_in_processing_or_pending.empty()) { - LOG_DEBUG(storage.log, "ExportPartition scheduler task: There are still parts in processing or pending, will not try to commit export partition"); + LOG_DEBUG(storage.log, "Export task scheduler: There are still parts in processing or pending, will not try to commit export partition"); return false; } diff --git a/src/Storages/MergeTree/ExportPartitionTaskScheduler.h b/src/Storages/MergeTree/ReplicatedExportTaskScheduler.h similarity index 85% rename from src/Storages/MergeTree/ExportPartitionTaskScheduler.h rename to src/Storages/MergeTree/ReplicatedExportTaskScheduler.h index ca518307f2b8..95637fd23252 100644 --- a/src/Storages/MergeTree/ExportPartitionTaskScheduler.h +++ b/src/Storages/MergeTree/ReplicatedExportTaskScheduler.h @@ -7,7 +7,7 @@ #include #include #include -#include +#include #include namespace DB @@ -16,13 +16,13 @@ namespace DB class Exception; class StorageReplicatedMergeTree; -struct ExportReplicatedMergeTreePartitionManifest; +struct ExportReplicatedMergeTreeTaskManifest; /// todo arthur remember to add check(lock, version) when updating stuff because maybe if we believe we have the lock, we might not actually have it -class ExportPartitionTaskScheduler +class ReplicatedExportTaskScheduler { public: - ExportPartitionTaskScheduler(StorageReplicatedMergeTree & storage); + ReplicatedExportTaskScheduler(StorageReplicatedMergeTree & storage); /// Returns the earliest future back-off deadline (unix seconds) among parts that were skipped /// this tick purely because they are still backing off, or nullopt if none. The caller can use @@ -35,12 +35,12 @@ class ExportPartitionTaskScheduler void handlePartExportCompletion( const std::string & export_key, const std::string & part_name, - const ExportReplicatedMergeTreePartitionManifest & manifest, + const ExportReplicatedMergeTreeTaskManifest & manifest, const StoragePtr & destination_storage, const MergeTreePartExportManifest::CompletionCallbackResult & result); void handlePartExportSuccess( - const ExportReplicatedMergeTreePartitionManifest & manifest, + const ExportReplicatedMergeTreeTaskManifest & manifest, const StoragePtr & destination_storage, const std::filesystem::path & processing_parts_path, const std::filesystem::path & processed_part_path, @@ -55,7 +55,7 @@ class ExportPartitionTaskScheduler const std::filesystem::path & export_path, const zkutil::ZooKeeperPtr & zk, const std::optional & exception, - const ExportReplicatedMergeTreePartitionManifest & manifest); + const ExportReplicatedMergeTreeTaskManifest & manifest); bool tryToMovePartToProcessed( const std::filesystem::path & export_path, @@ -99,17 +99,17 @@ class ExportPartitionTaskScheduler time_t registerLocalBackoff( const std::string & transaction_id, const std::string & part_name, - const ExportReplicatedMergeTreePartitionManifest & manifest); + const ExportReplicatedMergeTreeTaskManifest & manifest); /// Drop any back-off state for parts of (transaction_id) once they succeed or the task ends. void clearLocalBackoff(const std::string & transaction_id, const std::string & part_name); /// Remove back-off state for tasks whose transaction_id is no longer PENDING in the published /// model, bounding the map to the parts of currently-active tasks. - void pruneLocalBackoff(const ExportPartitionTaskEntriesContainer::index::type & model); + void pruneLocalBackoff(const ExportTaskEntriesContainer::index::type & model); public: - /// Snapshot of the local back-off map for system.partition_exports: + /// Snapshot of the local back-off map for system.distributed_exports: /// transaction_id -> part -> (attempts, next_retry_time). Briefly locks local_backoff_mutex; /// never held across ZooKeeper I/O. std::unordered_map getLocalBackoffSnapshot() const; diff --git a/src/Storages/MergeTree/ExportPartitionManifestUpdatingTask.cpp b/src/Storages/MergeTree/ReplicatedExportTaskUpdater.cpp similarity index 67% rename from src/Storages/MergeTree/ExportPartitionManifestUpdatingTask.cpp rename to src/Storages/MergeTree/ReplicatedExportTaskUpdater.cpp index 786308ec4ce4..f9ed96bfa10e 100644 --- a/src/Storages/MergeTree/ExportPartitionManifestUpdatingTask.cpp +++ b/src/Storages/MergeTree/ReplicatedExportTaskUpdater.cpp @@ -1,7 +1,7 @@ -#include +#include #include -#include -#include "Storages/MergeTree/ExportPartitionUtils.h" +#include +#include "Storages/MergeTree/ExportTaskUtils.h" #include "Common/logger_useful.h" #include #include @@ -15,13 +15,13 @@ namespace ProfileEvents { - extern const Event ExportPartitionZooKeeperRequests; - extern const Event ExportPartitionZooKeeperGet; - extern const Event ExportPartitionZooKeeperGetChildren; - extern const Event ExportPartitionZooKeeperGetChildrenWatch; - extern const Event ExportPartitionZooKeeperGetWatch; - extern const Event ExportPartitionZooKeeperRemoveRecursive; - extern const Event ExportPartitionZooKeeperMulti; + extern const Event ExportTaskZooKeeperRequests; + extern const Event ExportTaskZooKeeperGet; + extern const Event ExportTaskZooKeeperGetChildren; + extern const Event ExportTaskZooKeeperGetChildrenWatch; + extern const Event ExportTaskZooKeeperGetWatch; + extern const Event ExportTaskZooKeeperRemoveRecursive; + extern const Event ExportTaskZooKeeperMulti; } namespace DB @@ -47,7 +47,7 @@ namespace /// Describes pending commits struct CommitRecoveryWork { - ExportReplicatedMergeTreePartitionManifest metadata; + ExportReplicatedMergeTreeTaskManifest metadata; std::string entry_path; StoragePtr destination_storage; ContextPtr context; @@ -66,11 +66,11 @@ namespace const auto container_path = entry_path / "last_exception"; Strings children; - ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRequests); - ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperGetChildren); + ProfileEvents::increment(ProfileEvents::ExportTaskZooKeeperRequests); + ProfileEvents::increment(ProfileEvents::ExportTaskZooKeeperGetChildren); if (Coordination::Error::ZOK != zk->tryGetChildren(container_path, children)) { - LOG_WARNING(log, "ExportPartition Manifest Updating Task: failed to list last_exception leaves for {}, leaving in-memory copy untouched", log_key); + LOG_WARNING(log, "Export task updater: failed to list last_exception leaves for {}, leaving in-memory copy untouched", log_key); return std::nullopt; } @@ -82,8 +82,8 @@ namespace for (const auto & child : children) paths.emplace_back(container_path / child); - ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRequests); - ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperGet, paths.size()); + ProfileEvents::increment(ProfileEvents::ExportTaskZooKeeperRequests); + ProfileEvents::increment(ProfileEvents::ExportTaskZooKeeperGet, paths.size()); auto responses = zk->tryGet(paths); responses.waitForResponses(); @@ -100,7 +100,7 @@ namespace } catch (...) { - LOG_WARNING(log, "ExportPartition Manifest Updating Task: ZK error fetching last_exception leaf {} for {}, skipping", children[i], log_key); + LOG_WARNING(log, "Export task updater: ZK error fetching last_exception leaf {} for {}, skipping", children[i], log_key); continue; } @@ -115,7 +115,7 @@ namespace } catch (...) { - LOG_WARNING(log, "ExportPartition Manifest Updating Task: malformed last_exception JSON for {} (leaf {}), ignoring", log_key, children[i]); + LOG_WARNING(log, "Export task updater: malformed last_exception JSON for {} (leaf {}), ignoring", log_key, children[i]); } } @@ -133,11 +133,11 @@ namespace const auto container_path = entry_path / "processed"; Strings children; - ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRequests); - ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperGetChildren); + ProfileEvents::increment(ProfileEvents::ExportTaskZooKeeperRequests); + ProfileEvents::increment(ProfileEvents::ExportTaskZooKeeperGetChildren); if (Coordination::Error::ZOK != zk->tryGetChildren(container_path, children)) { - LOG_INFO(log, "ExportPartition Manifest Updating Task: failed to list processed leaves for {}, publishing sync-failed marker", log_key); + LOG_INFO(log, "Export task updater: failed to list processed leaves for {}, publishing sync-failed marker", log_key); out.emplace(String(zk_sync_failed_marker), std::vector{String(zk_sync_failed_marker)}); return out; } @@ -150,8 +150,8 @@ namespace for (const auto & child : children) paths.emplace_back(container_path / child); - ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRequests); - ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperGet, paths.size()); + ProfileEvents::increment(ProfileEvents::ExportTaskZooKeeperRequests); + ProfileEvents::increment(ProfileEvents::ExportTaskZooKeeperGet, paths.size()); auto responses = zk->tryGet(paths); responses.waitForResponses(); @@ -170,26 +170,26 @@ namespace } catch (...) { - LOG_WARNING(log, "ExportPartition Manifest Updating Task: ZK error fetching processed leaf {} for {}, publishing sync-failed marker", children[i], log_key); + LOG_WARNING(log, "Export task updater: ZK error fetching processed leaf {} for {}, publishing sync-failed marker", children[i], log_key); out.emplace(children[i], std::vector{String(zk_sync_failed_marker)}); continue; } if (response.error != Coordination::Error::ZOK) { - LOG_WARNING(log, "ExportPartition Manifest Updating Task: could not read processed leaf {} for {} (error {}), publishing sync-failed marker", children[i], log_key, response.error); + LOG_WARNING(log, "Export task updater: could not read processed leaf {} for {} (error {}), publishing sync-failed marker", children[i], log_key, response.error); out.emplace(children[i], std::vector{String(zk_sync_failed_marker)}); continue; } try { - auto entry = ExportReplicatedMergeTreePartitionProcessedPartEntry::fromJsonString(response.data); + auto entry = ExportReplicatedMergeTreeProcessedPartEntry::fromJsonString(response.data); out.emplace(std::move(entry.part_name), std::move(entry.paths_in_destination)); } catch (...) { - LOG_WARNING(log, "ExportPartition Manifest Updating Task: malformed processed JSON for {} (leaf {}), publishing sync-failed marker", log_key, children[i]); + LOG_WARNING(log, "Export task updater: malformed processed JSON for {} (leaf {}), publishing sync-failed marker", log_key, children[i]); out.emplace(children[i], std::vector{String(zk_sync_failed_marker)}); } } @@ -227,15 +227,15 @@ namespace } bool skipReadingDestinationFilePaths( - ExportReplicatedMergeTreePartitionTaskEntry::Status status, + ExportReplicatedMergeTreeTaskEntry::Status status, const std::map> & cached_paths, size_t number_of_parts) { - if (status == ExportReplicatedMergeTreePartitionTaskEntry::Status::PENDING) + if (status == ExportReplicatedMergeTreeTaskEntry::Status::PENDING) return false; if (destinationFilePathsMirrorHasSyncFailure(cached_paths)) return false; - if (status == ExportReplicatedMergeTreePartitionTaskEntry::Status::COMPLETED) + if (status == ExportReplicatedMergeTreeTaskEntry::Status::COMPLETED) return cached_paths.size() == number_of_parts; return true; } @@ -244,7 +244,7 @@ namespace /// Returns nullopt when the znode is absent (task has not committed yet, peer /// crashed before writing it, or transient ZK error). Callers should treat /// nullopt as "leave the in-memory copy untouched". - std::optional readCommitInfo( + std::optional readCommitInfo( const zkutil::ZooKeeperPtr & zk, const std::filesystem::path & entry_path, const std::string & log_key, @@ -253,18 +253,18 @@ namespace const auto commit_info_path = entry_path / "commit_info"; std::string data; - ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRequests); - ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperGet); + ProfileEvents::increment(ProfileEvents::ExportTaskZooKeeperRequests); + ProfileEvents::increment(ProfileEvents::ExportTaskZooKeeperGet); if (!zk->tryGet(commit_info_path, data)) return std::nullopt; try { - return ExportPartitionCommitInfoEntry::fromJsonString(data); + return ExportCommitInfoEntry::fromJsonString(data); } catch (...) { - LOG_WARNING(log, "ExportPartition Manifest Updating Task: malformed commit_info JSON for {}, ignoring", log_key); + LOG_WARNING(log, "Export task updater: malformed commit_info JSON for {}, ignoring", log_key); return std::nullopt; } } @@ -276,14 +276,14 @@ namespace const LoggerPtr & log, const ContextPtr & storage_context, StorageReplicatedMergeTree & storage, - const ExportReplicatedMergeTreePartitionManifest & metadata, + const ExportReplicatedMergeTreeTaskManifest & metadata, const time_t now, const bool is_pending, std::vector & deferred_commits ) { const bool task_timed_out = is_pending - && ExportPartitionUtils::isExportTaskTimedOut(metadata.create_time, metadata.task_timeout_seconds, now); + && ExportTaskUtils::isExportTaskTimedOut(metadata.create_time, metadata.task_timeout_seconds, now); if (task_timed_out) { @@ -292,7 +292,7 @@ namespace fs::path(entry_path) / "commit_lock", *zk, storage.getReplicaName()); if (!commit_lock) { - LOG_DEBUG(log, "ExportPartition Manifest Updating Task: commit in progress for {}, skipping timeout kill", entry_path); + LOG_DEBUG(log, "Export task updater: commit in progress for {}, skipping timeout kill", entry_path); return; } @@ -301,36 +301,36 @@ namespace Coordination::Stat status_stat; std::string status_string; - ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRequests); - ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperGet); + ProfileEvents::increment(ProfileEvents::ExportTaskZooKeeperRequests); + ProfileEvents::increment(ProfileEvents::ExportTaskZooKeeperGet); if (!zk->tryGet(status_path, status_string, &status_stat)) { - LOG_WARNING(log, "ExportPartition Manifest Updating Task: Failed to read status for {} while enforcing task timeout, skipping", entry_path); + LOG_WARNING(log, "Export task updater: Failed to read status for {} while enforcing task timeout, skipping", entry_path); return; } - const auto current_status = magic_enum::enum_cast(status_string); - if (!current_status || *current_status != ExportReplicatedMergeTreePartitionTaskEntry::Status::PENDING) + const auto current_status = magic_enum::enum_cast(status_string); + if (!current_status || *current_status != ExportReplicatedMergeTreeTaskEntry::Status::PENDING) { - LOG_DEBUG(log, "ExportPartition Manifest Updating Task: Task {} is not PENDING, can't set to KILLED, skipping", entry_path); + LOG_DEBUG(log, "Export task updater: Task {} is not PENDING, can't set to KILLED, skipping", entry_path); return; } const auto timeout_message = fmt::format( - "Export partition task timed out: exceeded export_merge_tree_partition_task_timeout_seconds={} (created at {}, now {})", + "Export partition task timed out: exceeded export_merge_tree_task_timeout_seconds={} (created at {}, now {})", metadata.task_timeout_seconds, metadata.create_time, now); - const auto killed_name = String(magic_enum::enum_name(ExportReplicatedMergeTreePartitionTaskEntry::Status::KILLED)); + const auto killed_name = String(magic_enum::enum_name(ExportReplicatedMergeTreeTaskEntry::Status::KILLED)); Coordination::Requests ops; - ExportPartitionUtils::appendExceptionOps( + ExportTaskUtils::appendExceptionOps( ops, zk, fs::path(entry_path), storage.getReplicaName(), /*part_name=*/"", timeout_message, log); ops.emplace_back(zkutil::makeSetRequest(status_path, killed_name, status_stat.version)); - ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRequests); - ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperMulti); + ProfileEvents::increment(ProfileEvents::ExportTaskZooKeeperRequests); + ProfileEvents::increment(ProfileEvents::ExportTaskZooKeeperMulti); Coordination::Responses responses; const auto rc = zk->tryMulti(ops, responses); @@ -338,7 +338,7 @@ namespace if (rc == Coordination::Error::ZOK) { LOG_WARNING(log, - "ExportPartition Manifest Updating Task: task {} exceeded task_timeout_seconds={}s, " + "Export task updater: task {} exceeded task_timeout_seconds={}s, " "transitioned PENDING -> KILLED (atomic with exception record)", entry_path, metadata.task_timeout_seconds); } @@ -348,7 +348,7 @@ namespace /// counter race, or ZNONODE (entry concurrently removed). In all cases the batch /// was rolled back atomically and the task will be re-evaluated on the next poll. LOG_DEBUG(log, - "ExportPartition Manifest Updating Task: atomic kill for {} failed (rc={}); " + "Export task updater: atomic kill for {} failed (rc={}); " "status was concurrently updated or a ZK op conflicted, will retry on next poll", entry_path, rc); } @@ -359,27 +359,27 @@ namespace } else if (is_pending) { - auto context = ExportPartitionUtils::getContextCopyWithTaskSettings(storage_context, metadata); + auto context = ExportTaskUtils::getContextCopyWithTaskSettings(storage_context, metadata); - ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRequests); - ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperGetChildren); + ProfileEvents::increment(ProfileEvents::ExportTaskZooKeeperRequests); + ProfileEvents::increment(ProfileEvents::ExportTaskZooKeeperGetChildren); std::vector parts_in_processing_or_pending; if (Coordination::Error::ZOK != zk->tryGetChildren(fs::path(entry_path) / "processing", parts_in_processing_or_pending)) { - LOG_WARNING(log, "ExportPartition Manifest Updating Task: Failed to get parts in processing or pending, skipping"); + LOG_WARNING(log, "Export task updater: Failed to get parts in processing or pending, skipping"); return; } if (parts_in_processing_or_pending.empty()) { - LOG_DEBUG(log, "ExportPartition Manifest Updating Task: Cleanup found PENDING for {} with all parts exported, deferring commit recovery to post-lock phase", entry_path); + LOG_DEBUG(log, "Export task updater: Cleanup found PENDING for {} with all parts exported, deferring commit recovery to post-lock phase", entry_path); const auto destination_storage_id = StorageID(QualifiedTableName {metadata.destination_database, metadata.destination_table}); const auto destination_storage = DatabaseCatalog::instance().tryGetTable(destination_storage_id, context); if (!destination_storage) { - LOG_WARNING(log, "ExportPartition Manifest Updating Task: Failed to reconstruct destination storage: {}, skipping", destination_storage_id.getNameForLogs()); + LOG_WARNING(log, "Export task updater: Failed to reconstruct destination storage: {}, skipping", destination_storage_id.getNameForLogs()); return; } @@ -395,32 +395,34 @@ namespace } } -ExportPartitionManifestUpdatingTask::ExportPartitionManifestUpdatingTask(StorageReplicatedMergeTree & storage_) +ReplicatedExportTaskUpdater::ReplicatedExportTaskUpdater(StorageReplicatedMergeTree & storage_) : storage(storage_) { } -std::vector ExportPartitionManifestUpdatingTask::getPartitionExportsInfo() const +std::vector ReplicatedExportTaskUpdater::getExportTasksInfo() const { const auto model = storage.export_partition_manifests.get(); if (!model) return {}; - const auto backoff = storage.export_merge_tree_partition_task_scheduler->getLocalBackoffSnapshot(); + const auto backoff = storage.export_task_scheduler->getLocalBackoffSnapshot(); - std::vector infos; + std::vector infos; infos.reserve(model->size()); - for (const auto & entry : model->get()) + for (const auto & entry : model->get()) { const auto & manifest = entry.manifest; - PartitionExportInfo info; + ExportTaskInfo info; info.destination_database = manifest.destination_database; info.destination_table = manifest.destination_table; - info.partition_id = manifest.partition_id; + info.partition_id = ExportTaskUtils::getPartitionIdOfParts(manifest.parts, storage.format_version); + info.source = String(magic_enum::enum_name(manifest.source)); + info.retry_of = manifest.retry_of; info.transaction_id = manifest.transaction_id; info.query_id = manifest.query_id; info.create_time = manifest.create_time; @@ -462,7 +464,7 @@ std::vector ExportPartitionManifestUpdatingTask::getPartiti return infos; } -void ExportPartitionManifestUpdatingTask::poll() +void ReplicatedExportTaskUpdater::poll() { /// Commit-recovery work collected while the storage-wide mutex is held. /// Executed AFTER the mutex is released. @@ -482,9 +484,12 @@ void ExportPartitionManifestUpdatingTask::poll() auto cleanup_lock = zkutil::EphemeralNodeHolder::tryCreate(cleanup_lock_path, *zk, storage.replica_name); if (cleanup_lock) { - LOG_DEBUG(log, "ExportPartition Manifest Updating Task: Cleanup lock acquired, will remove stale entries"); + LOG_DEBUG(log, "Export task updater: Cleanup lock acquired, will remove stale entries"); } + /// The `EXPORT` TTL resolves a finished task of its own, and may start the next group, as soon as it is published. + bool ttl_task_finished = false; + { /// M_task: serializes poll() vs handleStatusChanges(). We copy the current read-model into a /// private mutable container, mutate that copy across the ZooKeeper reads below, and publish @@ -494,18 +499,18 @@ void ExportPartitionManifestUpdatingTask::poll() const auto current_model = storage.export_partition_manifests.get(); auto working_model = current_model - ? std::make_unique(*current_model) - : std::make_unique(); + ? std::make_unique(*current_model) + : std::make_unique(); - auto & entries_by_key = working_model->get(); + auto & entries_by_key = working_model->get(); - LOG_DEBUG(log, "ExportPartition Manifest Updating Task: Polling for new entries for table {}. Current number of entries: {}", storage.getStorageID().getNameForLogs(), entries_by_key.size()); + LOG_DEBUG(log, "Export task updater: Polling for new entries for table {}. Current number of entries: {}", storage.getStorageID().getNameForLogs(), entries_by_key.size()); - ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRequests); - ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperGetChildrenWatch); + ProfileEvents::increment(ProfileEvents::ExportTaskZooKeeperRequests); + ProfileEvents::increment(ProfileEvents::ExportTaskZooKeeperGetChildrenWatch); Coordination::Stat stat; - const auto children = zk->getChildrenWatch(exports_path, &stat, storage.export_merge_tree_partition_watch_callback); + const auto children = zk->getChildrenWatch(exports_path, &stat, storage.export_task_watch_callback); const std::unordered_set zk_children(children.begin(), children.end()); const auto now = time(nullptr); @@ -515,40 +520,47 @@ void ExportPartitionManifestUpdatingTask::poll() /// Upload dangling commit files if any for (const auto & key : zk_children) { + /// A task is keyed by its transaction id and its descriptor never changes, so the + /// descriptor is read only the first time the task is seen. const std::string entry_path = fs::path(exports_path) / key; + const auto local_entry = entries_by_key.find(key); + const bool has_local_entry = local_entry != entries_by_key.end(); - ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRequests); - ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperGet); - std::string metadata_json; - if (!zk->tryGet(fs::path(entry_path) / "metadata.json", metadata_json)) + std::optional parsed_metadata; + if (!has_local_entry) { - LOG_WARNING(log, "ExportPartition Manifest Updating Task: Skipping {}: missing metadata.json", key); - continue; - } + ProfileEvents::increment(ProfileEvents::ExportTaskZooKeeperRequests); + ProfileEvents::increment(ProfileEvents::ExportTaskZooKeeperGet); + std::string metadata_json; + if (!zk->tryGet(fs::path(entry_path) / "metadata.json", metadata_json)) + { + LOG_WARNING(log, "Export task updater: Skipping {}: missing metadata.json", key); + continue; + } - ExportReplicatedMergeTreePartitionManifest metadata; - try - { - metadata = ExportReplicatedMergeTreePartitionManifest::fromJsonString(metadata_json); - } - catch (...) - { - /// A single unparseable metadata.json (e.g. genuinely corrupt, or written by a - /// future incompatible format) must not abort the whole poll and stall discovery, - /// cleanup and status convergence for every other task. Skip just this entry. - tryLogCurrentException(log, __PRETTY_FUNCTION__); - LOG_WARNING(log, "ExportPartition Manifest Updating Task: Skipping {}: could not parse metadata.json", key); - continue; - } + try + { + parsed_metadata = ExportReplicatedMergeTreeTaskManifest::fromJsonString(metadata_json); + } + catch (...) + { + /// A single unparseable metadata.json (e.g. genuinely corrupt, or written by a + /// future incompatible format) must not abort the whole poll and stall discovery, + /// cleanup and status convergence for every other task. Skip just this entry. + tryLogCurrentException(log, __PRETTY_FUNCTION__); + LOG_WARNING(log, "Export task updater: Skipping {}: could not parse metadata.json", key); + continue; + } - auto last_exception_per_replica = readLastExceptionPerReplica( - zk, fs::path(entry_path), key, log); + /// Tasks stored by earlier versions are keyed by partition and destination, and are not read. + if (parsed_metadata->transaction_id != key) + { + LOG_DEBUG(log, "Export task updater: Skipping {}: not keyed by its transaction id", key); + continue; + } + } - /// If the zk entry has been replaced with export_merge_tree_partition_force_export, checking only for the export key is not enough - /// we need to make sure it is the same transaction id. If it is not, it needs to be replaced. - const auto local_entry = entries_by_key.find(key); - const bool has_local_entry = local_entry != entries_by_key.end() - && local_entry->manifest.transaction_id == metadata.transaction_id; + const ExportReplicatedMergeTreeTaskManifest & metadata = has_local_entry ? local_entry->manifest : *parsed_metadata; std::string status_string; @@ -557,15 +569,15 @@ void ExportPartitionManifestUpdatingTask::poll() /// so we need to read the status from the ZK node directly. if (has_local_entry) { - ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRequests); - ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperGet); + ProfileEvents::increment(ProfileEvents::ExportTaskZooKeeperRequests); + ProfileEvents::increment(ProfileEvents::ExportTaskZooKeeperGet); zk->tryGet(fs::path(entry_path) / "status", status_string); } else { /// If we don't have a local entry, we need to arm a status watch to be notified when the status changes - std::weak_ptr weak_manifest_updater = storage.export_merge_tree_partition_manifest_updater; + std::weak_ptr weak_manifest_updater = storage.export_task_updater; auto status_watch_callback = std::make_shared([weak_manifest_updater, key](const Coordination::WatchResponse &) { /// If the table is dropped but the watch is not removed, we need to prevent use after free @@ -573,29 +585,40 @@ void ExportPartitionManifestUpdatingTask::poll() if (auto manifest_updater = weak_manifest_updater.lock()) { manifest_updater->addStatusChange(key); - manifest_updater->storage.export_merge_tree_partition_status_handling_task->schedule(); + manifest_updater->storage.export_task_status_handling_task->schedule(); } }); - ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRequests); - ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperGetWatch); - + ProfileEvents::increment(ProfileEvents::ExportTaskZooKeeperRequests); + ProfileEvents::increment(ProfileEvents::ExportTaskZooKeeperGetWatch); + zk->tryGetWatch(fs::path(entry_path) / "status", status_string, nullptr, status_watch_callback); } if (status_string.empty()) { - LOG_WARNING(log, "ExportPartition Manifest Updating Task: Skipping {}: missing status", key); + LOG_WARNING(log, "Export task updater: Skipping {}: missing status", key); continue; } - const auto status = magic_enum::enum_cast(status_string); + const auto status = magic_enum::enum_cast(status_string); if (!status) { - LOG_WARNING(log, "ExportPartition Manifest Updating Task: Invalid status {} for task {}, skipping", status_string, key); + LOG_WARNING(log, "Export task updater: Invalid status {} for task {}, skipping", status_string, key); continue; } + /// Tasks are never removed, so most of them are finished ones. Nothing but the status of a + /// finished task changes (a TTL export can resolve a failed task as completed), so there is + /// nothing else to refresh. + if (has_local_entry + && local_entry->status != ExportReplicatedMergeTreeTaskEntry::Status::PENDING + && local_entry->status == *status) + continue; + + auto last_exception_per_replica = readLastExceptionPerReplica( + zk, fs::path(entry_path), key, log); + const bool skip_processed_refresh = has_local_entry && skipReadingDestinationFilePaths(*status, local_entry->destination_file_paths_per_part, metadata.number_of_parts); @@ -617,7 +640,7 @@ void ExportPartitionManifestUpdatingTask::poll() storage, metadata, now, - *status == ExportReplicatedMergeTreePartitionTaskEntry::Status::PENDING, + *status == ExportReplicatedMergeTreeTaskEntry::Status::PENDING, deferred_commits); } @@ -631,11 +654,11 @@ void ExportPartitionManifestUpdatingTask::poll() readCommitInfo(zk, fs::path(entry_path), key, log), key, entries_by_key); - LOG_INFO(log, "ExportPartition Manifest Updating Task: Added new entry for task {}", key); + LOG_INFO(log, "Export task updater: Added new entry for task {}", key); continue; } - if (!local_entry->commit_info && *status == ExportReplicatedMergeTreePartitionTaskEntry::Status::COMPLETED) + if (!local_entry->commit_info && *status == ExportReplicatedMergeTreeTaskEntry::Status::COMPLETED) { local_entry->commit_info = readCommitInfo(zk, fs::path(entry_path), key, log); } @@ -650,21 +673,24 @@ void ExportPartitionManifestUpdatingTask::poll() if (status_changed) { local_entry->status = *status; - if (local_entry->status != ExportReplicatedMergeTreePartitionTaskEntry::Status::PENDING) + if (local_entry->status != ExportReplicatedMergeTreeTaskEntry::Status::PENDING) { + if (local_entry->manifest.source == ExportTaskSource::ttl) + ttl_task_finished = true; + /// terminal now - we no longer need to keep the data parts alive local_entry->part_references.clear(); /// looks like we missed a status change event, we should kill local operations. - if (local_entry->status == ExportReplicatedMergeTreePartitionTaskEntry::Status::KILLED) + if (local_entry->status == ExportReplicatedMergeTreeTaskEntry::Status::KILLED) { storage.killExportPart(local_entry->manifest.transaction_id); } } } - LOG_DEBUG(log, "ExportPartition Manifest Updating Task: Skipping {}: already exists", key); - + LOG_DEBUG(log, "Export task updater: Skipping {}: already exists", key); + } removeStaleEntries(zk_children, entries_by_key); @@ -675,25 +701,28 @@ void ExportPartitionManifestUpdatingTask::poll() /// `entries_by_key` (a reference into it) must not be used afterwards. storage.export_partition_manifests.set(std::move(working_model)); - LOG_DEBUG(log, "ExportPartition Manifest Updating task: finished polling for new entries. Number of entries: {}", entries_count); + LOG_DEBUG(log, "Export task updater: finished polling for new entries. Number of entries: {}", entries_count); } + if (ttl_task_finished) + storage.wakeUpExportTTL(); + /// Execute pending commits for (const auto & work : deferred_commits) { /// A replica exported the last part but the commit never landed. Try to fix it. try { - ExportPartitionUtils::commit(work.metadata, work.destination_storage, zk, log, work.entry_path, work.context, storage, storage.getReplicaName()); + ExportTaskUtils::commit(work.metadata, work.destination_storage, zk, log, work.entry_path, work.context, storage, storage.getReplicaName(), *storage.export_fence); } catch (const Exception & e) { LOG_WARNING(log, - "ExportPartition Manifest Updating Task: " + "Export task updater: " "Caught exception while committing export for {}: {}", work.entry_path, e.message()); - const bool became_failed = ExportPartitionUtils::handleCommitFailure( + const bool became_failed = ExportTaskUtils::handleCommitFailure( zk, work.entry_path, e.code(), @@ -704,22 +733,22 @@ void ExportPartitionManifestUpdatingTask::poll() if (became_failed) { LOG_WARNING(log, - "ExportPartition Manifest Updating Task: " + "Export task updater: " "Commit for {} transitioned to FAILED due to non-retryable error (code {})", work.entry_path, e.code()); } } } - storage.export_merge_tree_partition_select_task->schedule(); + storage.export_task_select_task->schedule(); } -void ExportPartitionManifestUpdatingTask::addTask( - const ExportReplicatedMergeTreePartitionManifest & metadata, - ExportReplicatedMergeTreePartitionTaskEntry::Status status, +void ReplicatedExportTaskUpdater::addTask( + const ExportReplicatedMergeTreeTaskManifest & metadata, + ExportReplicatedMergeTreeTaskEntry::Status status, std::map last_exception_per_replica, std::map> destination_file_paths_per_part, - std::optional commit_info, + std::optional commit_info, const std::string & key, auto & entries_by_key ) @@ -729,8 +758,8 @@ void ExportPartitionManifestUpdatingTask::addTask( /// If the status is PENDING, we grab references to the data parts to prevent them from being deleted from the disk /// Otherwise, the operation has already been completed and there is no need to keep the data parts alive /// You might also ask: why bother adding tasks that have already been completed (i.e, status != PENDING)? - /// The reason is the `partition_exports` table might miss entries if they are not added here. - if (status == ExportReplicatedMergeTreePartitionTaskEntry::Status::PENDING) + /// The reason is the `distributed_exports` table might miss entries if they are not added here. + if (status == ExportReplicatedMergeTreeTaskEntry::Status::PENDING) { for (const auto & part_name : metadata.parts) { @@ -742,7 +771,7 @@ void ExportPartitionManifestUpdatingTask::addTask( } /// Called from poll() under M_task (sole mutator), so no extra locking is required. - ExportReplicatedMergeTreePartitionTaskEntry entry { + ExportReplicatedMergeTreeTaskEntry entry { metadata, status, std::move(part_references), @@ -755,36 +784,36 @@ void ExportPartitionManifestUpdatingTask::addTask( { if (!entries_by_key.replace(it, entry)) LOG_ERROR(storage.log, - "ExportPartition Manifest Updating Task: failed to replace in-memory entry for {} (transaction_id {}). " + "Export task updater: failed to replace in-memory entry for {} (transaction_id {}). " "This most likely means another export already holds the same transaction_id (id collision); " - "this export will be missing from system.partition_exports.", + "this export will be missing from system.distributed_exports.", key, entry.getTransactionId()); } else if (!entries_by_key.insert(entry).second) { LOG_ERROR(storage.log, - "ExportPartition Manifest Updating Task: failed to insert in-memory entry for {} (transaction_id {}). " + "Export task updater: failed to insert in-memory entry for {} (transaction_id {}). " "Another entry already holds this transaction_id (id collision); " - "this export will be invisible in system.partition_exports.", + "this export will be invisible in system.distributed_exports.", key, entry.getTransactionId()); } } -void ExportPartitionManifestUpdatingTask::removeStaleEntries( +void ReplicatedExportTaskUpdater::removeStaleEntries( const std::unordered_set & zk_children, auto & entries_by_key ) { for (auto it = entries_by_key.begin(); it != entries_by_key.end();) { - const auto key = it->getCompositeKey(); + const auto key = it->getTransactionId(); if (zk_children.contains(key)) { ++it; continue; } - LOG_INFO(storage.log, "ExportPartition Manifest Updating Task: Export task {} was deleted, calling killExportPartition for transaction {}", key, it->manifest.transaction_id); + LOG_INFO(storage.log, "Export task updater: Export task {} was deleted, calling killExportTask for transaction {}", key, it->manifest.transaction_id); try { @@ -799,13 +828,13 @@ void ExportPartitionManifestUpdatingTask::removeStaleEntries( } } -void ExportPartitionManifestUpdatingTask::addStatusChange(const std::string & key) +void ReplicatedExportTaskUpdater::addStatusChange(const std::string & key) { std::lock_guard lock(status_changes_mutex); status_changes.emplace(key); } -void ExportPartitionManifestUpdatingTask::handleStatusChanges() +void ReplicatedExportTaskUpdater::handleStatusChanges() { /// copy the events to a local queue to avoid holding status_changes_mutex under M_task std::queue local_status_changes; @@ -827,20 +856,21 @@ void ExportPartitionManifestUpdatingTask::handleStatusChanges() std::lock_guard task_guard(background_task_serialization_mutex); auto zk = storage.getZooKeeper(); + bool ttl_task_finished = false; const bool had_changes = !local_status_changes.empty(); - LOG_DEBUG(log, "ExportPartition Manifest Updating task: handling status changes. Number of status changes: {}", local_status_changes.size()); + LOG_DEBUG(log, "Export task updater: handling status changes. Number of status changes: {}", local_status_changes.size()); const auto current_model = storage.export_partition_manifests.get(); auto working_model = current_model - ? std::make_unique(*current_model) - : std::make_unique(); - auto & entries_by_key = working_model->get(); + ? std::make_unique(*current_model) + : std::make_unique(); + auto & entries_by_key = working_model->get(); while (!local_status_changes.empty()) { const auto & key = local_status_changes.front(); - LOG_INFO(log, "ExportPartition Manifest Updating task: handling status change for task {}", key); + LOG_INFO(log, "Export task updater: handling status change for task {}", key); fiu_do_on(FailPoints::export_partition_status_change_throw, { @@ -857,26 +887,26 @@ void ExportPartitionManifestUpdatingTask::handleStatusChanges() const auto export_path = fs::path(storage.zookeeper_path) / "exports" / key; - ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRequests); - ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperGet); + ProfileEvents::increment(ProfileEvents::ExportTaskZooKeeperRequests); + ProfileEvents::increment(ProfileEvents::ExportTaskZooKeeperGet); /// get new status from zk std::string new_status_string; if (!zk->tryGet(export_path / "status", new_status_string)) { - LOG_WARNING(log, "ExportPartition Manifest Updating Task: Failed to get new status for task {}, skipping", key); + LOG_WARNING(log, "Export task updater: Failed to get new status for task {}, skipping", key); local_status_changes.pop(); continue; } - const auto new_status = magic_enum::enum_cast(new_status_string); + const auto new_status = magic_enum::enum_cast(new_status_string); if (!new_status) { - LOG_WARNING(log, "ExportPartition Manifest Updating Task: Invalid status {} for task {}, skipping", new_status_string, key); + LOG_WARNING(log, "Export task updater: Invalid status {} for task {}, skipping", new_status_string, key); local_status_changes.pop(); continue; } - LOG_INFO(log, "ExportPartition Manifest Updating task: status changed for task {}. New status: {}", key, magic_enum::enum_name(*new_status).data()); + LOG_INFO(log, "Export task updater: status changed for task {}. New status: {}", key, magic_enum::enum_name(*new_status).data()); auto fetched = readLastExceptionPerReplica( zk, export_path, key, log); @@ -888,18 +918,18 @@ void ExportPartitionManifestUpdatingTask::handleStatusChanges() it->destination_file_paths_per_part = std::move(destination_file_paths_per_part); } - if (*new_status == ExportReplicatedMergeTreePartitionTaskEntry::Status::COMPLETED) + if (*new_status == ExportReplicatedMergeTreeTaskEntry::Status::COMPLETED) { if (auto fetched_commit_info = readCommitInfo(zk, export_path, key, log)) it->commit_info = std::move(fetched_commit_info); } /// If status changed to KILLED, cancel local export operations - if (*new_status == ExportReplicatedMergeTreePartitionTaskEntry::Status::KILLED) + if (*new_status == ExportReplicatedMergeTreeTaskEntry::Status::KILLED) { try { - LOG_INFO(log, "ExportPartition Manifest Updating task: killing export partition for task {}", key); + LOG_INFO(log, "Export task updater: killing export partition for task {}", key); storage.killExportPart(it->manifest.transaction_id); } catch (...) @@ -914,10 +944,13 @@ void ExportPartitionManifestUpdatingTask::handleStatusChanges() it->status = *new_status; - if (it->status != ExportReplicatedMergeTreePartitionTaskEntry::Status::PENDING) + if (it->status != ExportReplicatedMergeTreeTaskEntry::Status::PENDING) { /// we no longer need to keep the data parts alive it->part_references.clear(); + + if (it->manifest.source == ExportTaskSource::ttl) + ttl_task_finished = true; } local_status_changes.pop(); @@ -927,12 +960,15 @@ void ExportPartitionManifestUpdatingTask::handleStatusChanges() /// so `entries_by_key` (a reference into it) must not be used afterwards. if (had_changes) storage.export_partition_manifests.set(std::move(working_model)); + + if (ttl_task_finished) + storage.wakeUpExportTTL(); } catch (...) { tryLogCurrentException(log, __PRETTY_FUNCTION__); - LOG_WARNING(log, "ExportPartition Manifest Updating task: exception thrown while handling status changes; nothing was published, requeuing the whole batch. Batch size: {}", batch.size()); + LOG_WARNING(log, "Export task updater: exception thrown while handling status changes; nothing was published, requeuing the whole batch. Batch size: {}", batch.size()); std::lock_guard lock(status_changes_mutex); @@ -949,7 +985,7 @@ void ExportPartitionManifestUpdatingTask::handleStatusChanges() std::swap(status_changes, requeued); } - LOG_DEBUG(log, "ExportPartition Manifest Updating task: pending status changes after requeue: {}", status_changes.size()); + LOG_DEBUG(log, "Export task updater: pending status changes after requeue: {}", status_changes.size()); throw; } diff --git a/src/Storages/MergeTree/ExportPartitionManifestUpdatingTask.h b/src/Storages/MergeTree/ReplicatedExportTaskUpdater.h similarity index 69% rename from src/Storages/MergeTree/ExportPartitionManifestUpdatingTask.h rename to src/Storages/MergeTree/ReplicatedExportTaskUpdater.h index b93f1b4bbdcf..ef6f97306006 100644 --- a/src/Storages/MergeTree/ExportPartitionManifestUpdatingTask.h +++ b/src/Storages/MergeTree/ReplicatedExportTaskUpdater.h @@ -4,18 +4,18 @@ #include #include #include -#include -#include +#include +#include namespace DB { class StorageReplicatedMergeTree; -struct ExportReplicatedMergeTreePartitionManifest; +struct ExportReplicatedMergeTreeTaskManifest; -class ExportPartitionManifestUpdatingTask +class ReplicatedExportTaskUpdater { public: - ExportPartitionManifestUpdatingTask(StorageReplicatedMergeTree & storage); + ReplicatedExportTaskUpdater(StorageReplicatedMergeTree & storage); void poll(); @@ -25,17 +25,17 @@ class ExportPartitionManifestUpdatingTask /// Returns a snapshot of every replicated partition export task tracked by this /// replica's in-memory mirror. No ZooKeeper traffic; safe to call from query threads. - std::vector getPartitionExportsInfo() const; + std::vector getExportTasksInfo() const; private: StorageReplicatedMergeTree & storage; void addTask( - const ExportReplicatedMergeTreePartitionManifest & metadata, - ExportReplicatedMergeTreePartitionTaskEntry::Status status, + const ExportReplicatedMergeTreeTaskManifest & metadata, + ExportReplicatedMergeTreeTaskEntry::Status status, std::map last_exception_per_replica, std::map> destination_file_paths_per_part, - std::optional commit_info, + std::optional commit_info, const std::string & key, auto & entries_by_key ); diff --git a/src/Storages/MergeTree/ReplicatedMergeTreeQueue.cpp b/src/Storages/MergeTree/ReplicatedMergeTreeQueue.cpp index 33fe206013ff..ae4168630fb9 100644 --- a/src/Storages/MergeTree/ReplicatedMergeTreeQueue.cpp +++ b/src/Storages/MergeTree/ReplicatedMergeTreeQueue.cpp @@ -119,6 +119,16 @@ bool ReplicatedMergeTreeQueue::isVirtualPart(const MergeTreeData::DataPartPtr & return !virtual_part_name.empty() && virtual_part_name != data_part->name; } +bool ReplicatedMergeTreeQueue::isGoingToBeMergedWithOtherParts(const MergeTreePartInfo & part_info) const +{ + std::lock_guard lock(state_mutex); + const auto virtual_part_name = virtual_parts.getContainingPart(part_info); + if (virtual_part_name.empty()) + return false; + const auto virtual_part_info = MergeTreePartInfo::fromPartName(virtual_part_name, format_version); + return virtual_part_info.min_block != part_info.min_block || virtual_part_info.max_block != part_info.max_block; +} + bool ReplicatedMergeTreeQueue::isGoingToBeDropped(const MergeTreePartInfo & part_info, MergeTreePartInfo * out_drop_range_info) const { std::lock_guard lock(state_mutex); diff --git a/src/Storages/MergeTree/ReplicatedMergeTreeQueue.h b/src/Storages/MergeTree/ReplicatedMergeTreeQueue.h index 454c9b3ec269..b795dd9e4a5c 100644 --- a/src/Storages/MergeTree/ReplicatedMergeTreeQueue.h +++ b/src/Storages/MergeTree/ReplicatedMergeTreeQueue.h @@ -486,6 +486,10 @@ class ReplicatedMergeTreeQueue /// Checks that part is already in virtual parts bool isVirtualPart(const MergeTreeData::DataPartPtr & data_part) const; + /// True if the part is going to be replaced by a part with a different block range, that is, merged + /// with other parts rather than mutated. + bool isGoingToBeMergedWithOtherParts(const MergeTreePartInfo & part_info) const; + /// Returns true if part_info is covered by some DROP_RANGE or DROP_PART bool isGoingToBeDropped(const MergeTreePartInfo & part_info, MergeTreePartInfo * out_drop_range_info = nullptr) const; bool isGoingToBeDroppedImpl(const MergeTreePartInfo & part_info, MergeTreePartInfo * out_drop_range_info) const; diff --git a/src/Storages/MergeTree/ReplicatedMergeTreeRestartingThread.cpp b/src/Storages/MergeTree/ReplicatedMergeTreeRestartingThread.cpp index 444ad5f2542d..499adb6d47a9 100644 --- a/src/Storages/MergeTree/ReplicatedMergeTreeRestartingThread.cpp +++ b/src/Storages/MergeTree/ReplicatedMergeTreeRestartingThread.cpp @@ -191,9 +191,10 @@ bool ReplicatedMergeTreeRestartingThread::runImpl() if (storage.getContext()->getServerSettings()[ServerSetting::allow_experimental_export_merge_tree_partition]) { - storage.export_merge_tree_partition_updating_task->activateAndSchedule(); - storage.export_merge_tree_partition_select_task->activateAndSchedule(); - storage.export_merge_tree_partition_status_handling_task->activateAndSchedule(); + storage.export_task_updating_task->activateAndSchedule(); + storage.export_task_select_task->activateAndSchedule(); + storage.export_task_status_handling_task->activateAndSchedule(); + storage.export_ttl_task->activateAndSchedule(); } storage.cleanup_thread.start(); @@ -263,6 +264,11 @@ bool ReplicatedMergeTreeRestartingThread::tryStartup() updateQuorumIfWeHavePart(); + /// Before this replica assigns merges: the `EXPORT` TTL starts tasks only if every replica + /// advertises that its merges respect the export states, so a stale marker must not outlive a + /// restart with partition export disabled. + storage.advertiseExportFeatures(zookeeper); + /// Anything above can throw a KeeperException if something is wrong with ZK. /// Anything below should not throw exceptions. diff --git a/src/Storages/MergeTree/registerStorageMergeTree.cpp b/src/Storages/MergeTree/registerStorageMergeTree.cpp index 753944860beb..e1e07d7d3642 100644 --- a/src/Storages/MergeTree/registerStorageMergeTree.cpp +++ b/src/Storages/MergeTree/registerStorageMergeTree.cpp @@ -1051,6 +1051,11 @@ static StoragePtr create(const StorageFactory::Arguments & args) merging_params.allow_tuple_element_aggregation = false; } + /// Before the table is created, e.g. in Keeper. + if (args.mode <= LoadingStrictnessLevel::CREATE) + MergeTreeData::validateExportTTL( + args.table_id, metadata, std::make_shared(*storage_settings), args.getLocalContext()); + if (replicated) { bool need_check_table_structure = true; diff --git a/src/Storages/MergeTree/tests/gtest_export_fence.cpp b/src/Storages/MergeTree/tests/gtest_export_fence.cpp new file mode 100644 index 000000000000..65282532a134 --- /dev/null +++ b/src/Storages/MergeTree/tests/gtest_export_fence.cpp @@ -0,0 +1,91 @@ +#include + +#include + +using namespace DB; + +namespace +{ + +MergeTreePartInfo part(const String & partition_id, Int64 min_block, Int64 max_block, UInt32 level = 0, Int64 mutation = 0) +{ + return MergeTreePartInfo(partition_id, min_block, max_block, level, mutation); +} + +} + +TEST(ExportFence, CompactRangesMergesOnlyOverlappingOrAdjacent) +{ + const auto compacted = ExportFenceUtils::compactRanges( + {part("p", 7, 8), part("p", 1, 2), part("p", 3, 5), part("q", 1, 1), part("p", 10, 12), part("p", 11, 11)}); + + ASSERT_EQ(compacted.size(), 4); + EXPECT_EQ(compacted[0].getPartitionId(), "p"); + EXPECT_EQ(compacted[0].min_block, 1); + EXPECT_EQ(compacted[0].max_block, 5); + /// Block 6 is a gap: it may still be committed later as a new part. + EXPECT_EQ(compacted[1].min_block, 7); + EXPECT_EQ(compacted[1].max_block, 8); + EXPECT_EQ(compacted[2].min_block, 10); + EXPECT_EQ(compacted[2].max_block, 12); + EXPECT_EQ(compacted[3].getPartitionId(), "q"); +} + +TEST(ExportFence, IsCoveredByUnion) +{ + const std::vector ranges{part("p", 1, 2), part("p", 3, 5), part("p", 8, 9)}; + + EXPECT_TRUE(ExportFenceUtils::isCoveredByUnion(part("p", 1, 5, 2), ranges)); + EXPECT_TRUE(ExportFenceUtils::isCoveredByUnion(part("p", 8, 9, 1), ranges)); + EXPECT_FALSE(ExportFenceUtils::isCoveredByUnion(part("p", 1, 9, 3), ranges)); + EXPECT_FALSE(ExportFenceUtils::isCoveredByUnion(part("p", 5, 6, 1), ranges)); + EXPECT_FALSE(ExportFenceUtils::isCoveredByUnion(part("q", 1, 2, 1), ranges)); +} + +TEST(ExportFence, ClassifyByOverlap) +{ + ExportFenceEntry entry; + entry.destination = "db.dst"; + entry.exported = {part("p", 1, 5)}; + entry.claimed = {part("p", 7, 7)}; + + /// A mutated or merged exported part keeps its block range, so it is still exported. + EXPECT_EQ(entry.classify(part("p", 1, 5, 3, 42)), PartExportState::EXPORTED); + EXPECT_EQ(entry.classify(part("p", 2, 3, 1)), PartExportState::EXPORTED); + EXPECT_EQ(entry.classify(part("p", 7, 7)), PartExportState::CLAIMED); + EXPECT_EQ(entry.classify(part("p", 6, 6)), PartExportState::NONE); + EXPECT_EQ(entry.classify(part("p", 8, 8)), PartExportState::NONE); + EXPECT_EQ(entry.classify(part("q", 1, 5)), PartExportState::NONE); +} + +TEST(ExportFence, CheckCanMerge) +{ + ExportFence fence; + ExportFenceEntry entry; + entry.destination = "db.dst"; + entry.exported = {part("p", 1, 5)}; + entry.claimed = {part("p", 7, 8)}; + fence.entries_by_partition["p"].push_back(entry); + + EXPECT_FALSE(fence.checkCanMerge(part("p", 1, 2), part("p", 3, 5)).has_value()); + EXPECT_FALSE(fence.checkCanMerge(part("p", 9, 9), part("p", 10, 10)).has_value()); + /// Claimed parts are being exported by name, so they are not merged even with each other. + EXPECT_TRUE(fence.checkCanMerge(part("p", 7, 7), part("p", 8, 8)).has_value()); + EXPECT_TRUE(fence.checkCanMerge(part("p", 1, 5), part("p", 6, 6)).has_value()); + EXPECT_TRUE(fence.checkCanMerge(part("p", 6, 6), part("p", 7, 7)).has_value()); + EXPECT_TRUE(fence.checkCanMerge(part("p", 3, 5), part("p", 7, 8)).has_value()); + + /// Two parts that are not exported must not merge across the range of an exported part that + /// no longer exists: the merged part would overlap it and look exported. + EXPECT_TRUE(fence.checkCanMerge(part("p", 0, 0), part("p", 6, 6)).has_value()); + EXPECT_TRUE(fence.checkCanMerge(part("p", 6, 6), part("p", 9, 9)).has_value()); + /// Exported parts may merge across an exported gap, and parts of any state across blocks without rows. + EXPECT_FALSE(fence.checkCanMerge(part("p", 1, 1), part("p", 5, 5)).has_value()); + EXPECT_FALSE(fence.checkCanMerge(part("p", 10, 10), part("p", 12, 12)).has_value()); + + /// Partitions nothing was exported from are not fenced. + EXPECT_FALSE(fence.checkCanMerge(part("q", 1, 1), part("q", 2, 2)).has_value()); + + EXPECT_EQ(fence.classify(part("p", 1, 1), "db.dst"), PartExportState::EXPORTED); + EXPECT_EQ(fence.classify(part("p", 1, 1), "db.other"), PartExportState::NONE); +} diff --git a/src/Storages/MergeTree/tests/gtest_export_partition_ordering.cpp b/src/Storages/MergeTree/tests/gtest_export_task_ordering.cpp similarity index 50% rename from src/Storages/MergeTree/tests/gtest_export_partition_ordering.cpp rename to src/Storages/MergeTree/tests/gtest_export_task_ordering.cpp index e77ab1c3a639..d057455167f5 100644 --- a/src/Storages/MergeTree/tests/gtest_export_partition_ordering.cpp +++ b/src/Storages/MergeTree/tests/gtest_export_task_ordering.cpp @@ -1,8 +1,9 @@ #include #include #include -#include -#include +#include +#include +#include #include #include #include @@ -28,12 +29,12 @@ namespace ErrorCodes namespace { - ExportReplicatedMergeTreePartitionManifest makeValidManifest() + ExportReplicatedMergeTreeTaskManifest makeValidManifest() { - ExportReplicatedMergeTreePartitionManifest manifest; + ExportReplicatedMergeTreeTaskManifest manifest; manifest.transaction_id = "tx1"; manifest.query_id = "query1"; - manifest.partition_id = "2020"; + manifest.parts = {"2020_1_1_0"}; manifest.destination_database = "db1"; manifest.destination_table = "table1"; manifest.source_replica = "r1"; @@ -51,52 +52,52 @@ namespace } } -class ExportPartitionOrderingTest : public ::testing::Test +class ExportTaskOrderingTest : public ::testing::Test { protected: - ExportPartitionTaskEntriesContainer container; - ExportPartitionTaskEntriesContainer::index::type & by_key; - ExportPartitionTaskEntriesContainer::index::type & by_create_time; + ExportTaskEntriesContainer container; + ExportTaskEntriesContainer::index::type & by_key; + ExportTaskEntriesContainer::index::type & by_create_time; - ExportPartitionOrderingTest() - : by_key(container.get()) - , by_create_time(container.get()) + ExportTaskOrderingTest() + : by_key(container.get()) + , by_create_time(container.get()) { } }; -class ExportPartitionManifestBackCompatTest : public ::testing::Test +class ExportTaskManifestBackCompatTest : public ::testing::Test { }; -TEST_F(ExportPartitionOrderingTest, IterationOrderMatchesCreateTime) +TEST_F(ExportTaskOrderingTest, IterationOrderMatchesCreateTime) { time_t base_time = 1000; - ExportReplicatedMergeTreePartitionManifest manifest1; - manifest1.partition_id = "2020"; + ExportReplicatedMergeTreeTaskManifest manifest1; + manifest1.parts = {"2020_1_1_0"}; manifest1.destination_database = "db1"; manifest1.destination_table = "table1"; manifest1.transaction_id = "tx1"; manifest1.create_time = base_time + 300; // Latest - ExportReplicatedMergeTreePartitionManifest manifest2; - manifest2.partition_id = "2021"; + ExportReplicatedMergeTreeTaskManifest manifest2; + manifest2.parts = {"2021_1_1_0"}; manifest2.destination_database = "db1"; manifest2.destination_table = "table1"; manifest2.transaction_id = "tx2"; manifest2.create_time = base_time + 100; // Middle - ExportReplicatedMergeTreePartitionManifest manifest3; - manifest3.partition_id = "2022"; + ExportReplicatedMergeTreeTaskManifest manifest3; + manifest3.parts = {"2022_1_1_0"}; manifest3.destination_database = "db1"; manifest3.destination_table = "table1"; manifest3.transaction_id = "tx3"; manifest3.create_time = base_time; // Oldest - ExportReplicatedMergeTreePartitionTaskEntry entry1{manifest1, ExportReplicatedMergeTreePartitionTaskEntry::Status::PENDING, {}, {}, {}, {}}; - ExportReplicatedMergeTreePartitionTaskEntry entry2{manifest2, ExportReplicatedMergeTreePartitionTaskEntry::Status::PENDING, {}, {}, {}, {}}; - ExportReplicatedMergeTreePartitionTaskEntry entry3{manifest3, ExportReplicatedMergeTreePartitionTaskEntry::Status::PENDING, {}, {}, {}, {}}; + ExportReplicatedMergeTreeTaskEntry entry1{manifest1, ExportReplicatedMergeTreeTaskEntry::Status::PENDING, {}, {}, {}, {}}; + ExportReplicatedMergeTreeTaskEntry entry2{manifest2, ExportReplicatedMergeTreeTaskEntry::Status::PENDING, {}, {}, {}, {}}; + ExportReplicatedMergeTreeTaskEntry entry3{manifest3, ExportReplicatedMergeTreeTaskEntry::Status::PENDING, {}, {}, {}, {}}; // Insert in reverse order by_key.insert(entry1); @@ -106,17 +107,17 @@ TEST_F(ExportPartitionOrderingTest, IterationOrderMatchesCreateTime) // Verify iteration order matches create_time (ascending) auto it = by_create_time.begin(); ASSERT_NE(it, by_create_time.end()); - EXPECT_EQ(it->manifest.partition_id, "2022"); // Oldest first + EXPECT_EQ(it->manifest.transaction_id, "tx3"); // Oldest first EXPECT_EQ(it->manifest.create_time, base_time); ++it; ASSERT_NE(it, by_create_time.end()); - EXPECT_EQ(it->manifest.partition_id, "2021"); + EXPECT_EQ(it->manifest.transaction_id, "tx2"); EXPECT_EQ(it->manifest.create_time, base_time + 100); ++it; ASSERT_NE(it, by_create_time.end()); - EXPECT_EQ(it->manifest.partition_id, "2020"); + EXPECT_EQ(it->manifest.transaction_id, "tx1"); EXPECT_EQ(it->manifest.create_time, base_time + 300); ++it; @@ -124,7 +125,7 @@ TEST_F(ExportPartitionOrderingTest, IterationOrderMatchesCreateTime) } -TEST_F(ExportPartitionManifestBackCompatTest, MissingSchemaMatchModeParsesAsNullopt) +TEST_F(ExportTaskManifestBackCompatTest, MissingSchemaMatchModeParsesAsNullopt) { auto manifest = makeValidManifest(); manifest.schema_match_mode = MergeTreePartExportSchemaMatchMode::NAME; @@ -136,25 +137,25 @@ TEST_F(ExportPartitionManifestBackCompatTest, MissingSchemaMatchModeParsesAsNull oss.exceptions(std::ios::failbit); Poco::JSON::Stringifier::stringify(json, oss); - auto parsed = ExportReplicatedMergeTreePartitionManifest::fromJsonString(oss.str()); + auto parsed = ExportReplicatedMergeTreeTaskManifest::fromJsonString(oss.str()); EXPECT_FALSE(parsed.schema_match_mode.has_value()); } -TEST_F(ExportPartitionManifestBackCompatTest, SchemaMatchModeRoundTripsForEveryValue) +TEST_F(ExportTaskManifestBackCompatTest, SchemaMatchModeRoundTripsForEveryValue) { for (const auto value : magic_enum::enum_values()) { auto manifest = makeValidManifest(); manifest.schema_match_mode = value; - auto parsed = ExportReplicatedMergeTreePartitionManifest::fromJsonString(manifest.toJsonString()); + auto parsed = ExportReplicatedMergeTreeTaskManifest::fromJsonString(manifest.toJsonString()); ASSERT_TRUE(parsed.schema_match_mode.has_value()) << "value=" << magic_enum::enum_name(value); EXPECT_EQ(*parsed.schema_match_mode, value) << "value=" << magic_enum::enum_name(value); } } -TEST_F(ExportPartitionManifestBackCompatTest, MissingIgnoreExtraSourceColumnsParsesAsNullopt) +TEST_F(ExportTaskManifestBackCompatTest, MissingIgnoreExtraSourceColumnsParsesAsNullopt) { auto manifest = makeValidManifest(); manifest.ignore_extra_source_columns = true; @@ -166,31 +167,31 @@ TEST_F(ExportPartitionManifestBackCompatTest, MissingIgnoreExtraSourceColumnsPar oss.exceptions(std::ios::failbit); Poco::JSON::Stringifier::stringify(json, oss); - auto parsed = ExportReplicatedMergeTreePartitionManifest::fromJsonString(oss.str()); + auto parsed = ExportReplicatedMergeTreeTaskManifest::fromJsonString(oss.str()); EXPECT_FALSE(parsed.ignore_extra_source_columns.has_value()); } -TEST_F(ExportPartitionManifestBackCompatTest, IgnoreExtraSourceColumnsRoundTripsForEveryValue) +TEST_F(ExportTaskManifestBackCompatTest, IgnoreExtraSourceColumnsRoundTripsForEveryValue) { for (const bool value : {false, true}) { auto manifest = makeValidManifest(); manifest.ignore_extra_source_columns = value; - auto parsed = ExportReplicatedMergeTreePartitionManifest::fromJsonString(manifest.toJsonString()); + auto parsed = ExportReplicatedMergeTreeTaskManifest::fromJsonString(manifest.toJsonString()); ASSERT_TRUE(parsed.ignore_extra_source_columns.has_value()) << "value=" << value; EXPECT_EQ(*parsed.ignore_extra_source_columns, value) << "value=" << value; } } -TEST_F(ExportPartitionManifestBackCompatTest, MissingSchemaMatchSettingsFallBackToDefaultsInWorkerContext) +TEST_F(ExportTaskManifestBackCompatTest, MissingSchemaMatchSettingsFallBackToDefaultsInWorkerContext) { auto manifest = makeValidManifest(); ASSERT_FALSE(manifest.schema_match_mode.has_value()); ASSERT_FALSE(manifest.ignore_extra_source_columns.has_value()); - auto worker_context = ExportPartitionUtils::getContextCopyWithTaskSettings(getContext().context, manifest); + auto worker_context = ExportTaskUtils::getContextCopyWithTaskSettings(getContext().context, manifest); EXPECT_EQ( worker_context->getSettingsRef()[Setting::export_merge_tree_part_schema_match_mode].value, @@ -200,14 +201,14 @@ TEST_F(ExportPartitionManifestBackCompatTest, MissingSchemaMatchSettingsFallBack false); } -TEST_F(ExportPartitionManifestBackCompatTest, SchemaMatchModeAppliedToWorkerContextForEveryValue) +TEST_F(ExportTaskManifestBackCompatTest, SchemaMatchModeAppliedToWorkerContextForEveryValue) { for (const auto value : magic_enum::enum_values()) { auto manifest = makeValidManifest(); manifest.schema_match_mode = value; - auto worker_context = ExportPartitionUtils::getContextCopyWithTaskSettings(getContext().context, manifest); + auto worker_context = ExportTaskUtils::getContextCopyWithTaskSettings(getContext().context, manifest); EXPECT_EQ( worker_context->getSettingsRef()[Setting::export_merge_tree_part_schema_match_mode].value, @@ -215,14 +216,14 @@ TEST_F(ExportPartitionManifestBackCompatTest, SchemaMatchModeAppliedToWorkerCont } } -TEST_F(ExportPartitionManifestBackCompatTest, IgnoreExtraSourceColumnsAppliedToWorkerContextForEveryValue) +TEST_F(ExportTaskManifestBackCompatTest, IgnoreExtraSourceColumnsAppliedToWorkerContextForEveryValue) { for (const bool value : {false, true}) { auto manifest = makeValidManifest(); manifest.ignore_extra_source_columns = value; - auto worker_context = ExportPartitionUtils::getContextCopyWithTaskSettings(getContext().context, manifest); + auto worker_context = ExportTaskUtils::getContextCopyWithTaskSettings(getContext().context, manifest); EXPECT_EQ( worker_context->getSettingsRef()[Setting::export_merge_tree_part_ignore_extra_source_columns].value, @@ -230,19 +231,53 @@ TEST_F(ExportPartitionManifestBackCompatTest, IgnoreExtraSourceColumnsAppliedToW } } -TEST(ExportPartitionRetryClassification, MissingPartIsFatalOnlyOnPlainPath) +TEST_F(ExportTaskManifestBackCompatTest, TaskSourceAndRetryOfRoundTrip) { - EXPECT_FALSE(ExportPartitionUtils::isNonRetryableExportError(ErrorCodes::NO_SUCH_DATA_PART)); - EXPECT_TRUE(ExportPartitionUtils::isNonRetryablePlainExportError(ErrorCodes::NO_SUCH_DATA_PART)); + auto manifest = makeValidManifest(); + manifest.source = ExportTaskSource::ttl; + manifest.destination_uuid = "00000000-0000-0000-0000-000000000001"; + manifest.retry_of = {"tx0", "tx00"}; + + const auto parsed = ExportReplicatedMergeTreeTaskManifest::fromJsonString(manifest.toJsonString()); + EXPECT_EQ(parsed.source, ExportTaskSource::ttl); + EXPECT_EQ(parsed.destination_uuid, manifest.destination_uuid); + EXPECT_EQ(parsed.retry_of, manifest.retry_of); + + /// A task of `EXPORT PARTITION` records neither. + const auto query_task = ExportReplicatedMergeTreeTaskManifest::fromJsonString(makeValidManifest().toJsonString()); + EXPECT_EQ(query_task.source, ExportTaskSource::query); + EXPECT_TRUE(query_task.retry_of.empty()); +} + +TEST(ExportTaskUtils, PartitionIdIsDerivedFromParts) +{ + EXPECT_EQ(ExportTaskUtils::getPartitionIdOfParts({"2020_1_1_0", "2020_2_5_1"}, MERGE_TREE_DATA_MIN_FORMAT_VERSION_WITH_CUSTOM_PARTITIONING), "2020"); + EXPECT_EQ(ExportTaskUtils::getPartitionIdOfParts({"all_3_3_0"}, MERGE_TREE_DATA_MIN_FORMAT_VERSION_WITH_CUSTOM_PARTITIONING), "all"); +} + +TEST(ExportTaskUtils, PartsCommittedByRetriedTaskAreLeftOut) +{ + const auto committed = ExportTTLUtils::rangesOfParts({"p_1_1_0", "p_2_2_0"}, MERGE_TREE_DATA_MIN_FORMAT_VERSION_WITH_CUSTOM_PARTITIONING); + + /// A mutated part keeps its block range, so it counts as committed as well. + const auto remaining = ExportTaskUtils::getPartsNotCommitted( + {"p_1_1_0_7", "p_2_2_0", "p_3_3_0"}, committed, MERGE_TREE_DATA_MIN_FORMAT_VERSION_WITH_CUSTOM_PARTITIONING); + EXPECT_EQ(remaining, std::vector{"p_3_3_0"}); +} + +TEST(ExportTaskRetryClassification, MissingPartIsFatalOnlyOnPlainPath) +{ + EXPECT_FALSE(ExportTaskUtils::isNonRetryableExportError(ErrorCodes::NO_SUCH_DATA_PART)); + EXPECT_TRUE(ExportTaskUtils::isNonRetryablePlainExportError(ErrorCodes::NO_SUCH_DATA_PART)); - EXPECT_FALSE(ExportPartitionUtils::isNonRetryableExportError(ErrorCodes::UNKNOWN_TABLE)); - EXPECT_TRUE(ExportPartitionUtils::isNonRetryablePlainExportError(ErrorCodes::UNKNOWN_TABLE)); + EXPECT_FALSE(ExportTaskUtils::isNonRetryableExportError(ErrorCodes::UNKNOWN_TABLE)); + EXPECT_TRUE(ExportTaskUtils::isNonRetryablePlainExportError(ErrorCodes::UNKNOWN_TABLE)); - EXPECT_TRUE(ExportPartitionUtils::isNonRetryableExportError(ErrorCodes::BAD_ARGUMENTS)); - EXPECT_TRUE(ExportPartitionUtils::isNonRetryablePlainExportError(ErrorCodes::BAD_ARGUMENTS)); + EXPECT_TRUE(ExportTaskUtils::isNonRetryableExportError(ErrorCodes::BAD_ARGUMENTS)); + EXPECT_TRUE(ExportTaskUtils::isNonRetryablePlainExportError(ErrorCodes::BAD_ARGUMENTS)); - EXPECT_FALSE(ExportPartitionUtils::isNonRetryableExportError(ErrorCodes::NETWORK_ERROR)); - EXPECT_FALSE(ExportPartitionUtils::isNonRetryablePlainExportError(ErrorCodes::NETWORK_ERROR)); + EXPECT_FALSE(ExportTaskUtils::isNonRetryableExportError(ErrorCodes::NETWORK_ERROR)); + EXPECT_FALSE(ExportTaskUtils::isNonRetryablePlainExportError(ErrorCodes::NETWORK_ERROR)); } } diff --git a/src/Storages/MergeTree/tests/gtest_export_ttl_index.cpp b/src/Storages/MergeTree/tests/gtest_export_ttl_index.cpp new file mode 100644 index 000000000000..986c5e37cdce --- /dev/null +++ b/src/Storages/MergeTree/tests/gtest_export_ttl_index.cpp @@ -0,0 +1,127 @@ +#include + +#include +#include +#include + +using namespace DB; + +namespace +{ + +MergeTreePartInfo part(Int64 min_block, Int64 max_block, UInt32 level = 0, Int64 mutation = 0) +{ + return MergeTreePartInfo("p", min_block, max_block, level, mutation); +} + +std::vector> blocks(const std::vector & ranges) +{ + std::vector> result; + for (const auto & range : ranges) + result.emplace_back(range.min_block, range.max_block); + return result; +} + +using Blocks = std::vector>; + +} + +TEST(ExportTTLIndex, ClaimThenCommit) +{ + ExportTTLIndexEntry entry; + entry.partition_id = "p"; + + entry.claim("t1", {part(1, 1), part(2, 2), part(4, 4)}); + EXPECT_EQ(blocks(entry.claimed.at("t1")), (Blocks{{1, 2}, {4, 4}})); + EXPECT_EQ(entry.classify(part(1, 2, 1)), PartExportState::CLAIMED); + EXPECT_EQ(entry.classify(part(3, 3)), PartExportState::NONE); + + entry.commitClaim("t1", {}); + EXPECT_TRUE(entry.claimed.empty()); + EXPECT_EQ(blocks(entry.exported), (Blocks{{1, 2}, {4, 4}})); + /// A mutated exported part keeps its block range. + EXPECT_EQ(entry.classify(part(4, 4, 0, 7)), PartExportState::EXPORTED); + EXPECT_EQ(entry.classify(part(3, 3)), PartExportState::NONE); + EXPECT_EQ(entry.maxBlock(), 4); +} + +TEST(ExportTTLIndex, RetryReclaimsExactRanges) +{ + ExportTTLIndexEntry entry; + entry.partition_id = "p"; + entry.claim("t1", {part(1, 1), part(2, 2)}); + + /// Part 2 was dropped before the retry, part 5 became eligible meanwhile. + entry.releaseClaim("t1"); + entry.claim("t2", {part(1, 1), part(5, 5)}); + + EXPECT_FALSE(entry.claimed.contains("t1")); + EXPECT_EQ(blocks(entry.claimed.at("t2")), (Blocks{{1, 1}, {5, 5}})); + EXPECT_EQ(entry.classify(part(2, 2)), PartExportState::NONE); + EXPECT_EQ(blocks(entry.allClaimed()), (Blocks{{1, 1}, {5, 5}})); +} + +TEST(ExportTTLIndex, MoveClaims) +{ + ExportTTLIndexEntry entry; + entry.partition_id = "p"; + entry.claim("t1", {part(1, 1)}); + entry.claim("t2", {part(2, 2)}); + + entry.moveClaims({"t1", "t2", "missing"}, "t3"); + EXPECT_EQ(entry.claimed.size(), 1); + EXPECT_EQ(blocks(entry.claimed.at("t3")), (Blocks{{1, 2}})); +} + +TEST(ExportTTLIndex, CommitAddsPartsWhoseClaimWasLost) +{ + ExportTTLIndexEntry entry; + entry.partition_id = "p"; + entry.commitClaim("t1", {part(3, 5)}); + EXPECT_EQ(blocks(entry.exported), (Blocks{{3, 5}})); +} + +TEST(ExportTTLIndex, JsonRoundTrip) +{ + ExportTTLIndexEntry entry; + entry.partition_id = "p"; + entry.exported = {part(0, 10), part(20, 30)}; + entry.claim("t1", {part(31, 31)}); + + const auto parsed = ExportTTLIndexEntry::fromJSONString("p", entry.toJSONString()); + EXPECT_EQ(parsed.partition_id, "p"); + EXPECT_EQ(blocks(parsed.exported), (Blocks{{0, 10}, {20, 30}})); + ASSERT_TRUE(parsed.claimed.contains("t1")); + EXPECT_EQ(blocks(parsed.claimed.at("t1")), (Blocks{{31, 31}})); + + EXPECT_TRUE(ExportTTLIndexEntry::fromJSONString("p", "").empty()); + EXPECT_ANY_THROW(ExportTTLIndexEntry::fromJSONString("p", R"({"exported":[[1]]})")); +} + +TEST(ExportTTLIndex, DestinationKeysOfDottedNamesDiffer) +{ + EXPECT_NE(ExportTTLUtils::destinationKey("db.x", "y", ""), ExportTTLUtils::destinationKey("db", "x.y", "")); + EXPECT_NE(ExportTTLUtils::destinationKey("db", "t", ""), ExportTTLUtils::destinationKey("db", "t", "00000000-0000-0000-0000-000000000001")); +} + +TEST(ExportTTLIndex, RangesOfParts) +{ + const auto ranges = ExportTTLUtils::rangesOfParts({"p_3_3_0", "p_1_2_1", "p_5_5_0_9"}, MERGE_TREE_DATA_MIN_FORMAT_VERSION_WITH_CUSTOM_PARTITIONING); + EXPECT_EQ(blocks(ranges), (Blocks{{1, 3}, {5, 5}})); +} + +TEST(ExportTTLIndex, EligibleOnceTheMaximumIsDue) +{ + TTLDescription description; + description.result_column = "plus(t, toIntervalDay(1))"; + const TTLDescriptions descriptions{description}; + + TTLInfoMap infos; + infos[description.result_column] = MergeTreeDataPartTTLInfo{.min = 100, .max = 200, .ttl_finished = {}}; + + EXPECT_FALSE(selectTTLDescriptionForTTLInfos(descriptions, infos, 150, /* use_max */ true).has_value()); + EXPECT_TRUE(selectTTLDescriptionForTTLInfos(descriptions, infos, 200, /* use_max */ true).has_value()); + + /// A part without the TTL info is never eligible. + EXPECT_FALSE(selectTTLDescriptionForTTLInfos(descriptions, {}, 1000, /* use_max */ true).has_value()); +} diff --git a/src/Storages/ObjectStorage/DataLakes/IDataLakeMetadata.h b/src/Storages/ObjectStorage/DataLakes/IDataLakeMetadata.h index 8784705e85b2..dcb6508ab1b0 100644 --- a/src/Storages/ObjectStorage/DataLakes/IDataLakeMetadata.h +++ b/src/Storages/ObjectStorage/DataLakes/IDataLakeMetadata.h @@ -225,7 +225,7 @@ class IDataLakeMetadata : boost::noncopyable throwNotImplemented("import"); } - virtual IStorage::ExportPartitionCommitInfo commitExportPartitionTransaction( + virtual IStorage::ExportCommitInfo commitExportTransaction( std::shared_ptr /* catalog */, const StorageID & /* table_id */, const String & /* transaction_id */, @@ -237,9 +237,14 @@ class IDataLakeMetadata : boost::noncopyable StorageObjectStorageConfigurationPtr /* configuration */, ContextPtr /* context */) { - throwNotImplemented("commitExportPartitionTransaction"); + throwNotImplemented("commitExportTransaction"); } + virtual bool isExportTransactionCommitted(const String & /* transaction_id */, ContextPtr /* context */) + { + throwNotImplemented("isExportTransactionCommitted"); + } + virtual bool optimize( const StorageMetadataPtr & /*metadata_snapshot*/, ContextPtr /*context*/, const std::optional & /*format_settings*/) { diff --git a/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergMetadata.cpp b/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergMetadata.cpp index 7f7d7211c680..1cc6ac42c5fd 100644 --- a/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergMetadata.cpp +++ b/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergMetadata.cpp @@ -162,7 +162,7 @@ String dumpMetadataObjectToString(const Poco::JSON::Object::Ptr & metadata_objec /// Check if a previous attempt already committed this transaction the snapshot /// (with our transaction_id embedded in its summary) is still present in the snapshots array /// unless an external engine ran expireSnapshots in the meantime. If found, skip re-committing. -bool isExportPartitionTransactionAlreadyCommitted(const Poco::JSON::Object::Ptr & metadata, const String & transaction_id) +bool isExportTransactionAlreadyCommitted(const Poco::JSON::Object::Ptr & metadata, const String & transaction_id) { const auto throw_error = [&](const std::string & missing_field_name) { @@ -1803,7 +1803,7 @@ std::vector recomputeExportPartitionValues( } -std::optional IcebergMetadata::commitImportPartitionTransactionImpl( +std::optional IcebergMetadata::commitImportPartitionTransactionImpl( FileNamesGenerator & filename_generator, Poco::JSON::Object::Ptr & metadata, Poco::JSON::Object::Ptr & partition_spec, @@ -1826,16 +1826,16 @@ std::optional IcebergMetadata::commitImport ContextPtr context) { /// this check also exists here because the metadata might have been updated upon retry attempts. - if (isExportPartitionTransactionAlreadyCommitted(metadata, transaction_id)) + if (isExportTransactionAlreadyCommitted(metadata, transaction_id)) { LOG_INFO(log, "Export transaction {} already committed, skipping re-commit", transaction_id); /// Surface a sentinel so the caller treats this as a successful attempt (non-empty /// commit info), persists a commit_info znode, and makes the situation visible in - /// system.partition_exports.committed_metadata_file. We do not know the + /// system.distributed_exports.committed_metadata_file. We do not know the /// original committer's paths from here. - IStorage::ExportPartitionCommitInfo already_committed_info; + IStorage::ExportCommitInfo already_committed_info; already_committed_info.iceberg_metadata_file = ""; return already_committed_info; } @@ -2095,7 +2095,7 @@ std::optional IcebergMetadata::commitImport tryLogCurrentException(log, "Post-publish work failed after Iceberg snapshot was committed; " "skipping manifest cleanup to preserve published snapshot"); - IStorage::ExportPartitionCommitInfo published_info; + IStorage::ExportCommitInfo published_info; published_info.iceberg_metadata_file = resolver.resolve(metadata_info.path); published_info.iceberg_manifest_list = storage_manifest_list_name; published_info.iceberg_manifest_file = storage_manifest_entry_name; @@ -2111,14 +2111,14 @@ std::optional IcebergMetadata::commitImport /// export task can persist them in ZooKeeper for observability. Only set here /// (not on the retry / "already committed" paths) so the struct reflects /// exactly what this attempt produced. - IStorage::ExportPartitionCommitInfo published_info; + IStorage::ExportCommitInfo published_info; published_info.iceberg_metadata_file = resolver.resolve(metadata_info.path); published_info.iceberg_manifest_list = storage_manifest_list_name; published_info.iceberg_manifest_file = storage_manifest_entry_name; return published_info; } -IStorage::ExportPartitionCommitInfo IcebergMetadata::commitExportPartitionTransaction( +IStorage::ExportCommitInfo IcebergMetadata::commitExportTransaction( std::shared_ptr catalog, const StorageID & table_id, const String & transaction_id, @@ -2152,12 +2152,12 @@ IStorage::ExportPartitionCommitInfo IcebergMetadata::commitExportPartitionTransa updated_metadata_file_info.compression_method, persistent_components.table_uuid); - if (isExportPartitionTransactionAlreadyCommitted(metadata, transaction_id)) + if (isExportTransactionAlreadyCommitted(metadata, transaction_id)) { LOG_INFO(log, "Export transaction {} already committed, skipping re-commit", transaction_id); - IStorage::ExportPartitionCommitInfo already_committed_info; + IStorage::ExportCommitInfo already_committed_info; already_committed_info.iceberg_metadata_file = ""; return already_committed_info; } @@ -2259,6 +2259,31 @@ IStorage::ExportPartitionCommitInfo IcebergMetadata::commitExportPartitionTransa attempt); } +bool IcebergMetadata::isExportTransactionCommitted(const String & transaction_id, ContextPtr context) +{ + const auto latest_metadata_file_info = getLatestOrExplicitMetadataFileAndVersion( + object_storage, + persistent_components.table_path, + data_lake_settings, + persistent_components.metadata_cache, + context, + getLogger("IcebergMetadata").get(), + persistent_components.table_uuid, + persistent_components.metadata_compression_method, + true); + + const auto metadata = getMetadataJSONObject( + latest_metadata_file_info.path, + object_storage, + persistent_components.metadata_cache, + context, + getLogger("IcebergMetadata"), + latest_metadata_file_info.compression_method, + persistent_components.table_uuid); + + return isExportTransactionAlreadyCommitted(metadata, transaction_id); +} + Poco::JSON::Object::Ptr IcebergMetadata::getMetadataJSON(ContextPtr local_context) const { auto [actual_data_snapshot, actual_table_state_snapshot] = getRelevantState(local_context); diff --git a/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergMetadata.h b/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergMetadata.h index 35845ae4415b..c6ef3492fc26 100644 --- a/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergMetadata.h +++ b/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergMetadata.h @@ -164,7 +164,7 @@ class IcebergMetadata : public IDataLakeMetadata /// data_file_paths contains the metadata-path for each exported data file (as recorded in /// ZooKeeper). For every path a co-located sidecar Avro file (same path, ".avro" extension) /// must exist in the object storage; it supplies record_count and file_size_in_bytes. - IStorage::ExportPartitionCommitInfo commitExportPartitionTransaction( + IStorage::ExportCommitInfo commitExportTransaction( std::shared_ptr catalog, const StorageID & table_id, const String & transaction_id, @@ -176,6 +176,10 @@ class IcebergMetadata : public IDataLakeMetadata StorageObjectStorageConfigurationPtr configuration, ContextPtr context) override; + /// Whether a snapshot of the latest metadata carries `transaction_id` in its summary. Like the + /// check done before committing, it cannot see snapshots removed by `expire_snapshots`. + bool isExportTransactionCommitted(const String & transaction_id, ContextPtr context) override; + CompressionMethod getCompressionMethod() const { return persistent_components.metadata_compression_method; } std::string getTableLocation() const override { return persistent_components.table_location; } @@ -250,11 +254,11 @@ class IcebergMetadata : public IDataLakeMetadata getRelevantDataSnapshotFromTableStateSnapshot(Iceberg::TableStateSnapshot table_state_snapshot, ContextPtr local_context) const; /// Non-empty return value means the attempt succeeded (covers both the normal - /// publish path and the `isExportPartitionTransactionAlreadyCommitted` short-circuit). - /// An empty `ExportPartitionCommitInfo` means the caller must retry. The + /// publish path and the `isExportTransactionAlreadyCommitted` short-circuit). + /// An empty `ExportCommitInfo` means the caller must retry. The /// short-circuit branch fills `iceberg_metadata_file` with a sentinel note since /// the original committer's paths are not trivially recoverable from inside this call. - std::optional commitImportPartitionTransactionImpl( + std::optional commitImportPartitionTransactionImpl( FileNamesGenerator & filename_generator, Poco::JSON::Object::Ptr & metadata, Poco::JSON::Object::Ptr & partition_spec, diff --git a/src/Storages/ObjectStorage/StorageObjectStorage.cpp b/src/Storages/ObjectStorage/StorageObjectStorage.cpp index e7913b4ec0fc..2b080304f207 100644 --- a/src/Storages/ObjectStorage/StorageObjectStorage.cpp +++ b/src/Storages/ObjectStorage/StorageObjectStorage.cpp @@ -904,45 +904,58 @@ IStorage::ImportResult StorageObjectStorage::import( local_context)); } -IStorage::ExportPartitionCommitInfo StorageObjectStorage::commitExportPartitionTransaction( +/// The commit file is in the directory of the table: for a path with wildcards, such as +/// `table/{_partition_id}/{_file}.parquet`, the directory before the first of them. +static String getExportCommitMarkerPath(const StorageObjectStorageConfiguration & configuration, const String & transaction_id) +{ + String directory = configuration.getRawPath().path; + if (const auto wildcard = directory.find('{'); wildcard != String::npos) + { + const auto slash = directory.rfind('/', wildcard); + directory = slash == String::npos ? "" : directory.substr(0, slash); + } + return directory.empty() ? "commit_" + transaction_id : directory + "/commit_" + transaction_id; +} + +IStorage::ExportCommitInfo StorageObjectStorage::commitExportTransaction( const String & transaction_id, - const String & partition_id, + const String & /* partition_id */, const Strings & exported_paths, - const IcebergCommitExportPartitionArguments & iceberg_commit_export_partition_arguments, + const IcebergCommitExportArguments & iceberg_commit_export_arguments, ContextPtr local_context) { if (isDataLake()) { /// Parse the Iceberg metadata snapshot (stored in ZooKeeper at export-start time) only to /// extract the schema-id and partition-spec-id that were current when the export began. - /// partition_columns and partition_types are derived inside commitExportPartitionTransaction + /// partition_columns and partition_types are derived inside commitExportTransaction /// from the same JSON; the representative source partition columns are carried here so the /// partition tuple can be recomputed through the destination transform. Poco::JSON::Parser iceberg_parser; Poco::JSON::Object::Ptr iceberg_metadata = - iceberg_parser.parse(iceberg_commit_export_partition_arguments.metadata_json_string).extract(); + iceberg_parser.parse(iceberg_commit_export_arguments.metadata_json_string).extract(); const auto original_schema_id = iceberg_metadata->getValue(Iceberg::f_current_schema_id); const auto partition_spec_id = iceberg_metadata->getValue(Iceberg::f_default_spec_id); configuration->lazyInitializeIfNeeded(object_storage, local_context); auto metadata_snapshot = getInMemoryMetadataPtr(local_context, false); - return configuration->getExternalMetadata()->commitExportPartitionTransaction( + return configuration->getExternalMetadata()->commitExportTransaction( catalog, storage_id, transaction_id, original_schema_id, partition_spec_id, - iceberg_commit_export_partition_arguments.partition_source_block, + iceberg_commit_export_arguments.partition_source_block, std::make_shared(metadata_snapshot->getSampleBlock()), exported_paths, configuration, local_context); } - const String commit_object = configuration->getRawPath().path + "/commit_" + partition_id + "_" + transaction_id; + const String commit_object = getExportCommitMarkerPath(*configuration, transaction_id); - ExportPartitionCommitInfo result; + ExportCommitInfo result; result.commit_marker_file = commit_object; /// if file already exists, nothing to be done @@ -964,6 +977,19 @@ IStorage::ExportPartitionCommitInfo StorageObjectStorage::commitExportPartitionT return result; } +bool StorageObjectStorage::isExportTransactionCommitted( + const String & transaction_id, + ContextPtr local_context) +{ + if (isDataLake()) + { + configuration->lazyInitializeIfNeeded(object_storage, local_context); + return configuration->getExternalMetadata()->isExportTransactionCommitted(transaction_id, local_context); + } + + return object_storage->exists(StoredObject(getExportCommitMarkerPath(*configuration, transaction_id))); +} + void StorageObjectStorage::truncate( const ASTPtr & /* query */, const StorageMetadataPtr & /* metadata_snapshot */, diff --git a/src/Storages/ObjectStorage/StorageObjectStorage.h b/src/Storages/ObjectStorage/StorageObjectStorage.h index 474e337e17ae..f5af08e647f3 100644 --- a/src/Storages/ObjectStorage/StorageObjectStorage.h +++ b/src/Storages/ObjectStorage/StorageObjectStorage.h @@ -96,11 +96,15 @@ class StorageObjectStorage : public IStorage, public IBackgroundOperation const std::optional & /* format_settings_ */, ContextPtr /* context */) override; - ExportPartitionCommitInfo commitExportPartitionTransaction( + ExportCommitInfo commitExportTransaction( const String & transaction_id, const String & partition_id, const Strings & exported_paths, - const IcebergCommitExportPartitionArguments & iceberg_commit_export_partition_arguments, + const IcebergCommitExportArguments & iceberg_commit_export_arguments, + ContextPtr local_context) override; + + bool isExportTransactionCommitted( + const String & transaction_id, ContextPtr local_context) override; void truncate( diff --git a/src/Storages/ObjectStorage/StorageObjectStorageCluster.cpp b/src/Storages/ObjectStorage/StorageObjectStorageCluster.cpp index aff028efef01..8edd541916ba 100644 --- a/src/Storages/ObjectStorage/StorageObjectStorageCluster.cpp +++ b/src/Storages/ObjectStorage/StorageObjectStorageCluster.cpp @@ -1165,30 +1165,39 @@ IStorage::ImportResult StorageObjectStorageCluster::import( context); } -IStorage::ExportPartitionCommitInfo StorageObjectStorageCluster::commitExportPartitionTransaction( +IStorage::ExportCommitInfo StorageObjectStorageCluster::commitExportTransaction( const String & transaction_id, const String & partition_id, const Strings & exported_paths, - const IcebergCommitExportPartitionArguments & iceberg_commit_export_partition_arguments, + const IcebergCommitExportArguments & iceberg_commit_export_arguments, ContextPtr local_context) { if (pure_storage) { - return pure_storage->commitExportPartitionTransaction( + return pure_storage->commitExportTransaction( transaction_id, partition_id, exported_paths, - iceberg_commit_export_partition_arguments, + iceberg_commit_export_arguments, local_context ); } - return IStorageCluster::commitExportPartitionTransaction( + return IStorageCluster::commitExportTransaction( transaction_id, partition_id, exported_paths, - iceberg_commit_export_partition_arguments, + iceberg_commit_export_arguments, local_context ); } +bool StorageObjectStorageCluster::isExportTransactionCommitted( + const String & transaction_id, + ContextPtr local_context) +{ + if (pure_storage) + return pure_storage->isExportTransactionCommitted(transaction_id, local_context); + return IStorageCluster::isExportTransactionCommitted(transaction_id, local_context); +} + } diff --git a/src/Storages/ObjectStorage/StorageObjectStorageCluster.h b/src/Storages/ObjectStorage/StorageObjectStorageCluster.h index 28cdaeb29643..db683ff02a3f 100644 --- a/src/Storages/ObjectStorage/StorageObjectStorageCluster.h +++ b/src/Storages/ObjectStorage/StorageObjectStorageCluster.h @@ -44,11 +44,15 @@ class StorageObjectStorageCluster : public IStorageCluster const std::optional & format_settings_, ContextPtr context) override; - ExportPartitionCommitInfo commitExportPartitionTransaction( + ExportCommitInfo commitExportTransaction( const String & transaction_id, const String & partition_id, const Strings & exported_paths, - const IcebergCommitExportPartitionArguments & iceberg_commit_export_partition_arguments, + const IcebergCommitExportArguments & iceberg_commit_export_arguments, + ContextPtr local_context) override; + + bool isExportTransactionCommitted( + const String & transaction_id, ContextPtr local_context) override; RemoteQueryExecutor::Extension getTaskIteratorExtension( diff --git a/src/Storages/StorageInMemoryMetadata.cpp b/src/Storages/StorageInMemoryMetadata.cpp index cea6a8a03446..3a93453b153c 100644 --- a/src/Storages/StorageInMemoryMetadata.cpp +++ b/src/Storages/StorageInMemoryMetadata.cpp @@ -312,12 +312,13 @@ TTLTableDescription StorageInMemoryMetadata::getTableTTLs() const bool StorageInMemoryMetadata::hasAnyTableTTL() const { - return hasAnyMoveTTL() || hasRowsTTL() || hasAnyRecompressionTTL() || hasAnyGroupByTTL() || hasAnyRowsWhereTTL(); + return hasAnyMoveTTL() || hasRowsTTL() || hasAnyRecompressionTTL() || hasAnyGroupByTTL() || hasAnyRowsWhereTTL() || hasAnyExportTTL(); } bool StorageInMemoryMetadata::hasOnlyRowsTTL() const { - bool has_any_other_ttl = hasAnyMoveTTL() || hasAnyRecompressionTTL() || hasAnyGroupByTTL() || hasAnyRowsWhereTTL() || hasAnyColumnTTL(); + bool has_any_other_ttl = hasAnyMoveTTL() || hasAnyRecompressionTTL() || hasAnyGroupByTTL() || hasAnyRowsWhereTTL() || hasAnyColumnTTL() + || hasAnyExportTTL(); return hasRowsTTL() && !has_any_other_ttl; } @@ -381,6 +382,16 @@ bool StorageInMemoryMetadata::hasAnyGroupByTTL() const return !table_ttl.group_by_ttl.empty(); } +TTLDescriptions StorageInMemoryMetadata::getExportTTLs() const +{ + return table_ttl.export_ttl; +} + +bool StorageInMemoryMetadata::hasAnyExportTTL() const +{ + return !table_ttl.export_ttl.empty(); +} + ColumnDependencies StorageInMemoryMetadata::getColumnDependencies( const NameSet & updated_columns, bool include_ttl_target, @@ -460,6 +471,9 @@ ColumnDependencies StorageInMemoryMetadata::getColumnDependencies( for (const auto & entry : getMoveTTLs()) add_dependent_columns(entry.expression_columns.getNames(), required_ttl_columns); + for (const auto & entry : getExportTTLs()) + add_dependent_columns(entry.expression_columns.getNames(), required_ttl_columns); + //TODO what about rows_where_ttl and group_by_ttl ?? for (const auto & column : indices_columns) diff --git a/src/Storages/StorageInMemoryMetadata.h b/src/Storages/StorageInMemoryMetadata.h index 7a1475bc4e82..04b2714b0dc9 100644 --- a/src/Storages/StorageInMemoryMetadata.h +++ b/src/Storages/StorageInMemoryMetadata.h @@ -204,6 +204,10 @@ struct StorageInMemoryMetadata TTLDescriptions getGroupByTTLs() const; bool hasAnyGroupByTTL() const; + /// Just wrapper for table TTLs, return the `EXPORT TO TABLE` TTL (at most one). + TTLDescriptions getExportTTLs() const; + bool hasAnyExportTTL() const; + using HasDependencyCallback = std::function; /// Returns columns, which will be needed to calculate dependencies (skip indices, projections, diff --git a/src/Storages/StorageMergeTree.cpp b/src/Storages/StorageMergeTree.cpp index 4d6798a12a1c..459882ac0933 100644 --- a/src/Storages/StorageMergeTree.cpp +++ b/src/Storages/StorageMergeTree.cpp @@ -52,9 +52,10 @@ #include #include #include -#include -#include -#include +#include +#include +#include +#include #include #include #include @@ -113,10 +114,9 @@ namespace Setting extern const SettingsBool throw_on_unsupported_query_inside_transaction; extern const SettingsUInt64 max_parts_to_move; extern const SettingsUpdateParallelMode update_parallel_mode; - extern const SettingsBool export_merge_tree_partition_force_export; - extern const SettingsUInt64 export_merge_tree_partition_retry_initial_backoff_seconds; - extern const SettingsUInt64 export_merge_tree_partition_retry_max_backoff_seconds; - extern const SettingsUInt64 export_merge_tree_partition_task_timeout_seconds; + extern const SettingsUInt64 export_merge_tree_retry_initial_backoff_seconds; + extern const SettingsUInt64 export_merge_tree_retry_max_backoff_seconds; + extern const SettingsUInt64 export_merge_tree_task_timeout_seconds; extern const SettingsBool output_format_parallel_formatting; extern const SettingsBool output_format_parquet_parallel_encoding; extern const SettingsParquetCompression output_format_parquet_compression_method; @@ -273,15 +273,29 @@ StorageMergeTree::StorageMergeTree( if (getContext()->getServerSettings()[ServerSetting::allow_experimental_export_merge_tree_partition]) { - partition_export_scheduler = std::make_shared(*this); + export_task_scheduler = std::make_shared(*this); - partition_export_task = getContext()->getSchedulePool().createTask( + export_task_scheduling_task = getContext()->getSchedulePool().createTask( getStorageID(), - getStorageID().getFullTableName() + " (StorageMergeTree::partition_export_task)", - [this] { partitionExportTask(); }); + getStorageID().getFullTableName() + " (StorageMergeTree::export_task_scheduling_task)", + [this] { exportTaskSchedulingTask(); }); /// Activated in startup(); deactivated during shutdown. - partition_export_task->deactivate(); + export_task_scheduling_task->deactivate(); + + export_ttl_index = std::make_shared(*this); + export_ttl_index->load(); + + /// Block numbers of parts exported by the `EXPORT` TTL must not be reused, even if the + /// parts no longer exist, or new parts would look exported. + increment.set(std::max(increment.value.load(), static_cast(export_ttl_index->maxBlock()))); + + export_ttl_scheduler = std::make_shared(*this, *export_ttl_index); + export_ttl_task = getContext()->getSchedulePool().createTask( + getStorageID(), + getStorageID().getFullTableName() + " (StorageMergeTree::export_ttl_task)", + [this] { exportTTLTask(); }); + export_ttl_task->deactivate(); } } @@ -304,10 +318,11 @@ void StorageMergeTree::startup() { /// Reload persisted partition-export tasks (and re-pin their parts) before background merges /// can start removing parts, then activate the scheduler task so PENDING tasks resume. - if (partition_export_scheduler) + if (export_task_scheduler) { - partition_export_scheduler->load(); - partition_export_task->activateAndSchedule(); + export_task_scheduler->load(); + export_task_scheduling_task->activateAndSchedule(); + export_ttl_task->activateAndSchedule(); } cleanup_thread.start(); @@ -346,8 +361,10 @@ void StorageMergeTree::flushAndPrepareForShutdown() merger_mutator.merges_blocker.cancelForever(); parts_mover.moves_blocker.cancelForever(); - if (partition_export_task) - partition_export_task->deactivate(); + if (export_task_scheduling_task) + export_task_scheduling_task->deactivate(); + if (export_ttl_task) + export_ttl_task->deactivate(); background_operations_assignee.finish(); background_moves_assignee.finish(); @@ -3726,7 +3743,7 @@ void StorageMergeTree::exportPartitionToTable(const PartitionCommand & command, /// The scheduler is created in the constructor whenever the server setting above is enabled, so /// this should always hold here. - if (!partition_export_scheduler) + if (!export_task_scheduler) throw Exception(ErrorCodes::SUPPORT_IS_DISABLED, "Partition export is not initialized for table {}", getStorageID().getNameForLogs()); @@ -3818,13 +3835,11 @@ void StorageMergeTree::exportPartitionToTable(const PartitionCommand & command, auto destination_snapshot = dest_storage->getInMemoryMetadataPtr(query_context, false); /// Positional CAST matching, like `INSERT INTO dest SELECT * FROM src`. - ExportPartitionUtils::verifyExportSchemaCastable( + ExportTaskUtils::verifyExportSchemaCastable( src_snapshot, destination_snapshot, dest_storage->getStorageID(), query_context); const String partition_id = getPartitionIDFromQuery(command.partition, query_context); - const bool force = query_context->getSettingsRef()[Setting::export_merge_tree_partition_force_export]; - DataPartsVector parts; { auto data_parts_lock = lockParts(); @@ -3834,6 +3849,25 @@ void StorageMergeTree::exportPartitionToTable(const PartitionCommand & command, if (parts.empty()) throw Exception(ErrorCodes::BAD_ARGUMENTS, "Partition {} doesn't exist", partition_id); + /// Every `EXPORT PARTITION` is a new task, even if the partition was exported before. + auto descriptor = buildExportTask(dest_storage_id, dest_storage, src_snapshot, destination_snapshot, parts, partition_id, query_context); + descriptor.transaction_id = toString(UUIDHelpers::generateV4()); + descriptor.query_id = query_context->getCurrentQueryId(); + descriptor.source = ExportTaskSource::query; + + std::vector part_references(parts.begin(), parts.end()); + export_task_scheduler->addTask(std::move(descriptor), std::move(part_references)); +} + +MergeTreeExportTask StorageMergeTree::buildExportTask( + const StorageID & dest_storage_id, + const StoragePtr & dest_storage, + const StorageMetadataPtr & src_snapshot, + const StorageMetadataPtr & destination_snapshot, + const DataPartsVector & parts, + const String & partition_id, + ContextPtr query_context) const +{ const bool throw_on_pending_mutations = query_context->getSettingsRef()[Setting::export_merge_tree_part_throw_on_pending_mutations]; const bool throw_on_pending_patch_parts = query_context->getSettingsRef()[Setting::export_merge_tree_part_throw_on_pending_patch_parts]; @@ -3858,29 +3892,28 @@ void StorageMergeTree::exportPartitionToTable(const PartitionCommand & command, "Partition {} can not be exported because the part {} has pending mutations. Either wait for the mutations to be applied or set `export_merge_tree_part_throw_on_pending_mutations` to false", partition_id, part->name); - if (alter_conversions->hasPatches()) + if (alter_conversions->hasPatches() && throw_on_pending_patch_parts) throw Exception(ErrorCodes::PENDING_MUTATIONS_NOT_ALLOWED, "Partition {} can not be exported because the part {} has pending patch parts. Either wait for the patch parts to be applied or set `export_merge_tree_part_throw_on_pending_patch_parts` to false", partition_id, part->name); } - MergeTreePartitionExportTask descriptor; - descriptor.transaction_id = toString(UUIDHelpers::generateV4()); - descriptor.query_id = query_context->getCurrentQueryId(); - descriptor.partition_id = partition_id; + MergeTreeExportTask descriptor; descriptor.source_database = getStorageID().database_name; descriptor.source_table = getStorageID().table_name; - descriptor.destination_database = dest_database; - descriptor.destination_table = dest_table; + descriptor.destination_database = dest_storage_id.database_name; + descriptor.destination_table = dest_storage_id.table_name; + if (const auto uuid = dest_storage->getStorageID().uuid; uuid != UUIDHelpers::Nil) + descriptor.destination_uuid = toString(uuid); descriptor.create_time = time(nullptr); - descriptor.status = MergeTreePartitionExportTask::Status::PENDING; + descriptor.status = MergeTreeExportTask::Status::PENDING; for (const auto & part : parts) descriptor.parts.push_back({part->name, /*done*/ false, /*paths*/ {}}); - descriptor.retry_initial_backoff_seconds = query_context->getSettingsRef()[Setting::export_merge_tree_partition_retry_initial_backoff_seconds]; - descriptor.retry_max_backoff_seconds = query_context->getSettingsRef()[Setting::export_merge_tree_partition_retry_max_backoff_seconds]; - descriptor.task_timeout_seconds = query_context->getSettingsRef()[Setting::export_merge_tree_partition_task_timeout_seconds]; + descriptor.retry_initial_backoff_seconds = query_context->getSettingsRef()[Setting::export_merge_tree_retry_initial_backoff_seconds]; + descriptor.retry_max_backoff_seconds = query_context->getSettingsRef()[Setting::export_merge_tree_retry_max_backoff_seconds]; + descriptor.task_timeout_seconds = query_context->getSettingsRef()[Setting::export_merge_tree_task_timeout_seconds]; descriptor.max_threads = query_context->getSettingsRef()[Setting::max_threads]; descriptor.parallel_formatting = query_context->getSettingsRef()[Setting::output_format_parallel_formatting]; descriptor.parquet_parallel_encoding = query_context->getSettingsRef()[Setting::output_format_parquet_parallel_encoding]; @@ -3901,7 +3934,7 @@ void StorageMergeTree::exportPartitionToTable(const PartitionCommand & command, if (dest_storage->isDataLake()) { #if USE_AVRO - descriptor.iceberg_metadata_json = ExportPartitionUtils::verifyAndExtractDestinationIcebergMetadataJson( + descriptor.iceberg_metadata_json = ExportTaskUtils::verifyAndExtractDestinationIcebergMetadataJson( src_snapshot, destination_snapshot, dest_storage, @@ -3917,37 +3950,36 @@ void StorageMergeTree::exportPartitionToTable(const PartitionCommand & command, } else { - ExportPartitionUtils::verifyPlainPartitionCompatibility( + ExportTaskUtils::verifyPlainPartitionCompatibility( src_snapshot, destination_snapshot, parts, partition_id, query_context); } - std::vector part_references(parts.begin(), parts.end()); - partition_export_scheduler->addTask(std::move(descriptor), std::move(part_references), force); + return descriptor; } -CancellationCode StorageMergeTree::killExportPartition(const String & transaction_id) +CancellationCode StorageMergeTree::killExportTask(const String & transaction_id) { - if (!partition_export_scheduler) + if (!export_task_scheduler) return CancellationCode::NotFound; - return partition_export_scheduler->kill(transaction_id); + return export_task_scheduler->kill(transaction_id); } -std::vector StorageMergeTree::getPartitionExportsInfo() const +std::vector StorageMergeTree::getExportTasksInfo() const { - if (!partition_export_scheduler) + if (!export_task_scheduler) return {}; - return partition_export_scheduler->getInfo(); + return export_task_scheduler->getInfo(); } -void StorageMergeTree::partitionExportTask() +void StorageMergeTree::exportTaskSchedulingTask() { /// Reschedule only while there is pending export work. When the scheduler reports no pending /// tasks the schedule-pool task goes idle (no periodic wakeups per table); it is re-armed by - /// triggerPartitionExportTask() on a new EXPORT PARTITION and by load() at startup. + /// triggerExportTaskScheduling() on a new EXPORT PARTITION and by load() at startup. bool has_pending_work = true; try { - has_pending_work = partition_export_scheduler->run(); + has_pending_work = export_task_scheduler->run(); } catch (...) { @@ -3957,12 +3989,37 @@ void StorageMergeTree::partitionExportTask() } if (has_pending_work) - partition_export_task->scheduleAfter(5000); + export_task_scheduling_task->scheduleAfter(5000); +} + +void StorageMergeTree::wakeUpExportTTL() +{ + if (export_ttl_task) + export_ttl_task->schedule(); +} + +void StorageMergeTree::exportTTLTask() +{ + UInt64 next_run_ms = 10000; + try + { + next_run_ms = export_ttl_scheduler->run(); + } + catch (...) + { + tryLogCurrentException(log, __PRETTY_FUNCTION__); + } + export_ttl_task->scheduleAfter(next_run_ms); +} + +ExportFencePtr StorageMergeTree::getExportFence() const +{ + return export_ttl_index ? export_ttl_index->getFence() : nullptr; } -void StorageMergeTree::triggerPartitionExportTask() +void StorageMergeTree::triggerExportTaskScheduling() { - if (partition_export_task) - partition_export_task->schedule(); + if (export_task_scheduling_task) + export_task_scheduling_task->schedule(); } } diff --git a/src/Storages/StorageMergeTree.h b/src/Storages/StorageMergeTree.h index d6c8c03c39db..90f1caa3d4e4 100644 --- a/src/Storages/StorageMergeTree.h +++ b/src/Storages/StorageMergeTree.h @@ -6,7 +6,8 @@ #include #include #include -#include +#include +#include #include #include #include @@ -30,6 +31,7 @@ namespace DB { class PreparedSetsCache; +class MergeTreeExportTTLIndex; using PreparedSetsCachePtr = std::shared_ptr; /** See the description of the data structure in MergeTreeData. @@ -129,26 +131,50 @@ class StorageMergeTree final : public MergeTreeData MergeTreeDeduplicationLog * getDeduplicationLog() { return deduplication_log.get(); } /// EXPORT PARTITION for a plain (non-replicated) MergeTree table. Coordinated locally by - /// `partition_export_scheduler`; the task descriptor is persisted on disk (no ZooKeeper). + /// `export_task_scheduler`; the task descriptor is persisted on disk (no ZooKeeper). void exportPartitionToTable(const PartitionCommand & command, ContextPtr query_context) override; - CancellationCode killExportPartition(const String & transaction_id) override; + CancellationCode killExportTask(const String & transaction_id) override; - /// Snapshot of local partition-export tasks for `system.partition_exports`. No disk I/O. - std::vector getPartitionExportsInfo() const override; + /// Snapshot of local partition-export tasks for `system.distributed_exports`. No disk I/O. + std::vector getExportTasksInfo() const override; + + /// Export states of parts for the `EXPORT` TTL, nullptr if partition export is disabled. + ExportFencePtr getExportFence() const; + ExportFencePtr getLatestExportFence() const override { return getExportFence(); } private: - friend class MergeTreePartitionExportScheduler; + friend class MergeTreeExportTaskScheduler; + friend class MergeTreeExportTTLScheduler; + friend class MergeTreeExportTTLIndex; + + /// Builds the descriptor of an export task of `parts` from the settings of `query_context`, and + /// validates the destination for them. The caller sets the transaction id and the source. + MergeTreeExportTask buildExportTask( + const StorageID & dest_storage_id, + const StoragePtr & dest_storage, + const StorageMetadataPtr & src_snapshot, + const StorageMetadataPtr & destination_snapshot, + const DataPartsVector & parts, + const String & partition_id, + ContextPtr query_context) const; + + /// The export index of the `EXPORT` TTL. Only created, like the scheduler, when partition export is enabled. + std::shared_ptr export_ttl_index; + BackgroundSchedulePoolTaskHolder export_ttl_task; + void exportTTLTask(); + /// E.g. when a task of the `EXPORT` TTL finished, so it is recorded as exported right away. + void wakeUpExportTTL(); /// Local coordinator + on-disk state for EXPORT PARTITION. Only created when the server setting /// `allow_experimental_export_merge_tree_partition` is enabled. - std::shared_ptr partition_export_scheduler; - BackgroundSchedulePoolTaskHolder partition_export_task; + std::shared_ptr export_task_scheduler; + BackgroundSchedulePoolTaskHolder export_task_scheduling_task; - /// Schedule-pool task body: drives partition_export_scheduler->run() periodically. - void partitionExportTask(); + /// Schedule-pool task body: drives export_task_scheduler->run() periodically. + void exportTaskSchedulingTask(); /// Wakes up the partition-export schedule-pool task (no-op when the feature is disabled). - void triggerPartitionExportTask(); + void triggerExportTaskScheduling(); /// Mutex and condvar for synchronous mutations wait diff --git a/src/Storages/StorageReplicatedMergeTree.cpp b/src/Storages/StorageReplicatedMergeTree.cpp index 6b8f70f249f1..5a4a76dfa9cb 100644 --- a/src/Storages/StorageReplicatedMergeTree.cpp +++ b/src/Storages/StorageReplicatedMergeTree.cpp @@ -83,8 +83,8 @@ #include #include #include -#include -#include +#include +#include #include #include @@ -136,8 +136,8 @@ #include #include "Interpreters/StorageID.h" #include "QueryPipeline/QueryPlanResourceHolder.h" -#include "Storages/ExportReplicatedMergeTreePartitionManifest.h" -#include "Storages/ExportReplicatedMergeTreePartitionTaskEntry.h" +#include "Storages/ExportReplicatedMergeTreeTaskManifest.h" +#include "Storages/ExportReplicatedMergeTreeTaskEntry.h" #include #include #include @@ -176,15 +176,15 @@ namespace ProfileEvents extern const Event ZooKeeperWatchTriggeredReplicatedMergeTreeLeaderElection; extern const Event ZooKeeperWatchTriggeredReplicatedMergeTreeReplicaSync; extern const Event ZooKeeperWatchTriggeredReplicatedMergeTreeMutations; - extern const Event ExportPartitionZooKeeperRequests; - extern const Event ExportPartitionZooKeeperGet; - extern const Event ExportPartitionZooKeeperGetChildren; - extern const Event ExportPartitionZooKeeperCreate; - extern const Event ExportPartitionZooKeeperSet; - extern const Event ExportPartitionZooKeeperRemove; - extern const Event ExportPartitionZooKeeperRemoveRecursive; - extern const Event ExportPartitionZooKeeperMulti; - extern const Event ExportPartitionZooKeeperExists; + extern const Event ExportTaskZooKeeperRequests; + extern const Event ExportTaskZooKeeperGet; + extern const Event ExportTaskZooKeeperGetChildren; + extern const Event ExportTaskZooKeeperCreate; + extern const Event ExportTaskZooKeeperSet; + extern const Event ExportTaskZooKeeperRemove; + extern const Event ExportTaskZooKeeperRemoveRecursive; + extern const Event ExportTaskZooKeeperMulti; + extern const Event ExportTaskZooKeeperExists; } namespace CurrentMetrics @@ -223,10 +223,9 @@ namespace Setting extern const SettingsUInt64 select_sequential_consistency; extern const SettingsBool update_sequential_consistency; extern const SettingsBool allow_experimental_export_merge_tree_part; - extern const SettingsBool export_merge_tree_partition_force_export; - extern const SettingsUInt64 export_merge_tree_partition_retry_initial_backoff_seconds; - extern const SettingsUInt64 export_merge_tree_partition_retry_max_backoff_seconds; - extern const SettingsUInt64 export_merge_tree_partition_task_timeout_seconds; + extern const SettingsUInt64 export_merge_tree_retry_initial_backoff_seconds; + extern const SettingsUInt64 export_merge_tree_retry_max_backoff_seconds; + extern const SettingsUInt64 export_merge_tree_task_timeout_seconds; extern const SettingsBool output_format_parallel_formatting; extern const SettingsBool output_format_parquet_parallel_encoding; extern const SettingsParquetCompression output_format_parquet_compression_method; @@ -505,7 +504,7 @@ StorageReplicatedMergeTree::StorageReplicatedMergeTree( , merge_strategy_picker(*this) , queue(*this, merge_strategy_picker) , fetcher(*this) - , export_partition_manifests(std::make_unique()) + , export_partition_manifests(std::make_unique()) , cleanup_thread(*this) , deduplication_hashes_cache(*this, "deduplication_hashes") , async_block_ids_cache(*this, "async_blocks") @@ -571,26 +570,33 @@ StorageReplicatedMergeTree::StorageReplicatedMergeTree( if (getContext()->getServerSettings()[ServerSetting::allow_experimental_export_merge_tree_partition]) { - export_merge_tree_partition_manifest_updater = std::make_shared(*this); + export_task_updater = std::make_shared(*this); - export_merge_tree_partition_task_scheduler = std::make_shared(*this); + export_task_scheduler = std::make_shared(*this); - export_merge_tree_partition_updating_task = getContext()->getSchedulePool().createTask( - getStorageID(), getStorageID().getFullTableName() + " (StorageReplicatedMergeTree::export_merge_tree_partition_updating_task)", [this] { exportMergeTreePartitionUpdatingTask(); }); + export_task_updating_task = getContext()->getSchedulePool().createTask( + getStorageID(), getStorageID().getFullTableName() + " (StorageReplicatedMergeTree::export_task_updating_task)", [this] { exportTaskUpdatingTask(); }); - export_merge_tree_partition_updating_task->deactivate(); + export_task_updating_task->deactivate(); - export_merge_tree_partition_status_handling_task = getContext()->getSchedulePool().createTask( - getStorageID(), getStorageID().getFullTableName() + " (StorageReplicatedMergeTree::export_merge_tree_partition_status_handling_task)", [this] { exportMergeTreePartitionStatusHandlingTask(); }); + export_task_status_handling_task = getContext()->getSchedulePool().createTask( + getStorageID(), getStorageID().getFullTableName() + " (StorageReplicatedMergeTree::export_task_status_handling_task)", [this] { exportTaskStatusHandlingTask(); }); - export_merge_tree_partition_status_handling_task->deactivate(); + export_task_status_handling_task->deactivate(); - export_merge_tree_partition_watch_callback = export_merge_tree_partition_updating_task->getWatchCallback(); + export_task_watch_callback = export_task_updating_task->getWatchCallback(); - export_merge_tree_partition_select_task = getContext()->getSchedulePool().createTask( - getStorageID(), getStorageID().getFullTableName() + " (StorageReplicatedMergeTree::export_merge_tree_partition_select_task)", [this] { selectPartsToExport(); }); + export_task_select_task = getContext()->getSchedulePool().createTask( + getStorageID(), getStorageID().getFullTableName() + " (StorageReplicatedMergeTree::export_task_select_task)", [this] { selectPartsToExport(); }); - export_merge_tree_partition_select_task->deactivate(); + export_task_select_task->deactivate(); + + export_fence = std::make_shared(zookeeper_path, log.load()); + + export_ttl_scheduler = std::make_shared(*this); + export_ttl_task = getContext()->getSchedulePool().createTask( + getStorageID(), getStorageID().getFullTableName() + " (StorageReplicatedMergeTree::export_ttl_task)", [this] { exportTTLTask(); }); + export_ttl_task->deactivate(); } @@ -1055,6 +1061,7 @@ void StorageReplicatedMergeTree::createNewZooKeeperNodesAttempt() const futures.push_back(zookeeper->asyncTryCreateNoThrow(zookeeper_path + "/quorum/failed_parts", String(), zkutil::CreateMode::Persistent)); futures.push_back(zookeeper->asyncTryCreateNoThrow(zookeeper_path + "/mutations", String(), zkutil::CreateMode::Persistent)); futures.push_back(zookeeper->asyncTryCreateNoThrow(zookeeper_path + "/exports", String(), zkutil::CreateMode::Persistent)); + futures.push_back(zookeeper->asyncTryCreateNoThrow(zookeeper_path + "/export_fence", String(), zkutil::CreateMode::Persistent)); futures.push_back(zookeeper->asyncTryCreateNoThrow(zookeeper_path + "/quorum/parallel", String(), zkutil::CreateMode::Persistent)); @@ -4582,7 +4589,10 @@ void StorageReplicatedMergeTree::mergeSelectingTask() if (partitions_to_merge_in.empty()) can_assign_merge = false; else + { merge_predicate = queue.getMergePredicate(zookeeper, partitions_to_merge_in); + merge_predicate->loadExportFence(zookeeper); + } } PreformattedMessage out_reason; @@ -4631,6 +4641,7 @@ void StorageReplicatedMergeTree::mergeSelectingTask() cleanup, nullptr, merge_predicate->getVersion(), + merge_predicate->getExportFenceVersion(), future_merged_part->merge_type); if (create_result == CreateMergeEntryResult::Ok) @@ -4769,12 +4780,12 @@ void StorageReplicatedMergeTree::mutationsFinalizingTask() } } -void StorageReplicatedMergeTree::exportMergeTreePartitionUpdatingTask() +void StorageReplicatedMergeTree::exportTaskUpdatingTask() { - auto component_guard = Coordination::setCurrentComponent("StorageReplicatedMergeTree::exportMergeTreePartitionUpdatingTask"); + auto component_guard = Coordination::setCurrentComponent("StorageReplicatedMergeTree::exportTaskUpdatingTask"); try { - export_merge_tree_partition_manifest_updater->poll(); + export_task_updater->poll(); } catch (const Coordination::Exception & e) { @@ -4791,7 +4802,7 @@ void StorageReplicatedMergeTree::exportMergeTreePartitionUpdatingTask() tryLogCurrentException(log, __PRETTY_FUNCTION__); } - export_merge_tree_partition_updating_task->scheduleAfter(30 * 1000); + export_task_updating_task->scheduleAfter(30 * 1000); } void StorageReplicatedMergeTree::selectPartsToExport() @@ -4810,7 +4821,7 @@ void StorageReplicatedMergeTree::selectPartsToExport() } else { - const auto earliest_backoff_retry = export_merge_tree_partition_task_scheduler->run(); + const auto earliest_backoff_retry = export_task_scheduler->run(); /// If a part is only waiting on its back-off deadline and that deadline is sooner than /// the default tick, wake up earlier so the retry is not delayed by up to a full tick. @@ -4827,15 +4838,15 @@ void StorageReplicatedMergeTree::selectPartsToExport() tryLogCurrentException(log, __PRETTY_FUNCTION__); } - export_merge_tree_partition_select_task->scheduleAfter(reschedule_ms); + export_task_select_task->scheduleAfter(reschedule_ms); } -void StorageReplicatedMergeTree::exportMergeTreePartitionStatusHandlingTask() +void StorageReplicatedMergeTree::exportTaskStatusHandlingTask() { - auto component_guard = Coordination::setCurrentComponent("StorageReplicatedMergeTree::exportMergeTreePartitionStatusHandlingTask"); + auto component_guard = Coordination::setCurrentComponent("StorageReplicatedMergeTree::exportTaskStatusHandlingTask"); try { - export_merge_tree_partition_manifest_updater->handleStatusChanges(); + export_task_updater->handleStatusChanges(); } catch (const Coordination::Exception & e) { @@ -4847,7 +4858,7 @@ void StorageReplicatedMergeTree::exportMergeTreePartitionStatusHandlingTask() else { /// if an exception is thrown, we might have unprocessed status changes, so we need to schedule the task again - export_merge_tree_partition_status_handling_task->scheduleAfter(5000); + export_task_status_handling_task->scheduleAfter(5000); } return; @@ -4855,13 +4866,13 @@ void StorageReplicatedMergeTree::exportMergeTreePartitionStatusHandlingTask() catch (...) { tryLogCurrentException(log, __PRETTY_FUNCTION__); - export_merge_tree_partition_status_handling_task->scheduleAfter(5000); + export_task_status_handling_task->scheduleAfter(5000); } } -std::vector StorageReplicatedMergeTree::getPartitionExportsInfo() const +std::vector StorageReplicatedMergeTree::getExportTasksInfo() const { - return export_merge_tree_partition_manifest_updater->getPartitionExportsInfo(); + return export_task_updater->getExportTasksInfo(); } StorageReplicatedMergeTree::CreateMergeEntryResult StorageReplicatedMergeTree::createLogEntryToMergeParts( @@ -4876,6 +4887,7 @@ StorageReplicatedMergeTree::CreateMergeEntryResult StorageReplicatedMergeTree::c bool cleanup, ReplicatedMergeTreeLogEntryData * out_log_entry, int32_t log_version, + int32_t export_fence_version, MergeType merge_type) { Strings exists_paths; @@ -4932,6 +4944,11 @@ StorageReplicatedMergeTree::CreateMergeEntryResult StorageReplicatedMergeTree::c ops.emplace_back(zkutil::makeSetRequest( fs::path(zookeeper_path) / "log", "", log_version)); /// Check and update version. + /// The merge was checked against the export states at this version: parts exported, being + /// exported and not exported by the `EXPORT` TTL must not be merged together. + if (export_fence_version >= 0 && export_fence) + ops.emplace_back(zkutil::makeCheckRequest(export_fence->getFencePath(), export_fence_version)); + Coordination::Error code = zookeeper->tryMulti(ops, responses); if (code == Coordination::Error::ZOK) @@ -6267,9 +6284,12 @@ void StorageReplicatedMergeTree::partialShutdown() if (getContext()->getServerSettings()[ServerSetting::allow_experimental_export_merge_tree_partition]) { - export_merge_tree_partition_updating_task->deactivate(); - export_merge_tree_partition_select_task->deactivate(); - export_merge_tree_partition_status_handling_task->deactivate(); + export_task_updating_task->deactivate(); + export_task_select_task->deactivate(); + export_task_status_handling_task->deactivate(); + export_ttl_task->deactivate(); + if (auto * scheduler = dynamic_cast(export_ttl_scheduler.get())) + scheduler->releaseSchedulerLock(); } cleanup_thread.stop(); @@ -6346,7 +6366,7 @@ void StorageReplicatedMergeTree::shutdown(bool) std::lock_guard lock(data_parts_exchange_ptr->rwlock); } - export_partition_manifests.set(std::make_unique()); + export_partition_manifests.set(std::make_unique()); { std::lock_guard lock(export_manifests_mutex); @@ -6746,6 +6766,7 @@ bool StorageReplicatedMergeTree::optimize( } auto merge_predicate = queue.getMergePredicate(zookeeper, std::move(partition_ids_hint)); + merge_predicate->loadExportFence(zookeeper); auto parts_collector = std::make_shared(*this, merge_predicate); const auto select_merge = [&]() -> std::expected @@ -6837,6 +6858,7 @@ bool StorageReplicatedMergeTree::optimize( cleanup, &merge_entry, merge_predicate->getVersion(), + merge_predicate->getExportFenceVersion(), select_merge_result.value()->merge_type); if (create_result == CreateMergeEntryResult::MissingPart) @@ -8679,48 +8701,14 @@ void StorageReplicatedMergeTree::exportPartitionToTable(const PartitionCommand & auto src_snapshot = getInMemoryMetadataPtr(query_context, false); auto destination_snapshot = dest_storage->getInMemoryMetadataPtr(query_context, false); - ExportPartitionUtils::verifyExportSchemaCastable( + ExportTaskUtils::verifyExportSchemaCastable( src_snapshot, destination_snapshot, dest_storage->getStorageID(), query_context); zkutil::ZooKeeperPtr zookeeper = getZooKeeperAndAssertNotReadonly(); const String partition_id = getPartitionIDFromQuery(command.partition, query_context); - - const auto exports_path = fs::path(zookeeper_path) / "exports"; - - const auto export_key = ExportPartitionUtils::compositeKey( - partition_id, dest_storage_id.getDatabaseName(), dest_storage_id.getTableName()); - - const auto partition_exports_path = fs::path(exports_path) / export_key; - - /// check if entry already exists - ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRequests); - ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperExists); - if (zookeeper->exists(partition_exports_path)) - { - LOG_INFO(log, "Export with key {} is already exported or it is being exported", export_key); - - if (!query_context->getSettingsRef()[Setting::export_merge_tree_partition_force_export]) - { - throw Exception(ErrorCodes::EXPORT_PARTITION_ALREADY_EXPORTED, "Export with key {} already exported or it is being exported. Set `export_merge_tree_partition_force_export` to overwrite it.", export_key); - } - - LOG_INFO(log, "Overwriting export with key {}", export_key); - - /// Not putting in ops (same transaction) because we can't construct a "tryRemoveRecursive" request. - /// It is possible that the zk being used does not support RemoveRecursive requests. - /// It is ok for this to be non transactional. Worst case scenario an on-going export is going to be killed and a new task won't be scheduled. - ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRequests); - ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRemoveRecursive); - zookeeper->tryRemoveRecursive(partition_exports_path); - } - - Coordination::Requests ops; - - ops.emplace_back(zkutil::makeCreateRequest(partition_exports_path, "", zkutil::CreateMode::Persistent)); DataPartsVector parts; - { auto data_parts_lock = lockParts(); parts = getDataPartsVectorInPartitionForInternalUsage(MergeTreeDataPartState::Active, partition_id, data_parts_lock); @@ -8731,6 +8719,40 @@ void StorageReplicatedMergeTree::exportPartitionToTable(const PartitionCommand & throw Exception(ErrorCodes::BAD_ARGUMENTS, "Partition {} doesn't exist", partition_id); } + /// Every `EXPORT PARTITION` is a new task, even if the partition was exported before. + auto manifest = buildExportTaskManifest(dest_storage_id, dest_storage, src_snapshot, destination_snapshot, parts, partition_id, query_context); + manifest.transaction_id = toString(UUIDHelpers::generateV4()); + manifest.query_id = query_context->getCurrentQueryId(); + manifest.source = ExportTaskSource::query; + + const auto task_path = fs::path(zookeeper_path) / "exports" / manifest.transaction_id; + + Coordination::Requests ops; + ExportTaskUtils::appendCreateExportTaskOps(ops, task_path, manifest); + + ProfileEvents::increment(ProfileEvents::ExportTaskZooKeeperRequests); + ProfileEvents::increment(ProfileEvents::ExportTaskZooKeeperMulti); + Coordination::Responses responses; + const auto code = zookeeper->tryMulti(ops, responses); + if (code != Coordination::Error::ZOK) + throw zkutil::KeeperException::fromPath(code, task_path); + + LOG_INFO(log, "Created export task {} of partition {} to {}, {} part(s)", + manifest.transaction_id, partition_id, dest_storage_id.getNameForLogs(), manifest.parts.size()); + + if (export_task_updating_task) + export_task_updating_task->schedule(); +} + +ExportReplicatedMergeTreeTaskManifest StorageReplicatedMergeTree::buildExportTaskManifest( + const StorageID & dest_storage_id, + const StoragePtr & dest_storage, + const StorageMetadataPtr & src_snapshot, + const StorageMetadataPtr & destination_snapshot, + const DataPartsVector & parts, + const String & partition_id, + ContextPtr query_context) const +{ const bool throw_on_pending_mutations = query_context->getSettingsRef()[Setting::export_merge_tree_part_throw_on_pending_mutations]; const bool throw_on_pending_patch_parts = query_context->getSettingsRef()[Setting::export_merge_tree_part_throw_on_pending_patch_parts]; @@ -8759,7 +8781,7 @@ void StorageReplicatedMergeTree::exportPartitionToTable(const PartitionCommand & part->name); } - if (alter_conversions->hasPatches()) + if (alter_conversions->hasPatches() && throw_on_pending_patch_parts) { throw Exception(ErrorCodes::PENDING_MUTATIONS_NOT_ALLOWED, "Partition {} can not be exported because the part {} has pending patch parts. Either wait for the patch parts to be applied or set `export_merge_tree_part_throw_on_pending_patch_parts` to false", @@ -8770,44 +8792,42 @@ void StorageReplicatedMergeTree::exportPartitionToTable(const PartitionCommand & part_names.push_back(part->name); } - /// TODO arthur somehow check if the list of parts is updated "enough" - - ExportReplicatedMergeTreePartitionManifest manifest; + const auto & settings = query_context->getSettingsRef(); - manifest.transaction_id = toString(UUIDHelpers::generateV4()); - manifest.query_id = query_context->getCurrentQueryId(); - manifest.partition_id = partition_id; - manifest.destination_database = dest_database; - manifest.destination_table = dest_table; + ExportReplicatedMergeTreeTaskManifest manifest; + manifest.destination_database = dest_storage_id.database_name; + manifest.destination_table = dest_storage_id.table_name; + if (const auto uuid = dest_storage->getStorageID().uuid; uuid != UUIDHelpers::Nil) + manifest.destination_uuid = toString(uuid); manifest.source_replica = replica_name; manifest.number_of_parts = part_names.size(); manifest.parts = part_names; manifest.create_time = time(nullptr); - manifest.retry_initial_backoff_seconds = query_context->getSettingsRef()[Setting::export_merge_tree_partition_retry_initial_backoff_seconds]; - manifest.retry_max_backoff_seconds = query_context->getSettingsRef()[Setting::export_merge_tree_partition_retry_max_backoff_seconds]; - manifest.task_timeout_seconds = query_context->getSettingsRef()[Setting::export_merge_tree_partition_task_timeout_seconds]; - manifest.max_threads = query_context->getSettingsRef()[Setting::max_threads]; - manifest.parallel_formatting = query_context->getSettingsRef()[Setting::output_format_parallel_formatting]; - manifest.parquet_parallel_encoding = query_context->getSettingsRef()[Setting::output_format_parquet_parallel_encoding]; - manifest.parquet_compression_method = query_context->getSettingsRef()[Setting::output_format_parquet_compression_method].toString(); - manifest.output_format_compression_level = query_context->getSettingsRef()[Setting::output_format_compression_level]; - manifest.parquet_row_group_size = query_context->getSettingsRef()[Setting::output_format_parquet_row_group_size]; - manifest.parquet_row_group_size_bytes = query_context->getSettingsRef()[Setting::output_format_parquet_row_group_size_bytes]; - manifest.max_bytes_per_file = query_context->getSettingsRef()[Setting::export_merge_tree_part_max_bytes_per_file]; - manifest.max_rows_per_file = query_context->getSettingsRef()[Setting::export_merge_tree_part_max_rows_per_file]; - - manifest.file_already_exists_policy = query_context->getSettingsRef()[Setting::export_merge_tree_part_file_already_exists_policy].value; - manifest.filename_pattern = query_context->getSettingsRef()[Setting::export_merge_tree_part_filename_pattern].value; - manifest.write_full_path_in_iceberg_metadata = query_context->getSettingsRef()[Setting::write_full_path_in_iceberg_metadata]; - manifest.allow_lossy_cast = query_context->getSettingsRef()[Setting::export_merge_tree_part_allow_lossy_cast]; - manifest.iceberg_partition_timezone = query_context->getSettingsRef()[Setting::iceberg_partition_timezone].toString(); - manifest.schema_match_mode = query_context->getSettingsRef()[Setting::export_merge_tree_part_schema_match_mode].value; - manifest.ignore_extra_source_columns = query_context->getSettingsRef()[Setting::export_merge_tree_part_ignore_extra_source_columns].value; + manifest.retry_initial_backoff_seconds = settings[Setting::export_merge_tree_retry_initial_backoff_seconds]; + manifest.retry_max_backoff_seconds = settings[Setting::export_merge_tree_retry_max_backoff_seconds]; + manifest.task_timeout_seconds = settings[Setting::export_merge_tree_task_timeout_seconds]; + manifest.max_threads = settings[Setting::max_threads]; + manifest.parallel_formatting = settings[Setting::output_format_parallel_formatting]; + manifest.parquet_parallel_encoding = settings[Setting::output_format_parquet_parallel_encoding]; + manifest.parquet_compression_method = settings[Setting::output_format_parquet_compression_method].toString(); + manifest.output_format_compression_level = settings[Setting::output_format_compression_level]; + manifest.parquet_row_group_size = settings[Setting::output_format_parquet_row_group_size]; + manifest.parquet_row_group_size_bytes = settings[Setting::output_format_parquet_row_group_size_bytes]; + manifest.max_bytes_per_file = settings[Setting::export_merge_tree_part_max_bytes_per_file]; + manifest.max_rows_per_file = settings[Setting::export_merge_tree_part_max_rows_per_file]; + + manifest.file_already_exists_policy = settings[Setting::export_merge_tree_part_file_already_exists_policy].value; + manifest.filename_pattern = settings[Setting::export_merge_tree_part_filename_pattern].value; + manifest.write_full_path_in_iceberg_metadata = settings[Setting::write_full_path_in_iceberg_metadata]; + manifest.allow_lossy_cast = settings[Setting::export_merge_tree_part_allow_lossy_cast]; + manifest.iceberg_partition_timezone = settings[Setting::iceberg_partition_timezone].toString(); + manifest.schema_match_mode = settings[Setting::export_merge_tree_part_schema_match_mode].value; + manifest.ignore_extra_source_columns = settings[Setting::export_merge_tree_part_ignore_extra_source_columns].value; if (dest_storage->isDataLake()) { #if USE_AVRO - manifest.iceberg_metadata_json = ExportPartitionUtils::verifyAndExtractDestinationIcebergMetadataJson( + manifest.iceberg_metadata_json = ExportTaskUtils::verifyAndExtractDestinationIcebergMetadataJson( src_snapshot, destination_snapshot, dest_storage, @@ -8815,83 +8835,86 @@ void StorageReplicatedMergeTree::exportPartitionToTable(const PartitionCommand & partition_id, query_context); - manifest.max_bytes_per_file = query_context->getSettingsRef()[Setting::iceberg_insert_max_bytes_in_data_file]; - manifest.max_rows_per_file = query_context->getSettingsRef()[Setting::iceberg_insert_max_rows_in_data_file]; + manifest.max_bytes_per_file = settings[Setting::iceberg_insert_max_bytes_in_data_file]; + manifest.max_rows_per_file = settings[Setting::iceberg_insert_max_rows_in_data_file]; #else throw Exception(ErrorCodes::NOT_IMPLEMENTED, "Data lake export requires Avro support"); #endif } else { - ExportPartitionUtils::verifyPlainPartitionCompatibility( + ExportTaskUtils::verifyPlainPartitionCompatibility( src_snapshot, destination_snapshot, parts, partition_id, query_context); } - ops.emplace_back(zkutil::makeCreateRequest( - fs::path(partition_exports_path) / "metadata.json", - manifest.toJsonString(), - zkutil::CreateMode::Persistent)); + return manifest; +} - /// Container for per-replica last_exception leaves; children are created lazily by the - /// first writer per replica (see ExportPartitionUtils::appendExceptionOps). - ops.emplace_back(zkutil::makeCreateRequest( - fs::path(partition_exports_path) / "last_exception", - "", - zkutil::CreateMode::Persistent)); +void StorageReplicatedMergeTree::checkAllReplicasSupportExportTTL(const zkutil::ZooKeeperPtr & zookeeper) const +{ + const auto replicas = zookeeper->getChildren(fs::path(zookeeper_path) / "replicas"); - ops.emplace_back(zkutil::makeCreateRequest( - fs::path(partition_exports_path) / "processing", - "", - zkutil::CreateMode::Persistent)); + Strings paths; + paths.reserve(replicas.size()); + for (const auto & replica : replicas) + paths.push_back(fs::path(zookeeper_path) / "replicas" / replica / "export_features"); - for (const auto & part : part_names) - { - ExportReplicatedMergeTreePartitionProcessingPartEntry entry; - entry.status = ExportReplicatedMergeTreePartitionProcessingPartEntry::Status::PENDING; - entry.part_name = part; + auto responses = zookeeper->exists(paths); - ops.emplace_back(zkutil::makeCreateRequest( - fs::path(partition_exports_path) / "processing" / part, - entry.toJsonString(), - zkutil::CreateMode::Persistent)); + Strings unsupported; + for (size_t i = 0; i < replicas.size(); ++i) + { + if (responses[i].error == Coordination::Error::ZNONODE) + unsupported.push_back(replicas[i]); + else if (responses[i].error != Coordination::Error::ZOK) + throw zkutil::KeeperException::fromPath(responses[i].error, paths[i]); } - - ops.emplace_back(zkutil::makeCreateRequest( - fs::path(partition_exports_path) / "processed", - "", - zkutil::CreateMode::Persistent)); - ops.emplace_back(zkutil::makeCreateRequest( - fs::path(partition_exports_path) / "locks", - "", - zkutil::CreateMode::Persistent)); - - /// status: IN_PROGRESS, COMPLETED, FAILED - ops.emplace_back(zkutil::makeCreateRequest( - fs::path(partition_exports_path) / "status", - "PENDING", - zkutil::CreateMode::Persistent)); + if (!unsupported.empty()) + throw Exception(ErrorCodes::SUPPORT_IS_DISABLED, + "Cannot export by TTL: replica(s) {} would merge exported parts with parts that are not exported. " + "Every replica must run a version that supports it with the server setting `allow_experimental_export_merge_tree_partition` " + "enabled; drop the replicas that are lost", + fmt::join(unsupported, ", ")); +} - ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRequests); - ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperMulti); - Coordination::Responses responses; - Coordination::Error code = zookeeper->tryMulti(ops, responses); +void StorageReplicatedMergeTree::advertiseExportFeatures(const zkutil::ZooKeeperPtr & zookeeper) const +{ + const String path = fs::path(replica_path) / "export_features"; - if (code != Coordination::Error::ZOK) + if (export_fence) { - if (code == Coordination::Error::ZNODEEXISTS - && zkutil::getFailedOpIndex(code, responses) == 0) - { - /// Lost the race on the root export node. Current code already - /// validated (exists / expired / force) — so this is *always* a race. - throw Exception(ErrorCodes::EXPORT_PARTITION_ALREADY_EXPORTED, - "Export with key {} was created concurrently by another replica. Retry if needed", - export_key); - } - throw zkutil::KeeperException::fromPath(code, partition_exports_path); + const auto code = zookeeper->tryCreate(path, "ttl", zkutil::CreateMode::Persistent); + if (code != Coordination::Error::ZOK && code != Coordination::Error::ZNODEEXISTS) + throw zkutil::KeeperException::fromPath(code, path); + return; } + + const auto code = zookeeper->tryRemove(path); + if (code != Coordination::Error::ZOK && code != Coordination::Error::ZNONODE) + throw zkutil::KeeperException::fromPath(code, path); +} + +void StorageReplicatedMergeTree::wakeUpExportTTL() +{ + if (export_ttl_task) + export_ttl_task->schedule(); } +void StorageReplicatedMergeTree::exportTTLTask() +{ + auto component_guard = Coordination::setCurrentComponent("StorageReplicatedMergeTree::exportTTLTask"); + UInt64 next_run_ms = 10000; + try + { + next_run_ms = export_ttl_scheduler->run(); + } + catch (...) + { + tryLogCurrentException(log, __PRETTY_FUNCTION__); + } + export_ttl_task->scheduleAfter(next_run_ms); +} void StorageReplicatedMergeTree::forgetPartition(const ASTPtr & partition, ContextPtr query_context) { @@ -8902,11 +8925,39 @@ void StorageReplicatedMergeTree::forgetPartition(const ASTPtr & partition, Conte String block_numbers_path = fs::path(zookeeper_path) / "block_numbers"; String partition_path = fs::path(block_numbers_path) / partition_id; - auto error_code = zookeeper->tryRemove(partition_path); + Coordination::Requests ops; + ops.emplace_back(zkutil::makeRemoveRequest(partition_path, -1)); + + /// Block numbers of the partition start over, so a new part could reuse the block range of a part + /// exported by the `EXPORT` TTL. The partition's index entries go with them, in the same transaction. + if (export_fence) + { + for (const auto & destination_key : export_fence->listDestinations(zookeeper)) + { + const auto versioned = export_fence->readIndexEntry(zookeeper, destination_key, partition_id); + if (versioned.version < 0) + continue; + + if (!versioned.entry.claimed.empty()) + throw Exception(ErrorCodes::CANNOT_FORGET_PARTITION, + "Partition {} is being exported by the EXPORT TTL (task {}), retry after it finishes", + partition_id, versioned.entry.claimed.begin()->first); + + ops.emplace_back(zkutil::makeRemoveRequest(export_fence->getIndexEntryPath(destination_key, partition_id), versioned.version)); + } + + if (ops.size() > 1) + ops.emplace_back(zkutil::makeSetRequest(export_fence->getFencePath(), "", -1)); + } + + Coordination::Responses responses; + auto error_code = zookeeper->tryMulti(ops, responses); if (error_code == Coordination::Error::ZOK) LOG_INFO(log, "Forget partition {}", partition_id); - else if (error_code == Coordination::Error::ZNONODE) + else if (error_code == Coordination::Error::ZNONODE && zkutil::getFailedOpIndex(error_code, responses) == 0) throw Exception(ErrorCodes::CANNOT_FORGET_PARTITION, "Partition {} is unknown", partition_id); + else if (error_code == Coordination::Error::ZNONODE || error_code == Coordination::Error::ZBADVERSION) + throw Exception(ErrorCodes::CANNOT_FORGET_PARTITION, "The EXPORT TTL state of partition {} changed concurrently, retry", partition_id); else throw zkutil::KeeperException::fromPath(error_code, partition_path); @@ -10346,10 +10397,10 @@ CancellationCode StorageReplicatedMergeTree::killPartMoveToShard(const UUID & ta return part_moves_between_shards_orchestrator.killPartMoveToShard(task_uuid); } -CancellationCode StorageReplicatedMergeTree::killExportPartition(const String & transaction_id) +CancellationCode StorageReplicatedMergeTree::killExportTask(const String & transaction_id) { - /// Called from a query thread (KILL EXPORT PARTITION via InterpreterKillQueryQuery), which does not have a component set. - auto component_guard = Coordination::setCurrentComponent("StorageReplicatedMergeTree::killExportPartition"); + /// Called from a query thread (KILL EXPORT via InterpreterKillQueryQuery), which does not have a component set. + auto component_guard = Coordination::setCurrentComponent("StorageReplicatedMergeTree::killExportTask"); /// KILL is serialized against the commit phase via commit_lock (see below), so a kill that /// succeeds cannot be overwritten by a concurrent commit. @@ -10375,7 +10426,7 @@ CancellationCode StorageReplicatedMergeTree::killExportPartition(const String & return CancellationCode::CancelCannotBeSent; } - const auto status_from_zk = magic_enum::enum_cast(status_from_zk_string); + const auto status_from_zk = magic_enum::enum_cast(status_from_zk_string); if (!status_from_zk) { @@ -10383,13 +10434,13 @@ CancellationCode StorageReplicatedMergeTree::killExportPartition(const String & return CancellationCode::CancelCannotBeSent; } - if (status_from_zk.value() != ExportReplicatedMergeTreePartitionTaskEntry::Status::PENDING) + if (status_from_zk.value() != ExportReplicatedMergeTreeTaskEntry::Status::PENDING) { LOG_INFO(log, "Export partition task is {}, can not cancel it", String(magic_enum::enum_name(status_from_zk.value()))); return CancellationCode::CancelCannotBeSent; } - if (zk->trySet(status_path, String(magic_enum::enum_name(ExportReplicatedMergeTreePartitionTaskEntry::Status::KILLED)), stat.version) != Coordination::Error::ZOK) + if (zk->trySet(status_path, String(magic_enum::enum_name(ExportReplicatedMergeTreeTaskEntry::Status::KILLED)), stat.version) != Coordination::Error::ZOK) { LOG_INFO(log, "Status has been updated while trying to kill the export partition task, can not cancel it"); return CancellationCode::CancelCannotBeSent; @@ -10402,56 +10453,23 @@ CancellationCode StorageReplicatedMergeTree::killExportPartition(const String & /// Read the published snapshot (shared_ptr copy, no lock, no ZooKeeper). The KILLED status set /// below propagates back into the mirror via the status watch -> handleStatusChanges. - bool local_entry_found = false; - bool local_entry_pending = false; - std::string local_composite_key; - if (const auto model = export_partition_manifests.get()) { - const auto & by_transaction_id = model->get(); + const auto & by_transaction_id = model->get(); const auto entry = by_transaction_id.find(transaction_id); - if (entry != by_transaction_id.end()) + if (entry != by_transaction_id.end() && entry->status != ExportReplicatedMergeTreeTaskEntry::Status::PENDING) { - local_entry_found = true; - local_entry_pending = entry->status == ExportReplicatedMergeTreePartitionTaskEntry::Status::PENDING; - local_composite_key = entry->getCompositeKey(); - } - } - - /// if we have the entry locally, no need to list from zk. we can save some requests. - if (local_entry_found) - { - LOG_INFO(log, "Export partition task found locally, trying to cancel it"); - /// found locally, no need to get children on zk - if (!local_entry_pending) - { - LOG_INFO(log, "Export partition task is not pending, can not cancel it"); + LOG_INFO(log, "Export task {} is not pending, can not cancel it", transaction_id); return CancellationCode::CancelCannotBeSent; } - - return try_set_status_to_killed(zk, fs::path(zookeeper_path) / "exports" / local_composite_key / "status"); } - else - { - LOG_INFO(log, "Export partition task not found locally, trying to find it on zk"); - /// for some reason, we don't have the entry locally. ls on zk to find the entry - const auto exports_path = fs::path(zookeeper_path) / "exports"; - - const auto export_keys = zk->getChildren(exports_path); - String export_key_to_be_cancelled; - for (const auto & export_key : export_keys) - { - std::string metadata_json; - if (!zk->tryGet(fs::path(exports_path) / export_key / "metadata.json", metadata_json)) - continue; - const auto manifest = ExportReplicatedMergeTreePartitionManifest::fromJsonString(metadata_json); - if (manifest.transaction_id == transaction_id) - { - LOG_INFO(log, "Export partition task found on zk, trying to cancel it"); - return try_set_status_to_killed(zk, fs::path(exports_path) / export_key / "status"); - } - } + /// Tasks are keyed by their transaction id, so a task the mirror does not know yet is found as well. + if (!transaction_id.empty() && transaction_id.find('/') == String::npos) + { + const auto status_path = fs::path(zookeeper_path) / "exports" / transaction_id / "status"; + if (zk->exists(status_path)) + return try_set_status_to_killed(zk, status_path); } LOG_INFO(log, "Export partition task not found, can not cancel it"); diff --git a/src/Storages/StorageReplicatedMergeTree.h b/src/Storages/StorageReplicatedMergeTree.h index 64e1b122e1c8..f1fb6b08c272 100644 --- a/src/Storages/StorageReplicatedMergeTree.h +++ b/src/Storages/StorageReplicatedMergeTree.h @@ -11,9 +11,10 @@ #include #include #include -#include -#include -#include +#include +#include +#include +#include #include #include #include @@ -376,7 +377,11 @@ class StorageReplicatedMergeTree final : public MergeTreeData using ShutdownDeadline = std::chrono::time_point; void waitForUniquePartsToBeFetchedByOtherReplicas(ShutdownDeadline shutdown_deadline); - std::vector getPartitionExportsInfo() const override; + std::vector getExportTasksInfo() const override; + + /// nullptr if partition export is disabled. + ReplicatedExportTTLIndexPtr getExportFence() const { return export_fence; } + ExportFencePtr getLatestExportFence() const override { return export_fence ? export_fence->getLatest() : nullptr; } private: std::atomic_bool are_restoring_replica {false}; @@ -402,8 +407,9 @@ class StorageReplicatedMergeTree final : public MergeTreeData friend class MergeFromLogEntryTask; friend class MutateFromLogEntryTask; friend class ReplicatedMergeMutateTaskBase; - friend class ExportPartitionManifestUpdatingTask; - friend class ExportPartitionTaskScheduler; + friend class ReplicatedExportTaskUpdater; + friend class ReplicatedExportTaskScheduler; + friend class ReplicatedExportTTLScheduler; using MergeStrategyPicker = ReplicatedMergeTreeMergeStrategyPicker; using LogEntry = ReplicatedMergeTreeLogEntry; @@ -516,21 +522,30 @@ class StorageReplicatedMergeTree final : public MergeTreeData /// A task that marks finished mutations as done. BackgroundSchedulePoolTaskHolder mutations_finalizing_task; - BackgroundSchedulePoolTaskHolder export_merge_tree_partition_updating_task; + BackgroundSchedulePoolTaskHolder export_task_updating_task; /// mostly handle kill operations - BackgroundSchedulePoolTaskHolder export_merge_tree_partition_status_handling_task; - std::shared_ptr export_merge_tree_partition_manifest_updater; + BackgroundSchedulePoolTaskHolder export_task_status_handling_task; + std::shared_ptr export_task_updater; - std::shared_ptr export_merge_tree_partition_task_scheduler; + std::shared_ptr export_task_scheduler; - Coordination::WatchCallbackPtr export_merge_tree_partition_watch_callback; + Coordination::WatchCallbackPtr export_task_watch_callback; - BackgroundSchedulePoolTaskHolder export_merge_tree_partition_select_task; + BackgroundSchedulePoolTaskHolder export_task_select_task; /// Immutable snapshot republished after each writer batch (part_references stripped). Readers /// (system table, scheduler, KILL) get() a consistent version with no lock and no ZooKeeper. - MultiVersion export_partition_manifests; + MultiVersion export_partition_manifests; + + /// The export index of the `EXPORT` TTL and the merge fence built from it. Only created when + /// partition export is enabled; a replica without it must not assign merges of parts exported by + /// the TTL, which is why it does not advertise `export_features`. + ReplicatedExportTTLIndexPtr export_fence; + + /// Runs `export_ttl_scheduler`. + BackgroundSchedulePoolTaskHolder export_ttl_task; + /// A thread that removes old parts, log entries, and blocks. ReplicatedMergeTreeCleanupThread cleanup_thread; @@ -766,10 +781,10 @@ class StorageReplicatedMergeTree final : public MergeTreeData void selectPartsToExport(); /// update in-memory list of partition exports - void exportMergeTreePartitionUpdatingTask(); + void exportTaskUpdatingTask(); /// handle status changes for export partition tasks - void exportMergeTreePartitionStatusHandlingTask(); + void exportTaskStatusHandlingTask(); /** Write the selected parts to merge into the log, * Call when merge_selecting_mutex is locked. @@ -789,6 +804,7 @@ class StorageReplicatedMergeTree final : public MergeTreeData bool cleanup, ReplicatedMergeTreeLogEntryData * out_log_entry, int32_t log_version, + int32_t export_fence_version, MergeType merge_type); CreateMergeEntryResult createLogEntryToMutatePart( @@ -959,7 +975,7 @@ class StorageReplicatedMergeTree final : public MergeTreeData void movePartitionToTable(const StoragePtr & dest_table, const ASTPtr & partition, ContextPtr query_context) override; void movePartitionToShard(const ASTPtr & partition, bool move_part, const String & to, ContextPtr query_context) override; CancellationCode killPartMoveToShard(const UUID & task_uuid) override; - CancellationCode killExportPartition(const String & transaction_id) override; + CancellationCode killExportTask(const String & transaction_id) override; void fetchPartition( const ASTPtr & partition, const StorageMetadataPtr & metadata_snapshot, @@ -970,6 +986,28 @@ class StorageReplicatedMergeTree final : public MergeTreeData void exportPartitionToTable(const PartitionCommand &, ContextPtr) override; + /// Builds the descriptor of an export task of `parts` from the settings of `query_context`, and + /// validates the destination for them. The caller sets the transaction id and the source. + ExportReplicatedMergeTreeTaskManifest buildExportTaskManifest( + const StorageID & dest_storage_id, + const StoragePtr & dest_storage, + const StorageMetadataPtr & src_snapshot, + const StorageMetadataPtr & destination_snapshot, + const DataPartsVector & parts, + const String & partition_id, + ContextPtr query_context) const; + + /// Throws unless every replica enforces the export states of parts when assigning merges. + void checkAllReplicasSupportExportTTL(const zkutil::ZooKeeperPtr & zookeeper) const; + + /// Creates (or removes, if partition export is disabled) `/export_features`. + void advertiseExportFeatures(const zkutil::ZooKeeperPtr & zookeeper) const; + + void exportTTLTask(); + + /// E.g. when a task of the `EXPORT` TTL finished, so the next group does not wait for the next check. + void wakeUpExportTTL(); + /// NOTE: there are no guarantees for concurrent merges. Dropping part can /// be concurrently merged into some covering part and dropPart will do /// nothing. There are some fundamental problems with it. But this is OK diff --git a/src/Storages/System/StorageSystemPartitionExports.cpp b/src/Storages/System/StorageSystemDistributedExports.cpp similarity index 89% rename from src/Storages/System/StorageSystemPartitionExports.cpp rename to src/Storages/System/StorageSystemDistributedExports.cpp index 92ff0d1bd563..b235054ca2a6 100644 --- a/src/Storages/System/StorageSystemPartitionExports.cpp +++ b/src/Storages/System/StorageSystemDistributedExports.cpp @@ -1,4 +1,4 @@ -#include +#include #include #include #include @@ -15,7 +15,7 @@ namespace DB { -ColumnsDescription StorageSystemPartitionExports::getColumnsDescription() +ColumnsDescription StorageSystemDistributedExports::getColumnsDescription() { auto last_exception_tuple = std::make_shared( DataTypes{ @@ -67,10 +67,14 @@ ColumnsDescription StorageSystemPartitionExports::getColumnsDescription() "For plain object storage destinations: path of the per-transaction commit marker file written by the destination. Empty for Iceberg destinations and for tasks that have not committed yet."}, {"local_backoff_per_part", std::make_shared(backoff_tuple), "Per-part retry back-off local to this node: parts currently waiting before their next attempt, with attempt count and the next eligible time. Not shared across replicas; empty if no part is backing off."}, + {"source", std::make_shared(), + "What created the task: `query` for `ALTER TABLE ... EXPORT PARTITION`, `ttl` for the table's `TTL ... EXPORT TO TABLE` expression."}, + {"retry_of", std::make_shared(std::make_shared()), + "For a TTL export task: transaction ids of earlier tasks that failed to export some of this task's parts. Its commit checks whether any of them landed at the destination after all. Empty otherwise."}, }; } -void StorageSystemPartitionExports::fillData(MutableColumns & res_columns, ContextPtr context, const ActionsDAG::Node * predicate, std::vector) const +void StorageSystemDistributedExports::fillData(MutableColumns & res_columns, ContextPtr context, const ActionsDAG::Node * predicate, std::vector) const { const auto access = context->getAccess(); const bool check_access_for_databases = !access->isGranted(AccessType::SHOW_TABLES); @@ -137,14 +141,14 @@ void StorageSystemPartitionExports::fillData(MutableColumns & res_columns, Conte const auto database = (*col_database)[i_storage].safeGet(); const auto table = (*col_table)[i_storage].safeGet(); - std::vector partition_exports_info; + std::vector export_tasks_info; { const IStorage * storage = merge_tree_tables[database][table].get(); if (const auto * merge_tree = dynamic_cast(storage)) - partition_exports_info = merge_tree->getPartitionExportsInfo(); + export_tasks_info = merge_tree->getExportTasksInfo(); } - for (const PartitionExportInfo & info : partition_exports_info) + for (const ExportTaskInfo & info : export_tasks_info) { std::size_t i = 0; res_columns[i++]->insert(database); @@ -196,6 +200,14 @@ void StorageSystemPartitionExports::fillData(MutableColumns & res_columns, Conte for (const auto & b : info.backoff_per_part) backoff_array.push_back(Tuple{b.part, b.attempts, b.next_retry_time}); res_columns[i++]->insert(backoff_array); + + res_columns[i++]->insert(info.source); + + Array retry_of_array; + retry_of_array.reserve(info.retry_of.size()); + for (const auto & transaction : info.retry_of) + retry_of_array.push_back(transaction); + res_columns[i++]->insert(retry_of_array); } } } diff --git a/src/Storages/System/StorageSystemDistributedExports.h b/src/Storages/System/StorageSystemDistributedExports.h new file mode 100644 index 000000000000..4f0af6e2292a --- /dev/null +++ b/src/Storages/System/StorageSystemDistributedExports.h @@ -0,0 +1,28 @@ +#pragma once + +#include + +namespace DB +{ + +class Context; + +/// system.distributed_exports: progress of the export tasks of every MergeTree-family table, created by +/// `EXPORT PARTITION` or by the `EXPORT` TTL, both of plain `MergeTree` (backed by on-disk task +/// descriptors) and `Replicated*MergeTree` (backed by the ZooKeeper manifest mirror). Both are read +/// from memory, so querying it touches neither disk nor ZooKeeper. Each export task is represented +/// by a single row. +class StorageSystemDistributedExports final : public IStorageSystemOneBlock +{ +public: + std::string getName() const override { return "SystemDistributedExports"; } + + static ColumnsDescription getColumnsDescription(); + +protected: + using IStorageSystemOneBlock::IStorageSystemOneBlock; + + void fillData(MutableColumns & res_columns, ContextPtr context, const ActionsDAG::Node *, std::vector) const override; +}; + +} diff --git a/src/Storages/System/StorageSystemPartitionExports.h b/src/Storages/System/StorageSystemPartitionExports.h deleted file mode 100644 index d7f8e5242618..000000000000 --- a/src/Storages/System/StorageSystemPartitionExports.h +++ /dev/null @@ -1,30 +0,0 @@ -#pragma once - -#include - -namespace DB -{ - -class Context; - -/// system.partition_exports: progress of EXPORT PARTITION tasks of every MergeTree-family table, -/// both plain `MergeTree` (backed by on-disk task descriptors) and `Replicated*MergeTree` (backed -/// by the ZooKeeper manifest mirror). Both are read from memory, so querying it touches neither -/// disk nor ZooKeeper. Each export task is represented by a single row. -/// -/// Also attached as `system.replicated_partition_exports`, a backwards-compatible alias from when -/// the two engines had separate tables. -class StorageSystemPartitionExports final : public IStorageSystemOneBlock -{ -public: - std::string getName() const override { return "SystemPartitionExports"; } - - static ColumnsDescription getColumnsDescription(); - -protected: - using IStorageSystemOneBlock::IStorageSystemOneBlock; - - void fillData(MutableColumns & res_columns, ContextPtr context, const ActionsDAG::Node *, std::vector) const override; -}; - -} diff --git a/src/Storages/System/StorageSystemTTLExports.cpp b/src/Storages/System/StorageSystemTTLExports.cpp new file mode 100644 index 000000000000..fef095e2e4f5 --- /dev/null +++ b/src/Storages/System/StorageSystemTTLExports.cpp @@ -0,0 +1,84 @@ +#include + +#include +#include +#include +#include +#include +#include +#include +#include + +namespace DB +{ + +ColumnsDescription StorageSystemTTLExports::getColumnsDescription() +{ + return ColumnsDescription + { + {"database", std::make_shared(), "Name of the source database."}, + {"table", std::make_shared(), "Name of the source table."}, + {"partition_id", std::make_shared(), "ID of the partition."}, + {"destination_database", std::make_shared(), "Name of the destination database of the EXPORT TTL."}, + {"destination_table", std::make_shared(), "Name of the destination table of the EXPORT TTL."}, + {"exported_parts", std::make_shared(), "Number of active parts whose rows were committed to the destination."}, + {"claimed_parts", std::make_shared(), "Number of active parts being exported, or waiting to be exported again after a failed task."}, + {"eligible_parts", std::make_shared(), "Number of active parts that are due for export and not exported yet."}, + {"eligible_bytes", std::make_shared(), "Size on disk of the eligible parts."}, + {"parts_held_by_delete_gate", std::make_shared(), + "Number of parts whose delete or column TTL is due, but that are kept from merges until they are exported."}, + {"first_eligible_time", std::make_shared(), "When the scheduler first saw an eligible part of the partition that is not exported yet, zero if there is none."}, + {"next_group_time", std::make_shared(), "When the eligible parts are exported at the latest, zero if nothing waits."}, + {"current_transaction_id", std::make_shared(), "Transaction id of the task exporting the partition now, see `system.distributed_exports`. Empty if there is none."}, + {"last_error", std::make_shared(), "Error of the last attempt to export the partition, empty if it succeeded."}, + {"scheduler_replica", std::make_shared(), + "The replica of a Replicated*MergeTree table that schedules its TTL exports; the other replicas show the state it stored. Empty for a plain MergeTree."}, + }; +} + +void StorageSystemTTLExports::fillData(MutableColumns & res_columns, ContextPtr context, const ActionsDAG::Node *, std::vector) const +{ + const auto access = context->getAccess(); + const bool check_access_for_databases = !access->isGranted(AccessType::SHOW_TABLES); + + for (const auto & [database_name, database] : DatabaseCatalog::instance().getDatabases(GetDatabasesOptions{.with_datalake_catalogs = false, .with_remote_databases = false})) + { + if (database->isExternal()) + continue; + + const bool check_access_for_tables = check_access_for_databases && !access->isGranted(AccessType::SHOW_TABLES, database_name); + + for (auto iterator = database->getTablesIterator(context); iterator->isValid(); iterator->next()) + { + const auto & table = iterator->table(); + const auto * merge_tree = dynamic_cast(table.get()); + if (!merge_tree) + continue; + + if (check_access_for_tables && !access->isGranted(AccessType::SHOW_TABLES, database_name, iterator->name())) + continue; + + for (const auto & info : merge_tree->getExportTTLInfo()) + { + size_t i = 0; + res_columns[i++]->insert(database_name); + res_columns[i++]->insert(iterator->name()); + res_columns[i++]->insert(info.partition_id); + res_columns[i++]->insert(info.destination_database); + res_columns[i++]->insert(info.destination_table); + res_columns[i++]->insert(info.exported_parts); + res_columns[i++]->insert(info.claimed_parts); + res_columns[i++]->insert(info.eligible_parts); + res_columns[i++]->insert(info.eligible_bytes); + res_columns[i++]->insert(info.parts_held_by_delete_gate); + res_columns[i++]->insert(static_cast(info.first_eligible_time)); + res_columns[i++]->insert(static_cast(info.next_group_time)); + res_columns[i++]->insert(info.current_transaction_id); + res_columns[i++]->insert(info.last_error); + res_columns[i++]->insert(info.scheduler_replica); + } + } + } +} + +} diff --git a/src/Storages/System/StorageSystemTTLExports.h b/src/Storages/System/StorageSystemTTLExports.h new file mode 100644 index 000000000000..6468ac1e4dd7 --- /dev/null +++ b/src/Storages/System/StorageSystemTTLExports.h @@ -0,0 +1,27 @@ +#pragma once + +#include + +namespace DB +{ + +class Context; + +/// system.ttl_exports: state of the `TTL ... EXPORT TO TABLE` expression of every MergeTree-family +/// table, one row per partition, as of the last tick of the table's TTL export scheduler. Every +/// replica of a `Replicated*MergeTree` table has the same rows. Read from memory, so querying it +/// touches neither disk nor ZooKeeper. +class StorageSystemTTLExports final : public IStorageSystemOneBlock +{ +public: + std::string getName() const override { return "SystemTTLExports"; } + + static ColumnsDescription getColumnsDescription(); + +protected: + using IStorageSystemOneBlock::IStorageSystemOneBlock; + + void fillData(MutableColumns & res_columns, ContextPtr context, const ActionsDAG::Node *, std::vector) const override; +}; + +} diff --git a/src/Storages/System/attachSystemTables.cpp b/src/Storages/System/attachSystemTables.cpp index f75b7dcae4bb..81d24e558481 100644 --- a/src/Storages/System/attachSystemTables.cpp +++ b/src/Storages/System/attachSystemTables.cpp @@ -1,5 +1,6 @@ #include -#include +#include +#include #include "config.h" #include @@ -269,9 +270,8 @@ void attachSystemTablesServer(ContextPtr context, IDatabase & system_database, b attach(context, system_database, "exports", "Contains a list of exports currently executing exports of MergeTree tables and their progress. Each export operation is represented by a single row."); if (context->getServerSettings()[ServerSetting::allow_experimental_export_merge_tree_partition]) { - attach(context, system_database, "partition_exports", "Contains a list of partition exports of MergeTree tables, both plain and replicated, and their progress. Each export operation is represented by a single row."); - /// Backwards-compatible alias from when plain and replicated exports had separate tables. - attach(context, system_database, "replicated_partition_exports", "Alias of system.partition_exports, kept for backwards compatibility. Returns the same rows, including exports of plain (non-replicated) MergeTree tables."); + attach(context, system_database, "distributed_exports", "Contains the export tasks of MergeTree tables, both plain and replicated, created by `EXPORT PARTITION` or by a `TTL ... EXPORT` expression, and their progress. Each task is represented by a single row."); + attach(context, system_database, "ttl_exports", "Contains the state of the `TTL ... EXPORT TO TABLE` expression of MergeTree tables, one row per partition: exported, claimed and eligible parts, when the next group is exported, and the last error."); } attach(context, system_database, "mutations", "Contains a list of mutations and their progress. Each mutation command is represented by a single row."); attachNoDescription(context, system_database, "replicas", "Contains information and status of all table replicas on current server. Each replica is represented by a single row."); diff --git a/src/Storages/TTLDescription.cpp b/src/Storages/TTLDescription.cpp index acc5da0bf5de..95b4cc19b03c 100644 --- a/src/Storages/TTLDescription.cpp +++ b/src/Storages/TTLDescription.cpp @@ -1135,6 +1135,7 @@ TTLDescription::TTLDescription(const TTLDescription & other) , aggregate_descriptions(other.aggregate_descriptions) , destination_type(other.destination_type) , destination_name(other.destination_name) + , destination_database(other.destination_database) , if_exists(other.if_exists) , recompression_codec(other.recompression_codec) { @@ -1166,6 +1167,7 @@ TTLDescription & TTLDescription::operator=(const TTLDescription & other) aggregate_descriptions = other.aggregate_descriptions; destination_type = other.destination_type; destination_name = other.destination_name; + destination_database = other.destination_database; if_exists = other.if_exists; if (other.recompression_codec) @@ -1296,6 +1298,7 @@ TTLDescription TTLDescription::getTTLFromAST( result.mode = ttl_element->mode; result.destination_type = ttl_element->destination_type; result.destination_name = ttl_element->destination_name; + result.destination_database = ttl_element->destination_database; result.if_exists = ttl_element->if_exists; if (ttl_element->mode == TTLMode::DELETE) @@ -1382,7 +1385,10 @@ TTLDescription TTLDescription::getTTLFromAST( } } - checkTTLExpression(expression, result.result_column, is_attach || context->getSettingsRef()[Setting::allow_suspicious_ttl_expressions]); + /// An export TTL decides when rows are copied elsewhere once and for all, so it must be a + /// deterministic function of the rows even when suspicious TTL expressions are allowed. + const bool allow_suspicious = result.mode != TTLMode::EXPORT && context->getSettingsRef()[Setting::allow_suspicious_ttl_expressions]; + checkTTLExpression(expression, result.result_column, is_attach || allow_suspicious); if (where_expression && !is_attach && !context->getSettingsRef()[Setting::allow_suspicious_ttl_expressions]) checkTTLExpressionForAggregateFunctions(where_expression, /*expression_kind=*/ "WHERE "); @@ -1398,6 +1404,7 @@ TTLTableDescription::TTLTableDescription(const TTLTableDescription & other) , move_ttl(other.move_ttl) , recompression_ttl(other.recompression_ttl) , group_by_ttl(other.group_by_ttl) + , export_ttl(other.export_ttl) { } @@ -1416,6 +1423,7 @@ TTLTableDescription & TTLTableDescription::operator=(const TTLTableDescription & move_ttl = other.move_ttl; recompression_ttl = other.recompression_ttl; group_by_ttl = other.group_by_ttl; + export_ttl = other.export_ttl; return *this; } @@ -1460,6 +1468,12 @@ TTLTableDescription TTLTableDescription::getTTLForTableFromAST( { result.group_by_ttl.emplace_back(std::move(ttl)); } + else if (ttl.mode == TTLMode::EXPORT) + { + if (!result.export_ttl.empty()) + throw Exception(ErrorCodes::BAD_TTL_EXPRESSION, "More than one EXPORT TTL expression is not allowed"); + result.export_ttl.emplace_back(std::move(ttl)); + } else { result.move_ttl.emplace_back(std::move(ttl)); diff --git a/src/Storages/TTLDescription.h b/src/Storages/TTLDescription.h index a8a462720c97..f775e663c42e 100644 --- a/src/Storages/TTLDescription.h +++ b/src/Storages/TTLDescription.h @@ -88,9 +88,13 @@ struct TTLDescription /// For example DISK or VOLUME DataDestinationType destination_type{}; - /// Name of destination disk or volume + /// Name of destination disk or volume, or of the table for `EXPORT TO TABLE`. String destination_name; + /// `EXPORT TO TABLE` only: database of the destination table. Empty means the database of the + /// table the TTL belongs to. + String destination_database; + /// If true, do nothing if DISK or VOLUME doesn't exist . /// Only valid for table MOVE TTLs. bool if_exists = false; @@ -131,6 +135,9 @@ struct TTLTableDescription TTLDescriptions group_by_ttl; + /// `EXPORT TO TABLE` TTL. At most one per table. + TTLDescriptions export_ttl; + TTLTableDescription() = default; TTLTableDescription(const TTLTableDescription & other); TTLTableDescription & operator=(const TTLTableDescription & other); diff --git a/src/Storages/TTLMode.h b/src/Storages/TTLMode.h index bbbdbee400ae..9741cf22b97b 100644 --- a/src/Storages/TTLMode.h +++ b/src/Storages/TTLMode.h @@ -10,6 +10,8 @@ enum class TTLMode : uint8_t MOVE, GROUP_BY, RECOMPRESS, + /// `TTL EXPORT TO TABLE [db.]table`: copy expired parts to an Iceberg or object storage table. + EXPORT, }; } diff --git a/tests/integration/helpers/export_partition_helpers.py b/tests/integration/helpers/export_partition_helpers.py index d0cb8eb3d379..1f84f377d21f 100644 --- a/tests/integration/helpers/export_partition_helpers.py +++ b/tests/integration/helpers/export_partition_helpers.py @@ -39,6 +39,17 @@ def skip_if_remote_database_disk_enabled(cluster): ) +# Every `EXPORT PARTITION` creates a new task, so a partition may have several. The newest one is +# the one a test just started; of tasks created in the same second, a pending one is preferred so +# that an older finished task does not satisfy a wait. +_NEWEST_TASK = "ORDER BY create_time DESC, status = 'PENDING' DESC LIMIT 1" + + +def _task_filter(source_table, dest_table, partition_id): + dest_filter = f" AND destination_table = '{dest_table}'" if dest_table else "" + return f"source_table = '{source_table}'{dest_filter} AND partition_id = '{partition_id}'" + + def wait_for_export_status( node, source_table, @@ -48,7 +59,7 @@ def wait_for_export_status( timeout=60, poll_interval=0.5, ): - """Poll `system.partition_exports` until status matches. + """Poll `system.distributed_exports` until the status of the newest task of the partition matches. *dest_table* may be ``None`` to skip filtering by destination table (useful for catalog-based tests where the destination is a database-qualified path). @@ -56,14 +67,10 @@ def wait_for_export_status( start_time = time.time() last_status = None while time.time() - start_time < timeout: - dest_filter = ( - f" AND destination_table = '{dest_table}'" if dest_table else "" - ) status = node.query( - f"SELECT status FROM system.partition_exports" - f" WHERE source_table = '{source_table}'" - f"{dest_filter}" - f" AND partition_id = '{partition_id}'" + f"SELECT status FROM system.distributed_exports" + f" WHERE {_task_filter(source_table, dest_table, partition_id)}" + f" {_NEWEST_TASK}" ).strip() last_status = status @@ -78,19 +85,55 @@ def wait_for_export_status( ) +def newest_active_part(node, table, partition_id): + """Name of the active part of *partition_id* with the highest block number.""" + return node.query( + f"SELECT name FROM system.parts" + f" WHERE database = currentDatabase() AND table = '{table}'" + f" AND partition_id = '{partition_id}' AND active" + f" ORDER BY max_block_number DESC LIMIT 1" + ).strip() + + +def assert_parts_not_merged_across_export_states(node, table, partition_id, part_names): + """Trigger merges of *partition_id* and check that none of *part_names* was merged. Each of them + must be in a different export state (exported, claimed, not exported) than its neighbours, or + be claimed by an export task of the `EXPORT` TTL. + + Both merge paths are tried: `OPTIMIZE FINAL` merges the whole partition or nothing, a plain + `OPTIMIZE` runs the selector background merges use and may still merge parts in the same state + with each other, which is allowed. + """ + error = node.query_and_get_error( + f"OPTIMIZE TABLE {table} PARTITION ID '{partition_id}' FINAL", + settings={"optimize_throw_if_noop": 1}, + ) + assert "export states" in error or "are being exported" in error, f"Unexpected error on {node.name}: {error}" + + node.query(f"OPTIMIZE TABLE {table} PARTITION ID '{partition_id}'") + + active_parts = node.query( + f"SELECT name FROM system.parts" + f" WHERE database = currentDatabase() AND table = '{table}'" + f" AND partition_id = '{partition_id}' AND active" + ).split() + merged = [name for name in part_names if name not in active_parts] + assert not merged, ( + f"Parts {merged} were merged across export states on {node.name}; active parts: {active_parts}" + ) + + def export_transaction_id( node, source_table, dest_table, partition_id, ): - """Return the current transaction id of a partition export, or an empty string if none.""" - dest_filter = f" AND destination_table = '{dest_table}'" if dest_table else "" + """Return the transaction id of the newest export task of a partition, or an empty string if none.""" return node.query( - f"SELECT transaction_id FROM system.partition_exports" - f" WHERE source_table = '{source_table}'" - f"{dest_filter}" - f" AND partition_id = '{partition_id}'" + f"SELECT transaction_id FROM system.distributed_exports" + f" WHERE {_task_filter(source_table, dest_table, partition_id)}" + f" {_NEWEST_TASK}" ).strip() @@ -103,10 +146,10 @@ def wait_for_new_export_transaction( timeout=60, poll_interval=0.2, ): - """Wait until the export entry carries a transaction id other than *previous_transaction_id*. + """Wait until the newest export task of the partition is not *previous_transaction_id*. - A force re-export replaces the entry. Without this wait, the COMPLETED status of the export - being replaced can still be visible in the in-memory mirror and satisfy a status wait + A re-export creates a new task. Without this wait, the COMPLETED status of the previous task + can still be the newest one visible in the in-memory mirror and satisfy a status wait immediately, before the new export has even started. """ start_time = time.time() @@ -125,6 +168,20 @@ def wait_for_new_export_transaction( ) +def commit_marker_lines(node, source_table, dest_table, partition_id): + """Number of files committed to the plain object storage destination *dest_table* by the export + tasks of *partition_id*: each task commits with a marker `commit_` that lists them. + """ + transaction_ids = node.query( + f"SELECT transaction_id FROM system.distributed_exports WHERE {_task_filter(source_table, dest_table, partition_id)}" + ).split() + + return sum( + int(node.query(f"SELECT count() FROM s3(s3_conn, filename='{dest_table}/commit_{transaction_id}*', format=LineAsString)")) + for transaction_id in transaction_ids + ) + + def wait_for_export_to_start( node, source_table, @@ -133,11 +190,11 @@ def wait_for_export_to_start( timeout=10, poll_interval=0.2, ): - """Poll until at least one row exists in `system.partition_exports`.""" + """Poll until at least one row exists in `system.distributed_exports`.""" start_time = time.time() while time.time() - start_time < timeout: count = node.query( - f"SELECT count() FROM system.partition_exports" + f"SELECT count() FROM system.distributed_exports" f" WHERE source_table = '{source_table}'" f" AND destination_table = '{dest_table}'" f" AND partition_id = '{partition_id}'" @@ -162,11 +219,11 @@ def wait_for_exception_count( timeout=60, poll_interval=0.5, ): - """Wait for exception_count to reach at least *min_exception_count*. + """Wait for exception_count of the newest task of the partition to reach at least *min_exception_count*. The default timeout is intentionally larger than one manifest-updater poll - cycle (~30s, see StorageReplicatedMergeTree::exportMergeTreePartitionUpdatingTask). - For a ReplicatedMergeTree source, `system.partition_exports` is served from the + cycle (~30s, see StorageReplicatedMergeTree::exportTaskUpdatingTask). + For a ReplicatedMergeTree source, `system.distributed_exports` is served from the in-memory mirror, which is refreshed on (a) the periodic poll tick and (b) status changes. While the task is still PENDING (e.g. transient part-export failures with a generous max_retries), no status watch fires, so newly written @@ -177,10 +234,9 @@ def wait_for_exception_count( last_exception_count = None while time.time() - start_time < timeout: exception_count_str = node.query( - f"SELECT exception_count FROM system.partition_exports" - f" WHERE source_table = '{source_table}'" - f" AND destination_table = '{dest_table}'" - f" AND partition_id = '{partition_id}'" + f"SELECT exception_count FROM system.distributed_exports" + f" WHERE {_task_filter(source_table, dest_table, partition_id)}" + f" {_NEWEST_TASK}" ).strip() if exception_count_str: diff --git a/tests/integration/test_export_merge_tree_part_to_iceberg/test.py b/tests/integration/test_export_merge_tree_part_to_iceberg/test.py index 053dec967036..4920c9c1933a 100644 --- a/tests/integration/test_export_merge_tree_part_to_iceberg/test.py +++ b/tests/integration/test_export_merge_tree_part_to_iceberg/test.py @@ -1220,6 +1220,39 @@ def test_export_part_with_castable_widening(cluster): node.query(f"DROP TABLE IF EXISTS {iceberg}") +def test_export_part_date_and_time_after_the_destination_schema_is_reloaded(cluster): + """ + Iceberg stores `Date` and `DateTime` as `date` and `timestamp`, so once a query reloads the + destination schema from its metadata it declares `Date32` and `DateTime64(6)` columns. Every + value fits in them, so the export must not be refused as a lossy cast. + """ + node = cluster.instances["node1"] + sfx = unique_suffix() + mt = f"mt_time_reload_{sfx}" + iceberg = f"iceberg_time_reload_{sfx}" + + make_mt(node, mt, "id Int32, year Int32, d Date, t DateTime", "year") + make_iceberg_s3(node, iceberg, "id Int32, year Int32, d Date, t DateTime", "year") + node.query(f"SELECT * FROM {iceberg}") + types = node.query( + f"SELECT type FROM system.columns WHERE database = currentDatabase() AND table = '{iceberg}' " + f"AND name IN ('d', 't') ORDER BY name" + ).split("\n") + assert "Date32" in types[0] and "DateTime64(6)" in types[1], f"The destination schema was not reloaded: {types}" + + node.query(f"INSERT INTO {mt} VALUES (1, 2020, '2020-05-01', '2020-05-01 10:20:30')") + part_2020 = get_part(node, mt, "2020") + + export_part(node, mt, part_2020, iceberg) + wait_for_export_part(node, mt, part_2020) + + result = node.query(f"SELECT id, d, t FROM {iceberg} ORDER BY id").strip() + assert result == "1\t2020-05-01\t2020-05-01 10:20:30.000000", f"Unexpected exported data:\n{result}" + + node.query(f"DROP TABLE IF EXISTS {mt} SYNC") + node.query(f"DROP TABLE IF EXISTS {iceberg}") + + def test_export_part_with_castable_narrowing_values_fit(cluster): """A lossy narrowing (id Int64 -> Int32) succeeds once the user opts in via export_merge_tree_part_allow_lossy_cast.""" diff --git a/tests/integration/test_export_mt_partition_to_object_storage/test.py b/tests/integration/test_export_mt_partition_to_object_storage/test.py index e09e47725f05..7e8c75d67c43 100644 --- a/tests/integration/test_export_mt_partition_to_object_storage/test.py +++ b/tests/integration/test_export_mt_partition_to_object_storage/test.py @@ -4,6 +4,7 @@ from helpers.cluster import ClickHouseCluster from helpers.export_partition_helpers import ( + commit_marker_lines, make_mt, unique_suffix, wait_for_export_status, @@ -107,10 +108,7 @@ def test_export_partition_without_keeper(cluster): assert node.query(f"SELECT count() FROM {s3_table} WHERE year = 2020") == "3\n" assert ( - node.query( - f"SELECT count() FROM s3(s3_conn, filename='{s3_table}/commit_2020_*', format=LineAsString)" - ) - != "0\n" + commit_marker_lines(node, mt_table, s3_table, "2020") != 0 ), "Commit file missing for partition 2020" diff --git a/tests/integration/test_export_partition_to_iceberg/common.py b/tests/integration/test_export_partition_to_iceberg/common.py index 313695118511..0e6177a73f87 100644 --- a/tests/integration/test_export_partition_to_iceberg/common.py +++ b/tests/integration/test_export_partition_to_iceberg/common.py @@ -58,7 +58,7 @@ def _destination_paths_has_sync_failed_marker(node, source_table, dest_table, pa """True when destination_file_paths contains the Keeper sync-failed marker value.""" result = node.query( f"SELECT has(arrayFlatten(mapValues(destination_file_paths)), '')" - f" FROM system.partition_exports" + f" FROM system.distributed_exports" f" WHERE source_table = '{source_table}'" f" AND destination_table = '{dest_table}'" f" AND partition_id = '{partition_id}'" diff --git a/tests/integration/test_export_partition_to_iceberg/test_failures.py b/tests/integration/test_export_partition_to_iceberg/test_failures.py index 75c52c1fe60f..7721c8ff099c 100644 --- a/tests/integration/test_export_partition_to_iceberg/test_failures.py +++ b/tests/integration/test_export_partition_to_iceberg/test_failures.py @@ -27,7 +27,7 @@ def test_failure_is_logged_in_system_table(cluster, source_engine): """ When a part export fails with a non-retryable error the export must be marked - FAILED in system.partition_exports with a non-zero exception_count. + FAILED in system.distributed_exports with a non-zero exception_count. Uses the export_part_non_retryable_throw failpoint (throws BAD_ARGUMENTS, a denylisted code) so the task fails fast without consuming any timeout budget. @@ -54,7 +54,7 @@ def test_failure_is_logged_in_system_table(cluster, source_engine): status = node.query( f""" - SELECT status FROM system.partition_exports + SELECT status FROM system.distributed_exports WHERE source_table = '{mt_table}' AND destination_table = '{iceberg_table}' AND partition_id = '2020' @@ -64,13 +64,13 @@ def test_failure_is_logged_in_system_table(cluster, source_engine): exception_count = int(node.query( f""" - SELECT any(exception_count) FROM system.partition_exports + SELECT any(exception_count) FROM system.distributed_exports WHERE source_table = '{mt_table}' AND destination_table = '{iceberg_table}' AND partition_id = '2020' """ ).strip()) - assert exception_count > 0, "Expected non-zero exception_count in system.partition_exports" + assert exception_count > 0, "Expected non-zero exception_count in system.distributed_exports" count = int(node.query(f"SELECT count() FROM {iceberg_table}").strip()) assert count == 0, f"Expected 0 rows in Iceberg table after a failed export, got {count}" @@ -124,7 +124,7 @@ def test_inject_short_living_failures(cluster): status = node.query( f""" - SELECT status FROM system.partition_exports + SELECT status FROM system.distributed_exports WHERE source_table = '{mt_table}' AND destination_table = '{iceberg_table}' AND partition_id = '2020' @@ -134,7 +134,7 @@ def test_inject_short_living_failures(cluster): exception_count = int(node.query( f""" - SELECT exception_count FROM system.partition_exports + SELECT exception_count FROM system.distributed_exports WHERE source_table = '{mt_table}' AND destination_table = '{iceberg_table}' AND partition_id = '2020' @@ -163,7 +163,7 @@ def test_export_partition_retryable_error_killed_on_timeout(cluster, source_engi # first retry. With the new model there is no budget and only the 5s timeout fails it. node.query( f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {iceberg_table}" - f" SETTINGS export_merge_tree_partition_task_timeout_seconds = 5," + f" SETTINGS export_merge_tree_task_timeout_seconds = 5," f" allow_insert_into_iceberg = 1" ) @@ -171,7 +171,7 @@ def test_export_partition_retryable_error_killed_on_timeout(cluster, source_engi # budget would already have transitioned the task to FAILED by now. time.sleep(15) status = node.query( - f"SELECT status FROM system.partition_exports" + f"SELECT status FROM system.distributed_exports" f" WHERE source_table = '{mt_table}'" f" AND destination_table = '{iceberg_table}'" f" AND partition_id = '2020'" @@ -188,7 +188,7 @@ def test_export_partition_retryable_error_killed_on_timeout(cluster, source_engi node.query("SYSTEM DISABLE FAILPOINT export_part_retryable_throw") exception_count = int(node.query( - f"SELECT any(exception_count) FROM system.partition_exports" + f"SELECT any(exception_count) FROM system.distributed_exports" f" WHERE source_table = '{mt_table}'" f" AND destination_table = '{iceberg_table}'" f" AND partition_id = '2020'" @@ -218,8 +218,8 @@ def test_export_partition_retryable_error_recovers_after_failpoint_cleared(clust try: node.query( f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {iceberg_table}" - f" SETTINGS export_merge_tree_partition_retry_initial_backoff_seconds = 1," - f" export_merge_tree_partition_retry_max_backoff_seconds = 2," + f" SETTINGS export_merge_tree_retry_initial_backoff_seconds = 1," + f" export_merge_tree_retry_max_backoff_seconds = 2," f" allow_insert_into_iceberg = 1" ) @@ -228,7 +228,7 @@ def test_export_partition_retryable_error_recovers_after_failpoint_cleared(clust wait_for_exception_count(node, mt_table, iceberg_table, "2020", min_exception_count=1, timeout=60) status = node.query( - f"SELECT status FROM system.partition_exports" + f"SELECT status FROM system.distributed_exports" f" WHERE source_table = '{mt_table}'" f" AND destination_table = '{iceberg_table}'" f" AND partition_id = '2020'" @@ -277,8 +277,8 @@ def test_export_partition_local_backoff_does_not_block_other_replica(cluster): try: replica1.query( f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {iceberg_table}" - f" SETTINGS export_merge_tree_partition_retry_initial_backoff_seconds = 1," - f" export_merge_tree_partition_retry_max_backoff_seconds = 2," + f" SETTINGS export_merge_tree_retry_initial_backoff_seconds = 1," + f" export_merge_tree_retry_max_backoff_seconds = 2," f" allow_insert_into_iceberg = 1" ) @@ -294,7 +294,7 @@ def test_export_partition_local_backoff_does_not_block_other_replica(cluster): backoff_replica1 = "0" while time.time() < deadline: backoff_replica1 = replica1.query( - f"SELECT length(local_backoff_per_part) FROM system.partition_exports" + f"SELECT length(local_backoff_per_part) FROM system.distributed_exports" f" WHERE source_table = '{mt_table}'" f" AND destination_table = '{iceberg_table}'" f" AND partition_id = '2020'" @@ -310,7 +310,7 @@ def test_export_partition_local_backoff_does_not_block_other_replica(cluster): # ... and it must NOT have leaked to replica2, which never attempted the part. # This is the core assertion: local back-off state is not shared across replicas. backoff_replica2 = replica2.query( - f"SELECT length(local_backoff_per_part) FROM system.partition_exports" + f"SELECT length(local_backoff_per_part) FROM system.distributed_exports" f" WHERE source_table = '{mt_table}'" f" AND destination_table = '{iceberg_table}'" f" AND partition_id = '2020'" @@ -364,7 +364,7 @@ def test_export_partition_scheduler_skipped_when_moves_stopped(cluster, source_e time.sleep(12) status = node.query( - f"SELECT status FROM system.partition_exports" + f"SELECT status FROM system.distributed_exports" f" WHERE source_table = '{mt_table}' AND destination_table = '{iceberg_table}'" f" AND partition_id = '2020'" ).strip() @@ -414,7 +414,7 @@ def test_export_partition_resumes_after_stop_moves(cluster, source_engine): time.sleep(5) status = node.query( - f"SELECT status FROM system.partition_exports" + f"SELECT status FROM system.distributed_exports" f" WHERE source_table = '{mt_table}' AND destination_table = '{iceberg_table}'" f" AND partition_id = '2020'" ).strip() @@ -475,7 +475,7 @@ def test_export_partition_resumes_after_stop_moves_during_export(cluster, source time.sleep(3) status = node.query( - f"SELECT status FROM system.partition_exports" + f"SELECT status FROM system.distributed_exports" f" WHERE source_table = '{mt_table}' AND destination_table = '{iceberg_table}'" f" AND partition_id = '2020'" ).strip() @@ -588,11 +588,11 @@ def test_post_publish_exception_preserves_snapshot(cluster): # After a post-publish exception the catch handler with published==true returns # the populated commit info (real metadata / manifest list / manifest file paths). - # ExportPartitionUtils::commit persists it to the commit_info znode, so the system + # ExportTaskUtils::commit persists it to the commit_info znode, so the system # table should show a real metadata path here, not the already-committed sentinel. committed_metadata_file = node.query( f""" - SELECT committed_metadata_file FROM system.partition_exports + SELECT committed_metadata_file FROM system.distributed_exports WHERE source_table = '{mt_table}' AND destination_table = '{iceberg_table}' AND partition_id = '2020' @@ -611,7 +611,7 @@ def test_post_publish_exception_preserves_snapshot(cluster): def test_export_task_timeout_kills_stuck_pending_task(cluster): """ - Verify that export_merge_tree_partition_task_timeout_seconds auto-kills a task + Verify that export_merge_tree_task_timeout_seconds auto-kills a task that remains PENDING past the deadline, transitioning it to KILLED with a descriptive last_exception. @@ -620,9 +620,9 @@ def test_export_task_timeout_kills_stuck_pending_task(cluster): retryable error, so the task never fails on its own and the timeout branch in tryCleanup is the actual mechanism under test. - Replicated-only: the failpoint lives in `ExportPartitionUtils::commit`, the + Replicated-only: the failpoint lives in `ExportTaskUtils::commit`, the ZooKeeper-coordinated commit routine. A plain MergeTree commits through - `MergeTreePartitionExportScheduler::tryCommit`, which the failpoint does not reach, so the + `MergeTreeExportTaskScheduler::tryCommit`, which the failpoint does not reach, so the export simply completes and there is nothing for the timeout to kill. """ node = cluster.instances["replica1"] @@ -636,7 +636,7 @@ def test_export_task_timeout_kills_stuck_pending_task(cluster): try: node.query( f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {iceberg_table}" - f" SETTINGS export_merge_tree_partition_task_timeout_seconds = 5," + f" SETTINGS export_merge_tree_task_timeout_seconds = 5," f" allow_insert_into_iceberg = 1" ) @@ -662,7 +662,7 @@ def test_export_task_timeout_kills_stuck_pending_task(cluster): arrayMap(x -> x.message, last_exception_per_replica), '\\n' ) - FROM system.partition_exports + FROM system.distributed_exports WHERE source_table = '{mt_table}' AND destination_table = '{iceberg_table}' AND partition_id = '2020' diff --git a/tests/integration/test_export_partition_to_iceberg/test_schema_match.py b/tests/integration/test_export_partition_to_iceberg/test_schema_match.py index 755b1095816c..ce3e6b3f9ced 100644 --- a/tests/integration/test_export_partition_to_iceberg/test_schema_match.py +++ b/tests/integration/test_export_partition_to_iceberg/test_schema_match.py @@ -2,10 +2,12 @@ from helpers.export_partition_helpers import ( EXTRA_SOURCE_COLUMN_MODES, + export_transaction_id, make_iceberg_s3, make_source, unique_suffix, wait_for_export_status, + wait_for_new_export_transaction, ) CLUSTER_INSTANCES = ["replica1"] @@ -18,7 +20,7 @@ def test_export_partition_column_count_mismatch_source_more_is_rejected(cluster, """ Source has 3 columns (id, year, extra), destination has 2 (id, year). The ALTER must be rejected synchronously with NUMBER_OF_COLUMNS_DOESNT_MATCH, - nothing must be scheduled in system.partition_exports, and the + nothing must be scheduled in system.distributed_exports, and the Iceberg table must remain empty. """ node = cluster.instances["replica1"] @@ -43,13 +45,13 @@ def test_export_partition_column_count_mismatch_source_more_is_rejected(cluster, ) rows_in_system_view = node.query( - f"SELECT count() FROM system.partition_exports " + f"SELECT count() FROM system.distributed_exports " f"WHERE source_table = '{mt_table}' " f" AND destination_table = '{iceberg_table}' " f" AND partition_id = '2020'" ).strip() assert rows_in_system_view == "0", ( - f"Expected no row in system.partition_exports after a " + f"Expected no row in system.distributed_exports after a " f"synchronously-rejected export, got {rows_in_system_view}." ) @@ -86,13 +88,13 @@ def test_export_partition_column_count_mismatch_source_fewer_is_rejected(cluster ) rows_in_system_view = node.query( - f"SELECT count() FROM system.partition_exports " + f"SELECT count() FROM system.distributed_exports " f"WHERE source_table = '{mt_table}' " f" AND destination_table = '{iceberg_table}' " f" AND partition_id = '2020'" ).strip() assert rows_in_system_view == "0", ( - f"Expected no row in system.partition_exports after a " + f"Expected no row in system.distributed_exports after a " f"synchronously-rejected export, got {rows_in_system_view}." ) @@ -205,13 +207,13 @@ def test_export_partition_column_count_mismatch_source_fewer_still_rejected_with ) rows_in_system_view = node.query( - f"SELECT count() FROM system.partition_exports " + f"SELECT count() FROM system.distributed_exports " f"WHERE source_table = '{mt_table}' " f" AND destination_table = '{iceberg_table}' " f" AND partition_id = '2020'" ).strip() assert rows_in_system_view == "0", ( - f"Expected no row in system.partition_exports after a " + f"Expected no row in system.distributed_exports after a " f"synchronously-rejected export, got {rows_in_system_view}." ) @@ -245,13 +247,13 @@ def test_export_partition_column_count_mismatch_source_fewer_reports_column_coun ) rows_in_system_view = node.query( - f"SELECT count() FROM system.partition_exports " + f"SELECT count() FROM system.distributed_exports " f"WHERE source_table = '{mt_table}' " f" AND destination_table = '{iceberg_table}' " f" AND partition_id = '2020'" ).strip() assert rows_in_system_view == "0", ( - f"Expected no row in system.partition_exports after a " + f"Expected no row in system.distributed_exports after a " f"synchronously-rejected export, got {rows_in_system_view}." ) @@ -286,13 +288,13 @@ def test_export_partition_column_count_mismatch_source_fewer_reports_column_coun ) rows_in_system_view = node.query( - f"SELECT count() FROM system.partition_exports " + f"SELECT count() FROM system.distributed_exports " f"WHERE source_table = '{mt_table}' " f" AND destination_table = '{iceberg_table}' " f" AND partition_id = '2020'" ).strip() assert rows_in_system_view == "0", ( - f"Expected no row in system.partition_exports after a " + f"Expected no row in system.distributed_exports after a " f"synchronously-rejected export, got {rows_in_system_view}." ) @@ -542,17 +544,20 @@ def test_export_partition_column_count_mismatch_into_partition_that_already_has_ f"Expected 2 rows after first export, got {count_after_first}" ) + first_transaction_id = export_transaction_id(node, mt_table, iceberg_table, "2020") + node.query(f"INSERT INTO {mt_table} VALUES (3, 2020, 'c'), (4, 2020, 'd')") node.query( f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {iceberg_table}", - settings={**ignore_extra_settings, "export_merge_tree_partition_force_export": 1}, + settings=ignore_extra_settings, ) + wait_for_new_export_transaction(node, mt_table, iceberg_table, "2020", first_transaction_id) wait_for_export_status(node=node, source_table=mt_table, dest_table=iceberg_table, partition_id="2020", expected_status="COMPLETED") count_after_second = int(node.query(f"SELECT count() FROM {iceberg_table}").strip()) assert count_after_second == 6, ( - f"Expected 6 rows (2 original + 2 duplicated by the forced re-export + 2 new) " + f"Expected 6 rows (2 original + 2 duplicated by the re-export + 2 new) " f"after re-exporting an already-populated partition, got {count_after_second}" ) @@ -723,7 +728,7 @@ def test_export_partition_runtime_cast_failure_propagates_async(cluster, source_ wait_for_export_status(node, mt_table, iceberg_table, "2020", "FAILED", timeout=60) exception_count = int(node.query( - f"SELECT any(exception_count) FROM system.partition_exports " + f"SELECT any(exception_count) FROM system.distributed_exports " f"WHERE source_table = '{mt_table}' " f" AND destination_table = '{iceberg_table}' " f" AND partition_id = '2020'" diff --git a/tests/integration/test_export_partition_to_object_storage/test_failures.py b/tests/integration/test_export_partition_to_object_storage/test_failures.py index 8c5a2aaa05f5..7b7b203835ea 100644 --- a/tests/integration/test_export_partition_to_object_storage/test_failures.py +++ b/tests/integration/test_export_partition_to_object_storage/test_failures.py @@ -2,6 +2,7 @@ import uuid from helpers.export_partition_helpers import ( + commit_marker_lines, setup_source_tables, skip_if_remote_database_disk_enabled, wait_for_exception_count, @@ -81,7 +82,7 @@ def test_kill_export(cluster, source_engine): # Kill only 2020 while S3 is blocked - retry mechanism keeps exports alive # ZooKeeper operations (KILL) proceed quickly since only S3 is blocked - node.query(f"KILL EXPORT PARTITION WHERE partition_id = '2020' and source_table = '{mt_table}' and destination_table = '{s3_table}'") + node.query(f"KILL EXPORT WHERE partition_id = '2020' and source_table = '{mt_table}' and destination_table = '{s3_table}'") # sleep for a while to let the kill to be processed time.sleep(2) @@ -90,19 +91,19 @@ def test_kill_export(cluster, source_engine): wait_for_export_status(node, mt_table, s3_table, "2021", "COMPLETED") # checking for the commit file because maybe the data file was too fast? - assert node.query(f"SELECT count() FROM s3(s3_conn, filename='{s3_table}/commit_2020_*', format=LineAsString)") == '0\n', "Partition 2020 was written to S3, it was not killed as expected" - assert node.query(f"SELECT count() FROM s3(s3_conn, filename='{s3_table}/commit_2021_*', format=LineAsString)") != f'0\n', "Partition 2021 was not written to S3, but it should have been" + assert commit_marker_lines(node, mt_table, s3_table, "2020") == 0, "Partition 2020 was written to S3, it was not killed as expected" + assert commit_marker_lines(node, mt_table, s3_table, "2021") != 0, "Partition 2021 was not written to S3, but it should have been" - # check system.partition_exports for the export, status should be KILLED - assert node.query(f"SELECT status FROM system.partition_exports WHERE partition_id = '2020' and source_table = '{mt_table}' and destination_table = '{s3_table}'") == 'KILLED\n', "Partition 2020 was not killed as expected" - assert node.query(f"SELECT status FROM system.partition_exports WHERE partition_id = '2021' and source_table = '{mt_table}' and destination_table = '{s3_table}'") == 'COMPLETED\n', "Partition 2021 was not completed, this is unexpected" + # check system.distributed_exports for the export, status should be KILLED + assert node.query(f"SELECT status FROM system.distributed_exports WHERE partition_id = '2020' and source_table = '{mt_table}' and destination_table = '{s3_table}'") == 'KILLED\n', "Partition 2020 was not killed as expected" + assert node.query(f"SELECT status FROM system.distributed_exports WHERE partition_id = '2021' and source_table = '{mt_table}' and destination_table = '{s3_table}'") == 'COMPLETED\n', "Partition 2021 was not completed, this is unexpected" # check the data did not land on s3 assert node.query(f"SELECT count() FROM {s3_table} WHERE year = 2020") == '0\n', "Partition 2020 was written to S3, it was not killed as expected" def test_kill_export_resilient_to_status_handling_failure(cluster): - """KILL EXPORT PARTITION must eventually take effect even when the first + """KILL EXPORT must eventually take effect even when the first attempt to handle the ZK status-change event throws (simulated via a ONCE failpoint). The re-queue + reschedule mechanism retries after ~5 s and the second attempt succeeds because the ONCE failpoint has already fired.""" @@ -142,7 +143,7 @@ def test_kill_export_resilient_to_status_handling_failure(cluster): node.query("SYSTEM ENABLE FAILPOINT export_partition_status_change_throw") node.query( - f"KILL EXPORT PARTITION WHERE partition_id = '2020'" + f"KILL EXPORT WHERE partition_id = '2020'" f" AND source_table = '{mt_table}' AND destination_table = '{s3_table}'") # sleep for a while to let the kill to be processed @@ -155,7 +156,7 @@ def test_kill_export_resilient_to_status_handling_failure(cluster): assert ( node.query( - f"SELECT status FROM system.partition_exports" + f"SELECT status FROM system.distributed_exports" f" WHERE partition_id = '2020'" f" AND source_table = '{mt_table}'" f" AND destination_table = '{s3_table}'" @@ -254,12 +255,8 @@ def test_concurrent_exports_to_different_targets(cluster): assert node.query(f"SELECT count() FROM {s3_table_b} WHERE year = 2020") == '3\n', "Second target did not receive expected rows" # And both should have a commit marker - assert node.query( - f"SELECT count() FROM s3(s3_conn, filename='{s3_table_a}/commit_2020_*', format=LineAsString)" - ) != '0\n', "Commit file missing for first target" - assert node.query( - f"SELECT count() FROM s3(s3_conn, filename='{s3_table_b}/commit_2020_*', format=LineAsString)" - ) != '0\n', "Commit file missing for second target" + assert commit_marker_lines(node, mt_table, s3_table_a, "2020") != 0, "Commit file missing for first target" + assert commit_marker_lines(node, mt_table, s3_table_b, "2020") != 0, "Commit file missing for second target" def test_failure_is_logged_in_system_table(cluster): @@ -303,7 +300,7 @@ def test_failure_is_logged_in_system_table(cluster): # so the test does not wait for the default (a day). node.query( f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {s3_table}" - f" SETTINGS export_merge_tree_partition_task_timeout_seconds = 5;" + f" SETTINGS export_merge_tree_task_timeout_seconds = 5;" ) # Wait for the timeout to kill the stuck task. The KILL is a Keeper operation @@ -316,7 +313,7 @@ def test_failure_is_logged_in_system_table(cluster): # Also verify we captured at least one exception and no commit file exists status = node.query( f""" - SELECT status FROM system.partition_exports + SELECT status FROM system.distributed_exports WHERE source_table = '{mt_table}' AND destination_table = '{s3_table}' AND partition_id = '2020' @@ -327,18 +324,16 @@ def test_failure_is_logged_in_system_table(cluster): exception_count = node.query( f""" - SELECT any(exception_count) FROM system.partition_exports + SELECT any(exception_count) FROM system.distributed_exports WHERE source_table = '{mt_table}' AND destination_table = '{s3_table}' AND partition_id = '2020' """ ) - assert int(exception_count.strip()) > 0, "Expected non-zero exception_count in system.partition_exports" + assert int(exception_count.strip()) > 0, "Expected non-zero exception_count in system.distributed_exports" # No commit should have been produced for this partition - assert node.query( - f"SELECT count() FROM s3(s3_conn, filename='{s3_table}/commit_2020_*', format=LineAsString)" - ) == '0\n', "Commit file exists despite forced S3 failures" + assert commit_marker_lines(node, mt_table, s3_table, "2020") == 0, "Commit file exists despite forced S3 failures" def test_inject_short_living_failures(cluster): @@ -383,7 +378,7 @@ def test_inject_short_living_failures(cluster): ) # wait for at least one exception to occur, but not enough to finish the export. - # Use the helper default (>= one manifest-updater poll cycle): system.partition_exports + # Use the helper default (>= one manifest-updater poll cycle): system.distributed_exports # is served from the in-memory mirror, and while the task stays PENDING the mirror only # picks up new exception leaves on the next poll tick (~30s) — see helper docstring. wait_for_exception_count(node, mt_table, s3_table, "2020", min_exception_count=1) @@ -393,12 +388,12 @@ def test_inject_short_living_failures(cluster): # Assert the export succeeded assert node.query(f"SELECT count() FROM {s3_table} WHERE year = 2020") == '3\n', "Export did not succeed" - assert node.query(f"SELECT count() FROM s3(s3_conn, filename='{s3_table}/commit_2020_*', format=LineAsString)") == '1\n', "Export did not succeed" + assert commit_marker_lines(node, mt_table, s3_table, "2020") == 1, "Export did not succeed" - # check system.partition_exports for the export + # check system.distributed_exports for the export assert node.query( f""" - SELECT status FROM system.partition_exports + SELECT status FROM system.distributed_exports WHERE source_table = '{mt_table}' AND destination_table = '{s3_table}' AND partition_id = '2020' @@ -407,7 +402,7 @@ def test_inject_short_living_failures(cluster): exception_count = node.query( f""" - SELECT exception_count FROM system.partition_exports + SELECT exception_count FROM system.distributed_exports WHERE source_table = '{mt_table}' AND destination_table = '{s3_table}' AND partition_id = '2020' @@ -464,8 +459,8 @@ def test_export_partition_retry_backoff(cluster, source_engine): node.query( f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {s3_table} " - f"SETTINGS export_merge_tree_partition_retry_initial_backoff_seconds = {initial_backoff_seconds}, " - f"export_merge_tree_partition_retry_max_backoff_seconds = {max_backoff_seconds}" + f"SETTINGS export_merge_tree_retry_initial_backoff_seconds = {initial_backoff_seconds}, " + f"export_merge_tree_retry_max_backoff_seconds = {max_backoff_seconds}" ) # Wait until the first failure is recorded. @@ -479,7 +474,7 @@ def test_export_partition_retry_backoff(cluster, source_engine): # the back-off is pacing retries. time.sleep(25) count_during_backoff = int(node.query( - f"SELECT exception_count FROM system.partition_exports" + f"SELECT exception_count FROM system.distributed_exports" f" WHERE source_table = '{mt_table}'" f" AND destination_table = '{s3_table}'" f" AND partition_id = '2020'" @@ -620,7 +615,7 @@ def test_export_partition_scheduler_skipped_when_moves_stopped(cluster, source_e time.sleep(10) status = node.query( - f"SELECT status FROM system.partition_exports" + f"SELECT status FROM system.distributed_exports" f" WHERE source_table = '{mt_table}' AND destination_table = '{s3_table}'" f" AND partition_id = '2020'" ).strip() @@ -664,7 +659,7 @@ def test_export_partition_resumes_after_stop_moves(cluster, source_engine): time.sleep(5) status = node.query( - f"SELECT status FROM system.partition_exports" + f"SELECT status FROM system.distributed_exports" f" WHERE source_table = '{mt_table}' AND destination_table = '{s3_table}'" f" AND partition_id = '2020'" ).strip() @@ -726,7 +721,7 @@ def test_export_partition_resumes_after_stop_moves_during_export(cluster, source time.sleep(3) status = node.query( - f"SELECT status FROM system.partition_exports" + f"SELECT status FROM system.distributed_exports" f" WHERE source_table = '{mt_table}' AND destination_table = '{s3_table}'" f" AND partition_id = '2020'" ).strip() @@ -761,7 +756,7 @@ def test_dispatch_fails_when_destination_dropped(cluster): wait_for_export_to_start(node, mt_table, s3_table, "2020") status = node.query( - f"SELECT status FROM system.partition_exports" + f"SELECT status FROM system.distributed_exports" f" WHERE source_table = '{mt_table}' AND destination_table = '{s3_table}'" f" AND partition_id = '2020'" ).strip() @@ -773,7 +768,7 @@ def test_dispatch_fails_when_destination_dropped(cluster): wait_for_export_status(node, mt_table, s3_table, "2020", "FAILED", timeout=60) last_exceptions = node.query( - f"SELECT last_exception_per_replica FROM system.partition_exports" + f"SELECT last_exception_per_replica FROM system.distributed_exports" f" WHERE source_table = '{mt_table}'" f" AND destination_table = '{s3_table}'" f" AND partition_id = '2020'" @@ -811,7 +806,7 @@ def test_dispatch_fails_when_destination_schema_incompatible(cluster): wait_for_export_status(node, mt_table, s3_table, "2020", "FAILED", timeout=60) last_exceptions = node.query( - f"SELECT last_exception_per_replica FROM system.partition_exports" + f"SELECT last_exception_per_replica FROM system.distributed_exports" f" WHERE source_table = '{mt_table}'" f" AND destination_table = '{s3_table}'" f" AND partition_id = '2020'" @@ -856,13 +851,13 @@ def test_export_task_timeout_kills_stuck_pending_task(cluster, source_engine): node.query( f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {s3_table}" - f" SETTINGS export_merge_tree_partition_task_timeout_seconds = 5" + f" SETTINGS export_merge_tree_task_timeout_seconds = 5" ) wait_for_export_status(node, mt_table, s3_table, "2020", "KILLED", timeout=90) last_exceptions = node.query( - f"SELECT last_exception_per_replica FROM system.partition_exports" + f"SELECT last_exception_per_replica FROM system.distributed_exports" f" WHERE source_table = '{mt_table}'" f" AND destination_table = '{s3_table}'" f" AND partition_id = '2020'" @@ -871,7 +866,5 @@ def test_export_task_timeout_kills_stuck_pending_task(cluster, source_engine): f"Expected the recorded exception to mention the timeout reason, got: {last_exceptions!r}" ) - assert node.query( - f"SELECT count() FROM s3(s3_conn, filename='{s3_table}/commit_2020_*', format=LineAsString)" - ) == "0\n", "Commit file exists despite the task timeout" + assert commit_marker_lines(node, mt_table, s3_table, "2020") == 0, "Commit file exists despite the task timeout" assert node.query(f"SELECT count() FROM {s3_table} WHERE year = 2020") == "0\n" diff --git a/tests/integration/test_export_partition_to_object_storage/test_lifecycle.py b/tests/integration/test_export_partition_to_object_storage/test_lifecycle.py index e1d19d3f46c9..3d676ac40768 100644 --- a/tests/integration/test_export_partition_to_object_storage/test_lifecycle.py +++ b/tests/integration/test_export_partition_to_object_storage/test_lifecycle.py @@ -37,10 +37,10 @@ def test_export_partition_file_already_exists_policy(cluster, source_engine): f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {s3_table}", ) - # check system.partition_exports for the export + # check system.distributed_exports for the export assert node.query( f""" - SELECT status FROM system.partition_exports + SELECT status FROM system.distributed_exports WHERE source_table = '{mt_table}' AND destination_table = '{s3_table}' AND partition_id = '2020' @@ -51,20 +51,21 @@ def test_export_partition_file_already_exists_policy(cluster, source_engine): wait_for_export_status(node, mt_table, s3_table, "2020", "COMPLETED") # plain object storage destinations surface the commit marker file path via - # system.partition_exports.committed_marker_file + # system.distributed_exports.committed_marker_file committed_marker_file = node.query( f""" - SELECT committed_marker_file FROM system.partition_exports + SELECT committed_marker_file FROM system.distributed_exports WHERE source_table = '{mt_table}' AND destination_table = '{s3_table}' AND partition_id = '2020' """ ).strip() + transaction_id = export_transaction_id(node, mt_table, s3_table, "2020") # `committed_marker_file` is the absolute key in the bucket (same convention as # `destination_file_paths`); it may carry the s3_conn URL's in-bucket prefix on # top of the table's `filename` argument, so use a "contains" check that does # not depend on knowing that prefix. - assert f"{s3_table}/commit_2020_" in committed_marker_file, \ + assert f"{s3_table}/commit_{transaction_id}" in committed_marker_file, \ f"Expected committed_marker_file under {s3_table}/, got: {committed_marker_file!r}" # Path relative to the `s3_conn` URL, derived from the absolute key without # assuming a particular URL prefix. @@ -73,64 +74,41 @@ def test_export_partition_file_already_exists_policy(cluster, source_engine): f"SELECT count() FROM s3(s3_conn, filename='{marker_relative_path}', format=LineAsString)" ) == '1\n', f"Commit marker file does not exist at {committed_marker_file!r}" - # try to export the partition - node.query( - f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {s3_table} SETTINGS export_merge_tree_partition_force_export=1" - ) - + def completed_tasks(): + return int(node.query( + f""" + SELECT count() FROM system.distributed_exports + WHERE source_table = '{mt_table}' + AND destination_table = '{s3_table}' + AND partition_id = '2020' + AND status = 'COMPLETED' + """ + )) + + # Exporting the partition again is allowed, and creates a new task. + node.query(f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {s3_table}") + transaction_id = wait_for_new_export_transaction(node, mt_table, s3_table, "2020", transaction_id) wait_for_export_status(node, mt_table, s3_table, "2020", "COMPLETED") - - assert node.query( - f""" - SELECT count() FROM system.partition_exports - WHERE source_table = '{mt_table}' - AND destination_table = '{s3_table}' - AND partition_id = '2020' - AND status = 'COMPLETED' - """ - ) == '1\n', "Expected the export to be marked as COMPLETED" + assert completed_tasks() == 2, "Expected both exports to be marked as COMPLETED" # overwrite policy node.query( - f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {s3_table} SETTINGS export_merge_tree_partition_force_export=1, export_merge_tree_part_file_already_exists_policy='overwrite'" + f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {s3_table} SETTINGS export_merge_tree_part_file_already_exists_policy='overwrite'" ) - - # wait for the export to finish + transaction_id = wait_for_new_export_transaction(node, mt_table, s3_table, "2020", transaction_id) wait_for_export_status(node, mt_table, s3_table, "2020", "COMPLETED") - - # check system.partition_exports for the export - # ideally we would make sure the transaction id is different, but I do not have the time to do that now - assert node.query( - f""" - SELECT count() FROM system.partition_exports - WHERE source_table = '{mt_table}' - AND destination_table = '{s3_table}' - AND partition_id = '2020' - AND status = 'COMPLETED' - """ - ) == '1\n', "Expected the export to be marked as COMPLETED" + assert completed_tasks() == 3, "Expected every export to be marked as COMPLETED" # last but not least, the error policy. The `overwrite` export above finished every part and # left a per-part commit marker proving it, so there is nothing for this export to write and # it completes by reusing those files. `error` only refuses destination files that no commit # marker covers -- see test_export_partition_error_policy_rejects_incomplete_part. node.query( - f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {s3_table} SETTINGS export_merge_tree_partition_force_export=1, export_merge_tree_part_file_already_exists_policy='error'", + f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {s3_table} SETTINGS export_merge_tree_part_file_already_exists_policy='error'", ) - - # wait for the export to finish + wait_for_new_export_transaction(node, mt_table, s3_table, "2020", transaction_id) wait_for_export_status(node, mt_table, s3_table, "2020", "COMPLETED") - - # check system.partition_exports for the export - assert node.query( - f""" - SELECT count() FROM system.partition_exports - WHERE source_table = '{mt_table}' - AND destination_table = '{s3_table}' - AND partition_id = '2020' - AND status = 'COMPLETED' - """ - ) == '1\n', "Expected the export to be marked as COMPLETED" + assert completed_tasks() == 4, "Expected every export to be marked as COMPLETED" def create_split_export_tables(node, mt_table, s3_table, replica_name, engine): @@ -154,7 +132,7 @@ def create_split_export_tables(node, mt_table, s3_table, replica_name, engine): def export_partition_split_into_files( - node, mt_table, s3_table, force=False, policy=None, previous_transaction_id=None, + node, mt_table, s3_table, policy=None, previous_transaction_id=None, expected_status="COMPLETED", ): """Export partition 2020 with one row per destination file and wait for *expected_status*. @@ -162,8 +140,6 @@ def export_partition_split_into_files( Only splits per row for a table built by `create_split_export_tables`. """ settings = ["export_merge_tree_part_max_rows_per_file = 1"] - if force: - settings.append("export_merge_tree_partition_force_export = 1") if policy: settings.append(f"export_merge_tree_part_file_already_exists_policy = '{policy}'") @@ -179,34 +155,33 @@ def export_partition_split_into_files( def recorded_export_paths(node, mt_table, s3_table): - """Destination file paths recorded for the exported parts, in the order the sink wrote them. + """Destination file paths recorded for the exported parts by the newest export of partition + 2020, in the order the sink wrote them. - This is what the commit phase turns into the partition commit marker. + This is what the commit phase turns into the commit marker. """ + transaction_id = export_transaction_id(node, mt_table, s3_table, "2020") paths = node.query( f""" SELECT arrayJoin(arrayFlatten(mapValues(destination_file_paths))) - FROM system.partition_exports - WHERE source_table = '{mt_table}' - AND destination_table = '{s3_table}' - AND partition_id = '2020' + FROM system.distributed_exports + WHERE transaction_id = '{transaction_id}' """ ) return [path for path in paths.splitlines() if path] def partition_commit_marker_lines(node, mt_table, s3_table): - """Data-file paths listed inside the partition-level commit marker.""" + """Data-file paths listed inside the commit marker of the newest export of partition 2020.""" + transaction_id = export_transaction_id(node, mt_table, s3_table, "2020") committed_marker_file = node.query( f""" - SELECT committed_marker_file FROM system.partition_exports - WHERE source_table = '{mt_table}' - AND destination_table = '{s3_table}' - AND partition_id = '2020' + SELECT committed_marker_file FROM system.distributed_exports + WHERE transaction_id = '{transaction_id}' """ ).strip() - assert f"{s3_table}/commit_2020_" in committed_marker_file, \ + assert f"{s3_table}/commit_{transaction_id}" in committed_marker_file, \ f"Expected committed_marker_file under {s3_table}/, got: {committed_marker_file!r}" marker_relative_path = committed_marker_file[committed_marker_file.index(f"{s3_table}/"):] @@ -264,7 +239,7 @@ def test_export_partition_skip_policy_reports_every_split_file(cluster, source_e # Re-export. Every destination file is already there, so `skip` short-circuits the part -- # but it must do so with the complete file list. export_partition_split_into_files( - node, mt_table, s3_table, force=True, policy="skip", + node, mt_table, s3_table, policy="skip", previous_transaction_id=first_transaction_id, ) @@ -320,7 +295,7 @@ def test_export_partition_skip_policy_reexports_incomplete_part(cluster, source_ f"Expected the per-part commit marker to be gone, got {surviving_markers}" export_partition_split_into_files( - node, mt_table, s3_table, force=True, policy="skip", + node, mt_table, s3_table, policy="skip", previous_transaction_id=first_transaction_id, ) @@ -366,7 +341,7 @@ def test_export_partition_error_policy_adopts_completed_part(cluster, source_eng # Stands in for the retry of a part whose success was never recorded: same part, same # destination paths, same commit marker, `error` policy. export_partition_split_into_files( - node, mt_table, s3_table, force=True, policy="error", + node, mt_table, s3_table, policy="error", previous_transaction_id=first_transaction_id, ) @@ -417,7 +392,7 @@ def test_export_partition_error_policy_rejects_incomplete_part(cluster, source_e cluster.minio_client.remove_object(cluster.minio_bucket, key) export_partition_split_into_files( - node, mt_table, s3_table, force=True, policy="error", + node, mt_table, s3_table, policy="error", previous_transaction_id=first_transaction_id, expected_status="FAILED", ) @@ -442,7 +417,7 @@ def test_export_partition_feature_is_disabled(cluster, source_engine): assert "experimental" in error, "Expected error about disabled feature" # make sure kill operation also throws - error = replica_with_export_disabled.query_and_get_error(f"KILL EXPORT PARTITION WHERE partition_id = '2020' and source_table = '{mt_table}' and destination_table = '{s3_table}'") + error = replica_with_export_disabled.query_and_get_error(f"KILL EXPORT WHERE partition_id = '2020' and source_table = '{mt_table}' and destination_table = '{s3_table}'") assert "experimental" in error, "Expected error about disabled feature" @@ -511,7 +486,7 @@ def test_export_partition_permissions(cluster, source_engine): # Verify system table shows COMPLETED status status = node.query( f""" - SELECT status FROM system.partition_exports + SELECT status FROM system.distributed_exports WHERE source_table = '{mt_table}' AND destination_table = '{s3_table}' AND partition_id = '2020' @@ -540,10 +515,10 @@ def test_multiple_exports_within_a_single_query(cluster, source_engine): assert node.query(f"SELECT count() FROM {s3_table} WHERE year = 2020") == '3\n', "Export did not succeed" assert node.query(f"SELECT count() FROM {s3_table} WHERE year = 2021") == '1\n', "Export did not succeed" - # check system.partition_exports for the exports + # check system.distributed_exports for the exports assert node.query( f""" - SELECT status FROM system.partition_exports + SELECT status FROM system.distributed_exports WHERE source_table = '{mt_table}' AND destination_table = '{s3_table}' AND partition_id = '2020' @@ -552,7 +527,7 @@ def test_multiple_exports_within_a_single_query(cluster, source_engine): assert node.query( f""" - SELECT status FROM system.partition_exports + SELECT status FROM system.distributed_exports WHERE source_table = '{mt_table}' AND destination_table = '{s3_table}' AND partition_id = '2021' @@ -762,7 +737,7 @@ def test_export_partition_with_mixed_computed_columns(cluster, source_engine): assert dest_result == expected, f"Exported data mismatch. Expected:\n{expected}\nGot:\n{dest_result}" status = node.query(f""" - SELECT status FROM system.partition_exports + SELECT status FROM system.distributed_exports WHERE source_table = '{mt_table}' AND destination_table = '{s3_table}' AND partition_id = '1' @@ -803,7 +778,8 @@ def test_export_partition_all_failure_modes(cluster, source_engine): """Cover the three values of `export_merge_tree_partition_all_on_error`. Set up an already-fully-exported source table, then re-run EXPORT PARTITION ALL - with each failure mode and assert the documented behavior. + with each failure mode. A partition that was exported before is exported again, so only a + partition that cannot be exported at all fails. """ node = cluster.instances["replica1"] @@ -838,32 +814,47 @@ def test_export_partition_all_failure_modes(cluster, source_engine): f"Expected 'no active partitions' error, got: {error}" ) - # throw_first (default): re-run aborts on the first conflicting partition. + def tasks_of(partition_id): + return int(node.query( + f"SELECT count() FROM system.distributed_exports" + f" WHERE source_table = '{mt_table}' AND destination_table = '{s3_table}' AND partition_id = '{partition_id}'" + )) + + # Every mode exports every partition again, as a new task: there are no conflicts to skip. + for expected_tasks, mode in enumerate(("throw_first", "collect", "skip_conflicts"), start=2): + node.query( + f"ALTER TABLE {mt_table} EXPORT PARTITION ALL TO TABLE {s3_table}" + f" SETTINGS export_merge_tree_partition_all_on_error = '{mode}'" + ) + for partition_id in ("2020", "2021", "2022"): + assert tasks_of(partition_id) == expected_tasks, ( + f"Expected {expected_tasks} tasks of partition {partition_id} after '{mode}'" + ) + wait_for_export_status(node, mt_table, s3_table, partition_id, "COMPLETED", timeout=60) + + # A partition whose rows map to several partitions of the destination cannot be exported. + bad_s3 = f"export_all_modes_bad_s3_{uid}" + node.query( + f"CREATE TABLE {bad_s3} (id UInt64, year UInt16)" + f" ENGINE = S3(s3_conn, filename='{bad_s3}', format=Parquet, partition_strategy='hive')" + f" PARTITION BY id" + ) + node.query(f"INSERT INTO {mt_table} VALUES (4, 2023), (5, 2023)") + + # throw_first (default): aborts on the first partition that fails. error = node.query_and_get_error( - f"ALTER TABLE {mt_table} EXPORT PARTITION ALL TO TABLE {s3_table}" + f"ALTER TABLE {mt_table} EXPORT PARTITION ALL TO TABLE {bad_s3}" f" SETTINGS export_merge_tree_partition_all_on_error = 'throw_first'" ) - assert "EXPORT_PARTITION_ALREADY_EXPORTED" in error, ( - f"Expected EXPORT_PARTITION_ALREADY_EXPORTED in error, got: {error}" - ) + assert error, "Expected the export of an incompatible partition to fail" - # collect: aggregated PARTITION_EXPORT_FAILED message lists every conflicting partition. + # collect: the aggregated PARTITION_EXPORT_FAILED message lists the failing partition. error = node.query_and_get_error( - f"ALTER TABLE {mt_table} EXPORT PARTITION ALL TO TABLE {s3_table}" + f"ALTER TABLE {mt_table} EXPORT PARTITION ALL TO TABLE {bad_s3}" f" SETTINGS export_merge_tree_partition_all_on_error = 'collect'" ) - assert "PARTITION_EXPORT_FAILED" in error, ( - f"Expected PARTITION_EXPORT_FAILED in error, got: {error}" - ) - for partition_id in ("2020", "2021", "2022"): - assert partition_id in error, ( - f"Expected aggregated error to mention partition {partition_id}, got: {error}" - ) - - # skip_conflicts: succeeds silently because every partition conflicts and is skipped. - node.query( - f"ALTER TABLE {mt_table} EXPORT PARTITION ALL TO TABLE {s3_table}" - f" SETTINGS export_merge_tree_partition_all_on_error = 'skip_conflicts'" + assert "PARTITION_EXPORT_FAILED" in error and "2023" in error, ( + f"Expected PARTITION_EXPORT_FAILED mentioning partition 2023, got: {error}" ) @@ -900,7 +891,7 @@ def test_export_partition_with_a_fully_deleted_part(cluster, source_engine): exported_files = node.query( f""" SELECT length(arrayFlatten(mapValues(destination_file_paths))) - FROM system.partition_exports + FROM system.distributed_exports WHERE source_table = '{mt_table}' AND destination_table = '{s3_table}' AND partition_id = '2020' @@ -943,7 +934,7 @@ def test_export_partition_where_every_row_is_deleted(cluster, source_engine): exported_files = node.query( f""" SELECT length(arrayFlatten(mapValues(destination_file_paths))) - FROM system.partition_exports + FROM system.distributed_exports WHERE source_table = '{mt_table}' AND destination_table = '{s3_table}' AND partition_id = '2020' diff --git a/tests/integration/test_export_partition_to_object_storage/test_validation.py b/tests/integration/test_export_partition_to_object_storage/test_validation.py index 1e40a46a6de3..d9f8874d8ec4 100644 --- a/tests/integration/test_export_partition_to_object_storage/test_validation.py +++ b/tests/integration/test_export_partition_to_object_storage/test_validation.py @@ -50,7 +50,7 @@ def test_export_partition_partition_column_castable_type_mismatch(cluster, sourc # With a String partition column the partition_id is the SipHash of the # value rather than the textual representation — look it up so we can # reference the partition explicitly in EXPORT PARTITION ID and in - # subsequent system.partition_exports queries. + # subsequent system.distributed_exports queries. partition_id = node.query( f"SELECT partition_id FROM system.parts " f"WHERE database = currentDatabase() AND table = '{mt_table}' " @@ -75,15 +75,15 @@ def test_export_partition_partition_column_castable_type_mismatch(cluster, sourc f"'year', got: {error!r}" ) - # Nothing scheduled: no row in system.partition_exports. + # Nothing scheduled: no row in system.distributed_exports. rows_in_system_view = node.query( - f"SELECT count() FROM system.partition_exports " + f"SELECT count() FROM system.distributed_exports " f"WHERE source_table = '{mt_table}' " f" AND destination_table = '{s3_table}' " f" AND partition_id = '{partition_id}'" ).strip() assert rows_in_system_view == "0", ( - f"Expected no row in system.partition_exports after a " + f"Expected no row in system.distributed_exports after a " f"synchronously-rejected export, got {rows_in_system_view}." ) @@ -210,7 +210,7 @@ def test_export_partition_coarser_source_rejected(cluster, source_engine): assert "BAD_ARGUMENTS" in error, f"expected BAD_ARGUMENTS, got: {error!r}" scheduled = node.query( - f"SELECT count() FROM system.partition_exports" + f"SELECT count() FROM system.distributed_exports" f" WHERE source_table = '{mt_table}' AND destination_table = '{s3_table}'" ).strip() assert scheduled == "0", f"expected nothing scheduled after a synchronous reject, got {scheduled}" diff --git a/tests/integration/test_export_ttl/__init__.py b/tests/integration/test_export_ttl/__init__.py new file mode 100644 index 000000000000..e69de29bb2d1 diff --git a/tests/integration/test_export_ttl/common.py b/tests/integration/test_export_ttl/common.py new file mode 100644 index 000000000000..6e275cb773f0 --- /dev/null +++ b/tests/integration/test_export_ttl/common.py @@ -0,0 +1,364 @@ +import contextlib +import json +import os +import time + +from helpers.export_partition_helpers import is_replicated_engine, make_iceberg_s3, unique_suffix +from helpers.iceberg_export_stats import fetch_manifest_entries + +# Helpers shared by the `TTL ... EXPORT TO TABLE` modules. + +# The columns of a source and of its Iceberg destination, which has no unsigned types. +COLUMNS = "id Int64, year Int32, t DateTime" + +# Rows whose TTL `t + INTERVAL 1 DAY` is due long ago, and rows that are not due before a test ends. +DUE = "now() - INTERVAL 10 DAY" +NOT_DUE = "now()" + +# A group is shipped on the first check after its parts are due. +FAST_TTL_SETTINGS = { + "ttl_export_check_period_seconds": 1, + "ttl_export_batch_window_seconds": 0, + "ttl_export_batch_max_delay_seconds": 0, +} + +PAUSE_EXPORT_FAILPOINT = "export_part_pause_before_schema_validation" + + +def settings_clause(settings): + return ", ".join(f"{name} = {value!r}" if isinstance(value, str) else f"{name} = {value}" for name, value in settings.items()) + + +def zookeeper_path(table): + return f"/clickhouse/tables/{table}" + + +def engine_clause(engine, table, replica_name="replica1"): + if is_replicated_engine(engine): + return f"ReplicatedMergeTree('{zookeeper_path(table)}', '{replica_name}')" + return "MergeTree()" + + +def source_ddl(table, columns, partition_by, ttl, engine="MergeTree", replica_name="replica1", order_by="tuple()", settings=None): + all_settings = dict(FAST_TTL_SETTINGS) + all_settings.update(settings or {}) + partition = f"PARTITION BY {partition_by}" if partition_by else "" + return ( + f"CREATE TABLE {table} ({columns}) ENGINE = {engine_clause(engine, table, replica_name)}" + f" {partition} ORDER BY {order_by} TTL {ttl} SETTINGS {settings_clause(all_settings)}" + ) + + +def create_source(node, table, columns, partition_by, ttl, **kwargs): + node.query(source_ddl(table, columns, partition_by, ttl, **kwargs)) + + +def create_source_error(node, table, columns, partition_by, ttl, **kwargs): + return node.query_and_get_error(source_ddl(table, columns, partition_by, ttl, **kwargs)) + + +def create_s3_hive(node, table, columns, partition_by): + node.query( + f"CREATE TABLE {table} ({columns})" + f" ENGINE = S3(s3_conn, filename='{table}', format=Parquet, partition_strategy='hive') PARTITION BY {partition_by}" + ) + + +def create_s3_wildcard(node, table, columns, partition_by): + node.query( + f"CREATE TABLE {table} ({columns})" + f" ENGINE = S3(s3_conn, filename='{table}/{{_partition_id}}/{{_file}}.parquet', format=Parquet, partition_strategy='wildcard')" + f" PARTITION BY {partition_by}" + ) + + +def wait_until(predicate, timeout=90, message="The condition did not hold", interval=0.5): + """Poll *predicate* until it returns a truthy value, and return it.""" + start = time.time() + last = None + while time.time() - start < timeout: + last = predicate() + if last: + return last + time.sleep(interval) + raise AssertionError(f"{message} within {timeout} s, last value: {last!r}") + + +def query_json(node, query): + result = node.query(f"{query} FORMAT JSONCompact", settings={"output_format_json_quote_64bit_integers": 0}) + return json.loads(result)["data"] + + +# `system.ttl_exports` + +TTL_STATE_COLUMNS = [ + "partition_id", + "exported_parts", + "claimed_parts", + "eligible_parts", + "parts_held_by_delete_gate", + "current_transaction_id", + "last_error", + "scheduler_replica", +] + + +def ttl_rows(node, table, columns=None): + """The rows of `system.ttl_exports` of *table*, by partition id.""" + columns = columns or TTL_STATE_COLUMNS + rows = query_json(node, f"SELECT {', '.join(columns)} FROM system.ttl_exports WHERE table = '{table}' ORDER BY partition_id") + return {row[0]: dict(zip(columns, row)) for row in rows} + + +def wait_for_same_ttl_rows(replicas, table, settled, timeout=90): + """Wait until every replica shows the same rows of `system.ttl_exports` for *table*, and they + satisfy *settled*.""" + start = time.time() + while True: + rows = [ttl_rows(replica, table) for replica in replicas] + if all(replica_rows == rows[0] for replica_rows in rows) and settled(rows[0]): + return rows[0] + assert time.time() - start < timeout, f"The rows of system.ttl_exports did not settle: {rows}" + time.sleep(0.5) + + +def first_eligible_time(node, table, partition_id): + """When the scheduler first saw an eligible part of the partition that is not exported, 0 if never.""" + return int(node.query( + f"SELECT toUnixTimestamp(first_eligible_time) FROM system.ttl_exports WHERE table = '{table}' AND partition_id = '{partition_id}'" + ).strip() or 0) + + +def partition_settled(rows, partition_id, exported=None): + """Nothing of the partition is claimed or waiting to be exported.""" + row = rows.get(partition_id) + if row is None or row["claimed_parts"] != 0 or row["eligible_parts"] != 0 or row["current_transaction_id"] != "": + return False + return exported is None or row["exported_parts"] == exported + + +def wait_for_partitions_exported(node, table, partition_ids, timeout=90): + """Wait until nothing of *partition_ids* is claimed or eligible and no TTL task is in flight.""" + def settled(): + rows = ttl_rows(node, table) + if any(not partition_settled(rows, partition_id) for partition_id in partition_ids): + return None + return pending_ttl_tasks(node, table) == 0 and rows + + return wait_until(settled, timeout, f"The TTL export of {table} partitions {partition_ids} did not settle") + + +def wait_for_last_error(node, table, partition_id, substring, timeout=90): + def has_error(): + row = ttl_rows(node, table).get(partition_id) + return row if row and substring in row["last_error"] else None + + return wait_until(has_error, timeout, f"No error containing {substring!r} for partition {partition_id} of {table}") + + +# `system.distributed_exports` + +def ttl_tasks(node, table): + rows = query_json( + node, + f"SELECT transaction_id, partition_id, status, parts, retry_of FROM system.distributed_exports" + f" WHERE source_table = '{table}' AND source = 'ttl' ORDER BY create_time, transaction_id", + ) + return [dict(zip(["transaction_id", "partition_id", "status", "parts", "retry_of"], row)) for row in rows] + + +def pending_ttl_tasks(node, table): + return sum(1 for task in ttl_tasks(node, table) if task["status"] == "PENDING") + + +def completed_ttl_tasks(node, table): + return [task for task in ttl_tasks(node, table) if task["status"] == "COMPLETED"] + + +# Destinations + +def committed_s3_files(node, table): + """Names of the data files that the commit files of an object storage destination reference.""" + lines = node.query(f"SELECT line FROM s3(s3_conn, filename='{table}/commit_*', format=LineAsString)").split() + return {os.path.basename(line) for line in lines} + + +def commit_files(node, table): + return int(node.query(f"SELECT uniqExact(_file) FROM s3(s3_conn, filename='{table}/commit_*', format=LineAsString)").strip()) + + +def s3_files(node, table, committed_only=True): + """{partition directory: sorted ids} of the data files of an object storage destination. Readers + must only consider files that a commit file references, so by default the others are ignored.""" + committed = committed_s3_files(node, table) if committed_only else None + result = {} + rows = query_json(node, f"SELECT _path, id FROM s3(s3_conn, filename='{table}/**.parquet', format=Parquet, structure='id UInt64')") + for path, row_id in rows: + if committed is not None and os.path.basename(path) not in committed: + continue + directory = path.split(f"{table}/", 1)[1].rsplit("/", 1)[0] + result.setdefault(directory, []).append(row_id) + return {directory: sorted(ids) for directory, ids in result.items()} + + +def s3_ids(node, table, committed_only=True): + return sorted(row_id for ids in s3_files(node, table, committed_only).values() for row_id in ids) + + +# Snapshots of each Iceberg destination when it was created, see `assert_one_snapshot_per_task`. +_snapshots_at_creation = {} + + +def create_iceberg(nodes, table, columns=COLUMNS, partition_by="year", attach=False): + """An Iceberg destination at the MinIO prefix named after the table. It is created on the first of + *nodes*, and the others attach to the same data. With *attach* the first one attaches too, for + example to the data of a dropped table of the same name, which a `DROP` leaves in place.""" + nodes = list(nodes) if isinstance(nodes, (list, tuple)) else [nodes] + make_iceberg_s3(nodes[0], table, columns, partition_by=partition_by, if_not_exists=attach) + for node in nodes[1:]: + make_iceberg_s3(node, table, columns, partition_by=partition_by, if_not_exists=True) + _snapshots_at_creation[table] = iceberg_snapshots(nodes[0], table) + + +def iceberg_ids(node, table): + return [int(x) for x in node.query(f"SELECT id FROM {table} ORDER BY id").split()] + + +def iceberg_snapshots(node, table): + return int(node.query( + f"SELECT count() FROM system.iceberg_history WHERE database = currentDatabase() AND table = '{table}'" + ).strip()) + + +def assert_one_snapshot_per_task(node, mt_table, iceberg_table): + """Every completed TTL task committed one snapshot since the destination was created, and no + other task committed.""" + snapshots = iceberg_snapshots(node, iceberg_table) - _snapshots_at_creation.get(iceberg_table, 0) + completed = completed_ttl_tasks(node, mt_table) + assert snapshots == len(completed), f"{snapshots} snapshots for {len(completed)} completed tasks: {ttl_tasks(node, mt_table)}" + + +def _partition_scalar(value): + """A partition field value, without the Avro-union `{type: value}` wrapper.""" + if isinstance(value, dict): + assert len(value) == 1, f"Unexpected partition union shape: {value!r}" + value = next(iter(value.values())) + return value + + +def iceberg_data_files(node, table): + """{file name: partition record} of the live data files of an Iceberg table, from its manifests.""" + query_id = f"iceberg_files_{unique_suffix()}" + node.query(f"SELECT * FROM {table}", query_id=query_id, settings={"iceberg_metadata_log_level": "manifest_file_entry"}) + files = {} + for entry in fetch_manifest_entries(node, query_id): + data_file = entry.get("data_file") or {} + if data_file.get("content", 0) not in (0, None) or entry.get("status") == 2: + continue + partition = {name: _partition_scalar(value) for name, value in (data_file.get("partition") or {}).items()} + files[os.path.basename(data_file["file_path"])] = partition + return files + + +def iceberg_orphan_files(node, table): + """Data files under the prefix of an Iceberg table that no manifest references.""" + written = { + os.path.basename(row[0]) + for row in query_json(node, f"SELECT DISTINCT _path FROM s3(s3_conn, filename='{table}/**.parquet', format=Parquet, structure='id Int64')") + } + return written - set(iceberg_data_files(node, table)) + + +def assert_iceberg_files_partitioned(node, table, field, expression): + """Every data file holds rows of one value of *expression*, which is the value of the partition + field *field* recorded for the file in the manifest. Returns the values.""" + files = iceberg_data_files(node, table) + rows = query_json(node, f"SELECT _path, groupUniqArray({expression}) FROM {table} GROUP BY _path") + assert rows, f"{table} has no rows" + values = set() + for path, file_values in rows: + assert len(file_values) == 1, f"The file {path} holds rows of several partitions: {file_values}" + name = os.path.basename(path) + assert name in files, f"The file {name} is not in the manifests: {files}" + recorded = files[name].get(field) + assert recorded is not None and int(recorded) == int(file_values[0]), ( + f"The file {name} holds {expression} = {file_values[0]}, the manifest records {files[name]}" + ) + values.add(int(file_values[0])) + return values + + +def assert_exactly_once(ids, expected): + assert sorted(ids) == sorted(expected), f"Expected exactly {sorted(expected)}, got {sorted(ids)}" + + +# Parts and merges + +def active_parts(node, table, partition_id=None): + condition = f" AND partition_id = '{partition_id}'" if partition_id is not None else "" + return node.query( + f"SELECT name FROM system.parts WHERE database = currentDatabase() AND table = '{table}' AND active{condition} ORDER BY name" + ).split() + + +def merged_parts(node, table): + """Names of the parts of *table* that were the source of a merge.""" + node.query("SYSTEM FLUSH LOGS part_log") + return set(node.query( + f"SELECT arrayJoin(merged_from) FROM system.part_log" + f" WHERE database = currentDatabase() AND table = '{table}' AND event_type = 'MergeParts'" + ).split()) + + +def assert_never_merged(node, table, part_names, seconds=5): + """Give merges *seconds* to happen, asking for them every second, and check that none of + *part_names* was merged.""" + start = time.time() + while time.time() - start < seconds: + node.query(f"OPTIMIZE TABLE {table}") + time.sleep(1) + active = active_parts(node, table) + missing = [name for name in part_names if name not in active] + assert not missing, f"Parts {missing} are no longer active on {node.name}; active parts: {active}" + merged = set(part_names) & merged_parts(node, table) + assert not merged, f"Parts {sorted(merged)} were merged on {node.name}" + + +def optimize_final_error(node, table, partition_id): + return node.query_and_get_error( + f"OPTIMIZE TABLE {table} PARTITION ID '{partition_id}' FINAL", settings={"optimize_throw_if_noop": 1} + ) + + +@contextlib.contextmanager +def group_in_flight(node): + """Pauses the export of the next part on *node*, which keeps its group claimed. Yields a function + that waits until the export is paused.""" + node.query(f"SYSTEM ENABLE FAILPOINT {PAUSE_EXPORT_FAILPOINT}") + try: + yield lambda: node.query(f"SYSTEM WAIT FAILPOINT {PAUSE_EXPORT_FAILPOINT} PAUSE") + finally: + node.query(f"SYSTEM DISABLE FAILPOINT {PAUSE_EXPORT_FAILPOINT}") + + +@contextlib.contextmanager +def failpoint(nodes, name): + for node in nodes: + node.query(f"SYSTEM ENABLE FAILPOINT {name}") + try: + yield + finally: + for node in nodes: + node.query(f"SYSTEM DISABLE FAILPOINT {name}") + + +# Replicas + +def scheduler_holder(node, table): + return node.query( + f"SELECT value FROM system.zookeeper WHERE path = '{zookeeper_path(table)}/export_ttl' AND name = 'scheduler_lock'" + ).strip() + + +def snapshot_refreshes(node): + return int(node.query("SELECT value FROM system.events WHERE event = 'ExportTTLIndexSnapshotRefreshes'").strip() or 0) diff --git a/tests/integration/test_export_ttl/configs/allow_experimental_export_partition.xml b/tests/integration/test_export_ttl/configs/allow_experimental_export_partition.xml new file mode 100644 index 000000000000..514cd710836a --- /dev/null +++ b/tests/integration/test_export_ttl/configs/allow_experimental_export_partition.xml @@ -0,0 +1,3 @@ + + 1 + diff --git a/tests/integration/test_export_ttl/configs/config.d/metadata_log.xml b/tests/integration/test_export_ttl/configs/config.d/metadata_log.xml new file mode 100644 index 000000000000..c1fece21745c --- /dev/null +++ b/tests/integration/test_export_ttl/configs/config.d/metadata_log.xml @@ -0,0 +1,7 @@ + + + system + iceberg_metadata_log
+ 10 +
+
diff --git a/tests/integration/test_export_ttl/configs/named_collections.xml b/tests/integration/test_export_ttl/configs/named_collections.xml new file mode 100644 index 000000000000..d46920b7ba88 --- /dev/null +++ b/tests/integration/test_export_ttl/configs/named_collections.xml @@ -0,0 +1,9 @@ + + + + http://minio1:9001/root/data + minio + ClickHouse_Minio_P@ssw0rd + + + \ No newline at end of file diff --git a/tests/integration/test_export_ttl/configs/users.d/profile.xml b/tests/integration/test_export_ttl/configs/users.d/profile.xml new file mode 100644 index 000000000000..e8c0fcdb33af --- /dev/null +++ b/tests/integration/test_export_ttl/configs/users.d/profile.xml @@ -0,0 +1,26 @@ + + + + 3 + 1 + + + + 3 + 10 + 1 + 1 + + + + 3 + 1 + 1 + + + + 3 + Asia/Tokyo + + + diff --git a/tests/integration/test_export_ttl/conftest.py b/tests/integration/test_export_ttl/conftest.py new file mode 100644 index 000000000000..f402a90f11bc --- /dev/null +++ b/tests/integration/test_export_ttl/conftest.py @@ -0,0 +1,65 @@ +import logging + +import pytest + +from helpers.cluster import ClickHouseCluster +from helpers.export_partition_helpers import SOURCE_ENGINE_IDS, SOURCE_ENGINES + +# `TTL ... EXPORT TO TABLE`, split across several modules so the harness can spread them over +# xdist workers (`--dist=loadfile` assigns a whole module to one worker). +# +# Each module declares the instances it needs in `CLUSTER_INSTANCES` and gets a cluster with only +# those, so modules that never touch a second replica do not start one. + +REPLICA = dict( + main_configs=[ + "configs/named_collections.xml", + "configs/allow_experimental_export_partition.xml", + "configs/config.d/metadata_log.xml", + ], + user_configs=["configs/users.d/profile.xml"], + with_minio=True, + stay_alive=True, + with_zookeeper=True, + keeper_required_feature_flags=["multi_read"], +) + +INSTANCES = {"replica1": REPLICA, "replica2": REPLICA} + + +@pytest.fixture(scope="module") +def cluster(request): + instance_names = getattr(request.module, "CLUSTER_INSTANCES", list(INSTANCES)) + try: + cluster = ClickHouseCluster(__file__) + for name in instance_names: + cluster.add_instance(name, **INSTANCES[name]) + logging.info("Starting cluster with instances %s...", instance_names) + cluster.start() + yield cluster + finally: + cluster.shutdown() + + +@pytest.fixture(autouse=True) +def drop_tables_after_test(cluster): + """Drop every table of the default database after each test, so the TTL schedulers of finished + tests stop running and do not disturb the next ones.""" + yield + for instance_name, instance in cluster.instances.items(): + try: + instance.query("SYSTEM DISABLE FAILPOINT export_part_pause_before_schema_validation") + tables = instance.query( + "SELECT name FROM system.tables WHERE database = 'default' FORMAT TabSeparated" + ).split() + if tables: + instance.query("".join(f"DROP TABLE IF EXISTS default.`{table}` SYNC;" for table in tables)) + except Exception as e: + logging.warning(f"drop_tables_after_test: cleanup failed on {instance_name}: {e}") + + +@pytest.fixture(params=SOURCE_ENGINES, ids=SOURCE_ENGINE_IDS) +def source_engine(request): + """The MergeTree flavour of the source table. A test that requests this fixture runs once per + engine; scenarios that need several replicas do not request it.""" + return request.param diff --git a/tests/integration/test_export_ttl/test_delete_gate.py b/tests/integration/test_export_ttl/test_delete_gate.py new file mode 100644 index 000000000000..f258e51edcb2 --- /dev/null +++ b/tests/integration/test_export_ttl/test_delete_gate.py @@ -0,0 +1,144 @@ +import time + +from helpers.export_partition_helpers import unique_suffix + +from .common import ( + COLUMNS, + DUE, + assert_exactly_once, + assert_one_snapshot_per_task, + create_iceberg, + create_source, + iceberg_ids, + ttl_rows, + wait_for_partitions_exported, + wait_until, +) + +CLUSTER_INSTANCES = ["replica1"] + +# While a table has an `EXPORT` TTL, TTL that deletes or rewrites rows waits until the part is +# exported, so no row is lost before it reaches the Iceberg destination. + + +def make_tables(node, engine, extra_ttl, columns=COLUMNS, destination_columns=None, order_by="tuple()"): + suffix = unique_suffix() + mt_table, iceberg_table = f"gate_mt_{suffix}", f"gate_iceberg_{suffix}" + create_iceberg(node, iceberg_table, columns=destination_columns or columns) + create_source( + node, mt_table, columns, "year", f"t + INTERVAL 1 DAY EXPORT TO TABLE {iceberg_table}{extra_ttl}", + engine=engine, order_by=order_by, + ) + return mt_table, iceberg_table + + +def held_parts_metric(node): + return int(node.query("SELECT value FROM system.metrics WHERE metric = 'ExportTTLPartsHeldByDeleteGate'").strip() or 0) + + +def try_to_apply_ttl(node, table): + node.query(f"OPTIMIZE TABLE {table} FINAL") + node.query(f"ALTER TABLE {table} MATERIALIZE TTL", settings={"mutations_sync": 2}) + + +def wait_for_ttl_applied(node, table, query, expected, timeout=60): + def applied(): + node.query(f"OPTIMIZE TABLE {table} FINAL") + return node.query(query) == expected + + wait_until(applied, timeout, f"The TTL was not applied to {table} after the export", interval=1) + + +def test_delete_waits_for_the_export(cluster, source_engine): + node = cluster.instances["replica1"] + mt_table, iceberg_table = make_tables(node, source_engine, ", t + INTERVAL 2 DAY DELETE") + + # Stopped moves pause the TTL export. + node.query(f"SYSTEM STOP MOVES {mt_table}") + node.query(f"INSERT INTO {mt_table} VALUES (1, 2020, {DUE}), (2, 2020, {DUE})") + try_to_apply_ttl(node, mt_table) + assert node.query(f"SELECT count() FROM {mt_table}") == "2\n", "Rows were deleted before they were exported" + + wait_until(lambda: ttl_rows(node, mt_table).get("2020", {}).get("parts_held_by_delete_gate") == 1, 30, + "The held part is not shown") + assert held_parts_metric(node) == 1 + + node.query(f"SYSTEM START MOVES {mt_table}") + wait_for_partitions_exported(node, mt_table, ["2020"]) + assert_exactly_once(iceberg_ids(node, iceberg_table), [1, 2]) + + wait_for_ttl_applied(node, mt_table, f"SELECT count() FROM {mt_table}", "0\n") + wait_until(lambda: held_parts_metric(node) == 0, 30, "The metric still counts held parts") + assert_exactly_once(iceberg_ids(node, iceberg_table), [1, 2]) + assert_one_snapshot_per_task(node, mt_table, iceberg_table) + + +def test_delete_where_waits_for_the_export(cluster, source_engine): + node = cluster.instances["replica1"] + mt_table, iceberg_table = make_tables(node, source_engine, ", t + INTERVAL 2 DAY DELETE WHERE id % 2 = 0") + + node.query(f"SYSTEM STOP MOVES {mt_table}") + node.query(f"INSERT INTO {mt_table} SELECT number, 2020, {DUE} FROM numbers(4)") + try_to_apply_ttl(node, mt_table) + assert node.query(f"SELECT count() FROM {mt_table}") == "4\n" + + node.query(f"SYSTEM START MOVES {mt_table}") + wait_for_partitions_exported(node, mt_table, ["2020"]) + wait_for_ttl_applied(node, mt_table, f"SELECT groupArray(id) FROM (SELECT id FROM {mt_table} ORDER BY id)", "[1,3]\n") + assert_exactly_once(iceberg_ids(node, iceberg_table), range(4)) + assert_one_snapshot_per_task(node, mt_table, iceberg_table) + + +def test_group_by_waits_for_the_export(cluster, source_engine): + node = cluster.instances["replica1"] + mt_table, iceberg_table = make_tables( + node, source_engine, ", t + INTERVAL 2 DAY GROUP BY year SET id = max(id)", order_by="year" + ) + + node.query(f"SYSTEM STOP MOVES {mt_table}") + node.query(f"INSERT INTO {mt_table} SELECT number, 2020, {DUE} FROM numbers(3)") + try_to_apply_ttl(node, mt_table) + assert node.query(f"SELECT count() FROM {mt_table}") == "3\n", "Rows were rolled up before they were exported" + + node.query(f"SYSTEM START MOVES {mt_table}") + wait_for_partitions_exported(node, mt_table, ["2020"]) + wait_for_ttl_applied(node, mt_table, f"SELECT groupArray(id) FROM {mt_table}", "[2]\n") + assert_exactly_once(iceberg_ids(node, iceberg_table), range(3)) + assert_one_snapshot_per_task(node, mt_table, iceberg_table) + + +def test_column_ttl_waits_for_the_export(cluster, source_engine): + node = cluster.instances["replica1"] + mt_table, iceberg_table = make_tables( + node, source_engine, "", + columns="id Int64, year Int32, t DateTime, v String TTL t + INTERVAL 2 DAY", + destination_columns="id Int64, year Int32, t DateTime, v String", + ) + + node.query(f"SYSTEM STOP MOVES {mt_table}") + node.query(f"INSERT INTO {mt_table} VALUES (1, 2020, {DUE}, 'kept')") + try_to_apply_ttl(node, mt_table) + assert node.query(f"SELECT v FROM {mt_table}") == "kept\n", "The column was reset before it was exported" + + node.query(f"SYSTEM START MOVES {mt_table}") + wait_for_partitions_exported(node, mt_table, ["2020"]) + wait_for_ttl_applied(node, mt_table, f"SELECT v FROM {mt_table}", "\n") + assert node.query(f"SELECT v FROM {iceberg_table}") == "kept\n" + assert_one_snapshot_per_task(node, mt_table, iceberg_table) + + +def test_removing_the_export_releases_held_rows(cluster, source_engine): + """Rows held for the export are deleted as soon as the `EXPORT` TTL is removed, without being exported.""" + node = cluster.instances["replica1"] + mt_table, iceberg_table = make_tables(node, source_engine, ", t + INTERVAL 2 DAY DELETE") + + node.query(f"SYSTEM STOP MOVES {mt_table}") + node.query(f"INSERT INTO {mt_table} VALUES (1, 2020, {DUE})") + try_to_apply_ttl(node, mt_table) + assert node.query(f"SELECT count() FROM {mt_table}") == "1\n" + + node.query(f"ALTER TABLE {mt_table} MODIFY TTL t + INTERVAL 2 DAY DELETE", settings={"mutations_sync": 2}) + wait_for_ttl_applied(node, mt_table, f"SELECT count() FROM {mt_table}", "0\n") + time.sleep(2) + assert iceberg_ids(node, iceberg_table) == [] + assert_one_snapshot_per_task(node, mt_table, iceberg_table) diff --git a/tests/integration/test_export_ttl/test_failures.py b/tests/integration/test_export_ttl/test_failures.py new file mode 100644 index 000000000000..32fe1f88d36b --- /dev/null +++ b/tests/integration/test_export_ttl/test_failures.py @@ -0,0 +1,236 @@ +import time + +from helpers.export_partition_helpers import unique_suffix + +from .common import ( + COLUMNS, + DUE, + assert_exactly_once, + assert_one_snapshot_per_task, + completed_ttl_tasks, + create_iceberg, + create_source, + failpoint, + group_in_flight, + iceberg_ids, + iceberg_orphan_files, + iceberg_snapshots, + ttl_rows, + ttl_tasks, + wait_for_last_error, + wait_for_partitions_exported, + wait_until, +) + +CLUSTER_INSTANCES = ["replica1"] + +# Failures of the tasks of the `EXPORT` TTL to an Iceberg destination: a failed group is retried as a +# new task that lists the failed ones in `retry_of`, only the task that completes commits a snapshot, +# and no row lands twice, whatever fails and when. + + +def make_tables(node, engine, settings=None, columns=COLUMNS, source_partition_by="year", spec="year"): + suffix = unique_suffix() + mt_table, iceberg_table = f"fail_mt_{suffix}", f"fail_iceberg_{suffix}" + create_iceberg(node, iceberg_table, columns=columns, partition_by=spec) + create_source( + node, mt_table, columns, source_partition_by, f"t + INTERVAL 1 DAY EXPORT TO TABLE {iceberg_table}", + engine=engine, settings=settings, + ) + return mt_table, iceberg_table + + +def exception_count(node, transaction_id): + return int(node.query( + f"SELECT sum(exception_count) FROM system.distributed_exports WHERE transaction_id = '{transaction_id}'" + ).strip() or 0) + + +def test_retryable_part_error_is_retried_by_the_same_task(cluster, source_engine): + node = cluster.instances["replica1"] + mt_table, iceberg_table = make_tables(node, source_engine, settings={"ttl_export_settings_profile": "ttl_export_quick_retry"}) + + with failpoint([node], "export_part_retryable_throw"): + node.query(f"INSERT INTO {mt_table} VALUES (1, 2020, {DUE})") + tasks = wait_until(lambda: ttl_tasks(node, mt_table), 60, "No task was started") + transaction_id = tasks[0]["transaction_id"] + wait_until(lambda: exception_count(node, transaction_id) > 0, 60, "The part did not fail") + assert iceberg_ids(node, iceberg_table) == [] + + wait_for_partitions_exported(node, mt_table, ["2020"]) + tasks = ttl_tasks(node, mt_table) + assert [(task["transaction_id"], task["status"]) for task in tasks] == [(transaction_id, "COMPLETED")], tasks + assert_exactly_once(iceberg_ids(node, iceberg_table), [1]) + assert_one_snapshot_per_task(node, mt_table, iceberg_table) + + +def test_non_retryable_part_error_is_retried_as_a_new_task(cluster, source_engine): + node = cluster.instances["replica1"] + mt_table, iceberg_table = make_tables(node, source_engine) + + with failpoint([node], "export_part_non_retryable_throw"): + node.query(f"INSERT INTO {mt_table} VALUES (1, 2020, {DUE})") + wait_until(lambda: any(task["status"] == "FAILED" for task in ttl_tasks(node, mt_table)), 60, "The task did not fail") + # Pause, so no retry fails once the failure is gone. On a replicated table this also pauses + # the part exports of a retry that is already running. + node.query(f"SYSTEM STOP MOVES {mt_table}") + assert iceberg_ids(node, iceberg_table) == [] + + node.query(f"SYSTEM START MOVES {mt_table}") + wait_for_partitions_exported(node, mt_table, ["2020"]) + + tasks = ttl_tasks(node, mt_table) + failed = {task["transaction_id"] for task in tasks if task["status"] == "FAILED"} + completed = completed_ttl_tasks(node, mt_table) + assert failed and len(completed) == 1, tasks + assert set(completed[0]["retry_of"]) == failed, (completed, failed) + assert_exactly_once(iceberg_ids(node, iceberg_table), [1]) + assert_one_snapshot_per_task(node, mt_table, iceberg_table) + + +def test_commit_failure_is_retried_with_the_new_parts(cluster): + """A group whose commit keeps failing times out, and is retried as a new task that also takes the + parts that became due meanwhile, and records the failed task in `retry_of`. Only the retry commits.""" + node = cluster.instances["replica1"] + mt_table, iceberg_table = make_tables(node, "ReplicatedMergeTree", settings={"ttl_export_settings_profile": "ttl_export_fail_fast"}) + node.query(f"SYSTEM STOP MERGES {mt_table}") + + with failpoint([node], "export_partition_commit_always_throw"): + node.query(f"INSERT INTO {mt_table} VALUES (1, 2020, {DUE})") + tasks = wait_until(lambda: ttl_tasks(node, mt_table), 60, "No task was started") + + # Pause the scheduler, so the retry is not started before the next part is due. + node.query(f"SYSTEM STOP MOVES {mt_table}") + failed_transaction_id = tasks[0]["transaction_id"] + wait_until( + lambda: ttl_tasks(node, mt_table)[0]["status"] == "KILLED", 120, "The failing task did not time out" + ) + node.query(f"INSERT INTO {mt_table} VALUES (2, 2020, {DUE})") + assert iceberg_ids(node, iceberg_table) == [] + + node.query(f"SYSTEM START MOVES {mt_table}") + wait_for_partitions_exported(node, mt_table, ["2020"]) + + completed = completed_ttl_tasks(node, mt_table) + assert [(len(task["parts"]), task["retry_of"]) for task in completed] == [(2, [failed_transaction_id])], completed + assert_exactly_once(iceberg_ids(node, iceberg_table), [1, 2]) + assert_one_snapshot_per_task(node, mt_table, iceberg_table) + + +def test_commit_that_landed_is_not_committed_again(cluster): + """The commit to Iceberg lands, but the task is not marked as completed. The retried commit finds + the transaction in the Iceberg metadata, so the rows are committed once.""" + node = cluster.instances["replica1"] + mt_table, iceberg_table = make_tables(node, "ReplicatedMergeTree") + snapshots = iceberg_snapshots(node, iceberg_table) + + with failpoint([node], "iceberg_export_after_commit_before_zk_completed"): + node.query(f"INSERT INTO {mt_table} VALUES (1, 2020, {DUE}), (2, 2020, {DUE})") + wait_for_partitions_exported(node, mt_table, ["2020"], timeout=120) + + assert_exactly_once(iceberg_ids(node, iceberg_table), [1, 2]) + assert iceberg_snapshots(node, iceberg_table) == snapshots + 1 + assert ttl_rows(node, mt_table)["2020"]["exported_parts"] == 1 + assert iceberg_orphan_files(node, iceberg_table) == set() + + +def test_killed_task_is_retried(cluster, source_engine): + """The killed task never commits; the file its part export writes afterwards is not in the table.""" + node = cluster.instances["replica1"] + mt_table, iceberg_table = make_tables(node, source_engine) + + with group_in_flight(node) as wait_paused: + node.query(f"INSERT INTO {mt_table} VALUES (1, 2020, {DUE})") + wait_paused() + killed = ttl_tasks(node, mt_table)[0]["transaction_id"] + node.query(f"KILL EXPORT WHERE transaction_id = '{killed}'") + wait_until(lambda: ttl_tasks(node, mt_table)[0]["status"] == "KILLED", 60, "The task was not killed") + + wait_for_partitions_exported(node, mt_table, ["2020"]) + completed = completed_ttl_tasks(node, mt_table) + assert len(completed) == 1 and completed[0]["retry_of"] == [killed], ttl_tasks(node, mt_table) + assert_exactly_once(iceberg_ids(node, iceberg_table), [1]) + assert_one_snapshot_per_task(node, mt_table, iceberg_table) + + +def test_restart_during_a_group(cluster, source_engine): + node = cluster.instances["replica1"] + mt_table, iceberg_table = make_tables(node, source_engine) + + with group_in_flight(node) as wait_paused: + node.query(f"INSERT INTO {mt_table} VALUES (1, 2020, {DUE}), (2, 2020, {DUE})") + wait_paused() + transaction_id = ttl_tasks(node, mt_table)[0]["transaction_id"] + node.restart_clickhouse(kill=True) + + wait_for_partitions_exported(node, mt_table, ["2020"], timeout=120) + tasks = ttl_tasks(node, mt_table) + assert [(task["transaction_id"], task["status"]) for task in tasks] == [(transaction_id, "COMPLETED")], tasks + assert_exactly_once(iceberg_ids(node, iceberg_table), [1, 2]) + assert_one_snapshot_per_task(node, mt_table, iceberg_table) + + +def test_restart_before_the_first_check(cluster, source_engine): + node = cluster.instances["replica1"] + mt_table, iceberg_table = make_tables(node, source_engine) + + node.query(f"INSERT INTO {mt_table} SELECT number, 2020, {DUE} FROM numbers(1000)") + node.restart_clickhouse() + + wait_for_partitions_exported(node, mt_table, ["2020"]) + assert_exactly_once(iceberg_ids(node, iceberg_table), range(1000)) + assert_one_snapshot_per_task(node, mt_table, iceberg_table) + + +def test_dropped_destination(cluster, source_engine): + """Without its destination the TTL exports nothing and reports why. A plain `MergeTree` table tells + apart a destination created again under the same name, and exports everything to it again: an + Iceberg table created again over the same data has the rows exported before twice. A replicated + table identifies the destination by name, and exports only what it did not export yet.""" + node = cluster.instances["replica1"] + mt_table, iceberg_table = make_tables(node, source_engine) + node.query(f"INSERT INTO {mt_table} VALUES (1, 2020, {DUE})") + wait_for_partitions_exported(node, mt_table, ["2020"]) + snapshots = iceberg_snapshots(node, iceberg_table) + + node.query(f"DROP TABLE {iceberg_table} SYNC") + node.query(f"INSERT INTO {mt_table} VALUES (2, 2020, {DUE})") + wait_for_last_error(node, mt_table, "2020", "does not exist") + time.sleep(2) + assert len(ttl_tasks(node, mt_table)) == 1 + + create_iceberg(node, iceberg_table, attach=True) + wait_until(lambda: len(completed_ttl_tasks(node, mt_table)) == 2, 60, "Nothing was exported to the new destination") + wait_for_partitions_exported(node, mt_table, ["2020"]) + assert ttl_rows(node, mt_table)["2020"]["last_error"] == "" + + exported_again = ttl_tasks(node, mt_table)[-1]["parts"] + if source_engine == "MergeTree": + assert len(exported_again) == 2, ttl_tasks(node, mt_table) + assert_exactly_once(iceberg_ids(node, iceberg_table), [1, 1, 2]) + else: + assert len(exported_again) == 1, ttl_tasks(node, mt_table) + assert_exactly_once(iceberg_ids(node, iceberg_table), [1, 2]) + # One snapshot for the task that exported to the table created again. + assert iceberg_snapshots(node, iceberg_table) == snapshots + 1 + + +def test_partition_failing_its_check_does_not_block_the_others(cluster, source_engine): + """A group whose rows fall into several days of a day-partitioned Iceberg table is not exported, + writes nothing, and does not count against `ttl_export_max_concurrent_groups`.""" + node = cluster.instances["replica1"] + mt_table, iceberg_table = make_tables( + node, source_engine, columns="id Int64, t DateTime", source_partition_by="toYYYYMM(t)", spec="toRelativeDayNum(t)", + settings={"ttl_export_max_concurrent_groups": 1}, + ) + + node.query(f"INSERT INTO {mt_table} VALUES (1, '2020-01-01 10:00:00'), (2, '2020-01-02 10:00:00')") + node.query(f"INSERT INTO {mt_table} VALUES (3, '2020-02-01 10:00:00'), (4, '2020-03-01 10:00:00'), (5, '2020-04-01 10:00:00')") + + wait_for_partitions_exported(node, mt_table, ["202002", "202003", "202004"]) + row = wait_for_last_error(node, mt_table, "202001", "multiple destination partitions") + assert row["eligible_parts"] == 1 and row["claimed_parts"] == 0 and row["current_transaction_id"] == "", row + assert all(task["partition_id"] != "202001" for task in ttl_tasks(node, mt_table)) + assert_exactly_once(iceberg_ids(node, iceberg_table), [3, 4, 5]) + assert_one_snapshot_per_task(node, mt_table, iceberg_table) + assert iceberg_orphan_files(node, iceberg_table) == set(), "A group that failed its check wrote files" diff --git a/tests/integration/test_export_ttl/test_iceberg_commits.py b/tests/integration/test_export_ttl/test_iceberg_commits.py new file mode 100644 index 000000000000..2605db4c1e20 --- /dev/null +++ b/tests/integration/test_export_ttl/test_iceberg_commits.py @@ -0,0 +1,141 @@ +import time + +from helpers.export_partition_helpers import make_iceberg_s3, unique_suffix + +from .common import ( + assert_exactly_once, + completed_ttl_tasks, + create_source, + iceberg_ids, + iceberg_snapshots, + pending_ttl_tasks, + ttl_rows, + ttl_tasks, + wait_for_partitions_exported, + wait_until, +) + +CLUSTER_INSTANCES = ["replica1"] + +# Every group of the `EXPORT` TTL is committed to an Iceberg destination as one transaction, i.e. +# one snapshot, and no row is committed twice. + +# An Iceberg `date` is read back as `Date32`, so the source uses it as well. +COLUMNS = "id Int32, year Int32, d Date32" +DUE = "today() - 10" + + +def make_tables(node, engine, settings=None, columns=COLUMNS): + suffix = unique_suffix() + mt_table, iceberg_table = f"commit_mt_{suffix}", f"commit_iceberg_{suffix}" + make_iceberg_s3(node, iceberg_table, columns, partition_by="year") + create_source( + node, mt_table, columns, "year", f"toDate(d) + INTERVAL 1 DAY EXPORT TO TABLE {iceberg_table}", + engine=engine, settings=settings, + ) + return mt_table, iceberg_table + + +def test_due_parts_are_exported_once(cluster, source_engine): + node = cluster.instances["replica1"] + mt_table, iceberg_table = make_tables(node, source_engine) + + node.query(f"INSERT INTO {mt_table} VALUES (1, 2020, {DUE}), (2, 2020, {DUE})") + node.query(f"INSERT INTO {mt_table} VALUES (3, 2021, today())") + wait_for_partitions_exported(node, mt_table, ["2020"]) + assert node.query(f"SELECT id, year FROM {iceberg_table} ORDER BY id") == "1\t2020\n2\t2020\n" + + node.query(f"INSERT INTO {mt_table} VALUES (4, 2020, {DUE})") + wait_until(lambda: len(completed_ttl_tasks(node, mt_table)) == 2, 60, "The second part was not exported") + wait_for_partitions_exported(node, mt_table, ["2020"]) + assert_exactly_once(iceberg_ids(node, iceberg_table), [1, 2, 4]) + assert node.query(f"SELECT count() FROM {iceberg_table} WHERE year = 2021") == "0\n" + + +def test_group_is_one_snapshot(cluster, source_engine): + node = cluster.instances["replica1"] + mt_table, iceberg_table = make_tables(node, source_engine) + snapshots = iceberg_snapshots(node, iceberg_table) + + node.query(f"SYSTEM STOP MERGES {mt_table}") + node.query(f"SYSTEM STOP MOVES {mt_table}") + for i in range(3): + node.query(f"INSERT INTO {mt_table} VALUES ({i}, 2020, {DUE})") + node.query(f"SYSTEM START MOVES {mt_table}") + wait_for_partitions_exported(node, mt_table, ["2020"]) + + tasks = ttl_tasks(node, mt_table) + assert len(tasks) == 1 and len(tasks[0]["parts"]) == 3, tasks + assert iceberg_snapshots(node, iceberg_table) == snapshots + 1 + assert_exactly_once(iceberg_ids(node, iceberg_table), range(3)) + + +def test_groups_limited_in_size_are_separate_snapshots(cluster, source_engine): + node = cluster.instances["replica1"] + mt_table, iceberg_table = make_tables(node, source_engine, settings={"ttl_export_max_parts_per_group": 2}) + snapshots = iceberg_snapshots(node, iceberg_table) + + node.query(f"SYSTEM STOP MERGES {mt_table}") + node.query(f"SYSTEM STOP MOVES {mt_table}") + for i in range(5): + node.query(f"INSERT INTO {mt_table} VALUES ({i}, 2020, {DUE})") + node.query(f"SYSTEM START MOVES {mt_table}") + + wait_until(lambda: len(completed_ttl_tasks(node, mt_table)) == 3, 120, "The partition was not exported in three groups") + wait_for_partitions_exported(node, mt_table, ["2020"]) + assert sorted(len(task["parts"]) for task in ttl_tasks(node, mt_table)) == [1, 2, 2] + assert iceberg_snapshots(node, iceberg_table) == snapshots + 3 + assert_exactly_once(iceberg_ids(node, iceberg_table), range(5)) + + +def test_concurrent_groups_commit_to_one_table(cluster, source_engine): + node = cluster.instances["replica1"] + mt_table, iceberg_table = make_tables(node, source_engine, settings={"ttl_export_max_concurrent_groups": 4}) + snapshots = iceberg_snapshots(node, iceberg_table) + + node.query(f"SYSTEM STOP MOVES {mt_table}") + node.query(f"INSERT INTO {mt_table} SELECT number, 2020 + number % 4, {DUE} FROM numbers(40)") + node.query(f"SYSTEM START MOVES {mt_table}") + + in_flight = [] + def settled(): + in_flight.append(pending_ttl_tasks(node, mt_table)) + return len(completed_ttl_tasks(node, mt_table)) == 4 + wait_until(settled, 120, "The four partitions were not exported") + wait_for_partitions_exported(node, mt_table, ["2020", "2021", "2022", "2023"]) + + assert iceberg_snapshots(node, iceberg_table) == snapshots + 4 + assert_exactly_once(iceberg_ids(node, iceberg_table), range(40)) + assert node.query(f"SELECT year, count() FROM {iceberg_table} GROUP BY year ORDER BY year") == "".join( + f"{year}\t10\n" for year in range(2020, 2024) + ) + + +def test_destination_schema_change_between_groups(cluster, source_engine): + """A column added to the destination after a group was exported: the next groups are checked + against the new schema, and either export or report why.""" + node = cluster.instances["replica1"] + mt_table, iceberg_table = make_tables(node, source_engine) + node.query(f"INSERT INTO {mt_table} VALUES (1, 2020, {DUE})") + wait_for_partitions_exported(node, mt_table, ["2020"]) + + node.query(f"ALTER TABLE {iceberg_table} ADD COLUMN extra Nullable(String)", settings={"allow_insert_into_iceberg": 1}) + node.query(f"INSERT INTO {mt_table} VALUES (2, 2021, {DUE})") + + def outcome(): + row = ttl_rows(node, mt_table).get("2021") + if row and row["last_error"]: + return "error" + if completed_ttl_tasks(node, mt_table) and len(completed_ttl_tasks(node, mt_table)) == 2: + return "exported" + return None + + result = wait_until(outcome, 60, "The group after the schema change neither exported nor failed") + time.sleep(5) + tasks = ttl_tasks(node, mt_table) + if result == "exported": + assert_exactly_once(iceberg_ids(node, iceberg_table), [1, 2]) + else: + assert_exactly_once(iceberg_ids(node, iceberg_table), [1]) + # A group that cannot be exported must not start a new task on every check. + assert len(tasks) <= 3, f"The failing group started {len(tasks)} tasks: {tasks}" diff --git a/tests/integration/test_export_ttl/test_lifecycle.py b/tests/integration/test_export_ttl/test_lifecycle.py new file mode 100644 index 000000000000..90cd34d79751 --- /dev/null +++ b/tests/integration/test_export_ttl/test_lifecycle.py @@ -0,0 +1,168 @@ +import time + +from helpers.export_partition_helpers import unique_suffix + +from .common import ( + COLUMNS, + DUE, + NOT_DUE, + active_parts, + assert_exactly_once, + assert_one_snapshot_per_task, + completed_ttl_tasks, + create_iceberg, + create_source, + group_in_flight, + iceberg_ids, + ttl_rows, + ttl_tasks, + wait_for_partitions_exported, + wait_until, + zookeeper_path, +) + +CLUSTER_INSTANCES = ["replica1"] + +# Changing, removing and adding back the `EXPORT` TTL expression, and forgetting partitions. + +TTL = "t + INTERVAL 1 DAY" + + +def make_tables(node, engine, settings=None): + suffix = unique_suffix() + mt_table, iceberg_table = f"life_mt_{suffix}", f"life_iceberg_{suffix}" + create_iceberg(node, iceberg_table) + create_source(node, mt_table, COLUMNS, "year", f"{TTL} EXPORT TO TABLE {iceberg_table}", engine=engine, settings=settings) + return mt_table, iceberg_table + + +def wait_for_no_ttl_rows(node, table): + wait_until(lambda: ttl_rows(node, table) == {}, 60, "The export index was not removed") + + +def test_removing_the_ttl_lifts_the_fence(cluster, source_engine): + node = cluster.instances["replica1"] + mt_table, iceberg_table = make_tables(node, source_engine) + node.query(f"SYSTEM STOP MERGES {mt_table}") + + node.query(f"INSERT INTO {mt_table} VALUES (1, 2020, {DUE})") + wait_for_partitions_exported(node, mt_table, ["2020"]) + node.query(f"INSERT INTO {mt_table} VALUES (2, 2020, {NOT_DUE})") + + node.query(f"ALTER TABLE {mt_table} REMOVE TTL") + node.query(f"SYSTEM START MERGES {mt_table}") + + # Once the index of the destination is removed, the exported part merges with the other one. + def merged(): + node.query(f"OPTIMIZE TABLE {mt_table} PARTITION ID '2020' FINAL") + return len(active_parts(node, mt_table)) == 1 + + wait_until(merged, 60, "Parts are still fenced after the EXPORT TTL was removed", interval=1) + wait_for_no_ttl_rows(node, mt_table) + + +def test_changing_the_destination(cluster, source_engine): + """Groups being exported to the previous destination are killed, what was exported to it is + forgotten, and everything due is exported to the new one.""" + node = cluster.instances["replica1"] + mt_table, first_table = make_tables(node, source_engine) + second_table = f"life_iceberg_second_{unique_suffix()}" + create_iceberg(node, second_table) + + node.query(f"INSERT INTO {mt_table} VALUES (1, 2020, {DUE})") + wait_for_partitions_exported(node, mt_table, ["2020"]) + + with group_in_flight(node) as wait_paused: + node.query(f"INSERT INTO {mt_table} VALUES (2, 2021, {DUE})") + wait_paused() + in_flight = [task for task in ttl_tasks(node, mt_table) if task["status"] == "PENDING"] + assert len(in_flight) == 1 + + node.query( + f"ALTER TABLE {mt_table} MODIFY TTL {TTL} EXPORT TO TABLE {second_table}", + settings={"materialize_ttl_after_modify": 0}, + ) + wait_until( + lambda: next(task for task in ttl_tasks(node, mt_table) if task["transaction_id"] == in_flight[0]["transaction_id"])["status"] == "KILLED", + 60, "The task exporting to the previous destination was not killed", + ) + + wait_for_partitions_exported(node, mt_table, ["2020", "2021"]) + wait_until(lambda: iceberg_ids(node, second_table) == [1, 2], 60, "Not everything was exported to the new destination") + assert_exactly_once(iceberg_ids(node, first_table), [1]) + rows = query_destinations(node, mt_table) + assert rows == {second_table}, rows + + +def query_destinations(node, table): + return set(node.query(f"SELECT DISTINCT destination_table FROM system.ttl_exports WHERE table = '{table}'").split()) + + +def test_changing_the_interval_exports_nothing_again(cluster, source_engine): + node = cluster.instances["replica1"] + mt_table, iceberg_table = make_tables(node, source_engine) + node.query(f"INSERT INTO {mt_table} VALUES (1, 2020, {DUE})") + wait_for_partitions_exported(node, mt_table, ["2020"]) + + node.query(f"ALTER TABLE {mt_table} MODIFY TTL t + INTERVAL 2 DAY EXPORT TO TABLE {iceberg_table}", settings={"mutations_sync": 2}) + time.sleep(4) + assert len(ttl_tasks(node, mt_table)) == 1 + assert ttl_rows(node, mt_table)["2020"]["exported_parts"] == 1 + assert_exactly_once(iceberg_ids(node, iceberg_table), [1]) + assert_one_snapshot_per_task(node, mt_table, iceberg_table) + + +def test_adding_the_ttl_back_exports_again(cluster, source_engine): + node = cluster.instances["replica1"] + mt_table, iceberg_table = make_tables(node, source_engine) + node.query(f"INSERT INTO {mt_table} VALUES (1, 2020, {DUE})") + wait_for_partitions_exported(node, mt_table, ["2020"]) + + node.query(f"ALTER TABLE {mt_table} REMOVE TTL") + wait_for_no_ttl_rows(node, mt_table) + + node.query(f"ALTER TABLE {mt_table} MODIFY TTL {TTL} EXPORT TO TABLE {iceberg_table}", settings={"mutations_sync": 2}) + wait_until(lambda: len(completed_ttl_tasks(node, mt_table)) == 2, 60, "The part was not exported again") + wait_for_partitions_exported(node, mt_table, ["2020"]) + assert_exactly_once(iceberg_ids(node, iceberg_table), [1, 1]) + assert_one_snapshot_per_task(node, mt_table, iceberg_table) + + +def test_attached_parts_are_exported_again(cluster, source_engine): + """Parts attached again get new block numbers, so they are not known as exported.""" + node = cluster.instances["replica1"] + mt_table, iceberg_table = make_tables(node, source_engine) + node.query(f"INSERT INTO {mt_table} VALUES (1, 2020, {DUE})") + wait_for_partitions_exported(node, mt_table, ["2020"]) + + node.query(f"ALTER TABLE {mt_table} DETACH PARTITION 2020") + node.query(f"ALTER TABLE {mt_table} ATTACH PARTITION 2020") + wait_until(lambda: len(completed_ttl_tasks(node, mt_table)) == 2, 60, "The attached part was not exported") + wait_for_partitions_exported(node, mt_table, ["2020"]) + assert_exactly_once(iceberg_ids(node, iceberg_table), [1, 1]) + assert_one_snapshot_per_task(node, mt_table, iceberg_table) + + +def test_forget_partition(cluster): + node = cluster.instances["replica1"] + # The cleanup thread removes the dropped parts from memory, after which the partition can be forgotten. + mt_table, iceberg_table = make_tables( + node, "ReplicatedMergeTree", settings={"old_parts_lifetime": 1, "cleanup_delay_period": 1, "max_cleanup_delay_period": 2} + ) + node.query(f"INSERT INTO {mt_table} VALUES (1, 2020, {DUE})") + wait_for_partitions_exported(node, mt_table, ["2020"]) + + index_path = f"{zookeeper_path(mt_table)}/export_ttl" + destination_key = node.query( + f"SELECT name FROM system.zookeeper WHERE path = '{index_path}' AND name != 'scheduler_lock'" + ).strip() + partitions_path = f"{index_path}/{destination_key}/partitions" + assert node.query(f"SELECT name FROM system.zookeeper WHERE path = '{partitions_path}'") == "2020\n" + + node.query(f"ALTER TABLE {mt_table} DROP PARTITION ID '2020'") + wait_until(lambda: active_parts(node, mt_table, "2020") == [] and node.query( + f"SELECT count() FROM system.parts WHERE database = currentDatabase() AND table = '{mt_table}' AND partition_id = '2020'" + ) == "0\n", 120, "The dropped parts were not removed") + + node.query(f"ALTER TABLE {mt_table} FORGET PARTITION ID '2020'") + assert node.query(f"SELECT count() FROM system.zookeeper WHERE path = '{partitions_path}'") == "0\n" diff --git a/tests/integration/test_export_ttl/test_merge_fence.py b/tests/integration/test_export_ttl/test_merge_fence.py new file mode 100644 index 000000000000..fe05d98a33d0 --- /dev/null +++ b/tests/integration/test_export_ttl/test_merge_fence.py @@ -0,0 +1,237 @@ +import time + +from helpers.export_partition_helpers import unique_suffix + +from .common import ( + COLUMNS, + DUE, + NOT_DUE, + active_parts, + assert_exactly_once, + assert_never_merged, + assert_one_snapshot_per_task, + completed_ttl_tasks, + create_iceberg, + create_source, + group_in_flight, + iceberg_ids, + optimize_final_error, + scheduler_holder, + ttl_rows, + ttl_tasks, + wait_for_partitions_exported, + wait_until, +) + +CLUSTER_INSTANCES = ["replica1", "replica2"] + +# Parts in different export states are never merged, so an exported range never covers rows that +# were not exported: a part is exported, being exported (claimed), or not exported, and merges only +# combine parts of the same state, never claimed ones, and never across blocks of another state. +# The destination is an Iceberg table. + + +def due_in(seconds): + return f"now() - INTERVAL 1 DAY + INTERVAL {seconds} SECOND" + + +def make_tables(node, engine, settings=None, columns=COLUMNS): + suffix = unique_suffix() + mt_table, iceberg_table = f"fence_mt_{suffix}", f"fence_iceberg_{suffix}" + create_iceberg(node, iceberg_table, columns=columns) + create_source( + node, mt_table, columns, "year", f"t + INTERVAL 1 DAY EXPORT TO TABLE {iceberg_table}", engine=engine, settings=settings, + ) + return mt_table, iceberg_table + + +def insert_parts(node, table, rows): + """One part per row, with the scheduler paused so they are shipped in one group.""" + node.query(f"SYSTEM STOP MOVES {table}") + for row in rows: + node.query(f"INSERT INTO {table} VALUES {row}") + node.query(f"SYSTEM START MOVES {table}") + + +def test_parts_in_different_states_never_merge(cluster, source_engine): + node = cluster.instances["replica1"] + mt_table, iceberg_table = make_tables(node, source_engine) + node.query(f"SYSTEM STOP MERGES {mt_table}") + + node.query(f"INSERT INTO {mt_table} VALUES (1, 2020, {DUE})") + wait_for_partitions_exported(node, mt_table, ["2020"]) + exported_part = active_parts(node, mt_table)[0] + node.query(f"INSERT INTO {mt_table} VALUES (2, 2020, {NOT_DUE})") + not_exported_part = [name for name in active_parts(node, mt_table) if name != exported_part][0] + + node.query(f"SYSTEM START MERGES {mt_table}") + assert "different export states" in optimize_final_error(node, mt_table, "2020") + assert_never_merged(node, mt_table, [exported_part, not_exported_part]) + assert_exactly_once(iceberg_ids(node, iceberg_table), [1]) + assert_one_snapshot_per_task(node, mt_table, iceberg_table) + + +def test_claimed_parts_never_merge(cluster, source_engine): + """The parts of a group being exported are not merged, not even with each other, until it commits.""" + node = cluster.instances["replica1"] + mt_table, iceberg_table = make_tables(node, source_engine) + node.query(f"SYSTEM STOP MERGES {mt_table}") + + with group_in_flight(node) as wait_paused: + insert_parts(node, mt_table, [f"(1, 2020, {DUE})", f"(2, 2020, {DUE})"]) + wait_paused() + claimed = active_parts(node, mt_table) + assert len(claimed) == 2 + assert ttl_rows(node, mt_table)["2020"]["claimed_parts"] == 2 + + node.query(f"SYSTEM START MERGES {mt_table}") + assert "are being exported" in optimize_final_error(node, mt_table, "2020") + assert_never_merged(node, mt_table, claimed, seconds=6) + assert [task["status"] for task in ttl_tasks(node, mt_table)] == ["PENDING"] + assert iceberg_ids(node, iceberg_table) == [] + + wait_for_partitions_exported(node, mt_table, ["2020"]) + assert_exactly_once(iceberg_ids(node, iceberg_table), [1, 2]) + assert_one_snapshot_per_task(node, mt_table, iceberg_table) + + # Once exported, they are in the same state and merge. + node.query(f"OPTIMIZE TABLE {mt_table} PARTITION ID '2020' FINAL") + assert len(active_parts(node, mt_table)) == 1 + + +def test_parts_do_not_merge_across_exported_blocks(cluster, source_engine): + """Two parts that are not exported, around a dropped exported part, are not merged: the merged + part would cover the exported block and would look exported.""" + node = cluster.instances["replica1"] + mt_table, iceberg_table = make_tables(node, source_engine) + node.query(f"SYSTEM STOP MERGES {mt_table}") + + node.query(f"INSERT INTO {mt_table} VALUES (1, 2020, {due_in(40)})") + node.query(f"INSERT INTO {mt_table} VALUES (2, 2020, {DUE})") + node.query(f"INSERT INTO {mt_table} VALUES (3, 2020, {due_in(40)})") + wait_until(lambda: ttl_rows(node, mt_table).get("2020", {}).get("exported_parts") == 1, 60, "The middle part was not exported") + wait_for_partitions_exported(node, mt_table, ["2020"]) + + first, middle, last = active_parts(node, mt_table) + node.query(f"ALTER TABLE {mt_table} DROP PART '{middle}'") + node.query(f"SYSTEM START MERGES {mt_table}") + + assert "export states than blocks between them" in optimize_final_error(node, mt_table, "2020") + assert_never_merged(node, mt_table, [first, last]) + + # Once due, both are exported. + wait_until(lambda: iceberg_ids(node, iceberg_table) == [1, 2, 3], 90, "The outer parts were not exported") + wait_for_partitions_exported(node, mt_table, ["2020"]) + assert sorted(part for task in ttl_tasks(node, mt_table)[1:] for part in task["parts"]) == [first, last] + assert_exactly_once(iceberg_ids(node, iceberg_table), [1, 2, 3]) + assert_one_snapshot_per_task(node, mt_table, iceberg_table) + + +def test_exported_parts_merge_with_each_other(cluster, source_engine): + node = cluster.instances["replica1"] + mt_table, iceberg_table = make_tables(node, source_engine) + node.query(f"SYSTEM STOP MERGES {mt_table}") + insert_parts(node, mt_table, [f"(1, 2020, {DUE})", f"(2, 2020, {DUE})"]) + wait_for_partitions_exported(node, mt_table, ["2020"]) + assert len(active_parts(node, mt_table)) == 2 + + node.query(f"SYSTEM START MERGES {mt_table}") + node.query(f"OPTIMIZE TABLE {mt_table} PARTITION ID '2020' FINAL") + assert len(active_parts(node, mt_table)) == 1 + + # The merged part is exported, so nothing is exported again. + time.sleep(3) + assert len(ttl_tasks(node, mt_table)) == 1 + assert ttl_rows(node, mt_table)["2020"]["exported_parts"] == 1 + assert_exactly_once(iceberg_ids(node, iceberg_table), [1, 2]) + assert_one_snapshot_per_task(node, mt_table, iceberg_table) + + +def test_eligible_parts_merge_while_the_group_is_collected(cluster, source_engine): + node = cluster.instances["replica1"] + mt_table, iceberg_table = make_tables(node, source_engine, settings={"ttl_export_batch_window_seconds": 8, "ttl_export_batch_max_delay_seconds": 600}) + node.query(f"SYSTEM STOP MERGES {mt_table}") + node.query(f"INSERT INTO {mt_table} VALUES (1, 2020, {DUE})") + node.query(f"INSERT INTO {mt_table} VALUES (2, 2020, {DUE})") + node.query(f"SYSTEM START MERGES {mt_table}") + node.query(f"OPTIMIZE TABLE {mt_table} PARTITION ID '2020' FINAL") + merged = active_parts(node, mt_table) + assert len(merged) == 1 + + wait_for_partitions_exported(node, mt_table, ["2020"]) + tasks = ttl_tasks(node, mt_table) + assert len(tasks) == 1 and tasks[0]["parts"] == merged, tasks + assert_exactly_once(iceberg_ids(node, iceberg_table), [1, 2]) + assert_one_snapshot_per_task(node, mt_table, iceberg_table) + + +def test_every_replica_enforces_the_fence(cluster): + replicas = [cluster.instances["replica1"], cluster.instances["replica2"]] + suffix = unique_suffix() + mt_table, iceberg_table = f"fence_mt_{suffix}", f"fence_iceberg_{suffix}" + create_iceberg(replicas, iceberg_table) + for replica in replicas: + create_source( + replica, mt_table, COLUMNS, "year", f"t + INTERVAL 1 DAY EXPORT TO TABLE {iceberg_table}", + engine="ReplicatedMergeTree", replica_name=replica.name, + ) + replica.query(f"SYSTEM STOP MERGES {mt_table}") + + replicas[0].query(f"INSERT INTO {mt_table} VALUES (1, 2020, {DUE})") + for replica in replicas: + replica.query(f"SYSTEM SYNC REPLICA {mt_table}") + wait_for_partitions_exported(replicas[0], mt_table, ["2020"]) + replicas[0].query(f"INSERT INTO {mt_table} VALUES (2, 2020, {NOT_DUE})") + for replica in replicas: + replica.query(f"SYSTEM SYNC REPLICA {mt_table}") + + parts = active_parts(replicas[0], mt_table) + assert len(parts) == 2 + holder = scheduler_holder(replicas[0], mt_table) + other = next(replica for replica in replicas if replica.name != holder) + + for replica in replicas: + replica.query(f"SYSTEM START MERGES {mt_table}") + assert "different export states" in optimize_final_error(other, mt_table, "2020") + for replica in replicas: + assert_never_merged(replica, mt_table, parts) + assert active_parts(replica, mt_table) == parts + assert_exactly_once(iceberg_ids(replica, iceberg_table), [1]) + assert_one_snapshot_per_task(replicas[0], mt_table, iceberg_table) + + +def test_mutation_keeps_parts_exported(cluster, source_engine): + """A mutation rewrites a part under the same block range, so an exported part stays exported.""" + node = cluster.instances["replica1"] + mt_table, iceberg_table = make_tables(node, source_engine, columns="id Int64, year Int32, t DateTime, v Int32") + + node.query(f"INSERT INTO {mt_table} VALUES (1, 2020, {DUE}, 7)") + wait_for_partitions_exported(node, mt_table, ["2020"]) + before = active_parts(node, mt_table) + + node.query(f"ALTER TABLE {mt_table} UPDATE v = 8 WHERE 1", settings={"mutations_sync": 2}) + assert active_parts(node, mt_table) != before + + time.sleep(3) + assert len(ttl_tasks(node, mt_table)) == 1 + assert ttl_rows(node, mt_table)["2020"]["exported_parts"] == 1 + assert_exactly_once(iceberg_ids(node, iceberg_table), [1]) + assert node.query(f"SELECT v FROM {iceberg_table}") == "7\n", "The destination has the rows as they were exported" + assert_one_snapshot_per_task(node, mt_table, iceberg_table) + + +def test_block_numbers_are_not_reused_after_a_restart(cluster): + """A plain `MergeTree` table allocates block numbers from its parts. The exported blocks of a + dropped partition must not be allocated again, or a new part would look exported.""" + node = cluster.instances["replica1"] + mt_table, iceberg_table = make_tables(node, "MergeTree") + node.query(f"INSERT INTO {mt_table} VALUES (1, 2020, {DUE})") + wait_for_partitions_exported(node, mt_table, ["2020"]) + + node.query(f"ALTER TABLE {mt_table} DROP PARTITION 2020") + node.restart_clickhouse() + + node.query(f"INSERT INTO {mt_table} VALUES (2, 2020, {DUE})") + wait_until(lambda: len(completed_ttl_tasks(node, mt_table)) == 2, 60, "The new part was not exported") + assert_exactly_once(iceberg_ids(node, iceberg_table), [1, 2]) + assert_one_snapshot_per_task(node, mt_table, iceberg_table) diff --git a/tests/integration/test_export_ttl/test_partition_keys_iceberg.py b/tests/integration/test_export_ttl/test_partition_keys_iceberg.py new file mode 100644 index 000000000000..ddbb17935dc3 --- /dev/null +++ b/tests/integration/test_export_ttl/test_partition_keys_iceberg.py @@ -0,0 +1,179 @@ +import pytest + +from helpers.export_partition_helpers import make_iceberg_s3, unique_suffix + +from .common import ( + assert_exactly_once, + assert_iceberg_files_partitioned, + create_source, + create_source_error, + iceberg_ids, + ttl_rows, + ttl_tasks, + wait_for_last_error, + wait_for_partitions_exported, +) + +CLUSTER_INSTANCES = ["replica1"] + +# The partition spec of an Iceberg destination against the partition key of the source. The spec +# is turned into a ClickHouse partition key, which is checked like the key of an object storage +# destination: structurally, per group from the min/max values of its parts, or refused at `CREATE` +# and `ALTER` when it can never hold. Where each file lands is checked against the partition values +# recorded in the Iceberg manifests. + +COLUMNS = "id Int64, t DateTime" + + +def make_tables(node, engine, columns, source_partition_by, spec, settings=None, ttl_column="t"): + suffix = unique_suffix() + mt_table, iceberg_table = f"ice_mt_{suffix}", f"ice_{suffix}" + make_iceberg_s3(node, iceberg_table, columns, partition_by=spec) + create_source( + node, mt_table, columns, source_partition_by, f"{ttl_column} + INTERVAL 1 DAY EXPORT TO TABLE {iceberg_table}", + engine=engine, settings=settings, + ) + return mt_table, iceberg_table + + +def source_partitions(node, table): + return node.query( + f"SELECT DISTINCT partition_id FROM system.parts WHERE database = currentDatabase() AND table = '{table}' AND active" + ).split() + + +REFUSED_SPECS = [ + pytest.param("icebergBucket(8, id)", id="bucket_of_a_column_outside_the_key"), + pytest.param("icebergBucket(8, t)", id="bucket_of_a_key_column"), + pytest.param("id", id="identity_of_a_column_outside_the_key"), +] + + +@pytest.mark.parametrize("spec", REFUSED_SPECS) +def test_spec_that_can_never_hold_is_refused(cluster, source_engine, spec): + node = cluster.instances["replica1"] + suffix = unique_suffix() + mt_table, iceberg_table = f"ice_mt_{suffix}", f"ice_{suffix}" + make_iceberg_s3(node, iceberg_table, COLUMNS, partition_by=spec) + + error = create_source_error(node, mt_table, COLUMNS, "toYYYYMM(t)", f"t + INTERVAL 1 DAY EXPORT TO TABLE {iceberg_table}", engine=source_engine) + assert "BAD_ARGUMENTS" in error and "partition key of the destination" in error, error + + create_source(node, mt_table, COLUMNS, "toYYYYMM(t)", "t + INTERVAL 30 DAY DELETE", engine=source_engine) + error = node.query_and_get_error(f"ALTER TABLE {mt_table} MODIFY TTL t + INTERVAL 1 DAY EXPORT TO TABLE {iceberg_table}") + assert "BAD_ARGUMENTS" in error, error + + +ACCEPTED_SPECS = [ + pytest.param("toYYYYMM(t)", "toYearNumSinceEpoch(t)", id="year"), + pytest.param("toYYYYMM(t)", "toMonthNumSinceEpoch(t)", id="month"), + pytest.param("toYYYYMM(t)", "toRelativeDayNum(t)", id="day"), + pytest.param("toYYYYMM(t)", "", id="unpartitioned"), + pytest.param("id", "id", id="identity"), +] + + +@pytest.mark.parametrize("source_partition_by, spec", ACCEPTED_SPECS) +def test_spec_that_can_hold_is_accepted(cluster, source_engine, source_partition_by, spec): + node = cluster.instances["replica1"] + mt_table, _ = make_tables(node, source_engine, COLUMNS, source_partition_by, spec) + assert "EXPORT TO TABLE" in node.query(f"SELECT create_table_query FROM system.tables WHERE name = '{mt_table}'") + + +def test_month_transform(cluster, source_engine): + node = cluster.instances["replica1"] + mt_table, iceberg_table = make_tables(node, source_engine, COLUMNS, "toYYYYMM(t)", "toMonthNumSinceEpoch(t)") + node.query( + f"INSERT INTO {mt_table} VALUES (1, '2020-01-05 10:00:00'), (2, '2020-01-20 10:00:00')," + f" (3, '2020-02-05 10:00:00'), (4, '2021-03-05 10:00:00')" + ) + wait_for_partitions_exported(node, mt_table, source_partitions(node, mt_table)) + + values = assert_iceberg_files_partitioned(node, iceberg_table, "t", "toMonthNumSinceEpoch(t)") + expected = {int(x) for x in node.query( + "SELECT toMonthNumSinceEpoch(toDateTime(x)) FROM values('x String', '2020-01-05 10:00:00', '2020-02-05 10:00:00', '2021-03-05 10:00:00')" + ).split()} + assert values == expected + assert_exactly_once(iceberg_ids(node, iceberg_table), [1, 2, 3, 4]) + + +def test_day_transform_is_checked_per_group(cluster, source_engine): + node = cluster.instances["replica1"] + mt_table, iceberg_table = make_tables(node, source_engine, COLUMNS, "toYYYYMM(t)", "toRelativeDayNum(t)") + node.query(f"INSERT INTO {mt_table} VALUES (1, '2020-01-05 01:00:00'), (2, '2020-01-05 23:00:00')") + node.query(f"INSERT INTO {mt_table} VALUES (3, '2020-02-05 10:00:00'), (4, '2020-02-06 10:00:00')") + + wait_for_partitions_exported(node, mt_table, ["202001"]) + wait_for_last_error(node, mt_table, "202002", "multiple destination partitions") + assert assert_iceberg_files_partitioned(node, iceberg_table, "t", "toRelativeDayNum(t)") == { + int(node.query("SELECT toRelativeDayNum(toDateTime('2020-01-05 10:00:00'))").strip()) + } + assert_exactly_once(iceberg_ids(node, iceberg_table), [1, 2]) + assert all(task["partition_id"] == "202001" for task in ttl_tasks(node, mt_table)) + + +def test_bucket_transform_matching_the_source_key(cluster, source_engine): + node = cluster.instances["replica1"] + columns = "id Int64, user_id Int64, t DateTime" + mt_table, iceberg_table = make_tables(node, source_engine, columns, "icebergBucket(8, user_id)", "icebergBucket(8, user_id)") + node.query(f"INSERT INTO {mt_table} SELECT number, number * 7, now() - INTERVAL 10 DAY FROM numbers(12)") + wait_for_partitions_exported(node, mt_table, source_partitions(node, mt_table)) + + values = assert_iceberg_files_partitioned(node, iceberg_table, "user_id", "icebergBucket(8, user_id)") + assert values == {int(x) for x in node.query("SELECT DISTINCT icebergBucket(8, number * 7) FROM numbers(12)").split()} + assert_exactly_once(iceberg_ids(node, iceberg_table), range(12)) + + +def test_truncate_transform(cluster, source_engine): + """A source partitioned by tens into a destination truncated to tens always holds; a source + partitioned by hundreds holds only for groups within one ten.""" + node = cluster.instances["replica1"] + columns = "id Int64, k Int64, t DateTime" + mt_table, iceberg_table = make_tables(node, source_engine, columns, "intDiv(k, 10)", "icebergTruncate(10, k)") + node.query(f"INSERT INTO {mt_table} VALUES (1, 11, now() - INTERVAL 10 DAY), (2, 15, now() - INTERVAL 10 DAY), (3, 23, now() - INTERVAL 10 DAY)") + wait_for_partitions_exported(node, mt_table, ["1", "2"]) + assert assert_iceberg_files_partitioned(node, iceberg_table, "k", "icebergTruncate(10, k)") == {10, 20} + assert_exactly_once(iceberg_ids(node, iceberg_table), [1, 2, 3]) + + mt_table, iceberg_table = make_tables(node, source_engine, columns, "intDiv(k, 100)", "icebergTruncate(10, k)") + node.query(f"INSERT INTO {mt_table} VALUES (1, 105, now() - INTERVAL 10 DAY), (2, 107, now() - INTERVAL 10 DAY)") + node.query(f"INSERT INTO {mt_table} VALUES (3, 210, now() - INTERVAL 10 DAY), (4, 250, now() - INTERVAL 10 DAY)") + wait_for_partitions_exported(node, mt_table, ["1"]) + wait_for_last_error(node, mt_table, "2", "multiple destination partitions") + assert assert_iceberg_files_partitioned(node, iceberg_table, "k", "icebergTruncate(10, k)") == {100} + assert_exactly_once(iceberg_ids(node, iceberg_table), [1, 2]) + + +def test_spec_of_several_fields(cluster, source_engine): + node = cluster.instances["replica1"] + columns = "id Int64, year Int32, t DateTime" + mt_table, iceberg_table = make_tables(node, source_engine, columns, "(year, toYYYYMM(t))", "(year, toMonthNumSinceEpoch(t))") + node.query( + f"INSERT INTO {mt_table} VALUES (1, 1, '2020-01-05 10:00:00'), (2, 2, '2020-01-05 10:00:00'), (3, 1, '2020-02-05 10:00:00')" + ) + wait_for_partitions_exported(node, mt_table, source_partitions(node, mt_table)) + + assert assert_iceberg_files_partitioned(node, iceberg_table, "year", "year") == {1, 2} + assert len(assert_iceberg_files_partitioned(node, iceberg_table, "t", "toMonthNumSinceEpoch(t)")) == 2 + assert_exactly_once(iceberg_ids(node, iceberg_table), [1, 2, 3]) + + +@pytest.mark.parametrize("profile, exported", [pytest.param("", True, id="utc"), pytest.param("ttl_export_tokyo", False, id="tokyo")]) +def test_day_transform_uses_the_partition_time_zone(cluster, profile, exported): + """14:00 and 16:00 UTC are on one day in UTC, and on two days in Tokyo. The check of a group uses + the time zone the destination partitions are computed in.""" + node = cluster.instances["replica1"] + mt_table, iceberg_table = make_tables( + node, "MergeTree", COLUMNS, "toYYYYMM(t)", "toRelativeDayNum(t)", + settings={"ttl_export_settings_profile": profile} if profile else None, + ) + node.query( + f"INSERT INTO {mt_table} VALUES (1, toDateTime('2020-01-01 14:00:00', 'UTC')), (2, toDateTime('2020-01-01 16:00:00', 'UTC'))" + ) + if exported: + wait_for_partitions_exported(node, mt_table, ["202001"]) + assert_exactly_once(iceberg_ids(node, iceberg_table), [1, 2]) + else: + wait_for_last_error(node, mt_table, "202001", "multiple destination partitions") + assert ttl_tasks(node, mt_table) == [] + assert ttl_rows(node, mt_table)["202001"]["eligible_parts"] == 1 diff --git a/tests/integration/test_export_ttl/test_partition_keys_object_storage.py b/tests/integration/test_export_ttl/test_partition_keys_object_storage.py new file mode 100644 index 000000000000..f18c583fd92b --- /dev/null +++ b/tests/integration/test_export_ttl/test_partition_keys_object_storage.py @@ -0,0 +1,225 @@ +from typing import NamedTuple, Optional + +import pytest + +from helpers.export_partition_helpers import unique_suffix + +from .common import ( + DUE, + commit_files, + create_s3_hive, + create_s3_wildcard, + create_source, + create_source_error, + s3_files, + ttl_tasks, + wait_for_last_error, + wait_for_partitions_exported, +) + +CLUSTER_INSTANCES = ["replica1"] + +# The partition key of an object storage destination against the partition key of the source. +# +# - Structural: the destination key is a function of the source key, so every group lands in one +# destination partition. +# - Dynamic: the destination key is monotonic in one column of the source key, so it is checked for +# each group, from the min/max values of its parts. +# - Rejected: neither can hold, so the `CREATE` or `ALTER` that adds the expression is refused. + + +class Case(NamedTuple): + columns: str + source_partition_by: str + destination_partition_by: str + wildcard: bool + # One insert per list, one part per source partition. + inserts: list + # Ids per destination directory, and the directory names when they are known. + expected_files: list + expected_directories: Optional[list] = None + ttl_column: str = "t" + + +STRUCTURAL_CASES = [ + pytest.param(Case( + "id UInt64, year UInt16, t DateTime", "year", "year", False, + [f"(1, 2020, {DUE}), (2, 2021, {DUE})"], + [[1], [2]], ["year=2020", "year=2021"], + ), id="identity"), + pytest.param(Case( + "id UInt64, t DateTime", "toYYYYMM(t)", "toYYYYMM(t)", True, + ["(1, '2020-01-05 10:00:00'), (2, '2020-02-05 10:00:00'), (3, '2020-02-06 10:00:00')"], + [[1], [2, 3]], ["202001", "202002"], + ), id="same_expression"), + pytest.param(Case( + "id UInt64, region String, t DateTime", "(region, toYYYYMM(t))", "region", False, + ["(1, 'eu', '2020-01-05 10:00:00'), (2, 'eu', '2020-02-05 10:00:00'), (3, 'us', '2020-01-05 10:00:00')"], + [[1, 2], [3]], ["region=eu", "region=us"], + ), id="subset_of_a_tuple"), + pytest.param(Case( + "id UInt64, a UInt8, b UInt8, t DateTime", "(a, b)", "(b, a)", False, + [f"(1, 1, 2, {DUE}), (2, 1, 3, {DUE})"], + [[1], [2]], ["b=2/a=1", "b=3/a=1"], + ), id="reordered_tuple"), + pytest.param(Case( + "id UInt64, year UInt16, t DateTime", "year", "year % 10", True, + [f"(1, 2020, {DUE}), (2, 2021, {DUE}), (3, 2031, {DUE})"], + [[1], [2, 3]], ["0", "1"], + ), id="function_of_the_key"), + pytest.param(Case( + "id UInt64, year UInt16, t DateTime", "year", "cityHash64(year) % 4", True, + [f"(1, 2020, {DUE}), (2, 2021, {DUE}), (3, 2022, {DUE}), (4, 2023, {DUE})"], + None, + ), id="hash_of_the_key"), + pytest.param(Case( + "id UInt64, year UInt16, t DateTime", "toString(year)", "year", False, + [f"(1, 2020, {DUE}), (2, 2021, {DUE})"], + [[1], [2]], ["year=2020", "year=2021"], + ), id="injective_wrapper"), + pytest.param(Case( + "id UInt64, t DateTime", "toYYYYMM(t)", "toYear(t)", True, + ["(1, '2020-01-05 10:00:00'), (2, '2020-02-05 10:00:00'), (3, '2021-03-05 10:00:00')"], + [[1, 2], [3]], ["2020", "2021"], + ), id="dynamic_month_into_year"), +] + + +def make_tables(node, case, engine): + suffix = unique_suffix() + mt_table, s3_table = f"pkey_mt_{suffix}", f"pkey_s3_{suffix}" + if case.wildcard: + create_s3_wildcard(node, s3_table, case.columns, case.destination_partition_by) + else: + create_s3_hive(node, s3_table, case.columns, case.destination_partition_by) + create_source( + node, mt_table, case.columns, case.source_partition_by, + f"{case.ttl_column} + INTERVAL 1 DAY EXPORT TO TABLE {s3_table}", engine=engine, + ) + return mt_table, s3_table + + +def source_partitions(node, table): + return node.query( + f"SELECT DISTINCT partition_id FROM system.parts WHERE database = currentDatabase() AND table = '{table}' AND active" + ).split() + + +@pytest.mark.parametrize("case", STRUCTURAL_CASES) +def test_every_group_lands_in_one_destination_partition(cluster, source_engine, case): + node = cluster.instances["replica1"] + mt_table, s3_table = make_tables(node, case, source_engine) + for values in case.inserts: + node.query(f"INSERT INTO {mt_table} VALUES {values}") + + wait_for_partitions_exported(node, mt_table, source_partitions(node, mt_table)) + files = s3_files(node, s3_table) + + if case.expected_files is None: + expected = {} + for row_id, year in [(1, 2020), (2, 2021), (3, 2022), (4, 2023)]: + directory = node.query(f"SELECT cityHash64(toUInt16({year})) % 4").strip() + expected.setdefault(directory, []).append(row_id) + assert files == expected, files + else: + assert sorted(files.values()) == sorted(case.expected_files), files + if case.expected_directories is not None: + assert sorted(files) == sorted(case.expected_directories), files + tasks = ttl_tasks(node, mt_table) + assert all(task["status"] == "COMPLETED" for task in tasks) + # Every task committed with a commit file in the directory of the table, also for wildcards. + assert commit_files(node, s3_table) == len(tasks) + + +def test_group_on_one_day_is_exported_to_a_daily_destination(cluster, source_engine): + """A monthly source into a daily destination: a group whose rows are on one day is exported, a + group spanning two days fails its check and is not exported.""" + node = cluster.instances["replica1"] + case = Case("id UInt64, d Date", "toYYYYMM(d)", "d", False, [], [], ttl_column="d") + mt_table, s3_table = make_tables(node, case, source_engine) + + node.query(f"INSERT INTO {mt_table} VALUES (1, '2020-01-01'), (2, '2020-01-01')") + node.query(f"INSERT INTO {mt_table} VALUES (3, '2020-02-01'), (4, '2020-02-02')") + + wait_for_partitions_exported(node, mt_table, ["202001"]) + row = wait_for_last_error(node, mt_table, "202002", "multiple destination partitions") + assert row["eligible_parts"] == 1 and row["claimed_parts"] == 0, row + assert s3_files(node, s3_table) == {"d=2020-01-01": [1, 2]} + assert s3_files(node, s3_table, committed_only=False) == {"d=2020-01-01": [1, 2]}, "A failed group wrote files" + + +def test_integer_ranges_are_checked_per_group(cluster, source_engine): + node = cluster.instances["replica1"] + case = Case("id UInt64, t DateTime", "intDiv(id, 1000)", "intDiv(id, 100)", True, [], []) + mt_table, s3_table = make_tables(node, case, source_engine) + + node.query(f"INSERT INTO {mt_table} VALUES (1000, {DUE}), (1050, {DUE})") + node.query(f"INSERT INTO {mt_table} VALUES (2000, {DUE}), (2150, {DUE})") + + wait_for_partitions_exported(node, mt_table, ["1"]) + wait_for_last_error(node, mt_table, "2", "multiple destination partitions") + assert s3_files(node, s3_table) == {"10": [1000, 1050]} + + +def test_cast_of_the_partition_column_is_checked_per_group(cluster, source_engine): + """The partition column is a `UInt32` in the source and a `UInt64` in the destination: the cast is + monotonic, so the values of a group are cast and checked like the others.""" + node = cluster.instances["replica1"] + suffix = unique_suffix() + mt_table, s3_table = f"pkey_mt_{suffix}", f"pkey_s3_{suffix}" + create_s3_wildcard(node, s3_table, "id UInt64, k UInt64, t DateTime", "intDiv(k, 100)") + create_source( + node, mt_table, "id UInt64, k UInt32, t DateTime", "intDiv(k, 1000)", + f"t + INTERVAL 1 DAY EXPORT TO TABLE {s3_table}", engine=source_engine, + ) + + node.query(f"INSERT INTO {mt_table} VALUES (1, 1000, {DUE}), (2, 1050, {DUE})") + node.query(f"INSERT INTO {mt_table} VALUES (3, 2000, {DUE}), (4, 2150, {DUE})") + + wait_for_partitions_exported(node, mt_table, ["1"]) + wait_for_last_error(node, mt_table, "2", "multiple destination partitions") + assert s3_files(node, s3_table) == {"10": [1, 2]} + + +REJECTED_CASES = [ + pytest.param( + ("id UInt64, year UInt16, t DateTime", "year", "id", False, None), + "not part of the source MergeTree partition key", id="column_outside_the_key", + ), + pytest.param( + ("id UInt64, t DateTime", "toYYYYMM(t)", "toDayOfWeek(t)", True, None), + "is not monotonic", id="not_monotonic", + ), + pytest.param( + ("id UInt64, k Nullable(UInt32), t DateTime", "intDiv(k, 100)", "intDiv(k, 10)", True, {"allow_nullable_key": 1}), + "Nullable", id="nullable_column", + ), + pytest.param( + ("id UInt64, year UInt16, t DateTime", "", "year", False, None), + "not part of the source MergeTree partition key", id="unpartitioned_source", + ), +] + + +@pytest.mark.parametrize("definition, message", REJECTED_CASES) +def test_incompatible_partition_key_is_refused(cluster, source_engine, definition, message): + node = cluster.instances["replica1"] + columns, source_partition_by, destination_partition_by, wildcard, settings = definition + suffix = unique_suffix() + mt_table, s3_table = f"pkey_mt_{suffix}", f"pkey_s3_{suffix}" + if wildcard: + create_s3_wildcard(node, s3_table, columns, destination_partition_by) + else: + create_s3_hive(node, s3_table, columns, destination_partition_by) + + error = create_source_error( + node, mt_table, columns, source_partition_by, f"t + INTERVAL 1 DAY EXPORT TO TABLE {s3_table}", + engine=source_engine, settings=settings, + ) + assert "BAD_ARGUMENTS" in error and message in error, error + + # The same through `ALTER`. + create_source(node, mt_table, columns, source_partition_by, "t + INTERVAL 30 DAY DELETE", engine=source_engine, settings=settings) + error = node.query_and_get_error(f"ALTER TABLE {mt_table} MODIFY TTL t + INTERVAL 1 DAY EXPORT TO TABLE {s3_table}") + assert "BAD_ARGUMENTS" in error and message in error, error + assert "EXPORT TO TABLE" not in node.query(f"SELECT create_table_query FROM system.tables WHERE name = '{mt_table}'") diff --git a/tests/integration/test_export_ttl/test_replication.py b/tests/integration/test_export_ttl/test_replication.py new file mode 100644 index 000000000000..3add6b76de98 --- /dev/null +++ b/tests/integration/test_export_ttl/test_replication.py @@ -0,0 +1,237 @@ +import time + +from helpers.export_partition_helpers import unique_suffix + +from .common import ( + COLUMNS, + DUE, + NOT_DUE, + assert_exactly_once, + assert_one_snapshot_per_task, + completed_ttl_tasks, + create_iceberg, + create_source, + first_eligible_time, + iceberg_ids, + pending_ttl_tasks, + scheduler_holder, + snapshot_refreshes, + ttl_rows, + ttl_tasks, + wait_for_partitions_exported, + wait_for_same_ttl_rows, + wait_until, +) + +CLUSTER_INSTANCES = ["replica1", "replica2"] + +# One replica of a `ReplicatedMergeTree` table schedules the groups of the `EXPORT` TTL, and stores +# its state in Keeper; every replica shows the same `system.ttl_exports`, and another replica takes +# over where the previous one left off. Every replica has the same Iceberg destination. + + +def make_replicated_tables(replicas, columns=COLUMNS, partition_by="year", spec="year", settings=None): + suffix = unique_suffix() + mt_table, iceberg_table = f"repl_mt_{suffix}", f"repl_iceberg_{suffix}" + create_iceberg(replicas, iceberg_table, columns=columns, partition_by=spec) + for replica in replicas: + create_source( + replica, mt_table, columns, partition_by, f"t + INTERVAL 1 DAY EXPORT TO TABLE {iceberg_table}", + engine="ReplicatedMergeTree", replica_name=replica.name, settings=settings, + ) + return mt_table, iceberg_table + + +def sync(replicas, table): + for replica in replicas: + replica.query(f"SYSTEM SYNC REPLICA {table}") + + +def holder_and_other(replicas, table): + name = wait_until(lambda: scheduler_holder(replicas[0], table), 60, "No replica schedules") + holder = next(replica for replica in replicas if replica.name == name) + other = next(replica for replica in replicas if replica.name != name) + return holder, other + + +def test_replicas_export_once(cluster): + replicas = [cluster.instances["replica1"], cluster.instances["replica2"]] + mt_table, iceberg_table = make_replicated_tables(replicas) + + replicas[0].query(f"INSERT INTO {mt_table} VALUES (1, 2020, {DUE})") + replicas[1].query(f"INSERT INTO {mt_table} VALUES (2, 2020, {DUE})") + sync(replicas, mt_table) + + rows = wait_for_same_ttl_rows( + replicas, mt_table, + lambda rows: "2020" in rows and rows["2020"]["claimed_parts"] == 0 and rows["2020"]["eligible_parts"] == 0, + ) + assert rows["2020"]["scheduler_replica"] in ("replica1", "replica2"), rows + + time.sleep(3) + tasks = ttl_tasks(replicas[0], mt_table) + assert tasks and all(task["status"] == "COMPLETED" for task in tasks), tasks + for replica in replicas: + assert ttl_tasks(replica, mt_table) == tasks + assert_exactly_once(iceberg_ids(replica, iceberg_table), [1, 2]) + assert_one_snapshot_per_task(replicas[0], mt_table, iceberg_table) + + +def test_state_is_the_same_on_every_replica(cluster): + """`system.ttl_exports` has the same rows on every replica, including the error of a group, which + only the replica that schedules sees. A day-partitioned Iceberg destination of a monthly source is + accepted, and a group whose rows are on two days fails when it is exported.""" + replicas = [cluster.instances["replica1"], cluster.instances["replica2"]] + mt_table, iceberg_table = make_replicated_tables( + replicas, columns="id Int64, t DateTime", partition_by="toYYYYMM(t)", spec="toRelativeDayNum(t)" + ) + + replicas[0].query(f"INSERT INTO {mt_table} VALUES (1, '2020-01-01 10:00:00'), (2, '2020-01-02 10:00:00')") + replicas[0].query(f"INSERT INTO {mt_table} VALUES (3, '2020-02-01 10:00:00')") + sync(replicas, mt_table) + + def settled(rows): + return ( + "202001" in rows and "202002" in rows + and rows["202001"]["last_error"] != "" and rows["202001"]["eligible_parts"] == 1 + and rows["202002"]["exported_parts"] == 1 and rows["202002"]["eligible_parts"] == 0 + ) + + rows = wait_for_same_ttl_rows(replicas, mt_table, settled) + assert "multiple destination partitions" in rows["202001"]["last_error"], rows + assert rows["202001"]["claimed_parts"] == 0 and rows["202001"]["current_transaction_id"] == "", rows + assert rows["202002"]["last_error"] == "", rows + assert rows["202001"]["scheduler_replica"] in ("replica1", "replica2"), rows + assert rows["202002"]["scheduler_replica"] == rows["202001"]["scheduler_replica"], rows + for replica in replicas: + assert_exactly_once(iceberg_ids(replica, iceberg_table), [3]) + assert_one_snapshot_per_task(replicas[0], mt_table, iceberg_table) + + +def test_index_snapshot_is_cached(cluster): + """The index of the exported parts is read again from Keeper only when it changes: while nothing is + exported, the scheduler and the merge selection of every replica use the cached one.""" + replicas = [cluster.instances["replica1"], cluster.instances["replica2"]] + mt_table, iceberg_table = make_replicated_tables(replicas) + + replicas[0].query(f"INSERT INTO {mt_table} VALUES (1, 2020, {DUE})") + replicas[0].query(f"INSERT INTO {mt_table} VALUES (2, 2021, {NOT_DUE})") + sync(replicas, mt_table) + wait_for_same_ttl_rows( + replicas, mt_table, + lambda rows: "2020" in rows and rows["2020"]["exported_parts"] == 1 and rows["2020"]["claimed_parts"] == 0, + ) + + # Every replica checks once a second, so a counter that grows during a few seconds means the + # index is read again without a change. + for replica in replicas: + start = time.time() + while True: + before = snapshot_refreshes(replica) + time.sleep(5) + after = snapshot_refreshes(replica) + if after == before: + break + assert time.time() - start < 60, f"The index snapshot is refreshed while idle: {before} -> {after} on {replica.name}" + + +def test_failover_resumes_the_batch(cluster): + """The replica that takes over resumes the batching window where the previous one left off, and + nothing is exported twice.""" + replicas = [cluster.instances["replica1"], cluster.instances["replica2"]] + mt_table, iceberg_table = make_replicated_tables( + replicas, settings={"ttl_export_batch_window_seconds": 60, "ttl_export_batch_max_delay_seconds": 600} + ) + holder, other = holder_and_other(replicas, mt_table) + + holder.query(f"INSERT INTO {mt_table} VALUES (1, 2020, {DUE})") + sync(replicas, mt_table) + first_eligible = wait_until(lambda: first_eligible_time(other, mt_table, "2020"), 60, "The batch was not stored") + assert first_eligible == first_eligible_time(holder, mt_table, "2020") + + holder.stop_clickhouse(kill=True) + try: + wait_until(lambda: scheduler_holder(other, mt_table) == other.name, 120, "The other replica did not take over") + wait_until(lambda: ttl_rows(other, mt_table).get("2020", {}).get("scheduler_replica") == other.name, 60, + "The rows do not show the new scheduler") + assert first_eligible_time(other, mt_table, "2020") == first_eligible, "The batch was restarted by the new scheduler" + assert ttl_tasks(other, mt_table) == [] + finally: + holder.start_clickhouse() + + other.query(f"INSERT INTO {mt_table} VALUES (2, 2020, {DUE})") + other.query(f"ALTER TABLE {mt_table} MODIFY SETTING ttl_export_batch_window_seconds = 0") + sync(replicas, mt_table) + wait_for_partitions_exported(other, mt_table, ["2020"], timeout=120) + assert len(completed_ttl_tasks(other, mt_table)) == 1 + for replica in replicas: + assert_exactly_once(iceberg_ids(replica, iceberg_table), [1, 2]) + assert_one_snapshot_per_task(other, mt_table, iceberg_table) + + +def test_detached_scheduler_hands_over(cluster): + replicas = [cluster.instances["replica1"], cluster.instances["replica2"]] + mt_table, iceberg_table = make_replicated_tables(replicas) + holder, other = holder_and_other(replicas, mt_table) + + holder.query(f"DETACH TABLE {mt_table}") + try: + wait_until(lambda: scheduler_holder(other, mt_table) == other.name, 60, "The other replica did not take over") + other.query(f"INSERT INTO {mt_table} VALUES (1, 2020, {DUE})") + wait_for_partitions_exported(other, mt_table, ["2020"]) + finally: + holder.query(f"ATTACH TABLE {mt_table}") + + sync(replicas, mt_table) + wait_for_same_ttl_rows(replicas, mt_table, lambda rows: rows.get("2020", {}).get("exported_parts") == 1) + assert_exactly_once(iceberg_ids(other, iceberg_table), [1]) + assert_one_snapshot_per_task(other, mt_table, iceberg_table) + + +def test_part_waits_until_the_scheduler_fetches_it(cluster): + """The scheduler groups the parts it has, so a part is exported once it fetched it.""" + replicas = [cluster.instances["replica1"], cluster.instances["replica2"]] + mt_table, iceberg_table = make_replicated_tables(replicas) + holder, other = holder_and_other(replicas, mt_table) + + holder.query(f"SYSTEM STOP FETCHES {mt_table}") + other.query(f"INSERT INTO {mt_table} VALUES (1, 2020, {DUE})") + try: + wait_until(lambda: ttl_rows(other, mt_table).get("2020", {}).get("eligible_parts") == 1, 30, + "The other replica does not show the part as eligible") + time.sleep(4) + assert ttl_tasks(other, mt_table) == [] + finally: + holder.query(f"SYSTEM START FETCHES {mt_table}") + + sync(replicas, mt_table) + wait_for_partitions_exported(holder, mt_table, ["2020"]) + assert len(completed_ttl_tasks(holder, mt_table)) == 1 + assert_exactly_once(iceberg_ids(holder, iceberg_table), [1]) + assert_one_snapshot_per_task(holder, mt_table, iceberg_table) + + +def test_stop_moves_pauses_on_the_scheduler_only(cluster): + """`SYSTEM STOP MOVES` pauses the export on the replica that schedules; on another replica it + does not.""" + replicas = [cluster.instances["replica1"], cluster.instances["replica2"]] + mt_table, iceberg_table = make_replicated_tables(replicas) + holder, other = holder_and_other(replicas, mt_table) + + other.query(f"SYSTEM STOP MOVES {mt_table}") + other.query(f"INSERT INTO {mt_table} VALUES (1, 2020, {DUE})") + sync(replicas, mt_table) + wait_for_partitions_exported(holder, mt_table, ["2020"]) + + holder.query(f"SYSTEM STOP MOVES {mt_table}") + holder.query(f"INSERT INTO {mt_table} VALUES (2, 2020, {DUE})") + sync(replicas, mt_table) + time.sleep(4) + assert len(ttl_tasks(holder, mt_table)) == 1 and pending_ttl_tasks(holder, mt_table) == 0 + + holder.query(f"SYSTEM START MOVES {mt_table}") + other.query(f"SYSTEM START MOVES {mt_table}") + wait_until(lambda: len(completed_ttl_tasks(holder, mt_table)) == 2, 60, "The export did not resume") + wait_for_partitions_exported(holder, mt_table, ["2020"]) + assert_exactly_once(iceberg_ids(holder, iceberg_table), [1, 2]) + assert_one_snapshot_per_task(holder, mt_table, iceberg_table) diff --git a/tests/integration/test_export_ttl/test_scheduling.py b/tests/integration/test_export_ttl/test_scheduling.py new file mode 100644 index 000000000000..21c20e08e066 --- /dev/null +++ b/tests/integration/test_export_ttl/test_scheduling.py @@ -0,0 +1,309 @@ +import time + +from helpers.export_partition_helpers import unique_suffix + +from .common import ( + COLUMNS, + DUE, + NOT_DUE, + assert_exactly_once, + assert_iceberg_files_partitioned, + assert_one_snapshot_per_task, + completed_ttl_tasks, + create_iceberg, + create_source, + failpoint, + first_eligible_time, + iceberg_ids, + iceberg_orphan_files, + iceberg_snapshots, + pending_ttl_tasks, + snapshot_refreshes, + ttl_rows, + ttl_tasks, + wait_for_partitions_exported, + wait_until, +) + +CLUSTER_INSTANCES = ["replica1"] + +# When the `EXPORT` TTL ships groups of parts to an Iceberg destination: on the checks after their +# parts are due, batched by the window, the maximum delay and the size threshold, limited in size and +# concurrency, one snapshot per group, and never twice however many checks go by. + + +def due_in(seconds): + """A value of `t` that makes the TTL `t + INTERVAL 1 DAY` due *seconds* after the insert.""" + return f"now() - INTERVAL 1 DAY + INTERVAL {seconds} SECOND" + + +def make_tables(node, engine, settings=None, prefix="sched"): + suffix = unique_suffix() + mt_table, iceberg_table = f"{prefix}_mt_{suffix}", f"{prefix}_iceberg_{suffix}" + create_iceberg(node, iceberg_table) + # No merge is assigned, so parts are shipped as inserted: a replica with stopped merges would + # still be assigned merges that it never runs, and a part that is being merged waits for it. + all_settings = {"max_bytes_to_merge_at_max_space_in_pool": 1} + all_settings.update(settings or {}) + create_source( + node, mt_table, COLUMNS, "year", f"t + INTERVAL 1 DAY EXPORT TO TABLE {iceberg_table}", + engine=engine, settings=all_settings, + ) + return mt_table, iceberg_table + + +def test_exports_due_parts_once(cluster, source_engine): + node = cluster.instances["replica1"] + mt_table, iceberg_table = make_tables(node, source_engine) + + node.query(f"INSERT INTO {mt_table} VALUES (1, 2020, {DUE}), (2, 2020, {DUE})") + node.query(f"INSERT INTO {mt_table} VALUES (3, 2021, {NOT_DUE})") + wait_for_partitions_exported(node, mt_table, ["2020"]) + assert_exactly_once(iceberg_ids(node, iceberg_table), [1, 2]) + assert assert_iceberg_files_partitioned(node, iceberg_table, "year", "year") == {2020} + + # Later checks and later due inserts never export a part again. + node.query(f"INSERT INTO {mt_table} VALUES (4, 2020, {DUE})") + wait_until(lambda: len(completed_ttl_tasks(node, mt_table)) == 2) + wait_for_partitions_exported(node, mt_table, ["2020"]) + time.sleep(3) + assert len(ttl_tasks(node, mt_table)) == 2 + assert_exactly_once(iceberg_ids(node, iceberg_table), [1, 2, 4]) + assert_one_snapshot_per_task(node, mt_table, iceberg_table) + assert iceberg_orphan_files(node, iceberg_table) == set() + + # The part that is not due stays in the source only. + row = ttl_rows(node, mt_table)["2021"] + assert row["eligible_parts"] == 0 and row["exported_parts"] == 0, row + assert [task["retry_of"] for task in ttl_tasks(node, mt_table)] == [[], []] + + +def test_part_becoming_due_wakes_the_scheduler(cluster, source_engine): + """A check computes when the next part becomes due, and the scheduler wakes up then instead of + waiting for the check period.""" + node = cluster.instances["replica1"] + mt_table, iceberg_table = make_tables(node, source_engine, settings={"ttl_export_check_period_seconds": 120}) + + node.query(f"INSERT INTO {mt_table} VALUES (1, 2020, {due_in(12)})") + inserted = time.time() + # The table starts with a check, which sees the part that is not due yet. + node.query(f"DETACH TABLE {mt_table}") + node.query(f"ATTACH TABLE {mt_table}") + + while time.time() - inserted < 9: + assert ttl_tasks(node, mt_table) == [], "The part was exported before it was due" + time.sleep(1) + + wait_for_partitions_exported(node, mt_table, ["2020"], timeout=40) + assert time.time() - inserted < 40, "The part was exported only by a periodic check" + assert_exactly_once(iceberg_ids(node, iceberg_table), [1]) + assert_one_snapshot_per_task(node, mt_table, iceberg_table) + + +def test_batching_window_spans_checks(cluster, source_engine): + """Parts that keep becoming due within the window are collected over several checks into one + group, which is committed as one snapshot once no new part came for the window.""" + node = cluster.instances["replica1"] + mt_table, iceberg_table = make_tables( + node, source_engine, + settings={"ttl_export_batch_window_seconds": 45, "ttl_export_batch_max_delay_seconds": 600, "ttl_export_batch_min_bytes": 0}, + ) + + first_eligible = set() + for i in range(5): + node.query(f"INSERT INTO {mt_table} VALUES ({i}, 2020, {DUE})") + assert ttl_tasks(node, mt_table) == [], "A group was shipped while parts kept coming" + first_eligible.add(first_eligible_time(node, mt_table, "2020")) + first_eligible.discard(0) + + assert len(first_eligible) == 1, f"The first eligible time of the group changed across checks: {first_eligible}" + + wait_for_partitions_exported(node, mt_table, ["2020"], timeout=120) + tasks = ttl_tasks(node, mt_table) + assert len(tasks) == 1 and len(tasks[0]["parts"]) == 5, tasks + assert_exactly_once(iceberg_ids(node, iceberg_table), range(5)) + assert_one_snapshot_per_task(node, mt_table, iceberg_table) + + +def test_maximum_delay_ships_while_parts_keep_coming(cluster, source_engine): + node = cluster.instances["replica1"] + mt_table, iceberg_table = make_tables( + node, source_engine, + settings={"ttl_export_batch_window_seconds": 15, "ttl_export_batch_max_delay_seconds": 25, "ttl_export_batch_min_bytes": 0}, + ) + + start = time.time() + shipped_at = None + i = 0 + while time.time() - start < 45: + node.query(f"INSERT INTO {mt_table} VALUES ({i}, 2020, {DUE})") + i += 1 + if shipped_at is None and ttl_tasks(node, mt_table): + shipped_at = time.time() - start + + assert shipped_at is not None, "No group was shipped although parts waited longer than the maximum delay" + assert shipped_at >= 20, f"A group was shipped after {shipped_at:.1f} s, before the maximum delay" + + wait_for_partitions_exported(node, mt_table, ["2020"], timeout=90) + assert_exactly_once(iceberg_ids(node, iceberg_table), range(i)) + assert_one_snapshot_per_task(node, mt_table, iceberg_table) + + +def test_size_threshold_ships_before_the_window(cluster, source_engine): + node = cluster.instances["replica1"] + mt_table, iceberg_table = make_tables( + node, source_engine, + settings={"ttl_export_batch_window_seconds": 600, "ttl_export_batch_max_delay_seconds": 600, "ttl_export_batch_min_bytes": 1}, + ) + node.query(f"INSERT INTO {mt_table} VALUES (1, 2020, {DUE})") + wait_for_partitions_exported(node, mt_table, ["2020"], timeout=30) + assert_exactly_once(iceberg_ids(node, iceberg_table), [1]) + assert_one_snapshot_per_task(node, mt_table, iceberg_table) + + +def test_group_size_limits_over_successive_checks(cluster, source_engine): + """A group is limited to `ttl_export_max_parts_per_group` parts; the rest of the partition is + shipped by the next groups, one at a time, each its own snapshot.""" + node = cluster.instances["replica1"] + mt_table, iceberg_table = make_tables( + node, source_engine, + settings={"ttl_export_batch_window_seconds": 15, "ttl_export_batch_max_delay_seconds": 600, + "ttl_export_batch_min_bytes": 0, "ttl_export_max_parts_per_group": 2}, + ) + + for i in range(3): + node.query(f"INSERT INTO {mt_table} VALUES ({i}, 2020, {DUE})") + + # Within the window nothing is exported. + time.sleep(4) + assert ttl_tasks(node, mt_table) == [] + assert ttl_rows(node, mt_table)["2020"]["eligible_parts"] == 3 + + in_flight = [] + def settled(): + in_flight.append(pending_ttl_tasks(node, mt_table)) + return len(completed_ttl_tasks(node, mt_table)) == 2 + wait_until(settled, 120, "The partition was not exported in two groups") + assert max(in_flight) <= 1, f"Two groups of one partition were in flight at once: {in_flight}" + + assert sorted(len(task["parts"]) for task in ttl_tasks(node, mt_table)) == [1, 2] + assert_exactly_once(iceberg_ids(node, iceberg_table), range(3)) + assert_one_snapshot_per_task(node, mt_table, iceberg_table) + + +def test_group_bytes_limit_ships_parts_one_by_one(cluster, source_engine): + node = cluster.instances["replica1"] + mt_table, iceberg_table = make_tables(node, source_engine, settings={"ttl_export_max_bytes_per_group": 1}) + node.query(f"SYSTEM STOP MOVES {mt_table}") + for i in range(3): + node.query(f"INSERT INTO {mt_table} VALUES ({i}, 2020, {DUE})") + node.query(f"SYSTEM START MOVES {mt_table}") + + wait_until(lambda: len(completed_ttl_tasks(node, mt_table)) == 3, 120, "The parts were not exported one by one") + assert [len(task["parts"]) for task in ttl_tasks(node, mt_table)] == [1, 1, 1] + assert_exactly_once(iceberg_ids(node, iceberg_table), range(3)) + assert_one_snapshot_per_task(node, mt_table, iceberg_table) + + +def test_concurrent_groups_are_capped(cluster, source_engine): + """At most `ttl_export_max_concurrent_groups` groups of the table are in flight, whatever the + number of partitions, and their commits to the one Iceberg table do not get lost.""" + node = cluster.instances["replica1"] + mt_table, iceberg_table = make_tables( + node, source_engine, + settings={"ttl_export_max_concurrent_groups": 2, "ttl_export_settings_profile": "ttl_export_quick_retry"}, + ) + + with failpoint([node], "export_part_retryable_throw"): + node.query(f"INSERT INTO {mt_table} VALUES (1, 2020, {DUE}), (2, 2021, {DUE}), (3, 2022, {DUE}), (4, 2023, {DUE})") + wait_until(lambda: pending_ttl_tasks(node, mt_table) == 2, 60, "Two groups were not started") + samples = [] + for _ in range(8): + samples.append(pending_ttl_tasks(node, mt_table)) + time.sleep(0.5) + assert max(samples) == 2, f"More groups than the limit were in flight: {samples}" + assert len(ttl_tasks(node, mt_table)) == 2, "A group was started while the limit was reached" + assert iceberg_ids(node, iceberg_table) == [] + + wait_for_partitions_exported(node, mt_table, ["2020", "2021", "2022", "2023"], timeout=120) + assert len(completed_ttl_tasks(node, mt_table)) == 4 + assert_exactly_once(iceberg_ids(node, iceberg_table), [1, 2, 3, 4]) + assert_one_snapshot_per_task(node, mt_table, iceberg_table) + assert assert_iceberg_files_partitioned(node, iceberg_table, "year", "year") == {2020, 2021, 2022, 2023} + + +def test_idle_checks_change_nothing(cluster, source_engine): + """Once everything due is exported, further checks start no task, commit nothing, do not read the + index again, and show the same state.""" + node = cluster.instances["replica1"] + mt_table, iceberg_table = make_tables(node, source_engine) + node.query(f"INSERT INTO {mt_table} VALUES (1, 2020, {DUE})") + node.query(f"INSERT INTO {mt_table} VALUES (2, 2021, {NOT_DUE})") + wait_for_partitions_exported(node, mt_table, ["2020"]) + + rows = ttl_rows(node, mt_table) + refreshes = snapshot_refreshes(node) + snapshots = iceberg_snapshots(node, iceberg_table) + time.sleep(10) + assert len(ttl_tasks(node, mt_table)) == 1 + assert snapshot_refreshes(node) == refreshes, "The index was read again although it did not change" + assert iceberg_snapshots(node, iceberg_table) == snapshots + assert ttl_rows(node, mt_table) == rows + assert_exactly_once(iceberg_ids(node, iceberg_table), [1]) + + +def test_stop_moves_pauses_and_start_moves_resumes(cluster, source_engine): + node = cluster.instances["replica1"] + mt_table, iceberg_table = make_tables(node, source_engine) + node.query(f"SYSTEM STOP MOVES {mt_table}") + node.query(f"INSERT INTO {mt_table} VALUES (1, 2020, {DUE})") + + time.sleep(4) + assert ttl_tasks(node, mt_table) == [] + assert ttl_rows(node, mt_table)["2020"]["eligible_parts"] == 1 + + node.query(f"SYSTEM START MOVES {mt_table}") + wait_for_partitions_exported(node, mt_table, ["2020"]) + assert_exactly_once(iceberg_ids(node, iceberg_table), [1]) + assert_one_snapshot_per_task(node, mt_table, iceberg_table) + + +def test_modified_settings_apply_on_the_next_check(cluster, source_engine): + node = cluster.instances["replica1"] + mt_table, iceberg_table = make_tables( + node, source_engine, settings={"ttl_export_batch_window_seconds": 600, "ttl_export_batch_max_delay_seconds": 600} + ) + node.query(f"INSERT INTO {mt_table} VALUES (1, 2020, {DUE})") + time.sleep(4) + assert ttl_tasks(node, mt_table) == [] + + node.query(f"ALTER TABLE {mt_table} MODIFY SETTING ttl_export_batch_window_seconds = 0, ttl_export_batch_max_delay_seconds = 0") + wait_for_partitions_exported(node, mt_table, ["2020"], timeout=30) + assert_exactly_once(iceberg_ids(node, iceberg_table), [1]) + assert_one_snapshot_per_task(node, mt_table, iceberg_table) + + +def test_plain_commit_is_recorded_without_waiting_for_the_period(cluster): + """On a plain `MergeTree` table the task commits first, and the index records the parts as exported + afterwards. The scheduler is woken by the commit, so it does not wait for the next check.""" + node = cluster.instances["replica1"] + period = 20 + mt_table, iceberg_table = make_tables(node, "MergeTree", settings={"ttl_export_check_period_seconds": period}) + node.query(f"INSERT INTO {mt_table} VALUES (1, 2020, {DUE})") + + completed_at = None + start = time.time() + while True: + now = time.time() + assert now - start < 3 * period, "The part was not exported" + if completed_at is None and completed_ttl_tasks(node, mt_table): + completed_at = now + rows = ttl_rows(node, mt_table) + if completed_at is not None and "2020" in rows and rows["2020"]["exported_parts"] == 1: + break + time.sleep(0.2) + + assert time.time() - completed_at < period / 2, "The commit was recorded only by a periodic check" + assert rows["2020"]["claimed_parts"] == 0 and rows["2020"]["scheduler_replica"] == "", rows + assert_exactly_once(iceberg_ids(node, iceberg_table), [1]) + assert_one_snapshot_per_task(node, mt_table, iceberg_table) diff --git a/tests/integration/test_storage_iceberg_with_spark/test_export_partition_iceberg.py b/tests/integration/test_storage_iceberg_with_spark/test_export_partition_iceberg.py index 59f6ebedd979..c31c03a0400e 100644 --- a/tests/integration/test_storage_iceberg_with_spark/test_export_partition_iceberg.py +++ b/tests/integration/test_storage_iceberg_with_spark/test_export_partition_iceberg.py @@ -686,10 +686,10 @@ def test_idempotency_after_commit_crash(export_cluster): # The already-committed early-exit in commitExportPartitionTransaction surfaces # a sentinel note in committed_metadata_file (the original committer's paths # are not recoverable from inside the call). The sentinel makes the situation - # visible in system.replicated_partition_exports rather than leaving the + # visible in system.distributed_exports rather than leaving the # commit_info columns empty. committed_metadata_file = node.query( - f"SELECT committed_metadata_file FROM system.replicated_partition_exports " + f"SELECT committed_metadata_file FROM system.distributed_exports " f"WHERE source_table = '{source}' AND partition_id = '{pid}'" ).strip() assert committed_metadata_file == "", ( diff --git a/tests/integration/test_storage_iceberg_with_spark/test_export_partition_iceberg_catalog.py b/tests/integration/test_storage_iceberg_with_spark/test_export_partition_iceberg_catalog.py index e05e762a5f84..b1f2717a44c7 100644 --- a/tests/integration/test_storage_iceberg_with_spark/test_export_partition_iceberg_catalog.py +++ b/tests/integration/test_storage_iceberg_with_spark/test_export_partition_iceberg_catalog.py @@ -374,7 +374,7 @@ def test_catalog_idempotent_retry(catalog_export_cluster): ) committed_metadata_file = node.query( - f"SELECT committed_metadata_file FROM system.replicated_partition_exports " + f"SELECT committed_metadata_file FROM system.distributed_exports " f"WHERE source_table = '{source}' AND partition_id = '{pid}'" ).strip() assert committed_metadata_file == "", ( diff --git a/tests/queries/0_stateless/02995_settings_26_3_13_20001_antalya.tsv b/tests/queries/0_stateless/02995_settings_26_3_13_20001_antalya.tsv index 99b62d08b097..72235022b9a8 100644 --- a/tests/queries/0_stateless/02995_settings_26_3_13_20001_antalya.tsv +++ b/tests/queries/0_stateless/02995_settings_26_3_13_20001_antalya.tsv @@ -472,7 +472,7 @@ export_merge_tree_partition_all_on_error throw_first export_merge_tree_partition_force_export 0 export_merge_tree_partition_manifest_ttl 86400 export_merge_tree_partition_max_retries 3 -export_merge_tree_partition_task_timeout_seconds 86400 +export_merge_tree_task_timeout_seconds 86400 external_storage_connect_timeout_sec 10 external_storage_max_read_bytes 0 external_storage_max_read_rows 0 diff --git a/tests/queries/0_stateless/03745_system_background_schedule_pool.sql b/tests/queries/0_stateless/03745_system_background_schedule_pool.sql index 870ba8e3b68a..bd23421ed3e8 100644 --- a/tests/queries/0_stateless/03745_system_background_schedule_pool.sql +++ b/tests/queries/0_stateless/03745_system_background_schedule_pool.sql @@ -15,9 +15,9 @@ DROP TABLE test_table_03745; DROP TABLE IF EXISTS test_merge_tree_03745; CREATE TABLE test_merge_tree_03745 (x UInt64, y String) ENGINE = MergeTree() ORDER BY x SETTINGS refresh_statistics_interval = '0'; INSERT INTO test_merge_tree_03745 VALUES (1, 'a'), (2, 'b'); --- Exclude the experimental, config-gated EXPORT PARTITION scheduler task so this test stays +-- Exclude the experimental, config-gated tasks of `EXPORT PARTITION` and `TTL ... EXPORT` so this test stays -- deterministic regardless of whether `allow_experimental_export_merge_tree_partition` is enabled. -SELECT pool, database, table, table_uuid != toUUIDOrDefault(0) AS has_uuid, log_name FROM system.background_schedule_pool WHERE database = currentDatabase() AND log_name NOT LIKE '%partition_export_task%' ORDER BY ALL; +SELECT pool, database, table, table_uuid != toUUIDOrDefault(0) AS has_uuid, log_name FROM system.background_schedule_pool WHERE database = currentDatabase() AND log_name NOT LIKE '%export_task_scheduling_task%' AND log_name NOT LIKE '%export_ttl_task%' ORDER BY ALL; DROP TABLE test_merge_tree_03745; -- Test 3: Distributed table (distributed pool) diff --git a/tests/queries/0_stateless/05027_export_partition_merge_tree.reference b/tests/queries/0_stateless/05027_export_partition_merge_tree.reference index c233b89e2101..4802db5dfdb2 100644 --- a/tests/queries/0_stateless/05027_export_partition_merge_tree.reference +++ b/tests/queries/0_stateless/05027_export_partition_merge_tree.reference @@ -6,9 +6,7 @@ Select from destination table (2020, 2021) 3 2020 4 2021 5 2021 -Re-exporting 2020 without force is rejected -EXPORT_PARTITION_ALREADY_EXPORTED -Export remaining partitions with EXPORT PARTITION ALL (skip existing) +Export partition 2022 Select from destination table (all partitions) 1 2020 2 2020 @@ -25,7 +23,12 @@ Roundtrip: create a table from the exported S3 data 5 2021 6 2022 7 2022 -system.partition_exports statuses -2020 COMPLETED 2 0 -2021 COMPLETED 2 0 -2022 COMPLETED 1 0 +system.distributed_exports statuses +2020 COMPLETED 2 0 query +2021 COMPLETED 2 0 query +2022 COMPLETED 1 0 query +Re-exporting 2020 creates a new task, whose files already exist and are skipped +2 2 +1 1 +2 1 +3 1 diff --git a/tests/queries/0_stateless/05028_export_partition_replicated_merge_tree.reference b/tests/queries/0_stateless/05028_export_partition_replicated_merge_tree.reference index c233b89e2101..4802db5dfdb2 100644 --- a/tests/queries/0_stateless/05028_export_partition_replicated_merge_tree.reference +++ b/tests/queries/0_stateless/05028_export_partition_replicated_merge_tree.reference @@ -6,9 +6,7 @@ Select from destination table (2020, 2021) 3 2020 4 2021 5 2021 -Re-exporting 2020 without force is rejected -EXPORT_PARTITION_ALREADY_EXPORTED -Export remaining partitions with EXPORT PARTITION ALL (skip existing) +Export partition 2022 Select from destination table (all partitions) 1 2020 2 2020 @@ -25,7 +23,12 @@ Roundtrip: create a table from the exported S3 data 5 2021 6 2022 7 2022 -system.partition_exports statuses -2020 COMPLETED 2 0 -2021 COMPLETED 2 0 -2022 COMPLETED 1 0 +system.distributed_exports statuses +2020 COMPLETED 2 0 query +2021 COMPLETED 2 0 query +2022 COMPLETED 1 0 query +Re-exporting 2020 creates a new task, whose files already exist and are skipped +2 2 +1 1 +2 1 +3 1 diff --git a/tests/queries/0_stateless/05029_export_partition_key_collision_merge_tree.reference b/tests/queries/0_stateless/05029_export_partition_key_collision_merge_tree.reference index f2309b3349d7..cb45ae09a759 100644 --- a/tests/queries/0_stateless/05029_export_partition_key_collision_merge_tree.reference +++ b/tests/queries/0_stateless/05029_export_partition_key_collision_merge_tree.reference @@ -8,5 +8,6 @@ Both destinations received the partition 2 2020 1 2020 2 2020 -Re-exporting to the first destination is still rejected -EXPORT_PARTITION_ALREADY_EXPORTED +Re-exporting to the first destination only adds a task of the first destination +{db} x.y 1 +{db}.x y 2 diff --git a/tests/queries/0_stateless/05030_export_partition_key_collision_replicated_merge_tree.reference b/tests/queries/0_stateless/05030_export_partition_key_collision_replicated_merge_tree.reference index f2309b3349d7..cb45ae09a759 100644 --- a/tests/queries/0_stateless/05030_export_partition_key_collision_replicated_merge_tree.reference +++ b/tests/queries/0_stateless/05030_export_partition_key_collision_replicated_merge_tree.reference @@ -8,5 +8,6 @@ Both destinations received the partition 2 2020 1 2020 2 2020 -Re-exporting to the first destination is still rejected -EXPORT_PARTITION_ALREADY_EXPORTED +Re-exporting to the first destination only adds a task of the first destination +{db} x.y 1 +{db}.x y 2 diff --git a/tests/queries/0_stateless/05053_export_ttl_syntax.reference b/tests/queries/0_stateless/05053_export_ttl_syntax.reference new file mode 100644 index 000000000000..844b030a4493 --- /dev/null +++ b/tests/queries/0_stateless/05053_export_ttl_syntax.reference @@ -0,0 +1,9 @@ +CREATE TABLE t\n(\n `d` DateTime\n)\nENGINE = MergeTree\nORDER BY d\nTTL d + toIntervalDay(1) EXPORT TO TABLE dst, d + toIntervalDay(2) +ALTER TABLE t\n (MODIFY TTL d + toIntervalDay(1) EXPORT TO TABLE db.dst) +ALTER TABLE t\n (MODIFY TTL d EXPORT TO TABLE `my db`.`my.table`) +TTL t + toIntervalYear(10) EXPORT TO TABLE export_ttl_destination, t + toIntervalYear(20) SETTINGS +TTL t + toIntervalYear(5) EXPORT TO TABLE export_ttl_destination SETTINGS +0 + +TTL t + toIntervalYear(10) EXPORT TO TABLE export_ttl_by_year SETTINGS +TTL t + toIntervalYear(10) EXPORT TO TABLE export_ttl_by_year SETTINGS diff --git a/tests/queries/0_stateless/05053_export_ttl_syntax.sql b/tests/queries/0_stateless/05053_export_ttl_syntax.sql new file mode 100644 index 000000000000..1abd8ce59aa8 --- /dev/null +++ b/tests/queries/0_stateless/05053_export_ttl_syntax.sql @@ -0,0 +1,109 @@ +-- Tags: no-fasttest +-- no-fasttest: the destination is an S3 table. + +SELECT formatQuery('CREATE TABLE t (d DateTime) ENGINE = MergeTree ORDER BY d TTL d + INTERVAL 1 DAY EXPORT TO TABLE dst, d + INTERVAL 2 DAY DELETE'); +SELECT formatQuery('ALTER TABLE t MODIFY TTL d + INTERVAL 1 DAY EXPORT TO TABLE db.dst'); +SELECT formatQuery('ALTER TABLE t MODIFY TTL d EXPORT TO TABLE `my db`.`my.table`'); +SELECT formatQuery('ALTER TABLE t MODIFY TTL d EXPORT TO TABLE dst WHERE 1'); -- { serverError SYNTAX_ERROR } +SELECT formatQuery('ALTER TABLE t MODIFY TTL d EXPORT dst'); -- { serverError SYNTAX_ERROR } + +DROP TABLE IF EXISTS export_ttl_source; +DROP TABLE IF EXISTS export_ttl_destination; +DROP TABLE IF EXISTS export_ttl_memory; +DROP TABLE IF EXISTS export_ttl_narrow; + +CREATE TABLE export_ttl_destination (id UInt64, year UInt16, t DateTime) +ENGINE = S3(s3_conn, filename = 'export_ttl_syntax_destination', format = Parquet, partition_strategy = 'hive') +PARTITION BY year; + +CREATE TABLE export_ttl_memory (id UInt64, year UInt16, t DateTime) ENGINE = Memory; + +CREATE TABLE export_ttl_narrow (id UInt64, year UInt16) +ENGINE = S3(s3_conn, filename = 'export_ttl_syntax_narrow', format = Parquet, partition_strategy = 'hive') +PARTITION BY year; + +SET allow_experimental_export_ttl = 0; +CREATE TABLE export_ttl_source (id UInt64, year UInt16, t DateTime) ENGINE = MergeTree PARTITION BY year ORDER BY id +TTL t + INTERVAL 10 YEAR EXPORT TO TABLE export_ttl_destination; -- { serverError SUPPORT_IS_DISABLED } + +SET allow_experimental_export_ttl = 1; + +CREATE TABLE export_ttl_source (id UInt64, year UInt16, t DateTime) ENGINE = MergeTree PARTITION BY year ORDER BY id +TTL t + INTERVAL 10 YEAR EXPORT TO TABLE export_ttl_missing; -- { serverError BAD_TTL_EXPRESSION } + +CREATE TABLE export_ttl_source (id UInt64, year UInt16, t DateTime) ENGINE = MergeTree PARTITION BY year ORDER BY id +TTL t + INTERVAL 10 YEAR EXPORT TO TABLE export_ttl_memory; -- { serverError BAD_TTL_EXPRESSION } + +CREATE TABLE export_ttl_source (id UInt64, year UInt16, t DateTime) ENGINE = MergeTree PARTITION BY year ORDER BY id +TTL t + INTERVAL 10 YEAR EXPORT TO TABLE export_ttl_narrow; -- { serverError NUMBER_OF_COLUMNS_DOESNT_MATCH } + +CREATE TABLE export_ttl_source (id UInt64, year UInt16, t DateTime) ENGINE = MergeTree PARTITION BY year ORDER BY id +TTL id EXPORT TO TABLE export_ttl_destination; -- { serverError BAD_TTL_EXPRESSION } + +CREATE TABLE export_ttl_source (id UInt64, year UInt16, t DateTime) ENGINE = MergeTree PARTITION BY year ORDER BY id +TTL t + INTERVAL 10 YEAR EXPORT TO TABLE export_ttl_destination, t + INTERVAL 20 YEAR EXPORT TO TABLE export_ttl_destination; -- { serverError BAD_TTL_EXPRESSION } + +CREATE TABLE export_ttl_source (id UInt64, year UInt16, t DateTime) ENGINE = MergeTree PARTITION BY year ORDER BY id +TTL t + INTERVAL 10 YEAR EXPORT TO TABLE export_ttl_destination, t + INTERVAL 20 YEAR DELETE; + +SELECT extract(create_table_query, 'TTL .*? SETTINGS') FROM system.tables WHERE database = currentDatabase() AND name = 'export_ttl_source'; + +-- The table cannot export to itself. +ALTER TABLE export_ttl_source MODIFY TTL t + INTERVAL 10 YEAR EXPORT TO TABLE export_ttl_source; -- { serverError BAD_TTL_EXPRESSION } +ALTER TABLE export_ttl_source MODIFY TTL t + INTERVAL 5 YEAR EXPORT TO TABLE export_ttl_destination; +SELECT extract(create_table_query, 'TTL .*? SETTINGS') FROM system.tables WHERE database = currentDatabase() AND name = 'export_ttl_source'; + +-- Nothing is due, so nothing is exported. +INSERT INTO export_ttl_source VALUES (1, 2020, now()); +SELECT count() FROM system.distributed_exports WHERE source_database = currentDatabase() AND source_table = 'export_ttl_source'; + +ALTER TABLE export_ttl_source REMOVE TTL; +SELECT extract(create_table_query, 'TTL .*? SETTINGS') FROM system.tables WHERE database = currentDatabase() AND name = 'export_ttl_source'; + +DROP TABLE export_ttl_source; + +-- A destination partition key that can never be proven to be single-valued over a group is refused. +DROP TABLE IF EXISTS export_ttl_monthly; +DROP TABLE IF EXISTS export_ttl_by_id; +DROP TABLE IF EXISTS export_ttl_by_year; +DROP TABLE IF EXISTS export_ttl_by_day_of_week; + +CREATE TABLE export_ttl_by_id (id UInt64, year UInt16, t DateTime) +ENGINE = S3(s3_conn, filename = 'export_ttl_syntax_by_id', format = Parquet, partition_strategy = 'hive') +PARTITION BY id; + +CREATE TABLE export_ttl_by_year (id UInt64, year UInt16, t DateTime) +ENGINE = S3(s3_conn, filename = 'export_ttl_syntax_by_year/{_partition_id}/{_file}.parquet', format = Parquet, partition_strategy = 'wildcard') +PARTITION BY toYear(t); + +CREATE TABLE export_ttl_by_day_of_week (id UInt64, year UInt16, t DateTime) +ENGINE = S3(s3_conn, filename = 'export_ttl_syntax_by_day_of_week/{_partition_id}/{_file}.parquet', format = Parquet, partition_strategy = 'wildcard') +PARTITION BY toDayOfWeek(t); + +-- `id` is not in the source partition key. +CREATE TABLE export_ttl_monthly (id UInt64, year UInt16, t DateTime) ENGINE = MergeTree PARTITION BY toYYYYMM(t) ORDER BY id +TTL t + INTERVAL 10 YEAR EXPORT TO TABLE export_ttl_by_id; -- { serverError BAD_ARGUMENTS } + +-- `toDayOfWeek` is not monotonic. +CREATE TABLE export_ttl_monthly (id UInt64, year UInt16, t DateTime) ENGINE = MergeTree PARTITION BY toYYYYMM(t) ORDER BY id +TTL t + INTERVAL 10 YEAR EXPORT TO TABLE export_ttl_by_day_of_week; -- { serverError BAD_ARGUMENTS } + +-- An unpartitioned source has no partition key to prove anything from. +CREATE TABLE export_ttl_monthly (id UInt64, year UInt16, t DateTime) ENGINE = MergeTree ORDER BY id +TTL t + INTERVAL 10 YEAR EXPORT TO TABLE export_ttl_destination; -- { serverError BAD_ARGUMENTS } + +-- A month is within a year, which is proven from the parts when they are exported. +CREATE TABLE export_ttl_monthly (id UInt64, year UInt16, t DateTime) ENGINE = MergeTree PARTITION BY toYYYYMM(t) ORDER BY id +TTL t + INTERVAL 10 YEAR EXPORT TO TABLE export_ttl_by_year; +SELECT extract(create_table_query, 'TTL .*? SETTINGS') FROM system.tables WHERE database = currentDatabase() AND name = 'export_ttl_monthly'; + +ALTER TABLE export_ttl_monthly MODIFY TTL t + INTERVAL 10 YEAR EXPORT TO TABLE export_ttl_by_id; -- { serverError BAD_ARGUMENTS } +SELECT extract(create_table_query, 'TTL .*? SETTINGS') FROM system.tables WHERE database = currentDatabase() AND name = 'export_ttl_monthly'; + +DROP TABLE export_ttl_monthly; +DROP TABLE export_ttl_by_id; +DROP TABLE export_ttl_by_year; +DROP TABLE export_ttl_by_day_of_week; +DROP TABLE export_ttl_destination; +DROP TABLE export_ttl_memory; +DROP TABLE export_ttl_narrow; diff --git a/tests/queries/0_stateless/05054_export_ttl_merge_tree.reference b/tests/queries/0_stateless/05054_export_ttl_merge_tree.reference new file mode 100644 index 000000000000..6b979b1b9b98 --- /dev/null +++ b/tests/queries/0_stateless/05054_export_ttl_merge_tree.reference @@ -0,0 +1,18 @@ +Destination after the first group +1 2020 +2 2020 +3 2020 +Tasks +ttl COMPLETED 2 2020 +Partition 2021 is not due +0 0 +Parts in different export states are not merged +refused +A part that becomes due later is exported on its own +1 2020 +2 2020 +3 2020 +6 2020 +2 2 3 +Removing the TTL forgets what was exported +0 diff --git a/tests/queries/0_stateless/05054_export_ttl_merge_tree.sh b/tests/queries/0_stateless/05054_export_ttl_merge_tree.sh new file mode 100755 index 000000000000..cb7d911fe007 --- /dev/null +++ b/tests/queries/0_stateless/05054_export_ttl_merge_tree.sh @@ -0,0 +1,13 @@ +#!/usr/bin/env bash +# Tags: no-fasttest, no-shared-merge-tree +# no-fasttest: requires S3 / MinIO. +# no-shared-merge-tree: this test exercises the EXPORT TTL of a plain (non-replicated) MergeTree. + +CUR_DIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd) +# shellcheck source=../shell_config.sh +. "$CUR_DIR"/../shell_config.sh + +# shellcheck source=./export_ttl.lib +. "$CUR_DIR"/export_ttl.lib + +run_export_ttl_test "MergeTree" diff --git a/tests/queries/0_stateless/05055_export_ttl_replicated_merge_tree.reference b/tests/queries/0_stateless/05055_export_ttl_replicated_merge_tree.reference new file mode 100644 index 000000000000..6b979b1b9b98 --- /dev/null +++ b/tests/queries/0_stateless/05055_export_ttl_replicated_merge_tree.reference @@ -0,0 +1,18 @@ +Destination after the first group +1 2020 +2 2020 +3 2020 +Tasks +ttl COMPLETED 2 2020 +Partition 2021 is not due +0 0 +Parts in different export states are not merged +refused +A part that becomes due later is exported on its own +1 2020 +2 2020 +3 2020 +6 2020 +2 2 3 +Removing the TTL forgets what was exported +0 diff --git a/tests/queries/0_stateless/05055_export_ttl_replicated_merge_tree.sh b/tests/queries/0_stateless/05055_export_ttl_replicated_merge_tree.sh new file mode 100755 index 000000000000..68c9a9120d16 --- /dev/null +++ b/tests/queries/0_stateless/05055_export_ttl_replicated_merge_tree.sh @@ -0,0 +1,12 @@ +#!/usr/bin/env bash +# Tags: no-fasttest, replica, no-replicated-database +# no-fasttest: requires S3 / MinIO. + +CUR_DIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd) +# shellcheck source=../shell_config.sh +. "$CUR_DIR"/../shell_config.sh + +# shellcheck source=./export_ttl.lib +. "$CUR_DIR"/export_ttl.lib + +run_export_ttl_test "ReplicatedMergeTree('/clickhouse/tables/$CLICKHOUSE_TEST_ZOOKEEPER_PREFIX/export_ttl_mt', 'r1')" diff --git a/tests/queries/0_stateless/05056_export_ttl_partition_key_check.reference b/tests/queries/0_stateless/05056_export_ttl_partition_key_check.reference new file mode 100644 index 000000000000..aebeef1397f6 --- /dev/null +++ b/tests/queries/0_stateless/05056_export_ttl_partition_key_check.reference @@ -0,0 +1,17 @@ +structural +dst_year +dst_month +dst_region +dst_ba +dst_year_mod +dst_hash +dst_year +dynamic +dst_to_year +dst_d +dst_hundreds +refused +dst_nullable +alter +TTL t + toIntervalDay(30) SETTINGS +TTL t + toIntervalDay(1) EXPORT TO TABLE dst_year_mod SETTINGS diff --git a/tests/queries/0_stateless/05056_export_ttl_partition_key_check.sql b/tests/queries/0_stateless/05056_export_ttl_partition_key_check.sql new file mode 100644 index 000000000000..4dad51d41972 --- /dev/null +++ b/tests/queries/0_stateless/05056_export_ttl_partition_key_check.sql @@ -0,0 +1,149 @@ +-- Tags: no-fasttest +-- no-fasttest: the destinations are S3 tables. + +-- The partition key of the destination of an EXPORT TTL is checked when the expression is added: +-- a key that is a function of the source partition key (structural) or monotonic in a single column +-- of it (checked per group later) is accepted, any other is refused. + +SET allow_experimental_export_ttl = 1; + +DROP TABLE IF EXISTS src; +DROP TABLE IF EXISTS dst_year; +DROP TABLE IF EXISTS dst_region; +DROP TABLE IF EXISTS dst_ba; +DROP TABLE IF EXISTS dst_id; +DROP TABLE IF EXISTS dst_d; +DROP TABLE IF EXISTS dst_month; +DROP TABLE IF EXISTS dst_year_mod; +DROP TABLE IF EXISTS dst_hash; +DROP TABLE IF EXISTS dst_to_year; +DROP TABLE IF EXISTS dst_day_of_week; +DROP TABLE IF EXISTS dst_hundreds; +DROP TABLE IF EXISTS dst_nullable; + +CREATE TABLE dst_year (id UInt64, year UInt16, region String, a UInt8, b UInt8, t DateTime, d Date) +ENGINE = S3(s3_conn, filename = 'export_ttl_pkey_check_year', format = Parquet, partition_strategy = 'hive') PARTITION BY year; +CREATE TABLE dst_region (id UInt64, year UInt16, region String, a UInt8, b UInt8, t DateTime, d Date) +ENGINE = S3(s3_conn, filename = 'export_ttl_pkey_check_region', format = Parquet, partition_strategy = 'hive') PARTITION BY region; +CREATE TABLE dst_ba (id UInt64, year UInt16, region String, a UInt8, b UInt8, t DateTime, d Date) +ENGINE = S3(s3_conn, filename = 'export_ttl_pkey_check_ba', format = Parquet, partition_strategy = 'hive') PARTITION BY (b, a); +CREATE TABLE dst_id (id UInt64, year UInt16, region String, a UInt8, b UInt8, t DateTime, d Date) +ENGINE = S3(s3_conn, filename = 'export_ttl_pkey_check_id', format = Parquet, partition_strategy = 'hive') PARTITION BY id; +CREATE TABLE dst_d (id UInt64, year UInt16, region String, a UInt8, b UInt8, t DateTime, d Date) +ENGINE = S3(s3_conn, filename = 'export_ttl_pkey_check_d', format = Parquet, partition_strategy = 'hive') PARTITION BY d; +CREATE TABLE dst_month (id UInt64, year UInt16, region String, a UInt8, b UInt8, t DateTime, d Date) +ENGINE = S3(s3_conn, filename = 'export_ttl_pkey_check_month/{_partition_id}/{_file}.parquet', format = Parquet, partition_strategy = 'wildcard') PARTITION BY toYYYYMM(t); +CREATE TABLE dst_year_mod (id UInt64, year UInt16, region String, a UInt8, b UInt8, t DateTime, d Date) +ENGINE = S3(s3_conn, filename = 'export_ttl_pkey_check_year_mod/{_partition_id}/{_file}.parquet', format = Parquet, partition_strategy = 'wildcard') PARTITION BY year % 10; +CREATE TABLE dst_hash (id UInt64, year UInt16, region String, a UInt8, b UInt8, t DateTime, d Date) +ENGINE = S3(s3_conn, filename = 'export_ttl_pkey_check_hash/{_partition_id}/{_file}.parquet', format = Parquet, partition_strategy = 'wildcard') PARTITION BY cityHash64(year) % 4; +CREATE TABLE dst_to_year (id UInt64, year UInt16, region String, a UInt8, b UInt8, t DateTime, d Date) +ENGINE = S3(s3_conn, filename = 'export_ttl_pkey_check_to_year/{_partition_id}/{_file}.parquet', format = Parquet, partition_strategy = 'wildcard') PARTITION BY toYear(t); +CREATE TABLE dst_day_of_week (id UInt64, year UInt16, region String, a UInt8, b UInt8, t DateTime, d Date) +ENGINE = S3(s3_conn, filename = 'export_ttl_pkey_check_day_of_week/{_partition_id}/{_file}.parquet', format = Parquet, partition_strategy = 'wildcard') PARTITION BY toDayOfWeek(t); +CREATE TABLE dst_hundreds (id UInt64, year UInt16, region String, a UInt8, b UInt8, t DateTime, d Date) +ENGINE = S3(s3_conn, filename = 'export_ttl_pkey_check_hundreds/{_partition_id}/{_file}.parquet', format = Parquet, partition_strategy = 'wildcard') PARTITION BY intDiv(id, 100); + +SELECT 'structural'; + +CREATE TABLE src (id UInt64, year UInt16, region String, a UInt8, b UInt8, t DateTime, d Date) ENGINE = MergeTree PARTITION BY year ORDER BY id +TTL t + INTERVAL 1 DAY EXPORT TO TABLE dst_year; +SELECT extract(create_table_query, 'EXPORT TO TABLE (\\w+)') FROM system.tables WHERE database = currentDatabase() AND name = 'src'; +DROP TABLE src; + +CREATE TABLE src (id UInt64, year UInt16, region String, a UInt8, b UInt8, t DateTime, d Date) ENGINE = MergeTree PARTITION BY toYYYYMM(t) ORDER BY id +TTL t + INTERVAL 1 DAY EXPORT TO TABLE dst_month; +SELECT extract(create_table_query, 'EXPORT TO TABLE (\\w+)') FROM system.tables WHERE database = currentDatabase() AND name = 'src'; +DROP TABLE src; + +CREATE TABLE src (id UInt64, year UInt16, region String, a UInt8, b UInt8, t DateTime, d Date) ENGINE = MergeTree PARTITION BY (region, toYYYYMM(t)) ORDER BY id +TTL t + INTERVAL 1 DAY EXPORT TO TABLE dst_region; +SELECT extract(create_table_query, 'EXPORT TO TABLE (\\w+)') FROM system.tables WHERE database = currentDatabase() AND name = 'src'; +DROP TABLE src; + +CREATE TABLE src (id UInt64, year UInt16, region String, a UInt8, b UInt8, t DateTime, d Date) ENGINE = MergeTree PARTITION BY (a, b) ORDER BY id +TTL t + INTERVAL 1 DAY EXPORT TO TABLE dst_ba; +SELECT extract(create_table_query, 'EXPORT TO TABLE (\\w+)') FROM system.tables WHERE database = currentDatabase() AND name = 'src'; +DROP TABLE src; + +CREATE TABLE src (id UInt64, year UInt16, region String, a UInt8, b UInt8, t DateTime, d Date) ENGINE = MergeTree PARTITION BY year ORDER BY id +TTL t + INTERVAL 1 DAY EXPORT TO TABLE dst_year_mod; +SELECT extract(create_table_query, 'EXPORT TO TABLE (\\w+)') FROM system.tables WHERE database = currentDatabase() AND name = 'src'; +DROP TABLE src; + +CREATE TABLE src (id UInt64, year UInt16, region String, a UInt8, b UInt8, t DateTime, d Date) ENGINE = MergeTree PARTITION BY year ORDER BY id +TTL t + INTERVAL 1 DAY EXPORT TO TABLE dst_hash; +SELECT extract(create_table_query, 'EXPORT TO TABLE (\\w+)') FROM system.tables WHERE database = currentDatabase() AND name = 'src'; +DROP TABLE src; + +CREATE TABLE src (id UInt64, year UInt16, region String, a UInt8, b UInt8, t DateTime, d Date) ENGINE = MergeTree PARTITION BY toString(year) ORDER BY id +TTL t + INTERVAL 1 DAY EXPORT TO TABLE dst_year; +SELECT extract(create_table_query, 'EXPORT TO TABLE (\\w+)') FROM system.tables WHERE database = currentDatabase() AND name = 'src'; +DROP TABLE src; + +SELECT 'dynamic'; + +CREATE TABLE src (id UInt64, year UInt16, region String, a UInt8, b UInt8, t DateTime, d Date) ENGINE = MergeTree PARTITION BY toYYYYMM(t) ORDER BY id +TTL t + INTERVAL 1 DAY EXPORT TO TABLE dst_to_year; +SELECT extract(create_table_query, 'EXPORT TO TABLE (\\w+)') FROM system.tables WHERE database = currentDatabase() AND name = 'src'; +DROP TABLE src; + +CREATE TABLE src (id UInt64, year UInt16, region String, a UInt8, b UInt8, t DateTime, d Date) ENGINE = MergeTree PARTITION BY toYYYYMM(d) ORDER BY id +TTL t + INTERVAL 1 DAY EXPORT TO TABLE dst_d; +SELECT extract(create_table_query, 'EXPORT TO TABLE (\\w+)') FROM system.tables WHERE database = currentDatabase() AND name = 'src'; +DROP TABLE src; + +CREATE TABLE src (id UInt64, year UInt16, region String, a UInt8, b UInt8, t DateTime, d Date) ENGINE = MergeTree PARTITION BY intDiv(id, 1000) ORDER BY id +TTL t + INTERVAL 1 DAY EXPORT TO TABLE dst_hundreds; +SELECT extract(create_table_query, 'EXPORT TO TABLE (\\w+)') FROM system.tables WHERE database = currentDatabase() AND name = 'src'; +DROP TABLE src; + +SELECT 'refused'; + +-- `id` is not in the source partition key. +CREATE TABLE src (id UInt64, year UInt16, region String, a UInt8, b UInt8, t DateTime, d Date) ENGINE = MergeTree PARTITION BY year ORDER BY id +TTL t + INTERVAL 1 DAY EXPORT TO TABLE dst_id; -- { serverError BAD_ARGUMENTS } + +-- `toDayOfWeek` is not monotonic. +CREATE TABLE src (id UInt64, year UInt16, region String, a UInt8, b UInt8, t DateTime, d Date) ENGINE = MergeTree PARTITION BY toYYYYMM(t) ORDER BY id +TTL t + INTERVAL 1 DAY EXPORT TO TABLE dst_day_of_week; -- { serverError BAD_ARGUMENTS } + +-- A source without a partition key into a partitioned destination. +CREATE TABLE src (id UInt64, year UInt16, region String, a UInt8, b UInt8, t DateTime, d Date) ENGINE = MergeTree ORDER BY id +TTL t + INTERVAL 1 DAY EXPORT TO TABLE dst_year; -- { serverError BAD_ARGUMENTS } + +-- A `Nullable` column that has to be checked per group: a NULL forms its own destination partition. +CREATE TABLE dst_nullable (id UInt64, k Nullable(UInt32), t DateTime) +ENGINE = S3(s3_conn, filename = 'export_ttl_pkey_check_nullable/{_partition_id}/{_file}.parquet', format = Parquet, partition_strategy = 'wildcard') PARTITION BY intDiv(k, 10); +CREATE TABLE src (id UInt64, k Nullable(UInt32), t DateTime) ENGINE = MergeTree PARTITION BY intDiv(k, 100) ORDER BY id +TTL t + INTERVAL 1 DAY EXPORT TO TABLE dst_nullable SETTINGS allow_nullable_key = 1; -- { serverError BAD_ARGUMENTS } + +-- The same destination is a function of a `Nullable` source key, which holds for every group. +CREATE TABLE src (id UInt64, k Nullable(UInt32), t DateTime) ENGINE = MergeTree PARTITION BY k ORDER BY id +TTL t + INTERVAL 1 DAY EXPORT TO TABLE dst_nullable SETTINGS allow_nullable_key = 1; +SELECT extract(create_table_query, 'EXPORT TO TABLE (\\w+)') FROM system.tables WHERE database = currentDatabase() AND name = 'src'; +DROP TABLE src; + +SELECT 'alter'; + +CREATE TABLE src (id UInt64, year UInt16, region String, a UInt8, b UInt8, t DateTime, d Date) ENGINE = MergeTree PARTITION BY year ORDER BY id +TTL t + INTERVAL 30 DAY DELETE; +ALTER TABLE src MODIFY TTL t + INTERVAL 1 DAY EXPORT TO TABLE dst_id; -- { serverError BAD_ARGUMENTS } +ALTER TABLE src MODIFY TTL t + INTERVAL 1 DAY EXPORT TO TABLE dst_day_of_week; -- { serverError BAD_ARGUMENTS } +SELECT extract(create_table_query, 'TTL .*? SETTINGS') FROM system.tables WHERE database = currentDatabase() AND name = 'src'; +ALTER TABLE src MODIFY TTL t + INTERVAL 1 DAY EXPORT TO TABLE dst_year_mod; +SELECT extract(create_table_query, 'TTL .*? SETTINGS') FROM system.tables WHERE database = currentDatabase() AND name = 'src'; +DROP TABLE src; + +DROP TABLE dst_year; +DROP TABLE dst_region; +DROP TABLE dst_ba; +DROP TABLE dst_id; +DROP TABLE dst_d; +DROP TABLE dst_month; +DROP TABLE dst_year_mod; +DROP TABLE dst_hash; +DROP TABLE dst_to_year; +DROP TABLE dst_day_of_week; +DROP TABLE dst_hundreds; +DROP TABLE dst_nullable; diff --git a/tests/queries/0_stateless/export_partition.lib b/tests/queries/0_stateless/export_partition.lib index 82d66d783c2b..c18ae42f0980 100644 --- a/tests/queries/0_stateless/export_partition.lib +++ b/tests/queries/0_stateless/export_partition.lib @@ -17,22 +17,23 @@ function partition_export_source_engine() fi } -# Poll `system.partition_exports` until the given export reaches the expected status (or timeout). -# $1 source table, $2 destination database, $3 destination table, $4 partition id, $5 expected status. +# Poll `system.distributed_exports` until the given number of export tasks of a partition exist and all of +# them completed (or timeout). Every `EXPORT PARTITION` creates a new task. +# $1 source table, $2 destination database, $3 destination table, $4 partition id, $5 number of tasks (1 by default). function wait_for_partition_export_status() { local source_table="$1" local destination_database="$2" local destination_table="$3" local partition_id="$4" - local expected="$5" + local expected="${5:-1} ${5:-1}" - local status + local state local i=0 while [ "$i" -lt 120 ] do - status=$(${CLICKHOUSE_CLIENT} --query "SELECT status FROM system.partition_exports WHERE source_table = '$source_table' AND destination_database = '$destination_database' AND destination_table = '$destination_table' AND partition_id = '$partition_id'") - if [ "$status" = "$expected" ] + state=$(${CLICKHOUSE_CLIENT} --query "SELECT count(), countIf(status = 'COMPLETED') FROM system.distributed_exports WHERE source_table = '$source_table' AND destination_database = '$destination_database' AND destination_table = '$destination_table' AND partition_id = '$partition_id'") + if [ "$state" = "$(echo -e "$expected")" ] then return 0 fi @@ -40,7 +41,7 @@ function wait_for_partition_export_status() i=$((i + 1)) done - echo "TIMEOUT waiting for the export of $partition_id to $destination_database.$destination_table to reach $expected (last: '$status')" + echo "TIMEOUT waiting for ${5:-1} completed export(s) of $partition_id to $destination_database.$destination_table (last tasks, completed: '$state')" return 1 } @@ -75,21 +76,18 @@ function run_partition_export_roundtrip_test() echo "Export partition 2020" ${CLICKHOUSE_CLIENT} --query "ALTER TABLE $mt_table EXPORT PARTITION ID '2020' TO TABLE $s3_table" - wait_for_partition_export_status "$mt_table" "$CLICKHOUSE_DATABASE" "$s3_table" "2020" "COMPLETED" + wait_for_partition_export_status "$mt_table" "$CLICKHOUSE_DATABASE" "$s3_table" "2020" echo "Export partition 2021" ${CLICKHOUSE_CLIENT} --query "ALTER TABLE $mt_table EXPORT PARTITION ID '2021' TO TABLE $s3_table" - wait_for_partition_export_status "$mt_table" "$CLICKHOUSE_DATABASE" "$s3_table" "2021" "COMPLETED" + wait_for_partition_export_status "$mt_table" "$CLICKHOUSE_DATABASE" "$s3_table" "2021" echo "Select from destination table (2020, 2021)" ${CLICKHOUSE_CLIENT} --query "SELECT * FROM $s3_table ORDER BY id" - echo "Re-exporting 2020 without force is rejected" - ${CLICKHOUSE_CLIENT} --query "ALTER TABLE $mt_table EXPORT PARTITION ID '2020' TO TABLE $s3_table" 2>&1 | grep -o "EXPORT_PARTITION_ALREADY_EXPORTED" | head -1 - - echo "Export remaining partitions with EXPORT PARTITION ALL (skip existing)" - ${CLICKHOUSE_CLIENT} --query "ALTER TABLE $mt_table EXPORT PARTITION ALL TO TABLE $s3_table SETTINGS export_merge_tree_partition_all_on_error = 'skip_conflicts'" - wait_for_partition_export_status "$mt_table" "$CLICKHOUSE_DATABASE" "$s3_table" "2022" "COMPLETED" + echo "Export partition 2022" + ${CLICKHOUSE_CLIENT} --query "ALTER TABLE $mt_table EXPORT PARTITION ID '2022' TO TABLE $s3_table" + wait_for_partition_export_status "$mt_table" "$CLICKHOUSE_DATABASE" "$s3_table" "2022" echo "Select from destination table (all partitions)" ${CLICKHOUSE_CLIENT} --query "SELECT * FROM $s3_table ORDER BY id" @@ -98,8 +96,14 @@ function run_partition_export_roundtrip_test() ${CLICKHOUSE_CLIENT} --query "CREATE TABLE $mt_roundtrip ENGINE = $roundtrip_engine PARTITION BY year ORDER BY tuple() AS SELECT * FROM $s3_table" ${CLICKHOUSE_CLIENT} --query "SELECT * FROM $mt_roundtrip ORDER BY id" - echo "system.partition_exports statuses" - ${CLICKHOUSE_CLIENT} --query "SELECT partition_id, status, parts_count, parts_to_do FROM system.partition_exports WHERE source_table = '$mt_table' AND destination_table = '$s3_table' ORDER BY partition_id" + echo "system.distributed_exports statuses" + ${CLICKHOUSE_CLIENT} --query "SELECT partition_id, status, parts_count, parts_to_do, source FROM system.distributed_exports WHERE source_table = '$mt_table' AND destination_table = '$s3_table' ORDER BY partition_id" + + echo "Re-exporting 2020 creates a new task, whose files already exist and are skipped" + ${CLICKHOUSE_CLIENT} --query "ALTER TABLE $mt_table EXPORT PARTITION ID '2020' TO TABLE $s3_table" + wait_for_partition_export_status "$mt_table" "$CLICKHOUSE_DATABASE" "$s3_table" "2020" 2 + ${CLICKHOUSE_CLIENT} --query "SELECT count(), uniqExact(transaction_id) FROM system.distributed_exports WHERE source_table = '$mt_table' AND destination_table = '$s3_table' AND partition_id = '2020'" + ${CLICKHOUSE_CLIENT} --query "SELECT id, count() FROM $s3_table WHERE year = 2020 GROUP BY id ORDER BY id" ${CLICKHOUSE_CLIENT} --query "DROP TABLE IF EXISTS $mt_table" ${CLICKHOUSE_CLIENT} --query "DROP TABLE IF EXISTS $s3_table" @@ -130,21 +134,23 @@ function run_partition_export_dotted_destination_test() echo "Export partition 2020 to the first destination" ${CLICKHOUSE_CLIENT} --query "ALTER TABLE $mt_table EXPORT PARTITION ID '2020' TO TABLE \`$dotted_database\`.y" - wait_for_partition_export_status "$mt_table" "$dotted_database" "y" "2020" "COMPLETED" + wait_for_partition_export_status "$mt_table" "$dotted_database" "y" "2020" echo "The same partition can be exported to the second destination" ${CLICKHOUSE_CLIENT} --query "ALTER TABLE $mt_table EXPORT PARTITION ID '2020' TO TABLE $CLICKHOUSE_DATABASE.\`x.y\`" - wait_for_partition_export_status "$mt_table" "$CLICKHOUSE_DATABASE" "x.y" "2020" "COMPLETED" + wait_for_partition_export_status "$mt_table" "$CLICKHOUSE_DATABASE" "x.y" "2020" echo "Both exports are tracked independently" - ${CLICKHOUSE_CLIENT} --query "SELECT replaceOne(destination_database, '$CLICKHOUSE_DATABASE', '{db}'), destination_table, status FROM system.partition_exports WHERE source_table = '$mt_table' ORDER BY 1, 2" + ${CLICKHOUSE_CLIENT} --query "SELECT replaceOne(destination_database, '$CLICKHOUSE_DATABASE', '{db}'), destination_table, status FROM system.distributed_exports WHERE source_table = '$mt_table' ORDER BY 1, 2" echo "Both destinations received the partition" ${CLICKHOUSE_CLIENT} --query "SELECT * FROM \`$dotted_database\`.y ORDER BY id" ${CLICKHOUSE_CLIENT} --query "SELECT * FROM $CLICKHOUSE_DATABASE.\`x.y\` ORDER BY id" - echo "Re-exporting to the first destination is still rejected" - ${CLICKHOUSE_CLIENT} --query "ALTER TABLE $mt_table EXPORT PARTITION ID '2020' TO TABLE \`$dotted_database\`.y" 2>&1 | grep -o "EXPORT_PARTITION_ALREADY_EXPORTED" | head -1 + echo "Re-exporting to the first destination only adds a task of the first destination" + ${CLICKHOUSE_CLIENT} --query "ALTER TABLE $mt_table EXPORT PARTITION ID '2020' TO TABLE \`$dotted_database\`.y" + wait_for_partition_export_status "$mt_table" "$dotted_database" "y" "2020" 2 + ${CLICKHOUSE_CLIENT} --query "SELECT replaceOne(destination_database, '$CLICKHOUSE_DATABASE', '{db}'), destination_table, count() FROM system.distributed_exports WHERE source_table = '$mt_table' GROUP BY 1, 2 ORDER BY 1, 2" ${CLICKHOUSE_CLIENT} --query "DROP TABLE IF EXISTS $CLICKHOUSE_DATABASE.\`x.y\`" ${CLICKHOUSE_CLIENT} --query "DROP TABLE IF EXISTS $mt_table" diff --git a/tests/queries/0_stateless/export_ttl.lib b/tests/queries/0_stateless/export_ttl.lib new file mode 100644 index 000000000000..00ce4e6171d2 --- /dev/null +++ b/tests/queries/0_stateless/export_ttl.lib @@ -0,0 +1,102 @@ +#!/usr/bin/env bash + +# Shared body of the `TTL ... EXPORT TO TABLE` tests, invoked once per source engine. The wrappers +# share the reference file, so the output must not depend on the engine. + +# Poll until the table has the given number of completed TTL export tasks and nothing of the +# partition is claimed or waiting to be exported (or timeout). +# $1 source table, $2 partition id, $3 number of completed tasks. +function wait_for_export_ttl() +{ + local source_table="$1" + local partition_id="$2" + local expected="$3" + + local state + local i=0 + while [ "$i" -lt 120 ] + do + state=$(${CLICKHOUSE_CLIENT} --query " + SELECT + (SELECT countIf(status = 'COMPLETED') FROM system.distributed_exports + WHERE source_database = currentDatabase() AND source_table = '$source_table' AND source = 'ttl'), + (SELECT count() FROM system.ttl_exports + WHERE database = currentDatabase() AND table = '$source_table' AND partition_id = '$partition_id' + AND claimed_parts = 0 AND eligible_parts = 0)") + if [ "$state" = "$(echo -e "$expected\t1")" ] + then + return 0 + fi + sleep 0.5 + i=$((i + 1)) + done + + echo "TIMEOUT waiting for $expected completed TTL export task(s) of $source_table (last: '$state')" + return 1 +} + +# $1 is the engine clause of the source table. +function run_export_ttl_test() +{ + local source_engine="$1" + + local mt_table="export_ttl_mt" + local s3_table="export_ttl_s3" + + ${CLICKHOUSE_CLIENT} --query "DROP TABLE IF EXISTS $mt_table" + ${CLICKHOUSE_CLIENT} --query "DROP TABLE IF EXISTS $s3_table" + + ${CLICKHOUSE_CLIENT} --query "CREATE TABLE $s3_table (id UInt64, year UInt16, t DateTime) ENGINE = S3(s3_conn, filename='export_ttl_${CLICKHOUSE_DATABASE}', format=Parquet, partition_strategy='hive') PARTITION BY year" + ${CLICKHOUSE_CLIENT} --allow_experimental_export_ttl 1 --query " + CREATE TABLE $mt_table (id UInt64, year UInt16, t DateTime) ENGINE = $source_engine + PARTITION BY year ORDER BY id + TTL t + INTERVAL 1 DAY EXPORT TO TABLE $s3_table + SETTINGS ttl_export_check_period_seconds = 1, ttl_export_batch_window_seconds = 0, ttl_export_batch_max_delay_seconds = 0" + + # Paused, so both parts of 2020 are exported in one group, and not merged before. + ${CLICKHOUSE_CLIENT} --query "SYSTEM STOP MOVES $mt_table" + ${CLICKHOUSE_CLIENT} --query "SYSTEM STOP MERGES $mt_table" + + ${CLICKHOUSE_CLIENT} --query "INSERT INTO $mt_table VALUES (1, 2020, now() - INTERVAL 10 DAY), (2, 2020, now() - INTERVAL 10 DAY)" + ${CLICKHOUSE_CLIENT} --query "INSERT INTO $mt_table VALUES (3, 2020, now() - INTERVAL 10 DAY)" + ${CLICKHOUSE_CLIENT} --query "INSERT INTO $mt_table VALUES (4, 2021, now())" + + ${CLICKHOUSE_CLIENT} --query "SYSTEM START MOVES $mt_table" + wait_for_export_ttl "$mt_table" "2020" 1 + + echo "Destination after the first group" + ${CLICKHOUSE_CLIENT} --query "SELECT id, year FROM $s3_table ORDER BY id" + + echo "Tasks" + ${CLICKHOUSE_CLIENT} --query "SELECT source, status, parts_count, partition_id FROM system.distributed_exports WHERE source_database = currentDatabase() AND source_table = '$mt_table'" + + echo "Partition 2021 is not due" + ${CLICKHOUSE_CLIENT} --query "SELECT eligible_parts, exported_parts FROM system.ttl_exports WHERE database = currentDatabase() AND table = '$mt_table' AND partition_id = '2021'" + + echo "Parts in different export states are not merged" + ${CLICKHOUSE_CLIENT} --query "INSERT INTO $mt_table VALUES (5, 2020, now())" + ${CLICKHOUSE_CLIENT} --query "SYSTEM START MERGES $mt_table" + ${CLICKHOUSE_CLIENT} --query "OPTIMIZE TABLE $mt_table PARTITION ID '2020' FINAL SETTINGS optimize_throw_if_noop = 1" 2>&1 | grep -q "different export states" && echo "refused" + ${CLICKHOUSE_CLIENT} --query "SYSTEM STOP MERGES $mt_table" + + echo "A part that becomes due later is exported on its own" + ${CLICKHOUSE_CLIENT} --query "INSERT INTO $mt_table VALUES (6, 2020, now() - INTERVAL 10 DAY)" + wait_for_export_ttl "$mt_table" "2020" 2 + ${CLICKHOUSE_CLIENT} --query "SELECT id, year FROM $s3_table ORDER BY id" + ${CLICKHOUSE_CLIENT} --query "SELECT count(), countIf(status = 'COMPLETED'), sum(parts_count) FROM system.distributed_exports WHERE source_database = currentDatabase() AND source_table = '$mt_table'" + + echo "Removing the TTL forgets what was exported" + ${CLICKHOUSE_CLIENT} --query "ALTER TABLE $mt_table REMOVE TTL" + local i=0 + while [ "$i" -lt 120 ] && [ "$(${CLICKHOUSE_CLIENT} --query "SELECT count() FROM system.ttl_exports WHERE database = currentDatabase() AND table = '$mt_table'")" != "0" ] + do + sleep 0.5 + i=$((i + 1)) + done + ${CLICKHOUSE_CLIENT} --query "SELECT count() FROM system.ttl_exports WHERE database = currentDatabase() AND table = '$mt_table'" + + ${CLICKHOUSE_CLIENT} --query "DROP TABLE $mt_table" + ${CLICKHOUSE_CLIENT} --query "DROP TABLE $s3_table" +} + +# vi: ft=bash From bcaa59609b04125ad669381d3097ee542b8a3133 Mon Sep 17 00:00:00 2001 From: Arthur Passos Date: Mon, 28 Sep 2026 11:47:54 -0300 Subject: [PATCH 02/15] more tests --- docs/en/antalya/ttl_export.md | 2 +- src/Storages/MergeTree/ExportTaskUtils.cpp | 15 +- src/Storages/MergeTree/MergeTreeData.cpp | 4 + src/Storages/TTLDescription.cpp | 10 +- src/Storages/TTLDescription.h | 5 + .../test_export_ttl/test_ttl_expressions.py | 283 ++++++++++++++++++ .../05057_export_ttl_expressions.reference | 8 + .../05057_export_ttl_expressions.sql | 95 ++++++ 8 files changed, 414 insertions(+), 8 deletions(-) create mode 100644 tests/integration/test_export_ttl/test_ttl_expressions.py create mode 100644 tests/queries/0_stateless/05057_export_ttl_expressions.reference create mode 100644 tests/queries/0_stateless/05057_export_ttl_expressions.sql diff --git a/docs/en/antalya/ttl_export.md b/docs/en/antalya/ttl_export.md index 18a7593762e0..d92140b8e8e4 100644 --- a/docs/en/antalya/ttl_export.md +++ b/docs/en/antalya/ttl_export.md @@ -54,7 +54,7 @@ A destination that is compatible only for some groups is accepted. With a source ## How parts are exported {#how-parts-are-exported} -A part becomes eligible once the maximum TTL value of its rows is due, the same rule move TTL uses, so a part is exported as a whole. The TTL does not have to be aligned with the partition key. A part written before the expression was added becomes eligible after `ALTER TABLE ... MATERIALIZE TTL`, which `ALTER TABLE ... MODIFY TTL` runs by default. +A part becomes eligible once the maximum TTL value of its rows is due, the same rule move TTL uses, so a part is exported as a whole. The TTL does not have to be aligned with the partition key. When it is not, a part may hold rows that are due at different times, and is exported once the last of them is due. A due part that no group has claimed yet may also merge with a part of its partition that is not due, and its rows then wait for the rows of that part. A part written before the expression was added becomes eligible after `ALTER TABLE ... MATERIALIZE TTL`, which `ALTER TABLE ... MODIFY TTL` runs by default. The eligible parts of a partition are collected into a group and exported together, when any of the following holds: diff --git a/src/Storages/MergeTree/ExportTaskUtils.cpp b/src/Storages/MergeTree/ExportTaskUtils.cpp index f7ea51db14f0..a420f85cae11 100644 --- a/src/Storages/MergeTree/ExportTaskUtils.cpp +++ b/src/Storages/MergeTree/ExportTaskUtils.cpp @@ -24,6 +24,7 @@ #include #include #include +#include #include #include #include @@ -1242,9 +1243,10 @@ namespace return true; } - /// `canBeSafelyCast` only widens numbers, but every `Date` fits in a `Date32` and every `DateTime` in a - /// `DateTime64` of any scale. Iceberg stores them as `date` and `timestamp`, so an Iceberg table declares - /// `Date32` and `DateTime64(6)` columns once it reloads its schema from its metadata. + /// `canBeSafelyCast` only widens numbers, but every `Date` fits in a `Date32`, every `DateTime` in a + /// `DateTime64` of any scale, and every `DateTime64` in one of a larger scale unless that scale is 9, + /// whose range ends in 2262. Iceberg stores them as `date` and `timestamp`, so an Iceberg table + /// declares `Date32` and `DateTime64(6)` columns once it reloads its schema from its metadata. bool isDateOrTimeWidening(const DataTypePtr & source_type, const DataTypePtr & destination_type) { if (isNullableOrLowCardinalityNullable(source_type) && !isNullableOrLowCardinalityNullable(destination_type)) @@ -1252,6 +1254,13 @@ namespace const auto source = removeNullable(removeLowCardinality(source_type)); const auto destination = removeNullable(removeLowCardinality(destination_type)); + if (isDateTime64(source) && isDateTime64(destination)) + { + const auto source_scale = assert_cast(*source).getScale(); + const auto destination_scale = assert_cast(*destination).getScale(); + return destination_scale >= source_scale && (destination_scale < 9 || source_scale == 9); + } + return (isDate(source) && isDate32(destination)) || (isDateTime(source) && isDateTime64(destination)); } diff --git a/src/Storages/MergeTree/MergeTreeData.cpp b/src/Storages/MergeTree/MergeTreeData.cpp index 512c4aba97e0..32f5ad705bbe 100644 --- a/src/Storages/MergeTree/MergeTreeData.cpp +++ b/src/Storages/MergeTree/MergeTreeData.cpp @@ -1337,6 +1337,10 @@ void MergeTreeData::validateExportTTL( throw Exception(ErrorCodes::SUPPORT_IS_DISABLED, "`TTL ... EXPORT TO TABLE` requires the server setting `allow_experimental_export_merge_tree_partition`"); + /// It decides when rows are copied elsewhere once and for all, so it must be a deterministic + /// function of the rows even when suspicious TTL expressions are allowed. + export_ttls.front().checkExpressionIsStrict(local_context); + const auto destination_id = getExportTTLDestination(table_id, export_ttls.front()); if (destination_id.database_name == table_id.database_name && destination_id.table_name == table_id.table_name) throw Exception(ErrorCodes::BAD_TTL_EXPRESSION, "A table cannot be the destination of its own EXPORT TTL"); diff --git a/src/Storages/TTLDescription.cpp b/src/Storages/TTLDescription.cpp index 95b4cc19b03c..f366fcdc197e 100644 --- a/src/Storages/TTLDescription.cpp +++ b/src/Storages/TTLDescription.cpp @@ -1237,6 +1237,11 @@ ExpressionAndSets TTLDescription::buildExpression(const ContextPtr & context) co return buildExpressionAndSets(ast, expression_columns, context); } +void TTLDescription::checkExpressionIsStrict(const ContextPtr & context) const +{ + checkTTLExpression(buildExpression(context).expression, result_column, /*allow_suspicious=*/ false); +} + ExpressionAndSets TTLDescription::buildWhereExpression(const ContextPtr & context) const { if (where_expression_ast) @@ -1385,10 +1390,7 @@ TTLDescription TTLDescription::getTTLFromAST( } } - /// An export TTL decides when rows are copied elsewhere once and for all, so it must be a - /// deterministic function of the rows even when suspicious TTL expressions are allowed. - const bool allow_suspicious = result.mode != TTLMode::EXPORT && context->getSettingsRef()[Setting::allow_suspicious_ttl_expressions]; - checkTTLExpression(expression, result.result_column, is_attach || allow_suspicious); + checkTTLExpression(expression, result.result_column, is_attach || context->getSettingsRef()[Setting::allow_suspicious_ttl_expressions]); if (where_expression && !is_attach && !context->getSettingsRef()[Setting::allow_suspicious_ttl_expressions]) checkTTLExpressionForAggregateFunctions(where_expression, /*expression_kind=*/ "WHERE "); diff --git a/src/Storages/TTLDescription.h b/src/Storages/TTLDescription.h index f775e663c42e..983be2974692 100644 --- a/src/Storages/TTLDescription.h +++ b/src/Storages/TTLDescription.h @@ -60,6 +60,11 @@ struct TTLDescription /// Expression actions evaluated from AST ExpressionAndSets buildExpression(const ContextPtr & context) const; + /// Throws unless the expression is a deterministic function of the columns with a date or time + /// result, whatever `allow_suspicious_ttl_expressions` is. The callers of `getTTLFromAST` pass that + /// setting as `is_attach`, so the checks there cannot tell it from an `ATTACH`. + void checkExpressionIsStrict(const ContextPtr & context) const; + /// Result column of this TTL expression String result_column; diff --git a/tests/integration/test_export_ttl/test_ttl_expressions.py b/tests/integration/test_export_ttl/test_ttl_expressions.py new file mode 100644 index 000000000000..7613e24e560c --- /dev/null +++ b/tests/integration/test_export_ttl/test_ttl_expressions.py @@ -0,0 +1,283 @@ +import time + +from helpers.export_partition_helpers import unique_suffix + +from .common import ( + active_parts, + assert_exactly_once, + assert_iceberg_files_partitioned, + assert_one_snapshot_per_task, + completed_ttl_tasks, + create_iceberg, + create_source, + iceberg_ids, + ttl_rows, + ttl_tasks, + wait_for_partitions_exported, + wait_until, +) + +CLUSTER_INSTANCES = ["replica1"] + +# `EXPORT` TTL expressions other than a constant interval after a `DateTime` column, against the +# partition key of the source and the partition spec of the Iceberg destination: an interval read +# from a column, a TTL that does not follow the partition key, functions of one or several columns, +# `Date`, `DateTime64` and `MATERIALIZED` columns, a `DELETE` TTL of another expression, and a +# change of the expression. + +# No merge is assigned, so parts are exported as inserted. +NO_MERGES = {"max_bytes_to_merge_at_max_space_in_pool": 1} + + +def make_tables(node, engine, columns, partition_by, spec, ttl, extra_ttl="", destination_columns=None, settings=None): + suffix = unique_suffix() + mt_table, iceberg_table = f"expr_mt_{suffix}", f"expr_iceberg_{suffix}" + create_iceberg(node, iceberg_table, columns=destination_columns or columns, partition_by=spec) + create_source( + node, mt_table, columns, partition_by, f"{ttl} EXPORT TO TABLE {iceberg_table}{extra_ttl}", + engine=engine, settings=settings, + ) + return mt_table, iceberg_table + + +def partitions_where(node, table, condition): + return node.query(f"SELECT DISTINCT _partition_id FROM {table} WHERE {condition} ORDER BY 1").split() + + +def values_of(node, table, expression, condition): + return {int(x) for x in node.query(f"SELECT DISTINCT {expression} FROM {table} WHERE {condition}").split()} + + +def assert_nothing_exported(node, table, partition_ids): + rows = ttl_rows(node, table) + for partition_id in partition_ids: + row = rows.get(partition_id) + assert row is None or (row["exported_parts"] == 0 and row["eligible_parts"] == 0), row + + +def test_retention_from_a_column(cluster, source_engine): + """Each row is kept for the number of days in its `retention` column. The source is partitioned by + it, so the rows of a part are due together: the partitions that are due are exported, the others + are not. The destination then declares `eventDate` as `Date32`, read back from the Iceberg + metadata, and a part inserted later is exported to it too.""" + node = cluster.instances["replica1"] + mt_table, iceberg_table = make_tables( + node, source_engine, "id Int64, eventDate Date, retention Int32", + "(eventDate, retention)", "(eventDate, retention)", "eventDate + toIntervalDay(retention)", + ) + due = "eventDate + toIntervalDay(retention) <= now()" + + node.query( + f"INSERT INTO {mt_table} VALUES (1, today() - 10, 5), (2, today() - 10, 30), (3, today() - 40, 30)," + f" (4, today() - 2, 1), (5, today(), 7)" + ) + due_partitions = partitions_where(node, mt_table, due) + assert len(due_partitions) == 3, due_partitions + wait_for_partitions_exported(node, mt_table, due_partitions) + time.sleep(3) + assert_exactly_once(iceberg_ids(node, iceberg_table), [1, 3, 4]) + assert_nothing_exported(node, mt_table, partitions_where(node, mt_table, f"NOT ({due})")) + + node.query(f"INSERT INTO {mt_table} VALUES (6, today() - 20, 10)") + wait_until(lambda: iceberg_ids(node, iceberg_table) == [1, 3, 4, 6], 60, "The part inserted later was not exported") + wait_for_partitions_exported(node, mt_table, partitions_where(node, mt_table, due)) + + assert assert_iceberg_files_partitioned(node, iceberg_table, "retention", "retention") == {1, 5, 10, 30} + assert assert_iceberg_files_partitioned(node, iceberg_table, "eventDate", "toRelativeDayNum(eventDate)") == values_of( + node, mt_table, "toRelativeDayNum(eventDate)", due + ) + assert_one_snapshot_per_task(node, mt_table, iceberg_table) + + +def test_part_is_exported_once_all_its_rows_are_due(cluster, source_engine): + """With a TTL that does not follow the partition key, a part holds rows that are due at different + times, and it is exported once the last of them is due.""" + node = cluster.instances["replica1"] + mt_table, iceberg_table = make_tables( + node, source_engine, "id Int64, customer_id Int64, eventDate Date", + "customer_id", "customer_id", "eventDate + INTERVAL 30 DAY", settings=NO_MERGES, + ) + + node.query(f"INSERT INTO {mt_table} VALUES (1, 1, today() - 40), (2, 1, today() - 35)") + node.query(f"INSERT INTO {mt_table} VALUES (3, 1, today() - 50), (4, 1, today())") + node.query(f"INSERT INTO {mt_table} VALUES (5, 2, today() - 45), (6, 3, today() - 10)") + wait_for_partitions_exported(node, mt_table, ["1", "2"]) + time.sleep(3) + + assert_exactly_once(iceberg_ids(node, iceberg_table), [1, 2, 5]) + row = ttl_rows(node, mt_table)["1"] + assert row["exported_parts"] == 1 and row["eligible_parts"] == 0, row + assert len(active_parts(node, mt_table, "1")) == 2 + assert_nothing_exported(node, mt_table, ["3"]) + assert assert_iceberg_files_partitioned(node, iceberg_table, "customer_id", "customer_id") == {1, 2} + assert_one_snapshot_per_task(node, mt_table, iceberg_table) + + +def test_due_part_merged_with_a_part_not_due_waits_for_it(cluster, source_engine): + """A due part that no group has claimed yet, e.g. while its group is collected, may merge with a + part of the same partition that is not due. The merged part is exported only once all its rows are + due, so the rows of the due part wait for those of the other one. This pins the current behaviour.""" + node = cluster.instances["replica1"] + mt_table, iceberg_table = make_tables( + node, source_engine, "id Int64, customer_id Int64, eventDate Date", + "customer_id", "customer_id", "eventDate + INTERVAL 30 DAY", + ) + + # Stopped moves pause the TTL export, so the due part is not claimed before the merge. + node.query(f"SYSTEM STOP MOVES {mt_table}") + node.query(f"INSERT INTO {mt_table} VALUES (1, 1, today() - 40)") + node.query(f"INSERT INTO {mt_table} VALUES (2, 1, today())") + def merged(): + node.query(f"OPTIMIZE TABLE {mt_table} PARTITION ID '1' FINAL") + return len(active_parts(node, mt_table, "1")) == 1 + + wait_until(merged, 60, "The parts were not merged", interval=1) + + node.query(f"SYSTEM START MOVES {mt_table}") + wait_until(lambda: "1" in ttl_rows(node, mt_table), 30, "The partition is not shown") + time.sleep(3) + + assert_nothing_exported(node, mt_table, ["1"]) + assert ttl_tasks(node, mt_table) == [] + assert iceberg_ids(node, iceberg_table) == [] + + +def test_ttl_of_the_start_of_the_month(cluster): + """A month is due as a whole, one month after it starts.""" + node = cluster.instances["replica1"] + mt_table, iceberg_table = make_tables( + node, "MergeTree", "id Int64, eventDate Date", + "toYYYYMM(eventDate)", "toMonthNumSinceEpoch(eventDate)", "toStartOfMonth(eventDate) + INTERVAL 1 MONTH", + ) + + node.query(f"INSERT INTO {mt_table} VALUES (1, '2020-01-05'), (2, '2020-01-31'), (3, '2020-02-01'), (4, today())") + wait_for_partitions_exported(node, mt_table, ["202001", "202002"]) + time.sleep(3) + + assert_exactly_once(iceberg_ids(node, iceberg_table), [1, 2, 3]) + assert assert_iceberg_files_partitioned(node, iceberg_table, "eventDate", "toMonthNumSinceEpoch(eventDate)") == values_of( + node, mt_table, "toMonthNumSinceEpoch(eventDate)", "id != 4" + ) + assert_nothing_exported(node, mt_table, partitions_where(node, mt_table, "id = 4")) + assert_one_snapshot_per_task(node, mt_table, iceberg_table) + + +def test_datetime64_column(cluster): + """A `DateTime64(3)` column is exported with its milliseconds to the `timestamp` column of Iceberg, + which holds microseconds and is declared as `DateTime64(6)`.""" + node = cluster.instances["replica1"] + mt_table, iceberg_table = make_tables( + node, "MergeTree", "id Int64, ts DateTime64(3)", + "toDate(ts)", "toRelativeDayNum(ts)", "ts + INTERVAL 1 HOUR", + destination_columns="id Int64, ts DateTime64(6)", + ) + + node.query(f"INSERT INTO {mt_table} VALUES (1, '2020-01-05 10:20:30.123'), (2, '2020-01-05 23:59:59.999'), (3, now64(3))") + wait_for_partitions_exported(node, mt_table, ["20200105"]) + node.query(f"INSERT INTO {mt_table} VALUES (4, '2020-01-06 00:00:00.001')") + wait_until(lambda: iceberg_ids(node, iceberg_table) == [1, 2, 4], 60, "The part inserted later was not exported") + wait_for_partitions_exported(node, mt_table, ["20200105", "20200106"]) + + assert node.query(f"SELECT id, ts FROM {iceberg_table} ORDER BY id") == ( + "1\t2020-01-05 10:20:30.123000\n2\t2020-01-05 23:59:59.999000\n4\t2020-01-06 00:00:00.001000\n" + ) + assert assert_iceberg_files_partitioned(node, iceberg_table, "ts", "toRelativeDayNum(ts)") == values_of( + node, mt_table, "toRelativeDayNum(ts)", "id != 3" + ) + assert_nothing_exported(node, mt_table, partitions_where(node, mt_table, "id = 3")) + assert_one_snapshot_per_task(node, mt_table, iceberg_table) + + +def test_ttl_of_a_materialized_column(cluster): + """The TTL and the partition key use a `MATERIALIZED` column, which is exported as an ordinary one.""" + node = cluster.instances["replica1"] + mt_table, iceberg_table = make_tables( + node, "MergeTree", "id Int64, ts DateTime, eventDate Date MATERIALIZED toDate(ts)", + "toYYYYMM(eventDate)", "toMonthNumSinceEpoch(eventDate)", "eventDate + INTERVAL 1 DAY", + destination_columns="id Int64, ts DateTime, eventDate Date", + ) + + node.query(f"INSERT INTO {mt_table} (id, ts) VALUES (1, '2020-03-10 10:00:00'), (2, now())") + wait_for_partitions_exported(node, mt_table, ["202003"]) + time.sleep(3) + + assert node.query(f"SELECT id, toDateTime(ts), eventDate FROM {iceberg_table} ORDER BY id") == "1\t2020-03-10 10:00:00\t2020-03-10\n" + assert_nothing_exported(node, mt_table, partitions_where(node, mt_table, "id = 2")) + assert_one_snapshot_per_task(node, mt_table, iceberg_table) + + +def test_ttl_of_several_columns(cluster): + """A row is due a week after the later of its creation and its last update.""" + node = cluster.instances["replica1"] + mt_table, iceberg_table = make_tables( + node, "MergeTree", "id Int64, created DateTime, updated DateTime", + "toYYYYMM(created)", "toMonthNumSinceEpoch(created)", "greatest(created, updated) + INTERVAL 7 DAY", + ) + + node.query( + f"INSERT INTO {mt_table} VALUES (1, '2020-01-05 10:00:00', '2020-01-06 10:00:00')," + f" (2, '2020-02-05 10:00:00', now()), (3, '2020-03-05 10:00:00', '2020-03-06 10:00:00')" + ) + wait_for_partitions_exported(node, mt_table, ["202001", "202003"]) + time.sleep(3) + + assert_exactly_once(iceberg_ids(node, iceberg_table), [1, 3]) + assert_nothing_exported(node, mt_table, ["202002"]) + assert_one_snapshot_per_task(node, mt_table, iceberg_table) + + +def test_delete_ttl_waits_for_the_rows_of_its_part_not_due_for_export(cluster, source_engine): + """With a `DELETE` TTL of another expression, a row due for deletion stays while its part holds + rows that are not due for export yet: the delete gate holds the part, and it is neither exported + nor are its rows deleted. A part whose rows are all due is exported, and then deleted.""" + node = cluster.instances["replica1"] + mt_table, iceberg_table = make_tables( + node, source_engine, "id Int64, customer_id Int64, eventDate Date", + "customer_id", "customer_id", "eventDate + INTERVAL 30 DAY", extra_ttl=", eventDate + INTERVAL 90 DAY DELETE", + ) + + node.query(f"INSERT INTO {mt_table} VALUES (1, 1, today() - 100), (2, 1, today()), (3, 2, today() - 100)") + wait_for_partitions_exported(node, mt_table, ["2"]) + + def deleted(): + node.query(f"OPTIMIZE TABLE {mt_table} PARTITION ID '2' FINAL") + return node.query(f"SELECT count() FROM {mt_table} WHERE customer_id = 2") == "0\n" + + wait_until(deleted, 60, "The exported rows were not deleted", interval=1) + + node.query(f"OPTIMIZE TABLE {mt_table} PARTITION ID '1' FINAL") + node.query(f"ALTER TABLE {mt_table} MATERIALIZE TTL", settings={"mutations_sync": 2}) + assert node.query(f"SELECT groupArray(id) FROM (SELECT id FROM {mt_table} ORDER BY id)") == "[1,2]\n" + wait_until( + lambda: ttl_rows(node, mt_table).get("1", {}).get("parts_held_by_delete_gate") == 1, 30, + "The held part is not shown", + ) + assert_nothing_exported(node, mt_table, ["1"]) + assert_exactly_once(iceberg_ids(node, iceberg_table), [3]) + assert_one_snapshot_per_task(node, mt_table, iceberg_table) + + +def test_changing_the_ttl_to_a_retention_column(cluster, source_engine): + """`MODIFY TTL` recalculates when each part is due under the new expression. A part exported under + the previous expression is not exported again, as the destination is the same.""" + node = cluster.instances["replica1"] + mt_table, iceberg_table = make_tables( + node, source_engine, "id Int64, eventDate Date, retention Int32", "id", "id", "eventDate + INTERVAL 30 DAY", + ) + + node.query(f"INSERT INTO {mt_table} VALUES (1, today() - 40, 5), (2, today() - 10, 5), (3, today() - 10, 30)") + wait_for_partitions_exported(node, mt_table, ["1"]) + assert_exactly_once(iceberg_ids(node, iceberg_table), [1]) + + node.query( + f"ALTER TABLE {mt_table} MODIFY TTL eventDate + toIntervalDay(retention) EXPORT TO TABLE {iceberg_table}", + settings={"mutations_sync": 2}, + ) + wait_until(lambda: iceberg_ids(node, iceberg_table) == [1, 2], 60, "The part due under the new expression was not exported") + wait_for_partitions_exported(node, mt_table, ["1", "2"]) + time.sleep(3) + + assert_exactly_once(iceberg_ids(node, iceberg_table), [1, 2]) + assert [len(task["parts"]) for task in completed_ttl_tasks(node, mt_table)] == [1, 1] + assert_nothing_exported(node, mt_table, ["3"]) + assert_one_snapshot_per_task(node, mt_table, iceberg_table) diff --git a/tests/queries/0_stateless/05057_export_ttl_expressions.reference b/tests/queries/0_stateless/05057_export_ttl_expressions.reference new file mode 100644 index 000000000000..2508fb1acb9f --- /dev/null +++ b/tests/queries/0_stateless/05057_export_ttl_expressions.reference @@ -0,0 +1,8 @@ +TTL d + toIntervalDay(retention) EXPORT TO TABLE export_ttl_expressions_destination SETTINGS +TTL toStartOfMonth(d) + toIntervalMonth(1) EXPORT TO TABLE export_ttl_expressions_destination SETTINGS +TTL greatest(t, toDateTime(d)) + toIntervalDay(1) EXPORT TO TABLE export_ttl_expressions_destination SETTINGS +TTL ifNull(nd, d) + toIntervalDay(1) EXPORT TO TABLE export_ttl_expressions_destination SETTINGS +TTL ifNull(nd, d) + toIntervalDay(1) EXPORT TO TABLE export_ttl_expressions_destination, t + toIntervalSecond(rand() % 10) SETTINGS +export_ttl_expressions_micros_to_micros +export_ttl_expressions_millis_to_micros +export_ttl_expressions_seconds_to_nanos diff --git a/tests/queries/0_stateless/05057_export_ttl_expressions.sql b/tests/queries/0_stateless/05057_export_ttl_expressions.sql new file mode 100644 index 000000000000..d5961d032320 --- /dev/null +++ b/tests/queries/0_stateless/05057_export_ttl_expressions.sql @@ -0,0 +1,95 @@ +-- Tags: no-fasttest +-- no-fasttest: the destinations are S3 tables. + +-- The expressions an `EXPORT` TTL accepts, and the date and time columns it may widen when exporting. + +DROP TABLE IF EXISTS export_ttl_expressions_source; +DROP TABLE IF EXISTS export_ttl_expressions_random; +DROP TABLE IF EXISTS export_ttl_expressions_destination; + +SET allow_experimental_export_ttl = 1; + +CREATE TABLE export_ttl_expressions_destination (id UInt64, d Date, retention UInt16, t DateTime, nd Nullable(Date)) +ENGINE = S3(s3_conn, filename = 'export_ttl_expressions_destination', format = Parquet, partition_strategy = 'hive') +PARTITION BY d; + +-- An interval read from a column. +CREATE TABLE export_ttl_expressions_source (id UInt64, d Date, retention UInt16, t DateTime, nd Nullable(Date)) +ENGINE = MergeTree PARTITION BY d ORDER BY id +TTL d + toIntervalDay(retention) EXPORT TO TABLE export_ttl_expressions_destination; +SELECT extract(create_table_query, 'TTL .*? SETTINGS') FROM system.tables WHERE database = currentDatabase() AND name = 'export_ttl_expressions_source'; + +-- Functions of one or several columns. +ALTER TABLE export_ttl_expressions_source MODIFY TTL toStartOfMonth(d) + INTERVAL 1 MONTH EXPORT TO TABLE export_ttl_expressions_destination; +SELECT extract(create_table_query, 'TTL .*? SETTINGS') FROM system.tables WHERE database = currentDatabase() AND name = 'export_ttl_expressions_source'; + +ALTER TABLE export_ttl_expressions_source MODIFY TTL greatest(t, toDateTime(d)) + INTERVAL 1 DAY EXPORT TO TABLE export_ttl_expressions_destination; +SELECT extract(create_table_query, 'TTL .*? SETTINGS') FROM system.tables WHERE database = currentDatabase() AND name = 'export_ttl_expressions_source'; + +ALTER TABLE export_ttl_expressions_source MODIFY TTL ifNull(nd, d) + INTERVAL 1 DAY EXPORT TO TABLE export_ttl_expressions_destination; +SELECT extract(create_table_query, 'TTL .*? SETTINGS') FROM system.tables WHERE database = currentDatabase() AND name = 'export_ttl_expressions_source'; + +-- The result must be a date or a time, not a `Nullable` one. +ALTER TABLE export_ttl_expressions_source MODIFY TTL nd + INTERVAL 1 DAY EXPORT TO TABLE export_ttl_expressions_destination; -- { serverError BAD_TTL_EXPRESSION } + +-- The expression must be a deterministic function of the columns. +CREATE TABLE export_ttl_expressions_random (id UInt64, d Date, retention UInt16, t DateTime, nd Nullable(Date)) +ENGINE = MergeTree PARTITION BY d ORDER BY id +TTL t + toIntervalSecond(rand() % 10) EXPORT TO TABLE export_ttl_expressions_destination; -- { serverError BAD_ARGUMENTS } +ALTER TABLE export_ttl_expressions_source MODIFY TTL t + toIntervalSecond(rand() % 10) EXPORT TO TABLE export_ttl_expressions_destination; -- { serverError BAD_ARGUMENTS } +ALTER TABLE export_ttl_expressions_source MODIFY TTL now() + INTERVAL 1 DAY EXPORT TO TABLE export_ttl_expressions_destination; -- { serverError BAD_ARGUMENTS } + +-- Even when suspicious TTL expressions are allowed, which still applies to the TTL that does not export. +SET allow_suspicious_ttl_expressions = 1; +CREATE TABLE export_ttl_expressions_random (id UInt64, d Date, retention UInt16, t DateTime, nd Nullable(Date)) +ENGINE = MergeTree PARTITION BY d ORDER BY id +TTL t + toIntervalSecond(rand() % 10) EXPORT TO TABLE export_ttl_expressions_destination; -- { serverError BAD_ARGUMENTS } +ALTER TABLE export_ttl_expressions_source MODIFY TTL t + toIntervalSecond(rand() % 10) EXPORT TO TABLE export_ttl_expressions_destination; -- { serverError BAD_ARGUMENTS } +ALTER TABLE export_ttl_expressions_source MODIFY TTL now() + INTERVAL 1 DAY EXPORT TO TABLE export_ttl_expressions_destination; -- { serverError BAD_ARGUMENTS } +ALTER TABLE export_ttl_expressions_source MODIFY TTL ifNull(nd, d) + INTERVAL 1 DAY EXPORT TO TABLE export_ttl_expressions_destination, t + toIntervalSecond(rand() % 10) DELETE; +SELECT extract(create_table_query, 'TTL .*? SETTINGS') FROM system.tables WHERE database = currentDatabase() AND name = 'export_ttl_expressions_source'; +SET allow_suspicious_ttl_expressions = 0; + +DROP TABLE export_ttl_expressions_source; +DROP TABLE export_ttl_expressions_destination; + +-- A `DateTime64` column may be exported to one of a larger scale, e.g. to the microseconds of an +-- Iceberg `timestamp`, unless that scale is 9, whose range ends in 2262. +DROP TABLE IF EXISTS export_ttl_expressions_micros; +DROP TABLE IF EXISTS export_ttl_expressions_nanos; +DROP TABLE IF EXISTS export_ttl_expressions_millis_to_micros; +DROP TABLE IF EXISTS export_ttl_expressions_micros_to_micros; +DROP TABLE IF EXISTS export_ttl_expressions_seconds_to_nanos; +DROP TABLE IF EXISTS export_ttl_expressions_nanos_to_micros; +DROP TABLE IF EXISTS export_ttl_expressions_millis_to_nanos; + +CREATE TABLE export_ttl_expressions_micros (id UInt64, d Date, ts DateTime64(6)) +ENGINE = S3(s3_conn, filename = 'export_ttl_expressions_micros', format = Parquet, partition_strategy = 'hive') +PARTITION BY d; + +CREATE TABLE export_ttl_expressions_nanos (id UInt64, d Date, ts DateTime64(9)) +ENGINE = S3(s3_conn, filename = 'export_ttl_expressions_nanos', format = Parquet, partition_strategy = 'hive') +PARTITION BY d; + +CREATE TABLE export_ttl_expressions_millis_to_micros (id UInt64, d Date, ts DateTime64(3)) ENGINE = MergeTree PARTITION BY d ORDER BY id +TTL ts + INTERVAL 1 DAY EXPORT TO TABLE export_ttl_expressions_micros; + +CREATE TABLE export_ttl_expressions_micros_to_micros (id UInt64, d Date, ts DateTime64(6)) ENGINE = MergeTree PARTITION BY d ORDER BY id +TTL ts + INTERVAL 1 DAY EXPORT TO TABLE export_ttl_expressions_micros; + +CREATE TABLE export_ttl_expressions_seconds_to_nanos (id UInt64, d Date, ts DateTime) ENGINE = MergeTree PARTITION BY d ORDER BY id +TTL ts + INTERVAL 1 DAY EXPORT TO TABLE export_ttl_expressions_nanos; + +CREATE TABLE export_ttl_expressions_nanos_to_micros (id UInt64, d Date, ts DateTime64(9)) ENGINE = MergeTree PARTITION BY d ORDER BY id +TTL ts + INTERVAL 1 DAY EXPORT TO TABLE export_ttl_expressions_micros; -- { serverError INCOMPATIBLE_COLUMNS } + +CREATE TABLE export_ttl_expressions_millis_to_nanos (id UInt64, d Date, ts DateTime64(3)) ENGINE = MergeTree PARTITION BY d ORDER BY id +TTL ts + INTERVAL 1 DAY EXPORT TO TABLE export_ttl_expressions_nanos; -- { serverError INCOMPATIBLE_COLUMNS } + +SELECT name FROM system.tables WHERE database = currentDatabase() AND match(name, '_to_') ORDER BY name; + +DROP TABLE export_ttl_expressions_millis_to_micros; +DROP TABLE export_ttl_expressions_micros_to_micros; +DROP TABLE export_ttl_expressions_seconds_to_nanos; +DROP TABLE export_ttl_expressions_micros; +DROP TABLE export_ttl_expressions_nanos; From 0c4df476f0bfd9d29dfe50bf518529d27554b70c Mon Sep 17 00:00:00 2001 From: Arthur Passos Date: Mon, 28 Sep 2026 12:00:54 -0300 Subject: [PATCH 03/15] fix fast test --- .../0_stateless/02221_system_zookeeper_unrestricted.reference | 4 ++++ .../02221_system_zookeeper_unrestricted_like.reference | 4 ++++ 2 files changed, 8 insertions(+) diff --git a/tests/queries/0_stateless/02221_system_zookeeper_unrestricted.reference b/tests/queries/0_stateless/02221_system_zookeeper_unrestricted.reference index 2a6bc4c98833..8242a05aa572 100644 --- a/tests/queries/0_stateless/02221_system_zookeeper_unrestricted.reference +++ b/tests/queries/0_stateless/02221_system_zookeeper_unrestricted.reference @@ -20,6 +20,10 @@ creator_info creator_info deduplication_hashes deduplication_hashes +export_features +export_features +export_fence +export_fence exports exports failed_parts diff --git a/tests/queries/0_stateless/02221_system_zookeeper_unrestricted_like.reference b/tests/queries/0_stateless/02221_system_zookeeper_unrestricted_like.reference index db94343ec003..a0ab94f91807 100644 --- a/tests/queries/0_stateless/02221_system_zookeeper_unrestricted_like.reference +++ b/tests/queries/0_stateless/02221_system_zookeeper_unrestricted_like.reference @@ -9,6 +9,8 @@ columns columns creator_info deduplication_hashes +export_features +export_fence exports failed_parts flags @@ -52,6 +54,8 @@ columns columns creator_info deduplication_hashes +export_features +export_fence exports failed_parts flags From 13a766a3a0ef6695398abc505f655225fcc83a38 Mon Sep 17 00:00:00 2001 From: Arthur Passos Date: Mon, 28 Sep 2026 16:17:18 -0300 Subject: [PATCH 04/15] more vibe coded changes --- src/Core/SettingsChangesHistory.cpp | 2 +- .../ExportReplicatedMergeTreeTaskManifest.h | 20 +- src/Storages/ExportRetriedTask.h | 106 ++++++++++ src/Storages/MergeTree/ExportTTLIndex.cpp | 71 ++----- src/Storages/MergeTree/ExportTTLIndex.h | 30 +-- src/Storages/MergeTree/ExportTTLScheduler.cpp | 197 +++++++----------- src/Storages/MergeTree/ExportTTLScheduler.h | 50 +++-- src/Storages/MergeTree/ExportTaskInfo.h | 3 +- src/Storages/MergeTree/ExportTaskUtils.cpp | 35 +--- src/Storages/MergeTree/ExportTaskUtils.h | 11 +- .../MergeTree/MergeTreeExportTTLScheduler.cpp | 1 + .../MergeTree/MergeTreeExportTTLScheduler.h | 6 +- src/Storages/MergeTree/MergeTreeExportTask.h | 16 +- .../MergeTreeExportTaskScheduler.cpp | 10 +- src/Storages/MergeTree/MergeTreeSettings.cpp | 2 +- .../MergeTree/ReplicatedExportTTLIndex.cpp | 5 - .../MergeTree/ReplicatedExportTTLIndex.h | 4 +- .../ReplicatedExportTTLScheduler.cpp | 113 +++++----- .../MergeTree/ReplicatedExportTTLScheduler.h | 4 +- .../MergeTree/ReplicatedExportTaskUpdater.cpp | 2 +- .../MergeTree/registerStorageMergeTree.cpp | 10 + .../tests/gtest_export_task_ordering.cpp | 22 +- .../tests/gtest_export_ttl_index.cpp | 11 + .../StorageSystemDistributedExports.cpp | 2 +- tests/integration/test_export_ttl/common.py | 4 +- .../test_export_ttl/test_failures.py | 15 +- .../test_export_ttl/test_replication.py | 37 ++-- .../test_export_ttl/test_server_setting.py | 56 +++++ 28 files changed, 477 insertions(+), 368 deletions(-) create mode 100644 src/Storages/ExportRetriedTask.h create mode 100644 tests/integration/test_export_ttl/test_server_setting.py diff --git a/src/Core/SettingsChangesHistory.cpp b/src/Core/SettingsChangesHistory.cpp index 300693a0b234..07fd1a0a71bc 100644 --- a/src/Core/SettingsChangesHistory.cpp +++ b/src/Core/SettingsChangesHistory.cpp @@ -1358,7 +1358,7 @@ const VersionToSettingsChangesMap & getMergeTreeSettingsChangesHistory() {"ttl_export_batch_max_delay_seconds", 600, 600, "New setting"}, {"ttl_export_batch_min_bytes", 256_MiB, 256_MiB, "New setting"}, {"ttl_export_max_parts_per_group", 100, 100, "New setting"}, - {"ttl_export_max_bytes_per_group", 10_GiB, 10_GiB, "New setting"}, + {"ttl_export_max_bytes_per_group", 100_GiB, 100_GiB, "New setting"}, {"ttl_export_max_concurrent_groups", 4, 4, "New setting"}, {"ttl_export_settings_profile", "", "", "New setting"}, {"packed_skip_index_max_bytes", 0, 0, "New setting. Pack any skip-index substream whose serialized on-disk size is at most this many bytes into a single skp_idx.packed archive per part; larger substreams stay in the standalone skp_idx_.idx2 / .mrk2 layout. Decision is made per substream at write time."}, diff --git a/src/Storages/ExportReplicatedMergeTreeTaskManifest.h b/src/Storages/ExportReplicatedMergeTreeTaskManifest.h index 832fbd83d126..6d3136fd97ad 100644 --- a/src/Storages/ExportReplicatedMergeTreeTaskManifest.h +++ b/src/Storages/ExportReplicatedMergeTreeTaskManifest.h @@ -7,6 +7,7 @@ #include #include #include +#include #include #include #include @@ -169,9 +170,9 @@ struct ExportReplicatedMergeTreeTaskManifest String destination_table; /// UUID of the destination table when the task was created, empty if it has none. String destination_uuid; - /// TTL export only: transaction ids of earlier tasks that failed to export some of these parts. - /// The commit checks whether any of them landed at the destination after all. - std::vector retry_of; + /// TTL export only: earlier tasks that failed to export some of these parts and whose commit may + /// still land. The commit checks whether any of them landed at the destination after all. + ExportRetriedTasks retry_of; String source_replica; size_t number_of_parts; std::vector parts; @@ -214,12 +215,7 @@ struct ExportReplicatedMergeTreeTaskManifest if (!destination_uuid.empty()) json.set("destination_uuid", destination_uuid); if (!retry_of.empty()) - { - Poco::JSON::Array::Ptr retry_of_array = new Poco::JSON::Array(); - for (const auto & transaction : retry_of) - retry_of_array->add(transaction); - json.set("retry_of", retry_of_array); - } + json.set("retry_of", ExportRetriedTaskUtils::toJSON(retry_of)); json.set("source_replica", source_replica); json.set("number_of_parts", number_of_parts); @@ -286,11 +282,7 @@ struct ExportReplicatedMergeTreeTaskManifest if (json->has("destination_uuid")) manifest.destination_uuid = json->getValue("destination_uuid"); if (json->has("retry_of")) - { - const auto retry_of_array = json->getArray("retry_of"); - for (size_t i = 0; i < retry_of_array->size(); ++i) - manifest.retry_of.push_back(retry_of_array->getElement(static_cast(i))); - } + manifest.retry_of = ExportRetriedTaskUtils::fromJSON(json->getArray("retry_of")); manifest.source_replica = json->getValue("source_replica"); manifest.number_of_parts = json->getValue("number_of_parts"); diff --git a/src/Storages/ExportRetriedTask.h b/src/Storages/ExportRetriedTask.h new file mode 100644 index 000000000000..70213e2a09a6 --- /dev/null +++ b/src/Storages/ExportRetriedTask.h @@ -0,0 +1,106 @@ +#pragma once + +#include +#include +#include +#include + +#include +#include +#include + +namespace DB +{ + +namespace ErrorCodes +{ + extern const int INCORRECT_DATA; +} + +/// A failed task of the `EXPORT` TTL whose parts a later task exports again, and whose commit may +/// still land at the destination, e.g. a request that the destination applies late. The later task +/// commits only the parts that no landed task exported, so it records the blocks of each of them. +struct ExportRetriedTask +{ + String transaction_id; + /// `[min_block, max_block]` of its parts, in the partition of the later task. + std::vector> block_ranges; + /// When the scheduler found it failed. + time_t failed_time = 0; + + bool operator==(const ExportRetriedTask &) const = default; +}; + +using ExportRetriedTasks = std::vector; + +namespace ExportRetriedTaskUtils +{ + +inline std::vector transactionIds(const ExportRetriedTasks & tasks) +{ + std::vector result; + result.reserve(tasks.size()); + for (const auto & task : tasks) + result.push_back(task.transaction_id); + return result; +} + +inline Poco::JSON::Array::Ptr toJSON(const ExportRetriedTasks & tasks) +{ + Poco::JSON::Array::Ptr array = new Poco::JSON::Array(); + for (const auto & task : tasks) + { + Poco::JSON::Array::Ptr ranges = new Poco::JSON::Array(); + for (const auto & [min_block, max_block] : task.block_ranges) + { + Poco::JSON::Array::Ptr range = new Poco::JSON::Array(); + range->add(min_block); + range->add(max_block); + ranges->add(range); + } + + Poco::JSON::Object::Ptr object = new Poco::JSON::Object(); + object->set("transaction_id", task.transaction_id); + object->set("block_ranges", ranges); + object->set("failed_time", static_cast(task.failed_time)); + array->add(object); + } + return array; +} + +inline ExportRetriedTasks fromJSON(const Poco::JSON::Array::Ptr & array) +{ + ExportRetriedTasks tasks; + if (!array) + return tasks; + + for (size_t i = 0; i < array->size(); ++i) + { + const auto object = array->getObject(static_cast(i)); + if (!object) + throw Exception(ErrorCodes::INCORRECT_DATA, "Invalid `retry_of` element in an export task descriptor"); + + ExportRetriedTask task; + task.transaction_id = object->getValue("transaction_id"); + task.failed_time = static_cast(object->optValue("failed_time", 0)); + + if (const auto ranges = object->getArray("block_ranges")) + { + for (size_t j = 0; j < ranges->size(); ++j) + { + const auto range = ranges->getArray(static_cast(j)); + if (!range || range->size() != 2) + throw Exception(ErrorCodes::INCORRECT_DATA, + "Invalid block range of retried task {} in an export task descriptor", task.transaction_id); + task.block_ranges.emplace_back(range->getElement(0), range->getElement(1)); + } + } + + tasks.push_back(std::move(task)); + } + return tasks; +} + +} + +} diff --git a/src/Storages/MergeTree/ExportTTLIndex.cpp b/src/Storages/MergeTree/ExportTTLIndex.cpp index 9cbb6461b1c2..02ae11b52f16 100644 --- a/src/Storages/MergeTree/ExportTTLIndex.cpp +++ b/src/Storages/MergeTree/ExportTTLIndex.cpp @@ -185,59 +185,6 @@ std::unique_ptr ExportTTLIndexSnapshot::build( return snapshot; } -String ExportTTLSchedulerState::toJSONString() const -{ - Poco::JSON::Object json; - json.set("scheduler_replica", scheduler_replica); - - Poco::JSON::Object::Ptr partitions_object = new Poco::JSON::Object(); - for (const auto & [partition_id, partition] : partitions) - { - Poco::JSON::Object::Ptr partition_object = new Poco::JSON::Object(); - partition_object->set("last_error", partition.last_error); - partition_object->set("first_eligible_time", static_cast(partition.first_eligible_time)); - partition_object->set("last_new_part_time", static_cast(partition.last_new_part_time)); - partitions_object->set(partition_id, partition_object); - } - json.set("partitions", partitions_object); - - std::ostringstream oss; // STYLE_CHECK_ALLOW_STD_STRING_STREAM - oss.exceptions(std::ios::failbit); - Poco::JSON::Stringifier::stringify(json, oss); - return oss.str(); -} - -ExportTTLSchedulerState ExportTTLSchedulerState::fromJSONString(const String & json_string) -{ - ExportTTLSchedulerState state; - if (json_string.empty()) - return state; - - Poco::JSON::Parser parser; - const auto json = parser.parse(json_string).extract(); - if (!json) - throw Exception(ErrorCodes::INCORRECT_DATA, "The state of the TTL export scheduler is not a JSON object"); - - if (json->has("scheduler_replica")) - state.scheduler_replica = json->getValue("scheduler_replica"); - - if (const auto partitions_object = json->getObject("partitions")) - { - for (const auto & partition_id : partitions_object->getNames()) - { - const auto partition_object = partitions_object->getObject(partition_id); - if (!partition_object) - continue; - auto & partition = state.partitions[partition_id]; - partition.last_error = partition_object->optValue("last_error", ""); - partition.first_eligible_time = static_cast(partition_object->optValue("first_eligible_time", 0)); - partition.last_new_part_time = static_cast(partition_object->optValue("last_new_part_time", 0)); - } - } - - return state; -} - namespace ExportTTLUtils { @@ -258,6 +205,24 @@ std::vector rangesOfParts(const std::vector & part_na return ExportFenceUtils::compactRanges(std::move(ranges)); } +std::vector> toBlockRanges(const std::vector & ranges) +{ + std::vector> result; + result.reserve(ranges.size()); + for (const auto & range : ranges) + result.emplace_back(range.min_block, range.max_block); + return result; +} + +std::vector fromBlockRanges(const String & partition_id, const std::vector> & block_ranges) +{ + std::vector ranges; + ranges.reserve(block_ranges.size()); + for (const auto & [min_block, max_block] : block_ranges) + ranges.emplace_back(partition_id, min_block, max_block, /* level */ 0, /* mutation */ 0); + return ExportFenceUtils::compactRanges(std::move(ranges)); +} + } } diff --git a/src/Storages/MergeTree/ExportTTLIndex.h b/src/Storages/MergeTree/ExportTTLIndex.h index b7c247df3698..1c1c3e717b08 100644 --- a/src/Storages/MergeTree/ExportTTLIndex.h +++ b/src/Storages/MergeTree/ExportTTLIndex.h @@ -4,9 +4,9 @@ #include #include -#include #include #include +#include #include namespace DB @@ -89,30 +89,6 @@ struct ExportTTLIndexSnapshot using ExportTTLIndexSnapshotPtr = std::shared_ptr; -/// What the replica that schedules the `EXPORT` TTL of a table knows beyond the index: stored with -/// the index of the destination whenever it changes, so every replica shows it, and a replica that -/// takes over the scheduling resumes the batching windows. -struct ExportTTLSchedulerState -{ - struct Partition - { - String last_error; - time_t first_eligible_time = 0; - time_t last_new_part_time = 0; - - bool operator==(const Partition &) const = default; - }; - - String scheduler_replica; - /// By partition id. A partition without an error and waiting for nothing is omitted. - std::map partitions; - - bool operator==(const ExportTTLSchedulerState &) const = default; - - String toJSONString() const; - static ExportTTLSchedulerState fromJSONString(const String & json_string); -}; - namespace ExportTTLUtils { /// Identifies a destination in the export index. The UUID, if not empty, tells apart a table that @@ -121,6 +97,10 @@ namespace ExportTTLUtils /// Ranges of the parts named `part_names`, compacted. std::vector rangesOfParts(const std::vector & part_names, MergeTreeDataFormatVersion format_version); + + /// `[min_block, max_block]` of `ranges`, as recorded for a retried task (see `ExportRetriedTask`). + std::vector> toBlockRanges(const std::vector & ranges); + std::vector fromBlockRanges(const String & partition_id, const std::vector> & block_ranges); } } diff --git a/src/Storages/MergeTree/ExportTTLScheduler.cpp b/src/Storages/MergeTree/ExportTTLScheduler.cpp index 82023407c474..4e75921b8747 100644 --- a/src/Storages/MergeTree/ExportTTLScheduler.cpp +++ b/src/Storages/MergeTree/ExportTTLScheduler.cpp @@ -40,6 +40,10 @@ namespace MergeTreeSetting namespace { +/// How long after its task failed a commit may still land: a destination may apply a request after +/// its sender gave up waiting for it. +constexpr time_t late_commit_window_seconds = 600; + /// The rule of move TTL: a part is eligible once the maximum TTL value of its rows is due. A part /// without the TTL info (written before the TTL was added and not materialized) is never eligible. bool isEligible(const IMergeTreeDataPart & part, const TTLDescriptions & export_ttls, time_t now) @@ -153,8 +157,6 @@ UInt64 ExportTTLScheduler::run() batches.clear(); last_errors.clear(); info_by_partition.clear(); - written_state.reset(); - was_scheduler = false; current_destination_key = destination_key; } current_destination_error = destination_error; @@ -198,43 +200,8 @@ UInt64 ExportTTLScheduler::run() return period * 1000; } - /// The replica that schedules owns the batching windows; the others show what it stored. - ExportTTLSchedulerState stored_state; - bool resume_stored_state = false; - { - bool was = false; - { - std::lock_guard lock(mutex); - was = was_scheduler; - was_scheduler = is_scheduler; - } - - if (!is_scheduler || !was) - { - if (auto state = readSchedulerState(destination_key)) - { - stored_state = std::move(*state); - resume_stored_state = is_scheduler; - } - } - } - - if (resume_stored_state) - { - std::lock_guard lock(mutex); - for (const auto & [partition_id, partition] : stored_state.partitions) - { - auto & batch = batches[partition_id]; - batch.first_eligible_time = partition.first_eligible_time; - batch.last_new_part_time = partition.last_new_part_time; - if (!partition.last_error.empty()) - last_errors[partition_id] = partition.last_error; - } - LOG_INFO(log, "This replica schedules the EXPORT TTL now, resuming the state stored by {}", stored_state.scheduler_replica); - } - const bool act = is_scheduler && !isPaused(); - const String scheduler_replica = is_scheduler ? getReplicaName() : stored_state.scheduler_replica; + const String scheduler_replica = is_scheduler ? getReplicaName() : getSchedulerReplica(); const time_t now = time(nullptr); @@ -278,32 +245,27 @@ UInt64 ExportTTLScheduler::run() PartitionBatch batch; String last_error; - if (is_scheduler) { std::lock_guard lock(mutex); batch = batches[partition_id]; if (const auto it = last_errors.find(partition_id); it != last_errors.end()) last_error = it->second; } - else if (const auto it = stored_state.partitions.find(partition_id); it != stored_state.partitions.end()) - { - batch.first_eligible_time = it->second.first_eligible_time; - batch.last_new_part_time = it->second.last_new_part_time; - last_error = it->second.last_error; - } PartitionView view; try { - view = observePartition(versioned.entry, parts, export_ttls, now, std::move(batch), is_scheduler, task_states); - view.info.last_error = last_error; + view = observePartition(versioned.entry, parts, export_ttls, now, std::move(batch), task_states); + + /// The error of acting on the partition is shown until the scheduler acts on it again. + if (is_scheduler && !act) + view.info.last_error = last_error; if (act) { /// Shown as it is after acting, e.g. with the parts of a recorded commit as exported. if (const auto updated_entry = actOnPartition(destination_key, destination, std::move(versioned), view, now, in_flight, task_states, context)) - view = observePartition(*updated_entry, parts, export_ttls, now, std::move(view.batch), is_scheduler, task_states); - view.info.last_error.clear(); + view = observePartition(*updated_entry, parts, export_ttls, now, std::move(view.batch), task_states); } } catch (...) @@ -323,19 +285,14 @@ UInt64 ExportTTLScheduler::run() next_tick = std::min(next_tick, view.next_eligible_time); std::lock_guard lock(mutex); - if (is_scheduler) - { - batches[partition_id] = std::move(view.batch); - if (view.info.last_error.empty()) - last_errors.erase(partition_id); - else - last_errors[partition_id] = view.info.last_error; - } + batches[partition_id] = std::move(view.batch); + if (view.info.last_error.empty()) + last_errors.erase(partition_id); + else + last_errors[partition_id] = view.info.last_error; info_by_partition[partition_id] = std::move(view.info); } - ExportTTLSchedulerState state_to_write; - bool write_state = false; { std::lock_guard lock(mutex); std::erase_if(info_by_partition, [&](const auto & item) { return !partition_ids.contains(item.first); }); @@ -346,39 +303,6 @@ UInt64 ExportTTLScheduler::run() for (const auto & [_, info] : info_by_partition) held += info.parts_held_by_delete_gate; parts_held_by_delete_gate.changeTo(held); - - if (is_scheduler) - { - state_to_write.scheduler_replica = scheduler_replica; - for (const auto & partition_id : partition_ids) - { - ExportTTLSchedulerState::Partition partition; - if (const auto it = batches.find(partition_id); it != batches.end()) - { - partition.first_eligible_time = it->second.first_eligible_time; - partition.last_new_part_time = it->second.last_new_part_time; - } - if (const auto it = last_errors.find(partition_id); it != last_errors.end()) - partition.last_error = it->second; - if (partition != ExportTTLSchedulerState::Partition{}) - state_to_write.partitions[partition_id] = std::move(partition); - } - write_state = !written_state || *written_state != state_to_write; - } - } - - if (write_state) - { - try - { - writeSchedulerState(destination_key, state_to_write); - std::lock_guard lock(mutex); - written_state = std::move(state_to_write); - } - catch (...) - { - tryLogCurrentException(log, "While storing the state of the TTL export scheduler"); - } } return static_cast(std::max(1, next_tick - now)) * 1000; @@ -412,7 +336,7 @@ ExportTTLScheduler::ResolvedClaims ExportTTLScheduler::resolveClaims( case TaskStatus::KILLED: case TaskStatus::MISSING: /// E.g. a commit that landed and then the task timed out before it was marked completed. - if (destination->isExportTransactionCommitted(transaction_id, context)) + if (state.reached_commit && destination->isExportTransactionCommitted(transaction_id, context)) { LOG_INFO(log, "Export task {} of partition {} did not complete, but it committed to the destination", transaction_id, entry.partition_id); @@ -422,7 +346,6 @@ ExportTTLScheduler::ResolvedClaims ExportTTLScheduler::resolveClaims( } result.failed.push_back(transaction_id); - result.retry_of.insert(result.retry_of.end(), state.retry_of.begin(), state.retry_of.end()); break; } } @@ -430,13 +353,61 @@ ExportTTLScheduler::ResolvedClaims ExportTTLScheduler::resolveClaims( return result; } +ExportRetriedTasks ExportTTLScheduler::collectRetriedTasks( + const ExportTTLIndexEntry & entry, + const std::vector & failed, + const StoragePtr & destination, + time_t now, + TaskStates & task_states, + const ContextPtr & context) +{ + ExportRetriedTasks result; + std::unordered_set added; + + for (const auto & transaction_id : failed) + { + const auto & state = getCachedTaskState(task_states, transaction_id); + + /// A task that did not export all its parts never committed, and a missing one was either + /// never created or checked when it went missing. + if (state.status != TaskStatus::MISSING && state.reached_commit && added.insert(transaction_id).second) + result.push_back(ExportRetriedTask{ + .transaction_id = transaction_id, + .block_ranges = ExportTTLUtils::toBlockRanges(entry.claimed.at(transaction_id)), + .failed_time = now, + }); + + for (const auto & retried : state.retry_of) + { + if (added.contains(retried.transaction_id)) + continue; + + /// Past the window, and with no commit of it in progress, the task is checked once more, + /// and not retried by the next groups if it did not land. + const bool may_land = now < retried.failed_time + late_commit_window_seconds + || isCommitInProgress(retried.transaction_id) + || destination->isExportTransactionCommitted(retried.transaction_id, context); + + if (!may_land) + { + LOG_DEBUG(log, "Export task {} of partition {} did not commit, it is no longer checked", retried.transaction_id, entry.partition_id); + continue; + } + + added.insert(retried.transaction_id); + result.push_back(retried); + } + } + + return result; +} + ExportTTLScheduler::PartitionView ExportTTLScheduler::observePartition( const ExportTTLIndexEntry & entry, const std::vector & parts, const TTLDescriptions & export_ttls, time_t now, PartitionBatch batch, - bool update_batch, TaskStates & task_states) { const auto settings = storage.getSettings(); @@ -489,27 +460,20 @@ ExportTTLScheduler::PartitionView ExportTTLScheduler::observePartition( } eligible_ranges = ExportFenceUtils::compactRanges(std::move(eligible_ranges)); - if (update_batch) + for (const auto & part : eligible) { - /// Resumed from the stored state: the parts seen by the previous scheduler are not new. - if (batch.seen_ranges.empty() && batch.first_eligible_time) - batch.seen_ranges = eligible_ranges; - - for (const auto & part : eligible) + if (!ExportFenceUtils::isCoveredByUnion(part->info, batch.seen_ranges)) { - if (!ExportFenceUtils::isCoveredByUnion(part->info, batch.seen_ranges)) - { - batch.last_new_part_time = now; - if (!batch.first_eligible_time) - batch.first_eligible_time = now; - } + batch.last_new_part_time = now; + if (!batch.first_eligible_time) + batch.first_eligible_time = now; } - - batch.seen_ranges = std::move(eligible_ranges); - if (eligible.empty()) - batch = PartitionBatch{}; } + batch.seen_ranges = std::move(eligible_ranges); + if (eligible.empty()) + batch = PartitionBatch{}; + info.eligible_parts = eligible.size(); info.first_eligible_time = eligible.empty() ? 0 : batch.first_eligible_time; @@ -614,12 +578,11 @@ std::optional ExportTTLScheduler::actOnPartition( for (const auto & part : group.parts) group_bytes += part->getBytesOnDisk(); + std::vector retried; for (const auto & transaction_id : resolved.failed) if (failed_with_parts.contains(transaction_id)) - group.retry_of.push_back(transaction_id); - group.retry_of.insert(group.retry_of.end(), resolved.retry_of.begin(), resolved.retry_of.end()); - std::sort(group.retry_of.begin(), group.retry_of.end()); - group.retry_of.erase(std::unique(group.retry_of.begin(), group.retry_of.end()), group.retry_of.end()); + retried.push_back(transaction_id); + group.retry_of = collectRetriedTasks(entry, retried, destination, now, task_states, context); for (const auto & part : view.shippable) { @@ -646,10 +609,10 @@ std::optional ExportTTLScheduler::actOnPartition( if (startGroup(group, context)) { ++in_flight; - task_states.insert_or_assign(group.transaction_id, TaskState{TaskStatus::PENDING, group.retry_of}); + task_states.insert_or_assign(group.transaction_id, TaskState{.status = TaskStatus::PENDING, .reached_commit = false, .retry_of = group.retry_of}); LOG_INFO(log, "Started export task {} of {} part(s) of partition {} to {}{}", group.transaction_id, group.parts.size(), partition_id, destination->getStorageID().getNameForLogs(), - group.retry_of.empty() ? "" : fmt::format(", retrying {}", fmt::join(group.retry_of, ", "))); + retried.empty() ? "" : fmt::format(", retrying {}", fmt::join(retried, ", "))); return std::move(group.entry.entry); } diff --git a/src/Storages/MergeTree/ExportTTLScheduler.h b/src/Storages/MergeTree/ExportTTLScheduler.h index db88a3775819..97befe414198 100644 --- a/src/Storages/MergeTree/ExportTTLScheduler.h +++ b/src/Storages/MergeTree/ExportTTLScheduler.h @@ -2,6 +2,7 @@ #include #include +#include #include #include #include @@ -31,7 +32,7 @@ struct ExportTTLPartitionInfo size_t eligible_bytes = 0; /// Parts held back from the delete TTL because they are not exported yet. size_t parts_held_by_delete_gate = 0; - /// When the scheduler first saw an eligible part that is not exported, 0 if there is none. + /// When this replica first saw an eligible part that is not exported, 0 if there is none. time_t first_eligible_time = 0; /// When the next group can start at the latest, 0 if nothing waits. time_t next_group_time = 0; @@ -52,11 +53,12 @@ struct ExportTTLPartitionInfo /// Eligible parts of a partition are shipped once no new eligible part appeared for the batching /// window, once the first of them waited for the maximum delay, or once they reach the size /// threshold. A task that failed without committing keeps its parts claimed, and they are retried -/// first by the next group, which records the failed tasks in `retry_of`. +/// first by the next group, which records in `retry_of` the failed tasks whose commit may still land. /// -/// Every replica observes the state of every partition, which `system.ttl_exports` shows. Only -/// the replica holding the scheduler lock acts on it: it resolves finished tasks, starts groups and -/// stores its state (`ExportTTLSchedulerState`) for the other replicas. +/// Every replica observes the state of every partition, which `system.ttl_exports` shows, and +/// tracks the batching windows of the parts it has, so a replica that takes over the scheduling +/// continues them. Only the replica holding the scheduler lock acts: it resolves finished tasks and +/// starts groups. /// /// Engines implement access to the index and to their export tasks. class ExportTTLScheduler @@ -87,7 +89,10 @@ class ExportTTLScheduler struct TaskState { TaskStatus status = TaskStatus::MISSING; - std::vector retry_of; + /// Whether it exported all its parts, which it does before committing: a task that did not + /// cannot have committed. Unknown counts as reached. + bool reached_commit = true; + ExportRetriedTasks retry_of; }; struct GroupToStart @@ -97,7 +102,7 @@ class ExportTTLScheduler StoragePtr destination; String partition_id; std::vector parts; - std::vector retry_of; + ExportRetriedTasks retry_of; /// The index entry with the claim of the group, to be stored with a check of its version. ExportTTLVersionedEntry entry; }; @@ -116,13 +121,16 @@ class ExportTTLScheduler /// unless its database is `Replicated`, so they identify it by name only. virtual bool identifiesDestinationByUUID() const = 0; - /// The state stored by the replica that schedules, nothing if there is none. - virtual std::optional readSchedulerState(const String & destination_key) = 0; - virtual void writeSchedulerState(const String & destination_key, const ExportTTLSchedulerState & state) = 0; virtual String getReplicaName() const = 0; + /// The replica holding the scheduler lock, empty if there is none. + virtual String getSchedulerReplica() = 0; + virtual TaskState getTaskState(const String & transaction_id) = 0; + /// Whether a replica is in the commit of the task, e.g. one that started before the task failed. + virtual bool isCommitInProgress(const String & transaction_id) = 0; + /// Stores `entry` with a check of its version. Returns false on a conflict. virtual bool updateIndexEntry(const String & destination_key, const ExportTTLVersionedEntry & entry) = 0; @@ -164,17 +172,13 @@ class ExportTTLScheduler using TaskStates = std::unordered_map; mutable std::mutex mutex; - /// By partition id, for the current destination. Kept by the replica that schedules. + /// By partition id, for the current destination. std::map batches; std::map last_errors; std::map info_by_partition; String current_destination_key; String current_destination_error; - /// Whether the previous tick scheduled, so a replica that takes over resumes the stored state. - bool was_scheduler = false; - std::optional written_state; - /// False once there is neither an `EXPORT` TTL nor an index left, so ticks do no Keeper reads. bool may_have_index = true; @@ -183,30 +187,38 @@ class ExportTTLScheduler ContextPtr makeContext() const; /// Resolves the claims of the index entry that do not belong to a task in flight. Returns the - /// failed tasks whose parts are still claimed, with their own `retry_of`, and the task in flight. + /// failed tasks whose parts are still claimed, and the task in flight. struct ResolvedClaims { std::vector failed; - std::vector retry_of; String in_flight; bool changed = false; }; ResolvedClaims resolveClaims(ExportTTLIndexEntry & entry, const StoragePtr & destination, TaskStates & task_states, const ContextPtr & context); + /// The failed tasks among `failed` and the ones they retried whose commit may still land, for + /// the `retry_of` of the group that retries the parts of `failed`. + ExportRetriedTasks collectRetriedTasks( + const ExportTTLIndexEntry & entry, + const std::vector & failed, + const StoragePtr & destination, + time_t now, + TaskStates & task_states, + const ContextPtr & context); + const TaskState & getCachedTaskState(TaskStates & task_states, const String & transaction_id); /// Kills the TTL tasks of a destination that is no longer the destination of the TTL, and /// removes its index once none of them holds a claim. void cleanupDestination(const String & destination_key, const std::map & index); - /// Read-only. `update_batch` is set on the replica that schedules, which owns the batching windows. + /// Read-only, except for the batching window of the partition, which it returns in the view. PartitionView observePartition( const ExportTTLIndexEntry & entry, const std::vector & parts, const TTLDescriptions & export_ttls, time_t now, PartitionBatch batch, - bool update_batch, TaskStates & task_states); /// On the replica that schedules: resolves finished tasks and starts a group if one is due. diff --git a/src/Storages/MergeTree/ExportTaskInfo.h b/src/Storages/MergeTree/ExportTaskInfo.h index 9310dddb87a3..74c21f4c5db5 100644 --- a/src/Storages/MergeTree/ExportTaskInfo.h +++ b/src/Storages/MergeTree/ExportTaskInfo.h @@ -65,7 +65,8 @@ struct ExportTaskInfo /// What created the task: `query` for `EXPORT PARTITION`, `ttl` for a `TTL ... EXPORT` expression. String source; - /// TTL export only: earlier tasks that failed to export some of this task's parts. + /// TTL export only: earlier tasks that failed to export some of this task's parts, and whose + /// commit may still land. std::vector retry_of; }; diff --git a/src/Storages/MergeTree/ExportTaskUtils.cpp b/src/Storages/MergeTree/ExportTaskUtils.cpp index a420f85cae11..d9b3e8e88d0c 100644 --- a/src/Storages/MergeTree/ExportTaskUtils.cpp +++ b/src/Storages/MergeTree/ExportTaskUtils.cpp @@ -489,26 +489,18 @@ namespace } std::vector getRangesCommittedByRetriedTasks( - const std::vector & retry_of, + const ExportRetriedTasks & retry_of, const StoragePtr & destination_storage, - const std::function>(const String &)> & get_task_parts, - MergeTreeDataFormatVersion format_version, + const String & partition_id, const ContextPtr & context) { std::vector ranges; - for (const auto & transaction_id : retry_of) + for (const auto & retried : retry_of) { - if (!destination_storage->isExportTransactionCommitted(transaction_id, context)) + if (!destination_storage->isExportTransactionCommitted(retried.transaction_id, context)) continue; - const auto parts = get_task_parts(transaction_id); - if (!parts) - throw Exception(ErrorCodes::CORRUPTED_DATA, - "Export task {} committed to the destination, but its description is lost, so it is not known which parts it exported. " - "Refusing to commit the task that retries it, which could export them again", - transaction_id); - - const auto task_ranges = ExportTTLUtils::rangesOfParts(*parts, format_version); + const auto task_ranges = ExportTTLUtils::fromBlockRanges(partition_id, retried.block_ranges); ranges.insert(ranges.end(), task_ranges.begin(), task_ranges.end()); } return ExportFenceUtils::compactRanges(std::move(ranges)); @@ -585,19 +577,8 @@ namespace std::vector paths_to_commit = exported.paths; if (!manifest.retry_of.empty()) { - const auto exports_path = fs::path(entry_path).parent_path(); const auto committed_ranges = getRangesCommittedByRetriedTasks( - manifest.retry_of, - destination_storage, - [&](const String & transaction_id) -> std::optional> - { - String metadata_json; - if (!zk->tryGet(exports_path / transaction_id / "metadata.json", metadata_json)) - return std::nullopt; - return ExportReplicatedMergeTreeTaskManifest::fromJsonString(metadata_json).parts; - }, - source_storage.format_version, - context_in); + manifest.retry_of, destination_storage, partition_id, context_in); if (!committed_ranges.empty()) { @@ -678,8 +659,8 @@ namespace export_fence.ensureDestination(zk, destination_key, fmt::format("{}.{}", manifest.destination_database, manifest.destination_table)); versioned.entry.commitClaim(manifest.transaction_id, ExportTTLUtils::rangesOfParts(manifest.parts, source_storage.format_version)); - for (const auto & transaction_id : manifest.retry_of) - versioned.entry.releaseClaim(transaction_id); + for (const auto & retried : manifest.retry_of) + versioned.entry.releaseClaim(retried.transaction_id); export_fence.appendUpdateEntryOps(ops, destination_key, versioned.entry, versioned.version); } diff --git a/src/Storages/MergeTree/ExportTaskUtils.h b/src/Storages/MergeTree/ExportTaskUtils.h index 82f25888dc61..8bfa499e207d 100644 --- a/src/Storages/MergeTree/ExportTaskUtils.h +++ b/src/Storages/MergeTree/ExportTaskUtils.h @@ -13,6 +13,7 @@ #include #include #include "Storages/IStorage.h" +#include #include #include #include @@ -58,14 +59,12 @@ namespace ExportTaskUtils /// Appends the ops that create the nodes of a new export task at `task_path` (`/exports/`). void appendCreateExportTaskOps(Coordination::Requests & ops, const std::string & task_path, const ExportReplicatedMergeTreeTaskManifest & manifest); - /// Block ranges committed to `destination_storage` by the tasks in `retry_of` that landed, e.g. a - /// commit that the destination applied after the task was considered failed. `get_task_parts` - /// returns the parts of a task, or nothing if it is unknown. + /// Block ranges of `partition_id` committed to `destination_storage` by the tasks in `retry_of` + /// that landed, e.g. a commit that the destination applied after the task was considered failed. std::vector getRangesCommittedByRetriedTasks( - const std::vector & retry_of, + const ExportRetriedTasks & retry_of, const StoragePtr & destination_storage, - const std::function>(const String &)> & get_task_parts, - MergeTreeDataFormatVersion format_version, + const String & partition_id, const ContextPtr & context); /// Parts of `part_names` whose rows are not in `committed_ranges`. diff --git a/src/Storages/MergeTree/MergeTreeExportTTLScheduler.cpp b/src/Storages/MergeTree/MergeTreeExportTTLScheduler.cpp index ee41e7edaea3..9e50c7e580b0 100644 --- a/src/Storages/MergeTree/MergeTreeExportTTLScheduler.cpp +++ b/src/Storages/MergeTree/MergeTreeExportTTLScheduler.cpp @@ -195,6 +195,7 @@ ExportTTLScheduler::TaskState MergeTreeExportTTLScheduler::getTaskState(const St case MergeTreeExportTask::Status::FAILED: state.status = TaskStatus::FAILED; break; case MergeTreeExportTask::Status::KILLED: state.status = TaskStatus::KILLED; break; } + state.reached_commit = task->allPartsDone(); state.retry_of = task->retry_of; return state; } diff --git a/src/Storages/MergeTree/MergeTreeExportTTLScheduler.h b/src/Storages/MergeTree/MergeTreeExportTTLScheduler.h index 07b3a19cef9d..aa0d912430bf 100644 --- a/src/Storages/MergeTree/MergeTreeExportTTLScheduler.h +++ b/src/Storages/MergeTree/MergeTreeExportTTLScheduler.h @@ -61,11 +61,11 @@ class MergeTreeExportTTLScheduler final : public ExportTTLScheduler bool isPaused() override; ExportTTLIndexSnapshotPtr getIndexSnapshot() override { return index.getSnapshot(); } bool identifiesDestinationByUUID() const override { return true; } - /// A single replica keeps its state in memory. - std::optional readSchedulerState(const String &) override { return std::nullopt; } - void writeSchedulerState(const String &, const ExportTTLSchedulerState &) override {} String getReplicaName() const override { return {}; } + String getSchedulerReplica() override { return {}; } TaskState getTaskState(const String & transaction_id) override; + /// A task is not marked failed or killed while it commits, and a restart ends every commit. + bool isCommitInProgress(const String &) override { return false; } bool updateIndexEntry(const String & destination_key, const ExportTTLVersionedEntry & entry) override { return index.update(destination_key, entry); } bool startGroup(const GroupToStart & group, const ContextPtr & context) override; bool isPartBeingMerged(const MergeTreeDataPartPtr & part) override; diff --git a/src/Storages/MergeTree/MergeTreeExportTask.h b/src/Storages/MergeTree/MergeTreeExportTask.h index 9967e1a61755..5fd485fc7ff4 100644 --- a/src/Storages/MergeTree/MergeTreeExportTask.h +++ b/src/Storages/MergeTree/MergeTreeExportTask.h @@ -9,6 +9,7 @@ #include #include #include +#include #include #include @@ -62,9 +63,9 @@ struct MergeTreeExportTask String destination_uuid; time_t create_time = 0; ExportTaskSource source = ExportTaskSource::query; - /// TTL tasks whose parts this task exports again because they failed. Their commits may still + /// TTL tasks whose parts this task exports again because they failed, and whose commits may still /// land, which the commit of this task checks. - std::vector retry_of; + ExportRetriedTasks retry_of; /// Work + progress std::vector parts; @@ -161,12 +162,7 @@ struct MergeTreeExportTask json.set("create_time", create_time); json.set("source", String(magic_enum::enum_name(source))); if (!retry_of.empty()) - { - Poco::JSON::Array::Ptr retry_of_array = new Poco::JSON::Array(); - for (const auto & transaction : retry_of) - retry_of_array->add(transaction); - json.set("retry_of", retry_of_array); - } + json.set("retry_of", ExportRetriedTaskUtils::toJSON(retry_of)); json.set("status", String(magic_enum::enum_name(status))); Poco::JSON::Array::Ptr parts_array = new Poco::JSON::Array(); @@ -253,9 +249,7 @@ struct MergeTreeExportTask throw Exception(ErrorCodes::INCORRECT_DATA, "Unknown source '{}' in export task descriptor", source_str); } - if (const auto retry_of_array = json->getArray("retry_of")) - for (size_t i = 0; i < retry_of_array->size(); ++i) - task.retry_of.push_back(retry_of_array->getElement(static_cast(i))); + task.retry_of = ExportRetriedTaskUtils::fromJSON(json->getArray("retry_of")); const auto status_str = json->getValue("status"); if (const auto status = magic_enum::enum_cast(status_str)) diff --git a/src/Storages/MergeTree/MergeTreeExportTaskScheduler.cpp b/src/Storages/MergeTree/MergeTreeExportTaskScheduler.cpp index e602bd355435..738ae015dc4b 100644 --- a/src/Storages/MergeTree/MergeTreeExportTaskScheduler.cpp +++ b/src/Storages/MergeTree/MergeTreeExportTaskScheduler.cpp @@ -143,7 +143,7 @@ std::vector MergeTreeExportTaskScheduler::getInfo() const ? "" : ExportTaskUtils::getPartitionIdOfParts(descriptor.partNames(), storage.format_version); info.source = String(magic_enum::enum_name(descriptor.source)); - info.retry_of = descriptor.retry_of; + info.retry_of = ExportRetriedTaskUtils::transactionIds(descriptor.retry_of); info.transaction_id = descriptor.transaction_id; info.query_id = descriptor.query_id; info.parts = descriptor.partNames(); @@ -591,13 +591,7 @@ void MergeTreeExportTaskScheduler::tryCommit(const String & transaction_id) const auto committed_ranges = ExportTaskUtils::getRangesCommittedByRetriedTasks( descriptor_copy.retry_of, destination_storage, - [this](const String & retried_transaction_id) -> std::optional> - { - if (const auto task = getTask(retried_transaction_id)) - return task->partNames(); - return std::nullopt; - }, - storage.format_version, + ExportTaskUtils::getPartitionIdOfParts(descriptor_copy.partNames(), storage.format_version), *context); if (!committed_ranges.empty()) diff --git a/src/Storages/MergeTree/MergeTreeSettings.cpp b/src/Storages/MergeTree/MergeTreeSettings.cpp index 8615a9850873..f64246c70a91 100644 --- a/src/Storages/MergeTree/MergeTreeSettings.cpp +++ b/src/Storages/MergeTree/MergeTreeSettings.cpp @@ -839,7 +839,7 @@ Possible values: Maximum number of parts exported together by one task of the `EXPORT` TTL. Parts of a failed task are always retried together, even if there are more of them. )", EXPERIMENTAL) \ - DECLARE(UInt64, ttl_export_max_bytes_per_group, 10_GiB, R"( + DECLARE(UInt64, ttl_export_max_bytes_per_group, 100_GiB, R"( Maximum size on disk of the parts exported together by one task of the `EXPORT` TTL. A single part bigger than this is exported on its own. 0 means unlimited. )", EXPERIMENTAL) \ diff --git a/src/Storages/MergeTree/ReplicatedExportTTLIndex.cpp b/src/Storages/MergeTree/ReplicatedExportTTLIndex.cpp index 62ad5a889c60..b92f1ea07b0b 100644 --- a/src/Storages/MergeTree/ReplicatedExportTTLIndex.cpp +++ b/src/Storages/MergeTree/ReplicatedExportTTLIndex.cpp @@ -49,11 +49,6 @@ String ReplicatedExportTTLIndex::getIndexEntryPath(const String & destination_ke return fs::path(getDestinationPath(destination_key)) / "partitions" / partition_id; } -String ReplicatedExportTTLIndex::getSchedulerStatePath(const String & destination_key) const -{ - return fs::path(getDestinationPath(destination_key)) / "state"; -} - int32_t ReplicatedExportTTLIndex::readFenceVersion(const zkutil::ZooKeeperPtr & zookeeper) const { const auto path = getFencePath(); diff --git a/src/Storages/MergeTree/ReplicatedExportTTLIndex.h b/src/Storages/MergeTree/ReplicatedExportTTLIndex.h index a5e5ebd33d31..32d9104fa6b9 100644 --- a/src/Storages/MergeTree/ReplicatedExportTTLIndex.h +++ b/src/Storages/MergeTree/ReplicatedExportTTLIndex.h @@ -20,8 +20,7 @@ namespace DB /// Layout under ``: /// - `export_ttl/`: holds `ExportTTLDestination` of one destination; /// - `export_ttl//partitions/`: an `ExportTTLIndexEntry`; -/// - `export_ttl//state`: the `ExportTTLSchedulerState` of the replica that schedules; -/// - `export_ttl/scheduler_lock`: ephemeral, held by the replica that schedules TTL exports; +/// - `export_ttl/scheduler_lock`: ephemeral, held by the replica that schedules TTL exports, whose name it holds; /// - `export_fence`: bumped in every transaction that changes an index entry. A merge is assigned /// with a check of the version its predicate read the index at, so no merge is assigned from a /// stale view of the export states. @@ -35,7 +34,6 @@ class ReplicatedExportTTLIndex String getSchedulerLockPath() const; String getDestinationPath(const String & destination_key) const; String getIndexEntryPath(const String & destination_key, const String & partition_id) const; - String getSchedulerStatePath(const String & destination_key) const; /// Destination keys that have an index. std::vector listDestinations(const zkutil::ZooKeeperPtr & zookeeper) const; diff --git a/src/Storages/MergeTree/ReplicatedExportTTLScheduler.cpp b/src/Storages/MergeTree/ReplicatedExportTTLScheduler.cpp index 7a962f171766..2731a3df8506 100644 --- a/src/Storages/MergeTree/ReplicatedExportTTLScheduler.cpp +++ b/src/Storages/MergeTree/ReplicatedExportTTLScheduler.cpp @@ -82,37 +82,25 @@ ExportTTLIndexSnapshotPtr ReplicatedExportTTLScheduler::getIndexSnapshot() return replicated_storage.export_fence->getSnapshot(replicated_storage.getZooKeeper()); } -std::optional ReplicatedExportTTLScheduler::readSchedulerState(const String & destination_key) +String ReplicatedExportTTLScheduler::getReplicaName() const { - String data; - if (!replicated_storage.getZooKeeper()->tryGet(replicated_storage.export_fence->getSchedulerStatePath(destination_key), data)) - return std::nullopt; - return ExportTTLSchedulerState::fromJSONString(data); + return replicated_storage.getReplicaName(); } -void ReplicatedExportTTLScheduler::writeSchedulerState(const String & destination_key, const ExportTTLSchedulerState & state) +String ReplicatedExportTTLScheduler::getSchedulerReplica() { - const auto zookeeper = replicated_storage.getZooKeeper(); - const auto & fence = *replicated_storage.export_fence; - const auto path = fence.getSchedulerStatePath(destination_key); - const auto data = state.toJSONString(); + const auto zookeeper = replicated_storage.tryGetZooKeeper(); + if (!zookeeper || zookeeper->expired()) + return {}; - auto code = zookeeper->trySet(path, data); - if (code == Coordination::Error::ZNONODE) - { - fence.ensureDestination(zookeeper, destination_key, destination_key); - code = zookeeper->tryCreate(path, data, zkutil::CreateMode::Persistent); - if (code == Coordination::Error::ZNODEEXISTS) - code = zookeeper->trySet(path, data); - } - - if (code != Coordination::Error::ZOK) - throw zkutil::KeeperException::fromPath(code, path); + String replica; + zookeeper->tryGet(replicated_storage.export_fence->getSchedulerLockPath(), replica); + return replica; } -String ReplicatedExportTTLScheduler::getReplicaName() const +bool ReplicatedExportTTLScheduler::isCommitInProgress(const String & transaction_id) { - return replicated_storage.getReplicaName(); + return replicated_storage.getZooKeeper()->exists(fs::path(replicated_storage.zookeeper_path) / "exports" / transaction_id / "commit_lock"); } ExportTTLScheduler::TaskState ReplicatedExportTTLScheduler::getTaskState(const String & transaction_id) @@ -130,43 +118,74 @@ ExportTTLScheduler::TaskState ReplicatedExportTTLScheduler::getTaskState(const S UNREACHABLE(); }; + const auto zookeeper = replicated_storage.getZooKeeper(); + const fs::path task_path = fs::path(replicated_storage.zookeeper_path) / "exports" / transaction_id; + + /// Read from Keeper rather than from the mirror of the tasks, which may lag behind the parts + /// the task exported. Unknown counts as reached. + const auto all_parts_processed = [&](size_t parts_count) + { + Coordination::Stat stat; + if (!zookeeper->exists(task_path / "processed", &stat)) + return true; + return static_cast(stat.numChildren) >= parts_count; + }; + + TaskState state; + std::optional parts_count; + /// The in-memory mirror of the tasks may lag behind Keeper, which only delays a retry. A task it /// does not know yet, e.g. just created, is read from Keeper, so it is never taken for missing. + bool found_in_mirror = false; if (const auto tasks = replicated_storage.export_partition_manifests.get()) { const auto & by_transaction_id = tasks->get(); if (const auto it = by_transaction_id.find(transaction_id); it != by_transaction_id.end()) - return TaskState{.status = to_task_status(it->status), .retry_of = it->manifest.retry_of}; + { + state.status = to_task_status(it->status); + state.retry_of = it->manifest.retry_of; + parts_count = it->manifest.parts.size(); + found_in_mirror = true; + } } - const auto zookeeper = replicated_storage.getZooKeeper(); - const fs::path task_path = fs::path(replicated_storage.zookeeper_path) / "exports" / transaction_id; - const Strings paths{task_path / "status", task_path / "metadata.json"}; + if (!found_in_mirror) + { + const Strings paths{task_path / "status", task_path / "metadata.json"}; - auto responses = zookeeper->tryGet(paths); - responses.waitForResponses(); + auto responses = zookeeper->tryGet(paths); + responses.waitForResponses(); - TaskState state; - if (responses[0].error == Coordination::Error::ZNONODE) - return state; - if (responses[0].error != Coordination::Error::ZOK) - throw zkutil::KeeperException::fromPath(responses[0].error, paths[0]); + if (responses[0].error == Coordination::Error::ZNONODE) + return state; + if (responses[0].error != Coordination::Error::ZOK) + throw zkutil::KeeperException::fromPath(responses[0].error, paths[0]); - if (const auto status = magic_enum::enum_cast(responses[0].data)) - { - state.status = to_task_status(*status); - } - else - { - /// Treated as failed: the recovery check still decides whether it committed. - LOG_WARNING(log, "Export task {} has an unknown status {}", transaction_id, responses[0].data); - state.status = TaskStatus::FAILED; + if (const auto status = magic_enum::enum_cast(responses[0].data)) + { + state.status = to_task_status(*status); + } + else + { + /// Treated as failed: the recovery check still decides whether it committed. + LOG_WARNING(log, "Export task {} has an unknown status {}", transaction_id, responses[0].data); + state.status = TaskStatus::FAILED; + } + + if (responses[1].error == Coordination::Error::ZOK) + { + auto manifest = ExportReplicatedMergeTreeTaskManifest::fromJsonString(responses[1].data); + state.retry_of = std::move(manifest.retry_of); + parts_count = manifest.parts.size(); + } + else if (responses[1].error != Coordination::Error::ZNONODE) + { + throw zkutil::KeeperException::fromPath(responses[1].error, paths[1]); + } } - if (responses[1].error == Coordination::Error::ZOK) - state.retry_of = ExportReplicatedMergeTreeTaskManifest::fromJsonString(responses[1].data).retry_of; - else if (responses[1].error != Coordination::Error::ZNONODE) - throw zkutil::KeeperException::fromPath(responses[1].error, paths[1]); + if (parts_count && (state.status == TaskStatus::FAILED || state.status == TaskStatus::KILLED)) + state.reached_commit = all_parts_processed(*parts_count); return state; } diff --git a/src/Storages/MergeTree/ReplicatedExportTTLScheduler.h b/src/Storages/MergeTree/ReplicatedExportTTLScheduler.h index 58fe00905e74..633f1ac74bab 100644 --- a/src/Storages/MergeTree/ReplicatedExportTTLScheduler.h +++ b/src/Storages/MergeTree/ReplicatedExportTTLScheduler.h @@ -27,10 +27,10 @@ class ReplicatedExportTTLScheduler final : public ExportTTLScheduler bool isPaused() override; ExportTTLIndexSnapshotPtr getIndexSnapshot() override; bool identifiesDestinationByUUID() const override { return false; } - std::optional readSchedulerState(const String & destination_key) override; - void writeSchedulerState(const String & destination_key, const ExportTTLSchedulerState & state) override; String getReplicaName() const override; + String getSchedulerReplica() override; TaskState getTaskState(const String & transaction_id) override; + bool isCommitInProgress(const String & transaction_id) override; bool updateIndexEntry(const String & destination_key, const ExportTTLVersionedEntry & entry) override; bool startGroup(const GroupToStart & group, const ContextPtr & context) override; bool isPartBeingMerged(const MergeTreeDataPartPtr & part) override; diff --git a/src/Storages/MergeTree/ReplicatedExportTaskUpdater.cpp b/src/Storages/MergeTree/ReplicatedExportTaskUpdater.cpp index f9ed96bfa10e..d6fd42c38584 100644 --- a/src/Storages/MergeTree/ReplicatedExportTaskUpdater.cpp +++ b/src/Storages/MergeTree/ReplicatedExportTaskUpdater.cpp @@ -422,7 +422,7 @@ std::vector ReplicatedExportTaskUpdater::getExportTasksInfo() co info.destination_table = manifest.destination_table; info.partition_id = ExportTaskUtils::getPartitionIdOfParts(manifest.parts, storage.format_version); info.source = String(magic_enum::enum_name(manifest.source)); - info.retry_of = manifest.retry_of; + info.retry_of = ExportRetriedTaskUtils::transactionIds(manifest.retry_of); info.transaction_id = manifest.transaction_id; info.query_id = manifest.query_id; info.create_time = manifest.create_time; diff --git a/src/Storages/MergeTree/registerStorageMergeTree.cpp b/src/Storages/MergeTree/registerStorageMergeTree.cpp index e1e07d7d3642..213c298e5bbf 100644 --- a/src/Storages/MergeTree/registerStorageMergeTree.cpp +++ b/src/Storages/MergeTree/registerStorageMergeTree.cpp @@ -80,6 +80,7 @@ namespace ServerSetting { extern const ServerSettingsString default_replica_name; extern const ServerSettingsString default_replica_path; + extern const ServerSettingsBool allow_experimental_export_merge_tree_partition; } namespace ErrorCodes @@ -1051,6 +1052,15 @@ static StoragePtr create(const StorageFactory::Arguments & args) merging_params.allow_tuple_element_aggregation = false; } + /// Without the setting the table has neither the export index nor the merge fence, so merges + /// could mix exported parts with the others, whose rows would then never be exported. + if (metadata.hasAnyExportTTL() + && !args.getContext()->getServerSettings()[ServerSetting::allow_experimental_export_merge_tree_partition]) + throw Exception(ErrorCodes::SUPPORT_IS_DISABLED, + "Table {} has a `TTL ... EXPORT TO TABLE` expression, which requires the server setting " + "`allow_experimental_export_merge_tree_partition`", + args.table_id.getNameForLogs()); + /// Before the table is created, e.g. in Keeper. if (args.mode <= LoadingStrictnessLevel::CREATE) MergeTreeData::validateExportTTL( diff --git a/src/Storages/MergeTree/tests/gtest_export_task_ordering.cpp b/src/Storages/MergeTree/tests/gtest_export_task_ordering.cpp index d057455167f5..c48231fd9df8 100644 --- a/src/Storages/MergeTree/tests/gtest_export_task_ordering.cpp +++ b/src/Storages/MergeTree/tests/gtest_export_task_ordering.cpp @@ -4,6 +4,7 @@ #include #include #include +#include #include #include #include @@ -236,12 +237,16 @@ TEST_F(ExportTaskManifestBackCompatTest, TaskSourceAndRetryOfRoundTrip) auto manifest = makeValidManifest(); manifest.source = ExportTaskSource::ttl; manifest.destination_uuid = "00000000-0000-0000-0000-000000000001"; - manifest.retry_of = {"tx0", "tx00"}; + manifest.retry_of = { + ExportRetriedTask{.transaction_id = "tx0", .block_ranges = {{1, 3}, {7, 7}}, .failed_time = 1700000000}, + ExportRetriedTask{.transaction_id = "tx00", .block_ranges = {{1, 1}}, .failed_time = 1600000000}, + }; const auto parsed = ExportReplicatedMergeTreeTaskManifest::fromJsonString(manifest.toJsonString()); EXPECT_EQ(parsed.source, ExportTaskSource::ttl); EXPECT_EQ(parsed.destination_uuid, manifest.destination_uuid); EXPECT_EQ(parsed.retry_of, manifest.retry_of); + EXPECT_EQ(ExportRetriedTaskUtils::transactionIds(parsed.retry_of), (std::vector{"tx0", "tx00"})); /// A task of `EXPORT PARTITION` records neither. const auto query_task = ExportReplicatedMergeTreeTaskManifest::fromJsonString(makeValidManifest().toJsonString()); @@ -249,6 +254,21 @@ TEST_F(ExportTaskManifestBackCompatTest, TaskSourceAndRetryOfRoundTrip) EXPECT_TRUE(query_task.retry_of.empty()); } +TEST(ExportTaskUtils, PlainTaskRetryOfRoundTrip) +{ + MergeTreeExportTask task; + task.transaction_id = "tx1"; + task.source = ExportTaskSource::ttl; + task.parts.push_back({.part_name = "p_1_3_1", .done = true, .paths_in_destination = {"a.parquet"}}); + task.retry_of = {ExportRetriedTask{.transaction_id = "tx0", .block_ranges = {{1, 3}}, .failed_time = 1700000000}}; + + const auto parsed = MergeTreeExportTask::fromJsonString(task.toJsonString()); + EXPECT_EQ(parsed.retry_of, task.retry_of); + + task.retry_of.clear(); + EXPECT_TRUE(MergeTreeExportTask::fromJsonString(task.toJsonString()).retry_of.empty()); +} + TEST(ExportTaskUtils, PartitionIdIsDerivedFromParts) { EXPECT_EQ(ExportTaskUtils::getPartitionIdOfParts({"2020_1_1_0", "2020_2_5_1"}, MERGE_TREE_DATA_MIN_FORMAT_VERSION_WITH_CUSTOM_PARTITIONING), "2020"); diff --git a/src/Storages/MergeTree/tests/gtest_export_ttl_index.cpp b/src/Storages/MergeTree/tests/gtest_export_ttl_index.cpp index 986c5e37cdce..186cdb94e34b 100644 --- a/src/Storages/MergeTree/tests/gtest_export_ttl_index.cpp +++ b/src/Storages/MergeTree/tests/gtest_export_ttl_index.cpp @@ -110,6 +110,17 @@ TEST(ExportTTLIndex, RangesOfParts) EXPECT_EQ(blocks(ranges), (Blocks{{1, 3}, {5, 5}})); } +TEST(ExportTTLIndex, BlockRangesOfRetriedTasks) +{ + const std::vector ranges{part(1, 3), part(7, 7)}; + const auto block_ranges = ExportTTLUtils::toBlockRanges(ranges); + EXPECT_EQ(block_ranges, (Blocks{{1, 3}, {7, 7}})); + + const auto restored = ExportTTLUtils::fromBlockRanges("p", {{4, 5}, {1, 3}}); + EXPECT_EQ(blocks(restored), (Blocks{{1, 5}})); + EXPECT_EQ(restored.front().getPartitionId(), "p"); +} + TEST(ExportTTLIndex, EligibleOnceTheMaximumIsDue) { TTLDescription description; diff --git a/src/Storages/System/StorageSystemDistributedExports.cpp b/src/Storages/System/StorageSystemDistributedExports.cpp index b235054ca2a6..8ffe7a69ea0c 100644 --- a/src/Storages/System/StorageSystemDistributedExports.cpp +++ b/src/Storages/System/StorageSystemDistributedExports.cpp @@ -70,7 +70,7 @@ ColumnsDescription StorageSystemDistributedExports::getColumnsDescription() {"source", std::make_shared(), "What created the task: `query` for `ALTER TABLE ... EXPORT PARTITION`, `ttl` for the table's `TTL ... EXPORT TO TABLE` expression."}, {"retry_of", std::make_shared(std::make_shared()), - "For a TTL export task: transaction ids of earlier tasks that failed to export some of this task's parts. Its commit checks whether any of them landed at the destination after all. Empty otherwise."}, + "For a TTL export task: transaction ids of earlier tasks that failed to export some of this task's parts after exporting all of theirs, so their commit may still land. Its commit checks whether any of them landed at the destination after all. Empty otherwise."}, }; } diff --git a/tests/integration/test_export_ttl/common.py b/tests/integration/test_export_ttl/common.py index 6e275cb773f0..237e4acdc84c 100644 --- a/tests/integration/test_export_ttl/common.py +++ b/tests/integration/test_export_ttl/common.py @@ -110,12 +110,12 @@ def ttl_rows(node, table, columns=None): return {row[0]: dict(zip(columns, row)) for row in rows} -def wait_for_same_ttl_rows(replicas, table, settled, timeout=90): +def wait_for_same_ttl_rows(replicas, table, settled, timeout=90, columns=None): """Wait until every replica shows the same rows of `system.ttl_exports` for *table*, and they satisfy *settled*.""" start = time.time() while True: - rows = [ttl_rows(replica, table) for replica in replicas] + rows = [ttl_rows(replica, table, columns) for replica in replicas] if all(replica_rows == rows[0] for replica_rows in rows) and settled(rows[0]): return rows[0] assert time.time() - start < timeout, f"The rows of system.ttl_exports did not settle: {rows}" diff --git a/tests/integration/test_export_ttl/test_failures.py b/tests/integration/test_export_ttl/test_failures.py index 32fe1f88d36b..e1c283a69abc 100644 --- a/tests/integration/test_export_ttl/test_failures.py +++ b/tests/integration/test_export_ttl/test_failures.py @@ -25,8 +25,9 @@ CLUSTER_INSTANCES = ["replica1"] # Failures of the tasks of the `EXPORT` TTL to an Iceberg destination: a failed group is retried as a -# new task that lists the failed ones in `retry_of`, only the task that completes commits a snapshot, -# and no row lands twice, whatever fails and when. +# new task, which lists in `retry_of` the failed ones that exported all their parts, since only those +# may have committed. Only the task that completes commits a snapshot, and no row lands twice, +# whatever fails and when. def make_tables(node, engine, settings=None, columns=COLUMNS, source_partition_by="year", spec="year"): @@ -65,6 +66,7 @@ def test_retryable_part_error_is_retried_by_the_same_task(cluster, source_engine def test_non_retryable_part_error_is_retried_as_a_new_task(cluster, source_engine): + """The failed task did not export its part, so it did not commit: the retry does not check it.""" node = cluster.instances["replica1"] mt_table, iceberg_table = make_tables(node, source_engine) @@ -83,7 +85,7 @@ def test_non_retryable_part_error_is_retried_as_a_new_task(cluster, source_engin failed = {task["transaction_id"] for task in tasks if task["status"] == "FAILED"} completed = completed_ttl_tasks(node, mt_table) assert failed and len(completed) == 1, tasks - assert set(completed[0]["retry_of"]) == failed, (completed, failed) + assert completed[0]["retry_of"] == [], (completed, failed) assert_exactly_once(iceberg_ids(node, iceberg_table), [1]) assert_one_snapshot_per_task(node, mt_table, iceberg_table) @@ -135,7 +137,8 @@ def test_commit_that_landed_is_not_committed_again(cluster): def test_killed_task_is_retried(cluster, source_engine): - """The killed task never commits; the file its part export writes afterwards is not in the table.""" + """The killed task never commits; the file its part export writes afterwards is not in the table. + It was killed before exporting its part, so the retry does not check it.""" node = cluster.instances["replica1"] mt_table, iceberg_table = make_tables(node, source_engine) @@ -145,10 +148,12 @@ def test_killed_task_is_retried(cluster, source_engine): killed = ttl_tasks(node, mt_table)[0]["transaction_id"] node.query(f"KILL EXPORT WHERE transaction_id = '{killed}'") wait_until(lambda: ttl_tasks(node, mt_table)[0]["status"] == "KILLED", 60, "The task was not killed") + # The retry starts while the part export of the killed task is still paused. + wait_until(lambda: len(ttl_tasks(node, mt_table)) == 2, 60, "The killed task was not retried") wait_for_partitions_exported(node, mt_table, ["2020"]) completed = completed_ttl_tasks(node, mt_table) - assert len(completed) == 1 and completed[0]["retry_of"] == [killed], ttl_tasks(node, mt_table) + assert len(completed) == 1 and completed[0]["retry_of"] == [], (killed, ttl_tasks(node, mt_table)) assert_exactly_once(iceberg_ids(node, iceberg_table), [1]) assert_one_snapshot_per_task(node, mt_table, iceberg_table) diff --git a/tests/integration/test_export_ttl/test_replication.py b/tests/integration/test_export_ttl/test_replication.py index 3add6b76de98..63f024439bb6 100644 --- a/tests/integration/test_export_ttl/test_replication.py +++ b/tests/integration/test_export_ttl/test_replication.py @@ -6,6 +6,7 @@ COLUMNS, DUE, NOT_DUE, + TTL_STATE_COLUMNS, assert_exactly_once, assert_one_snapshot_per_task, completed_ttl_tasks, @@ -18,6 +19,7 @@ snapshot_refreshes, ttl_rows, ttl_tasks, + wait_for_last_error, wait_for_partitions_exported, wait_for_same_ttl_rows, wait_until, @@ -25,9 +27,10 @@ CLUSTER_INSTANCES = ["replica1", "replica2"] -# One replica of a `ReplicatedMergeTree` table schedules the groups of the `EXPORT` TTL, and stores -# its state in Keeper; every replica shows the same `system.ttl_exports`, and another replica takes -# over where the previous one left off. Every replica has the same Iceberg destination. +# One replica of a `ReplicatedMergeTree` table schedules the groups of the `EXPORT` TTL. Every +# replica tracks the batching windows of its parts and shows the same `system.ttl_exports`, except +# for the error of acting on a partition, and another replica takes over where the previous one left +# off. Every replica has the same Iceberg destination. def make_replicated_tables(replicas, columns=COLUMNS, partition_by="year", spec="year", settings=None): @@ -77,14 +80,15 @@ def test_replicas_export_once(cluster): assert_one_snapshot_per_task(replicas[0], mt_table, iceberg_table) -def test_state_is_the_same_on_every_replica(cluster): - """`system.ttl_exports` has the same rows on every replica, including the error of a group, which +def test_rows_are_the_same_on_every_replica(cluster): + """`system.ttl_exports` has the same rows on every replica, except the error of a group, which only the replica that schedules sees. A day-partitioned Iceberg destination of a monthly source is accepted, and a group whose rows are on two days fails when it is exported.""" replicas = [cluster.instances["replica1"], cluster.instances["replica2"]] mt_table, iceberg_table = make_replicated_tables( replicas, columns="id Int64, t DateTime", partition_by="toYYYYMM(t)", spec="toRelativeDayNum(t)" ) + holder, other = holder_and_other(replicas, mt_table) replicas[0].query(f"INSERT INTO {mt_table} VALUES (1, '2020-01-01 10:00:00'), (2, '2020-01-02 10:00:00')") replicas[0].query(f"INSERT INTO {mt_table} VALUES (3, '2020-02-01 10:00:00')") @@ -93,16 +97,18 @@ def test_state_is_the_same_on_every_replica(cluster): def settled(rows): return ( "202001" in rows and "202002" in rows - and rows["202001"]["last_error"] != "" and rows["202001"]["eligible_parts"] == 1 + and rows["202001"]["eligible_parts"] == 1 and rows["202002"]["exported_parts"] == 1 and rows["202002"]["eligible_parts"] == 0 ) - rows = wait_for_same_ttl_rows(replicas, mt_table, settled) - assert "multiple destination partitions" in rows["202001"]["last_error"], rows + columns = [column for column in TTL_STATE_COLUMNS if column != "last_error"] + rows = wait_for_same_ttl_rows(replicas, mt_table, settled, columns=columns) assert rows["202001"]["claimed_parts"] == 0 and rows["202001"]["current_transaction_id"] == "", rows - assert rows["202002"]["last_error"] == "", rows - assert rows["202001"]["scheduler_replica"] in ("replica1", "replica2"), rows - assert rows["202002"]["scheduler_replica"] == rows["202001"]["scheduler_replica"], rows + assert rows["202001"]["scheduler_replica"] == holder.name and rows["202002"]["scheduler_replica"] == holder.name, rows + + wait_for_last_error(holder, mt_table, "202001", "multiple destination partitions") + assert ttl_rows(holder, mt_table)["202002"]["last_error"] == "" + assert all(row["last_error"] == "" for row in ttl_rows(other, mt_table).values()), ttl_rows(other, mt_table) for replica in replicas: assert_exactly_once(iceberg_ids(replica, iceberg_table), [3]) assert_one_snapshot_per_task(replicas[0], mt_table, iceberg_table) @@ -136,8 +142,8 @@ def test_index_snapshot_is_cached(cluster): def test_failover_resumes_the_batch(cluster): - """The replica that takes over resumes the batching window where the previous one left off, and - nothing is exported twice.""" + """Every replica tracks the batching window of the parts it has, so the replica that takes over + continues it instead of starting it again, and nothing is exported twice.""" replicas = [cluster.instances["replica1"], cluster.instances["replica2"]] mt_table, iceberg_table = make_replicated_tables( replicas, settings={"ttl_export_batch_window_seconds": 60, "ttl_export_batch_max_delay_seconds": 600} @@ -146,8 +152,9 @@ def test_failover_resumes_the_batch(cluster): holder.query(f"INSERT INTO {mt_table} VALUES (1, 2020, {DUE})") sync(replicas, mt_table) - first_eligible = wait_until(lambda: first_eligible_time(other, mt_table, "2020"), 60, "The batch was not stored") - assert first_eligible == first_eligible_time(holder, mt_table, "2020") + first_eligible = wait_until(lambda: first_eligible_time(other, mt_table, "2020"), 60, "The other replica does not track the batch") + # The replicas saw the part at their own checks, after it was fetched. + assert abs(first_eligible - first_eligible_time(holder, mt_table, "2020")) <= 5 holder.stop_clickhouse(kill=True) try: diff --git a/tests/integration/test_export_ttl/test_server_setting.py b/tests/integration/test_export_ttl/test_server_setting.py new file mode 100644 index 000000000000..bc92b8f55637 --- /dev/null +++ b/tests/integration/test_export_ttl/test_server_setting.py @@ -0,0 +1,56 @@ +from helpers.export_partition_helpers import unique_suffix + +from .common import ( + COLUMNS, + DUE, + assert_exactly_once, + create_iceberg, + create_source, + iceberg_ids, + ttl_rows, + wait_for_partitions_exported, + wait_until, +) + +CLUSTER_INSTANCES = ["replica1"] + +# Without the server setting `allow_experimental_export_merge_tree_partition` a table has neither the +# export index nor the merge fence, so merges could mix exported parts with the others, whose rows +# would then never be exported. A table with an `EXPORT` TTL is therefore not loaded without it. + +SETTING_CONFIG = "/etc/clickhouse-server/config.d/allow_experimental_export_partition.xml" + + +def set_server_setting(node, value): + node.replace_in_config( + SETTING_CONFIG, + f"{1 - value}", + f"{value}", + ) + node.restart_clickhouse() + + +def test_table_is_not_attached_without_the_server_setting(cluster, source_engine): + node = cluster.instances["replica1"] + suffix = unique_suffix() + mt_table, iceberg_table = f"setting_mt_{suffix}", f"setting_iceberg_{suffix}" + create_iceberg(node, iceberg_table) + create_source(node, mt_table, COLUMNS, "year", f"t + INTERVAL 1 DAY EXPORT TO TABLE {iceberg_table}", engine=source_engine) + node.query(f"INSERT INTO {mt_table} VALUES (1, 2020, {DUE})") + wait_for_partitions_exported(node, mt_table, ["2020"]) + + # Detached, so that the server starts without loading it. + node.query(f"DETACH TABLE {mt_table} PERMANENTLY") + set_server_setting(node, 0) + try: + error = node.query_and_get_error(f"ATTACH TABLE {mt_table}") + assert "SUPPORT_IS_DISABLED" in error and "allow_experimental_export_merge_tree_partition" in error, error + finally: + set_server_setting(node, 1) + + node.query(f"ATTACH TABLE {mt_table}") + wait_until( + lambda: ttl_rows(node, mt_table).get("2020", {}).get("exported_parts") == 1, 60, + "The exported part is not known as exported after the table is attached again", + ) + assert_exactly_once(iceberg_ids(node, iceberg_table), [1]) From af151b0ec0a045cce365c42ef0c4aed157066510 Mon Sep 17 00:00:00 2001 From: Arthur Passos Date: Mon, 28 Sep 2026 17:51:42 -0300 Subject: [PATCH 05/15] hard code allow lossy casts --- src/Storages/MergeTree/ExportTTLIndex.cpp | 6 +++ src/Storages/MergeTree/ExportTTLIndex.h | 5 ++ src/Storages/MergeTree/ExportTTLScheduler.cpp | 2 + src/Storages/MergeTree/MergeTreeData.cpp | 5 +- .../tests/gtest_export_task_ordering.cpp | 46 +++++++++++++++++++ .../test_export_ttl/test_ttl_expressions.py | 17 +++++++ .../05057_export_ttl_expressions.reference | 5 +- .../05057_export_ttl_expressions.sql | 31 +++++++------ 8 files changed, 101 insertions(+), 16 deletions(-) diff --git a/src/Storages/MergeTree/ExportTTLIndex.cpp b/src/Storages/MergeTree/ExportTTLIndex.cpp index 02ae11b52f16..1629accdbb1e 100644 --- a/src/Storages/MergeTree/ExportTTLIndex.cpp +++ b/src/Storages/MergeTree/ExportTTLIndex.cpp @@ -1,5 +1,6 @@ #include +#include #include #include #include @@ -188,6 +189,11 @@ std::unique_ptr ExportTTLIndexSnapshot::build( namespace ExportTTLUtils { +void allowLossyCasts(Context & context) +{ + context.setSetting("export_merge_tree_part_allow_lossy_cast", true); +} + String destinationKey(const String & database, const String & table, const String & uuid) { return escapeForFileName(database) + "." + escapeForFileName(table) + "." + (uuid.empty() ? String("none") : escapeForFileName(uuid)); diff --git a/src/Storages/MergeTree/ExportTTLIndex.h b/src/Storages/MergeTree/ExportTTLIndex.h index 1c1c3e717b08..ab3906d336dd 100644 --- a/src/Storages/MergeTree/ExportTTLIndex.h +++ b/src/Storages/MergeTree/ExportTTLIndex.h @@ -1,5 +1,6 @@ #pragma once +#include #include #include #include @@ -91,6 +92,10 @@ using ExportTTLIndexSnapshotPtr = std::shared_ptr; namespace ExportTTLUtils { + /// The `EXPORT` TTL always allows lossy casts: its tasks run in the background, so a session + /// that opted in at `CREATE` or `ALTER` would not reach them, and Iceberg has no unsigned types. + void allowLossyCasts(Context & context); + /// Identifies a destination in the export index. The UUID, if not empty, tells apart a table that /// was dropped and created again under the same name, which holds none of the exported rows. String destinationKey(const String & database, const String & table, const String & uuid); diff --git a/src/Storages/MergeTree/ExportTTLScheduler.cpp b/src/Storages/MergeTree/ExportTTLScheduler.cpp index 4e75921b8747..d47e8a39b948 100644 --- a/src/Storages/MergeTree/ExportTTLScheduler.cpp +++ b/src/Storages/MergeTree/ExportTTLScheduler.cpp @@ -95,6 +95,8 @@ ContextPtr ExportTTLScheduler::makeContext() const context->setSetting("export_merge_tree_part_throw_on_pending_mutations", false); context->setSetting("export_merge_tree_part_throw_on_pending_patch_parts", false); + ExportTTLUtils::allowLossyCasts(*context); + return context; } diff --git a/src/Storages/MergeTree/MergeTreeData.cpp b/src/Storages/MergeTree/MergeTreeData.cpp index 32f5ad705bbe..fb6c39d2bcf5 100644 --- a/src/Storages/MergeTree/MergeTreeData.cpp +++ b/src/Storages/MergeTree/MergeTreeData.cpp @@ -23,6 +23,7 @@ #include #include #include +#include #include #include #include @@ -1359,7 +1360,9 @@ void MergeTreeData::validateExportTTL( const auto source_metadata = std::make_shared(new_metadata); const auto destination_metadata = destination->getInMemoryMetadataPtr(local_context, false); - ExportTaskUtils::verifyExportSchemaCastable(source_metadata, destination_metadata, destination->getStorageID(), local_context); + auto schema_context = Context::createCopy(local_context); + ExportTTLUtils::allowLossyCasts(*schema_context); + ExportTaskUtils::verifyExportSchemaCastable(source_metadata, destination_metadata, destination->getStorageID(), schema_context); /// Whether the rows of a group of parts land in a single destination partition can only be proven /// from the parts, which is done when the group is exported. What can never be proven is refused here. diff --git a/src/Storages/MergeTree/tests/gtest_export_task_ordering.cpp b/src/Storages/MergeTree/tests/gtest_export_task_ordering.cpp index c48231fd9df8..abc63a588bb3 100644 --- a/src/Storages/MergeTree/tests/gtest_export_task_ordering.cpp +++ b/src/Storages/MergeTree/tests/gtest_export_task_ordering.cpp @@ -6,7 +6,9 @@ #include #include #include +#include #include +#include #include #include #include @@ -26,6 +28,7 @@ namespace ErrorCodes extern const int UNKNOWN_TABLE; extern const int BAD_ARGUMENTS; extern const int NETWORK_ERROR; + extern const int INCOMPATIBLE_COLUMNS; } namespace @@ -269,6 +272,49 @@ TEST(ExportTaskUtils, PlainTaskRetryOfRoundTrip) EXPECT_TRUE(MergeTreeExportTask::fromJsonString(task.toJsonString()).retry_of.empty()); } +namespace +{ + /// Whether a manual export, which does not allow lossy casts by default, may export a column of + /// type `from` to a column of type `to`. + bool isCastAllowedByDefault(const String & from, const String & to) + { + tryRegisterFunctions(); + const auto metadata_with_column_of_type = [](const String & type) + { + auto metadata = std::make_shared(); + metadata->setColumns(ColumnsDescription(NamesAndTypesList{{"ts", DataTypeFactory::instance().get(type)}})); + return metadata; + }; + + try + { + ExportTaskUtils::verifyExportSchemaCastable( + metadata_with_column_of_type(from), metadata_with_column_of_type(to), StorageID("db", "destination"), getContext().context); + return true; + } + catch (const Exception & e) + { + if (e.code() == ErrorCodes::INCOMPATIBLE_COLUMNS) + return false; + throw; + } + } +} + +TEST(ExportTaskUtils, DateAndTimeWideningIsNotLossy) +{ + EXPECT_TRUE(isCastAllowedByDefault("Date", "Date32")); + EXPECT_TRUE(isCastAllowedByDefault("DateTime", "DateTime64(6)")); + EXPECT_TRUE(isCastAllowedByDefault("DateTime", "DateTime64(9)")); + EXPECT_TRUE(isCastAllowedByDefault("DateTime64(3)", "DateTime64(6)")); + EXPECT_TRUE(isCastAllowedByDefault("DateTime64(9)", "DateTime64(9)")); + + EXPECT_FALSE(isCastAllowedByDefault("DateTime64(9)", "DateTime64(6)")); + /// The range of scale 9 ends in 2262. + EXPECT_FALSE(isCastAllowedByDefault("DateTime64(3)", "DateTime64(9)")); + EXPECT_FALSE(isCastAllowedByDefault("DateTime64(6)", "DateTime")); +} + TEST(ExportTaskUtils, PartitionIdIsDerivedFromParts) { EXPECT_EQ(ExportTaskUtils::getPartitionIdOfParts({"2020_1_1_0", "2020_2_5_1"}, MERGE_TREE_DATA_MIN_FORMAT_VERSION_WITH_CUSTOM_PARTITIONING), "2020"); diff --git a/tests/integration/test_export_ttl/test_ttl_expressions.py b/tests/integration/test_export_ttl/test_ttl_expressions.py index 7613e24e560c..4cb0bbb316a2 100644 --- a/tests/integration/test_export_ttl/test_ttl_expressions.py +++ b/tests/integration/test_export_ttl/test_ttl_expressions.py @@ -188,6 +188,23 @@ def test_datetime64_column(cluster): assert_one_snapshot_per_task(node, mt_table, iceberg_table) +def test_unsigned_columns_are_exported_to_signed_ones(cluster): + """Iceberg has no unsigned types, so a `UInt32` column is exported to an `int`, read back as + `Int32`. The TTL allows the lossy cast, which changes the values that do not fit.""" + node = cluster.instances["replica1"] + mt_table, iceberg_table = make_tables( + node, "MergeTree", "id UInt32, retention UInt16, eventDate Date", + "(eventDate, retention)", "", "eventDate + toIntervalDay(retention)", + destination_columns="id Int32, retention Int32, eventDate Date", + ) + + node.query(f"INSERT INTO {mt_table} VALUES (1, 5, today() - 10), (4000000000, 5, today() - 10)") + wait_for_partitions_exported(node, mt_table, partitions_where(node, mt_table, "1")) + + assert node.query(f"SELECT id, retention FROM {iceberg_table} ORDER BY id") == "-294967296\t5\n1\t5\n" + assert_one_snapshot_per_task(node, mt_table, iceberg_table) + + def test_ttl_of_a_materialized_column(cluster): """The TTL and the partition key use a `MATERIALIZED` column, which is exported as an ordinary one.""" node = cluster.instances["replica1"] diff --git a/tests/queries/0_stateless/05057_export_ttl_expressions.reference b/tests/queries/0_stateless/05057_export_ttl_expressions.reference index 2508fb1acb9f..5011c681c328 100644 --- a/tests/queries/0_stateless/05057_export_ttl_expressions.reference +++ b/tests/queries/0_stateless/05057_export_ttl_expressions.reference @@ -3,6 +3,7 @@ TTL toStartOfMonth(d) + toIntervalMonth(1) EXPORT TO TABLE export_ttl_expression TTL greatest(t, toDateTime(d)) + toIntervalDay(1) EXPORT TO TABLE export_ttl_expressions_destination SETTINGS TTL ifNull(nd, d) + toIntervalDay(1) EXPORT TO TABLE export_ttl_expressions_destination SETTINGS TTL ifNull(nd, d) + toIntervalDay(1) EXPORT TO TABLE export_ttl_expressions_destination, t + toIntervalSecond(rand() % 10) SETTINGS -export_ttl_expressions_micros_to_micros export_ttl_expressions_millis_to_micros -export_ttl_expressions_seconds_to_nanos +export_ttl_expressions_millis_to_nanos +export_ttl_expressions_nanos_to_micros +export_ttl_expressions_unsigned_to_signed diff --git a/tests/queries/0_stateless/05057_export_ttl_expressions.sql b/tests/queries/0_stateless/05057_export_ttl_expressions.sql index d5961d032320..9a9385d9a983 100644 --- a/tests/queries/0_stateless/05057_export_ttl_expressions.sql +++ b/tests/queries/0_stateless/05057_export_ttl_expressions.sql @@ -53,15 +53,17 @@ SET allow_suspicious_ttl_expressions = 0; DROP TABLE export_ttl_expressions_source; DROP TABLE export_ttl_expressions_destination; --- A `DateTime64` column may be exported to one of a larger scale, e.g. to the microseconds of an --- Iceberg `timestamp`, unless that scale is 9, whose range ends in 2262. +-- The `EXPORT` TTL allows lossy casts, whatever `export_merge_tree_part_allow_lossy_cast` is, e.g. +-- nanoseconds to the microseconds of an Iceberg `timestamp`, or an unsigned column to a signed one. DROP TABLE IF EXISTS export_ttl_expressions_micros; DROP TABLE IF EXISTS export_ttl_expressions_nanos; +DROP TABLE IF EXISTS export_ttl_expressions_signed; DROP TABLE IF EXISTS export_ttl_expressions_millis_to_micros; -DROP TABLE IF EXISTS export_ttl_expressions_micros_to_micros; -DROP TABLE IF EXISTS export_ttl_expressions_seconds_to_nanos; DROP TABLE IF EXISTS export_ttl_expressions_nanos_to_micros; DROP TABLE IF EXISTS export_ttl_expressions_millis_to_nanos; +DROP TABLE IF EXISTS export_ttl_expressions_unsigned_to_signed; + +SET export_merge_tree_part_allow_lossy_cast = 0; CREATE TABLE export_ttl_expressions_micros (id UInt64, d Date, ts DateTime64(6)) ENGINE = S3(s3_conn, filename = 'export_ttl_expressions_micros', format = Parquet, partition_strategy = 'hive') @@ -71,25 +73,28 @@ CREATE TABLE export_ttl_expressions_nanos (id UInt64, d Date, ts DateTime64(9)) ENGINE = S3(s3_conn, filename = 'export_ttl_expressions_nanos', format = Parquet, partition_strategy = 'hive') PARTITION BY d; +CREATE TABLE export_ttl_expressions_signed (id Int32, d Date, ts DateTime) +ENGINE = S3(s3_conn, filename = 'export_ttl_expressions_signed', format = Parquet, partition_strategy = 'hive') +PARTITION BY d; + CREATE TABLE export_ttl_expressions_millis_to_micros (id UInt64, d Date, ts DateTime64(3)) ENGINE = MergeTree PARTITION BY d ORDER BY id TTL ts + INTERVAL 1 DAY EXPORT TO TABLE export_ttl_expressions_micros; -CREATE TABLE export_ttl_expressions_micros_to_micros (id UInt64, d Date, ts DateTime64(6)) ENGINE = MergeTree PARTITION BY d ORDER BY id +CREATE TABLE export_ttl_expressions_nanos_to_micros (id UInt64, d Date, ts DateTime64(9)) ENGINE = MergeTree PARTITION BY d ORDER BY id TTL ts + INTERVAL 1 DAY EXPORT TO TABLE export_ttl_expressions_micros; -CREATE TABLE export_ttl_expressions_seconds_to_nanos (id UInt64, d Date, ts DateTime) ENGINE = MergeTree PARTITION BY d ORDER BY id +CREATE TABLE export_ttl_expressions_millis_to_nanos (id UInt64, d Date, ts DateTime64(3)) ENGINE = MergeTree PARTITION BY d ORDER BY id TTL ts + INTERVAL 1 DAY EXPORT TO TABLE export_ttl_expressions_nanos; -CREATE TABLE export_ttl_expressions_nanos_to_micros (id UInt64, d Date, ts DateTime64(9)) ENGINE = MergeTree PARTITION BY d ORDER BY id -TTL ts + INTERVAL 1 DAY EXPORT TO TABLE export_ttl_expressions_micros; -- { serverError INCOMPATIBLE_COLUMNS } - -CREATE TABLE export_ttl_expressions_millis_to_nanos (id UInt64, d Date, ts DateTime64(3)) ENGINE = MergeTree PARTITION BY d ORDER BY id -TTL ts + INTERVAL 1 DAY EXPORT TO TABLE export_ttl_expressions_nanos; -- { serverError INCOMPATIBLE_COLUMNS } +CREATE TABLE export_ttl_expressions_unsigned_to_signed (id UInt32, d Date, ts DateTime) ENGINE = MergeTree PARTITION BY d ORDER BY id +TTL ts + INTERVAL 1 DAY EXPORT TO TABLE export_ttl_expressions_signed; SELECT name FROM system.tables WHERE database = currentDatabase() AND match(name, '_to_') ORDER BY name; DROP TABLE export_ttl_expressions_millis_to_micros; -DROP TABLE export_ttl_expressions_micros_to_micros; -DROP TABLE export_ttl_expressions_seconds_to_nanos; +DROP TABLE export_ttl_expressions_nanos_to_micros; +DROP TABLE export_ttl_expressions_millis_to_nanos; +DROP TABLE export_ttl_expressions_unsigned_to_signed; DROP TABLE export_ttl_expressions_micros; DROP TABLE export_ttl_expressions_nanos; +DROP TABLE export_ttl_expressions_signed; From f42cbc253c4034f8731eee534ef902ae885b7c85 Mon Sep 17 00:00:00 2001 From: Arthur Passos Date: Tue, 29 Sep 2026 09:36:53 -0300 Subject: [PATCH 06/15] do not support ttl on cas for now --- docs/en/antalya/partition_export.md | 2 +- docs/en/antalya/ttl_export.md | 17 +++++++++-------- src/Storages/MergeTree/MergeTreeData.cpp | 13 +++++++++++++ src/Storages/MergeTree/MergeTreeData.h | 4 ++++ src/Storages/StorageMergeTree.cpp | 3 +++ .../0_stateless/05053_export_ttl_syntax.sql | 3 ++- .../0_stateless/05054_export_ttl_merge_tree.sh | 3 ++- .../05056_export_ttl_partition_key_check.sql | 3 ++- .../05057_export_ttl_expressions.sql | 3 ++- 9 files changed, 38 insertions(+), 13 deletions(-) diff --git a/docs/en/antalya/partition_export.md b/docs/en/antalya/partition_export.md index b40162c75d64..d09a28ac81e0 100644 --- a/docs/en/antalya/partition_export.md +++ b/docs/en/antalya/partition_export.md @@ -299,7 +299,7 @@ Status values include: ### Source columns {#source-columns} - `source` — `query` for a task of `EXPORT PARTITION`, `ttl` for a task of the table's `TTL ... EXPORT TO TABLE` expression. -- `retry_of` — for a task of the `EXPORT` TTL: transaction ids of earlier tasks that failed to export some of its parts. Its commit checks whether any of them landed at the destination after all. +- `retry_of` — for a task of the `EXPORT` TTL: transaction ids of earlier tasks that failed to export some of its parts after exporting all of theirs, so their commit may still land. Its commit checks whether any of them landed at the destination after all. To pick the latest exception across replicas: diff --git a/docs/en/antalya/ttl_export.md b/docs/en/antalya/ttl_export.md index d92140b8e8e4..e79e097426d4 100644 --- a/docs/en/antalya/ttl_export.md +++ b/docs/en/antalya/ttl_export.md @@ -36,8 +36,8 @@ Here rows are exported to `events_archive` 30 days after `event_time`, and delet ## Requirements {#requirements} -- The server setting `allow_experimental_export_merge_tree_partition` must be enabled on every replica, and the query setting `allow_experimental_export_ttl` must be enabled for the `CREATE` or `ALTER` that adds the expression. -- The destination must exist when the expression is added. It must be an Apache Iceberg or object storage table that `EXPORT PARTITION` can export to, and its schema must be castable from the source schema, see [`EXPORT PARTITION` requirements](/docs/en/antalya/partition_export.md#requirements). An unqualified table name refers to the database of the source table. +- The server setting `allow_experimental_export_merge_tree_partition` must be enabled on every replica, and the query setting `allow_experimental_export_ttl` must be enabled for the `CREATE` or `ALTER` that adds the expression. A table with the expression is not loaded, e.g. at a restart or by `ATTACH TABLE`, while the server setting is disabled: without it, merges would not keep the exported parts apart from the others. +- The destination must exist when the expression is added. It must be an Apache Iceberg or object storage table that `EXPORT PARTITION` can export to, and its schema must be castable from the source schema, see [`EXPORT PARTITION` requirements](/docs/en/antalya/partition_export.md#requirements). Unlike `EXPORT PARTITION`, the TTL allows lossy casts, as if `export_merge_tree_part_allow_lossy_cast` were enabled: e.g. a `UInt32` column is exported to the `int` of an Iceberg table, which is read back as `Int32`, and values that do not fit change. An unqualified table name refers to the database of the source table. - A table can have at most one `EXPORT` TTL expression, without `WHERE` or `GROUP BY`. The expression must be deterministic and return a `Date` or `DateTime`. - The rows of a group must land in a single partition of the destination, see [Partition key of the destination](#destination-partition-key). @@ -66,13 +66,13 @@ A group has at most `ttl_export_max_parts_per_group` parts and `ttl_export_max_b To keep exported rows apart from the others, parts that are exported, parts that are being exported and parts that are not exported are never merged together. Parts that are being exported are not merged at all until their group commits. On a `Replicated*MergeTree` table, every replica enforces this, which is why a replica advertises that it supports it, and groups are only started while every replica does. -One replica of a `Replicated*MergeTree` table schedules the groups; the others take part in exporting them like in `EXPORT PARTITION`. The state of the scheduling, i.e. the batch timers and the last errors, is kept in Keeper, so when another replica takes over, it continues where the previous one left off. +One replica of a `Replicated*MergeTree` table schedules the groups; the others take part in exporting them like in `EXPORT PARTITION`. Every replica tracks the batch timers of the parts it has, so when another replica takes over, it continues them, give or take how much later it got the parts. ## Failures and retries {#failures-and-retries} -A group that fails, is killed or times out is retried on the next check as a new task, with a new transaction id. The retry contains every part of the failed group, plus any part that became eligible meanwhile. Its `retry_of` column lists the failed tasks. Throttling comes from the export tasks themselves: per-part retry back-off, `export_merge_tree_task_timeout_seconds` and the commit attempts. +A group that fails, is killed or times out is retried on the next check as a new task, with a new transaction id. The retry contains every part of the failed group, plus any part that became eligible meanwhile. Throttling comes from the export tasks themselves: per-part retry back-off, `export_merge_tree_task_timeout_seconds` and the commit attempts. -Before a failed task is retried, the destination is checked for its commit, in case it committed but was not marked as completed. A failed task may also land at the destination after that check, e.g. a request that completes late. The commit of the retry therefore checks each task in `retry_of` again, and leaves out the files of the parts a landed task exported. +A task commits only once it exported all its parts, so a task that failed before that cannot have committed, and nothing checks it. A task that failed after that is checked at the destination before it is retried, in case it committed but was not marked as completed. It may also land after that check, e.g. a request that the destination applies late, so the retry lists it in its `retry_of` column, with the blocks of its parts. The commit of the retry checks each task in `retry_of` again, and leaves out the files of the parts a landed task exported. A task stays in the `retry_of` of later retries for 10 minutes after it failed, or longer while a replica is still in its commit; it is then checked a last time, and kept only if it landed. `KILL EXPORT` of a task of the TTL makes it retry. To stop exporting, use `SYSTEM STOP MOVES`, which pauses the TTL export of the table, or remove the expression. On a `Replicated*MergeTree` table, `SYSTEM STOP MOVES` pauses it only on the replica that schedules the groups, shown in the `scheduler_replica` column of `system.ttl_exports`, so run it on every replica, e.g. with `ON CLUSTER`. @@ -96,7 +96,7 @@ While the destination does not exist, e.g. it was dropped or is not loaded yet, - `allow_experimental_export_ttl` — allows adding a `TTL ... EXPORT TO TABLE` expression. -The settings of the export tasks, e.g. the output format settings, `export_merge_tree_part_file_already_exists_policy` and `export_merge_tree_task_timeout_seconds`, are taken from the settings profile named by `ttl_export_settings_profile`, or the default profile. +The settings of the export tasks, e.g. the output format settings, `export_merge_tree_part_file_already_exists_policy` and `export_merge_tree_task_timeout_seconds`, are taken from the settings profile named by `ttl_export_settings_profile`, or the default profile. The settings of the session that adds the expression do not apply to them. `export_merge_tree_part_allow_lossy_cast` is always enabled. ### MergeTree settings {#merge-tree-settings} @@ -107,13 +107,13 @@ The settings of the export tasks, e.g. the output format settings, `export_merge | `ttl_export_batch_max_delay_seconds` | `600` | A group is exported at the latest this long after its first part became eligible. | | `ttl_export_batch_min_bytes` | `256 MiB` | A group is exported as soon as its parts reach this size. `0` disables it. | | `ttl_export_max_parts_per_group` | `100` | Maximum number of parts of a group. Parts of a failed group are always retried together. | -| `ttl_export_max_bytes_per_group` | `10 GiB` | Maximum size of a group. A bigger part is exported on its own. `0` means unlimited. | +| `ttl_export_max_bytes_per_group` | `100 GiB` | Maximum size of a group. A bigger part is exported on its own. `0` means unlimited. | | `ttl_export_max_concurrent_groups` | `4` | Maximum number of groups of the table being exported at the same time. | | `ttl_export_settings_profile` | `''` | Settings profile of the export tasks. | ## Monitoring {#monitoring} -`system.ttl_exports` has one row per partition of a table with an `EXPORT` TTL expression, as of the last check. For a `Replicated*MergeTree` table, every replica has the same rows: the batch timers, the task being exported and the last error come from Keeper, and are those of the replica that schedules the groups, shown in `scheduler_replica`. The part counts are those of the parts of the replica, so they differ while a replica is fetching parts. +`system.ttl_exports` has one row per partition of a table with an `EXPORT` TTL expression, as of the last check. For a `Replicated*MergeTree` table, every replica shows what was exported and the task being exported, which come from Keeper. The part counts and the batch timers are those of the parts of the replica, so they differ while a replica is fetching parts. The error of starting a group, e.g. because its rows would land in several destination partitions, is shown by the replica that schedules the groups, shown in `scheduler_replica`. ```sql SELECT partition_id, exported_parts, claimed_parts, eligible_parts, parts_held_by_delete_gate, @@ -134,3 +134,4 @@ What was exported is read from Keeper again only when it changes, i.e. when a gr - Export tasks are not removed, so they accumulate in Keeper (or in the data directory of a plain `MergeTree` table), and in `system.distributed_exports`. - On a plain object storage destination, files that no commit file references may remain, e.g. when a part of a failed group is mutated before its retry, its new name gives a new file, and the file of the failed attempt is left. Readers must follow the commit files, see [`EXPORT PARTITION`](/docs/en/antalya/partition_export.md). - Parts that are attached again, e.g. by `ALTER TABLE ... ATTACH PARTITION`, get new block numbers and are exported again. +- A plain `MergeTree` table whose first disk is content-addressed cannot have the expression: it keeps what was exported and its export tasks in files on that disk, which does not support how they are written. A `Replicated*MergeTree` table keeps them in Keeper, so it can. diff --git a/src/Storages/MergeTree/MergeTreeData.cpp b/src/Storages/MergeTree/MergeTreeData.cpp index fb6c39d2bcf5..7d5044dc1357 100644 --- a/src/Storages/MergeTree/MergeTreeData.cpp +++ b/src/Storages/MergeTree/MergeTreeData.cpp @@ -1320,9 +1320,22 @@ void MergeTreeData::checkExportTTL( if (describe(export_ttls) == describe(old_metadata.getExportTTLs())) return; + checkExportTTLIsSupportedByDisk(new_metadata); validateExportTTL(getStorageID(), new_metadata, getSettings(), local_context); } +void MergeTreeData::checkExportTTLIsSupportedByDisk(const StorageInMemoryMetadata & metadata) const +{ + if (supportsReplication() || !metadata.hasAnyExportTTL()) + return; + + const auto disk = getDisks().front(); + if (disk->isContentAddressed()) + throw Exception(ErrorCodes::NOT_IMPLEMENTED, + "`TTL ... EXPORT TO TABLE` is not supported for a MergeTree table on the content-addressed disk {}", + disk->getName()); +} + void MergeTreeData::validateExportTTL( const StorageID & table_id, const StorageInMemoryMetadata & new_metadata, const MergeTreeSettingsPtr & settings, ContextPtr local_context) { diff --git a/src/Storages/MergeTree/MergeTreeData.h b/src/Storages/MergeTree/MergeTreeData.h index 578ea56fcfbe..c962c68fffeb 100644 --- a/src/Storages/MergeTree/MergeTreeData.h +++ b/src/Storages/MergeTree/MergeTreeData.h @@ -1766,6 +1766,10 @@ class MergeTreeData : public WithMutableContext, public IStorage, public IBackgr /// Validates the `TTL ... EXPORT TO TABLE` expression of an ALTER, if it changes it. void checkExportTTL(const StorageInMemoryMetadata & new_metadata, const StorageInMemoryMetadata & old_metadata, ContextPtr local_context) const; + /// A plain `MergeTree` keeps its export index and tasks in files on its first disk, which are + /// written and then renamed: a content-addressed disk cannot rename a file written before. + void checkExportTTLIsSupportedByDisk(const StorageInMemoryMetadata & metadata) const; + public: /// Validates the `TTL ... EXPORT TO TABLE` expression of the table `table_id` being created or /// altered. The partition key of the destination is checked against the parts of every group diff --git a/src/Storages/StorageMergeTree.cpp b/src/Storages/StorageMergeTree.cpp index 459882ac0933..e939b265e5e5 100644 --- a/src/Storages/StorageMergeTree.cpp +++ b/src/Storages/StorageMergeTree.cpp @@ -254,6 +254,9 @@ StorageMergeTree::StorageMergeTree( , cleanup_thread(*this) , support_transaction(supportTransaction(getDisks(), log.load())) { + if (mode <= LoadingStrictnessLevel::CREATE) + checkExportTTLIsSupportedByDisk(metadata_); + initializeDirectoriesAndFormatVersion(relative_data_path_, LoadingStrictnessLevel::ATTACH <= mode, date_column_name); loadDataParts(LoadingStrictnessLevel::FORCE_RESTORE <= mode, std::nullopt); diff --git a/tests/queries/0_stateless/05053_export_ttl_syntax.sql b/tests/queries/0_stateless/05053_export_ttl_syntax.sql index 1abd8ce59aa8..4f9351563c91 100644 --- a/tests/queries/0_stateless/05053_export_ttl_syntax.sql +++ b/tests/queries/0_stateless/05053_export_ttl_syntax.sql @@ -1,5 +1,6 @@ --- Tags: no-fasttest +-- Tags: no-fasttest, no-cas-storage -- no-fasttest: the destination is an S3 table. +-- no-cas-storage: the EXPORT TTL of a plain MergeTree is not supported on a content-addressed disk. SELECT formatQuery('CREATE TABLE t (d DateTime) ENGINE = MergeTree ORDER BY d TTL d + INTERVAL 1 DAY EXPORT TO TABLE dst, d + INTERVAL 2 DAY DELETE'); SELECT formatQuery('ALTER TABLE t MODIFY TTL d + INTERVAL 1 DAY EXPORT TO TABLE db.dst'); diff --git a/tests/queries/0_stateless/05054_export_ttl_merge_tree.sh b/tests/queries/0_stateless/05054_export_ttl_merge_tree.sh index cb7d911fe007..1caeb464317c 100755 --- a/tests/queries/0_stateless/05054_export_ttl_merge_tree.sh +++ b/tests/queries/0_stateless/05054_export_ttl_merge_tree.sh @@ -1,7 +1,8 @@ #!/usr/bin/env bash -# Tags: no-fasttest, no-shared-merge-tree +# Tags: no-fasttest, no-shared-merge-tree, no-cas-storage # no-fasttest: requires S3 / MinIO. # no-shared-merge-tree: this test exercises the EXPORT TTL of a plain (non-replicated) MergeTree. +# no-cas-storage: the EXPORT TTL of a plain MergeTree is not supported on a content-addressed disk. CUR_DIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd) # shellcheck source=../shell_config.sh diff --git a/tests/queries/0_stateless/05056_export_ttl_partition_key_check.sql b/tests/queries/0_stateless/05056_export_ttl_partition_key_check.sql index 4dad51d41972..9be060e848f8 100644 --- a/tests/queries/0_stateless/05056_export_ttl_partition_key_check.sql +++ b/tests/queries/0_stateless/05056_export_ttl_partition_key_check.sql @@ -1,5 +1,6 @@ --- Tags: no-fasttest +-- Tags: no-fasttest, no-cas-storage -- no-fasttest: the destinations are S3 tables. +-- no-cas-storage: the EXPORT TTL of a plain MergeTree is not supported on a content-addressed disk. -- The partition key of the destination of an EXPORT TTL is checked when the expression is added: -- a key that is a function of the source partition key (structural) or monotonic in a single column diff --git a/tests/queries/0_stateless/05057_export_ttl_expressions.sql b/tests/queries/0_stateless/05057_export_ttl_expressions.sql index 9a9385d9a983..960a3bfeb149 100644 --- a/tests/queries/0_stateless/05057_export_ttl_expressions.sql +++ b/tests/queries/0_stateless/05057_export_ttl_expressions.sql @@ -1,5 +1,6 @@ --- Tags: no-fasttest +-- Tags: no-fasttest, no-cas-storage -- no-fasttest: the destinations are S3 tables. +-- no-cas-storage: the EXPORT TTL of a plain MergeTree is not supported on a content-addressed disk. -- The expressions an `EXPORT` TTL accepts, and the date and time columns it may widen when exporting. From 2d9105065db94526ff0a6d5a9e130bb9bf9eeb2e Mon Sep 17 00:00:00 2001 From: Arthur Passos Date: Tue, 29 Sep 2026 09:51:58 -0300 Subject: [PATCH 07/15] vibe fix a test --- tests/integration/test_export_ttl/common.py | 10 +++++++--- tests/integration/test_export_ttl/test_scheduling.py | 10 ++++++++-- 2 files changed, 15 insertions(+), 5 deletions(-) diff --git a/tests/integration/test_export_ttl/common.py b/tests/integration/test_export_ttl/common.py index 237e4acdc84c..7a94c0bed6c1 100644 --- a/tests/integration/test_export_ttl/common.py +++ b/tests/integration/test_export_ttl/common.py @@ -137,11 +137,15 @@ def partition_settled(rows, partition_id, exported=None): return exported is None or row["exported_parts"] == exported -def wait_for_partitions_exported(node, table, partition_ids, timeout=90): - """Wait until nothing of *partition_ids* is claimed or eligible and no TTL task is in flight.""" +def wait_for_partitions_exported(node, table, partition_ids, timeout=90, exported=None): + """Wait until nothing of *partition_ids* is claimed or eligible and no TTL task is in flight. + + A check that runs before a part is due already publishes that partition with nothing eligible, so + pass *exported* when the wait must see the parts recorded as exported rather than merely not due. + """ def settled(): rows = ttl_rows(node, table) - if any(not partition_settled(rows, partition_id) for partition_id in partition_ids): + if any(not partition_settled(rows, partition_id, exported) for partition_id in partition_ids): return None return pending_ttl_tasks(node, table) == 0 and rows diff --git a/tests/integration/test_export_ttl/test_scheduling.py b/tests/integration/test_export_ttl/test_scheduling.py index 21c20e08e066..fe4e5314c8ae 100644 --- a/tests/integration/test_export_ttl/test_scheduling.py +++ b/tests/integration/test_export_ttl/test_scheduling.py @@ -82,7 +82,12 @@ def test_part_becoming_due_wakes_the_scheduler(cluster, source_engine): """A check computes when the next part becomes due, and the scheduler wakes up then instead of waiting for the check period.""" node = cluster.instances["replica1"] - mt_table, iceberg_table = make_tables(node, source_engine, settings={"ttl_export_check_period_seconds": 120}) + # The batch window is disabled so the wake at the due time exports the part. The default window + # would hold it for another minute, past the bound that distinguishes this wake from the check period. + mt_table, iceberg_table = make_tables( + node, source_engine, + settings={"ttl_export_check_period_seconds": 120, "ttl_export_batch_window_seconds": 0, "ttl_export_batch_max_delay_seconds": 0}, + ) node.query(f"INSERT INTO {mt_table} VALUES (1, 2020, {due_in(12)})") inserted = time.time() @@ -94,7 +99,8 @@ def test_part_becoming_due_wakes_the_scheduler(cluster, source_engine): assert ttl_tasks(node, mt_table) == [], "The part was exported before it was due" time.sleep(1) - wait_for_partitions_exported(node, mt_table, ["2020"], timeout=40) + # That check already shows the partition as idle, so wait until the part is recorded as exported. + wait_for_partitions_exported(node, mt_table, ["2020"], timeout=40, exported=1) assert time.time() - inserted < 40, "The part was exported only by a periodic check" assert_exactly_once(iceberg_ids(node, iceberg_table), [1]) assert_one_snapshot_per_task(node, mt_table, iceberg_table) From 19df41198629abdd78a22f8a27523cdafa898f5e Mon Sep 17 00:00:00 2001 From: Arthur Passos Date: Tue, 29 Sep 2026 11:51:16 -0300 Subject: [PATCH 08/15] more vibe coded stuff --- .../ReplicatedMergeTreeMergePredicate.cpp | 4 +- .../ReplicatedMergeTreeMergePredicate.h | 8 ++-- src/Storages/MergeTree/ExportTTLIndex.h | 2 +- src/Storages/MergeTree/ExportTaskUtils.cpp | 8 ++-- src/Storages/MergeTree/ExportTaskUtils.h | 2 +- src/Storages/MergeTree/MergeTreeData.cpp | 3 ++ src/Storages/MergeTree/MergeTreeData.h | 3 ++ .../MergeTree/ReplicatedExportTTLIndex.cpp | 43 +++++++++++-------- .../MergeTree/ReplicatedExportTTLIndex.h | 31 ++++++------- .../ReplicatedExportTTLScheduler.cpp | 24 +++++------ .../ReplicatedExportTaskScheduler.cpp | 2 +- .../MergeTree/ReplicatedExportTaskUpdater.cpp | 2 +- src/Storages/StorageReplicatedMergeTree.cpp | 31 ++++++------- src/Storages/StorageReplicatedMergeTree.h | 9 ++-- .../test_export_ttl/test_lifecycle.py | 8 ++-- .../test_export_ttl/test_replication.py | 41 ++++++++++++++++++ ...21_system_zookeeper_unrestricted.reference | 6 ++- ...stem_zookeeper_unrestricted_like.reference | 6 ++- 18 files changed, 146 insertions(+), 87 deletions(-) diff --git a/src/Storages/MergeTree/Compaction/MergePredicates/ReplicatedMergeTreeMergePredicate.cpp b/src/Storages/MergeTree/Compaction/MergePredicates/ReplicatedMergeTreeMergePredicate.cpp index 75cf801840e5..b4ecf4638d22 100644 --- a/src/Storages/MergeTree/Compaction/MergePredicates/ReplicatedMergeTreeMergePredicate.cpp +++ b/src/Storages/MergeTree/Compaction/MergePredicates/ReplicatedMergeTreeMergePredicate.cpp @@ -129,9 +129,9 @@ ReplicatedMergeTreeZooKeeperMergePredicate::ReplicatedMergeTreeZooKeeperMergePre void ReplicatedMergeTreeZooKeeperMergePredicate::loadExportFence(zkutil::ZooKeeperPtr & zookeeper) { - if (const auto export_ttl_index = queue.storage.getExportFence()) + if (const auto export_ttl_index = queue.storage.getExportTTLIndex()) { - std::tie(export_fence, export_fence_version) = export_ttl_index->getForMergeAssignment(zookeeper); + std::tie(export_fence, export_index_version) = export_ttl_index->getForMergeAssignment(zookeeper); export_fence_ptr = export_fence.get(); } delete_gate = queue.storage.getExportTTLDeleteGate(); diff --git a/src/Storages/MergeTree/Compaction/MergePredicates/ReplicatedMergeTreeMergePredicate.h b/src/Storages/MergeTree/Compaction/MergePredicates/ReplicatedMergeTreeMergePredicate.h index d29401c2ae41..4b3493f0da91 100644 --- a/src/Storages/MergeTree/Compaction/MergePredicates/ReplicatedMergeTreeMergePredicate.h +++ b/src/Storages/MergeTree/Compaction/MergePredicates/ReplicatedMergeTreeMergePredicate.h @@ -69,9 +69,9 @@ class ReplicatedMergeTreeZooKeeperMergePredicate final : public ReplicatedMergeT /// other users of the predicate do not merge parts and skip the Keeper reads. void loadExportFence(zkutil::ZooKeeperPtr & zookeeper); - /// The version of the "export_fence" node the export states were read at, or -1 if they were not - /// loaded. A merge must be assigned with a check of it, so it is not assigned from stale export states. - int32_t getExportFenceVersion() const { return export_fence_version; } + /// The version of the "export_ttl/version" node the export states were read at, or -1 if they were + /// not loaded. A merge must be assigned with a check of it, so it is not assigned from stale export states. + int32_t getExportIndexVersion() const { return export_index_version; } /// Returns true if there's a drop range covering new_drop_range_info bool isGoingToBeDropped(const MergeTreePartInfo & new_drop_range_info, MergeTreePartInfo * out_drop_range_info = nullptr) const; @@ -86,7 +86,7 @@ class ReplicatedMergeTreeZooKeeperMergePredicate final : public ReplicatedMergeT std::shared_ptr inprogress_quorum_part; int32_t merges_version = -1; - int32_t export_fence_version = -1; + int32_t export_index_version = -1; }; using ReplicatedMergeTreeMergePredicatePtr = std::shared_ptr; diff --git a/src/Storages/MergeTree/ExportTTLIndex.h b/src/Storages/MergeTree/ExportTTLIndex.h index ab3906d336dd..fc785bc4141c 100644 --- a/src/Storages/MergeTree/ExportTTLIndex.h +++ b/src/Storages/MergeTree/ExportTTLIndex.h @@ -75,7 +75,7 @@ struct ExportTTLVersionedEntry /// The whole export index of a table as of one version, with the merge fence built from it. struct ExportTTLIndexSnapshot { - /// Version of the `export_fence` node of a `ReplicatedMergeTree` it was read at: every change of + /// Version of the `export_ttl/version` node of a `ReplicatedMergeTree` it was read at: every change of /// the index bumps it, so the snapshot stays valid while it does not change. -1 for a plain `MergeTree`. int32_t version = -1; diff --git a/src/Storages/MergeTree/ExportTaskUtils.cpp b/src/Storages/MergeTree/ExportTaskUtils.cpp index d9b3e8e88d0c..d7d8bd1ab77a 100644 --- a/src/Storages/MergeTree/ExportTaskUtils.cpp +++ b/src/Storages/MergeTree/ExportTaskUtils.cpp @@ -525,7 +525,7 @@ namespace const ContextPtr & context_in, MergeTreeData & source_storage, const String & replica_name, - const ReplicatedExportTTLIndex & export_fence) + const ReplicatedExportTTLIndex & export_ttl_index) { /// Failpoint used by integration tests to force persistent commit failure and exercise /// the commit-attempts budget / FAILED state transition. @@ -654,14 +654,14 @@ namespace if (is_ttl_task) { - auto versioned = export_fence.readIndexEntry(zk, destination_key, partition_id); + auto versioned = export_ttl_index.readIndexEntry(zk, destination_key, partition_id); if (versioned.version < 0) - export_fence.ensureDestination(zk, destination_key, fmt::format("{}.{}", manifest.destination_database, manifest.destination_table)); + export_ttl_index.ensureDestination(zk, destination_key, fmt::format("{}.{}", manifest.destination_database, manifest.destination_table)); versioned.entry.commitClaim(manifest.transaction_id, ExportTTLUtils::rangesOfParts(manifest.parts, source_storage.format_version)); for (const auto & retried : manifest.retry_of) versioned.entry.releaseClaim(retried.transaction_id); - export_fence.appendUpdateEntryOps(ops, destination_key, versioned.entry, versioned.version); + export_ttl_index.appendUpdateEntryOps(ops, destination_key, versioned.entry, versioned.version); } ProfileEvents::increment(ProfileEvents::ExportTaskZooKeeperRequests); diff --git a/src/Storages/MergeTree/ExportTaskUtils.h b/src/Storages/MergeTree/ExportTaskUtils.h index 8bfa499e207d..da6ae6d6605c 100644 --- a/src/Storages/MergeTree/ExportTaskUtils.h +++ b/src/Storages/MergeTree/ExportTaskUtils.h @@ -120,7 +120,7 @@ namespace ExportTaskUtils const ContextPtr & context, MergeTreeData & source_storage, const String & replica_name, - const ReplicatedExportTTLIndex & export_fence + const ReplicatedExportTTLIndex & export_ttl_index ); /// Handles a commit-phase failure for a replicated partition export: diff --git a/src/Storages/MergeTree/MergeTreeData.cpp b/src/Storages/MergeTree/MergeTreeData.cpp index 7d5044dc1357..05932a8be6b3 100644 --- a/src/Storages/MergeTree/MergeTreeData.cpp +++ b/src/Storages/MergeTree/MergeTreeData.cpp @@ -1322,6 +1322,9 @@ void MergeTreeData::checkExportTTL( checkExportTTLIsSupportedByDisk(new_metadata); validateExportTTL(getStorageID(), new_metadata, getSettings(), local_context); + + /// The other replicas apply the ALTER from the replication log without validating it. + checkReplicasSupportExportTTL(); } void MergeTreeData::checkExportTTLIsSupportedByDisk(const StorageInMemoryMetadata & metadata) const diff --git a/src/Storages/MergeTree/MergeTreeData.h b/src/Storages/MergeTree/MergeTreeData.h index c962c68fffeb..15dfee1ad92b 100644 --- a/src/Storages/MergeTree/MergeTreeData.h +++ b/src/Storages/MergeTree/MergeTreeData.h @@ -1770,6 +1770,9 @@ class MergeTreeData : public WithMutableContext, public IStorage, public IBackgr /// written and then renamed: a content-addressed disk cannot rename a file written before. void checkExportTTLIsSupportedByDisk(const StorageInMemoryMetadata & metadata) const; + /// Throws unless every replica keeps exported parts apart from the others when it assigns merges. + virtual void checkReplicasSupportExportTTL() const {} + public: /// Validates the `TTL ... EXPORT TO TABLE` expression of the table `table_id` being created or /// altered. The partition key of the destination is checked against the parts of every group diff --git a/src/Storages/MergeTree/ReplicatedExportTTLIndex.cpp b/src/Storages/MergeTree/ReplicatedExportTTLIndex.cpp index b92f1ea07b0b..4236247a0837 100644 --- a/src/Storages/MergeTree/ReplicatedExportTTLIndex.cpp +++ b/src/Storages/MergeTree/ReplicatedExportTTLIndex.cpp @@ -24,14 +24,14 @@ ReplicatedExportTTLIndex::ReplicatedExportTTLIndex(String zookeeper_path_, Logge { } -String ReplicatedExportTTLIndex::getFencePath() const +String ReplicatedExportTTLIndex::getRootPath() const { - return fs::path(zookeeper_path) / "export_fence"; + return fs::path(zookeeper_path) / "export_ttl"; } -String ReplicatedExportTTLIndex::getRootPath() const +String ReplicatedExportTTLIndex::getVersionPath() const { - return fs::path(zookeeper_path) / "export_ttl"; + return fs::path(getRootPath()) / "version"; } String ReplicatedExportTTLIndex::getSchedulerLockPath() const @@ -39,9 +39,14 @@ String ReplicatedExportTTLIndex::getSchedulerLockPath() const return fs::path(getRootPath()) / "scheduler_lock"; } +String ReplicatedExportTTLIndex::getDestinationsPath() const +{ + return fs::path(getRootPath()) / "destinations"; +} + String ReplicatedExportTTLIndex::getDestinationPath(const String & destination_key) const { - return fs::path(getRootPath()) / destination_key; + return fs::path(getDestinationsPath()) / destination_key; } String ReplicatedExportTTLIndex::getIndexEntryPath(const String & destination_key, const String & partition_id) const @@ -49,16 +54,19 @@ String ReplicatedExportTTLIndex::getIndexEntryPath(const String & destination_ke return fs::path(getDestinationPath(destination_key)) / "partitions" / partition_id; } -int32_t ReplicatedExportTTLIndex::readFenceVersion(const zkutil::ZooKeeperPtr & zookeeper) const +int32_t ReplicatedExportTTLIndex::readVersion(const zkutil::ZooKeeperPtr & zookeeper) const { - const auto path = getFencePath(); + const auto path = getVersionPath(); Coordination::Stat stat; if (!zookeeper->exists(path, &stat)) { - const auto code = zookeeper->tryCreate(path, "", zkutil::CreateMode::Persistent); - if (code != Coordination::Error::ZOK && code != Coordination::Error::ZNODEEXISTS) - throw zkutil::KeeperException::fromPath(code, path); + for (const auto & node : {getRootPath(), path}) + { + const auto code = zookeeper->tryCreate(node, "", zkutil::CreateMode::Persistent); + if (code != Coordination::Error::ZOK && code != Coordination::Error::ZNODEEXISTS) + throw zkutil::KeeperException::fromPath(code, node); + } zookeeper->exists(path, &stat); } return stat.version; @@ -67,13 +75,11 @@ int32_t ReplicatedExportTTLIndex::readFenceVersion(const zkutil::ZooKeeperPtr & std::vector ReplicatedExportTTLIndex::listDestinations(const zkutil::ZooKeeperPtr & zookeeper) const { Strings children; - const auto code = zookeeper->tryGetChildren(getRootPath(), children); + const auto code = zookeeper->tryGetChildren(getDestinationsPath(), children); if (code == Coordination::Error::ZNONODE) return {}; if (code != Coordination::Error::ZOK) - throw zkutil::KeeperException::fromPath(code, getRootPath()); - - std::erase_if(children, [](const String & child) { return child == "scheduler_lock"; }); + throw zkutil::KeeperException::fromPath(code, getDestinationsPath()); return children; } @@ -133,6 +139,7 @@ void ReplicatedExportTTLIndex::ensureDestination( { const std::vector> nodes{ {getRootPath(), ""}, + {getDestinationsPath(), ""}, {getDestinationPath(destination_key), description}, {fs::path(getDestinationPath(destination_key)) / "partitions", ""}, }; @@ -154,13 +161,13 @@ void ReplicatedExportTTLIndex::appendUpdateEntryOps( else ops.emplace_back(zkutil::makeSetRequest(path, entry.toJSONString(), version)); - ops.emplace_back(zkutil::makeSetRequest(getFencePath(), "", -1)); + ops.emplace_back(zkutil::makeSetRequest(getVersionPath(), "", -1)); } void ReplicatedExportTTLIndex::removeDestination(const zkutil::ZooKeeperPtr & zookeeper, const String & destination_key) const { zookeeper->tryRemoveRecursive(getDestinationPath(destination_key)); - zookeeper->trySet(getFencePath(), "", -1); + zookeeper->trySet(getVersionPath(), "", -1); LOG_INFO(log, "Removed the TTL export index of destination {}", destination_key); } @@ -168,7 +175,7 @@ ExportTTLIndexSnapshotPtr ReplicatedExportTTLIndex::getSnapshot(const zkutil::Zo { /// Read before the index, so the index is at least as new as the version it is cached for. Every /// change of the index bumps the version in the same transaction. - const auto version = readFenceVersion(zookeeper); + const auto version = readVersion(zookeeper); if (auto cached = latest.get(); cached->version == version) return cached; @@ -183,7 +190,7 @@ ExportTTLIndexSnapshotPtr ReplicatedExportTTLIndex::getSnapshot(const zkutil::Zo ProfileEvents::increment(ProfileEvents::ExportTTLIndexSnapshotRefreshes); auto snapshot = ExportTTLIndexSnapshot::build(version, std::move(entries)); - LOG_DEBUG(log, "Read the export index at version {} of the fence: {} partition(s) with exported or claimed parts", + LOG_DEBUG(log, "Read the export index at version {}: {} partition(s) with exported or claimed parts", version, snapshot->fence->entries_by_partition.size()); latest.set(std::move(snapshot)); diff --git a/src/Storages/MergeTree/ReplicatedExportTTLIndex.h b/src/Storages/MergeTree/ReplicatedExportTTLIndex.h index 32d9104fa6b9..44c1942ec747 100644 --- a/src/Storages/MergeTree/ReplicatedExportTTLIndex.h +++ b/src/Storages/MergeTree/ReplicatedExportTTLIndex.h @@ -17,20 +17,20 @@ namespace DB /// The `TTL ... EXPORT` state of a `ReplicatedMergeTree` table in Keeper, and the merge fence built from it. /// -/// Layout under ``: -/// - `export_ttl/`: holds `ExportTTLDestination` of one destination; -/// - `export_ttl//partitions/`: an `ExportTTLIndexEntry`; -/// - `export_ttl/scheduler_lock`: ephemeral, held by the replica that schedules TTL exports, whose name it holds; -/// - `export_fence`: bumped in every transaction that changes an index entry. A merge is assigned -/// with a check of the version its predicate read the index at, so no merge is assigned from a -/// stale view of the export states. +/// Layout under `/export_ttl`: +/// - `version`: bumped in every transaction that changes an index entry. Replicas cache the index +/// for its version, and a merge is assigned with a check of the version its predicate read the +/// index at, so no merge is assigned from a stale view of the export states; +/// - `scheduler_lock`: ephemeral, held by the replica that schedules TTL exports, whose name it holds; +/// - `destinations/`: holds the name of one destination; +/// - `destinations//partitions/`: an `ExportTTLIndexEntry`. class ReplicatedExportTTLIndex { public: ReplicatedExportTTLIndex(String zookeeper_path_, LoggerPtr log_); - String getFencePath() const; String getRootPath() const; + String getVersionPath() const; String getSchedulerLockPath() const; String getDestinationPath(const String & destination_key) const; String getIndexEntryPath(const String & destination_key, const String & partition_id) const; @@ -47,19 +47,19 @@ class ReplicatedExportTTLIndex /// Creates the nodes of `destination_key`, so entries can be created in a transaction. void ensureDestination(const zkutil::ZooKeeperPtr & zookeeper, const String & destination_key, const String & description) const; - /// Appends the ops that store `entry`, with a check of `version` (see `ExportTTLVersionedEntry`), and bump the fence. + /// Appends the ops that store `entry`, with a check of `version` (see `ExportTTLVersionedEntry`), and bump the version of the index. void appendUpdateEntryOps( Coordination::Requests & ops, const String & destination_key, const ExportTTLIndexEntry & entry, int32_t version) const; - /// Removes the index of `destination_key` and bumps the fence. Not transactional: an interrupted - /// removal leaves fewer entries, which only lifts more of the fence. + /// Removes the index of `destination_key` and bumps the version of the index. Not transactional: + /// an interrupted removal leaves fewer entries, which only lifts more of the fence. void removeDestination(const zkutil::ZooKeeperPtr & zookeeper, const String & destination_key) const; - /// The whole index as of the current version of the fence node. It is read again only when that - /// version changed, so while nothing changes this costs one `exists`. + /// The whole index as of the current version of the `version` node. It is read again only when + /// that version changed, so while nothing changes this costs one `exists`. ExportTTLIndexSnapshotPtr getSnapshot(const zkutil::ZooKeeperPtr & zookeeper); - /// The export states as of the returned version of the fence node. + /// The export states as of the returned version of the `version` node. std::pair getForMergeAssignment(const zkutil::ZooKeeperPtr & zookeeper); /// The export states read by the last `getSnapshot`, without reading Keeper. @@ -73,7 +73,8 @@ class ReplicatedExportTTLIndex std::mutex refresh_mutex; MultiVersion latest; - int32_t readFenceVersion(const zkutil::ZooKeeperPtr & zookeeper) const; + String getDestinationsPath() const; + int32_t readVersion(const zkutil::ZooKeeperPtr & zookeeper) const; }; using ReplicatedExportTTLIndexPtr = std::shared_ptr; diff --git a/src/Storages/MergeTree/ReplicatedExportTTLScheduler.cpp b/src/Storages/MergeTree/ReplicatedExportTTLScheduler.cpp index 2731a3df8506..36fb341dd9da 100644 --- a/src/Storages/MergeTree/ReplicatedExportTTLScheduler.cpp +++ b/src/Storages/MergeTree/ReplicatedExportTTLScheduler.cpp @@ -57,13 +57,13 @@ bool ReplicatedExportTTLScheduler::acquireSchedulerLock() lock_holder.reset(); lock_zookeeper.reset(); - const auto & fence = *replicated_storage.export_fence; - const auto root_path = fence.getRootPath(); + const auto & export_ttl_index = *replicated_storage.export_ttl_index; + const auto root_path = export_ttl_index.getRootPath(); if (const auto code = zookeeper->tryCreate(root_path, "", zkutil::CreateMode::Persistent); code != Coordination::Error::ZOK && code != Coordination::Error::ZNODEEXISTS) throw zkutil::KeeperException::fromPath(code, root_path); - lock_holder = zkutil::EphemeralNodeHolder::tryCreate(fence.getSchedulerLockPath(), *zookeeper, replicated_storage.getReplicaName()); + lock_holder = zkutil::EphemeralNodeHolder::tryCreate(export_ttl_index.getSchedulerLockPath(), *zookeeper, replicated_storage.getReplicaName()); if (!lock_holder) return false; @@ -79,7 +79,7 @@ bool ReplicatedExportTTLScheduler::isPaused() ExportTTLIndexSnapshotPtr ReplicatedExportTTLScheduler::getIndexSnapshot() { - return replicated_storage.export_fence->getSnapshot(replicated_storage.getZooKeeper()); + return replicated_storage.export_ttl_index->getSnapshot(replicated_storage.getZooKeeper()); } String ReplicatedExportTTLScheduler::getReplicaName() const @@ -94,7 +94,7 @@ String ReplicatedExportTTLScheduler::getSchedulerReplica() return {}; String replica; - zookeeper->tryGet(replicated_storage.export_fence->getSchedulerLockPath(), replica); + zookeeper->tryGet(replicated_storage.export_ttl_index->getSchedulerLockPath(), replica); return replica; } @@ -193,13 +193,13 @@ ExportTTLScheduler::TaskState ReplicatedExportTTLScheduler::getTaskState(const S bool ReplicatedExportTTLScheduler::updateIndexEntry(const String & destination_key, const ExportTTLVersionedEntry & entry) { const auto zookeeper = replicated_storage.getZooKeeper(); - const auto & fence = *replicated_storage.export_fence; + const auto & export_ttl_index = *replicated_storage.export_ttl_index; if (entry.version < 0) - fence.ensureDestination(zookeeper, destination_key, destination_key); + export_ttl_index.ensureDestination(zookeeper, destination_key, destination_key); Coordination::Requests ops; - fence.appendUpdateEntryOps(ops, destination_key, entry.entry, entry.version); + export_ttl_index.appendUpdateEntryOps(ops, destination_key, entry.entry, entry.version); Coordination::Responses responses; const auto code = zookeeper->tryMulti(ops, responses); @@ -241,16 +241,16 @@ bool ReplicatedExportTTLScheduler::startGroup(const GroupToStart & group, const manifest.source = ExportTaskSource::ttl; manifest.retry_of = group.retry_of; - const auto & fence = *replicated_storage.export_fence; + const auto & export_ttl_index = *replicated_storage.export_ttl_index; if (group.entry.version < 0) - fence.ensureDestination(zookeeper, group.destination_key, group.destination->getStorageID().getNameForLogs()); + export_ttl_index.ensureDestination(zookeeper, group.destination_key, group.destination->getStorageID().getNameForLogs()); const auto task_path = fs::path(replicated_storage.zookeeper_path) / "exports" / group.transaction_id; Coordination::Requests ops; ops.emplace_back(zkutil::makeCheckRequest(fs::path(replicated_storage.zookeeper_path) / "log", merge_predicate->getVersion())); ExportTaskUtils::appendCreateExportTaskOps(ops, task_path, manifest); - fence.appendUpdateEntryOps(ops, group.destination_key, group.entry.entry, group.entry.version); + export_ttl_index.appendUpdateEntryOps(ops, group.destination_key, group.entry.entry, group.entry.version); ProfileEvents::increment(ProfileEvents::ExportTaskZooKeeperRequests); ProfileEvents::increment(ProfileEvents::ExportTaskZooKeeperMulti); @@ -283,7 +283,7 @@ void ReplicatedExportTTLScheduler::killTask(const String & transaction_id) void ReplicatedExportTTLScheduler::removeDestination(const String & destination_key) { - replicated_storage.export_fence->removeDestination(replicated_storage.getZooKeeper(), destination_key); + replicated_storage.export_ttl_index->removeDestination(replicated_storage.getZooKeeper(), destination_key); } } diff --git a/src/Storages/MergeTree/ReplicatedExportTaskScheduler.cpp b/src/Storages/MergeTree/ReplicatedExportTaskScheduler.cpp index f7b055d93215..629992a9f4ba 100644 --- a/src/Storages/MergeTree/ReplicatedExportTaskScheduler.cpp +++ b/src/Storages/MergeTree/ReplicatedExportTaskScheduler.cpp @@ -408,7 +408,7 @@ void ReplicatedExportTaskScheduler::handlePartExportSuccess( try { auto context = ExportTaskUtils::getContextCopyWithTaskSettings(storage.getContext(), manifest); - ExportTaskUtils::commit(manifest, destination_storage, zk, storage.log.load(), export_path, context, storage, storage.replica_name, *storage.export_fence); + ExportTaskUtils::commit(manifest, destination_storage, zk, storage.log.load(), export_path, context, storage, storage.replica_name, *storage.export_ttl_index); } catch (const Exception & e) { diff --git a/src/Storages/MergeTree/ReplicatedExportTaskUpdater.cpp b/src/Storages/MergeTree/ReplicatedExportTaskUpdater.cpp index d6fd42c38584..19ffb7be647b 100644 --- a/src/Storages/MergeTree/ReplicatedExportTaskUpdater.cpp +++ b/src/Storages/MergeTree/ReplicatedExportTaskUpdater.cpp @@ -713,7 +713,7 @@ void ReplicatedExportTaskUpdater::poll() /// A replica exported the last part but the commit never landed. Try to fix it. try { - ExportTaskUtils::commit(work.metadata, work.destination_storage, zk, log, work.entry_path, work.context, storage, storage.getReplicaName(), *storage.export_fence); + ExportTaskUtils::commit(work.metadata, work.destination_storage, zk, log, work.entry_path, work.context, storage, storage.getReplicaName(), *storage.export_ttl_index); } catch (const Exception & e) { diff --git a/src/Storages/StorageReplicatedMergeTree.cpp b/src/Storages/StorageReplicatedMergeTree.cpp index 5a4a76dfa9cb..7ba1ce895616 100644 --- a/src/Storages/StorageReplicatedMergeTree.cpp +++ b/src/Storages/StorageReplicatedMergeTree.cpp @@ -591,7 +591,7 @@ StorageReplicatedMergeTree::StorageReplicatedMergeTree( export_task_select_task->deactivate(); - export_fence = std::make_shared(zookeeper_path, log.load()); + export_ttl_index = std::make_shared(zookeeper_path, log.load()); export_ttl_scheduler = std::make_shared(*this); export_ttl_task = getContext()->getSchedulePool().createTask( @@ -1061,7 +1061,8 @@ void StorageReplicatedMergeTree::createNewZooKeeperNodesAttempt() const futures.push_back(zookeeper->asyncTryCreateNoThrow(zookeeper_path + "/quorum/failed_parts", String(), zkutil::CreateMode::Persistent)); futures.push_back(zookeeper->asyncTryCreateNoThrow(zookeeper_path + "/mutations", String(), zkutil::CreateMode::Persistent)); futures.push_back(zookeeper->asyncTryCreateNoThrow(zookeeper_path + "/exports", String(), zkutil::CreateMode::Persistent)); - futures.push_back(zookeeper->asyncTryCreateNoThrow(zookeeper_path + "/export_fence", String(), zkutil::CreateMode::Persistent)); + futures.push_back(zookeeper->asyncTryCreateNoThrow(zookeeper_path + "/export_ttl", String(), zkutil::CreateMode::Persistent)); + futures.push_back(zookeeper->asyncTryCreateNoThrow(zookeeper_path + "/export_ttl/version", String(), zkutil::CreateMode::Persistent)); futures.push_back(zookeeper->asyncTryCreateNoThrow(zookeeper_path + "/quorum/parallel", String(), zkutil::CreateMode::Persistent)); @@ -4641,7 +4642,7 @@ void StorageReplicatedMergeTree::mergeSelectingTask() cleanup, nullptr, merge_predicate->getVersion(), - merge_predicate->getExportFenceVersion(), + merge_predicate->getExportIndexVersion(), future_merged_part->merge_type); if (create_result == CreateMergeEntryResult::Ok) @@ -4887,7 +4888,7 @@ StorageReplicatedMergeTree::CreateMergeEntryResult StorageReplicatedMergeTree::c bool cleanup, ReplicatedMergeTreeLogEntryData * out_log_entry, int32_t log_version, - int32_t export_fence_version, + int32_t export_index_version, MergeType merge_type) { Strings exists_paths; @@ -4946,8 +4947,8 @@ StorageReplicatedMergeTree::CreateMergeEntryResult StorageReplicatedMergeTree::c /// The merge was checked against the export states at this version: parts exported, being /// exported and not exported by the `EXPORT` TTL must not be merged together. - if (export_fence_version >= 0 && export_fence) - ops.emplace_back(zkutil::makeCheckRequest(export_fence->getFencePath(), export_fence_version)); + if (export_index_version >= 0 && export_ttl_index) + ops.emplace_back(zkutil::makeCheckRequest(export_ttl_index->getVersionPath(), export_index_version)); Coordination::Error code = zookeeper->tryMulti(ops, responses); @@ -6858,7 +6859,7 @@ bool StorageReplicatedMergeTree::optimize( cleanup, &merge_entry, merge_predicate->getVersion(), - merge_predicate->getExportFenceVersion(), + merge_predicate->getExportIndexVersion(), select_merge_result.value()->merge_type); if (create_result == CreateMergeEntryResult::MissingPart) @@ -8872,8 +8873,8 @@ void StorageReplicatedMergeTree::checkAllReplicasSupportExportTTL(const zkutil:: if (!unsupported.empty()) throw Exception(ErrorCodes::SUPPORT_IS_DISABLED, - "Cannot export by TTL: replica(s) {} would merge exported parts with parts that are not exported. " - "Every replica must run a version that supports it with the server setting `allow_experimental_export_merge_tree_partition` " + "The `EXPORT` TTL is not supported by replica(s) {}, which would merge exported parts with parts that are not exported. " + "Every replica must run a version that supports it, with the server setting `allow_experimental_export_merge_tree_partition` " "enabled; drop the replicas that are lost", fmt::join(unsupported, ", ")); } @@ -8882,7 +8883,7 @@ void StorageReplicatedMergeTree::advertiseExportFeatures(const zkutil::ZooKeeper { const String path = fs::path(replica_path) / "export_features"; - if (export_fence) + if (export_ttl_index) { const auto code = zookeeper->tryCreate(path, "ttl", zkutil::CreateMode::Persistent); if (code != Coordination::Error::ZOK && code != Coordination::Error::ZNODEEXISTS) @@ -8930,11 +8931,11 @@ void StorageReplicatedMergeTree::forgetPartition(const ASTPtr & partition, Conte /// Block numbers of the partition start over, so a new part could reuse the block range of a part /// exported by the `EXPORT` TTL. The partition's index entries go with them, in the same transaction. - if (export_fence) + if (export_ttl_index) { - for (const auto & destination_key : export_fence->listDestinations(zookeeper)) + for (const auto & destination_key : export_ttl_index->listDestinations(zookeeper)) { - const auto versioned = export_fence->readIndexEntry(zookeeper, destination_key, partition_id); + const auto versioned = export_ttl_index->readIndexEntry(zookeeper, destination_key, partition_id); if (versioned.version < 0) continue; @@ -8943,11 +8944,11 @@ void StorageReplicatedMergeTree::forgetPartition(const ASTPtr & partition, Conte "Partition {} is being exported by the EXPORT TTL (task {}), retry after it finishes", partition_id, versioned.entry.claimed.begin()->first); - ops.emplace_back(zkutil::makeRemoveRequest(export_fence->getIndexEntryPath(destination_key, partition_id), versioned.version)); + ops.emplace_back(zkutil::makeRemoveRequest(export_ttl_index->getIndexEntryPath(destination_key, partition_id), versioned.version)); } if (ops.size() > 1) - ops.emplace_back(zkutil::makeSetRequest(export_fence->getFencePath(), "", -1)); + ops.emplace_back(zkutil::makeSetRequest(export_ttl_index->getVersionPath(), "", -1)); } Coordination::Responses responses; diff --git a/src/Storages/StorageReplicatedMergeTree.h b/src/Storages/StorageReplicatedMergeTree.h index f1fb6b08c272..fc908340e407 100644 --- a/src/Storages/StorageReplicatedMergeTree.h +++ b/src/Storages/StorageReplicatedMergeTree.h @@ -380,8 +380,8 @@ class StorageReplicatedMergeTree final : public MergeTreeData std::vector getExportTasksInfo() const override; /// nullptr if partition export is disabled. - ReplicatedExportTTLIndexPtr getExportFence() const { return export_fence; } - ExportFencePtr getLatestExportFence() const override { return export_fence ? export_fence->getLatest() : nullptr; } + ReplicatedExportTTLIndexPtr getExportTTLIndex() const { return export_ttl_index; } + ExportFencePtr getLatestExportFence() const override { return export_ttl_index ? export_ttl_index->getLatest() : nullptr; } private: std::atomic_bool are_restoring_replica {false}; @@ -541,7 +541,7 @@ class StorageReplicatedMergeTree final : public MergeTreeData /// The export index of the `EXPORT` TTL and the merge fence built from it. Only created when /// partition export is enabled; a replica without it must not assign merges of parts exported by /// the TTL, which is why it does not advertise `export_features`. - ReplicatedExportTTLIndexPtr export_fence; + ReplicatedExportTTLIndexPtr export_ttl_index; /// Runs `export_ttl_scheduler`. BackgroundSchedulePoolTaskHolder export_ttl_task; @@ -804,7 +804,7 @@ class StorageReplicatedMergeTree final : public MergeTreeData bool cleanup, ReplicatedMergeTreeLogEntryData * out_log_entry, int32_t log_version, - int32_t export_fence_version, + int32_t export_index_version, MergeType merge_type); CreateMergeEntryResult createLogEntryToMutatePart( @@ -999,6 +999,7 @@ class StorageReplicatedMergeTree final : public MergeTreeData /// Throws unless every replica enforces the export states of parts when assigning merges. void checkAllReplicasSupportExportTTL(const zkutil::ZooKeeperPtr & zookeeper) const; + void checkReplicasSupportExportTTL() const override { checkAllReplicasSupportExportTTL(getZooKeeper()); } /// Creates (or removes, if partition export is disabled) `/export_features`. void advertiseExportFeatures(const zkutil::ZooKeeperPtr & zookeeper) const; diff --git a/tests/integration/test_export_ttl/test_lifecycle.py b/tests/integration/test_export_ttl/test_lifecycle.py index 90cd34d79751..8ef07c9847a7 100644 --- a/tests/integration/test_export_ttl/test_lifecycle.py +++ b/tests/integration/test_export_ttl/test_lifecycle.py @@ -152,11 +152,9 @@ def test_forget_partition(cluster): node.query(f"INSERT INTO {mt_table} VALUES (1, 2020, {DUE})") wait_for_partitions_exported(node, mt_table, ["2020"]) - index_path = f"{zookeeper_path(mt_table)}/export_ttl" - destination_key = node.query( - f"SELECT name FROM system.zookeeper WHERE path = '{index_path}' AND name != 'scheduler_lock'" - ).strip() - partitions_path = f"{index_path}/{destination_key}/partitions" + destinations_path = f"{zookeeper_path(mt_table)}/export_ttl/destinations" + destination_key = node.query(f"SELECT name FROM system.zookeeper WHERE path = '{destinations_path}'").strip() + partitions_path = f"{destinations_path}/{destination_key}/partitions" assert node.query(f"SELECT name FROM system.zookeeper WHERE path = '{partitions_path}'") == "2020\n" node.query(f"ALTER TABLE {mt_table} DROP PARTITION ID '2020'") diff --git a/tests/integration/test_export_ttl/test_replication.py b/tests/integration/test_export_ttl/test_replication.py index 63f024439bb6..af147659cbb6 100644 --- a/tests/integration/test_export_ttl/test_replication.py +++ b/tests/integration/test_export_ttl/test_replication.py @@ -23,6 +23,7 @@ wait_for_partitions_exported, wait_for_same_ttl_rows, wait_until, + zookeeper_path, ) CLUSTER_INSTANCES = ["replica1", "replica2"] @@ -176,6 +177,46 @@ def test_failover_resumes_the_batch(cluster): assert_one_snapshot_per_task(other, mt_table, iceberg_table) +def test_alter_is_refused_while_a_replica_does_not_support_it(cluster): + """The other replicas apply an ALTER from the replication log without validating it. A replica that + does not keep exported parts apart from the others when it merges, e.g. because it runs an older + version, does not advertise `export_features`, and an ALTER that adds the `EXPORT` TTL is refused + while there is one.""" + replicas = [cluster.instances["replica1"], cluster.instances["replica2"]] + suffix = unique_suffix() + mt_table, iceberg_table = f"repl_mt_{suffix}", f"repl_iceberg_{suffix}" + create_iceberg(replicas, iceberg_table) + for replica in replicas: + create_source( + replica, mt_table, COLUMNS, "year", "t + INTERVAL 10 YEAR DELETE", + engine="ReplicatedMergeTree", replica_name=replica.name, + ) + add_export_ttl = f"ALTER TABLE {mt_table} MODIFY TTL t + INTERVAL 1 DAY EXPORT TO TABLE {iceberg_table}" + + export_features = f"{zookeeper_path(mt_table)}/replicas/replica2/export_features" + zk = cluster.get_kazoo_client("zoo1") + try: + zk.delete(export_features) + try: + error = replicas[0].query_and_get_error(add_export_ttl) + assert "SUPPORT_IS_DISABLED" in error and "replica2" in error, error + finally: + # A replica advertises it when its table starts up. + replicas[1].restart_clickhouse() + wait_until(lambda: zk.exists(export_features), 60, "replica2 does not advertise export_features again") + finally: + zk.stop() + zk.close() + + replicas[0].query(add_export_ttl) + replicas[0].query(f"INSERT INTO {mt_table} VALUES (1, 2020, {DUE})") + sync(replicas, mt_table) + wait_for_partitions_exported(replicas[0], mt_table, ["2020"]) + for replica in replicas: + assert_exactly_once(iceberg_ids(replica, iceberg_table), [1]) + assert_one_snapshot_per_task(replicas[0], mt_table, iceberg_table) + + def test_detached_scheduler_hands_over(cluster): replicas = [cluster.instances["replica1"], cluster.instances["replica2"]] mt_table, iceberg_table = make_replicated_tables(replicas) diff --git a/tests/queries/0_stateless/02221_system_zookeeper_unrestricted.reference b/tests/queries/0_stateless/02221_system_zookeeper_unrestricted.reference index 8242a05aa572..8e617317fd28 100644 --- a/tests/queries/0_stateless/02221_system_zookeeper_unrestricted.reference +++ b/tests/queries/0_stateless/02221_system_zookeeper_unrestricted.reference @@ -22,8 +22,8 @@ deduplication_hashes deduplication_hashes export_features export_features -export_fence -export_fence +export_ttl +export_ttl exports exports failed_parts @@ -86,3 +86,5 @@ table_shared_id table_shared_id temp temp +version +version diff --git a/tests/queries/0_stateless/02221_system_zookeeper_unrestricted_like.reference b/tests/queries/0_stateless/02221_system_zookeeper_unrestricted_like.reference index a0ab94f91807..ec5847d877c8 100644 --- a/tests/queries/0_stateless/02221_system_zookeeper_unrestricted_like.reference +++ b/tests/queries/0_stateless/02221_system_zookeeper_unrestricted_like.reference @@ -10,7 +10,7 @@ columns creator_info deduplication_hashes export_features -export_fence +export_ttl exports failed_parts flags @@ -42,6 +42,7 @@ quorum replicas table_shared_id temp +version ------------------------- 1 abandonable_lock-insert @@ -55,7 +56,7 @@ columns creator_info deduplication_hashes export_features -export_fence +export_ttl exports failed_parts flags @@ -87,3 +88,4 @@ quorum replicas table_shared_id temp +version From 0435e835282ebe918fdc1fdb9f798b353c13cf97 Mon Sep 17 00:00:00 2001 From: Arthur Passos Date: Tue, 29 Sep 2026 14:37:26 -0300 Subject: [PATCH 09/15] fast teste --- .../queries/0_stateless/01700_system_zookeeper_path_in.reference | 1 + 1 file changed, 1 insertion(+) diff --git a/tests/queries/0_stateless/01700_system_zookeeper_path_in.reference b/tests/queries/0_stateless/01700_system_zookeeper_path_in.reference index 0c2bba9bf880..794199157468 100644 --- a/tests/queries/0_stateless/01700_system_zookeeper_path_in.reference +++ b/tests/queries/0_stateless/01700_system_zookeeper_path_in.reference @@ -15,3 +15,4 @@ failed_parts in_progress last_part parallel +version From 07e2f4628977eb138484e38413a89460cd5db56c Mon Sep 17 00:00:00 2001 From: Arthur Passos Date: Wed, 30 Sep 2026 08:44:49 -0300 Subject: [PATCH 10/15] vibe coded simplification --- src/Storages/MergeTree/ExportTTLIndex.cpp | 77 +++---- src/Storages/MergeTree/ExportTTLIndex.h | 34 +-- src/Storages/MergeTree/ExportTTLScheduler.cpp | 206 +++++++----------- src/Storages/MergeTree/ExportTTLScheduler.h | 17 +- src/Storages/MergeTree/ExportTaskUtils.cpp | 2 - .../tests/gtest_export_ttl_index.cpp | 67 ++++-- src/Storages/StorageReplicatedMergeTree.cpp | 10 +- src/Storages/StorageReplicatedMergeTree.h | 2 +- 8 files changed, 199 insertions(+), 216 deletions(-) diff --git a/src/Storages/MergeTree/ExportTTLIndex.cpp b/src/Storages/MergeTree/ExportTTLIndex.cpp index 1629accdbb1e..eb7bf98b35c3 100644 --- a/src/Storages/MergeTree/ExportTTLIndex.cpp +++ b/src/Storages/MergeTree/ExportTTLIndex.cpp @@ -16,6 +16,7 @@ namespace DB namespace ErrorCodes { extern const int INCORRECT_DATA; + extern const int LOGICAL_ERROR; } namespace @@ -58,25 +59,16 @@ Int64 ExportTTLIndexEntry::maxBlock() const Int64 result = 0; for (const auto & range : exported) result = std::max(result, range.max_block); - for (const auto & [_, ranges] : claimed) - for (const auto & range : ranges) + if (claim) + for (const auto & range : claim->ranges) result = std::max(result, range.max_block); return result; } -std::vector ExportTTLIndexEntry::allClaimed() const -{ - std::vector result; - for (const auto & [_, ranges] : claimed) - result.insert(result.end(), ranges.begin(), ranges.end()); - return ExportFenceUtils::compactRanges(std::move(result)); -} - PartExportState ExportTTLIndexEntry::classify(const MergeTreePartInfo & part) const { - for (const auto & [_, ranges] : claimed) - if (ExportFenceUtils::intersectsAny(part, ranges)) - return PartExportState::CLAIMED; + if (claim && ExportFenceUtils::intersectsAny(part, claim->ranges)) + return PartExportState::CLAIMED; if (ExportFenceUtils::intersectsAny(part, exported)) return PartExportState::EXPORTED; return PartExportState::NONE; @@ -87,58 +79,48 @@ ExportFenceEntry ExportTTLIndexEntry::toFenceEntry(const String & destination) c ExportFenceEntry entry; entry.destination = destination; entry.exported = exported; - entry.claimed = allClaimed(); + if (claim) + entry.claimed = claim->ranges; return entry; } -void ExportTTLIndexEntry::claim(const String & transaction_id, const std::vector & parts) +void ExportTTLIndexEntry::startClaim(const String & transaction_id, const std::vector & parts) { - auto & ranges = claimed[transaction_id]; + if (claim) + throw Exception(ErrorCodes::LOGICAL_ERROR, "Export task {} cannot claim parts of partition {}, which export task {} claimed", + transaction_id, partition_id, claim->transaction_id); + + std::vector ranges; + ranges.reserve(parts.size()); for (const auto & part : parts) ranges.emplace_back(partition_id, part.min_block, part.max_block, 0, 0); - ranges = ExportFenceUtils::compactRanges(std::move(ranges)); -} - -void ExportTTLIndexEntry::moveClaims(const std::vector & from, const String & to) -{ - auto & target = claimed[to]; - for (const auto & transaction_id : from) - { - const auto it = claimed.find(transaction_id); - if (it == claimed.end() || transaction_id == to) - continue; - target.insert(target.end(), it->second.begin(), it->second.end()); - claimed.erase(it); - } - target = ExportFenceUtils::compactRanges(std::move(target)); + claim = Claim{.transaction_id = transaction_id, .ranges = ExportFenceUtils::compactRanges(std::move(ranges))}; } void ExportTTLIndexEntry::commitClaim(const String & transaction_id, const std::vector & parts) { - if (const auto it = claimed.find(transaction_id); it != claimed.end()) + if (claim && claim->transaction_id == transaction_id) { - exported.insert(exported.end(), it->second.begin(), it->second.end()); - claimed.erase(it); + exported.insert(exported.end(), claim->ranges.begin(), claim->ranges.end()); + claim.reset(); } for (const auto & part : parts) exported.emplace_back(partition_id, part.min_block, part.max_block, 0, 0); exported = ExportFenceUtils::compactRanges(std::move(exported)); } -void ExportTTLIndexEntry::releaseClaim(const String & transaction_id) -{ - claimed.erase(transaction_id); -} - String ExportTTLIndexEntry::toJSONString() const { Poco::JSON::Object json; json.set("exported", rangesToJSON(exported)); - Poco::JSON::Object::Ptr claimed_object = new Poco::JSON::Object(); - for (const auto & [transaction_id, ranges] : claimed) - claimed_object->set(transaction_id, rangesToJSON(ranges)); - json.set("claimed", claimed_object); + if (claim) + { + Poco::JSON::Object::Ptr claim_object = new Poco::JSON::Object(); + claim_object->set("transaction_id", claim->transaction_id); + claim_object->set("ranges", rangesToJSON(claim->ranges)); + json.set("claim", claim_object); + } std::ostringstream oss; // STYLE_CHECK_ALLOW_STD_STRING_STREAM oss.exceptions(std::ios::failbit); @@ -160,11 +142,12 @@ ExportTTLIndexEntry ExportTTLIndexEntry::fromJSONString(const String & partition entry.exported = ExportFenceUtils::compactRanges(rangesFromJSON(partition_id, json->getArray("exported"))); - if (const auto claimed_object = json->getObject("claimed")) + if (const auto claim_object = json->getObject("claim")) { - for (const auto & transaction_id : claimed_object->getNames()) - entry.claimed[transaction_id] = ExportFenceUtils::compactRanges( - rangesFromJSON(partition_id, claimed_object->getArray(transaction_id))); + entry.claim = Claim{ + .transaction_id = claim_object->getValue("transaction_id"), + .ranges = ExportFenceUtils::compactRanges(rangesFromJSON(partition_id, claim_object->getArray("ranges"))), + }; } return entry; diff --git a/src/Storages/MergeTree/ExportTTLIndex.h b/src/Storages/MergeTree/ExportTTLIndex.h index fc785bc4141c..144fb5330b0c 100644 --- a/src/Storages/MergeTree/ExportTTLIndex.h +++ b/src/Storages/MergeTree/ExportTTLIndex.h @@ -7,6 +7,7 @@ #include #include +#include #include #include @@ -24,41 +25,44 @@ namespace DB /// committed to the destination: /// - creating a task claims the ranges of its parts; /// - committing a task moves its claim to `exported`; -/// - retrying failed tasks moves their claims to the new task; +/// - retrying a failed task moves its claim to the new task; /// - resolving a failed task that does not need a retry releases its claim. struct ExportTTLIndexEntry { + /// Ranges owned by a TTL export task that did not commit: in flight, or failed and waiting to be retried. + struct Claim + { + String transaction_id; + /// Compacted. + std::vector ranges; + }; + String partition_id; /// Ranges committed to the destination, compacted. std::vector exported; - /// Ranges owned by TTL export tasks that did not commit, by transaction id, compacted. A task - /// may be in flight, or failed and waiting to be retried. - std::map> claimed; + /// A partition has at most one claim: a group starts only while no task of the partition is in + /// flight, and takes over the claim of the failed task it retries. + std::optional claim; - bool empty() const { return exported.empty() && claimed.empty(); } + bool empty() const { return exported.empty() && !claim; } /// Highest block number of any exported or claimed range, 0 if there is none. Int64 maxBlock() const; - std::vector allClaimed() const; - PartExportState classify(const MergeTreePartInfo & part) const; ExportFenceEntry toFenceEntry(const String & destination) const; - /// Adds the ranges of `parts` to the claim of `transaction_id`. - void claim(const String & transaction_id, const std::vector & parts); - - /// Moves the claims of `from` to `to`. - void moveClaims(const std::vector & from, const String & to); + /// Claims the ranges of `parts` for `transaction_id`. Throws if the entry has a claim already. + void startClaim(const String & transaction_id, const std::vector & parts); - /// Moves the claim of `transaction_id` to the exported ranges, adding `parts` as well: a task - /// may commit parts whose claim was lost. + /// Moves the claim to the exported ranges if `transaction_id` holds it, adding `parts` as well: + /// a task may commit parts whose claim was lost. void commitClaim(const String & transaction_id, const std::vector & parts); - void releaseClaim(const String & transaction_id); + void releaseClaim() { claim.reset(); } String toJSONString() const; static ExportTTLIndexEntry fromJSONString(const String & partition_id, const String & json_string); diff --git a/src/Storages/MergeTree/ExportTTLScheduler.cpp b/src/Storages/MergeTree/ExportTTLScheduler.cpp index d47e8a39b948..b92dff1bf4a8 100644 --- a/src/Storages/MergeTree/ExportTTLScheduler.cpp +++ b/src/Storages/MergeTree/ExportTTLScheduler.cpp @@ -14,9 +14,6 @@ #include #include -#include -#include - namespace CurrentMetrics { extern const Metric ExportTTLPartsHeldByDeleteGate; @@ -221,9 +218,9 @@ UInt64 ExportTTLScheduler::run() if (act) { for (const auto & [_, versioned] : index) - for (const auto & [transaction_id, ranges] : versioned.entry.claimed) - if (getCachedTaskState(task_states, transaction_id).status == TaskStatus::PENDING) - ++in_flight; + if (const auto & claim = versioned.entry.claim; + claim && getCachedTaskState(task_states, claim->transaction_id).status == TaskStatus::PENDING) + ++in_flight; } std::set partition_ids; @@ -310,46 +307,42 @@ UInt64 ExportTTLScheduler::run() return static_cast(std::max(1, next_tick - now)) * 1000; } -ExportTTLScheduler::ResolvedClaims ExportTTLScheduler::resolveClaims( +ExportTTLScheduler::ResolvedClaim ExportTTLScheduler::resolveClaim( ExportTTLIndexEntry & entry, const StoragePtr & destination, TaskStates & task_states, const ContextPtr & context) { - ResolvedClaims result; - - std::vector transaction_ids; - for (const auto & [transaction_id, _] : entry.claimed) - transaction_ids.push_back(transaction_id); + ResolvedClaim result; + if (!entry.claim) + return result; - for (const auto & transaction_id : transaction_ids) + const auto transaction_id = entry.claim->transaction_id; + const auto & state = getCachedTaskState(task_states, transaction_id); + switch (state.status) { - const auto & state = getCachedTaskState(task_states, transaction_id); - switch (state.status) - { - case TaskStatus::PENDING: - result.in_flight = transaction_id; - break; - - case TaskStatus::COMPLETED: - /// The commit of a plain `MergeTree` records it here, after the task is marked completed. + case TaskStatus::PENDING: + result.in_flight = transaction_id; + break; + + case TaskStatus::COMPLETED: + /// The commit of a plain `MergeTree` records it here, after the task is marked completed. + entry.commitClaim(transaction_id, {}); + result.changed = true; + break; + + case TaskStatus::FAILED: + case TaskStatus::KILLED: + case TaskStatus::MISSING: + /// E.g. a commit that landed and then the task timed out before it was marked completed. + if (state.reached_commit && destination->isExportTransactionCommitted(transaction_id, context)) + { + LOG_INFO(log, "Export task {} of partition {} did not complete, but it committed to the destination", + transaction_id, entry.partition_id); entry.commitClaim(transaction_id, {}); result.changed = true; break; + } - case TaskStatus::FAILED: - case TaskStatus::KILLED: - case TaskStatus::MISSING: - /// E.g. a commit that landed and then the task timed out before it was marked completed. - if (state.reached_commit && destination->isExportTransactionCommitted(transaction_id, context)) - { - LOG_INFO(log, "Export task {} of partition {} did not complete, but it committed to the destination", - transaction_id, entry.partition_id); - entry.commitClaim(transaction_id, {}); - result.changed = true; - break; - } - - result.failed.push_back(transaction_id); - break; - } + result.failed = transaction_id; + break; } return result; @@ -357,48 +350,39 @@ ExportTTLScheduler::ResolvedClaims ExportTTLScheduler::resolveClaims( ExportRetriedTasks ExportTTLScheduler::collectRetriedTasks( const ExportTTLIndexEntry & entry, - const std::vector & failed, + const String & failed, const StoragePtr & destination, time_t now, TaskStates & task_states, const ContextPtr & context) { ExportRetriedTasks result; - std::unordered_set added; - - for (const auto & transaction_id : failed) + const auto & state = getCachedTaskState(task_states, failed); + + /// A task that did not export all its parts never committed, and a missing one was either + /// never created or checked when it went missing. + if (state.status != TaskStatus::MISSING && state.reached_commit) + result.push_back(ExportRetriedTask{ + .transaction_id = failed, + .block_ranges = ExportTTLUtils::toBlockRanges(entry.claim->ranges), + .failed_time = now, + }); + + for (const auto & retried : state.retry_of) { - const auto & state = getCachedTaskState(task_states, transaction_id); - - /// A task that did not export all its parts never committed, and a missing one was either - /// never created or checked when it went missing. - if (state.status != TaskStatus::MISSING && state.reached_commit && added.insert(transaction_id).second) - result.push_back(ExportRetriedTask{ - .transaction_id = transaction_id, - .block_ranges = ExportTTLUtils::toBlockRanges(entry.claimed.at(transaction_id)), - .failed_time = now, - }); - - for (const auto & retried : state.retry_of) - { - if (added.contains(retried.transaction_id)) - continue; + /// Past the window, and with no commit of it in progress, the task is checked once more, + /// and not retried by the next groups if it did not land. + const bool may_land = now < retried.failed_time + late_commit_window_seconds + || isCommitInProgress(retried.transaction_id) + || destination->isExportTransactionCommitted(retried.transaction_id, context); - /// Past the window, and with no commit of it in progress, the task is checked once more, - /// and not retried by the next groups if it did not land. - const bool may_land = now < retried.failed_time + late_commit_window_seconds - || isCommitInProgress(retried.transaction_id) - || destination->isExportTransactionCommitted(retried.transaction_id, context); - - if (!may_land) - { - LOG_DEBUG(log, "Export task {} of partition {} did not commit, it is no longer checked", retried.transaction_id, entry.partition_id); - continue; - } - - added.insert(retried.transaction_id); - result.push_back(retried); + if (!may_land) + { + LOG_DEBUG(log, "Export task {} of partition {} did not commit, it is no longer checked", retried.transaction_id, entry.partition_id); + continue; } + + result.push_back(retried); } return result; @@ -496,11 +480,11 @@ ExportTTLScheduler::PartitionView ExportTTLScheduler::observePartition( } bool retry_pending = false; - for (const auto & [transaction_id, _] : entry.claimed) + if (entry.claim) { - const auto status = getCachedTaskState(task_states, transaction_id).status; + const auto status = getCachedTaskState(task_states, entry.claim->transaction_id).status; if (status == TaskStatus::PENDING) - info.current_transaction_id = transaction_id; + info.current_transaction_id = entry.claim->transaction_id; else if (status != TaskStatus::COMPLETED) retry_pending = true; } @@ -527,33 +511,21 @@ std::optional ExportTTLScheduler::actOnPartition( const auto & partition_id = entry.partition_id; const auto settings = storage.getSettings(); - auto resolved = resolveClaims(entry, destination, task_states, context); + auto resolved = resolveClaim(entry, destination, task_states, context); info.current_transaction_id = resolved.in_flight; + /// The claimed parts are the parts of the failed task that still exist. std::vector retry_parts; - std::unordered_set failed_with_parts; - for (const auto & part : view.claimed_parts) - { - for (const auto & transaction_id : resolved.failed) - { - if (ExportFenceUtils::intersectsAny(part->info, entry.claimed.at(transaction_id))) - { - retry_parts.push_back(part); - failed_with_parts.insert(transaction_id); - break; - } - } - } + if (!resolved.failed.empty()) + retry_parts = view.claimed_parts; /// The parts of a failed task that no longer exist, e.g. were dropped, have nothing left to export. - for (const auto & transaction_id : resolved.failed) + if (!resolved.failed.empty() && retry_parts.empty()) { - if (failed_with_parts.contains(transaction_id)) - continue; - LOG_INFO(log, "Export task {} of partition {} failed and none of its parts exists anymore, releasing its claim", - transaction_id, partition_id); - entry.releaseClaim(transaction_id); + resolved.failed, partition_id); + entry.releaseClaim(); + resolved.failed.clear(); resolved.changed = true; } @@ -573,18 +545,15 @@ std::optional ExportTTLScheduler::actOnPartition( group.destination = destination; group.partition_id = partition_id; - /// Every claimed part of a failed task is retried: a part left out would lose its claim and + /// Every claimed part of the failed task is retried: a part left out would lose its claim and /// could be exported again later although the failed task may still land. group.parts = retry_parts; size_t group_bytes = 0; for (const auto & part : group.parts) group_bytes += part->getBytesOnDisk(); - std::vector retried; - for (const auto & transaction_id : resolved.failed) - if (failed_with_parts.contains(transaction_id)) - retried.push_back(transaction_id); - group.retry_of = collectRetriedTasks(entry, retried, destination, now, task_states, context); + if (!resolved.failed.empty()) + group.retry_of = collectRetriedTasks(entry, resolved.failed, destination, now, task_states, context); for (const auto & part : view.shippable) { @@ -598,15 +567,15 @@ std::optional ExportTTLScheduler::actOnPartition( if (!group.parts.empty()) { + /// The group takes over the claim of the failed task it retries, if any. group.entry = versioned; - for (const auto & transaction_id : failed_with_parts) - group.entry.entry.releaseClaim(transaction_id); + group.entry.entry.releaseClaim(); std::vector infos; infos.reserve(group.parts.size()); for (const auto & part : group.parts) infos.push_back(part->info); - group.entry.entry.claim(group.transaction_id, infos); + group.entry.entry.startClaim(group.transaction_id, infos); if (startGroup(group, context)) { @@ -614,7 +583,7 @@ std::optional ExportTTLScheduler::actOnPartition( task_states.insert_or_assign(group.transaction_id, TaskState{.status = TaskStatus::PENDING, .reached_commit = false, .retry_of = group.retry_of}); LOG_INFO(log, "Started export task {} of {} part(s) of partition {} to {}{}", group.transaction_id, group.parts.size(), partition_id, destination->getStorageID().getNameForLogs(), - retried.empty() ? "" : fmt::format(", retrying {}", fmt::join(retried, ", "))); + resolved.failed.empty() ? "" : fmt::format(", retrying {}", resolved.failed)); return std::move(group.entry.entry); } @@ -639,29 +608,22 @@ void ExportTTLScheduler::cleanupDestination(const String & destination_key, cons TaskStates task_states; bool claims_left = false; - for (auto [partition_id, versioned] : index) + for (auto [_, versioned] : index) { - bool changed = false; - - std::vector transaction_ids; - for (const auto & [transaction_id, _] : versioned.entry.claimed) - transaction_ids.push_back(transaction_id); + if (!versioned.entry.claim) + continue; - for (const auto & transaction_id : transaction_ids) + const auto transaction_id = versioned.entry.claim->transaction_id; + if (getCachedTaskState(task_states, transaction_id).status == TaskStatus::PENDING) { - if (getCachedTaskState(task_states, transaction_id).status == TaskStatus::PENDING) - { - LOG_INFO(log, "Killing export task {}: the EXPORT TTL no longer exports to destination {}", transaction_id, destination_key); - killTask(transaction_id); - claims_left = true; - continue; - } - - versioned.entry.releaseClaim(transaction_id); - changed = true; + LOG_INFO(log, "Killing export task {}: the EXPORT TTL no longer exports to destination {}", transaction_id, destination_key); + killTask(transaction_id); + claims_left = true; + continue; } - if (changed && !updateIndexEntry(destination_key, versioned)) + versioned.entry.releaseClaim(); + if (!updateIndexEntry(destination_key, versioned)) claims_left = true; } diff --git a/src/Storages/MergeTree/ExportTTLScheduler.h b/src/Storages/MergeTree/ExportTTLScheduler.h index 97befe414198..e3f949c38644 100644 --- a/src/Storages/MergeTree/ExportTTLScheduler.h +++ b/src/Storages/MergeTree/ExportTTLScheduler.h @@ -186,21 +186,22 @@ class ExportTTLScheduler ContextPtr makeContext() const; - /// Resolves the claims of the index entry that do not belong to a task in flight. Returns the - /// failed tasks whose parts are still claimed, and the task in flight. - struct ResolvedClaims + /// Resolves the claim of the index entry unless its task is in flight: a task that completed, or + /// that failed but committed to the destination, has its claim moved to the exported ranges. + struct ResolvedClaim { - std::vector failed; + /// The task in flight, or the failed task whose parts are still claimed; empty if there is none. String in_flight; + String failed; bool changed = false; }; - ResolvedClaims resolveClaims(ExportTTLIndexEntry & entry, const StoragePtr & destination, TaskStates & task_states, const ContextPtr & context); + ResolvedClaim resolveClaim(ExportTTLIndexEntry & entry, const StoragePtr & destination, TaskStates & task_states, const ContextPtr & context); - /// The failed tasks among `failed` and the ones they retried whose commit may still land, for - /// the `retry_of` of the group that retries the parts of `failed`. + /// The task `failed`, which holds the claim of `entry`, and the ones it retried whose commit may + /// still land, for the `retry_of` of the group that retries its parts. ExportRetriedTasks collectRetriedTasks( const ExportTTLIndexEntry & entry, - const std::vector & failed, + const String & failed, const StoragePtr & destination, time_t now, TaskStates & task_states, diff --git a/src/Storages/MergeTree/ExportTaskUtils.cpp b/src/Storages/MergeTree/ExportTaskUtils.cpp index d7d8bd1ab77a..77e41c6fbb5a 100644 --- a/src/Storages/MergeTree/ExportTaskUtils.cpp +++ b/src/Storages/MergeTree/ExportTaskUtils.cpp @@ -659,8 +659,6 @@ namespace export_ttl_index.ensureDestination(zk, destination_key, fmt::format("{}.{}", manifest.destination_database, manifest.destination_table)); versioned.entry.commitClaim(manifest.transaction_id, ExportTTLUtils::rangesOfParts(manifest.parts, source_storage.format_version)); - for (const auto & retried : manifest.retry_of) - versioned.entry.releaseClaim(retried.transaction_id); export_ttl_index.appendUpdateEntryOps(ops, destination_key, versioned.entry, versioned.version); } diff --git a/src/Storages/MergeTree/tests/gtest_export_ttl_index.cpp b/src/Storages/MergeTree/tests/gtest_export_ttl_index.cpp index 186cdb94e34b..21368c09c70d 100644 --- a/src/Storages/MergeTree/tests/gtest_export_ttl_index.cpp +++ b/src/Storages/MergeTree/tests/gtest_export_ttl_index.cpp @@ -3,6 +3,8 @@ #include #include #include +#include +#include using namespace DB; @@ -31,13 +33,16 @@ TEST(ExportTTLIndex, ClaimThenCommit) ExportTTLIndexEntry entry; entry.partition_id = "p"; - entry.claim("t1", {part(1, 1), part(2, 2), part(4, 4)}); - EXPECT_EQ(blocks(entry.claimed.at("t1")), (Blocks{{1, 2}, {4, 4}})); + entry.startClaim("t1", {part(1, 1), part(2, 2), part(4, 4)}); + ASSERT_TRUE(entry.claim); + EXPECT_EQ(entry.claim->transaction_id, "t1"); + EXPECT_EQ(blocks(entry.claim->ranges), (Blocks{{1, 2}, {4, 4}})); + EXPECT_EQ(blocks(entry.toFenceEntry("db.t").claimed), (Blocks{{1, 2}, {4, 4}})); EXPECT_EQ(entry.classify(part(1, 2, 1)), PartExportState::CLAIMED); EXPECT_EQ(entry.classify(part(3, 3)), PartExportState::NONE); entry.commitClaim("t1", {}); - EXPECT_TRUE(entry.claimed.empty()); + EXPECT_FALSE(entry.claim); EXPECT_EQ(blocks(entry.exported), (Blocks{{1, 2}, {4, 4}})); /// A mutated exported part keeps its block range. EXPECT_EQ(entry.classify(part(4, 4, 0, 7)), PartExportState::EXPORTED); @@ -49,29 +54,41 @@ TEST(ExportTTLIndex, RetryReclaimsExactRanges) { ExportTTLIndexEntry entry; entry.partition_id = "p"; - entry.claim("t1", {part(1, 1), part(2, 2)}); + entry.startClaim("t1", {part(1, 1), part(2, 2)}); /// Part 2 was dropped before the retry, part 5 became eligible meanwhile. - entry.releaseClaim("t1"); - entry.claim("t2", {part(1, 1), part(5, 5)}); + entry.releaseClaim(); + entry.startClaim("t2", {part(1, 1), part(5, 5)}); - EXPECT_FALSE(entry.claimed.contains("t1")); - EXPECT_EQ(blocks(entry.claimed.at("t2")), (Blocks{{1, 1}, {5, 5}})); + ASSERT_TRUE(entry.claim); + EXPECT_EQ(entry.claim->transaction_id, "t2"); + EXPECT_EQ(blocks(entry.claim->ranges), (Blocks{{1, 1}, {5, 5}})); EXPECT_EQ(entry.classify(part(2, 2)), PartExportState::NONE); - EXPECT_EQ(blocks(entry.allClaimed()), (Blocks{{1, 1}, {5, 5}})); + EXPECT_EQ(entry.maxBlock(), 5); } -TEST(ExportTTLIndex, MoveClaims) +/// A second claim is a `LOGICAL_ERROR`, which aborts debug and sanitizer builds instead of throwing. +#ifndef DEBUG_OR_SANITIZER_BUILD +TEST(ExportTTLIndex, OneClaimAtATime) { ExportTTLIndexEntry entry; entry.partition_id = "p"; - entry.claim("t1", {part(1, 1)}); - entry.claim("t2", {part(2, 2)}); + entry.startClaim("t1", {part(1, 1)}); - entry.moveClaims({"t1", "t2", "missing"}, "t3"); - EXPECT_EQ(entry.claimed.size(), 1); - EXPECT_EQ(blocks(entry.claimed.at("t3")), (Blocks{{1, 2}})); + EXPECT_THROW(entry.startClaim("t2", {part(2, 2)}), Exception); + EXPECT_EQ(entry.claim->transaction_id, "t1"); + EXPECT_EQ(blocks(entry.claim->ranges), (Blocks{{1, 1}})); } +#else +TEST(ExportTTLIndexDeathTest, OneClaimAtATime) +{ + ExportTTLIndexEntry entry; + entry.partition_id = "p"; + entry.startClaim("t1", {part(1, 1)}); + + EXPECT_DEATH(entry.startClaim("t2", {part(2, 2)}), "cannot claim parts of partition p"); +} +#endif TEST(ExportTTLIndex, CommitAddsPartsWhoseClaimWasLost) { @@ -79,6 +96,13 @@ TEST(ExportTTLIndex, CommitAddsPartsWhoseClaimWasLost) entry.partition_id = "p"; entry.commitClaim("t1", {part(3, 5)}); EXPECT_EQ(blocks(entry.exported), (Blocks{{3, 5}})); + + /// The claim of another task is kept. + entry.startClaim("t2", {part(7, 7)}); + entry.commitClaim("t1", {part(6, 6)}); + EXPECT_EQ(blocks(entry.exported), (Blocks{{3, 6}})); + ASSERT_TRUE(entry.claim); + EXPECT_EQ(entry.claim->transaction_id, "t2"); } TEST(ExportTTLIndex, JsonRoundTrip) @@ -86,13 +110,18 @@ TEST(ExportTTLIndex, JsonRoundTrip) ExportTTLIndexEntry entry; entry.partition_id = "p"; entry.exported = {part(0, 10), part(20, 30)}; - entry.claim("t1", {part(31, 31)}); - const auto parsed = ExportTTLIndexEntry::fromJSONString("p", entry.toJSONString()); + auto parsed = ExportTTLIndexEntry::fromJSONString("p", entry.toJSONString()); + EXPECT_EQ(blocks(parsed.exported), (Blocks{{0, 10}, {20, 30}})); + EXPECT_FALSE(parsed.claim); + + entry.startClaim("t1", {part(31, 31)}); + parsed = ExportTTLIndexEntry::fromJSONString("p", entry.toJSONString()); EXPECT_EQ(parsed.partition_id, "p"); EXPECT_EQ(blocks(parsed.exported), (Blocks{{0, 10}, {20, 30}})); - ASSERT_TRUE(parsed.claimed.contains("t1")); - EXPECT_EQ(blocks(parsed.claimed.at("t1")), (Blocks{{31, 31}})); + ASSERT_TRUE(parsed.claim); + EXPECT_EQ(parsed.claim->transaction_id, "t1"); + EXPECT_EQ(blocks(parsed.claim->ranges), (Blocks{{31, 31}})); EXPECT_TRUE(ExportTTLIndexEntry::fromJSONString("p", "").empty()); EXPECT_ANY_THROW(ExportTTLIndexEntry::fromJSONString("p", R"({"exported":[[1]]})")); diff --git a/src/Storages/StorageReplicatedMergeTree.cpp b/src/Storages/StorageReplicatedMergeTree.cpp index 7ba1ce895616..85fdecc5336c 100644 --- a/src/Storages/StorageReplicatedMergeTree.cpp +++ b/src/Storages/StorageReplicatedMergeTree.cpp @@ -8879,6 +8879,12 @@ void StorageReplicatedMergeTree::checkAllReplicasSupportExportTTL(const zkutil:: fmt::join(unsupported, ", ")); } +void StorageReplicatedMergeTree::checkReplicasSupportExportTTL() const +{ + auto component_guard = Coordination::setCurrentComponent("StorageReplicatedMergeTree::checkReplicasSupportExportTTL"); + checkAllReplicasSupportExportTTL(getZooKeeper()); +} + void StorageReplicatedMergeTree::advertiseExportFeatures(const zkutil::ZooKeeperPtr & zookeeper) const { const String path = fs::path(replica_path) / "export_features"; @@ -8939,10 +8945,10 @@ void StorageReplicatedMergeTree::forgetPartition(const ASTPtr & partition, Conte if (versioned.version < 0) continue; - if (!versioned.entry.claimed.empty()) + if (versioned.entry.claim) throw Exception(ErrorCodes::CANNOT_FORGET_PARTITION, "Partition {} is being exported by the EXPORT TTL (task {}), retry after it finishes", - partition_id, versioned.entry.claimed.begin()->first); + partition_id, versioned.entry.claim->transaction_id); ops.emplace_back(zkutil::makeRemoveRequest(export_ttl_index->getIndexEntryPath(destination_key, partition_id), versioned.version)); } diff --git a/src/Storages/StorageReplicatedMergeTree.h b/src/Storages/StorageReplicatedMergeTree.h index fc908340e407..2e26423ae31f 100644 --- a/src/Storages/StorageReplicatedMergeTree.h +++ b/src/Storages/StorageReplicatedMergeTree.h @@ -999,7 +999,7 @@ class StorageReplicatedMergeTree final : public MergeTreeData /// Throws unless every replica enforces the export states of parts when assigning merges. void checkAllReplicasSupportExportTTL(const zkutil::ZooKeeperPtr & zookeeper) const; - void checkReplicasSupportExportTTL() const override { checkAllReplicasSupportExportTTL(getZooKeeper()); } + void checkReplicasSupportExportTTL() const override; /// Creates (or removes, if partition export is disabled) `/export_features`. void advertiseExportFeatures(const zkutil::ZooKeeperPtr & zookeeper) const; From 65981365a6ab3c33743e8266ce7529539d181e4c Mon Sep 17 00:00:00 2001 From: Arthur Passos Date: Wed, 30 Sep 2026 10:08:24 -0300 Subject: [PATCH 11/15] docs --- docs/en/antalya/ttl_export.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/docs/en/antalya/ttl_export.md b/docs/en/antalya/ttl_export.md index e79e097426d4..1ff9ef6bc643 100644 --- a/docs/en/antalya/ttl_export.md +++ b/docs/en/antalya/ttl_export.md @@ -36,7 +36,7 @@ Here rows are exported to `events_archive` 30 days after `event_time`, and delet ## Requirements {#requirements} -- The server setting `allow_experimental_export_merge_tree_partition` must be enabled on every replica, and the query setting `allow_experimental_export_ttl` must be enabled for the `CREATE` or `ALTER` that adds the expression. A table with the expression is not loaded, e.g. at a restart or by `ATTACH TABLE`, while the server setting is disabled: without it, merges would not keep the exported parts apart from the others. +- The server setting `allow_experimental_export_merge_tree_partition` must be enabled on every replica, and the query setting `allow_experimental_export_ttl` must be enabled for the `CREATE` or `ALTER` that adds the expression. A table with the expression is not loaded, e.g. at a restart or by `ATTACH TABLE`, while the server setting is disabled: without it, merges would not keep the exported parts apart from the others. For the same reason, an `ALTER` that adds or changes the expression of a `Replicated*MergeTree` table is refused while a replica does not support it, e.g. because it runs an older version or has the server setting disabled. - The destination must exist when the expression is added. It must be an Apache Iceberg or object storage table that `EXPORT PARTITION` can export to, and its schema must be castable from the source schema, see [`EXPORT PARTITION` requirements](/docs/en/antalya/partition_export.md#requirements). Unlike `EXPORT PARTITION`, the TTL allows lossy casts, as if `export_merge_tree_part_allow_lossy_cast` were enabled: e.g. a `UInt32` column is exported to the `int` of an Iceberg table, which is read back as `Int32`, and values that do not fit change. An unqualified table name refers to the database of the source table. - A table can have at most one `EXPORT` TTL expression, without `WHERE` or `GROUP BY`. The expression must be deterministic and return a `Date` or `DateTime`. - The rows of a group must land in a single partition of the destination, see [Partition key of the destination](#destination-partition-key). From b39db8f49665ae08ff974be8d52fa1558f410b3a Mon Sep 17 00:00:00 2001 From: Arthur Passos Date: Wed, 30 Sep 2026 11:02:05 -0300 Subject: [PATCH 12/15] first docs pass --- docs/en/antalya/partition_export.md | 4 +--- docs/en/antalya/ttl_export.md | 12 ++++-------- 2 files changed, 5 insertions(+), 11 deletions(-) diff --git a/docs/en/antalya/partition_export.md b/docs/en/antalya/partition_export.md index d09a28ac81e0..08727a4e17aa 100644 --- a/docs/en/antalya/partition_export.md +++ b/docs/en/antalya/partition_export.md @@ -9,14 +9,12 @@ The `ALTER TABLE EXPORT PARTITION` command exports entire partitions from `Merge The set of parts that are exported is based on the list of parts the replica that received the export command sees. On `Replicated*MergeTree`, the other replicas will assist in the export process if they have those parts locally. Otherwise they will ignore it. -Every `EXPORT PARTITION` creates a new export task, identified by its transaction id, that exports the parts the partition has at that moment. Exporting a partition that was exported before is allowed and exports its parts again; whether that duplicates rows depends on the destination and on `export_merge_tree_part_file_already_exists_policy` (with the default `skip` and file name pattern, files already written to a plain object storage destination are reused). Preventing duplicates is up to the user. The setting `export_merge_tree_partition_force_export` is obsolete and has no effect. +Exporting a partition that was exported before is allowed and exports its parts again; Preventing duplicates is up to the user. Export tasks of both engines, including those of a [`TTL ... EXPORT TO TABLE`](/docs/en/antalya/ttl_export.md) expression, can be observed through `system.distributed_exports`, one row per task. Manual exports are independent of the `EXPORT` TTL: they neither read nor update what the TTL has exported. The export task can be killed by issuing the kill command: `KILL EXPORT `. -Tasks are stored under `/exports/` for a `Replicated*MergeTree` table, and in `exports/.json` in the data directory of a plain `MergeTree` table. Tasks stored by earlier versions of this experimental feature, keyed by partition and destination, are neither read nor migrated. - The task is persistent - it should be resumed after crashes, failures and etc. A part with no surviving rows writes no file. This happens when every row of the part was removed by a lightweight delete: the part is still exported, and counts as done, but it contributes nothing to the destination. If that is true of every part of the partition, the export produces no files at all and there is nothing to commit, so the task reaches `COMPLETED` without touching the destination. Such a part therefore has no entry in the `destination_file_paths` column of `system.distributed_exports`. diff --git a/docs/en/antalya/ttl_export.md b/docs/en/antalya/ttl_export.md index 1ff9ef6bc643..3592824d286b 100644 --- a/docs/en/antalya/ttl_export.md +++ b/docs/en/antalya/ttl_export.md @@ -13,7 +13,7 @@ doc_type: 'reference' A `TTL EXPORT TO TABLE [database.]table` expression exports the rows of a `MergeTree`-family table to an Apache Iceberg or plain object storage table in the background, once their TTL is due. It is meant for tiering: recent data stays in `MergeTree`, older data is kept in the destination. -The TTL never exports a row twice, also across failures, restarts and retries: what was exported is recorded per partition, and parts are exported in groups, each committed to the destination in one transaction. The export uses the same machinery as [`EXPORT PARTITION`](/docs/en/antalya/partition_export.md): each group is an export task shown in `system.distributed_exports` with `source = 'ttl'`. +The TTL never exports a part twice, also across failures, restarts and retries: what was exported is recorded, and parts are exported in groups, each committed to the destination in one transaction. The export uses the same machinery as [`EXPORT PARTITION`](/docs/en/antalya/partition_export.md): each group is an export task shown in `system.distributed_exports` with `source = 'ttl'`. Both plain `MergeTree` and `Replicated*MergeTree` tables are supported. @@ -37,17 +37,13 @@ Here rows are exported to `events_archive` 30 days after `event_time`, and delet ## Requirements {#requirements} - The server setting `allow_experimental_export_merge_tree_partition` must be enabled on every replica, and the query setting `allow_experimental_export_ttl` must be enabled for the `CREATE` or `ALTER` that adds the expression. A table with the expression is not loaded, e.g. at a restart or by `ATTACH TABLE`, while the server setting is disabled: without it, merges would not keep the exported parts apart from the others. For the same reason, an `ALTER` that adds or changes the expression of a `Replicated*MergeTree` table is refused while a replica does not support it, e.g. because it runs an older version or has the server setting disabled. -- The destination must exist when the expression is added. It must be an Apache Iceberg or object storage table that `EXPORT PARTITION` can export to, and its schema must be castable from the source schema, see [`EXPORT PARTITION` requirements](/docs/en/antalya/partition_export.md#requirements). Unlike `EXPORT PARTITION`, the TTL allows lossy casts, as if `export_merge_tree_part_allow_lossy_cast` were enabled: e.g. a `UInt32` column is exported to the `int` of an Iceberg table, which is read back as `Int32`, and values that do not fit change. An unqualified table name refers to the database of the source table. +- The destination must exist when the expression is added. It must be an Apache Iceberg or object storage table that `EXPORT PARTITION` can export to, and its schema must be castable from the source schema, see [`EXPORT PARTITION` requirements](/docs/en/antalya/partition_export.md#requirements). - A table can have at most one `EXPORT` TTL expression, without `WHERE` or `GROUP BY`. The expression must be deterministic and return a `Date` or `DateTime`. - The rows of a group must land in a single partition of the destination, see [Partition key of the destination](#destination-partition-key). ## Partition key of the destination {#destination-partition-key} -Each partition expression of the destination (or each Iceberg partition transform) must either be a function of the source partition key, e.g. the same expression, or be a monotonic function of a single non-`Nullable` column of the source partition key. In the second case, whether the rows of a group land in a single destination partition depends on the rows, so it is checked when the group is exported, from the minimum and maximum values of the column in its parts. - -A destination that can never be compatible is refused by the `CREATE` or `ALTER` that adds the expression, e.g. a destination partitioned by a column that is not in the source partition key, by a function that is not monotonic like `toDayOfWeek(event_time)` or the Iceberg `bucket` transform, or a partitioned destination of an unpartitioned source. - -A destination that is compatible only for some groups is accepted. With a source `PARTITION BY toYYYYMM(event_time)`: +A destination that is compatible only for some groups is accepted, and might fail at runtime. With a source `PARTITION BY toYYYYMM(event_time)`: - `PARTITION BY toYear(event_time)` is compatible for every group, since a month is within a year; - `PARTITION BY toDate(event_time)` is compatible only for a group whose rows are all on the same day. Any other group is not exported: no export task is created, the error is shown in the `last_error` column of `system.ttl_exports`, and the group is tried again on every check. @@ -72,7 +68,7 @@ One replica of a `Replicated*MergeTree` table schedules the groups; the others t A group that fails, is killed or times out is retried on the next check as a new task, with a new transaction id. The retry contains every part of the failed group, plus any part that became eligible meanwhile. Throttling comes from the export tasks themselves: per-part retry back-off, `export_merge_tree_task_timeout_seconds` and the commit attempts. -A task commits only once it exported all its parts, so a task that failed before that cannot have committed, and nothing checks it. A task that failed after that is checked at the destination before it is retried, in case it committed but was not marked as completed. It may also land after that check, e.g. a request that the destination applies late, so the retry lists it in its `retry_of` column, with the blocks of its parts. The commit of the retry checks each task in `retry_of` again, and leaves out the files of the parts a landed task exported. A task stays in the `retry_of` of later retries for 10 minutes after it failed, or longer while a replica is still in its commit; it is then checked a last time, and kept only if it landed. +A task that failed after uploading all parts and before marking it as committed might have committed to the destination. It is retried nevertheless. The retry task will therefore check if the previous task transaction has been completed by asking the destination storage. If it has been completed, it'll only export the delta parts. Otherwise, it'll export it all again. `KILL EXPORT` of a task of the TTL makes it retry. To stop exporting, use `SYSTEM STOP MOVES`, which pauses the TTL export of the table, or remove the expression. On a `Replicated*MergeTree` table, `SYSTEM STOP MOVES` pauses it only on the replica that schedules the groups, shown in the `scheduler_replica` column of `system.ttl_exports`, so run it on every replica, e.g. with `ON CLUSTER`. From daed786a6104d438c4ab15dba21ffafb9d4ca30a Mon Sep 17 00:00:00 2001 From: Arthur Passos Date: Wed, 30 Sep 2026 11:02:49 -0300 Subject: [PATCH 13/15] add vibe coded hourly export --- .../test_export_ttl/test_rolling_exports.py | 188 ++++++++++++++++++ 1 file changed, 188 insertions(+) create mode 100644 tests/integration/test_export_ttl/test_rolling_exports.py diff --git a/tests/integration/test_export_ttl/test_rolling_exports.py b/tests/integration/test_export_ttl/test_rolling_exports.py new file mode 100644 index 000000000000..561f07c6791a --- /dev/null +++ b/tests/integration/test_export_ttl/test_rolling_exports.py @@ -0,0 +1,188 @@ +import itertools +import logging +import time +from collections import Counter + +from helpers.export_partition_helpers import unique_suffix + +from .common import ( + assert_one_snapshot_per_task, + completed_ttl_tasks, + create_iceberg, + create_source, + merged_parts, + query_json, + scheduler_holder, + wait_for_partitions_exported, + wait_until, +) + +CLUSTER_INSTANCES = ["replica1", "replica2"] + +# A TTL of seconds that does not follow the partition key, on partitions that keep receiving rows: +# rows are exported in successive groups as they become due while more keep coming, never before +# they are due and never twice. When a row is due and when it was first read from the destination +# are both taken from the clock of the servers. + +COLUMNS = "id Int64, customer_id Int64, t DateTime" +CUSTOMERS = [1, 2, 3] +PARTITION_IDS = [str(customer) for customer in CUSTOMERS] +TTL_SECONDS = 10 +INSERT_SECONDS = 60 +# How long after it is due a row may take to be read from the destination when parts are not merged: +# the maximum batch delay, a group in flight and the next group, with room for sanitizer builds. The +# rows due in the first `INSERT_SECONDS - MAX_LAG_SECONDS` seconds are thus exported while rows keep +# coming. +MAX_LAG_SECONDS = 40 +# Parts keep becoming due, so a group is shipped by the window or, while it does not close, by the maximum delay. +BATCH = {"ttl_export_batch_window_seconds": 3, "ttl_export_batch_max_delay_seconds": 6} +NO_MERGES = {"max_bytes_to_merge_at_max_space_in_pool": 1} + + +def make_tables(nodes, engine, settings=None): + suffix = unique_suffix() + mt_table, iceberg_table = f"roll_mt_{suffix}", f"roll_iceberg_{suffix}" + create_iceberg(nodes, iceberg_table, columns=COLUMNS, partition_by="customer_id") + for node in nodes: + create_source( + node, mt_table, COLUMNS, "customer_id", f"t + INTERVAL {TTL_SECONDS} SECOND EXPORT TO TABLE {iceberg_table}", + engine=engine, replica_name=node.name, settings={**BATCH, **(settings or {})}, + ) + return mt_table, iceberg_table + + +class RollingExport: + """The rows inserted, one part per partition at a time, and when each was first read from the destination.""" + + def __init__(self, reader, iceberg_table): + self.reader = reader + self.iceberg_table = iceberg_table + self.inserted = 0 + self.due = {} + self.first_read = {} + + def insert(self, node, mt_table): + rows = ", ".join(f"({self.inserted + i}, {customer}, now())" for i, customer in enumerate(CUSTOMERS)) + node.query(f"INSERT INTO {mt_table} VALUES {rows}") + self.inserted += len(CUSTOMERS) + + def read(self): + """Reads the destination, which must hold no row twice and no row that is not due. The time of a + row is taken by `nowInBlock` after the row was read, so a row read before it was due was exported + before it was due. `assert_rows_match_the_source` checks the `t` that the due time is taken from.""" + rows = query_json( + self.reader, + f"SELECT id, toUnixTimestamp(t) + {TTL_SECONDS}, toUnixTimestamp(nowInBlock()) FROM {self.iceberg_table}", + ) + ids = [row_id for row_id, _, _ in rows] + twice = sorted(row_id for row_id, count in Counter(ids).items() if count > 1) + assert not twice, f"Rows exported twice: {twice}" + early = {row_id: due - read_at for row_id, due, read_at in rows if read_at < due} + assert not early, f"Rows exported before they were due, by seconds: {early}" + for row_id, due, read_at in rows: + self.due[row_id] = due + self.first_read.setdefault(row_id, read_at) + return ids + + def lags(self): + return {row_id: self.first_read[row_id] - due for row_id, due in self.due.items()} + + +def roll(insert_node, mt_table, reader, iceberg_table, merge=None): + """Inserts a part into every partition at most every second for `INSERT_SECONDS`, calling *merge* + after each insert, then waits until every row is exported, reading the destination all along.""" + export = RollingExport(reader, iceberg_table) + start = time.time() + while time.time() - start < INSERT_SECONDS: + iteration_start = time.time() + export.insert(insert_node, mt_table) + if merge: + merge() + export.read() + time.sleep(max(0.0, 1 - (time.time() - iteration_start))) + + wait_until(lambda: sorted(export.read()) == list(range(export.inserted)), 120, "Not every row was exported") + lags = export.lags() + logging.info("%d rows were exported at most %d s after they were due", len(lags), max(lags.values())) + return export + + +def assert_rows_match_the_source(node, mt_table, iceberg_table): + """The destination holds every row of the source once, with the same values.""" + query = "SELECT id, customer_id, toUnixTimestamp(t) FROM {} ORDER BY id" + assert node.query(query.format(iceberg_table)) == node.query(query.format(mt_table)) + + +def assert_lag_is_bounded(export): + late = {row_id: lag for row_id, lag in export.lags().items() if lag > MAX_LAG_SECONDS} + assert not late, f"Rows exported more than {MAX_LAG_SECONDS} s after they were due, by seconds: {late}" + + +def assert_several_groups_per_partition(node, mt_table): + groups = Counter(task["partition_id"] for task in completed_ttl_tasks(node, mt_table)) + assert all(groups[partition_id] > 1 for partition_id in PARTITION_IDS), f"Groups by partition: {groups}" + + +def wait_until_settled(node, mt_table, iceberg_table): + wait_for_partitions_exported(node, mt_table, PARTITION_IDS) + # A replica refreshes the tasks it shows from Keeper in the background, so the last ones may show + # up after the partitions settled. + deadline = time.time() + 30 + while True: + try: + assert_one_snapshot_per_task(node, mt_table, iceberg_table) + return + except AssertionError: + if time.time() > deadline: + raise + time.sleep(0.5) + + +def test_rows_are_exported_as_they_become_due(cluster, source_engine): + """Without merges, a part holds the rows of one insert and is due with them, so every row is + exported within a bounded time after it is due, in several groups per partition.""" + node = cluster.instances["replica1"] + mt_table, iceberg_table = make_tables([node], source_engine, NO_MERGES) + + export = roll(node, mt_table, node, iceberg_table) + + assert_lag_is_bounded(export) + assert_several_groups_per_partition(node, mt_table) + wait_until_settled(node, mt_table, iceberg_table) + assert_rows_match_the_source(node, mt_table, iceberg_table) + + +def test_merges_never_export_rows_early(cluster, source_engine): + """A part that no group claimed yet may merge with newer parts of its partition, and its rows then + wait for theirs, so how late the rows are exported depends on the merges. Still, no row is + exported before it is due or twice, and every row is exported once the inserts stop.""" + node = cluster.instances["replica1"] + mt_table, iceberg_table = make_tables([node], source_engine) + partitions = itertools.cycle(PARTITION_IDS) + + export = roll( + node, mt_table, node, iceberg_table, + merge=lambda: node.query(f"OPTIMIZE TABLE {mt_table} PARTITION ID '{next(partitions)}'"), + ) + + assert merged_parts(node, mt_table), "No part was merged" + wait_until_settled(node, mt_table, iceberg_table) + assert_rows_match_the_source(node, mt_table, iceberg_table) + + +def test_rows_inserted_on_another_replica_are_exported_as_they_become_due(cluster): + """The replica that schedules the groups only ships the parts it has, so the rows inserted on + another replica are exported once it fetched them, within the same bound.""" + replicas = [cluster.instances["replica1"], cluster.instances["replica2"]] + mt_table, iceberg_table = make_tables(replicas, "ReplicatedMergeTree", NO_MERGES) + scheduler = wait_until(lambda: scheduler_holder(replicas[0], mt_table), 60, "No replica schedules the TTL export") + inserter = next(replica for replica in replicas if replica.name != scheduler) + + export = roll(inserter, mt_table, inserter, iceberg_table) + + assert_lag_is_bounded(export) + assert_several_groups_per_partition(inserter, mt_table) + wait_until_settled(inserter, mt_table, iceberg_table) + for replica in replicas: + replica.query(f"SYSTEM SYNC REPLICA {mt_table}") + assert_rows_match_the_source(replica, mt_table, iceberg_table) From c8f1ab1f459ee4f4e2d169cba6c25097f3a0e2b5 Mon Sep 17 00:00:00 2001 From: Arthur Passos Date: Wed, 30 Sep 2026 14:52:48 -0300 Subject: [PATCH 14/15] vibe code ai review --- .../ExportReplicatedMergeTreeTaskManifest.h | 6 -- src/Storages/MergeTree/ExportTTLIndex.cpp | 9 ++- src/Storages/MergeTree/ExportTTLIndex.h | 9 ++- src/Storages/MergeTree/ExportTTLScheduler.cpp | 59 ++++++++++++------- src/Storages/MergeTree/ExportTTLScheduler.h | 17 +++--- src/Storages/MergeTree/ExportTaskUtils.cpp | 3 +- .../MergeTree/MergeTreeExportTTLScheduler.cpp | 29 +++++---- .../MergeTree/MergeTreeExportTTLScheduler.h | 1 - src/Storages/MergeTree/MergeTreeExportTask.h | 6 -- .../ReplicatedExportTTLScheduler.cpp | 39 ++++++------ .../MergeTree/ReplicatedExportTTLScheduler.h | 1 - .../tests/gtest_export_task_ordering.cpp | 2 - .../tests/gtest_export_ttl_index.cpp | 3 +- src/Storages/StorageMergeTree.cpp | 2 - src/Storages/StorageReplicatedMergeTree.cpp | 2 - .../test_export_ttl/test_failures.py | 14 ++--- .../test_export_ttl/test_merge_fence.py | 35 +++++++++++ 17 files changed, 139 insertions(+), 98 deletions(-) diff --git a/src/Storages/ExportReplicatedMergeTreeTaskManifest.h b/src/Storages/ExportReplicatedMergeTreeTaskManifest.h index 6d3136fd97ad..286783385480 100644 --- a/src/Storages/ExportReplicatedMergeTreeTaskManifest.h +++ b/src/Storages/ExportReplicatedMergeTreeTaskManifest.h @@ -168,8 +168,6 @@ struct ExportReplicatedMergeTreeTaskManifest ExportTaskSource source = ExportTaskSource::query; String destination_database; String destination_table; - /// UUID of the destination table when the task was created, empty if it has none. - String destination_uuid; /// TTL export only: earlier tasks that failed to export some of these parts and whose commit may /// still land. The commit checks whether any of them landed at the destination after all. ExportRetriedTasks retry_of; @@ -212,8 +210,6 @@ struct ExportReplicatedMergeTreeTaskManifest json.set("source", String(magic_enum::enum_name(source))); json.set("destination_database", destination_database); json.set("destination_table", destination_table); - if (!destination_uuid.empty()) - json.set("destination_uuid", destination_uuid); if (!retry_of.empty()) json.set("retry_of", ExportRetriedTaskUtils::toJSON(retry_of)); json.set("source_replica", source_replica); @@ -279,8 +275,6 @@ struct ExportReplicatedMergeTreeTaskManifest } manifest.destination_database = json->getValue("destination_database"); manifest.destination_table = json->getValue("destination_table"); - if (json->has("destination_uuid")) - manifest.destination_uuid = json->getValue("destination_uuid"); if (json->has("retry_of")) manifest.retry_of = ExportRetriedTaskUtils::fromJSON(json->getArray("retry_of")); manifest.source_replica = json->getValue("source_replica"); diff --git a/src/Storages/MergeTree/ExportTTLIndex.cpp b/src/Storages/MergeTree/ExportTTLIndex.cpp index eb7bf98b35c3..27e63a521b9e 100644 --- a/src/Storages/MergeTree/ExportTTLIndex.cpp +++ b/src/Storages/MergeTree/ExportTTLIndex.cpp @@ -104,6 +104,11 @@ void ExportTTLIndexEntry::commitClaim(const String & transaction_id, const std:: exported.insert(exported.end(), claim->ranges.begin(), claim->ranges.end()); claim.reset(); } + addExported(parts); +} + +void ExportTTLIndexEntry::addExported(const std::vector & parts) +{ for (const auto & part : parts) exported.emplace_back(partition_id, part.min_block, part.max_block, 0, 0); exported = ExportFenceUtils::compactRanges(std::move(exported)); @@ -177,9 +182,9 @@ void allowLossyCasts(Context & context) context.setSetting("export_merge_tree_part_allow_lossy_cast", true); } -String destinationKey(const String & database, const String & table, const String & uuid) +String destinationKey(const String & database, const String & table) { - return escapeForFileName(database) + "." + escapeForFileName(table) + "." + (uuid.empty() ? String("none") : escapeForFileName(uuid)); + return escapeForFileName(database) + "." + escapeForFileName(table); } std::vector rangesOfParts(const std::vector & part_names, MergeTreeDataFormatVersion format_version) diff --git a/src/Storages/MergeTree/ExportTTLIndex.h b/src/Storages/MergeTree/ExportTTLIndex.h index 144fb5330b0c..64db2a8272a9 100644 --- a/src/Storages/MergeTree/ExportTTLIndex.h +++ b/src/Storages/MergeTree/ExportTTLIndex.h @@ -62,6 +62,9 @@ struct ExportTTLIndexEntry /// a task may commit parts whose claim was lost. void commitClaim(const String & transaction_id, const std::vector & parts); + /// Adds the ranges of `parts` to the exported ranges. + void addExported(const std::vector & parts); + void releaseClaim() { claim.reset(); } String toJSONString() const; @@ -100,9 +103,9 @@ namespace ExportTTLUtils /// that opted in at `CREATE` or `ALTER` would not reach them, and Iceberg has no unsigned types. void allowLossyCasts(Context & context); - /// Identifies a destination in the export index. The UUID, if not empty, tells apart a table that - /// was dropped and created again under the same name, which holds none of the exported rows. - String destinationKey(const String & database, const String & table, const String & uuid); + /// Identifies a destination in the export index, by name: a table dropped and created again under + /// the same name is the same destination. + String destinationKey(const String & database, const String & table); /// Ranges of the parts named `part_names`, compacted. std::vector rangesOfParts(const std::vector & part_names, MergeTreeDataFormatVersion format_version); diff --git a/src/Storages/MergeTree/ExportTTLScheduler.cpp b/src/Storages/MergeTree/ExportTTLScheduler.cpp index b92dff1bf4a8..aae883df8c03 100644 --- a/src/Storages/MergeTree/ExportTTLScheduler.cpp +++ b/src/Storages/MergeTree/ExportTTLScheduler.cpp @@ -97,6 +97,15 @@ ContextPtr ExportTTLScheduler::makeContext() const return context; } +std::vector ExportTTLScheduler::GroupToStart::partsWithRows() const +{ + std::vector result; + for (const auto & part : parts) + if (part->rows_count != 0) + result.push_back(part); + return result; +} + const ExportTTLScheduler::TaskState & ExportTTLScheduler::getCachedTaskState(TaskStates & task_states, const String & transaction_id) { auto it = task_states.find(transaction_id); @@ -132,17 +141,9 @@ UInt64 ExportTTLScheduler::run() const auto destination_id = storage.getExportTTLDestination(export_ttls.front()); destination = DatabaseCatalog::instance().tryGetTable(destination_id, context); if (destination) - { - const auto uuid = destination->getStorageID().uuid; - destination_key = ExportTTLUtils::destinationKey( - destination_id.database_name, - destination_id.table_name, - identifiesDestinationByUUID() && uuid != UUIDHelpers::Nil ? toString(uuid) : ""); - } + destination_key = ExportTTLUtils::destinationKey(destination_id.database_name, destination_id.table_name); else - { destination_error = fmt::format("The destination table {} of the EXPORT TTL does not exist", destination_id.getNameForLogs()); - } } /// A destination that cannot be resolved may be created again or not loaded yet, so what was @@ -422,10 +423,9 @@ ExportTTLScheduler::PartitionView ExportTTLScheduler::observePartition( continue; } - if (part->rows_count == 0) - continue; - - if (!isEligible(*part, export_ttls, now)) + /// A part without rows, e.g. emptied by a mutation, waits for no row. Nothing of it is exported, + /// but its group records it as exported, so that it merges with the exported parts around it. + if (part->rows_count != 0 && !isEligible(*part, export_ttls, now)) { for (const auto & [_, ttl_info] : part->ttl_infos.export_ttl) if (ttl_info.max > now && (!view.next_eligible_time || ttl_info.max < view.next_eligible_time)) @@ -552,9 +552,6 @@ std::optional ExportTTLScheduler::actOnPartition( for (const auto & part : group.parts) group_bytes += part->getBytesOnDisk(); - if (!resolved.failed.empty()) - group.retry_of = collectRetriedTasks(entry, resolved.failed, destination, now, task_states, context); - for (const auto & part : view.shippable) { if (max_parts && group.parts.size() >= max_parts) @@ -567,6 +564,12 @@ std::optional ExportTTLScheduler::actOnPartition( if (!group.parts.empty()) { + /// A group of parts without rows only exports nothing, so it has no task: it records them + /// as exported when it starts. + const size_t parts_with_rows = group.partsWithRows().size(); + if (parts_with_rows && !resolved.failed.empty()) + group.retry_of = collectRetriedTasks(entry, resolved.failed, destination, now, task_states, context); + /// The group takes over the claim of the failed task it retries, if any. group.entry = versioned; group.entry.entry.releaseClaim(); @@ -575,15 +578,27 @@ std::optional ExportTTLScheduler::actOnPartition( infos.reserve(group.parts.size()); for (const auto & part : group.parts) infos.push_back(part->info); - group.entry.entry.startClaim(group.transaction_id, infos); + + if (parts_with_rows) + group.entry.entry.startClaim(group.transaction_id, infos); + else + group.entry.entry.addExported(infos); if (startGroup(group, context)) { - ++in_flight; - task_states.insert_or_assign(group.transaction_id, TaskState{.status = TaskStatus::PENDING, .reached_commit = false, .retry_of = group.retry_of}); - LOG_INFO(log, "Started export task {} of {} part(s) of partition {} to {}{}", - group.transaction_id, group.parts.size(), partition_id, destination->getStorageID().getNameForLogs(), - resolved.failed.empty() ? "" : fmt::format(", retrying {}", resolved.failed)); + if (parts_with_rows) + { + ++in_flight; + task_states.insert_or_assign(group.transaction_id, TaskState{.status = TaskStatus::PENDING, .reached_commit = false, .retry_of = group.retry_of}); + LOG_INFO(log, "Started export task {} of {} part(s) of partition {} to {}{}", + group.transaction_id, parts_with_rows, partition_id, destination->getStorageID().getNameForLogs(), + resolved.failed.empty() ? "" : fmt::format(", retrying {}", resolved.failed)); + } + else + { + LOG_INFO(log, "Recorded {} part(s) of partition {} without rows as exported to {}", + group.parts.size(), partition_id, destination->getStorageID().getNameForLogs()); + } return std::move(group.entry.entry); } diff --git a/src/Storages/MergeTree/ExportTTLScheduler.h b/src/Storages/MergeTree/ExportTTLScheduler.h index e3f949c38644..af110b1c3d90 100644 --- a/src/Storages/MergeTree/ExportTTLScheduler.h +++ b/src/Storages/MergeTree/ExportTTLScheduler.h @@ -101,10 +101,15 @@ class ExportTTLScheduler String destination_key; StoragePtr destination; String partition_id; + /// Every part of the group, including the parts without rows, which have nothing to export. std::vector parts; ExportRetriedTasks retry_of; - /// The index entry with the claim of the group, to be stored with a check of its version. + /// The index entry with the claim of the group, to be stored with a check of its version. A + /// group of parts without rows only has no task, and its entry records them as exported. ExportTTLVersionedEntry entry; + + /// The parts that the task of the group exports. + std::vector partsWithRows() const; }; /// Only one replica schedules, which keeps the replicas from conflicting on the index. The lock @@ -116,11 +121,6 @@ class ExportTTLScheduler virtual ExportTTLIndexSnapshotPtr getIndexSnapshot() = 0; - /// Whether the key of the destination in the index contains its UUID, which tells apart a table - /// that was dropped and created again. Replicas have different UUIDs for the same destination - /// unless its database is `Replicated`, so they identify it by name only. - virtual bool identifiesDestinationByUUID() const = 0; - virtual String getReplicaName() const = 0; /// The replica holding the scheduler lock, empty if there is none. @@ -134,8 +134,9 @@ class ExportTTLScheduler /// Stores `entry` with a check of its version. Returns false on a conflict. virtual bool updateIndexEntry(const String & destination_key, const ExportTTLVersionedEntry & entry) = 0; - /// Creates the export task of `group` and stores its index entry. Returns false on a conflict, - /// e.g. a merge was assigned or the index changed meanwhile. + /// Creates the export task of the parts of `group` that have rows, if any, and stores its index + /// entry, provided no part of the group is being merged. Returns false on a conflict, e.g. a merge + /// was assigned or the index changed meanwhile. virtual bool startGroup(const GroupToStart & group, const ContextPtr & context) = 0; /// Whether the part is a source of an assigned merge that changes its block range. diff --git a/src/Storages/MergeTree/ExportTaskUtils.cpp b/src/Storages/MergeTree/ExportTaskUtils.cpp index 77e41c6fbb5a..ee40f6f5520a 100644 --- a/src/Storages/MergeTree/ExportTaskUtils.cpp +++ b/src/Storages/MergeTree/ExportTaskUtils.cpp @@ -640,8 +640,7 @@ namespace const std::string commit_info_path = fs::path(entry_path) / "commit_info"; const bool is_ttl_task = manifest.source == ExportTaskSource::ttl; - /// Replicas identify the destination by name, see `ReplicatedExportTTLScheduler`. - const auto destination_key = ExportTTLUtils::destinationKey(manifest.destination_database, manifest.destination_table, ""); + const auto destination_key = ExportTTLUtils::destinationKey(manifest.destination_database, manifest.destination_table); /// The index entry of a TTL task is stored with a check of its version, because the TTL /// scheduler may change it meanwhile, e.g. when it resolves another claim of the partition. diff --git a/src/Storages/MergeTree/MergeTreeExportTTLScheduler.cpp b/src/Storages/MergeTree/MergeTreeExportTTLScheduler.cpp index 9e50c7e580b0..9255b03e7842 100644 --- a/src/Storages/MergeTree/MergeTreeExportTTLScheduler.cpp +++ b/src/Storages/MergeTree/MergeTreeExportTTLScheduler.cpp @@ -15,6 +15,7 @@ #include #include +#include namespace fs = std::filesystem; @@ -202,16 +203,21 @@ ExportTTLScheduler::TaskState MergeTreeExportTTLScheduler::getTaskState(const St bool MergeTreeExportTTLScheduler::startGroup(const GroupToStart & group, const ContextPtr & context) { - const auto source_metadata = plain_storage.getInMemoryMetadataPtr(context, false); - const auto destination_metadata = group.destination->getInMemoryMetadataPtr(context, false); - ExportTaskUtils::verifyExportSchemaCastable(source_metadata, destination_metadata, group.destination->getStorageID(), context); - - MergeTreeData::DataPartsVector parts(group.parts.begin(), group.parts.end()); - auto descriptor = plain_storage.buildExportTask( - group.destination->getStorageID(), group.destination, source_metadata, destination_metadata, parts, group.partition_id, context); - descriptor.transaction_id = group.transaction_id; - descriptor.source = ExportTaskSource::ttl; - descriptor.retry_of = group.retry_of; + const auto parts_with_rows = group.partsWithRows(); + std::optional descriptor; + if (!parts_with_rows.empty()) + { + const auto source_metadata = plain_storage.getInMemoryMetadataPtr(context, false); + const auto destination_metadata = group.destination->getInMemoryMetadataPtr(context, false); + ExportTaskUtils::verifyExportSchemaCastable(source_metadata, destination_metadata, group.destination->getStorageID(), context); + + MergeTreeData::DataPartsVector parts(parts_with_rows.begin(), parts_with_rows.end()); + descriptor = plain_storage.buildExportTask( + group.destination->getStorageID(), group.destination, source_metadata, destination_metadata, parts, group.partition_id, context); + descriptor->transaction_id = group.transaction_id; + descriptor->source = ExportTaskSource::ttl; + descriptor->retry_of = group.retry_of; + } { /// Merge selection holds it while it checks the fence and tags the parts it merges. @@ -226,7 +232,8 @@ bool MergeTreeExportTTLScheduler::startGroup(const GroupToStart & group, const C return false; } - plain_storage.export_task_scheduler->addTask(std::move(descriptor), std::vector(group.parts.begin(), group.parts.end())); + if (descriptor) + plain_storage.export_task_scheduler->addTask(std::move(*descriptor), parts_with_rows); return true; } diff --git a/src/Storages/MergeTree/MergeTreeExportTTLScheduler.h b/src/Storages/MergeTree/MergeTreeExportTTLScheduler.h index aa0d912430bf..a59d8e6a67c2 100644 --- a/src/Storages/MergeTree/MergeTreeExportTTLScheduler.h +++ b/src/Storages/MergeTree/MergeTreeExportTTLScheduler.h @@ -60,7 +60,6 @@ class MergeTreeExportTTLScheduler final : public ExportTTLScheduler bool acquireSchedulerLock() override { return true; } bool isPaused() override; ExportTTLIndexSnapshotPtr getIndexSnapshot() override { return index.getSnapshot(); } - bool identifiesDestinationByUUID() const override { return true; } String getReplicaName() const override { return {}; } String getSchedulerReplica() override { return {}; } TaskState getTaskState(const String & transaction_id) override; diff --git a/src/Storages/MergeTree/MergeTreeExportTask.h b/src/Storages/MergeTree/MergeTreeExportTask.h index 5fd485fc7ff4..2ebb501a226d 100644 --- a/src/Storages/MergeTree/MergeTreeExportTask.h +++ b/src/Storages/MergeTree/MergeTreeExportTask.h @@ -59,8 +59,6 @@ struct MergeTreeExportTask String source_table; String destination_database; String destination_table; - /// Empty if the destination has no UUID. - String destination_uuid; time_t create_time = 0; ExportTaskSource source = ExportTaskSource::query; /// TTL tasks whose parts this task exports again because they failed, and whose commits may still @@ -157,8 +155,6 @@ struct MergeTreeExportTask json.set("source_table", source_table); json.set("destination_database", destination_database); json.set("destination_table", destination_table); - if (!destination_uuid.empty()) - json.set("destination_uuid", destination_uuid); json.set("create_time", create_time); json.set("source", String(magic_enum::enum_name(source))); if (!retry_of.empty()) @@ -236,8 +232,6 @@ struct MergeTreeExportTask task.source_table = json->getValue("source_table"); task.destination_database = json->getValue("destination_database"); task.destination_table = json->getValue("destination_table"); - if (json->has("destination_uuid")) - task.destination_uuid = json->getValue("destination_uuid"); task.create_time = json->getValue("create_time"); if (json->has("source")) diff --git a/src/Storages/MergeTree/ReplicatedExportTTLScheduler.cpp b/src/Storages/MergeTree/ReplicatedExportTTLScheduler.cpp index 36fb341dd9da..0e9a9393f7cc 100644 --- a/src/Storages/MergeTree/ReplicatedExportTTLScheduler.cpp +++ b/src/Storages/MergeTree/ReplicatedExportTTLScheduler.cpp @@ -216,8 +216,8 @@ bool ReplicatedExportTTLScheduler::startGroup(const GroupToStart & group, const auto zookeeper = replicated_storage.getZooKeeperAndAssertNotReadonly(); replicated_storage.checkAllReplicasSupportExportTTL(zookeeper); - /// Its `/log` version is checked when the task is created, so no merge can be assigned between - /// checking the parts below and claiming them. + /// Its `/log` version is checked when the index entry is stored, so no merge can be assigned between + /// checking the parts below, including the ones without rows, and claiming them. const auto merge_predicate = replicated_storage.queue.getMergePredicate(zookeeper, PartitionIdsHint{group.partition_id}); for (const auto & part : group.parts) { @@ -230,26 +230,29 @@ bool ReplicatedExportTTLScheduler::startGroup(const GroupToStart & group, const return false; } - const auto source_metadata = replicated_storage.getInMemoryMetadataPtr(context, false); - const auto destination_metadata = group.destination->getInMemoryMetadataPtr(context, false); - ExportTaskUtils::verifyExportSchemaCastable(source_metadata, destination_metadata, group.destination->getStorageID(), context); + Coordination::Requests ops; + ops.emplace_back(zkutil::makeCheckRequest(fs::path(replicated_storage.zookeeper_path) / "log", merge_predicate->getVersion())); - MergeTreeData::DataPartsVector parts(group.parts.begin(), group.parts.end()); - auto manifest = replicated_storage.buildExportTaskManifest( - group.destination->getStorageID(), group.destination, source_metadata, destination_metadata, parts, group.partition_id, context); - manifest.transaction_id = group.transaction_id; - manifest.source = ExportTaskSource::ttl; - manifest.retry_of = group.retry_of; + const auto parts_with_rows = group.partsWithRows(); + if (!parts_with_rows.empty()) + { + const auto source_metadata = replicated_storage.getInMemoryMetadataPtr(context, false); + const auto destination_metadata = group.destination->getInMemoryMetadataPtr(context, false); + ExportTaskUtils::verifyExportSchemaCastable(source_metadata, destination_metadata, group.destination->getStorageID(), context); + + MergeTreeData::DataPartsVector parts(parts_with_rows.begin(), parts_with_rows.end()); + auto manifest = replicated_storage.buildExportTaskManifest( + group.destination->getStorageID(), group.destination, source_metadata, destination_metadata, parts, group.partition_id, context); + manifest.transaction_id = group.transaction_id; + manifest.source = ExportTaskSource::ttl; + manifest.retry_of = group.retry_of; + + ExportTaskUtils::appendCreateExportTaskOps(ops, fs::path(replicated_storage.zookeeper_path) / "exports" / group.transaction_id, manifest); + } const auto & export_ttl_index = *replicated_storage.export_ttl_index; if (group.entry.version < 0) export_ttl_index.ensureDestination(zookeeper, group.destination_key, group.destination->getStorageID().getNameForLogs()); - - const auto task_path = fs::path(replicated_storage.zookeeper_path) / "exports" / group.transaction_id; - - Coordination::Requests ops; - ops.emplace_back(zkutil::makeCheckRequest(fs::path(replicated_storage.zookeeper_path) / "log", merge_predicate->getVersion())); - ExportTaskUtils::appendCreateExportTaskOps(ops, task_path, manifest); export_ttl_index.appendUpdateEntryOps(ops, group.destination_key, group.entry.entry, group.entry.version); ProfileEvents::increment(ProfileEvents::ExportTaskZooKeeperRequests); @@ -259,7 +262,7 @@ bool ReplicatedExportTTLScheduler::startGroup(const GroupToStart & group, const if (code == Coordination::Error::ZOK) { - if (replicated_storage.export_task_updating_task) + if (!parts_with_rows.empty() && replicated_storage.export_task_updating_task) replicated_storage.export_task_updating_task->schedule(); return true; } diff --git a/src/Storages/MergeTree/ReplicatedExportTTLScheduler.h b/src/Storages/MergeTree/ReplicatedExportTTLScheduler.h index 633f1ac74bab..664c8ccbb38a 100644 --- a/src/Storages/MergeTree/ReplicatedExportTTLScheduler.h +++ b/src/Storages/MergeTree/ReplicatedExportTTLScheduler.h @@ -26,7 +26,6 @@ class ReplicatedExportTTLScheduler final : public ExportTTLScheduler bool acquireSchedulerLock() override; bool isPaused() override; ExportTTLIndexSnapshotPtr getIndexSnapshot() override; - bool identifiesDestinationByUUID() const override { return false; } String getReplicaName() const override; String getSchedulerReplica() override; TaskState getTaskState(const String & transaction_id) override; diff --git a/src/Storages/MergeTree/tests/gtest_export_task_ordering.cpp b/src/Storages/MergeTree/tests/gtest_export_task_ordering.cpp index abc63a588bb3..107a0b5bea25 100644 --- a/src/Storages/MergeTree/tests/gtest_export_task_ordering.cpp +++ b/src/Storages/MergeTree/tests/gtest_export_task_ordering.cpp @@ -239,7 +239,6 @@ TEST_F(ExportTaskManifestBackCompatTest, TaskSourceAndRetryOfRoundTrip) { auto manifest = makeValidManifest(); manifest.source = ExportTaskSource::ttl; - manifest.destination_uuid = "00000000-0000-0000-0000-000000000001"; manifest.retry_of = { ExportRetriedTask{.transaction_id = "tx0", .block_ranges = {{1, 3}, {7, 7}}, .failed_time = 1700000000}, ExportRetriedTask{.transaction_id = "tx00", .block_ranges = {{1, 1}}, .failed_time = 1600000000}, @@ -247,7 +246,6 @@ TEST_F(ExportTaskManifestBackCompatTest, TaskSourceAndRetryOfRoundTrip) const auto parsed = ExportReplicatedMergeTreeTaskManifest::fromJsonString(manifest.toJsonString()); EXPECT_EQ(parsed.source, ExportTaskSource::ttl); - EXPECT_EQ(parsed.destination_uuid, manifest.destination_uuid); EXPECT_EQ(parsed.retry_of, manifest.retry_of); EXPECT_EQ(ExportRetriedTaskUtils::transactionIds(parsed.retry_of), (std::vector{"tx0", "tx00"})); diff --git a/src/Storages/MergeTree/tests/gtest_export_ttl_index.cpp b/src/Storages/MergeTree/tests/gtest_export_ttl_index.cpp index 21368c09c70d..c0237be288b7 100644 --- a/src/Storages/MergeTree/tests/gtest_export_ttl_index.cpp +++ b/src/Storages/MergeTree/tests/gtest_export_ttl_index.cpp @@ -129,8 +129,7 @@ TEST(ExportTTLIndex, JsonRoundTrip) TEST(ExportTTLIndex, DestinationKeysOfDottedNamesDiffer) { - EXPECT_NE(ExportTTLUtils::destinationKey("db.x", "y", ""), ExportTTLUtils::destinationKey("db", "x.y", "")); - EXPECT_NE(ExportTTLUtils::destinationKey("db", "t", ""), ExportTTLUtils::destinationKey("db", "t", "00000000-0000-0000-0000-000000000001")); + EXPECT_NE(ExportTTLUtils::destinationKey("db.x", "y"), ExportTTLUtils::destinationKey("db", "x.y")); } TEST(ExportTTLIndex, RangesOfParts) diff --git a/src/Storages/StorageMergeTree.cpp b/src/Storages/StorageMergeTree.cpp index e939b265e5e5..3c5afd3c8d2d 100644 --- a/src/Storages/StorageMergeTree.cpp +++ b/src/Storages/StorageMergeTree.cpp @@ -3906,8 +3906,6 @@ MergeTreeExportTask StorageMergeTree::buildExportTask( descriptor.source_table = getStorageID().table_name; descriptor.destination_database = dest_storage_id.database_name; descriptor.destination_table = dest_storage_id.table_name; - if (const auto uuid = dest_storage->getStorageID().uuid; uuid != UUIDHelpers::Nil) - descriptor.destination_uuid = toString(uuid); descriptor.create_time = time(nullptr); descriptor.status = MergeTreeExportTask::Status::PENDING; diff --git a/src/Storages/StorageReplicatedMergeTree.cpp b/src/Storages/StorageReplicatedMergeTree.cpp index 85fdecc5336c..e11de103c754 100644 --- a/src/Storages/StorageReplicatedMergeTree.cpp +++ b/src/Storages/StorageReplicatedMergeTree.cpp @@ -8798,8 +8798,6 @@ ExportReplicatedMergeTreeTaskManifest StorageReplicatedMergeTree::buildExportTas ExportReplicatedMergeTreeTaskManifest manifest; manifest.destination_database = dest_storage_id.database_name; manifest.destination_table = dest_storage_id.table_name; - if (const auto uuid = dest_storage->getStorageID().uuid; uuid != UUIDHelpers::Nil) - manifest.destination_uuid = toString(uuid); manifest.source_replica = replica_name; manifest.number_of_parts = part_names.size(); manifest.parts = part_names; diff --git a/tests/integration/test_export_ttl/test_failures.py b/tests/integration/test_export_ttl/test_failures.py index e1c283a69abc..9d94cad95e61 100644 --- a/tests/integration/test_export_ttl/test_failures.py +++ b/tests/integration/test_export_ttl/test_failures.py @@ -188,10 +188,8 @@ def test_restart_before_the_first_check(cluster, source_engine): def test_dropped_destination(cluster, source_engine): - """Without its destination the TTL exports nothing and reports why. A plain `MergeTree` table tells - apart a destination created again under the same name, and exports everything to it again: an - Iceberg table created again over the same data has the rows exported before twice. A replicated - table identifies the destination by name, and exports only what it did not export yet.""" + """Without its destination the TTL exports nothing and reports why. The destination is identified + by name, so once it is created again, only what was not exported yet is exported to it.""" node = cluster.instances["replica1"] mt_table, iceberg_table = make_tables(node, source_engine) node.query(f"INSERT INTO {mt_table} VALUES (1, 2020, {DUE})") @@ -210,12 +208,8 @@ def test_dropped_destination(cluster, source_engine): assert ttl_rows(node, mt_table)["2020"]["last_error"] == "" exported_again = ttl_tasks(node, mt_table)[-1]["parts"] - if source_engine == "MergeTree": - assert len(exported_again) == 2, ttl_tasks(node, mt_table) - assert_exactly_once(iceberg_ids(node, iceberg_table), [1, 1, 2]) - else: - assert len(exported_again) == 1, ttl_tasks(node, mt_table) - assert_exactly_once(iceberg_ids(node, iceberg_table), [1, 2]) + assert len(exported_again) == 1, ttl_tasks(node, mt_table) + assert_exactly_once(iceberg_ids(node, iceberg_table), [1, 2]) # One snapshot for the task that exported to the table created again. assert iceberg_snapshots(node, iceberg_table) == snapshots + 1 diff --git a/tests/integration/test_export_ttl/test_merge_fence.py b/tests/integration/test_export_ttl/test_merge_fence.py index fe05d98a33d0..d8e024868e80 100644 --- a/tests/integration/test_export_ttl/test_merge_fence.py +++ b/tests/integration/test_export_ttl/test_merge_fence.py @@ -220,6 +220,41 @@ def test_mutation_keeps_parts_exported(cluster, source_engine): assert_one_snapshot_per_task(node, mt_table, iceberg_table) +def test_parts_without_rows_are_recorded_as_exported(cluster, source_engine): + """A part that a mutation emptied, kept by `remove_empty_parts = 0`, has nothing to export, but its + group records it as exported: otherwise it would stay apart from the exported parts around it, and + the partition could never be merged into one part. A group of such parts only has no task.""" + node = cluster.instances["replica1"] + # No merge is assigned until the end, so the emptied part stays between the others. + mt_table, iceberg_table = make_tables( + node, source_engine, settings={"remove_empty_parts": 0, "max_bytes_to_merge_at_max_space_in_pool": 1} + ) + rows_of_parts = f"SELECT rows FROM system.parts WHERE database = currentDatabase() AND table = '{mt_table}' AND active ORDER BY name" + + node.query(f"SYSTEM STOP MOVES {mt_table}") + for row in [f"(1, 2020, {DUE})", f"(2, 2020, {NOT_DUE})", f"(3, 2020, {DUE})"]: + node.query(f"INSERT INTO {mt_table} VALUES {row}") + node.query(f"ALTER TABLE {mt_table} DELETE WHERE id = 2", settings={"mutations_sync": 2}) + assert node.query(rows_of_parts).split() == ["1", "0", "1"] + node.query(f"SYSTEM START MOVES {mt_table}") + + wait_for_partitions_exported(node, mt_table, ["2020"], exported=3) + tasks = ttl_tasks(node, mt_table) + assert len(tasks) == 1 and len(tasks[0]["parts"]) == 2, tasks + assert_exactly_once(iceberg_ids(node, iceberg_table), [1, 3]) + + node.query(f"INSERT INTO {mt_table} VALUES (4, 2020, {NOT_DUE})") + node.query(f"ALTER TABLE {mt_table} DELETE WHERE id = 4", settings={"mutations_sync": 2}) + wait_for_partitions_exported(node, mt_table, ["2020"], exported=4) + assert len(ttl_tasks(node, mt_table)) == 1 + + node.query(f"ALTER TABLE {mt_table} RESET SETTING max_bytes_to_merge_at_max_space_in_pool") + node.query(f"OPTIMIZE TABLE {mt_table} PARTITION ID '2020' FINAL") + assert len(active_parts(node, mt_table)) == 1 + assert_exactly_once(iceberg_ids(node, iceberg_table), [1, 3]) + assert_one_snapshot_per_task(node, mt_table, iceberg_table) + + def test_block_numbers_are_not_reused_after_a_restart(cluster): """A plain `MergeTree` table allocates block numbers from its parts. The exported blocks of a dropped partition must not be allocated again, or a new part would look exported.""" From f2a4f39eac79d54da6c3af9435b055c7fa0eb158 Mon Sep 17 00:00:00 2001 From: Arthur Passos Date: Thu, 1 Oct 2026 11:45:04 -0300 Subject: [PATCH 15/15] Simplify the scheduler of the `EXPORT` TTL - `MergeTreeData::exportParts` starts an export task and updates the export index atomically for both engines, and `exportPartitionToTable` is shared by `MergeTree` and `ReplicatedMergeTree`. - One `ExportTTLScheduler` class over an `IExportTTLIndex` (`MergeTreeExportTTLIndex` or `ReplicatedExportTTLIndex`) replaces the per-engine schedulers. - Every check ships the due parts of each partition as one group, with at most one group in flight per partition, so `ttl_export_check_period_seconds` (now 60 by default) is the batch interval. The batch settings and the `first_eligible_time` and `next_group_time` columns of `system.ttl_exports` are removed; `system.ttl_exports` is computed when it is queried. - `system.ttl_exports` reads nothing from Keeper: every replica of a table with an `EXPORT` TTL keeps its copy of the export index and of the scheduler replica up to date with watches on `version` and `scheduler_lock`. A table without one takes no scheduler lock and arms no watches. - A failed task is retried with exactly its claimed parts under the same `commit_id`, which replaces `retry_of`, so the destination commits them once even if the failed task landed after all. - A part without rows, e.g. emptied by a mutation, is exported as no files by `EXPORT PART`, `EXPORT PARTITION` and the TTL alike, without reading its min/max index, which is not initialized. --- docs/en/antalya/partition_export.md | 12 +- docs/en/antalya/ttl_export.md | 38 +- src/Core/SettingsChangesHistory.cpp | 8 +- .../ExportReplicatedMergeTreeTaskManifest.h | 14 +- src/Storages/ExportRetriedTask.h | 106 --- .../MergeTreeMergePredicate.cpp | 2 +- src/Storages/MergeTree/ExportPartTask.cpp | 44 +- src/Storages/MergeTree/ExportTTLIndex.cpp | 40 +- src/Storages/MergeTree/ExportTTLIndex.h | 31 +- src/Storages/MergeTree/ExportTTLScheduler.cpp | 750 +++++++----------- src/Storages/MergeTree/ExportTTLScheduler.h | 204 +---- src/Storages/MergeTree/ExportTaskInfo.h | 6 +- src/Storages/MergeTree/ExportTaskUtils.cpp | 75 +- src/Storages/MergeTree/ExportTaskUtils.h | 17 - src/Storages/MergeTree/IExportTTLIndex.h | 41 + src/Storages/MergeTree/MergeTreeData.cpp | 122 ++- src/Storages/MergeTree/MergeTreeData.h | 59 +- ...eduler.cpp => MergeTreeExportTTLIndex.cpp} | 85 +- .../MergeTree/MergeTreeExportTTLIndex.h | 52 ++ .../MergeTree/MergeTreeExportTTLScheduler.h | 79 -- src/Storages/MergeTree/MergeTreeExportTask.h | 13 +- .../MergeTreeExportTaskScheduler.cpp | 55 +- src/Storages/MergeTree/MergeTreeSettings.cpp | 30 +- .../MergeTree/ReplicatedExportTTLIndex.cpp | 148 +++- .../MergeTree/ReplicatedExportTTLIndex.h | 66 +- .../ReplicatedExportTTLScheduler.cpp | 292 ------- .../MergeTree/ReplicatedExportTTLScheduler.h | 48 -- .../MergeTree/ReplicatedExportTaskUpdater.cpp | 2 +- .../ReplicatedMergeTreeRestartingThread.cpp | 1 + .../tests/gtest_export_task_ordering.cpp | 56 +- .../tests/gtest_export_ttl_index.cpp | 53 +- src/Storages/StorageMergeTree.cpp | 173 ++-- src/Storages/StorageMergeTree.h | 20 +- src/Storages/StorageReplicatedMergeTree.cpp | 262 +++--- src/Storages/StorageReplicatedMergeTree.h | 20 +- .../StorageSystemDistributedExports.cpp | 11 +- .../System/StorageSystemTTLExports.cpp | 5 - src/Storages/System/attachSystemTables.cpp | 2 +- .../test.py | 47 ++ .../test_lifecycle.py | 52 +- tests/integration/test_export_ttl/common.py | 35 +- .../test_export_ttl/test_failures.py | 27 +- .../test_export_ttl/test_iceberg_commits.py | 20 +- .../test_export_ttl/test_merge_fence.py | 22 +- .../test_export_ttl/test_replication.py | 80 +- .../test_export_ttl/test_rolling_exports.py | 9 +- .../test_export_ttl/test_scheduling.py | 213 ++--- tests/queries/0_stateless/export_ttl.lib | 2 +- 48 files changed, 1445 insertions(+), 2104 deletions(-) delete mode 100644 src/Storages/ExportRetriedTask.h create mode 100644 src/Storages/MergeTree/IExportTTLIndex.h rename src/Storages/MergeTree/{MergeTreeExportTTLScheduler.cpp => MergeTreeExportTTLIndex.cpp} (61%) create mode 100644 src/Storages/MergeTree/MergeTreeExportTTLIndex.h delete mode 100644 src/Storages/MergeTree/MergeTreeExportTTLScheduler.h delete mode 100644 src/Storages/MergeTree/ReplicatedExportTTLScheduler.cpp delete mode 100644 src/Storages/MergeTree/ReplicatedExportTTLScheduler.h diff --git a/docs/en/antalya/partition_export.md b/docs/en/antalya/partition_export.md index 08727a4e17aa..f242a6842ac7 100644 --- a/docs/en/antalya/partition_export.md +++ b/docs/en/antalya/partition_export.md @@ -17,7 +17,7 @@ The export task can be killed by issuing the kill command: `KILL EXPORT //_.`. To ensure atomicity, a commit file containing the relative paths of all exported parts is also shipped. A data file should only be considered part of the dataset if a commit file references it. The commit file will be named using the following convention: `/commit_`. +Each MergeTree part will become a separate file with the following name convention: `//_.`. To ensure atomicity, a commit file containing the relative paths of all exported parts is also shipped. A data file should only be considered part of the dataset if a commit file references it. The commit file will be named using the following convention: `/commit_`, where the commit id is the transaction id of the export, see [`commit_id`](#source-columns). ## Plain (non-replicated) MergeTree {#plain-non-replicated-mergetree} @@ -239,7 +239,7 @@ committed_manifest_file: committed_marker_file: data/commit_9b2c1e5a-3f47-4c8e-8a1d-6f0b2d4e7c31 local_backoff_per_part: [] source: query -retry_of: [] +commit_id: 9b2c1e5a-3f47-4c8e-8a1d-6f0b2d4e7c31 Row 2: ────── @@ -265,7 +265,7 @@ committed_manifest_file: data/metadata/-m0.avro committed_marker_file: local_backoff_per_part: [] source: query -retry_of: [] +commit_id: d0e4f7a2-8c19-4b6d-9e3a-1f5c7b2e9d40 2 rows in set. Elapsed: 0.019 sec. @@ -289,7 +289,7 @@ Status values include: ### Commit info columns -- `committed_metadata_file` — for Iceberg destinations: path of the new `vN.metadata.json` written by the commit. Empty for non-Iceberg destinations and before the commit lands. If the commit was already finished by a previous run (detected via the transaction id stored in the snapshot summary), this column carries a human-readable sentinel string instead of a path because the original committer's paths are not recoverable from inside the impl. +- `committed_metadata_file` — for Iceberg destinations: path of the new `vN.metadata.json` written by the commit. Empty for non-Iceberg destinations and before the commit lands. If the commit was already finished by a previous run (detected via the commit id stored in the snapshot summary), this column carries a human-readable sentinel string instead of a path because the original committer's paths are not recoverable from inside the impl. - `committed_manifest_list` — for Iceberg destinations: path of the manifest list file (`snap-*.avro`) referenced by the new snapshot. Empty under the same conditions as `committed_metadata_file`. - `committed_manifest_file` — for Iceberg destinations: path of the manifest file referenced by `committed_manifest_list`. Empty under the same conditions as `committed_metadata_file`. - `committed_marker_file` — for plain object storage destinations: path of the per-transaction commit marker file written by the destination. Empty for Iceberg destinations and for tasks that have not committed yet. @@ -297,7 +297,7 @@ Status values include: ### Source columns {#source-columns} - `source` — `query` for a task of `EXPORT PARTITION`, `ttl` for a task of the table's `TTL ... EXPORT TO TABLE` expression. -- `retry_of` — for a task of the `EXPORT` TTL: transaction ids of earlier tasks that failed to export some of its parts after exporting all of theirs, so their commit may still land. Its commit checks whether any of them landed at the destination after all. +- `commit_id` — the id the task commits to the destination under, and checks the destination for. It is the transaction id of the task, except for a task of the `EXPORT` TTL that retries a failed one: it keeps the commit id of the failed task, so that the destination commits their parts once. To pick the latest exception across replicas: diff --git a/docs/en/antalya/ttl_export.md b/docs/en/antalya/ttl_export.md index 3592824d286b..9343b7e1e07f 100644 --- a/docs/en/antalya/ttl_export.md +++ b/docs/en/antalya/ttl_export.md @@ -29,10 +29,10 @@ PARTITION BY toYYYYMM(event_time) ORDER BY (user_id, event_time) TTL event_time + INTERVAL 30 DAY EXPORT TO TABLE events_archive, event_time + INTERVAL 90 DAY DELETE -SETTINGS ttl_export_batch_window_seconds = 300; +SETTINGS ttl_export_check_period_seconds = 300; ``` -Here rows are exported to `events_archive` 30 days after `event_time`, and deleted from `events` after 90 days, but not before they are exported. +Here rows are exported to `events_archive` 30 days after `event_time`, in groups shipped every 5 minutes, and deleted from `events` after 90 days, but not before they are exported. ## Requirements {#requirements} @@ -50,25 +50,19 @@ A destination that is compatible only for some groups is accepted, and might fai ## How parts are exported {#how-parts-are-exported} -A part becomes eligible once the maximum TTL value of its rows is due, the same rule move TTL uses, so a part is exported as a whole. The TTL does not have to be aligned with the partition key. When it is not, a part may hold rows that are due at different times, and is exported once the last of them is due. A due part that no group has claimed yet may also merge with a part of its partition that is not due, and its rows then wait for the rows of that part. A part written before the expression was added becomes eligible after `ALTER TABLE ... MATERIALIZE TTL`, which `ALTER TABLE ... MODIFY TTL` runs by default. +A part becomes eligible once the maximum TTL value of its rows is due, the same rule move TTL uses, so a part is exported as a whole. The TTL does not have to be aligned with the partition key. When it is not, a part may hold rows that are due at different times, and is exported once the last of them is due. A due part that no group has claimed yet may also merge with a part of its partition that is not due, and its rows then wait for the rows of that part. A part written before the expression was added becomes eligible after `ALTER TABLE ... MATERIALIZE TTL`, which `ALTER TABLE ... MODIFY TTL` runs by default. A part without rows, e.g. one kept by `remove_empty_parts = 0` after a mutation deleted all its rows, is eligible at once: it is exported with its group, writing nothing, so that it merges with the exported parts around it. -The eligible parts of a partition are collected into a group and exported together, when any of the following holds: - -- no new eligible part appeared in the partition for `ttl_export_batch_window_seconds`; -- the first of them became eligible more than `ttl_export_batch_max_delay_seconds` ago; -- their size reaches `ttl_export_batch_min_bytes`. - -A group has at most `ttl_export_max_parts_per_group` parts and `ttl_export_max_bytes_per_group` bytes, and contains parts of one partition only. Each partition has at most one group being exported at a time, and the table at most `ttl_export_max_concurrent_groups`. Merges of eligible parts continue while a group is being collected; a part that is being merged waits for the merge. +Every `ttl_export_check_period_seconds`, a check exports the eligible parts of each partition together, as one group. Each partition has at most one group being exported at a time: while it is, the parts that become eligible wait, and the first check after it committed exports them as the next group. The check period is thus the batch interval: a longer one gives fewer and bigger groups, i.e. fewer commits to the destination, and a longer delay before rows are exported. A part that is being merged waits for the merge. To keep exported rows apart from the others, parts that are exported, parts that are being exported and parts that are not exported are never merged together. Parts that are being exported are not merged at all until their group commits. On a `Replicated*MergeTree` table, every replica enforces this, which is why a replica advertises that it supports it, and groups are only started while every replica does. -One replica of a `Replicated*MergeTree` table schedules the groups; the others take part in exporting them like in `EXPORT PARTITION`. Every replica tracks the batch timers of the parts it has, so when another replica takes over, it continues them, give or take how much later it got the parts. +One replica of a `Replicated*MergeTree` table schedules the groups; the others take part in exporting them like in `EXPORT PARTITION`. When another replica takes over, its next check exports what the previous one did not. ## Failures and retries {#failures-and-retries} -A group that fails, is killed or times out is retried on the next check as a new task, with a new transaction id. The retry contains every part of the failed group, plus any part that became eligible meanwhile. Throttling comes from the export tasks themselves: per-part retry back-off, `export_merge_tree_task_timeout_seconds` and the commit attempts. +A group that fails, is killed or times out is retried on the next check as a new task, with a new transaction id and the commit id of the failed task, shown in the `commit_id` column of `system.distributed_exports`. The retry contains exactly the parts of the failed group; the parts that became eligible meanwhile wait for the next group. Throttling comes from the export tasks themselves: per-part retry back-off, `export_merge_tree_task_timeout_seconds` and the commit attempts. -A task that failed after uploading all parts and before marking it as committed might have committed to the destination. It is retried nevertheless. The retry task will therefore check if the previous task transaction has been completed by asking the destination storage. If it has been completed, it'll only export the delta parts. Otherwise, it'll export it all again. +A task that failed may have committed to the destination nevertheless, e.g. when its commit landed and the task timed out before it was marked as completed. Before retrying it, the scheduler asks the destination whether its commit id was committed: if it was, the parts are recorded as exported and nothing is retried. Otherwise the retry commits under the same commit id, so that the destination commits the rows once even if the commit of the failed task lands later. `KILL EXPORT` of a task of the TTL makes it retry. To stop exporting, use `SYSTEM STOP MOVES`, which pauses the TTL export of the table, or remove the expression. On a `Replicated*MergeTree` table, `SYSTEM STOP MOVES` pauses it only on the replica that schedules the groups, shown in the `scheduler_replica` column of `system.ttl_exports`, so run it on every replica, e.g. with `ON CLUSTER`. @@ -82,7 +76,7 @@ While a table has an `EXPORT` TTL expression, TTL that deletes or rewrites rows When the `EXPORT` TTL expression is removed, e.g. by `ALTER TABLE ... REMOVE TTL` or a `MODIFY TTL` without it, or its destination changes, groups being exported to the previous destination are killed, and what was exported to it is forgotten once they finished. Merges are then unrestricted again. Adding the expression back later exports every eligible part again, including parts that were exported before. -While the destination does not exist, e.g. it was dropped or is not loaded yet, nothing is exported, what was exported to it is kept, and the `last_error` column of `system.ttl_exports` says so. A destination that is dropped and created again under the same name is a new destination for a plain `MergeTree` table, so its eligible parts are exported to it again: an Iceberg table created again over the data of the dropped one then has the rows exported before twice. A `Replicated*MergeTree` table identifies the destination by its name only, because its replicas may have different UUIDs for it, so what was exported to the dropped table is not exported again. +While the destination does not exist, e.g. it was dropped or is not loaded yet, nothing is exported, what was exported to it is kept, and the `last_error` column of `system.ttl_exports` says so. The destination is identified by its name, so a destination that is dropped and created again under the same name is the same destination: what was exported to the dropped table is not exported again, and a group being exported when it was dropped may commit to the new table. `ALTER TABLE ... FORGET PARTITION` forgets what was exported from the partition, and is refused while a group of the partition is being exported. @@ -98,36 +92,28 @@ The settings of the export tasks, e.g. the output format settings, `export_merge | Setting | Default | Description | |---|---|---| -| `ttl_export_check_period_seconds` | `10` | How often eligible parts are looked for. | -| `ttl_export_batch_window_seconds` | `60` | A group is exported once no new eligible part appeared for this long. | -| `ttl_export_batch_max_delay_seconds` | `600` | A group is exported at the latest this long after its first part became eligible. | -| `ttl_export_batch_min_bytes` | `256 MiB` | A group is exported as soon as its parts reach this size. `0` disables it. | -| `ttl_export_max_parts_per_group` | `100` | Maximum number of parts of a group. Parts of a failed group are always retried together. | -| `ttl_export_max_bytes_per_group` | `100 GiB` | Maximum size of a group. A bigger part is exported on its own. `0` means unlimited. | -| `ttl_export_max_concurrent_groups` | `4` | Maximum number of groups of the table being exported at the same time. | +| `ttl_export_check_period_seconds` | `60` | How often the eligible parts are exported, i.e. the batch interval. | | `ttl_export_settings_profile` | `''` | Settings profile of the export tasks. | ## Monitoring {#monitoring} -`system.ttl_exports` has one row per partition of a table with an `EXPORT` TTL expression, as of the last check. For a `Replicated*MergeTree` table, every replica shows what was exported and the task being exported, which come from Keeper. The part counts and the batch timers are those of the parts of the replica, so they differ while a replica is fetching parts. The error of starting a group, e.g. because its rows would land in several destination partitions, is shown by the replica that schedules the groups, shown in `scheduler_replica`. +`system.ttl_exports` has one row per partition of a table with an `EXPORT` TTL expression, computed when it is queried, without reading Keeper. For a `Replicated*MergeTree` table, every replica shows what was exported and the task being exported, from the copy of the export index that it keeps up to date with Keeper watches; a replica shows the table once it has read the index. The part counts are those of the parts of the replica, so they differ while a replica is fetching parts. The error of the last check, e.g. because the rows of a group would land in several destination partitions, is shown by the replica that schedules the groups, shown in `scheduler_replica`. ```sql SELECT partition_id, exported_parts, claimed_parts, eligible_parts, parts_held_by_delete_gate, - first_eligible_time, next_group_time, current_transaction_id, last_error, scheduler_replica + current_transaction_id, last_error, scheduler_replica FROM system.ttl_exports WHERE table = 'events'; ``` - `exported_parts`, `claimed_parts` and `eligible_parts` count the active parts that are exported, that are being exported (or waiting to be retried), and that are due for export. -- `next_group_time` is when the eligible parts are exported at the latest. - `current_transaction_id` is the task exporting the partition now, see `system.distributed_exports`. - `scheduler_replica` is the replica that schedules the groups of a `Replicated*MergeTree` table, and is empty for a plain `MergeTree` table. -What was exported is read from Keeper again only when it changes, i.e. when a group is claimed, committed or released: the checks and the merge selection otherwise use a cached copy. The `ExportTTLIndexSnapshotRefreshes` profile event counts how often it is read again, so on an idle table it does not grow. +Every replica reads what was exported from Keeper again only when it changes, i.e. when a group is claimed, committed or released, and uses its copy otherwise: the checks and the merge selection only check that it is current, and `system.ttl_exports` does not. The `ExportTTLIndexSnapshotRefreshes` profile event counts how often it is read again, so on an idle table it does not grow. ## Limitations {#limitations} -- Export tasks are not removed, so they accumulate in Keeper (or in the data directory of a plain `MergeTree` table), and in `system.distributed_exports`. - On a plain object storage destination, files that no commit file references may remain, e.g. when a part of a failed group is mutated before its retry, its new name gives a new file, and the file of the failed attempt is left. Readers must follow the commit files, see [`EXPORT PARTITION`](/docs/en/antalya/partition_export.md). - Parts that are attached again, e.g. by `ALTER TABLE ... ATTACH PARTITION`, get new block numbers and are exported again. - A plain `MergeTree` table whose first disk is content-addressed cannot have the expression: it keeps what was exported and its export tasks in files on that disk, which does not support how they are written. A `Replicated*MergeTree` table keeps them in Keeper, so it can. diff --git a/src/Core/SettingsChangesHistory.cpp b/src/Core/SettingsChangesHistory.cpp index 07fd1a0a71bc..22416c40712c 100644 --- a/src/Core/SettingsChangesHistory.cpp +++ b/src/Core/SettingsChangesHistory.cpp @@ -1353,13 +1353,7 @@ const VersionToSettingsChangesMap & getMergeTreeSettingsChangesHistory() { addSettingsChanges(merge_tree_settings_changes_history, "26.6", { - {"ttl_export_check_period_seconds", 10, 10, "New setting"}, - {"ttl_export_batch_window_seconds", 60, 60, "New setting"}, - {"ttl_export_batch_max_delay_seconds", 600, 600, "New setting"}, - {"ttl_export_batch_min_bytes", 256_MiB, 256_MiB, "New setting"}, - {"ttl_export_max_parts_per_group", 100, 100, "New setting"}, - {"ttl_export_max_bytes_per_group", 100_GiB, 100_GiB, "New setting"}, - {"ttl_export_max_concurrent_groups", 4, 4, "New setting"}, + {"ttl_export_check_period_seconds", 60, 60, "New setting"}, {"ttl_export_settings_profile", "", "", "New setting"}, {"packed_skip_index_max_bytes", 0, 0, "New setting. Pack any skip-index substream whose serialized on-disk size is at most this many bytes into a single skp_idx.packed archive per part; larger substreams stay in the standalone skp_idx_.idx2 / .mrk2 layout. Decision is made per substream at write time."}, {"allow_tuple_element_aggregation", false, false, "New setting"}, diff --git a/src/Storages/ExportReplicatedMergeTreeTaskManifest.h b/src/Storages/ExportReplicatedMergeTreeTaskManifest.h index 286783385480..046c1170210a 100644 --- a/src/Storages/ExportReplicatedMergeTreeTaskManifest.h +++ b/src/Storages/ExportReplicatedMergeTreeTaskManifest.h @@ -7,7 +7,6 @@ #include #include #include -#include #include #include #include @@ -168,9 +167,9 @@ struct ExportReplicatedMergeTreeTaskManifest ExportTaskSource source = ExportTaskSource::query; String destination_database; String destination_table; - /// TTL export only: earlier tasks that failed to export some of these parts and whose commit may - /// still land. The commit checks whether any of them landed at the destination after all. - ExportRetriedTasks retry_of; + /// Id the task commits to the destination under: its transaction id, or for a task of the `EXPORT` + /// TTL that retries a failed one, the commit id of the failed task. + String commit_id; String source_replica; size_t number_of_parts; std::vector parts; @@ -210,8 +209,7 @@ struct ExportReplicatedMergeTreeTaskManifest json.set("source", String(magic_enum::enum_name(source))); json.set("destination_database", destination_database); json.set("destination_table", destination_table); - if (!retry_of.empty()) - json.set("retry_of", ExportRetriedTaskUtils::toJSON(retry_of)); + json.set("commit_id", commit_id); json.set("source_replica", source_replica); json.set("number_of_parts", number_of_parts); @@ -275,8 +273,8 @@ struct ExportReplicatedMergeTreeTaskManifest } manifest.destination_database = json->getValue("destination_database"); manifest.destination_table = json->getValue("destination_table"); - if (json->has("retry_of")) - manifest.retry_of = ExportRetriedTaskUtils::fromJSON(json->getArray("retry_of")); + /// Tasks created before commit ids existed commit under their transaction id. + manifest.commit_id = json->optValue("commit_id", manifest.transaction_id); manifest.source_replica = json->getValue("source_replica"); manifest.number_of_parts = json->getValue("number_of_parts"); diff --git a/src/Storages/ExportRetriedTask.h b/src/Storages/ExportRetriedTask.h deleted file mode 100644 index 70213e2a09a6..000000000000 --- a/src/Storages/ExportRetriedTask.h +++ /dev/null @@ -1,106 +0,0 @@ -#pragma once - -#include -#include -#include -#include - -#include -#include -#include - -namespace DB -{ - -namespace ErrorCodes -{ - extern const int INCORRECT_DATA; -} - -/// A failed task of the `EXPORT` TTL whose parts a later task exports again, and whose commit may -/// still land at the destination, e.g. a request that the destination applies late. The later task -/// commits only the parts that no landed task exported, so it records the blocks of each of them. -struct ExportRetriedTask -{ - String transaction_id; - /// `[min_block, max_block]` of its parts, in the partition of the later task. - std::vector> block_ranges; - /// When the scheduler found it failed. - time_t failed_time = 0; - - bool operator==(const ExportRetriedTask &) const = default; -}; - -using ExportRetriedTasks = std::vector; - -namespace ExportRetriedTaskUtils -{ - -inline std::vector transactionIds(const ExportRetriedTasks & tasks) -{ - std::vector result; - result.reserve(tasks.size()); - for (const auto & task : tasks) - result.push_back(task.transaction_id); - return result; -} - -inline Poco::JSON::Array::Ptr toJSON(const ExportRetriedTasks & tasks) -{ - Poco::JSON::Array::Ptr array = new Poco::JSON::Array(); - for (const auto & task : tasks) - { - Poco::JSON::Array::Ptr ranges = new Poco::JSON::Array(); - for (const auto & [min_block, max_block] : task.block_ranges) - { - Poco::JSON::Array::Ptr range = new Poco::JSON::Array(); - range->add(min_block); - range->add(max_block); - ranges->add(range); - } - - Poco::JSON::Object::Ptr object = new Poco::JSON::Object(); - object->set("transaction_id", task.transaction_id); - object->set("block_ranges", ranges); - object->set("failed_time", static_cast(task.failed_time)); - array->add(object); - } - return array; -} - -inline ExportRetriedTasks fromJSON(const Poco::JSON::Array::Ptr & array) -{ - ExportRetriedTasks tasks; - if (!array) - return tasks; - - for (size_t i = 0; i < array->size(); ++i) - { - const auto object = array->getObject(static_cast(i)); - if (!object) - throw Exception(ErrorCodes::INCORRECT_DATA, "Invalid `retry_of` element in an export task descriptor"); - - ExportRetriedTask task; - task.transaction_id = object->getValue("transaction_id"); - task.failed_time = static_cast(object->optValue("failed_time", 0)); - - if (const auto ranges = object->getArray("block_ranges")) - { - for (size_t j = 0; j < ranges->size(); ++j) - { - const auto range = ranges->getArray(static_cast(j)); - if (!range || range->size() != 2) - throw Exception(ErrorCodes::INCORRECT_DATA, - "Invalid block range of retried task {} in an export task descriptor", task.transaction_id); - task.block_ranges.emplace_back(range->getElement(0), range->getElement(1)); - } - } - - tasks.push_back(std::move(task)); - } - return tasks; -} - -} - -} diff --git a/src/Storages/MergeTree/Compaction/MergePredicates/MergeTreeMergePredicate.cpp b/src/Storages/MergeTree/Compaction/MergePredicates/MergeTreeMergePredicate.cpp index 43dff066d2e5..db0b2e276b12 100644 --- a/src/Storages/MergeTree/Compaction/MergePredicates/MergeTreeMergePredicate.cpp +++ b/src/Storages/MergeTree/Compaction/MergePredicates/MergeTreeMergePredicate.cpp @@ -24,7 +24,7 @@ MergeTreeMergePredicate::MergeTreeMergePredicate(const StorageMergeTree & storag , merge_mutate_lock(merge_mutate_lock_) , committing_blocks(storage.getCommittingBlocks()) , min_update_block(getMinUpdateBlockNumber(committing_blocks)) - , export_fence(storage.getExportFence()) + , export_fence(storage.getLatestExportFence()) , delete_gate(storage.getExportTTLDeleteGate()) { auto patches_vector = getPatchPartInfos(storage); diff --git a/src/Storages/MergeTree/ExportPartTask.cpp b/src/Storages/MergeTree/ExportPartTask.cpp index 8ee7d54f1e9d..94c9237f7638 100644 --- a/src/Storages/MergeTree/ExportPartTask.cpp +++ b/src/Storages/MergeTree/ExportPartTask.cpp @@ -236,8 +236,12 @@ bool ExportPartTask::executeStep() MergeTreeSequentialSourceType read_type = MergeTreeSequentialSourceType::Export; + /// A part without rows, e.g. emptied by a mutation, has nothing to write, and its min/max index, + /// which places the rows in the destination, is not initialized. It is exported as no files. + const bool has_rows = manifest.data_part->rows_count != 0; + Block block_with_partition_values; - if (metadata_snapshot->hasPartitionKey()) + if (metadata_snapshot->hasPartitionKey() && has_rows) { /// todo arthur do I need to init minmax_idx? block_with_partition_values = manifest.data_part->getMinMaxIndex()->getBlock(storage); @@ -309,18 +313,26 @@ bool ExportPartTask::executeStep() FailPointInjection::pauseFailPoint(FailPoints::export_part_pause_before_schema_validation); - auto import_result = destination_storage->import( - filename, - block_with_partition_values, - new_file_path_callback, - manifest.file_already_exists_policy, - manifest.settings[Setting::export_merge_tree_part_max_bytes_per_file], - manifest.settings[Setting::export_merge_tree_part_max_rows_per_file], - manifest.iceberg_metadata_json, - getFormatSettings(local_context), - local_context); + std::optional import_result; + if (has_rows) + { + import_result = destination_storage->import( + filename, + block_with_partition_values, + new_file_path_callback, + manifest.file_already_exists_policy, + manifest.settings[Setting::export_merge_tree_part_max_bytes_per_file], + manifest.settings[Setting::export_merge_tree_part_max_rows_per_file], + manifest.iceberg_metadata_json, + getFormatSettings(local_context), + local_context); + } - if (import_result.already_exported) + if (!import_result) + { + LOG_INFO(getLogger("ExportPartTask"), "Part {} has no rows, nothing to export", manifest.data_part->name); + } + else if (import_result->already_exported) { /// An earlier attempt at this part already wrote the whole set of destination files and /// committed it, but died before the result was recorded durably. Adopt those files as @@ -328,14 +340,14 @@ bool ExportPartTask::executeStep() ProfileEvents::increment(ProfileEvents::PartsExportDuplicated); LOG_INFO(getLogger("ExportPartTask"), "Part {} was already exported as {} file(s), reusing them", - manifest.data_part->name, import_result.exported_paths.size()); + manifest.data_part->name, import_result->exported_paths.size()); - for (const auto & exported_path : import_result.exported_paths) + for (const auto & exported_path : import_result->exported_paths) new_file_path_callback(exported_path); } else { - sink = std::move(import_result.sink); + sink = std::move(import_result->sink); bool apply_deleted_mask = true; bool read_with_direct_io = local_context->getSettingsRef()[Setting::min_bytes_to_use_direct_io] > manifest.data_part->getBytesOnDisk(); @@ -419,7 +431,7 @@ bool ExportPartTask::executeStep() /// For the direct EXPORT PART → Iceberg path there is no deferred-commit callback /// (the partition-export path provides one that writes to ZooKeeper). /// Commit the Iceberg metadata inline here so the rows become visible immediately. - if (destination_storage->isDataLake() && !manifest.completion_callback) + if (import_result && destination_storage->isDataLake() && !manifest.completion_callback) { IStorage::IcebergCommitExportArguments iceberg_args; iceberg_args.metadata_json_string = manifest.iceberg_metadata_json; diff --git a/src/Storages/MergeTree/ExportTTLIndex.cpp b/src/Storages/MergeTree/ExportTTLIndex.cpp index 27e63a521b9e..6d7d72821613 100644 --- a/src/Storages/MergeTree/ExportTTLIndex.cpp +++ b/src/Storages/MergeTree/ExportTTLIndex.cpp @@ -84,7 +84,7 @@ ExportFenceEntry ExportTTLIndexEntry::toFenceEntry(const String & destination) c return entry; } -void ExportTTLIndexEntry::startClaim(const String & transaction_id, const std::vector & parts) +void ExportTTLIndexEntry::startClaim(const String & commit_id, const String & transaction_id, const std::vector & parts) { if (claim) throw Exception(ErrorCodes::LOGICAL_ERROR, "Export task {} cannot claim parts of partition {}, which export task {} claimed", @@ -94,12 +94,20 @@ void ExportTTLIndexEntry::startClaim(const String & transaction_id, const std::v ranges.reserve(parts.size()); for (const auto & part : parts) ranges.emplace_back(partition_id, part.min_block, part.max_block, 0, 0); - claim = Claim{.transaction_id = transaction_id, .ranges = ExportFenceUtils::compactRanges(std::move(ranges))}; + claim = Claim{.commit_id = commit_id, .transaction_id = transaction_id, .ranges = ExportFenceUtils::compactRanges(std::move(ranges))}; } -void ExportTTLIndexEntry::commitClaim(const String & transaction_id, const std::vector & parts) +void ExportTTLIndexEntry::retryClaim(const String & transaction_id) { - if (claim && claim->transaction_id == transaction_id) + if (!claim) + throw Exception(ErrorCodes::LOGICAL_ERROR, "Export task {} cannot retry a claim of partition {}, which has none", + transaction_id, partition_id); + claim->transaction_id = transaction_id; +} + +void ExportTTLIndexEntry::commitClaim(const String & commit_id, const std::vector & parts) +{ + if (claim && claim->commit_id == commit_id) { exported.insert(exported.end(), claim->ranges.begin(), claim->ranges.end()); claim.reset(); @@ -122,6 +130,7 @@ String ExportTTLIndexEntry::toJSONString() const if (claim) { Poco::JSON::Object::Ptr claim_object = new Poco::JSON::Object(); + claim_object->set("commit_id", claim->commit_id); claim_object->set("transaction_id", claim->transaction_id); claim_object->set("ranges", rangesToJSON(claim->ranges)); json.set("claim", claim_object); @@ -149,8 +158,11 @@ ExportTTLIndexEntry ExportTTLIndexEntry::fromJSONString(const String & partition if (const auto claim_object = json->getObject("claim")) { + auto transaction_id = claim_object->getValue("transaction_id"); + auto commit_id = claim_object->optValue("commit_id", transaction_id); entry.claim = Claim{ - .transaction_id = claim_object->getValue("transaction_id"), + .commit_id = std::move(commit_id), + .transaction_id = std::move(transaction_id), .ranges = ExportFenceUtils::compactRanges(rangesFromJSON(partition_id, claim_object->getArray("ranges"))), }; } @@ -199,24 +211,6 @@ std::vector rangesOfParts(const std::vector & part_na return ExportFenceUtils::compactRanges(std::move(ranges)); } -std::vector> toBlockRanges(const std::vector & ranges) -{ - std::vector> result; - result.reserve(ranges.size()); - for (const auto & range : ranges) - result.emplace_back(range.min_block, range.max_block); - return result; -} - -std::vector fromBlockRanges(const String & partition_id, const std::vector> & block_ranges) -{ - std::vector ranges; - ranges.reserve(block_ranges.size()); - for (const auto & [min_block, max_block] : block_ranges) - ranges.emplace_back(partition_id, min_block, max_block, /* level */ 0, /* mutation */ 0); - return ExportFenceUtils::compactRanges(std::move(ranges)); -} - } } diff --git a/src/Storages/MergeTree/ExportTTLIndex.h b/src/Storages/MergeTree/ExportTTLIndex.h index 64db2a8272a9..6a8407420c11 100644 --- a/src/Storages/MergeTree/ExportTTLIndex.h +++ b/src/Storages/MergeTree/ExportTTLIndex.h @@ -8,7 +8,6 @@ #include #include #include -#include #include namespace DB @@ -25,13 +24,18 @@ namespace DB /// committed to the destination: /// - creating a task claims the ranges of its parts; /// - committing a task moves its claim to `exported`; -/// - retrying a failed task moves its claim to the new task; +/// - retrying a failed task moves its claim to the new task, which commits under the same id; /// - resolving a failed task that does not need a retry releases its claim. struct ExportTTLIndexEntry { /// Ranges owned by a TTL export task that did not commit: in flight, or failed and waiting to be retried. struct Claim { + /// Id the ranges are committed to the destination under. The retries of a failed task commit + /// under the id of the first task, so the destination commits the rows once if a failed task + /// landed after all. + String commit_id; + /// The task that exports the ranges now. String transaction_id; /// Compacted. std::vector ranges; @@ -55,20 +59,25 @@ struct ExportTTLIndexEntry ExportFenceEntry toFenceEntry(const String & destination) const; - /// Claims the ranges of `parts` for `transaction_id`. Throws if the entry has a claim already. - void startClaim(const String & transaction_id, const std::vector & parts); + /// Claims the ranges of `parts` for the task `transaction_id`, which commits them under + /// `commit_id`. Throws if the entry has a claim already. + void startClaim(const String & commit_id, const String & transaction_id, const std::vector & parts); - /// Moves the claim to the exported ranges if `transaction_id` holds it, adding `parts` as well: - /// a task may commit parts whose claim was lost. - void commitClaim(const String & transaction_id, const std::vector & parts); + /// Moves the claim to the task `transaction_id`, which retries the task that holds it. + void retryClaim(const String & transaction_id); - /// Adds the ranges of `parts` to the exported ranges. - void addExported(const std::vector & parts); + /// Moves the claim to the exported ranges if it is committed under `commit_id`, adding `parts` as + /// well: a task may commit parts whose claim was lost. + void commitClaim(const String & commit_id, const std::vector & parts); void releaseClaim() { claim.reset(); } String toJSONString() const; static ExportTTLIndexEntry fromJSONString(const String & partition_id, const String & json_string); + +private: + /// Adds the ranges of `parts` to the exported ranges. + void addExported(const std::vector & parts); }; /// An index entry as read, with the version to update it with a check. @@ -109,10 +118,6 @@ namespace ExportTTLUtils /// Ranges of the parts named `part_names`, compacted. std::vector rangesOfParts(const std::vector & part_names, MergeTreeDataFormatVersion format_version); - - /// `[min_block, max_block]` of `ranges`, as recorded for a retried task (see `ExportRetriedTask`). - std::vector> toBlockRanges(const std::vector & ranges); - std::vector fromBlockRanges(const String & partition_id, const std::vector> & block_ranges); } } diff --git a/src/Storages/MergeTree/ExportTTLScheduler.cpp b/src/Storages/MergeTree/ExportTTLScheduler.cpp index aae883df8c03..ab4b3c1ea078 100644 --- a/src/Storages/MergeTree/ExportTTLScheduler.cpp +++ b/src/Storages/MergeTree/ExportTTLScheduler.cpp @@ -3,7 +3,9 @@ #include #include #include +#include #include +#include #include #include #include @@ -25,56 +27,62 @@ namespace DB namespace MergeTreeSetting { extern const MergeTreeSettingsUInt64 ttl_export_check_period_seconds; - extern const MergeTreeSettingsUInt64 ttl_export_batch_window_seconds; - extern const MergeTreeSettingsUInt64 ttl_export_batch_max_delay_seconds; - extern const MergeTreeSettingsUInt64 ttl_export_batch_min_bytes; - extern const MergeTreeSettingsUInt64 ttl_export_max_parts_per_group; - extern const MergeTreeSettingsUInt64 ttl_export_max_bytes_per_group; - extern const MergeTreeSettingsUInt64 ttl_export_max_concurrent_groups; extern const MergeTreeSettingsString ttl_export_settings_profile; } namespace { -/// How long after its task failed a commit may still land: a destination may apply a request after -/// its sender gave up waiting for it. -constexpr time_t late_commit_window_seconds = 600; +/// The rule of move TTL: a part is due once the maximum TTL value of its rows is due. A part without +/// the TTL info (written before the TTL was added and not materialized) is never due. A part without +/// rows, e.g. emptied by a mutation, waits for no row: it is exported with its group, writing nothing, +/// so that it merges with the exported parts around it. +bool isDue(const IMergeTreeDataPart & part, const TTLDescriptions & export_ttls, time_t now) +{ + return part.rows_count == 0 + || selectTTLDescriptionForTTLInfos(export_ttls, part.ttl_infos.export_ttl, now, /* use_max */ true).has_value(); +} -/// The rule of move TTL: a part is eligible once the maximum TTL value of its rows is due. A part -/// without the TTL info (written before the TTL was added and not materialized) is never eligible. -bool isEligible(const IMergeTreeDataPart & part, const TTLDescriptions & export_ttls, time_t now) +std::map> getPartsByPartition(const MergeTreeData & storage) { - return selectTTLDescriptionForTTLInfos(export_ttls, part.ttl_infos.export_ttl, now, /* use_max */ true).has_value(); + std::map> result; + for (const auto & part : storage.getDataPartsVectorForInternalUsage()) + if (!part->info.isPatch()) + result[part->info.getPartitionId()].push_back(part); + return result; } +std::set getPartitionIds( + const std::map & entries, const std::map> & parts_by_partition) +{ + std::set result; + for (const auto & [partition_id, _] : entries) + result.insert(partition_id); + for (const auto & [partition_id, _] : parts_by_partition) + result.insert(partition_id); + return result; } -ExportTTLScheduler::ExportTTLScheduler(MergeTreeData & storage_) - : storage(storage_) - , log(getLogger(fmt::format("{} (ExportTTLScheduler)", storage_.getLogName()))) - , parts_held_by_delete_gate(CurrentMetrics::ExportTTLPartsHeldByDeleteGate, 0) +String getDestinationKey(const StorageID & destination_id) { + return ExportTTLUtils::destinationKey(destination_id.database_name, destination_id.table_name); } -String ExportTTLScheduler::getDestinationKey() const +const std::map & getEntries(const ExportTTLIndexSnapshot & snapshot, const String & destination_key) { - std::lock_guard lock(mutex); - return current_destination_key; + static const std::map no_entries; + const auto it = snapshot.entries.find(destination_key); + return it == snapshot.entries.end() ? no_entries : it->second; } -std::vector ExportTTLScheduler::getInfo() const +} + +ExportTTLScheduler::ExportTTLScheduler(MergeTreeData & storage_, IExportTTLIndex & index_) + : storage(storage_) + , ttl_index(index_) + , log(getLogger(fmt::format("{} (ExportTTLScheduler)", storage_.getLogName()))) + , parts_held_by_delete_gate(CurrentMetrics::ExportTTLPartsHeldByDeleteGate, 0) { - std::lock_guard lock(mutex); - std::vector result; - result.reserve(info_by_partition.size()); - for (const auto & [_, info] : info_by_partition) - { - result.push_back(info); - if (result.back().last_error.empty()) - result.back().last_error = current_destination_error; - } - return result; } ContextPtr ExportTTLScheduler::makeContext() const @@ -97,553 +105,359 @@ ContextPtr ExportTTLScheduler::makeContext() const return context; } -std::vector ExportTTLScheduler::GroupToStart::partsWithRows() const +bool ExportTTLScheduler::isPaused() const { - std::vector result; - for (const auto & part : parts) - if (part->rows_count != 0) - result.push_back(part); - return result; + return storage.parts_mover.moves_blocker.isCancelled(); } -const ExportTTLScheduler::TaskState & ExportTTLScheduler::getCachedTaskState(TaskStates & task_states, const String & transaction_id) +void ExportTTLScheduler::updatePartsHeldByDeleteGate() { - auto it = task_states.find(transaction_id); - if (it == task_states.end()) - it = task_states.emplace(transaction_id, getTaskState(transaction_id)).first; - return it->second; + const auto gate = storage.getExportTTLDeleteGate(); + const auto snapshot = ttl_index.getLatest(); + /// Until the index is read, it is not known which parts are exported. + if (gate.enabled && !snapshot) + return; + + size_t held = 0; + if (gate.enabled) + { + for (const auto & part : storage.getDataPartsVectorForInternalUsage()) + if (!part->info.isPatch() && gate.check(part->name, part->info, part->ttl_infos, snapshot->fence.get())) + ++held; + } + parts_held_by_delete_gate.changeTo(held); } UInt64 ExportTTLScheduler::run() { - const auto settings = storage.getSettings(); - const time_t period = static_cast(std::max(1, (*settings)[MergeTreeSetting::ttl_export_check_period_seconds])); - + const UInt64 period_ms = std::max(1, (*storage.getSettings())[MergeTreeSetting::ttl_export_check_period_seconds]) * 1000; const auto metadata = storage.getInMemoryMetadataPtr(nullptr, false); const auto export_ttls = metadata->getExportTTLs(); + updatePartsHeldByDeleteGate(); + + const auto clear_errors = [this] { std::lock_guard lock(mutex); - if (!export_ttls.empty()) - may_have_index = true; - if (!may_have_index) - return period * 1000; - } - - const auto context = makeContext(); + last_errors.clear(); + table_error.clear(); + }; - /// Every replica resolves the destination, because the delete gate of its merges depends on it. - StoragePtr destination; - String destination_key; - String destination_error; - if (!export_ttls.empty()) + /// Without an `EXPORT` TTL there is only the index of an earlier one to clean up, so a table that + /// never had one neither takes the scheduler lock nor reads Keeper. + if (export_ttls.empty()) { - const auto destination_id = storage.getExportTTLDestination(export_ttls.front()); - destination = DatabaseCatalog::instance().tryGetTable(destination_id, context); - if (destination) - destination_key = ExportTTLUtils::destinationKey(destination_id.database_name, destination_id.table_name); - else - destination_error = fmt::format("The destination table {} of the EXPORT TTL does not exist", destination_id.getNameForLogs()); + const auto latest = ttl_index.getLatest(); + if (!latest || latest->entries.empty()) + { + ttl_index.releaseSchedulerLock(); + clear_errors(); + return period_ms; + } } - /// A destination that cannot be resolved may be created again or not loaded yet, so what was - /// exported to it is kept, and its parts stay held from the delete TTL. - const bool destination_missing = !export_ttls.empty() && !destination; - + if (!ttl_index.tryAcquireSchedulerLock()) { - std::lock_guard lock(mutex); - if (!destination_missing && current_destination_key != destination_key) - { - batches.clear(); - last_errors.clear(); - info_by_partition.clear(); - current_destination_key = destination_key; - } - current_destination_error = destination_error; + clear_errors(); + return period_ms; } - const auto snapshot = getIndexSnapshot(); + if (isPaused()) + return period_ms; + + const auto snapshot = ttl_index.getSnapshot(); if (export_ttls.empty() && snapshot->entries.empty()) { - std::lock_guard lock(mutex); - may_have_index = false; - batches.clear(); - last_errors.clear(); - info_by_partition.clear(); - parts_held_by_delete_gate.changeTo(0); - return period * 1000; + ttl_index.releaseSchedulerLock(); + clear_errors(); + return period_ms; } - const bool is_scheduler = acquireSchedulerLock(); - if (is_scheduler && !destination_missing) + const auto destination_id = export_ttls.empty() ? StorageID::createEmpty() : storage.getExportTTLDestination(export_ttls.front()); + const String destination_key = export_ttls.empty() ? "" : getDestinationKey(destination_id); + const time_t now = time(nullptr); + + for (const auto & [key, entries] : snapshot->entries) { - for (const auto & [key, index] : snapshot->entries) - { - if (key == destination_key) - continue; + if (key == destination_key) + continue; - try - { - cleanupDestination(key, index); - } - catch (...) - { - tryLogCurrentException(log, fmt::format("While removing the TTL export index of destination {}", key)); - } + try + { + cleanupDestination(key, entries); + } + catch (...) + { + tryLogCurrentException(log, fmt::format("While removing the TTL export index of destination {}", key)); } } - if (!destination) - { - if (is_scheduler && !destination_error.empty()) - LOG_WARNING(log, "{}, nothing is exported", destination_error); - return period * 1000; - } - - const bool act = is_scheduler && !isPaused(); - const String scheduler_replica = is_scheduler ? getReplicaName() : getSchedulerReplica(); - - const time_t now = time(nullptr); + if (export_ttls.empty()) + return period_ms; - static const std::map no_entries; - const auto index_it = snapshot->entries.find(destination_key); - const auto & index = index_it == snapshot->entries.end() ? no_entries : index_it->second; - - std::map> parts_by_partition; - for (const auto & part : storage.getDataPartsVectorForInternalUsage()) - if (!part->info.isPatch()) - parts_by_partition[part->info.getPartitionId()].push_back(part); + const auto context = makeContext(); - TaskStates task_states; - size_t in_flight = 0; - if (act) + /// A destination that cannot be resolved may be created again or not loaded yet, so what was + /// exported to it is kept, and its parts stay held from the delete TTL. + const auto destination = DatabaseCatalog::instance().tryGetTable(destination_id, context); + if (!destination) { - for (const auto & [_, versioned] : index) - if (const auto & claim = versioned.entry.claim; - claim && getCachedTaskState(task_states, claim->transaction_id).status == TaskStatus::PENDING) - ++in_flight; + auto error = fmt::format("The destination table {} of the EXPORT TTL does not exist", destination_id.getNameForLogs()); + LOG_WARNING(log, "{}, nothing is exported", error); + std::lock_guard lock(mutex); + table_error = std::move(error); + return period_ms; } - std::set partition_ids; - for (const auto & [partition_id, _] : index) - partition_ids.insert(partition_id); - for (const auto & [partition_id, _] : parts_by_partition) - partition_ids.insert(partition_id); + const auto & entries = getEntries(*snapshot, destination_key); + const auto parts_by_partition = getPartsByPartition(storage); - time_t next_tick = now + period; - const std::vector no_parts; - for (const auto & partition_id : partition_ids) + std::map errors; + for (const auto & partition_id : getPartitionIds(entries, parts_by_partition)) { ExportTTLVersionedEntry versioned; - if (const auto it = index.find(partition_id); it != index.end()) + if (const auto it = entries.find(partition_id); it != entries.end()) versioned = it->second; else versioned.entry.partition_id = partition_id; const auto parts_it = parts_by_partition.find(partition_id); - const auto & parts = parts_it == parts_by_partition.end() ? no_parts : parts_it->second; - PartitionBatch batch; - String last_error; - { - std::lock_guard lock(mutex); - batch = batches[partition_id]; - if (const auto it = last_errors.find(partition_id); it != last_errors.end()) - last_error = it->second; - } - - PartitionView view; try { - view = observePartition(versioned.entry, parts, export_ttls, now, std::move(batch), task_states); - - /// The error of acting on the partition is shown until the scheduler acts on it again. - if (is_scheduler && !act) - view.info.last_error = last_error; - - if (act) - { - /// Shown as it is after acting, e.g. with the parts of a recorded commit as exported. - if (const auto updated_entry = actOnPartition(destination_key, destination, std::move(versioned), view, now, in_flight, task_states, context)) - view = observePartition(*updated_entry, parts, export_ttls, now, std::move(view.batch), task_states); - } + schedulePartition( + destination_key, destination, std::move(versioned), + parts_it == parts_by_partition.end() ? std::vector{} : parts_it->second, + export_ttls, now, context); } catch (...) { tryLogCurrentException(log, fmt::format("While exporting partition {} by TTL", partition_id)); - view.info.partition_id = partition_id; - view.info.last_error = getCurrentExceptionMessage(/* with_stacktrace */ false); + errors[partition_id] = getCurrentExceptionMessage(/* with_stacktrace */ false); } - - view.info.destination_database = destination->getStorageID().database_name; - view.info.destination_table = destination->getStorageID().table_name; - view.info.scheduler_replica = scheduler_replica; - - if (view.info.next_group_time > now) - next_tick = std::min(next_tick, view.info.next_group_time); - if (view.next_eligible_time > now) - next_tick = std::min(next_tick, view.next_eligible_time); - - std::lock_guard lock(mutex); - batches[partition_id] = std::move(view.batch); - if (view.info.last_error.empty()) - last_errors.erase(partition_id); - else - last_errors[partition_id] = view.info.last_error; - info_by_partition[partition_id] = std::move(view.info); - } - - { - std::lock_guard lock(mutex); - std::erase_if(info_by_partition, [&](const auto & item) { return !partition_ids.contains(item.first); }); - std::erase_if(batches, [&](const auto & item) { return !partition_ids.contains(item.first); }); - std::erase_if(last_errors, [&](const auto & item) { return !partition_ids.contains(item.first); }); - - size_t held = 0; - for (const auto & [_, info] : info_by_partition) - held += info.parts_held_by_delete_gate; - parts_held_by_delete_gate.changeTo(held); - } - - return static_cast(std::max(1, next_tick - now)) * 1000; -} - -ExportTTLScheduler::ResolvedClaim ExportTTLScheduler::resolveClaim( - ExportTTLIndexEntry & entry, const StoragePtr & destination, TaskStates & task_states, const ContextPtr & context) -{ - ResolvedClaim result; - if (!entry.claim) - return result; - - const auto transaction_id = entry.claim->transaction_id; - const auto & state = getCachedTaskState(task_states, transaction_id); - switch (state.status) - { - case TaskStatus::PENDING: - result.in_flight = transaction_id; - break; - - case TaskStatus::COMPLETED: - /// The commit of a plain `MergeTree` records it here, after the task is marked completed. - entry.commitClaim(transaction_id, {}); - result.changed = true; - break; - - case TaskStatus::FAILED: - case TaskStatus::KILLED: - case TaskStatus::MISSING: - /// E.g. a commit that landed and then the task timed out before it was marked completed. - if (state.reached_commit && destination->isExportTransactionCommitted(transaction_id, context)) - { - LOG_INFO(log, "Export task {} of partition {} did not complete, but it committed to the destination", - transaction_id, entry.partition_id); - entry.commitClaim(transaction_id, {}); - result.changed = true; - break; - } - - result.failed = transaction_id; - break; } - return result; + std::lock_guard lock(mutex); + last_errors = std::move(errors); + table_error.clear(); + return period_ms; } -ExportRetriedTasks ExportTTLScheduler::collectRetriedTasks( - const ExportTTLIndexEntry & entry, - const String & failed, +void ExportTTLScheduler::schedulePartition( + const String & destination_key, const StoragePtr & destination, - time_t now, - TaskStates & task_states, - const ContextPtr & context) -{ - ExportRetriedTasks result; - const auto & state = getCachedTaskState(task_states, failed); - - /// A task that did not export all its parts never committed, and a missing one was either - /// never created or checked when it went missing. - if (state.status != TaskStatus::MISSING && state.reached_commit) - result.push_back(ExportRetriedTask{ - .transaction_id = failed, - .block_ranges = ExportTTLUtils::toBlockRanges(entry.claim->ranges), - .failed_time = now, - }); - - for (const auto & retried : state.retry_of) - { - /// Past the window, and with no commit of it in progress, the task is checked once more, - /// and not retried by the next groups if it did not land. - const bool may_land = now < retried.failed_time + late_commit_window_seconds - || isCommitInProgress(retried.transaction_id) - || destination->isExportTransactionCommitted(retried.transaction_id, context); - - if (!may_land) - { - LOG_DEBUG(log, "Export task {} of partition {} did not commit, it is no longer checked", retried.transaction_id, entry.partition_id); - continue; - } - - result.push_back(retried); - } - - return result; -} - -ExportTTLScheduler::PartitionView ExportTTLScheduler::observePartition( - const ExportTTLIndexEntry & entry, + ExportTTLVersionedEntry versioned, const std::vector & parts, const TTLDescriptions & export_ttls, time_t now, - PartitionBatch batch, - TaskStates & task_states) + const ContextPtr & context) { - const auto settings = storage.getSettings(); + auto & entry = versioned.entry; + const auto & partition_id = entry.partition_id; - PartitionView view; - auto & info = view.info; - info.partition_id = entry.partition_id; + /// Whether `entry` differs from the stored one. + bool changed = false; + /// The commit id of the failed task that holds the claim, which the group retries. + String retried_commit_id; - std::vector eligible; - for (const auto & part : parts) + if (entry.claim) { - const auto state = entry.classify(part->info); - if (state == PartExportState::EXPORTED) + const auto claim = *entry.claim; + const auto status = storage.getExportTaskStatus(claim.transaction_id); + if (status == MergeTreeData::ExportTaskStatus::PENDING) + return; + + if (status == MergeTreeData::ExportTaskStatus::COMPLETED) { - ++info.exported_parts; - continue; + /// The commit of a plain `MergeTree` records it here, after the task is marked completed. + entry.commitClaim(claim.commit_id, {}); + changed = true; } - - if (part->ttl_infos.part_min_ttl && part->ttl_infos.part_min_ttl <= now) - ++info.parts_held_by_delete_gate; - - if (state == PartExportState::CLAIMED) + else if (destination->isExportTransactionCommitted(claim.commit_id, context)) { - ++info.claimed_parts; - view.claimed_parts.push_back(part); - continue; + /// E.g. a commit that landed and then the task timed out before it was marked completed. + LOG_INFO(log, "Export task {} of partition {} did not complete, but it committed to the destination", + claim.transaction_id, partition_id); + entry.commitClaim(claim.commit_id, {}); + changed = true; } - - /// A part without rows, e.g. emptied by a mutation, waits for no row. Nothing of it is exported, - /// but its group records it as exported, so that it merges with the exported parts around it. - if (part->rows_count != 0 && !isEligible(*part, export_ttls, now)) + else { - for (const auto & [_, ttl_info] : part->ttl_infos.export_ttl) - if (ttl_info.max > now && (!view.next_eligible_time || ttl_info.max < view.next_eligible_time)) - view.next_eligible_time = ttl_info.max; - continue; + retried_commit_id = claim.commit_id; } - - eligible.push_back(part); - if (!isPartBeingMerged(part)) - view.shippable.push_back(part); - } - - std::vector eligible_ranges; - for (const auto & part : eligible) - { - eligible_ranges.push_back(part->info); - info.eligible_bytes += part->getBytesOnDisk(); } - eligible_ranges = ExportFenceUtils::compactRanges(std::move(eligible_ranges)); - for (const auto & part : eligible) + std::vector group; + for (const auto & part : parts) { - if (!ExportFenceUtils::isCoveredByUnion(part->info, batch.seen_ranges)) + const auto state = entry.classify(part->info); + if (!retried_commit_id.empty()) { - batch.last_new_part_time = now; - if (!batch.first_eligible_time) - batch.first_eligible_time = now; + /// Exactly the claimed parts, so that the retry commits the rows of the failed task and no + /// other: the parts that became due meanwhile wait for the next group. + if (state == PartExportState::CLAIMED) + group.push_back(part); + } + else if (state == PartExportState::NONE && isDue(*part, export_ttls, now) && !storage.isPartBeingMerged(part->info)) + { + group.push_back(part); } } - batch.seen_ranges = std::move(eligible_ranges); - if (eligible.empty()) - batch = PartitionBatch{}; - - info.eligible_parts = eligible.size(); - info.first_eligible_time = eligible.empty() ? 0 : batch.first_eligible_time; - - std::sort(view.shippable.begin(), view.shippable.end(), [](const auto & lhs, const auto & rhs) { return lhs->info.min_block < rhs->info.min_block; }); - - size_t shippable_bytes = 0; - for (const auto & part : view.shippable) - shippable_bytes += part->getBytesOnDisk(); - - if (!view.shippable.empty() && batch.first_eligible_time) - { - const auto window = static_cast((*settings)[MergeTreeSetting::ttl_export_batch_window_seconds]); - const auto max_delay = static_cast((*settings)[MergeTreeSetting::ttl_export_batch_max_delay_seconds]); - const UInt64 min_bytes = (*settings)[MergeTreeSetting::ttl_export_batch_min_bytes]; - - info.next_group_time = std::min(batch.last_new_part_time + window, batch.first_eligible_time + max_delay); - view.batch_ready = now >= info.next_group_time || (min_bytes && shippable_bytes >= min_bytes); - } - - bool retry_pending = false; - if (entry.claim) - { - const auto status = getCachedTaskState(task_states, entry.claim->transaction_id).status; - if (status == TaskStatus::PENDING) - info.current_transaction_id = entry.claim->transaction_id; - else if (status != TaskStatus::COMPLETED) - retry_pending = true; - } - - if (info.current_transaction_id.empty() && (view.batch_ready || retry_pending)) - info.next_group_time = now; - - view.batch = std::move(batch); - return view; -} - -std::optional ExportTTLScheduler::actOnPartition( - const String & destination_key, - const StoragePtr & destination, - ExportTTLVersionedEntry versioned, - PartitionView & view, - time_t now, - size_t & in_flight, - TaskStates & task_states, - const ContextPtr & context) -{ - auto & entry = versioned.entry; - auto & info = view.info; - const auto & partition_id = entry.partition_id; - const auto settings = storage.getSettings(); - - auto resolved = resolveClaim(entry, destination, task_states, context); - info.current_transaction_id = resolved.in_flight; - - /// The claimed parts are the parts of the failed task that still exist. - std::vector retry_parts; - if (!resolved.failed.empty()) - retry_parts = view.claimed_parts; - /// The parts of a failed task that no longer exist, e.g. were dropped, have nothing left to export. - if (!resolved.failed.empty() && retry_parts.empty()) + if (!retried_commit_id.empty() && group.empty()) { LOG_INFO(log, "Export task {} of partition {} failed and none of its parts exists anymore, releasing its claim", - resolved.failed, partition_id); + entry.claim->transaction_id, partition_id); entry.releaseClaim(); - resolved.failed.clear(); - resolved.changed = true; - } + changed = true; + retried_commit_id.clear(); - const bool ready = !retry_parts.empty() || view.batch_ready; - const UInt64 max_concurrent = (*settings)[MergeTreeSetting::ttl_export_max_concurrent_groups]; + for (const auto & part : parts) + if (entry.classify(part->info) == PartExportState::NONE && isDue(*part, export_ttls, now) && !storage.isPartBeingMerged(part->info)) + group.push_back(part); + } - if (ready && resolved.in_flight.empty() && (!max_concurrent || in_flight < max_concurrent)) + if (!group.empty()) { - info.next_group_time = now; - - const UInt64 max_parts = (*settings)[MergeTreeSetting::ttl_export_max_parts_per_group]; - const UInt64 max_bytes = (*settings)[MergeTreeSetting::ttl_export_max_bytes_per_group]; - - GroupToStart group; - group.transaction_id = toString(UUIDHelpers::generateV4()); - group.destination_key = destination_key; - group.destination = destination; - group.partition_id = partition_id; - - /// Every claimed part of the failed task is retried: a part left out would lose its claim and - /// could be exported again later although the failed task may still land. - group.parts = retry_parts; - size_t group_bytes = 0; - for (const auto & part : group.parts) - group_bytes += part->getBytesOnDisk(); - - for (const auto & part : view.shippable) - { - if (max_parts && group.parts.size() >= max_parts) - break; - if (max_bytes && !group.parts.empty() && group_bytes + part->getBytesOnDisk() > max_bytes) - break; - group.parts.push_back(part); - group_bytes += part->getBytesOnDisk(); - } + MergeTreeData::ExportPartsRequest request; + request.destination = destination; + request.partition_id = partition_id; + request.parts.assign(group.begin(), group.end()); + request.source = ExportTaskSource::ttl; + request.transaction_id = toString(UUIDHelpers::generateV4()); + request.commit_id = retried_commit_id.empty() ? request.transaction_id : retried_commit_id; + + std::vector infos; + infos.reserve(group.size()); + for (const auto & part : group) + infos.push_back(part->info); + + MergeTreeData::ExportTTLIndexUpdate update{.destination_key = destination_key, .entry = versioned}; + auto & updated = update.entry.entry; + if (retried_commit_id.empty()) + updated.startClaim(request.commit_id, request.transaction_id, infos); + else + updated.retryClaim(request.transaction_id); - if (!group.parts.empty()) + if (storage.exportParts(request, context, &update)) { - /// A group of parts without rows only exports nothing, so it has no task: it records them - /// as exported when it starts. - const size_t parts_with_rows = group.partsWithRows().size(); - if (parts_with_rows && !resolved.failed.empty()) - group.retry_of = collectRetriedTasks(entry, resolved.failed, destination, now, task_states, context); - - /// The group takes over the claim of the failed task it retries, if any. - group.entry = versioned; - group.entry.entry.releaseClaim(); - - std::vector infos; - infos.reserve(group.parts.size()); - for (const auto & part : group.parts) - infos.push_back(part->info); - - if (parts_with_rows) - group.entry.entry.startClaim(group.transaction_id, infos); + if (retried_commit_id.empty()) + LOG_INFO(log, "Started export task {} of {} part(s) of partition {} to {}", + request.transaction_id, group.size(), partition_id, destination->getStorageID().getNameForLogs()); else - group.entry.entry.addExported(infos); - - if (startGroup(group, context)) - { - if (parts_with_rows) - { - ++in_flight; - task_states.insert_or_assign(group.transaction_id, TaskState{.status = TaskStatus::PENDING, .reached_commit = false, .retry_of = group.retry_of}); - LOG_INFO(log, "Started export task {} of {} part(s) of partition {} to {}{}", - group.transaction_id, parts_with_rows, partition_id, destination->getStorageID().getNameForLogs(), - resolved.failed.empty() ? "" : fmt::format(", retrying {}", resolved.failed)); - } - else - { - LOG_INFO(log, "Recorded {} part(s) of partition {} without rows as exported to {}", - group.parts.size(), partition_id, destination->getStorageID().getNameForLogs()); - } - return std::move(group.entry.entry); - } - - LOG_DEBUG(log, "Merges or the export index of partition {} changed while starting an export task, will retry", partition_id); + LOG_INFO(log, "Started export task {} of {} part(s) of partition {} to {}, retrying commit {}", + request.transaction_id, group.size(), partition_id, destination->getStorageID().getNameForLogs(), retried_commit_id); + return; } - } - if (!resolved.changed) - return std::nullopt; - - if (!updateIndexEntry(destination_key, versioned)) - { - LOG_DEBUG(log, "The export index of partition {} changed concurrently, will retry", partition_id); - return std::nullopt; + LOG_DEBUG(log, "Merges or the export index of partition {} changed while starting an export task, will retry", partition_id); } - return std::move(entry); + if (changed && !ttl_index.updateEntry(destination_key, versioned)) + LOG_DEBUG(log, "The export index of partition {} changed concurrently, will retry", partition_id); } -void ExportTTLScheduler::cleanupDestination(const String & destination_key, const std::map & index) +void ExportTTLScheduler::cleanupDestination(const String & destination_key, const std::map & entries) { - TaskStates task_states; bool claims_left = false; - for (auto [_, versioned] : index) + for (auto [_, versioned] : entries) { if (!versioned.entry.claim) continue; const auto transaction_id = versioned.entry.claim->transaction_id; - if (getCachedTaskState(task_states, transaction_id).status == TaskStatus::PENDING) + if (storage.getExportTaskStatus(transaction_id) == MergeTreeData::ExportTaskStatus::PENDING) { LOG_INFO(log, "Killing export task {}: the EXPORT TTL no longer exports to destination {}", transaction_id, destination_key); - killTask(transaction_id); + storage.killExportTask(transaction_id); claims_left = true; continue; } versioned.entry.releaseClaim(); - if (!updateIndexEntry(destination_key, versioned)) + if (!ttl_index.updateEntry(destination_key, versioned)) claims_left = true; } if (!claims_left) - removeDestination(destination_key); + ttl_index.removeDestination(destination_key); +} + +std::vector ExportTTLScheduler::getInfo() const +{ + const auto metadata = storage.getInMemoryMetadataPtr(nullptr, false); + const auto export_ttls = metadata->getExportTTLs(); + if (export_ttls.empty()) + return {}; + + /// Until the index is read, it is not known what was exported, so the table has no rows. + const auto snapshot = ttl_index.getLatest(); + if (!snapshot) + return {}; + + const auto destination_id = storage.getExportTTLDestination(export_ttls.front()); + const auto & entries = getEntries(*snapshot, getDestinationKey(destination_id)); + const auto parts_by_partition = getPartsByPartition(storage); + const auto scheduler_replica = ttl_index.getSchedulerReplica(); + const auto gate = storage.getExportTTLDeleteGate(); + const time_t now = time(nullptr); + + std::map errors; + String error; + { + std::lock_guard lock(mutex); + errors = last_errors; + error = table_error; + } + + std::vector result; + for (const auto & partition_id : getPartitionIds(entries, parts_by_partition)) + { + auto & info = result.emplace_back(); + info.partition_id = partition_id; + info.destination_database = destination_id.database_name; + info.destination_table = destination_id.table_name; + info.scheduler_replica = scheduler_replica; + + ExportTTLIndexEntry entry; + entry.partition_id = partition_id; + if (const auto it = entries.find(partition_id); it != entries.end()) + entry = it->second.entry; + + if (const auto it = parts_by_partition.find(partition_id); it != parts_by_partition.end()) + { + for (const auto & part : it->second) + { + if (gate.check(part->name, part->info, part->ttl_infos, snapshot->fence.get())) + ++info.parts_held_by_delete_gate; + + const auto state = entry.classify(part->info); + if (state == PartExportState::EXPORTED) + { + ++info.exported_parts; + } + else if (state == PartExportState::CLAIMED) + { + ++info.claimed_parts; + } + else if (isDue(*part, export_ttls, now)) + { + ++info.eligible_parts; + info.eligible_bytes += part->getBytesOnDisk(); + } + } + } + + if (entry.claim && storage.getKnownExportTaskStatus(entry.claim->transaction_id) == MergeTreeData::ExportTaskStatus::PENDING) + info.current_transaction_id = entry.claim->transaction_id; + + const auto error_it = errors.find(partition_id); + info.last_error = error_it == errors.end() ? error : error_it->second; + } + + return result; } } diff --git a/src/Storages/MergeTree/ExportTTLScheduler.h b/src/Storages/MergeTree/ExportTTLScheduler.h index af110b1c3d90..2d4515d0844b 100644 --- a/src/Storages/MergeTree/ExportTTLScheduler.h +++ b/src/Storages/MergeTree/ExportTTLScheduler.h @@ -2,22 +2,21 @@ #include #include -#include #include #include +#include #include #include #include #include #include -#include -#include #include namespace DB { +class IExportTTLIndex; class MergeTreeData; /// What `system.ttl_exports` shows about one partition of a table with a `TTL ... EXPORT` expression. @@ -32,10 +31,6 @@ struct ExportTTLPartitionInfo size_t eligible_bytes = 0; /// Parts held back from the delete TTL because they are not exported yet. size_t parts_held_by_delete_gate = 0; - /// When this replica first saw an eligible part that is not exported, 0 if there is none. - time_t first_eligible_time = 0; - /// When the next group can start at the latest, 0 if nothing waits. - time_t next_group_time = 0; /// Transaction id of the task exporting the partition now, empty if there is none. String current_transaction_id; String last_error; @@ -45,195 +40,62 @@ struct ExportTTLPartitionInfo /// The background task of a table with a `TTL EXPORT TO TABLE ` expression. /// -/// On every tick it ships groups of eligible parts (the maximum TTL value of their rows is due) to -/// the destination as export tasks of the table, one partition per group, and never exports a part -/// twice: what was exported is recorded in the export index (see `ExportTTLIndexEntry`), and the -/// merge fence keeps parts in different export states from merging. +/// Every check ships the eligible parts (the maximum TTL value of their rows is due) of each partition +/// to the destination as one export task of the table, and never exports a part twice: what was +/// exported is recorded in the export index (see `ExportTTLIndexEntry`), and the merge fence keeps +/// parts in different export states from merging. A partition has at most one task in flight, so the +/// check period is the batch interval. A task that failed without committing keeps its parts claimed, +/// and a new task retries exactly them under the same commit id, so the destination commits them once +/// even if the failed task landed after all. /// -/// Eligible parts of a partition are shipped once no new eligible part appeared for the batching -/// window, once the first of them waited for the maximum delay, or once they reach the size -/// threshold. A task that failed without committing keeps its parts claimed, and they are retried -/// first by the next group, which records in `retry_of` the failed tasks whose commit may still land. -/// -/// Every replica observes the state of every partition, which `system.ttl_exports` shows, and -/// tracks the batching windows of the parts it has, so a replica that takes over the scheduling -/// continues them. Only the replica holding the scheduler lock acts: it resolves finished tasks and -/// starts groups. -/// -/// Engines implement access to the index and to their export tasks. -class ExportTTLScheduler +/// Only the replica holding the scheduler lock checks. `system.ttl_exports` is computed when it is +/// queried, from the copy of the index that the replica keeps (see `IExportTTLIndex::getLatest`) and +/// its parts, without reading Keeper. +class ExportTTLScheduler final { public: - explicit ExportTTLScheduler(MergeTreeData & storage_); - virtual ~ExportTTLScheduler() = default; + ExportTTLScheduler(MergeTreeData & storage_, IExportTTLIndex & index_); - /// One tick. Returns the number of milliseconds until the next one. + /// One check. Returns the number of milliseconds until the next one. UInt64 run(); std::vector getInfo() const; - /// Key in the export index of the current destination, empty if it is not known yet. - String getDestinationKey() const; - -protected: - enum class TaskStatus : UInt8 - { - PENDING, - COMPLETED, - FAILED, - KILLED, - /// No such task, e.g. a crash between claiming its parts and creating it. - MISSING, - }; - - struct TaskState - { - TaskStatus status = TaskStatus::MISSING; - /// Whether it exported all its parts, which it does before committing: a task that did not - /// cannot have committed. Unknown counts as reached. - bool reached_commit = true; - ExportRetriedTasks retry_of; - }; - - struct GroupToStart - { - String transaction_id; - String destination_key; - StoragePtr destination; - String partition_id; - /// Every part of the group, including the parts without rows, which have nothing to export. - std::vector parts; - ExportRetriedTasks retry_of; - /// The index entry with the claim of the group, to be stored with a check of its version. A - /// group of parts without rows only has no task, and its entry records them as exported. - ExportTTLVersionedEntry entry; - - /// The parts that the task of the group exports. - std::vector partsWithRows() const; - }; - - /// Only one replica schedules, which keeps the replicas from conflicting on the index. The lock - /// is kept until it is lost or the table shuts down. Correctness does not depend on it. - virtual bool acquireSchedulerLock() = 0; - - /// E.g. `SYSTEM STOP MOVES`. - virtual bool isPaused() = 0; - - virtual ExportTTLIndexSnapshotPtr getIndexSnapshot() = 0; - - virtual String getReplicaName() const = 0; - - /// The replica holding the scheduler lock, empty if there is none. - virtual String getSchedulerReplica() = 0; - - virtual TaskState getTaskState(const String & transaction_id) = 0; - - /// Whether a replica is in the commit of the task, e.g. one that started before the task failed. - virtual bool isCommitInProgress(const String & transaction_id) = 0; - - /// Stores `entry` with a check of its version. Returns false on a conflict. - virtual bool updateIndexEntry(const String & destination_key, const ExportTTLVersionedEntry & entry) = 0; - - /// Creates the export task of the parts of `group` that have rows, if any, and stores its index - /// entry, provided no part of the group is being merged. Returns false on a conflict, e.g. a merge - /// was assigned or the index changed meanwhile. - virtual bool startGroup(const GroupToStart & group, const ContextPtr & context) = 0; - - /// Whether the part is a source of an assigned merge that changes its block range. - virtual bool isPartBeingMerged(const MergeTreeDataPartPtr & part) = 0; - - virtual void killTask(const String & transaction_id) = 0; - virtual void removeDestination(const String & destination_key) = 0; - +private: MergeTreeData & storage; + IExportTTLIndex & ttl_index; const LoggerPtr log; -private: - struct PartitionBatch - { - /// Block ranges of the eligible parts already seen, so a merge of seen parts is not new. - std::vector seen_ranges; - time_t first_eligible_time = 0; - time_t last_new_part_time = 0; - }; - - /// What every replica can tell about a partition without acting on it. - struct PartitionView - { - ExportTTLPartitionInfo info; - PartitionBatch batch; - std::vector claimed_parts; - /// Eligible parts that are neither exported nor claimed nor being merged, by block number. - std::vector shippable; - bool batch_ready = false; - /// When a part that is not due yet becomes eligible, 0 if there is none. - time_t next_eligible_time = 0; - }; - - using TaskStates = std::unordered_map; - mutable std::mutex mutex; - /// By partition id, for the current destination. - std::map batches; + /// Kept by the replica that schedules: the error of the last check of each partition, and why the + /// table exports nothing. std::map last_errors; - std::map info_by_partition; - String current_destination_key; - String current_destination_error; - - /// False once there is neither an `EXPORT` TTL nor an index left, so ticks do no Keeper reads. - bool may_have_index = true; + String table_error; CurrentMetrics::Increment parts_held_by_delete_gate; ContextPtr makeContext() const; - /// Resolves the claim of the index entry unless its task is in flight: a task that completed, or - /// that failed but committed to the destination, has its claim moved to the exported ranges. - struct ResolvedClaim - { - /// The task in flight, or the failed task whose parts are still claimed; empty if there is none. - String in_flight; - String failed; - bool changed = false; - }; - ResolvedClaim resolveClaim(ExportTTLIndexEntry & entry, const StoragePtr & destination, TaskStates & task_states, const ContextPtr & context); - - /// The task `failed`, which holds the claim of `entry`, and the ones it retried whose commit may - /// still land, for the `retry_of` of the group that retries its parts. - ExportRetriedTasks collectRetriedTasks( - const ExportTTLIndexEntry & entry, - const String & failed, - const StoragePtr & destination, - time_t now, - TaskStates & task_states, - const ContextPtr & context); - - const TaskState & getCachedTaskState(TaskStates & task_states, const String & transaction_id); + /// E.g. `SYSTEM STOP MOVES`. + bool isPaused() const; - /// Kills the TTL tasks of a destination that is no longer the destination of the TTL, and - /// removes its index once none of them holds a claim. - void cleanupDestination(const String & destination_key, const std::map & index); + void updatePartsHeldByDeleteGate(); - /// Read-only, except for the batching window of the partition, which it returns in the view. - PartitionView observePartition( - const ExportTTLIndexEntry & entry, - const std::vector & parts, - const TTLDescriptions & export_ttls, - time_t now, - PartitionBatch batch, - TaskStates & task_states); - - /// On the replica that schedules: resolves finished tasks and starts a group if one is due. - /// Returns the index entry of the partition if it was changed. - std::optional actOnPartition( + /// Resolves the claim of the partition unless its task is in flight. A failed task is retried with + /// exactly its claimed parts; otherwise every due part of the partition that is not being merged + /// is exported. + void schedulePartition( const String & destination_key, const StoragePtr & destination, ExportTTLVersionedEntry versioned, - PartitionView & view, + const std::vector & parts, + const TTLDescriptions & export_ttls, time_t now, - size_t & in_flight, - TaskStates & task_states, const ContextPtr & context); + + /// Kills the TTL tasks of a destination that is no longer the destination of the TTL, and + /// removes its index once none of them holds a claim. + void cleanupDestination(const String & destination_key, const std::map & entries); }; } diff --git a/src/Storages/MergeTree/ExportTaskInfo.h b/src/Storages/MergeTree/ExportTaskInfo.h index 74c21f4c5db5..b6c4f752a92a 100644 --- a/src/Storages/MergeTree/ExportTaskInfo.h +++ b/src/Storages/MergeTree/ExportTaskInfo.h @@ -65,9 +65,9 @@ struct ExportTaskInfo /// What created the task: `query` for `EXPORT PARTITION`, `ttl` for a `TTL ... EXPORT` expression. String source; - /// TTL export only: earlier tasks that failed to export some of this task's parts, and whose - /// commit may still land. - std::vector retry_of; + /// Id the task commits to the destination under: its transaction id, or for a TTL export task + /// that retries a failed one, the commit id of the failed task. + String commit_id; }; } diff --git a/src/Storages/MergeTree/ExportTaskUtils.cpp b/src/Storages/MergeTree/ExportTaskUtils.cpp index ee40f6f5520a..397319645590 100644 --- a/src/Storages/MergeTree/ExportTaskUtils.cpp +++ b/src/Storages/MergeTree/ExportTaskUtils.cpp @@ -457,7 +457,6 @@ namespace { result.paths.emplace_back(path_in_destination); } - result.paths_by_part[processed_parts[i]] = processed_part_entry.paths_in_destination; } return result; @@ -488,34 +487,6 @@ namespace ops.emplace_back(zkutil::makeCreateRequest(path / "status", "PENDING", zkutil::CreateMode::Persistent)); } - std::vector getRangesCommittedByRetriedTasks( - const ExportRetriedTasks & retry_of, - const StoragePtr & destination_storage, - const String & partition_id, - const ContextPtr & context) - { - std::vector ranges; - for (const auto & retried : retry_of) - { - if (!destination_storage->isExportTransactionCommitted(retried.transaction_id, context)) - continue; - - const auto task_ranges = ExportTTLUtils::fromBlockRanges(partition_id, retried.block_ranges); - ranges.insert(ranges.end(), task_ranges.begin(), task_ranges.end()); - } - return ExportFenceUtils::compactRanges(std::move(ranges)); - } - - std::vector getPartsNotCommitted( - const std::vector & part_names, const std::vector & committed_ranges, MergeTreeDataFormatVersion format_version) - { - std::vector result; - for (const auto & part_name : part_names) - if (!ExportFenceUtils::isCoveredByUnion(MergeTreePartInfo::fromPartName(part_name, format_version), committed_ranges)) - result.push_back(part_name); - return result; - } - void commit( const ExportReplicatedMergeTreeTaskManifest & manifest, const StoragePtr & destination_storage, @@ -572,44 +543,21 @@ namespace const auto partition_id = getPartitionIdOfParts(manifest.parts, source_storage.format_version); - /// A task this one retries may have landed after it was considered failed. Its parts are - /// then in the destination already, so only the files of the other parts are committed. - std::vector paths_to_commit = exported.paths; - if (!manifest.retry_of.empty()) - { - const auto committed_ranges = getRangesCommittedByRetriedTasks( - manifest.retry_of, destination_storage, partition_id, context_in); - - if (!committed_ranges.empty()) - { - paths_to_commit.clear(); - for (const auto & part_name : getPartsNotCommitted(manifest.parts, committed_ranges, source_storage.format_version)) - { - const auto it = exported.paths_by_part.find(part_name); - if (it != exported.paths_by_part.end()) - paths_to_commit.insert(paths_to_commit.end(), it->second.begin(), it->second.end()); - } - - LOG_INFO(log, "Export task: a task retried by {} committed some of its parts, committing {} of {} files", - entry_path, paths_to_commit.size(), exported.paths.size()); - } - } - IStorage::ExportCommitInfo destination_commit_info; - if (paths_to_commit.empty()) + if (exported.paths.empty()) { LOG_INFO(log, "Export task: {} has no destination files to commit", entry_path); } else { destination_commit_info = commitExportOnDestination( - manifest.transaction_id, + manifest.commit_id, partition_id, manifest.iceberg_metadata_json, manifest.write_full_path_in_iceberg_metadata, manifest.iceberg_partition_timezone, - paths_to_commit, + exported.paths, manifest.parts, destination_storage, source_storage, @@ -657,7 +605,7 @@ namespace if (versioned.version < 0) export_ttl_index.ensureDestination(zk, destination_key, fmt::format("{}.{}", manifest.destination_database, manifest.destination_table)); - versioned.entry.commitClaim(manifest.transaction_id, ExportTTLUtils::rangesOfParts(manifest.parts, source_storage.format_version)); + versioned.entry.commitClaim(manifest.commit_id, ExportTTLUtils::rangesOfParts(manifest.parts, source_storage.format_version)); export_ttl_index.appendUpdateEntryOps(ops, destination_key, versioned.entry, versioned.version); } @@ -965,8 +913,8 @@ namespace /// A source partition is not split in the destination when every destination partition expression is /// single-valued over it. That holds structurally when the expression is a deterministic function of the /// source partition key, because rows agreeing on the source key then agree on it as well; the remaining - /// expressions have to be proven from the partition's min/max values. Without `parts`, only checks that - /// such a proof is possible. + /// expressions have to be proven from the partition's min/max values. Without `parts` that have rows, + /// only checks that such a proof is possible. void verifyPartitionKeyCompatibility( const KeyDescription & source_key, const KeyDescription & destination_key, @@ -998,11 +946,18 @@ namespace const auto minmax_column_names = minmax_columns.getNames(); const auto minmax_column_types = minmax_columns.getTypes(); - /// Compute the global min/max index of the parts + /// Compute the global min/max index of the parts. A part without rows, e.g. emptied by a + /// mutation, has no values to place, and its min/max index is not initialized. IMergeTreeDataPart::MinMaxIndex minmax; + bool has_rows = false; for (const auto & part : parts) + { + if (part->rows_count == 0) + continue; minmax.merge(*part->getMinMaxIndex()); - const auto * minmax_of_parts = parts.empty() ? nullptr : &minmax; + has_rows = true; + } + const auto * minmax_of_parts = has_rows ? &minmax : nullptr; /* 1. If there is a structural match between the source and destination key, we accept it diff --git a/src/Storages/MergeTree/ExportTaskUtils.h b/src/Storages/MergeTree/ExportTaskUtils.h index da6ae6d6605c..3ebcd2e24b17 100644 --- a/src/Storages/MergeTree/ExportTaskUtils.h +++ b/src/Storages/MergeTree/ExportTaskUtils.h @@ -3,7 +3,6 @@ #include #include #include -#include #include #include #include @@ -13,7 +12,6 @@ #include #include #include "Storages/IStorage.h" -#include #include #include #include @@ -51,26 +49,11 @@ namespace ExportTaskUtils /// Destination paths recorded by those leaves, flattened. A leaf may carry none, so this /// can legitimately be shorter than `processed_parts_count`. std::vector paths; - - /// The same paths by the name of the part they were exported from. - std::map> paths_by_part; }; /// Appends the ops that create the nodes of a new export task at `task_path` (`/exports/`). void appendCreateExportTaskOps(Coordination::Requests & ops, const std::string & task_path, const ExportReplicatedMergeTreeTaskManifest & manifest); - /// Block ranges of `partition_id` committed to `destination_storage` by the tasks in `retry_of` - /// that landed, e.g. a commit that the destination applied after the task was considered failed. - std::vector getRangesCommittedByRetriedTasks( - const ExportRetriedTasks & retry_of, - const StoragePtr & destination_storage, - const String & partition_id, - const ContextPtr & context); - - /// Parts of `part_names` whose rows are not in `committed_ranges`. - std::vector getPartsNotCommitted( - const std::vector & part_names, const std::vector & committed_ranges, MergeTreeDataFormatVersion format_version); - /// Reads the destination paths recorded under `/processed`. ExportedPaths getExportedPaths(const LoggerPtr & log, const zkutil::ZooKeeperPtr & zk, const std::string & export_path); diff --git a/src/Storages/MergeTree/IExportTTLIndex.h b/src/Storages/MergeTree/IExportTTLIndex.h new file mode 100644 index 000000000000..f494b2542e51 --- /dev/null +++ b/src/Storages/MergeTree/IExportTTLIndex.h @@ -0,0 +1,41 @@ +#pragma once + +#include +#include +#include + +namespace DB +{ + +/// Where a table keeps the export index of its `EXPORT` TTL (see `ExportTTLIndexEntry`): in Keeper +/// for a `ReplicatedMergeTree`, in the data directory for a plain `MergeTree`. +class IExportTTLIndex +{ +public: + virtual ~IExportTTLIndex() = default; + + /// The whole index, read again only if it changed since it was last read. + virtual ExportTTLIndexSnapshotPtr getSnapshot() = 0; + + /// The index as of its last read, without reading it. nullptr if it was not read yet. + virtual ExportTTLIndexSnapshotPtr getLatest() const = 0; + + /// Stores `versioned` if the stored version is still `versioned.version`. Returns false otherwise. + virtual bool updateEntry(const String & destination_key, const ExportTTLVersionedEntry & versioned) = 0; + + virtual void removeDestination(const String & destination_key) = 0; + + /// Only one replica schedules the exports of the `EXPORT` TTL, which keeps the replicas from + /// conflicting on the index; correctness does not depend on it. The lock is kept until it is lost + /// or released. + virtual bool tryAcquireSchedulerLock() = 0; + + /// Released when the table has nothing left to schedule, so that it does not hold Keeper nodes. + virtual void releaseSchedulerLock() = 0; + + /// The replica holding the scheduler lock, empty if there is none or the table is not replicated. + /// Does not read it. + virtual String getSchedulerReplica() const = 0; +}; + +} diff --git a/src/Storages/MergeTree/MergeTreeData.cpp b/src/Storages/MergeTree/MergeTreeData.cpp index 05932a8be6b3..53317876c43f 100644 --- a/src/Storages/MergeTree/MergeTreeData.cpp +++ b/src/Storages/MergeTree/MergeTreeData.cpp @@ -25,6 +25,7 @@ #include #include #include +#include #include #include #include @@ -272,6 +273,7 @@ namespace Setting extern const SettingsBool export_merge_tree_part_throw_on_pending_mutations; extern const SettingsBool export_merge_tree_part_throw_on_pending_patch_parts; extern const SettingsBool allow_insert_into_iceberg; + extern const SettingsExportPartitionAllOnError export_merge_tree_partition_all_on_error; } namespace MergeTreeSetting @@ -413,6 +415,8 @@ namespace ErrorCodes extern const int UNKNOWN_TABLE; extern const int FILE_ALREADY_EXISTS; extern const int PENDING_MUTATIONS_NOT_ALLOWED; + extern const int EXPORT_PARTITION_ALREADY_EXPORTED; + extern const int PARTITION_EXPORT_FAILED; } namespace FailPoints @@ -1288,13 +1292,24 @@ ExportTTLDeleteGate MergeTreeData::getExportTTLDeleteGate() const { ExportTTLDeleteGate gate; const auto metadata = getInMemoryMetadataPtr(nullptr, false); - gate.enabled = metadata->hasAnyExportTTL(); - if (gate.enabled && export_ttl_scheduler) - gate.destination_key = export_ttl_scheduler->getDestinationKey(); + const auto export_ttls = metadata->getExportTTLs(); + gate.enabled = !export_ttls.empty(); + if (gate.enabled) + { + const auto destination = getExportTTLDestination(export_ttls.front()); + gate.destination_key = ExportTTLUtils::destinationKey(destination.database_name, destination.table_name); + } gate.now = time(nullptr); return gate; } +ExportFencePtr MergeTreeData::getLatestExportFence() const +{ + const auto * index = getExportTTLIndex(); + const auto snapshot = index ? index->getLatest() : nullptr; + return snapshot ? snapshot->fence : nullptr; +} + std::vector MergeTreeData::getExportTTLInfo() const { if (!export_ttl_scheduler) @@ -7551,6 +7566,107 @@ void MergeTreeData::killExportPart(const String & transaction_id) }); } +void MergeTreeData::exportPartitionToTable(const PartitionCommand & command, ContextPtr query_context) +{ + if (!query_context->getServerSettings()[ServerSetting::allow_experimental_export_merge_tree_partition]) + throw Exception(ErrorCodes::SUPPORT_IS_DISABLED, + "Exporting merge tree partition is experimental. Set the server setting `allow_experimental_export_merge_tree_partition` to enable it " + "(on all replicas of a replicated table).\n" + "If you are exporting to an Apache Iceberg table, you also need to enable the setting `allow_insert_into_iceberg` on the initiator " + "(query, session or profile) - the export task inherits it."); + + /// EXPORT PARTITION ALL: expand into one sub-call per active partition id. + /// Failure handling is controlled by `export_merge_tree_partition_all_on_error`. + if (const auto * partition_ast = command.partition->as(); partition_ast && partition_ast->all) + { + auto partition_id_set = getAllPartitionIds(); + if (partition_id_set.empty()) + throw Exception(ErrorCodes::BAD_ARGUMENTS, "Table {} has no active partitions to export", getStorageID().getNameForLogs()); + + /// Sort for deterministic ordering (so failure messages and tests are stable). + std::vector partition_ids(partition_id_set.begin(), partition_id_set.end()); + std::sort(partition_ids.begin(), partition_ids.end()); + + const auto & on_error_setting = query_context->getSettingsRef()[Setting::export_merge_tree_partition_all_on_error]; + const ExportPartitionAllOnError on_error = on_error_setting.value; + + LOG_INFO(log, "EXPORT PARTITION ALL: scheduling export for {} partitions, on_error={}", + partition_ids.size(), on_error_setting.toString()); + + std::vector> failures; /// (partition_id, message) + size_t skipped_conflicts = 0; + + for (const auto & partition_id : partition_ids) + { + PartitionCommand sub = command; + auto synthetic = make_intrusive(); + synthetic->setPartitionID(make_intrusive(partition_id)); + sub.partition = synthetic; + + try + { + exportPartitionToTable(sub, query_context); + } + catch (const Exception & e) + { + switch (on_error) + { + case ExportPartitionAllOnError::throw_first: + throw; + case ExportPartitionAllOnError::skip_conflicts: + if (e.code() == ErrorCodes::EXPORT_PARTITION_ALREADY_EXPORTED) + { + ++skipped_conflicts; + LOG_INFO(log, "EXPORT PARTITION ALL: skipping partition {} (already exported): {}", + partition_id, e.message()); + break; + } + throw; + case ExportPartitionAllOnError::collect: + LOG_WARNING(log, "EXPORT PARTITION ALL: partition {} failed: {}", partition_id, e.message()); + failures.emplace_back(partition_id, e.message()); + break; + } + } + } + + if (!failures.empty()) + { + String aggregated = fmt::format( + "EXPORT PARTITION ALL: {}/{} partitions failed to schedule. Per-partition errors:", + failures.size(), partition_ids.size()); + for (const auto & [pid, msg] : failures) + aggregated += fmt::format("\n {}: {}", pid, msg); + throw Exception(ErrorCodes::PARTITION_EXPORT_FAILED, "{}", aggregated); + } + + if (skipped_conflicts > 0) + LOG_INFO(log, "EXPORT PARTITION ALL: skipped {} partitions due to existing exports", skipped_conflicts); + + return; + } + + const auto dest_database = query_context->resolveDatabase(command.to_database); + const auto dest_storage = DatabaseCatalog::instance().getTable({dest_database, command.to_table}, query_context); + if (dest_storage->getStorageID() == getStorageID()) + throw Exception(ErrorCodes::BAD_ARGUMENTS, "Exporting to the same table is not allowed"); + + ExportPartsRequest request; + request.destination = dest_storage; + request.partition_id = getPartitionIDFromQuery(command.partition, query_context); + request.source = ExportTaskSource::query; + request.query_id = query_context->getCurrentQueryId(); + { + auto data_parts_lock = lockParts(); + request.parts = getDataPartsVectorInPartitionForInternalUsage(MergeTreeDataPartState::Active, request.partition_id, data_parts_lock); + } + + if (request.parts.empty()) + throw Exception(ErrorCodes::BAD_ARGUMENTS, "Partition {} doesn't exist", request.partition_id); + + exportParts(request, query_context); +} + void MergeTreeData::movePartitionToShard(const ASTPtr & /*partition*/, bool /*move_part*/, const String & /*to*/, ContextPtr /*query_context*/) { throw Exception(ErrorCodes::NOT_IMPLEMENTED, "MOVE PARTITION TO SHARD is not supported by storage {}", getName()); diff --git a/src/Storages/MergeTree/MergeTreeData.h b/src/Storages/MergeTree/MergeTreeData.h index 15dfee1ad92b..5b30b349c540 100644 --- a/src/Storages/MergeTree/MergeTreeData.h +++ b/src/Storages/MergeTree/MergeTreeData.h @@ -18,6 +18,8 @@ #include #include #include +#include +#include #include #include #include @@ -54,6 +56,7 @@ namespace DB class ExportTTLScheduler; struct ExportTTLPartitionInfo; +class IExportTTLIndex; /// Number of streams is not number parts, but number or parts*files, hence 100. const size_t DEFAULT_DELAYED_STREAMS_FOR_PARALLEL_WRITE = 100; @@ -1111,10 +1114,36 @@ class MergeTreeData : public WithMutableContext, public IStorage, public IBackgr void killExportPart(const String & transaction_id); - virtual void exportPartitionToTable(const PartitionCommand &, ContextPtr) + /// An export of parts of one partition, by `EXPORT PARTITION` or by the `EXPORT` TTL. + struct ExportPartsRequest { - throw Exception(ErrorCodes::NOT_IMPLEMENTED, "EXPORT PARTITION is not implemented for engine {}", getName()); - } + StoragePtr destination; + String partition_id; + /// Parts without rows are exported as no files. + DataPartsVector parts; + ExportTaskSource source = ExportTaskSource::query; + /// Generated if empty. + String transaction_id; + /// The id the task commits to the destination under, the transaction id if empty. A retry of a + /// task of the `EXPORT` TTL commits under the id of the task it retries. + String commit_id; + String query_id; + }; + + /// The export index entry of the `EXPORT` TTL to store with an export. + struct ExportTTLIndexUpdate + { + String destination_key; + ExportTTLVersionedEntry entry; + }; + + /// Checks the export and creates its task. With `index_update`, also + /// stores that entry in the same step and fails if a part is being merged. Returns false on a conflict. + virtual bool exportParts(const ExportPartsRequest & request, ContextPtr local_context, const ExportTTLIndexUpdate * index_update = nullptr) = 0; + + /// `ALTER TABLE ... EXPORT PARTITION`, including `EXPORT PARTITION ALL`: every call is a new task, + /// even if the partition was exported before. + void exportPartitionToTable(const PartitionCommand & command, ContextPtr query_context); /// Snapshot of this table's partition-export tasks for `system.distributed_exports`, taken from /// an in-memory mirror: no disk or ZooKeeper I/O, so it is safe to call from query threads. @@ -1786,13 +1815,35 @@ class MergeTreeData : public WithMutableContext, public IStorage, public IBackgr ExportTTLDeleteGate getExportTTLDeleteGate() const; + /// The export index of the `EXPORT` TTL, nullptr if the table has none. + virtual IExportTTLIndex * getExportTTLIndex() const = 0; + /// The export states of parts as last seen, without reading Keeper. May be older than the /// index, which only makes the delete gate hold more parts. - virtual ExportFencePtr getLatestExportFence() const { return nullptr; } + ExportFencePtr getLatestExportFence() const; /// For `system.ttl_exports`, empty if the table has no `EXPORT` TTL scheduler. std::vector getExportTTLInfo() const; + /// The status of an export task of the table, as the `EXPORT` TTL resolves the claim of the task. + enum class ExportTaskStatus : uint8_t + { + PENDING, + COMPLETED, + FAILED, + KILLED, + }; + + /// Nullopt if there is no such task, e.g. after a crash between claiming its parts and creating it. + virtual std::optional getExportTaskStatus(const String & transaction_id) const = 0; + + /// As `getExportTaskStatus`, from what this replica knows without reading Keeper: also nullopt for + /// a task that it does not know yet. + virtual std::optional getKnownExportTaskStatus(const String & transaction_id) const = 0; + + /// Whether the part is a source of an assigned merge that changes its block range. + virtual bool isPartBeingMerged(const MergeTreePartInfo & part_info) const = 0; + protected: /// Created by the engines that support `TTL ... EXPORT`. std::shared_ptr export_ttl_scheduler; diff --git a/src/Storages/MergeTree/MergeTreeExportTTLScheduler.cpp b/src/Storages/MergeTree/MergeTreeExportTTLIndex.cpp similarity index 61% rename from src/Storages/MergeTree/MergeTreeExportTTLScheduler.cpp rename to src/Storages/MergeTree/MergeTreeExportTTLIndex.cpp index 9255b03e7842..976041b42e51 100644 --- a/src/Storages/MergeTree/MergeTreeExportTTLScheduler.cpp +++ b/src/Storages/MergeTree/MergeTreeExportTTLIndex.cpp @@ -1,4 +1,4 @@ -#include +#include #include #include @@ -7,15 +7,12 @@ #include #include #include -#include -#include #include #include #include #include #include -#include namespace fs = std::filesystem; @@ -110,7 +107,7 @@ void MergeTreeExportTTLIndex::load() publishSnapshot(); } -bool MergeTreeExportTTLIndex::update(const String & destination_key, const ExportTTLVersionedEntry & versioned) +bool MergeTreeExportTTLIndex::updateEntry(const String & destination_key, const ExportTTLVersionedEntry & versioned) { std::lock_guard lock(mutex); @@ -170,82 +167,4 @@ void MergeTreeExportTTLIndex::publishSnapshot() snapshot.set(ExportTTLIndexSnapshot::build(/* version */ -1, entries)); } -MergeTreeExportTTLScheduler::MergeTreeExportTTLScheduler(StorageMergeTree & storage_, MergeTreeExportTTLIndex & index_) - : ExportTTLScheduler(storage_) - , plain_storage(storage_) - , index(index_) -{ -} - -bool MergeTreeExportTTLScheduler::isPaused() -{ - return plain_storage.parts_mover.moves_blocker.isCancelled(); -} - -ExportTTLScheduler::TaskState MergeTreeExportTTLScheduler::getTaskState(const String & transaction_id) -{ - TaskState state; - const auto task = plain_storage.export_task_scheduler->getTask(transaction_id); - if (!task) - return state; - - switch (task->status) - { - case MergeTreeExportTask::Status::PENDING: state.status = TaskStatus::PENDING; break; - case MergeTreeExportTask::Status::COMPLETED: state.status = TaskStatus::COMPLETED; break; - case MergeTreeExportTask::Status::FAILED: state.status = TaskStatus::FAILED; break; - case MergeTreeExportTask::Status::KILLED: state.status = TaskStatus::KILLED; break; - } - state.reached_commit = task->allPartsDone(); - state.retry_of = task->retry_of; - return state; -} - -bool MergeTreeExportTTLScheduler::startGroup(const GroupToStart & group, const ContextPtr & context) -{ - const auto parts_with_rows = group.partsWithRows(); - std::optional descriptor; - if (!parts_with_rows.empty()) - { - const auto source_metadata = plain_storage.getInMemoryMetadataPtr(context, false); - const auto destination_metadata = group.destination->getInMemoryMetadataPtr(context, false); - ExportTaskUtils::verifyExportSchemaCastable(source_metadata, destination_metadata, group.destination->getStorageID(), context); - - MergeTreeData::DataPartsVector parts(parts_with_rows.begin(), parts_with_rows.end()); - descriptor = plain_storage.buildExportTask( - group.destination->getStorageID(), group.destination, source_metadata, destination_metadata, parts, group.partition_id, context); - descriptor->transaction_id = group.transaction_id; - descriptor->source = ExportTaskSource::ttl; - descriptor->retry_of = group.retry_of; - } - - { - /// Merge selection holds it while it checks the fence and tags the parts it merges. - std::lock_guard lock(plain_storage.currently_processing_in_background_mutex); - for (const auto & part : group.parts) - { - if (part->getState() != MergeTreeDataPartState::Active || plain_storage.currently_merging_mutating_parts.contains(part->info)) - return false; - } - - if (!index.update(group.destination_key, group.entry)) - return false; - } - - if (descriptor) - plain_storage.export_task_scheduler->addTask(std::move(*descriptor), parts_with_rows); - return true; -} - -bool MergeTreeExportTTLScheduler::isPartBeingMerged(const MergeTreeDataPartPtr & part) -{ - std::lock_guard lock(plain_storage.currently_processing_in_background_mutex); - return plain_storage.currently_merging_mutating_parts.contains(part->info); -} - -void MergeTreeExportTTLScheduler::killTask(const String & transaction_id) -{ - plain_storage.export_task_scheduler->kill(transaction_id); -} - } diff --git a/src/Storages/MergeTree/MergeTreeExportTTLIndex.h b/src/Storages/MergeTree/MergeTreeExportTTLIndex.h new file mode 100644 index 000000000000..991709ee53e2 --- /dev/null +++ b/src/Storages/MergeTree/MergeTreeExportTTLIndex.h @@ -0,0 +1,52 @@ +#pragma once + +#include +#include + +#include +#include + +namespace DB +{ + +class StorageMergeTree; + +/// The export index of the `EXPORT` TTL of a plain `MergeTree`, in the data directory of the table: +/// `export_ttl//partitions/.json`, each file written atomically. +/// Versions are kept in memory only, for the compare-and-set of the scheduler. There is one +/// scheduler, so its lock is always held. +class MergeTreeExportTTLIndex final : public IExportTTLIndex +{ +public: + explicit MergeTreeExportTTLIndex(StorageMergeTree & storage_); + + void load(); + + ExportTTLIndexSnapshotPtr getSnapshot() override { return snapshot.get(); } + ExportTTLIndexSnapshotPtr getLatest() const override { return snapshot.get(); } + bool updateEntry(const String & destination_key, const ExportTTLVersionedEntry & versioned) override; + void removeDestination(const String & destination_key) override; + bool tryAcquireSchedulerLock() override { return true; } + void releaseSchedulerLock() override {} + String getSchedulerReplica() const override { return {}; } + + /// Highest block number in any entry, 0 if there is none. + Int64 maxBlock() const; + +private: + StorageMergeTree & storage; + + mutable std::mutex mutex; + /// By destination key, then by partition id. + std::map> entries; + MultiVersion snapshot; + + String getRootPath() const; + String getDestinationPath(const String & destination_key) const; + String getEntryPath(const String & destination_key, const String & partition_id) const; + + /// Caller must hold `mutex`. + void publishSnapshot(); +}; + +} diff --git a/src/Storages/MergeTree/MergeTreeExportTTLScheduler.h b/src/Storages/MergeTree/MergeTreeExportTTLScheduler.h deleted file mode 100644 index a59d8e6a67c2..000000000000 --- a/src/Storages/MergeTree/MergeTreeExportTTLScheduler.h +++ /dev/null @@ -1,79 +0,0 @@ -#pragma once - -#include -#include - -#include -#include - -namespace DB -{ - -class StorageMergeTree; - -/// The export index of the `EXPORT` TTL of a plain `MergeTree`, in the data directory of the table: -/// `export_ttl//partitions/.json`, each file written atomically. -/// Versions are kept in memory only, for the compare-and-set of the scheduler. -class MergeTreeExportTTLIndex -{ -public: - explicit MergeTreeExportTTLIndex(StorageMergeTree & storage_); - - void load(); - - /// Stores `entry` if the stored version is still `entry.version`. Returns false otherwise. - bool update(const String & destination_key, const ExportTTLVersionedEntry & entry); - - void removeDestination(const String & destination_key); - - ExportTTLIndexSnapshotPtr getSnapshot() const { return snapshot.get(); } - ExportFencePtr getFence() const { return snapshot.get()->fence; } - - /// Highest block number in any entry, 0 if there is none. - Int64 maxBlock() const; - -private: - StorageMergeTree & storage; - - mutable std::mutex mutex; - /// By destination key, then by partition id. - std::map> entries; - MultiVersion snapshot; - - String getRootPath() const; - String getDestinationPath(const String & destination_key) const; - String getEntryPath(const String & destination_key, const String & partition_id) const; - - /// Caller must hold `mutex`. - void publishSnapshot(); -}; - -/// `ExportTTLScheduler` of a plain `MergeTree`. A group is claimed in the index before its task is -/// created, under the lock that merge selection holds, so no merge of its parts can be selected -/// meanwhile. A crash in between leaves a claim without a task, which the next tick retries. -class MergeTreeExportTTLScheduler final : public ExportTTLScheduler -{ -public: - MergeTreeExportTTLScheduler(StorageMergeTree & storage_, MergeTreeExportTTLIndex & index_); - -protected: - bool acquireSchedulerLock() override { return true; } - bool isPaused() override; - ExportTTLIndexSnapshotPtr getIndexSnapshot() override { return index.getSnapshot(); } - String getReplicaName() const override { return {}; } - String getSchedulerReplica() override { return {}; } - TaskState getTaskState(const String & transaction_id) override; - /// A task is not marked failed or killed while it commits, and a restart ends every commit. - bool isCommitInProgress(const String &) override { return false; } - bool updateIndexEntry(const String & destination_key, const ExportTTLVersionedEntry & entry) override { return index.update(destination_key, entry); } - bool startGroup(const GroupToStart & group, const ContextPtr & context) override; - bool isPartBeingMerged(const MergeTreeDataPartPtr & part) override; - void killTask(const String & transaction_id) override; - void removeDestination(const String & destination_key) override { index.removeDestination(destination_key); } - -private: - StorageMergeTree & plain_storage; - MergeTreeExportTTLIndex & index; -}; - -} diff --git a/src/Storages/MergeTree/MergeTreeExportTask.h b/src/Storages/MergeTree/MergeTreeExportTask.h index 2ebb501a226d..d26e65b9021a 100644 --- a/src/Storages/MergeTree/MergeTreeExportTask.h +++ b/src/Storages/MergeTree/MergeTreeExportTask.h @@ -9,7 +9,6 @@ #include #include #include -#include #include #include @@ -61,9 +60,9 @@ struct MergeTreeExportTask String destination_table; time_t create_time = 0; ExportTaskSource source = ExportTaskSource::query; - /// TTL tasks whose parts this task exports again because they failed, and whose commits may still - /// land, which the commit of this task checks. - ExportRetriedTasks retry_of; + /// Id the task commits to the destination under: its transaction id, or for a task of the `EXPORT` + /// TTL that retries a failed one, the commit id of the failed task. + String commit_id; /// Work + progress std::vector parts; @@ -157,8 +156,7 @@ struct MergeTreeExportTask json.set("destination_table", destination_table); json.set("create_time", create_time); json.set("source", String(magic_enum::enum_name(source))); - if (!retry_of.empty()) - json.set("retry_of", ExportRetriedTaskUtils::toJSON(retry_of)); + json.set("commit_id", commit_id); json.set("status", String(magic_enum::enum_name(status))); Poco::JSON::Array::Ptr parts_array = new Poco::JSON::Array(); @@ -243,7 +241,8 @@ struct MergeTreeExportTask throw Exception(ErrorCodes::INCORRECT_DATA, "Unknown source '{}' in export task descriptor", source_str); } - task.retry_of = ExportRetriedTaskUtils::fromJSON(json->getArray("retry_of")); + /// Descriptors written before commit ids existed commit under their transaction id. + task.commit_id = json->optValue("commit_id", task.transaction_id); const auto status_str = json->getValue("status"); if (const auto status = magic_enum::enum_cast(status_str)) diff --git a/src/Storages/MergeTree/MergeTreeExportTaskScheduler.cpp b/src/Storages/MergeTree/MergeTreeExportTaskScheduler.cpp index 738ae015dc4b..e976eb0f915c 100644 --- a/src/Storages/MergeTree/MergeTreeExportTaskScheduler.cpp +++ b/src/Storages/MergeTree/MergeTreeExportTaskScheduler.cpp @@ -20,7 +20,6 @@ #include #include #include -#include #include @@ -143,7 +142,7 @@ std::vector MergeTreeExportTaskScheduler::getInfo() const ? "" : ExportTaskUtils::getPartitionIdOfParts(descriptor.partNames(), storage.format_version); info.source = String(magic_enum::enum_name(descriptor.source)); - info.retry_of = ExportRetriedTaskUtils::transactionIds(descriptor.retry_of); + info.commit_id = descriptor.commit_id; info.transaction_id = descriptor.transaction_id; info.query_id = descriptor.query_id; info.parts = descriptor.partNames(); @@ -568,47 +567,7 @@ void MergeTreeExportTaskScheduler::tryCommit(const String & transaction_id) IStorage::ExportCommitInfo destination_commit_info; try { - auto exported_paths = descriptor_copy.collectExportedPaths(); - - std::optional context; - StoragePtr destination_storage; - const auto get_destination = [&] - { - if (destination_storage) - return; - destination_storage = DatabaseCatalog::instance().tryGetTable(destination_storage_id, storage.getContext()); - if (!destination_storage) - throw Exception(ErrorCodes::UNKNOWN_TABLE, "Destination table {} not found for export commit", - destination_storage_id.getNameForLogs()); - context = ExportTaskUtils::getContextCopyWithTaskSettings(storage.getContext(), descriptor_copy); - }; - - /// A task this one retries may have landed after it was considered failed. Its parts are - /// then in the destination already, so only the files of the other parts are committed. - if (!descriptor_copy.retry_of.empty()) - { - get_destination(); - const auto committed_ranges = ExportTaskUtils::getRangesCommittedByRetriedTasks( - descriptor_copy.retry_of, - destination_storage, - ExportTaskUtils::getPartitionIdOfParts(descriptor_copy.partNames(), storage.format_version), - *context); - - if (!committed_ranges.empty()) - { - const auto parts_to_commit = ExportTaskUtils::getPartsNotCommitted( - descriptor_copy.partNames(), committed_ranges, storage.format_version); - const std::unordered_set parts_to_commit_set(parts_to_commit.begin(), parts_to_commit.end()); - - exported_paths.clear(); - for (const auto & part : descriptor_copy.parts) - if (parts_to_commit_set.contains(part.part_name)) - exported_paths.insert(exported_paths.end(), part.paths_in_destination.begin(), part.paths_in_destination.end()); - - LOG_INFO(storage.log, "Export task: a task retried by {} committed some of its parts, committing the files of {} of {} parts", - transaction_id, parts_to_commit.size(), descriptor_copy.parts.size()); - } - } + const auto exported_paths = descriptor_copy.collectExportedPaths(); /// Every part is done and its paths were recorded durably at completion, so an empty set /// here is not a lost-data symptom: an Iceberg destination writes no data file for a part @@ -621,12 +580,16 @@ void MergeTreeExportTaskScheduler::tryCommit(const String & transaction_id) } else { - get_destination(); + const auto destination_storage = DatabaseCatalog::instance().tryGetTable(destination_storage_id, storage.getContext()); + if (!destination_storage) + throw Exception(ErrorCodes::UNKNOWN_TABLE, "Destination table {} not found for export commit", + destination_storage_id.getNameForLogs()); + const auto context = ExportTaskUtils::getContextCopyWithTaskSettings(storage.getContext(), descriptor_copy); LOG_INFO(storage.log, "Export task: all parts exported for task {}, committing", transaction_id); destination_commit_info = ExportTaskUtils::commitExportOnDestination( - descriptor_copy.transaction_id, + descriptor_copy.commit_id, ExportTaskUtils::getPartitionIdOfParts(descriptor_copy.partNames(), storage.format_version), descriptor_copy.iceberg_metadata_json, descriptor_copy.write_full_path_in_iceberg_metadata, @@ -635,7 +598,7 @@ void MergeTreeExportTaskScheduler::tryCommit(const String & transaction_id) descriptor_copy.partNames(), destination_storage, storage, - *context); + context); } success = true; diff --git a/src/Storages/MergeTree/MergeTreeSettings.cpp b/src/Storages/MergeTree/MergeTreeSettings.cpp index f64246c70a91..f860859b8219 100644 --- a/src/Storages/MergeTree/MergeTreeSettings.cpp +++ b/src/Storages/MergeTree/MergeTreeSettings.cpp @@ -820,32 +820,10 @@ Possible values: - [load_existing_rows_count_for_old_parts](#load_existing_rows_count_for_old_parts) setting )", 0) \ - DECLARE(UInt64, ttl_export_check_period_seconds, 10, R"( - How often the `TTL ... EXPORT TO TABLE` scheduler of the table looks for parts to export, in seconds. - )", EXPERIMENTAL) \ - DECLARE(UInt64, ttl_export_batch_window_seconds, 60, R"( - The eligible parts of a partition are exported once no new eligible part appeared for this many - seconds, so that parts which become eligible together are exported together. - )", EXPERIMENTAL) \ - DECLARE(UInt64, ttl_export_batch_max_delay_seconds, 600, R"( - The eligible parts of a partition are exported at the latest this many seconds after the first of - them was seen, even if new eligible parts keep appearing. - )", EXPERIMENTAL) \ - DECLARE(UInt64, ttl_export_batch_min_bytes, 256_MiB, R"( - The eligible parts of a partition are exported as soon as their size on disk reaches this many - bytes. 0 disables the threshold. - )", EXPERIMENTAL) \ - DECLARE(UInt64, ttl_export_max_parts_per_group, 100, R"( - Maximum number of parts exported together by one task of the `EXPORT` TTL. Parts of a failed task - are always retried together, even if there are more of them. - )", EXPERIMENTAL) \ - DECLARE(UInt64, ttl_export_max_bytes_per_group, 100_GiB, R"( - Maximum size on disk of the parts exported together by one task of the `EXPORT` TTL. A single part - bigger than this is exported on its own. 0 means unlimited. - )", EXPERIMENTAL) \ - DECLARE(UInt64, ttl_export_max_concurrent_groups, 4, R"( - Maximum number of tasks of the `EXPORT` TTL of the table that run at the same time. There is at - most one per partition. + DECLARE(UInt64, ttl_export_check_period_seconds, 60, R"( + How often the `TTL ... EXPORT TO TABLE` scheduler of the table exports the eligible parts, in + seconds. Each check exports the eligible parts of a partition together, unless an export task of the + partition is still running, so this is the interval of the batches. )", EXPERIMENTAL) \ DECLARE(String, ttl_export_settings_profile, "", R"( Settings profile whose settings the tasks of the `EXPORT` TTL run with, e.g. the output format diff --git a/src/Storages/MergeTree/ReplicatedExportTTLIndex.cpp b/src/Storages/MergeTree/ReplicatedExportTTLIndex.cpp index 4236247a0837..f162897ac93b 100644 --- a/src/Storages/MergeTree/ReplicatedExportTTLIndex.cpp +++ b/src/Storages/MergeTree/ReplicatedExportTTLIndex.cpp @@ -17,13 +17,121 @@ namespace ProfileEvents namespace DB { -ReplicatedExportTTLIndex::ReplicatedExportTTLIndex(String zookeeper_path_, LoggerPtr log_) +namespace ErrorCodes +{ + extern const int NO_ZOOKEEPER; +} + +ReplicatedExportTTLIndex::ReplicatedExportTTLIndex( + String zookeeper_path_, + String replica_name_, + std::function get_zookeeper_, + std::function is_readonly_, + LoggerPtr log_) : zookeeper_path(std::move(zookeeper_path_)) + , replica_name(std::move(replica_name_)) + , get_zookeeper(std::move(get_zookeeper_)) + , is_readonly(std::move(is_readonly_)) , log(std::move(log_)) , latest(ExportTTLIndexSnapshot::build(/* version */ -1, {})) { } +ReplicatedExportTTLIndex::~ReplicatedExportTTLIndex() +{ + releaseSchedulerLock(); +} + +zkutil::ZooKeeperPtr ReplicatedExportTTLIndex::getZooKeeper() const +{ + auto zookeeper = get_zookeeper(); + if (!zookeeper) + throw Exception(ErrorCodes::NO_ZOOKEEPER, "Cannot get ZooKeeper"); + return zookeeper; +} + +void ReplicatedExportTTLIndex::releaseSchedulerLock() +{ + std::lock_guard lock(lock_mutex); + lock_holder.reset(); + lock_zookeeper.reset(); +} + +bool ReplicatedExportTTLIndex::tryAcquireSchedulerLock() +{ + auto zookeeper = get_zookeeper(); + if (!zookeeper || zookeeper->expired() || is_readonly()) + { + releaseSchedulerLock(); + return false; + } + + std::lock_guard lock(lock_mutex); + if (lock_holder && lock_zookeeper == zookeeper) + return true; + + lock_holder.reset(); + lock_zookeeper.reset(); + + const auto root_path = getRootPath(); + if (const auto code = zookeeper->tryCreate(root_path, "", zkutil::CreateMode::Persistent); + code != Coordination::Error::ZOK && code != Coordination::Error::ZNODEEXISTS) + throw zkutil::KeeperException::fromPath(code, root_path); + + lock_holder = zkutil::EphemeralNodeHolder::tryCreate(getSchedulerLockPath(), *zookeeper, replica_name); + if (!lock_holder) + return false; + + lock_zookeeper = zookeeper; + LOG_INFO(log, "This replica schedules the EXPORT TTL of the table"); + return true; +} + +String ReplicatedExportTTLIndex::getSchedulerReplica() const +{ + { + std::lock_guard lock(lock_mutex); + if (lock_holder && !lock_zookeeper->expired()) + return replica_name; + } + + std::lock_guard lock(scheduler_replica_mutex); + return scheduler_replica; +} + +void ReplicatedExportTTLIndex::refreshAndWatch( + const zkutil::ZooKeeperPtr & zookeeper, const Coordination::WatchCallbackPtr & watch, bool has_export_ttl) +{ + if (!has_export_ttl) + { + Coordination::Stat stat; + const int32_t version = zookeeper->exists(getVersionPath(), &stat) ? stat.version : -1; + const auto snapshot = refresh(zookeeper, version); + loaded = true; + if (snapshot->entries.empty()) + { + std::lock_guard lock(scheduler_replica_mutex); + scheduler_replica.clear(); + return; + } + } + + /// The watches are set before the nodes are read, so a change after a read always calls this again. + String replica; + if (zookeeper->existsWatch(getSchedulerLockPath(), nullptr, watch)) + zookeeper->tryGet(getSchedulerLockPath(), replica); + { + std::lock_guard lock(scheduler_replica_mutex); + scheduler_replica = std::move(replica); + } + + /// Without the `version` node there is no index yet: it is created before the first entry. + Coordination::Stat stat; + const int32_t version = zookeeper->existsWatch(getVersionPath(), &stat, watch) ? stat.version : -1; + refresh(zookeeper, version); + loaded = true; +} + String ReplicatedExportTTLIndex::getRootPath() const { return fs::path(zookeeper_path) / "export_ttl"; @@ -164,19 +272,49 @@ void ReplicatedExportTTLIndex::appendUpdateEntryOps( ops.emplace_back(zkutil::makeSetRequest(getVersionPath(), "", -1)); } -void ReplicatedExportTTLIndex::removeDestination(const zkutil::ZooKeeperPtr & zookeeper, const String & destination_key) const +bool ReplicatedExportTTLIndex::updateEntry(const String & destination_key, const ExportTTLVersionedEntry & versioned) +{ + const auto zookeeper = getZooKeeper(); + if (versioned.version < 0) + ensureDestination(zookeeper, destination_key, destination_key); + + Coordination::Requests ops; + appendUpdateEntryOps(ops, destination_key, versioned.entry, versioned.version); + + Coordination::Responses responses; + const auto code = zookeeper->tryMulti(ops, responses); + if (code == Coordination::Error::ZOK) + return true; + if (code == Coordination::Error::ZBADVERSION || code == Coordination::Error::ZNODEEXISTS || code == Coordination::Error::ZNONODE) + return false; + zkutil::KeeperMultiException::check(code, ops, responses); + return false; +} + +void ReplicatedExportTTLIndex::removeDestination(const String & destination_key) { + const auto zookeeper = getZooKeeper(); zookeeper->tryRemoveRecursive(getDestinationPath(destination_key)); zookeeper->trySet(getVersionPath(), "", -1); LOG_INFO(log, "Removed the TTL export index of destination {}", destination_key); } +ExportTTLIndexSnapshotPtr ReplicatedExportTTLIndex::getSnapshot() +{ + return getSnapshot(getZooKeeper()); +} + ExportTTLIndexSnapshotPtr ReplicatedExportTTLIndex::getSnapshot(const zkutil::ZooKeeperPtr & zookeeper) { - /// Read before the index, so the index is at least as new as the version it is cached for. Every - /// change of the index bumps the version in the same transaction. - const auto version = readVersion(zookeeper); + auto snapshot = refresh(zookeeper, readVersion(zookeeper)); + loaded = true; + return snapshot; +} +ExportTTLIndexSnapshotPtr ReplicatedExportTTLIndex::refresh(const zkutil::ZooKeeperPtr & zookeeper, int32_t version) +{ + /// `version` is read before the index, so the index is at least as new as the version it is cached + /// for. Every change of the index bumps the version in the same transaction. if (auto cached = latest.get(); cached->version == version) return cached; diff --git a/src/Storages/MergeTree/ReplicatedExportTTLIndex.h b/src/Storages/MergeTree/ReplicatedExportTTLIndex.h index 44c1942ec747..8e40cc06d9a1 100644 --- a/src/Storages/MergeTree/ReplicatedExportTTLIndex.h +++ b/src/Storages/MergeTree/ReplicatedExportTTLIndex.h @@ -5,8 +5,11 @@ #include #include #include +#include #include +#include +#include #include #include #include @@ -24,10 +27,42 @@ namespace DB /// - `scheduler_lock`: ephemeral, held by the replica that schedules TTL exports, whose name it holds; /// - `destinations/`: holds the name of one destination; /// - `destinations//partitions/`: an `ExportTTLIndexEntry`. -class ReplicatedExportTTLIndex +/// +/// Every replica of a table with an `EXPORT` TTL, or with an index left to clean up, watches `version` +/// and `scheduler_lock`, so that its copy of the index and of the lock holder follow Keeper, and +/// `system.ttl_exports` reads neither from Keeper. +class ReplicatedExportTTLIndex final : public IExportTTLIndex { public: - ReplicatedExportTTLIndex(String zookeeper_path_, LoggerPtr log_); + /// `get_zookeeper` returns the current session of the table, or nullptr if there is none. A + /// replica does not schedule while `is_readonly`. + ReplicatedExportTTLIndex( + String zookeeper_path_, + String replica_name_, + std::function get_zookeeper_, + std::function is_readonly_, + LoggerPtr log_); + ~ReplicatedExportTTLIndex() override; + + /// Reads the index again only when the version of the `version` node changed, so while nothing + /// changes this costs one `exists`. + ExportTTLIndexSnapshotPtr getSnapshot() override; + ExportTTLIndexSnapshotPtr getLatest() const override { return loaded ? latest.get() : nullptr; } + bool updateEntry(const String & destination_key, const ExportTTLVersionedEntry & versioned) override; + /// Not transactional: an interrupted removal leaves fewer entries, which only lifts more of the fence. + void removeDestination(const String & destination_key) override; + /// Needs a session that is not expired, and holds the lock while that session lives and the + /// replica is not read-only. + bool tryAcquireSchedulerLock() override; + /// Also e.g. when the replica goes read-only, so another replica can take over. + void releaseSchedulerLock() override; + String getSchedulerReplica() const override; + + /// Reads the index again if it changed, and the holder of the scheduler lock, with `watch` on the + /// nodes they are read from, so that it is called again when they change. Creates no node. + /// Without `has_export_ttl`, an empty index is read without watches: the table has nothing to + /// follow until an `EXPORT` TTL is added, which calls this again. + void refreshAndWatch(const zkutil::ZooKeeperPtr & zookeeper, const Coordination::WatchCallbackPtr & watch, bool has_export_ttl); String getRootPath() const; String getVersionPath() const; @@ -51,30 +86,39 @@ class ReplicatedExportTTLIndex void appendUpdateEntryOps( Coordination::Requests & ops, const String & destination_key, const ExportTTLIndexEntry & entry, int32_t version) const; - /// Removes the index of `destination_key` and bumps the version of the index. Not transactional: - /// an interrupted removal leaves fewer entries, which only lifts more of the fence. - void removeDestination(const zkutil::ZooKeeperPtr & zookeeper, const String & destination_key) const; - - /// The whole index as of the current version of the `version` node. It is read again only when - /// that version changed, so while nothing changes this costs one `exists`. ExportTTLIndexSnapshotPtr getSnapshot(const zkutil::ZooKeeperPtr & zookeeper); /// The export states as of the returned version of the `version` node. std::pair getForMergeAssignment(const zkutil::ZooKeeperPtr & zookeeper); - /// The export states read by the last `getSnapshot`, without reading Keeper. - ExportFencePtr getLatest() const { return latest.get()->fence; } - private: const String zookeeper_path; + const String replica_name; + const std::function get_zookeeper; + const std::function is_readonly; const LoggerPtr log; /// Serializes reading the index again, which readers of `latest` do not wait for. std::mutex refresh_mutex; MultiVersion latest; + /// Whether `latest` was read from Keeper at least once. + std::atomic loaded = false; + + mutable std::mutex scheduler_replica_mutex; + /// The holder of the scheduler lock as of the last `refreshAndWatch`. + String scheduler_replica; + + mutable std::mutex lock_mutex; + /// Keeps the session the lock was created in alive: the holder refers to it. + zkutil::ZooKeeperPtr lock_zookeeper; + zkutil::EphemeralNodeHolderPtr lock_holder; String getDestinationsPath() const; int32_t readVersion(const zkutil::ZooKeeperPtr & zookeeper) const; + zkutil::ZooKeeperPtr getZooKeeper() const; + + /// The index as of `version` of the `version` node, read again unless it is cached for it. + ExportTTLIndexSnapshotPtr refresh(const zkutil::ZooKeeperPtr & zookeeper, int32_t version); }; using ReplicatedExportTTLIndexPtr = std::shared_ptr; diff --git a/src/Storages/MergeTree/ReplicatedExportTTLScheduler.cpp b/src/Storages/MergeTree/ReplicatedExportTTLScheduler.cpp deleted file mode 100644 index 0e9a9393f7cc..000000000000 --- a/src/Storages/MergeTree/ReplicatedExportTTLScheduler.cpp +++ /dev/null @@ -1,292 +0,0 @@ -#include - -#include -#include -#include -#include -#include -#include -#include -#include -#include - -#include - -namespace fs = std::filesystem; - -namespace ProfileEvents -{ - extern const Event ExportTaskZooKeeperRequests; - extern const Event ExportTaskZooKeeperMulti; -} - -namespace DB -{ - -ReplicatedExportTTLScheduler::ReplicatedExportTTLScheduler(StorageReplicatedMergeTree & storage_) - : ExportTTLScheduler(storage_) - , replicated_storage(storage_) -{ -} - -ReplicatedExportTTLScheduler::~ReplicatedExportTTLScheduler() -{ - releaseSchedulerLock(); -} - -void ReplicatedExportTTLScheduler::releaseSchedulerLock() -{ - std::lock_guard lock(lock_mutex); - lock_holder.reset(); - lock_zookeeper.reset(); -} - -bool ReplicatedExportTTLScheduler::acquireSchedulerLock() -{ - auto zookeeper = replicated_storage.tryGetZooKeeper(); - if (!zookeeper || zookeeper->expired() || replicated_storage.is_readonly) - { - releaseSchedulerLock(); - return false; - } - - std::lock_guard lock(lock_mutex); - if (lock_holder && lock_zookeeper == zookeeper) - return true; - - lock_holder.reset(); - lock_zookeeper.reset(); - - const auto & export_ttl_index = *replicated_storage.export_ttl_index; - const auto root_path = export_ttl_index.getRootPath(); - if (const auto code = zookeeper->tryCreate(root_path, "", zkutil::CreateMode::Persistent); - code != Coordination::Error::ZOK && code != Coordination::Error::ZNODEEXISTS) - throw zkutil::KeeperException::fromPath(code, root_path); - - lock_holder = zkutil::EphemeralNodeHolder::tryCreate(export_ttl_index.getSchedulerLockPath(), *zookeeper, replicated_storage.getReplicaName()); - if (!lock_holder) - return false; - - lock_zookeeper = zookeeper; - LOG_INFO(log, "This replica schedules the EXPORT TTL of the table"); - return true; -} - -bool ReplicatedExportTTLScheduler::isPaused() -{ - return replicated_storage.parts_mover.moves_blocker.isCancelled(); -} - -ExportTTLIndexSnapshotPtr ReplicatedExportTTLScheduler::getIndexSnapshot() -{ - return replicated_storage.export_ttl_index->getSnapshot(replicated_storage.getZooKeeper()); -} - -String ReplicatedExportTTLScheduler::getReplicaName() const -{ - return replicated_storage.getReplicaName(); -} - -String ReplicatedExportTTLScheduler::getSchedulerReplica() -{ - const auto zookeeper = replicated_storage.tryGetZooKeeper(); - if (!zookeeper || zookeeper->expired()) - return {}; - - String replica; - zookeeper->tryGet(replicated_storage.export_ttl_index->getSchedulerLockPath(), replica); - return replica; -} - -bool ReplicatedExportTTLScheduler::isCommitInProgress(const String & transaction_id) -{ - return replicated_storage.getZooKeeper()->exists(fs::path(replicated_storage.zookeeper_path) / "exports" / transaction_id / "commit_lock"); -} - -ExportTTLScheduler::TaskState ReplicatedExportTTLScheduler::getTaskState(const String & transaction_id) -{ - using Status = ExportReplicatedMergeTreeTaskEntry::Status; - const auto to_task_status = [](Status status) -> TaskStatus - { - switch (status) - { - case Status::PENDING: return TaskStatus::PENDING; - case Status::COMPLETED: return TaskStatus::COMPLETED; - case Status::FAILED: return TaskStatus::FAILED; - case Status::KILLED: return TaskStatus::KILLED; - } - UNREACHABLE(); - }; - - const auto zookeeper = replicated_storage.getZooKeeper(); - const fs::path task_path = fs::path(replicated_storage.zookeeper_path) / "exports" / transaction_id; - - /// Read from Keeper rather than from the mirror of the tasks, which may lag behind the parts - /// the task exported. Unknown counts as reached. - const auto all_parts_processed = [&](size_t parts_count) - { - Coordination::Stat stat; - if (!zookeeper->exists(task_path / "processed", &stat)) - return true; - return static_cast(stat.numChildren) >= parts_count; - }; - - TaskState state; - std::optional parts_count; - - /// The in-memory mirror of the tasks may lag behind Keeper, which only delays a retry. A task it - /// does not know yet, e.g. just created, is read from Keeper, so it is never taken for missing. - bool found_in_mirror = false; - if (const auto tasks = replicated_storage.export_partition_manifests.get()) - { - const auto & by_transaction_id = tasks->get(); - if (const auto it = by_transaction_id.find(transaction_id); it != by_transaction_id.end()) - { - state.status = to_task_status(it->status); - state.retry_of = it->manifest.retry_of; - parts_count = it->manifest.parts.size(); - found_in_mirror = true; - } - } - - if (!found_in_mirror) - { - const Strings paths{task_path / "status", task_path / "metadata.json"}; - - auto responses = zookeeper->tryGet(paths); - responses.waitForResponses(); - - if (responses[0].error == Coordination::Error::ZNONODE) - return state; - if (responses[0].error != Coordination::Error::ZOK) - throw zkutil::KeeperException::fromPath(responses[0].error, paths[0]); - - if (const auto status = magic_enum::enum_cast(responses[0].data)) - { - state.status = to_task_status(*status); - } - else - { - /// Treated as failed: the recovery check still decides whether it committed. - LOG_WARNING(log, "Export task {} has an unknown status {}", transaction_id, responses[0].data); - state.status = TaskStatus::FAILED; - } - - if (responses[1].error == Coordination::Error::ZOK) - { - auto manifest = ExportReplicatedMergeTreeTaskManifest::fromJsonString(responses[1].data); - state.retry_of = std::move(manifest.retry_of); - parts_count = manifest.parts.size(); - } - else if (responses[1].error != Coordination::Error::ZNONODE) - { - throw zkutil::KeeperException::fromPath(responses[1].error, paths[1]); - } - } - - if (parts_count && (state.status == TaskStatus::FAILED || state.status == TaskStatus::KILLED)) - state.reached_commit = all_parts_processed(*parts_count); - - return state; -} - -bool ReplicatedExportTTLScheduler::updateIndexEntry(const String & destination_key, const ExportTTLVersionedEntry & entry) -{ - const auto zookeeper = replicated_storage.getZooKeeper(); - const auto & export_ttl_index = *replicated_storage.export_ttl_index; - - if (entry.version < 0) - export_ttl_index.ensureDestination(zookeeper, destination_key, destination_key); - - Coordination::Requests ops; - export_ttl_index.appendUpdateEntryOps(ops, destination_key, entry.entry, entry.version); - - Coordination::Responses responses; - const auto code = zookeeper->tryMulti(ops, responses); - if (code == Coordination::Error::ZOK) - return true; - if (code == Coordination::Error::ZBADVERSION || code == Coordination::Error::ZNODEEXISTS || code == Coordination::Error::ZNONODE) - return false; - zkutil::KeeperMultiException::check(code, ops, responses); - return false; -} - -bool ReplicatedExportTTLScheduler::startGroup(const GroupToStart & group, const ContextPtr & context) -{ - auto zookeeper = replicated_storage.getZooKeeperAndAssertNotReadonly(); - replicated_storage.checkAllReplicasSupportExportTTL(zookeeper); - - /// Its `/log` version is checked when the index entry is stored, so no merge can be assigned between - /// checking the parts below, including the ones without rows, and claiming them. - const auto merge_predicate = replicated_storage.queue.getMergePredicate(zookeeper, PartitionIdsHint{group.partition_id}); - for (const auto & part : group.parts) - { - const auto covering_part = merge_predicate->getCoveringVirtualPart(part->name); - if (covering_part.empty()) - return false; - - const auto covering_info = MergeTreePartInfo::fromPartName(covering_part, replicated_storage.format_version); - if (covering_info.min_block != part->info.min_block || covering_info.max_block != part->info.max_block) - return false; - } - - Coordination::Requests ops; - ops.emplace_back(zkutil::makeCheckRequest(fs::path(replicated_storage.zookeeper_path) / "log", merge_predicate->getVersion())); - - const auto parts_with_rows = group.partsWithRows(); - if (!parts_with_rows.empty()) - { - const auto source_metadata = replicated_storage.getInMemoryMetadataPtr(context, false); - const auto destination_metadata = group.destination->getInMemoryMetadataPtr(context, false); - ExportTaskUtils::verifyExportSchemaCastable(source_metadata, destination_metadata, group.destination->getStorageID(), context); - - MergeTreeData::DataPartsVector parts(parts_with_rows.begin(), parts_with_rows.end()); - auto manifest = replicated_storage.buildExportTaskManifest( - group.destination->getStorageID(), group.destination, source_metadata, destination_metadata, parts, group.partition_id, context); - manifest.transaction_id = group.transaction_id; - manifest.source = ExportTaskSource::ttl; - manifest.retry_of = group.retry_of; - - ExportTaskUtils::appendCreateExportTaskOps(ops, fs::path(replicated_storage.zookeeper_path) / "exports" / group.transaction_id, manifest); - } - - const auto & export_ttl_index = *replicated_storage.export_ttl_index; - if (group.entry.version < 0) - export_ttl_index.ensureDestination(zookeeper, group.destination_key, group.destination->getStorageID().getNameForLogs()); - export_ttl_index.appendUpdateEntryOps(ops, group.destination_key, group.entry.entry, group.entry.version); - - ProfileEvents::increment(ProfileEvents::ExportTaskZooKeeperRequests); - ProfileEvents::increment(ProfileEvents::ExportTaskZooKeeperMulti); - Coordination::Responses responses; - const auto code = zookeeper->tryMulti(ops, responses); - - if (code == Coordination::Error::ZOK) - { - if (!parts_with_rows.empty() && replicated_storage.export_task_updating_task) - replicated_storage.export_task_updating_task->schedule(); - return true; - } - - if (code == Coordination::Error::ZBADVERSION || code == Coordination::Error::ZNODEEXISTS) - return false; - - zkutil::KeeperMultiException::check(code, ops, responses); - return false; -} - -bool ReplicatedExportTTLScheduler::isPartBeingMerged(const MergeTreeDataPartPtr & part) -{ - return replicated_storage.queue.isGoingToBeMergedWithOtherParts(part->info); -} - -void ReplicatedExportTTLScheduler::killTask(const String & transaction_id) -{ - replicated_storage.killExportTask(transaction_id); -} - -void ReplicatedExportTTLScheduler::removeDestination(const String & destination_key) -{ - replicated_storage.export_ttl_index->removeDestination(replicated_storage.getZooKeeper(), destination_key); -} - -} diff --git a/src/Storages/MergeTree/ReplicatedExportTTLScheduler.h b/src/Storages/MergeTree/ReplicatedExportTTLScheduler.h deleted file mode 100644 index 664c8ccbb38a..000000000000 --- a/src/Storages/MergeTree/ReplicatedExportTTLScheduler.h +++ /dev/null @@ -1,48 +0,0 @@ -#pragma once - -#include -#include - -#include - -namespace DB -{ - -class StorageReplicatedMergeTree; - -/// `ExportTTLScheduler` of a `ReplicatedMergeTree`: the index is in Keeper (see -/// `ReplicatedExportTTLIndex`), a group is claimed in the transaction that creates its task, -/// with a check that no merge was assigned meanwhile, and one replica schedules at a time. -class ReplicatedExportTTLScheduler final : public ExportTTLScheduler -{ -public: - explicit ReplicatedExportTTLScheduler(StorageReplicatedMergeTree & storage_); - ~ReplicatedExportTTLScheduler() override; - - /// E.g. when the replica goes read-only, so another replica can take over. - void releaseSchedulerLock(); - -protected: - bool acquireSchedulerLock() override; - bool isPaused() override; - ExportTTLIndexSnapshotPtr getIndexSnapshot() override; - String getReplicaName() const override; - String getSchedulerReplica() override; - TaskState getTaskState(const String & transaction_id) override; - bool isCommitInProgress(const String & transaction_id) override; - bool updateIndexEntry(const String & destination_key, const ExportTTLVersionedEntry & entry) override; - bool startGroup(const GroupToStart & group, const ContextPtr & context) override; - bool isPartBeingMerged(const MergeTreeDataPartPtr & part) override; - void killTask(const String & transaction_id) override; - void removeDestination(const String & destination_key) override; - -private: - StorageReplicatedMergeTree & replicated_storage; - - std::mutex lock_mutex; - /// Keeps the session the lock was created in alive: the holder refers to it. - zkutil::ZooKeeperPtr lock_zookeeper; - zkutil::EphemeralNodeHolderPtr lock_holder; -}; - -} diff --git a/src/Storages/MergeTree/ReplicatedExportTaskUpdater.cpp b/src/Storages/MergeTree/ReplicatedExportTaskUpdater.cpp index 19ffb7be647b..3e020509c40b 100644 --- a/src/Storages/MergeTree/ReplicatedExportTaskUpdater.cpp +++ b/src/Storages/MergeTree/ReplicatedExportTaskUpdater.cpp @@ -422,7 +422,7 @@ std::vector ReplicatedExportTaskUpdater::getExportTasksInfo() co info.destination_table = manifest.destination_table; info.partition_id = ExportTaskUtils::getPartitionIdOfParts(manifest.parts, storage.format_version); info.source = String(magic_enum::enum_name(manifest.source)); - info.retry_of = ExportRetriedTaskUtils::transactionIds(manifest.retry_of); + info.commit_id = manifest.commit_id; info.transaction_id = manifest.transaction_id; info.query_id = manifest.query_id; info.create_time = manifest.create_time; diff --git a/src/Storages/MergeTree/ReplicatedMergeTreeRestartingThread.cpp b/src/Storages/MergeTree/ReplicatedMergeTreeRestartingThread.cpp index 499adb6d47a9..672e2d79a0d2 100644 --- a/src/Storages/MergeTree/ReplicatedMergeTreeRestartingThread.cpp +++ b/src/Storages/MergeTree/ReplicatedMergeTreeRestartingThread.cpp @@ -195,6 +195,7 @@ bool ReplicatedMergeTreeRestartingThread::runImpl() storage.export_task_select_task->activateAndSchedule(); storage.export_task_status_handling_task->activateAndSchedule(); storage.export_ttl_task->activateAndSchedule(); + storage.export_ttl_index_updating_task->activateAndSchedule(); } storage.cleanup_thread.start(); diff --git a/src/Storages/MergeTree/tests/gtest_export_task_ordering.cpp b/src/Storages/MergeTree/tests/gtest_export_task_ordering.cpp index 107a0b5bea25..6be3eadbfa28 100644 --- a/src/Storages/MergeTree/tests/gtest_export_task_ordering.cpp +++ b/src/Storages/MergeTree/tests/gtest_export_task_ordering.cpp @@ -3,7 +3,6 @@ #include #include #include -#include #include #include #include @@ -235,39 +234,50 @@ TEST_F(ExportTaskManifestBackCompatTest, IgnoreExtraSourceColumnsAppliedToWorker } } -TEST_F(ExportTaskManifestBackCompatTest, TaskSourceAndRetryOfRoundTrip) +TEST_F(ExportTaskManifestBackCompatTest, TaskSourceAndCommitIdRoundTrip) { auto manifest = makeValidManifest(); manifest.source = ExportTaskSource::ttl; - manifest.retry_of = { - ExportRetriedTask{.transaction_id = "tx0", .block_ranges = {{1, 3}, {7, 7}}, .failed_time = 1700000000}, - ExportRetriedTask{.transaction_id = "tx00", .block_ranges = {{1, 1}}, .failed_time = 1600000000}, - }; + manifest.commit_id = "tx0"; const auto parsed = ExportReplicatedMergeTreeTaskManifest::fromJsonString(manifest.toJsonString()); EXPECT_EQ(parsed.source, ExportTaskSource::ttl); - EXPECT_EQ(parsed.retry_of, manifest.retry_of); - EXPECT_EQ(ExportRetriedTaskUtils::transactionIds(parsed.retry_of), (std::vector{"tx0", "tx00"})); + EXPECT_EQ(parsed.commit_id, "tx0"); +} + +TEST_F(ExportTaskManifestBackCompatTest, MissingCommitIdIsTheTransactionId) +{ + auto manifest = makeValidManifest(); + manifest.commit_id = "tx0"; + + Poco::JSON::Parser parser; + auto json = parser.parse(manifest.toJsonString()).extract(); + json->remove("commit_id"); + json->set("retry_of", Poco::JSON::Array::Ptr(new Poco::JSON::Array())); + std::ostringstream oss; + oss.exceptions(std::ios::failbit); + Poco::JSON::Stringifier::stringify(json, oss); - /// A task of `EXPORT PARTITION` records neither. - const auto query_task = ExportReplicatedMergeTreeTaskManifest::fromJsonString(makeValidManifest().toJsonString()); - EXPECT_EQ(query_task.source, ExportTaskSource::query); - EXPECT_TRUE(query_task.retry_of.empty()); + EXPECT_EQ(ExportReplicatedMergeTreeTaskManifest::fromJsonString(oss.str()).commit_id, "tx1"); } -TEST(ExportTaskUtils, PlainTaskRetryOfRoundTrip) +TEST_F(ExportTaskManifestBackCompatTest, PlainTaskMissingCommitIdIsTheTransactionId) { MergeTreeExportTask task; task.transaction_id = "tx1"; + task.commit_id = "tx0"; task.source = ExportTaskSource::ttl; task.parts.push_back({.part_name = "p_1_3_1", .done = true, .paths_in_destination = {"a.parquet"}}); - task.retry_of = {ExportRetriedTask{.transaction_id = "tx0", .block_ranges = {{1, 3}}, .failed_time = 1700000000}}; + EXPECT_EQ(MergeTreeExportTask::fromJsonString(task.toJsonString()).commit_id, "tx0"); - const auto parsed = MergeTreeExportTask::fromJsonString(task.toJsonString()); - EXPECT_EQ(parsed.retry_of, task.retry_of); + Poco::JSON::Parser parser; + auto json = parser.parse(task.toJsonString()).extract(); + json->remove("commit_id"); + std::ostringstream oss; + oss.exceptions(std::ios::failbit); + Poco::JSON::Stringifier::stringify(json, oss); - task.retry_of.clear(); - EXPECT_TRUE(MergeTreeExportTask::fromJsonString(task.toJsonString()).retry_of.empty()); + EXPECT_EQ(MergeTreeExportTask::fromJsonString(oss.str()).commit_id, "tx1"); } namespace @@ -319,16 +329,6 @@ TEST(ExportTaskUtils, PartitionIdIsDerivedFromParts) EXPECT_EQ(ExportTaskUtils::getPartitionIdOfParts({"all_3_3_0"}, MERGE_TREE_DATA_MIN_FORMAT_VERSION_WITH_CUSTOM_PARTITIONING), "all"); } -TEST(ExportTaskUtils, PartsCommittedByRetriedTaskAreLeftOut) -{ - const auto committed = ExportTTLUtils::rangesOfParts({"p_1_1_0", "p_2_2_0"}, MERGE_TREE_DATA_MIN_FORMAT_VERSION_WITH_CUSTOM_PARTITIONING); - - /// A mutated part keeps its block range, so it counts as committed as well. - const auto remaining = ExportTaskUtils::getPartsNotCommitted( - {"p_1_1_0_7", "p_2_2_0", "p_3_3_0"}, committed, MERGE_TREE_DATA_MIN_FORMAT_VERSION_WITH_CUSTOM_PARTITIONING); - EXPECT_EQ(remaining, std::vector{"p_3_3_0"}); -} - TEST(ExportTaskRetryClassification, MissingPartIsFatalOnlyOnPlainPath) { EXPECT_FALSE(ExportTaskUtils::isNonRetryableExportError(ErrorCodes::NO_SUCH_DATA_PART)); diff --git a/src/Storages/MergeTree/tests/gtest_export_ttl_index.cpp b/src/Storages/MergeTree/tests/gtest_export_ttl_index.cpp index c0237be288b7..90597ba70ab6 100644 --- a/src/Storages/MergeTree/tests/gtest_export_ttl_index.cpp +++ b/src/Storages/MergeTree/tests/gtest_export_ttl_index.cpp @@ -33,8 +33,9 @@ TEST(ExportTTLIndex, ClaimThenCommit) ExportTTLIndexEntry entry; entry.partition_id = "p"; - entry.startClaim("t1", {part(1, 1), part(2, 2), part(4, 4)}); + entry.startClaim("t1", "t1", {part(1, 1), part(2, 2), part(4, 4)}); ASSERT_TRUE(entry.claim); + EXPECT_EQ(entry.claim->commit_id, "t1"); EXPECT_EQ(entry.claim->transaction_id, "t1"); EXPECT_EQ(blocks(entry.claim->ranges), (Blocks{{1, 2}, {4, 4}})); EXPECT_EQ(blocks(entry.toFenceEntry("db.t").claimed), (Blocks{{1, 2}, {4, 4}})); @@ -50,21 +51,24 @@ TEST(ExportTTLIndex, ClaimThenCommit) EXPECT_EQ(entry.maxBlock(), 4); } -TEST(ExportTTLIndex, RetryReclaimsExactRanges) +TEST(ExportTTLIndex, RetryKeepsTheCommitIdAndRanges) { ExportTTLIndexEntry entry; entry.partition_id = "p"; - entry.startClaim("t1", {part(1, 1), part(2, 2)}); - - /// Part 2 was dropped before the retry, part 5 became eligible meanwhile. - entry.releaseClaim(); - entry.startClaim("t2", {part(1, 1), part(5, 5)}); + entry.startClaim("t1", "t1", {part(1, 1), part(2, 2)}); + entry.retryClaim("t2"); ASSERT_TRUE(entry.claim); + EXPECT_EQ(entry.claim->commit_id, "t1"); EXPECT_EQ(entry.claim->transaction_id, "t2"); - EXPECT_EQ(blocks(entry.claim->ranges), (Blocks{{1, 1}, {5, 5}})); - EXPECT_EQ(entry.classify(part(2, 2)), PartExportState::NONE); - EXPECT_EQ(entry.maxBlock(), 5); + EXPECT_EQ(blocks(entry.claim->ranges), (Blocks{{1, 2}})); + + /// The claim is committed under its commit id, whichever task commits it. + entry.commitClaim("t2", {}); + ASSERT_TRUE(entry.claim); + entry.commitClaim("t1", {}); + EXPECT_FALSE(entry.claim); + EXPECT_EQ(blocks(entry.exported), (Blocks{{1, 2}})); } /// A second claim is a `LOGICAL_ERROR`, which aborts debug and sanitizer builds instead of throwing. @@ -73,9 +77,9 @@ TEST(ExportTTLIndex, OneClaimAtATime) { ExportTTLIndexEntry entry; entry.partition_id = "p"; - entry.startClaim("t1", {part(1, 1)}); + entry.startClaim("t1", "t1", {part(1, 1)}); - EXPECT_THROW(entry.startClaim("t2", {part(2, 2)}), Exception); + EXPECT_THROW(entry.startClaim("t2", "t2", {part(2, 2)}), Exception); EXPECT_EQ(entry.claim->transaction_id, "t1"); EXPECT_EQ(blocks(entry.claim->ranges), (Blocks{{1, 1}})); } @@ -84,9 +88,9 @@ TEST(ExportTTLIndexDeathTest, OneClaimAtATime) { ExportTTLIndexEntry entry; entry.partition_id = "p"; - entry.startClaim("t1", {part(1, 1)}); + entry.startClaim("t1", "t1", {part(1, 1)}); - EXPECT_DEATH(entry.startClaim("t2", {part(2, 2)}), "cannot claim parts of partition p"); + EXPECT_DEATH(entry.startClaim("t2", "t2", {part(2, 2)}), "cannot claim parts of partition p"); } #endif @@ -98,7 +102,7 @@ TEST(ExportTTLIndex, CommitAddsPartsWhoseClaimWasLost) EXPECT_EQ(blocks(entry.exported), (Blocks{{3, 5}})); /// The claim of another task is kept. - entry.startClaim("t2", {part(7, 7)}); + entry.startClaim("t2", "t2", {part(7, 7)}); entry.commitClaim("t1", {part(6, 6)}); EXPECT_EQ(blocks(entry.exported), (Blocks{{3, 6}})); ASSERT_TRUE(entry.claim); @@ -115,14 +119,20 @@ TEST(ExportTTLIndex, JsonRoundTrip) EXPECT_EQ(blocks(parsed.exported), (Blocks{{0, 10}, {20, 30}})); EXPECT_FALSE(parsed.claim); - entry.startClaim("t1", {part(31, 31)}); + entry.startClaim("c1", "t1", {part(31, 31)}); parsed = ExportTTLIndexEntry::fromJSONString("p", entry.toJSONString()); EXPECT_EQ(parsed.partition_id, "p"); EXPECT_EQ(blocks(parsed.exported), (Blocks{{0, 10}, {20, 30}})); ASSERT_TRUE(parsed.claim); + EXPECT_EQ(parsed.claim->commit_id, "c1"); EXPECT_EQ(parsed.claim->transaction_id, "t1"); EXPECT_EQ(blocks(parsed.claim->ranges), (Blocks{{31, 31}})); + /// A claim stored before commit ids existed is committed under its transaction id. + parsed = ExportTTLIndexEntry::fromJSONString("p", R"({"exported":[],"claim":{"transaction_id":"t1","ranges":[[31,31]]}})"); + ASSERT_TRUE(parsed.claim); + EXPECT_EQ(parsed.claim->commit_id, "t1"); + EXPECT_TRUE(ExportTTLIndexEntry::fromJSONString("p", "").empty()); EXPECT_ANY_THROW(ExportTTLIndexEntry::fromJSONString("p", R"({"exported":[[1]]})")); } @@ -138,17 +148,6 @@ TEST(ExportTTLIndex, RangesOfParts) EXPECT_EQ(blocks(ranges), (Blocks{{1, 3}, {5, 5}})); } -TEST(ExportTTLIndex, BlockRangesOfRetriedTasks) -{ - const std::vector ranges{part(1, 3), part(7, 7)}; - const auto block_ranges = ExportTTLUtils::toBlockRanges(ranges); - EXPECT_EQ(block_ranges, (Blocks{{1, 3}, {7, 7}})); - - const auto restored = ExportTTLUtils::fromBlockRanges("p", {{4, 5}, {1, 3}}); - EXPECT_EQ(blocks(restored), (Blocks{{1, 5}})); - EXPECT_EQ(restored.front().getPartitionId(), "p"); -} - TEST(ExportTTLIndex, EligibleOnceTheMaximumIsDue) { TTLDescription description; diff --git a/src/Storages/StorageMergeTree.cpp b/src/Storages/StorageMergeTree.cpp index 3c5afd3c8d2d..053b919f6032 100644 --- a/src/Storages/StorageMergeTree.cpp +++ b/src/Storages/StorageMergeTree.cpp @@ -29,7 +29,6 @@ #include #include #include -#include #include #include #include @@ -54,7 +53,8 @@ #include #include #include -#include +#include +#include #include #include #include @@ -130,7 +130,6 @@ namespace Setting extern const SettingsBool export_merge_tree_part_throw_on_pending_mutations; extern const SettingsBool export_merge_tree_part_throw_on_pending_patch_parts; extern const SettingsBool export_merge_tree_part_allow_lossy_cast; - extern const SettingsExportPartitionAllOnError export_merge_tree_partition_all_on_error; extern const SettingsString export_merge_tree_part_filename_pattern; extern const SettingsBool write_full_path_in_iceberg_metadata; extern const SettingsUInt64 iceberg_insert_max_bytes_in_data_file; @@ -185,8 +184,6 @@ namespace ErrorCodes extern const int PART_IS_TEMPORARILY_LOCKED; extern const int FAULT_INJECTED; extern const int INCOMPATIBLE_COLUMNS; - extern const int EXPORT_PARTITION_ALREADY_EXPORTED; - extern const int PARTITION_EXPORT_FAILED; extern const int PENDING_MUTATIONS_NOT_ALLOWED; } @@ -293,7 +290,7 @@ StorageMergeTree::StorageMergeTree( /// parts no longer exist, or new parts would look exported. increment.set(std::max(increment.value.load(), static_cast(export_ttl_index->maxBlock()))); - export_ttl_scheduler = std::make_shared(*this, *export_ttl_index); + export_ttl_scheduler = std::make_shared(*this, *export_ttl_index); export_ttl_task = getContext()->getSchedulePool().createTask( getStorageID(), getStorageID().getFullTableName() + " (StorageMergeTree::export_ttl_task)", @@ -3737,129 +3734,50 @@ CommittingBlocksSet StorageMergeTree::getCommittingBlocks() const return committing_blocks; } -void StorageMergeTree::exportPartitionToTable(const PartitionCommand & command, ContextPtr query_context) +bool StorageMergeTree::exportParts(const ExportPartsRequest & request, ContextPtr local_context, const ExportTTLIndexUpdate * index_update) { - if (!query_context->getServerSettings()[ServerSetting::allow_experimental_export_merge_tree_partition]) - throw Exception(ErrorCodes::SUPPORT_IS_DISABLED, - "Exporting merge tree partition is experimental. Set the server setting `allow_experimental_export_merge_tree_partition` to enable it.\n" - "If you are exporting to an Apache Iceberg table, you also need to enable the setting `allow_insert_into_iceberg`."); - - /// The scheduler is created in the constructor whenever the server setting above is enabled, so - /// this should always hold here. + /// The scheduler is created in the constructor whenever partition export is enabled, so this + /// should always hold here. if (!export_task_scheduler) throw Exception(ErrorCodes::SUPPORT_IS_DISABLED, "Partition export is not initialized for table {}", getStorageID().getNameForLogs()); - /// EXPORT PARTITION ALL: expand into one sub-call per active partition id. - /// Failure handling is controlled by `export_merge_tree_partition_all_on_error`. - if (const auto * partition_ast = command.partition->as(); partition_ast && partition_ast->all) - { - auto partition_id_set = getAllPartitionIds(); - if (partition_id_set.empty()) - throw Exception(ErrorCodes::BAD_ARGUMENTS, - "Table {} has no active partitions to export", getStorageID().getNameForLogs()); - - std::vector partition_ids(partition_id_set.begin(), partition_id_set.end()); - std::sort(partition_ids.begin(), partition_ids.end()); - - const auto & on_error_setting = query_context->getSettingsRef()[Setting::export_merge_tree_partition_all_on_error]; - const ExportPartitionAllOnError on_error = on_error_setting.value; - - LOG_INFO(log, "EXPORT PARTITION ALL: scheduling export for {} partitions, on_error={}", - partition_ids.size(), on_error_setting.toString()); + const auto & destination = request.destination; + if (!destination->supportsImport(local_context)) + throw Exception(ErrorCodes::NOT_IMPLEMENTED, "Destination storage {} does not support MergeTree parts or uses unsupported partitioning", destination->getName()); - std::vector> failures; /// (partition_id, message) - size_t skipped_conflicts = 0; + const auto source_metadata = getInMemoryMetadataPtr(local_context, false); + const auto destination_metadata = destination->getInMemoryMetadataPtr(local_context, false); - for (const auto & partition_id : partition_ids) - { - PartitionCommand sub = command; - auto synthetic = make_intrusive(); - synthetic->setPartitionID(make_intrusive(partition_id)); - sub.partition = synthetic; + /// Positional CAST matching, like `INSERT INTO dest SELECT * FROM src`. + ExportTaskUtils::verifyExportSchemaCastable(source_metadata, destination_metadata, destination->getStorageID(), local_context); - try - { - exportPartitionToTable(sub, query_context); - } - catch (const Exception & e) - { - switch (on_error) - { - case ExportPartitionAllOnError::throw_first: - throw; - case ExportPartitionAllOnError::skip_conflicts: - if (e.code() == ErrorCodes::EXPORT_PARTITION_ALREADY_EXPORTED) - { - ++skipped_conflicts; - LOG_INFO(log, "EXPORT PARTITION ALL: skipping partition {} (already exported): {}", - partition_id, e.message()); - break; - } - throw; - case ExportPartitionAllOnError::collect: - LOG_WARNING(log, "EXPORT PARTITION ALL: partition {} failed: {}", partition_id, e.message()); - failures.emplace_back(partition_id, e.message()); - break; - } - } - } + auto descriptor = buildExportTask( + destination->getStorageID(), destination, source_metadata, destination_metadata, request.parts, request.partition_id, local_context); + descriptor.transaction_id = request.transaction_id.empty() ? toString(UUIDHelpers::generateV4()) : request.transaction_id; + descriptor.query_id = request.query_id; + descriptor.source = request.source; + descriptor.commit_id = request.commit_id.empty() ? descriptor.transaction_id : request.commit_id; - if (!failures.empty()) + if (index_update) + { + /// Merge selection holds it while it checks the fence and tags the parts it merges. + std::lock_guard lock(currently_processing_in_background_mutex); + for (const auto & part : request.parts) { - String aggregated = fmt::format( - "EXPORT PARTITION ALL: {}/{} partitions failed to schedule. Per-partition errors:", - failures.size(), partition_ids.size()); - for (const auto & [pid, msg] : failures) - aggregated += fmt::format("\n {}: {}", pid, msg); - throw Exception(ErrorCodes::PARTITION_EXPORT_FAILED, "{}", aggregated); + if (part->getState() != MergeTreeDataPartState::Active || currently_merging_mutating_parts.contains(part->info)) + return false; } - if (skipped_conflicts > 0) - LOG_INFO(log, "EXPORT PARTITION ALL: skipped {} partitions due to existing exports", skipped_conflicts); - - return; - } - - const auto dest_database = query_context->resolveDatabase(command.to_database); - const auto dest_table = command.to_table; - const auto dest_storage_id = StorageID(dest_database, dest_table); - auto dest_storage = DatabaseCatalog::instance().getTable({dest_database, dest_table}, query_context); - - if (dest_storage->getStorageID() == this->getStorageID()) - { - throw Exception(ErrorCodes::BAD_ARGUMENTS, "Exporting to the same table is not allowed"); - } - - if (!dest_storage->supportsImport(query_context)) - throw Exception(ErrorCodes::NOT_IMPLEMENTED, "Destination storage {} does not support MergeTree parts or uses unsupported partitioning", dest_storage->getName()); - - auto src_snapshot = getInMemoryMetadataPtr(query_context, false); - auto destination_snapshot = dest_storage->getInMemoryMetadataPtr(query_context, false); - - /// Positional CAST matching, like `INSERT INTO dest SELECT * FROM src`. - ExportTaskUtils::verifyExportSchemaCastable( - src_snapshot, destination_snapshot, dest_storage->getStorageID(), query_context); - - const String partition_id = getPartitionIDFromQuery(command.partition, query_context); - - DataPartsVector parts; - { - auto data_parts_lock = lockParts(); - parts = getDataPartsVectorInPartitionForInternalUsage(MergeTreeDataPartState::Active, partition_id, data_parts_lock); + if (!export_ttl_index->updateEntry(index_update->destination_key, index_update->entry)) + return false; } - if (parts.empty()) - throw Exception(ErrorCodes::BAD_ARGUMENTS, "Partition {} doesn't exist", partition_id); - - /// Every `EXPORT PARTITION` is a new task, even if the partition was exported before. - auto descriptor = buildExportTask(dest_storage_id, dest_storage, src_snapshot, destination_snapshot, parts, partition_id, query_context); - descriptor.transaction_id = toString(UUIDHelpers::generateV4()); - descriptor.query_id = query_context->getCurrentQueryId(); - descriptor.source = ExportTaskSource::query; - - std::vector part_references(parts.begin(), parts.end()); - export_task_scheduler->addTask(std::move(descriptor), std::move(part_references)); + const auto transaction_id = descriptor.transaction_id; + export_task_scheduler->addTask(std::move(descriptor), std::vector(request.parts.begin(), request.parts.end())); + LOG_INFO(log, "Created export task {} of partition {} to {}, {} part(s)", + transaction_id, request.partition_id, destination->getStorageID().getNameForLogs(), request.parts.size()); + return true; } MergeTreeExportTask StorageMergeTree::buildExportTask( @@ -4013,9 +3931,30 @@ void StorageMergeTree::exportTTLTask() export_ttl_task->scheduleAfter(next_run_ms); } -ExportFencePtr StorageMergeTree::getExportFence() const +IExportTTLIndex * StorageMergeTree::getExportTTLIndex() const +{ + return export_ttl_index.get(); +} + +std::optional StorageMergeTree::getExportTaskStatus(const String & transaction_id) const +{ + const auto task = export_task_scheduler->getTask(transaction_id); + if (!task) + return std::nullopt; + + switch (task->status) + { + case MergeTreeExportTask::Status::PENDING: return ExportTaskStatus::PENDING; + case MergeTreeExportTask::Status::COMPLETED: return ExportTaskStatus::COMPLETED; + case MergeTreeExportTask::Status::FAILED: return ExportTaskStatus::FAILED; + case MergeTreeExportTask::Status::KILLED: return ExportTaskStatus::KILLED; + } +} + +bool StorageMergeTree::isPartBeingMerged(const MergeTreePartInfo & part_info) const { - return export_ttl_index ? export_ttl_index->getFence() : nullptr; + std::lock_guard lock(currently_processing_in_background_mutex); + return currently_merging_mutating_parts.contains(part_info); } void StorageMergeTree::triggerExportTaskScheduling() diff --git a/src/Storages/StorageMergeTree.h b/src/Storages/StorageMergeTree.h index 90f1caa3d4e4..cc3f6e3b88d3 100644 --- a/src/Storages/StorageMergeTree.h +++ b/src/Storages/StorageMergeTree.h @@ -130,22 +130,28 @@ class StorageMergeTree final : public MergeTreeData MergeTreeDeduplicationLog * getDeduplicationLog() { return deduplication_log.get(); } - /// EXPORT PARTITION for a plain (non-replicated) MergeTree table. Coordinated locally by - /// `export_task_scheduler`; the task descriptor is persisted on disk (no ZooKeeper). - void exportPartitionToTable(const PartitionCommand & command, ContextPtr query_context) override; + /// Coordinated locally by `export_task_scheduler`; the task descriptor is persisted on disk (no + /// ZooKeeper). With an index update, the parts are claimed under the lock that merge selection + /// holds, before the task is created: a crash in between leaves a claim without a task. + bool exportParts(const ExportPartsRequest & request, ContextPtr local_context, const ExportTTLIndexUpdate * index_update) override; CancellationCode killExportTask(const String & transaction_id) override; /// Snapshot of local partition-export tasks for `system.distributed_exports`. No disk I/O. std::vector getExportTasksInfo() const override; - /// Export states of parts for the `EXPORT` TTL, nullptr if partition export is disabled. - ExportFencePtr getExportFence() const; - ExportFencePtr getLatestExportFence() const override { return getExportFence(); } + /// Nullptr if partition export is disabled. + IExportTTLIndex * getExportTTLIndex() const override; + + std::optional getExportTaskStatus(const String & transaction_id) const override; + std::optional getKnownExportTaskStatus(const String & transaction_id) const override + { + return getExportTaskStatus(transaction_id); + } + bool isPartBeingMerged(const MergeTreePartInfo & part_info) const override; private: friend class MergeTreeExportTaskScheduler; - friend class MergeTreeExportTTLScheduler; friend class MergeTreeExportTTLIndex; /// Builds the descriptor of an export task of `parts` from the settings of `query_context`, and diff --git a/src/Storages/StorageReplicatedMergeTree.cpp b/src/Storages/StorageReplicatedMergeTree.cpp index e11de103c754..d9ca42e798f8 100644 --- a/src/Storages/StorageReplicatedMergeTree.cpp +++ b/src/Storages/StorageReplicatedMergeTree.cpp @@ -5,6 +5,7 @@ #include #include +#include #include #include #include "Common/ZooKeeper/IKeeper.h" @@ -83,7 +84,7 @@ #include #include #include -#include +#include #include #include @@ -241,7 +242,6 @@ namespace Setting extern const SettingsBool export_merge_tree_part_allow_lossy_cast; extern const SettingsMergeTreePartExportSchemaMatchMode export_merge_tree_part_schema_match_mode; extern const SettingsBool export_merge_tree_part_ignore_extra_source_columns; - extern const SettingsExportPartitionAllOnError export_merge_tree_partition_all_on_error; extern const SettingsString export_merge_tree_part_filename_pattern; extern const SettingsBool write_full_path_in_iceberg_metadata; extern const SettingsBool allow_insert_into_iceberg; @@ -363,8 +363,6 @@ namespace ErrorCodes extern const int TIMEOUT_EXCEEDED; extern const int INVALID_SETTING_VALUE; extern const int PENDING_MUTATIONS_NOT_ALLOWED; - extern const int EXPORT_PARTITION_ALREADY_EXPORTED; - extern const int PARTITION_EXPORT_FAILED; } namespace ServerSetting @@ -591,12 +589,17 @@ StorageReplicatedMergeTree::StorageReplicatedMergeTree( export_task_select_task->deactivate(); - export_ttl_index = std::make_shared(zookeeper_path, log.load()); + export_ttl_index = std::make_shared( + zookeeper_path, replica_name, [this] { return tryGetZooKeeper(); }, [this] { return is_readonly.load(); }, log.load()); - export_ttl_scheduler = std::make_shared(*this); + export_ttl_scheduler = std::make_shared(*this, *export_ttl_index); export_ttl_task = getContext()->getSchedulePool().createTask( getStorageID(), getStorageID().getFullTableName() + " (StorageReplicatedMergeTree::export_ttl_task)", [this] { exportTTLTask(); }); export_ttl_task->deactivate(); + export_ttl_index_updating_task = getContext()->getSchedulePool().createTask( + getStorageID(), getStorageID().getFullTableName() + " (StorageReplicatedMergeTree::export_ttl_index_updating_task)", + [this] { exportTTLIndexUpdatingTask(); }); + export_ttl_index_updating_task->deactivate(); } @@ -6289,8 +6292,8 @@ void StorageReplicatedMergeTree::partialShutdown() export_task_select_task->deactivate(); export_task_status_handling_task->deactivate(); export_ttl_task->deactivate(); - if (auto * scheduler = dynamic_cast(export_ttl_scheduler.get())) - scheduler->releaseSchedulerLock(); + export_ttl_index_updating_task->deactivate(); + export_ttl_index->releaseSchedulerLock(); } cleanup_thread.stop(); @@ -6995,6 +6998,10 @@ bool StorageReplicatedMergeTree::executeMetadataAlter(const StorageReplicatedMer resetSerializationHints(parts_lock); } + /// The index is watched only while the table has an `EXPORT` TTL, which may have been added. + if (export_ttl_index_updating_task) + export_ttl_index_updating_task->schedule(); + return true; } @@ -8600,149 +8607,75 @@ void StorageReplicatedMergeTree::fetchPartition( LOG_TRACE(log, "Fetch took {} sec. ({} tries)", watch.elapsedSeconds(), try_no); } -void StorageReplicatedMergeTree::exportPartitionToTable(const PartitionCommand & command, ContextPtr query_context) +bool StorageReplicatedMergeTree::exportParts(const ExportPartsRequest & request, ContextPtr local_context, const ExportTTLIndexUpdate * index_update) { - auto component_guard = Coordination::setCurrentComponent("StorageReplicatedMergeTree::exportPartitionToTable"); - if (!query_context->getServerSettings()[ServerSetting::allow_experimental_export_merge_tree_partition]) - { - throw Exception(ErrorCodes::SUPPORT_IS_DISABLED, - "Exporting merge tree partition is experimental. Set the server setting `allow_experimental_export_merge_tree_partition` to enable it (on all replicas).\n" - "If you are exporting to an Apache Iceberg table, you also need to enable the setting `allow_insert_into_iceberg` on the initiator (query, session or profile) - replicas inherit it from the scheduled task."); - } - - /// EXPORT PARTITION ALL: expand into one sub-call per active partition id. - /// Failure handling is controlled by `export_merge_tree_partition_all_on_error`. - if (const auto * partition_ast = command.partition->as(); partition_ast && partition_ast->all) - { - auto partition_id_set = getAllPartitionIds(); - if (partition_id_set.empty()) - throw Exception(ErrorCodes::BAD_ARGUMENTS, - "Table {} has no active partitions to export", - getStorageID().getNameForLogs()); + auto component_guard = Coordination::setCurrentComponent("StorageReplicatedMergeTree::exportParts"); - /// Sort for deterministic ordering (so failure messages and tests are stable). - std::vector partition_ids(partition_id_set.begin(), partition_id_set.end()); - std::sort(partition_ids.begin(), partition_ids.end()); + const auto & destination = request.destination; + if (!destination->supportsImport(local_context)) + throw Exception(ErrorCodes::NOT_IMPLEMENTED, "Destination storage {} does not support MergeTree parts or uses unsupported partitioning", destination->getName()); - const auto & on_error_setting = query_context->getSettingsRef()[Setting::export_merge_tree_partition_all_on_error]; - const ExportPartitionAllOnError on_error = on_error_setting.value; + const auto source_metadata = getInMemoryMetadataPtr(local_context, false); + const auto destination_metadata = destination->getInMemoryMetadataPtr(local_context, false); + ExportTaskUtils::verifyExportSchemaCastable(source_metadata, destination_metadata, destination->getStorageID(), local_context); - LOG_INFO(log, "EXPORT PARTITION ALL: scheduling export for {} partitions, on_error={}", - partition_ids.size(), on_error_setting.toString()); + auto zookeeper = getZooKeeperAndAssertNotReadonly(); - std::vector> failures; /// (partition_id, message) - size_t skipped_conflicts = 0; + Coordination::Requests ops; + if (index_update) + { + checkAllReplicasSupportExportTTL(zookeeper); - for (const auto & partition_id : partition_ids) + /// Its `/log` version is checked when the index entry is stored, so no merge can be assigned between + /// checking the parts below and claiming them. + const auto merge_predicate = queue.getMergePredicate(zookeeper, PartitionIdsHint{request.partition_id}); + for (const auto & part : request.parts) { - PartitionCommand sub = command; - auto synthetic = make_intrusive(); - synthetic->setPartitionID(make_intrusive(partition_id)); - sub.partition = synthetic; - - try - { - exportPartitionToTable(sub, query_context); - } - catch (const Exception & e) - { - switch (on_error) - { - case ExportPartitionAllOnError::throw_first: - throw; - case ExportPartitionAllOnError::skip_conflicts: - if (e.code() == ErrorCodes::EXPORT_PARTITION_ALREADY_EXPORTED) - { - ++skipped_conflicts; - LOG_INFO(log, - "EXPORT PARTITION ALL: skipping partition {} (already exported / concurrent): {}", - partition_id, e.message()); - break; - } - throw; - case ExportPartitionAllOnError::collect: - LOG_WARNING(log, "EXPORT PARTITION ALL: partition {} failed: {}", - partition_id, e.message()); - failures.emplace_back(partition_id, e.message()); - break; - } - } - } + const auto covering_part = merge_predicate->getCoveringVirtualPart(part->name); + if (covering_part.empty()) + return false; - if (!failures.empty()) - { - String aggregated = fmt::format( - "EXPORT PARTITION ALL: {}/{} partitions failed to schedule. Per-partition errors:", - failures.size(), partition_ids.size()); - for (const auto & [pid, msg] : failures) - aggregated += fmt::format("\n {}: {}", pid, msg); - throw Exception(ErrorCodes::PARTITION_EXPORT_FAILED, "{}", aggregated); + const auto covering_info = MergeTreePartInfo::fromPartName(covering_part, format_version); + if (covering_info.min_block != part->info.min_block || covering_info.max_block != part->info.max_block) + return false; } - if (skipped_conflicts > 0) - LOG_INFO(log, "EXPORT PARTITION ALL: skipped {} partitions due to existing exports", - skipped_conflicts); - - return; - } - - const auto dest_database = query_context->resolveDatabase(command.to_database); - const auto dest_table = command.to_table; - const auto dest_storage_id = StorageID(dest_database, dest_table); - auto dest_storage = DatabaseCatalog::instance().getTable({dest_database, dest_table}, query_context); - - if (dest_storage->getStorageID() == this->getStorageID()) - { - throw Exception(ErrorCodes::BAD_ARGUMENTS, "Exporting to the same table is not allowed"); + ops.emplace_back(zkutil::makeCheckRequest(fs::path(zookeeper_path) / "log", merge_predicate->getVersion())); } - if (!dest_storage->supportsImport(query_context)) - throw Exception(ErrorCodes::NOT_IMPLEMENTED, "Destination storage {} does not support MergeTree parts or uses unsupported partitioning", dest_storage->getName()); - - auto src_snapshot = getInMemoryMetadataPtr(query_context, false); - auto destination_snapshot = dest_storage->getInMemoryMetadataPtr(query_context, false); - - ExportTaskUtils::verifyExportSchemaCastable( - src_snapshot, destination_snapshot, dest_storage->getStorageID(), query_context); - - zkutil::ZooKeeperPtr zookeeper = getZooKeeperAndAssertNotReadonly(); - - const String partition_id = getPartitionIDFromQuery(command.partition, query_context); + auto manifest = buildExportTaskManifest( + destination->getStorageID(), destination, source_metadata, destination_metadata, request.parts, request.partition_id, local_context); + manifest.transaction_id = request.transaction_id.empty() ? toString(UUIDHelpers::generateV4()) : request.transaction_id; + manifest.query_id = request.query_id; + manifest.source = request.source; + manifest.commit_id = request.commit_id.empty() ? manifest.transaction_id : request.commit_id; - DataPartsVector parts; - { - auto data_parts_lock = lockParts(); - parts = getDataPartsVectorInPartitionForInternalUsage(MergeTreeDataPartState::Active, partition_id, data_parts_lock); - } + ExportTaskUtils::appendCreateExportTaskOps(ops, fs::path(zookeeper_path) / "exports" / manifest.transaction_id, manifest); - if (parts.empty()) + if (index_update) { - throw Exception(ErrorCodes::BAD_ARGUMENTS, "Partition {} doesn't exist", partition_id); + if (index_update->entry.version < 0) + export_ttl_index->ensureDestination(zookeeper, index_update->destination_key, destination->getStorageID().getNameForLogs()); + export_ttl_index->appendUpdateEntryOps(ops, index_update->destination_key, index_update->entry.entry, index_update->entry.version); } - /// Every `EXPORT PARTITION` is a new task, even if the partition was exported before. - auto manifest = buildExportTaskManifest(dest_storage_id, dest_storage, src_snapshot, destination_snapshot, parts, partition_id, query_context); - manifest.transaction_id = toString(UUIDHelpers::generateV4()); - manifest.query_id = query_context->getCurrentQueryId(); - manifest.source = ExportTaskSource::query; - - const auto task_path = fs::path(zookeeper_path) / "exports" / manifest.transaction_id; - - Coordination::Requests ops; - ExportTaskUtils::appendCreateExportTaskOps(ops, task_path, manifest); - ProfileEvents::increment(ProfileEvents::ExportTaskZooKeeperRequests); ProfileEvents::increment(ProfileEvents::ExportTaskZooKeeperMulti); Coordination::Responses responses; const auto code = zookeeper->tryMulti(ops, responses); if (code != Coordination::Error::ZOK) - throw zkutil::KeeperException::fromPath(code, task_path); + { + if (index_update && (code == Coordination::Error::ZBADVERSION || code == Coordination::Error::ZNODEEXISTS)) + return false; + zkutil::KeeperMultiException::check(code, ops, responses); + } LOG_INFO(log, "Created export task {} of partition {} to {}, {} part(s)", - manifest.transaction_id, partition_id, dest_storage_id.getNameForLogs(), manifest.parts.size()); + manifest.transaction_id, request.partition_id, destination->getStorageID().getNameForLogs(), manifest.parts.size()); if (export_task_updating_task) export_task_updating_task->schedule(); + return true; } ExportReplicatedMergeTreeTaskManifest StorageReplicatedMergeTree::buildExportTaskManifest( @@ -8921,6 +8854,87 @@ void StorageReplicatedMergeTree::exportTTLTask() export_ttl_task->scheduleAfter(next_run_ms); } +void StorageReplicatedMergeTree::exportTTLIndexUpdatingTask() +{ + auto component_guard = Coordination::setCurrentComponent("StorageReplicatedMergeTree::exportTTLIndexUpdatingTask"); + try + { + const auto metadata_snapshot = getInMemoryMetadataPtr(getContext(), false); + export_ttl_index->refreshAndWatch( + getZooKeeper(), export_ttl_index_updating_task->getWatchCallback(), /* has_export_ttl */ !metadata_snapshot->getExportTTLs().empty()); + } + catch (const Coordination::Exception & e) + { + tryLogCurrentException(log, __PRETTY_FUNCTION__); + /// The restarting thread schedules this task again once there is a new session. + if (e.code == Coordination::Error::ZSESSIONEXPIRED) + { + restarting_thread.wakeup(); + return; + } + export_ttl_index_updating_task->scheduleAfter(QUEUE_UPDATE_ERROR_SLEEP_MS); + } + catch (...) + { + tryLogCurrentException(log, __PRETTY_FUNCTION__); + export_ttl_index_updating_task->scheduleAfter(QUEUE_UPDATE_ERROR_SLEEP_MS); + } +} + +namespace +{ + +MergeTreeData::ExportTaskStatus toExportTaskStatus(ExportReplicatedMergeTreeTaskEntry::Status status) +{ + using Status = ExportReplicatedMergeTreeTaskEntry::Status; + switch (status) + { + case Status::PENDING: return MergeTreeData::ExportTaskStatus::PENDING; + case Status::COMPLETED: return MergeTreeData::ExportTaskStatus::COMPLETED; + case Status::FAILED: return MergeTreeData::ExportTaskStatus::FAILED; + case Status::KILLED: return MergeTreeData::ExportTaskStatus::KILLED; + } + UNREACHABLE(); +} + +} + +std::optional StorageReplicatedMergeTree::getKnownExportTaskStatus(const String & transaction_id) const +{ + if (const auto tasks = export_partition_manifests.get()) + { + const auto & by_transaction_id = tasks->get(); + if (const auto it = by_transaction_id.find(transaction_id); it != by_transaction_id.end()) + return toExportTaskStatus(it->status); + } + return std::nullopt; +} + +std::optional StorageReplicatedMergeTree::getExportTaskStatus(const String & transaction_id) const +{ + /// The in-memory mirror of the tasks may lag behind Keeper, which only delays a retry. A task it + /// does not know yet, e.g. just created, is read from Keeper, so it is never taken for missing. + if (const auto status = getKnownExportTaskStatus(transaction_id)) + return status; + + const auto status_path = fs::path(zookeeper_path) / "exports" / transaction_id / "status"; + String status_str; + if (!getZooKeeper()->tryGet(status_path, status_str)) + return std::nullopt; + + if (const auto status = magic_enum::enum_cast(status_str)) + return toExportTaskStatus(*status); + + /// Treated as failed: the scheduler still checks whether it committed. + LOG_WARNING(log, "Export task {} has an unknown status {}", transaction_id, status_str); + return ExportTaskStatus::FAILED; +} + +bool StorageReplicatedMergeTree::isPartBeingMerged(const MergeTreePartInfo & part_info) const +{ + return queue.isGoingToBeMergedWithOtherParts(part_info); +} + void StorageReplicatedMergeTree::forgetPartition(const ASTPtr & partition, ContextPtr query_context) { auto component_guard = Coordination::setCurrentComponent("StorageReplicatedMergeTree::forgetPartition"); diff --git a/src/Storages/StorageReplicatedMergeTree.h b/src/Storages/StorageReplicatedMergeTree.h index 2e26423ae31f..778b6be66cee 100644 --- a/src/Storages/StorageReplicatedMergeTree.h +++ b/src/Storages/StorageReplicatedMergeTree.h @@ -380,8 +380,11 @@ class StorageReplicatedMergeTree final : public MergeTreeData std::vector getExportTasksInfo() const override; /// nullptr if partition export is disabled. - ReplicatedExportTTLIndexPtr getExportTTLIndex() const { return export_ttl_index; } - ExportFencePtr getLatestExportFence() const override { return export_ttl_index ? export_ttl_index->getLatest() : nullptr; } + ReplicatedExportTTLIndex * getExportTTLIndex() const override { return export_ttl_index.get(); } + + std::optional getExportTaskStatus(const String & transaction_id) const override; + std::optional getKnownExportTaskStatus(const String & transaction_id) const override; + bool isPartBeingMerged(const MergeTreePartInfo & part_info) const override; private: std::atomic_bool are_restoring_replica {false}; @@ -409,7 +412,6 @@ class StorageReplicatedMergeTree final : public MergeTreeData friend class ReplicatedMergeMutateTaskBase; friend class ReplicatedExportTaskUpdater; friend class ReplicatedExportTaskScheduler; - friend class ReplicatedExportTTLScheduler; using MergeStrategyPicker = ReplicatedMergeTreeMergeStrategyPicker; using LogEntry = ReplicatedMergeTreeLogEntry; @@ -546,6 +548,9 @@ class StorageReplicatedMergeTree final : public MergeTreeData /// Runs `export_ttl_scheduler`. BackgroundSchedulePoolTaskHolder export_ttl_task; + /// Keeps the copy of `export_ttl_index` up to date, woken by its watches. + BackgroundSchedulePoolTaskHolder export_ttl_index_updating_task; + /// A thread that removes old parts, log entries, and blocks. ReplicatedMergeTreeCleanupThread cleanup_thread; @@ -983,8 +988,11 @@ class StorageReplicatedMergeTree final : public MergeTreeData bool fetch_part, ContextPtr query_context) override; void forgetPartition(const ASTPtr & partition, ContextPtr query_context) override; - - void exportPartitionToTable(const PartitionCommand &, ContextPtr) override; + + /// Creates the task in one Keeper transaction. With an index update, the same transaction stores + /// the entry and checks the version of `/log` that the merge predicate read, so no merge of the + /// parts can be assigned between checking them and claiming them. + bool exportParts(const ExportPartsRequest & request, ContextPtr local_context, const ExportTTLIndexUpdate * index_update) override; /// Builds the descriptor of an export task of `parts` from the settings of `query_context`, and /// validates the destination for them. The caller sets the transaction id and the source. @@ -1006,6 +1014,8 @@ class StorageReplicatedMergeTree final : public MergeTreeData void exportTTLTask(); + void exportTTLIndexUpdatingTask(); + /// E.g. when a task of the `EXPORT` TTL finished, so the next group does not wait for the next check. void wakeUpExportTTL(); diff --git a/src/Storages/System/StorageSystemDistributedExports.cpp b/src/Storages/System/StorageSystemDistributedExports.cpp index 8ffe7a69ea0c..d85287f09576 100644 --- a/src/Storages/System/StorageSystemDistributedExports.cpp +++ b/src/Storages/System/StorageSystemDistributedExports.cpp @@ -69,8 +69,8 @@ ColumnsDescription StorageSystemDistributedExports::getColumnsDescription() "Per-part retry back-off local to this node: parts currently waiting before their next attempt, with attempt count and the next eligible time. Not shared across replicas; empty if no part is backing off."}, {"source", std::make_shared(), "What created the task: `query` for `ALTER TABLE ... EXPORT PARTITION`, `ttl` for the table's `TTL ... EXPORT TO TABLE` expression."}, - {"retry_of", std::make_shared(std::make_shared()), - "For a TTL export task: transaction ids of earlier tasks that failed to export some of this task's parts after exporting all of theirs, so their commit may still land. Its commit checks whether any of them landed at the destination after all. Empty otherwise."}, + {"commit_id", std::make_shared(), + "ID the task commits to the destination under. It is the transaction ID, except for a TTL export task that retries a failed one: it commits under the ID of the failed task, so the destination commits the rows once if the failed task landed after all."}, }; } @@ -202,12 +202,7 @@ void StorageSystemDistributedExports::fillData(MutableColumns & res_columns, Con res_columns[i++]->insert(backoff_array); res_columns[i++]->insert(info.source); - - Array retry_of_array; - retry_of_array.reserve(info.retry_of.size()); - for (const auto & transaction : info.retry_of) - retry_of_array.push_back(transaction); - res_columns[i++]->insert(retry_of_array); + res_columns[i++]->insert(info.commit_id); } } } diff --git a/src/Storages/System/StorageSystemTTLExports.cpp b/src/Storages/System/StorageSystemTTLExports.cpp index fef095e2e4f5..d1ae864c22e2 100644 --- a/src/Storages/System/StorageSystemTTLExports.cpp +++ b/src/Storages/System/StorageSystemTTLExports.cpp @@ -1,7 +1,6 @@ #include #include -#include #include #include #include @@ -27,8 +26,6 @@ ColumnsDescription StorageSystemTTLExports::getColumnsDescription() {"eligible_bytes", std::make_shared(), "Size on disk of the eligible parts."}, {"parts_held_by_delete_gate", std::make_shared(), "Number of parts whose delete or column TTL is due, but that are kept from merges until they are exported."}, - {"first_eligible_time", std::make_shared(), "When the scheduler first saw an eligible part of the partition that is not exported yet, zero if there is none."}, - {"next_group_time", std::make_shared(), "When the eligible parts are exported at the latest, zero if nothing waits."}, {"current_transaction_id", std::make_shared(), "Transaction id of the task exporting the partition now, see `system.distributed_exports`. Empty if there is none."}, {"last_error", std::make_shared(), "Error of the last attempt to export the partition, empty if it succeeded."}, {"scheduler_replica", std::make_shared(), @@ -71,8 +68,6 @@ void StorageSystemTTLExports::fillData(MutableColumns & res_columns, ContextPtr res_columns[i++]->insert(info.eligible_parts); res_columns[i++]->insert(info.eligible_bytes); res_columns[i++]->insert(info.parts_held_by_delete_gate); - res_columns[i++]->insert(static_cast(info.first_eligible_time)); - res_columns[i++]->insert(static_cast(info.next_group_time)); res_columns[i++]->insert(info.current_transaction_id); res_columns[i++]->insert(info.last_error); res_columns[i++]->insert(info.scheduler_replica); diff --git a/src/Storages/System/attachSystemTables.cpp b/src/Storages/System/attachSystemTables.cpp index 81d24e558481..b6c7e9a1d7f9 100644 --- a/src/Storages/System/attachSystemTables.cpp +++ b/src/Storages/System/attachSystemTables.cpp @@ -271,7 +271,7 @@ void attachSystemTablesServer(ContextPtr context, IDatabase & system_database, b if (context->getServerSettings()[ServerSetting::allow_experimental_export_merge_tree_partition]) { attach(context, system_database, "distributed_exports", "Contains the export tasks of MergeTree tables, both plain and replicated, created by `EXPORT PARTITION` or by a `TTL ... EXPORT` expression, and their progress. Each task is represented by a single row."); - attach(context, system_database, "ttl_exports", "Contains the state of the `TTL ... EXPORT TO TABLE` expression of MergeTree tables, one row per partition: exported, claimed and eligible parts, when the next group is exported, and the last error."); + attach(context, system_database, "ttl_exports", "Contains the state of the `TTL ... EXPORT TO TABLE` expression of MergeTree tables, one row per partition: exported, claimed and eligible parts, the task exporting it, and the last error."); } attach(context, system_database, "mutations", "Contains a list of mutations and their progress. Each mutation command is represented by a single row."); attachNoDescription(context, system_database, "replicas", "Contains information and status of all table replicas on current server. Each replica is represented by a single row."); diff --git a/tests/integration/test_export_merge_tree_part_to_iceberg/test.py b/tests/integration/test_export_merge_tree_part_to_iceberg/test.py index 4920c9c1933a..00b9b6c74d94 100644 --- a/tests/integration/test_export_merge_tree_part_to_iceberg/test.py +++ b/tests/integration/test_export_merge_tree_part_to_iceberg/test.py @@ -11,6 +11,7 @@ test_export_part_all_iceberg_types – schema covering all major Iceberg data types test_export_multiple_parts_to_iceberg – two parts from different partitions land together test_export_part_with_year_transform_partition – toYearNumSinceEpoch() partition expression + test_export_part_without_rows – a part emptied by a mutation exports as no files test_export_part_with_bucket_partition – icebergBucket(N, col) partition expression test_export_part_partition_key_mismatch_is_rejected – mismatched partition spec rejected synchronously test_export_part_multi_column_partition_key_success – composite (a, b, c) partition key round-trips @@ -399,6 +400,52 @@ def test_export_part_with_year_transform_partition(cluster): node.query(f"DROP TABLE IF EXISTS {iceberg}") +def test_export_part_without_rows(cluster): + """ + A part that a mutation emptied, kept by `remove_empty_parts = 0`, has no min/max index. Exporting + it succeeds without writing or committing anything, although the destination partition, a year, + is proven from the min/max index for a source partition, a month, with rows. + """ + node = cluster.instances["node1"] + sfx = unique_suffix() + mt = f"mt_no_rows_{sfx}" + iceberg = f"iceberg_no_rows_{sfx}" + + cols = "id Int64, event_date Date" + make_mt(node, mt, cols, "toYYYYMM(event_date)", order_by="id", extra_settings="remove_empty_parts = 0") + make_iceberg_s3(node, iceberg, cols, "toYearNumSinceEpoch(event_date)") + + node.query(f"INSERT INTO {mt} VALUES (1, '2020-01-10')") + node.query(f"ALTER TABLE {mt} DELETE WHERE 1", settings={"mutations_sync": 2}) + part = get_part(node, mt, "202001") + rows = node.query( + f"SELECT rows FROM system.parts WHERE database = currentDatabase() AND table = '{mt}' AND name = '{part}'" + ).strip() + assert rows == "0", f"Expected part {part} without rows, got {rows}" + + snapshots = int(node.query( + f"SELECT count() FROM system.iceberg_history WHERE database = currentDatabase() AND table = '{iceberg}'" + ).strip()) + + export_part(node, mt, part, iceberg) + wait_for_export_part(node, mt, part) + + errors = node.query( + f"SELECT error FROM system.part_log WHERE event_type = 'ExportPart' " + f"AND database = currentDatabase() AND table = '{mt}' AND part_name = '{part}'" + ).split() + assert errors == ["0"], f"Expected one successful export of part {part}, got errors {errors}" + + count = int(node.query(f"SELECT count() FROM {iceberg}").strip()) + assert count == 0, f"Expected the Iceberg table to stay empty, got {count} rows" + assert int(node.query( + f"SELECT count() FROM system.iceberg_history WHERE database = currentDatabase() AND table = '{iceberg}'" + ).strip()) == snapshots, "An export without files committed a snapshot" + + node.query(f"DROP TABLE IF EXISTS {mt} SYNC") + node.query(f"DROP TABLE IF EXISTS {iceberg}") + + def test_export_part_partition_column_lossless_widening(cluster): """A lossless widening of a partition column (year Int32 -> Int64) round-trips.""" node = cluster.instances["node1"] diff --git a/tests/integration/test_export_partition_to_iceberg/test_lifecycle.py b/tests/integration/test_export_partition_to_iceberg/test_lifecycle.py index 0bb14bfed4b5..98ec3f38da29 100644 --- a/tests/integration/test_export_partition_to_iceberg/test_lifecycle.py +++ b/tests/integration/test_export_partition_to_iceberg/test_lifecycle.py @@ -14,7 +14,8 @@ CLUSTER_INSTANCES = ["replica1"] # The happy paths of `EXPORT PARTITION` into an Iceberg destination: one partition, several -# partitions, all of them, and the column statistics carried by the resulting manifest entry. +# partitions, all of them, partitions whose rows were deleted, and the column statistics carried by +# the resulting manifest entry. # --------------------------------------------------------------------------- @@ -186,6 +187,55 @@ def test_export_partition_where_every_row_is_deleted(cluster, source_engine): assert count == 0, f"Expected the Iceberg table to stay empty, got {count} rows" +def iceberg_snapshots(node, table: str) -> int: + return int(node.query( + f"SELECT count() FROM system.iceberg_history WHERE database = currentDatabase() AND table = '{table}'" + ).strip()) + + +def test_export_partition_of_parts_without_rows(cluster, source_engine): + """ + A mutation that deletes every row of a part leaves a part without rows, kept by + `remove_empty_parts = 0`. Unlike after a lightweight delete, such a part has no min/max index. + The export must not need it: the source partition is a month, and the destination one a year, + which only the min/max index of the parts with rows can prove the month to lie in. A partition + of only such parts is exported as no files, so the task reaches COMPLETED without committing. + """ + node = cluster.instances["replica1"] + + uid = unique_suffix() + mt_table = f"mt_no_rows_{uid}" + iceberg_table = f"iceberg_no_rows_{uid}" + columns = "id Int64, event_date Date" + + make_source( + node, mt_table, columns, "toYYYYMM(event_date)", + engine=source_engine, replica_name="replica1", + extra_settings="remove_empty_parts = 0, max_bytes_to_merge_at_max_space_in_pool = 1", + ) + make_iceberg_s3(node, iceberg_table, columns, partition_by="toYearNumSinceEpoch(event_date)") + + node.query(f"INSERT INTO {mt_table} VALUES (1, '2020-01-10')") + node.query(f"INSERT INTO {mt_table} VALUES (2, '2020-01-20')") + node.query(f"ALTER TABLE {mt_table} DELETE WHERE 1", settings={"mutations_sync": 2}) + + rows = node.query( + f"SELECT rows FROM system.parts WHERE database = currentDatabase() AND table = '{mt_table}' AND active" + ).split() + assert rows == ["0", "0"], f"Expected two parts without rows, got {rows}" + + snapshots = iceberg_snapshots(node, iceberg_table) + node.query( + f"ALTER TABLE {mt_table} EXPORT PARTITION ID '202001' TO TABLE {iceberg_table}", + settings={"allow_insert_into_iceberg": 1}, + ) + wait_for_export_status(node, mt_table, iceberg_table, "202001", "COMPLETED") + + count = int(node.query(f"SELECT count() FROM {iceberg_table}").strip()) + assert count == 0, f"Expected the Iceberg table to stay empty, got {count} rows" + assert iceberg_snapshots(node, iceberg_table) == snapshots, "An export without files committed a snapshot" + + def setup_stats_tables(node, mt_table: str, iceberg_table: str, engine: str = "ReplicatedMergeTree"): """Local variant of setup_tables using the wider schema with a Nullable column.""" columns = "id Int32, name String, tag Nullable(String), year Int32" diff --git a/tests/integration/test_export_ttl/common.py b/tests/integration/test_export_ttl/common.py index 7a94c0bed6c1..06d5927d2278 100644 --- a/tests/integration/test_export_ttl/common.py +++ b/tests/integration/test_export_ttl/common.py @@ -18,8 +18,6 @@ # A group is shipped on the first check after its parts are due. FAST_TTL_SETTINGS = { "ttl_export_check_period_seconds": 1, - "ttl_export_batch_window_seconds": 0, - "ttl_export_batch_max_delay_seconds": 0, } PAUSE_EXPORT_FAILPOINT = "export_part_pause_before_schema_validation" @@ -122,13 +120,6 @@ def wait_for_same_ttl_rows(replicas, table, settled, timeout=90, columns=None): time.sleep(0.5) -def first_eligible_time(node, table, partition_id): - """When the scheduler first saw an eligible part of the partition that is not exported, 0 if never.""" - return int(node.query( - f"SELECT toUnixTimestamp(first_eligible_time) FROM system.ttl_exports WHERE table = '{table}' AND partition_id = '{partition_id}'" - ).strip() or 0) - - def partition_settled(rows, partition_id, exported=None): """Nothing of the partition is claimed or waiting to be exported.""" row = rows.get(partition_id) @@ -140,8 +131,8 @@ def partition_settled(rows, partition_id, exported=None): def wait_for_partitions_exported(node, table, partition_ids, timeout=90, exported=None): """Wait until nothing of *partition_ids* is claimed or eligible and no TTL task is in flight. - A check that runs before a part is due already publishes that partition with nothing eligible, so - pass *exported* when the wait must see the parts recorded as exported rather than merely not due. + Before a part is due its partition already shows nothing eligible, so pass *exported* when the + wait must see the parts recorded as exported rather than merely not due. """ def settled(): rows = ttl_rows(node, table) @@ -165,10 +156,10 @@ def has_error(): def ttl_tasks(node, table): rows = query_json( node, - f"SELECT transaction_id, partition_id, status, parts, retry_of FROM system.distributed_exports" + f"SELECT transaction_id, partition_id, status, parts, commit_id FROM system.distributed_exports" f" WHERE source_table = '{table}' AND source = 'ttl' ORDER BY create_time, transaction_id", ) - return [dict(zip(["transaction_id", "partition_id", "status", "parts", "retry_of"], row)) for row in rows] + return [dict(zip(["transaction_id", "partition_id", "status", "parts", "commit_id"], row)) for row in rows] def pending_ttl_tasks(node, table): @@ -234,12 +225,22 @@ def iceberg_snapshots(node, table): ).strip()) +def completed_ttl_tasks_with_files(node, table): + """Transaction ids of the completed TTL tasks that wrote files. A task of parts without rows + writes none, so it has nothing to commit.""" + return node.query( + f"SELECT transaction_id FROM system.distributed_exports" + f" WHERE source_table = '{table}' AND source = 'ttl' AND status = 'COMPLETED'" + f" AND arrayExists(paths -> notEmpty(paths), mapValues(destination_file_paths))" + ).split() + + def assert_one_snapshot_per_task(node, mt_table, iceberg_table): - """Every completed TTL task committed one snapshot since the destination was created, and no - other task committed.""" + """Every completed TTL task that wrote files committed one snapshot since the destination was + created, and no other task committed.""" snapshots = iceberg_snapshots(node, iceberg_table) - _snapshots_at_creation.get(iceberg_table, 0) - completed = completed_ttl_tasks(node, mt_table) - assert snapshots == len(completed), f"{snapshots} snapshots for {len(completed)} completed tasks: {ttl_tasks(node, mt_table)}" + committing = completed_ttl_tasks_with_files(node, mt_table) + assert snapshots == len(committing), f"{snapshots} snapshots for {len(committing)} completed tasks with files: {ttl_tasks(node, mt_table)}" def _partition_scalar(value): diff --git a/tests/integration/test_export_ttl/test_failures.py b/tests/integration/test_export_ttl/test_failures.py index 9d94cad95e61..0f36d68f85b6 100644 --- a/tests/integration/test_export_ttl/test_failures.py +++ b/tests/integration/test_export_ttl/test_failures.py @@ -25,9 +25,9 @@ CLUSTER_INSTANCES = ["replica1"] # Failures of the tasks of the `EXPORT` TTL to an Iceberg destination: a failed group is retried as a -# new task, which lists in `retry_of` the failed ones that exported all their parts, since only those -# may have committed. Only the task that completes commits a snapshot, and no row lands twice, -# whatever fails and when. +# new task with exactly its parts and the same `commit_id`, so that the destination commits them once +# even if the failed task committed after all. Only the task that completes commits a snapshot, and no +# row lands twice, whatever fails and when. def make_tables(node, engine, settings=None, columns=COLUMNS, source_partition_by="year", spec="year"): @@ -66,7 +66,7 @@ def test_retryable_part_error_is_retried_by_the_same_task(cluster, source_engine def test_non_retryable_part_error_is_retried_as_a_new_task(cluster, source_engine): - """The failed task did not export its part, so it did not commit: the retry does not check it.""" + """Every retry of the failed task commits under the commit id of the first one.""" node = cluster.instances["replica1"] mt_table, iceberg_table = make_tables(node, source_engine) @@ -85,14 +85,15 @@ def test_non_retryable_part_error_is_retried_as_a_new_task(cluster, source_engin failed = {task["transaction_id"] for task in tasks if task["status"] == "FAILED"} completed = completed_ttl_tasks(node, mt_table) assert failed and len(completed) == 1, tasks - assert completed[0]["retry_of"] == [], (completed, failed) + commit_ids = {task["commit_id"] for task in tasks} + assert len(commit_ids) == 1 and commit_ids <= failed, tasks assert_exactly_once(iceberg_ids(node, iceberg_table), [1]) assert_one_snapshot_per_task(node, mt_table, iceberg_table) -def test_commit_failure_is_retried_with_the_new_parts(cluster): - """A group whose commit keeps failing times out, and is retried as a new task that also takes the - parts that became due meanwhile, and records the failed task in `retry_of`. Only the retry commits.""" +def test_commit_failure_is_retried_with_the_same_parts(cluster): + """A group whose commit keeps failing times out, and is retried as a new task with exactly its part + and its commit id. The part that became due meanwhile is exported by the next group.""" node = cluster.instances["replica1"] mt_table, iceberg_table = make_tables(node, "ReplicatedMergeTree", settings={"ttl_export_settings_profile": "ttl_export_fail_fast"}) node.query(f"SYSTEM STOP MERGES {mt_table}") @@ -114,7 +115,8 @@ def test_commit_failure_is_retried_with_the_new_parts(cluster): wait_for_partitions_exported(node, mt_table, ["2020"]) completed = completed_ttl_tasks(node, mt_table) - assert [(len(task["parts"]), task["retry_of"]) for task in completed] == [(2, [failed_transaction_id])], completed + assert sorted((len(task["parts"]), task["commit_id"] == failed_transaction_id) for task in completed) == [(1, False), (1, True)], completed + assert all(task["commit_id"] in (failed_transaction_id, task["transaction_id"]) for task in completed), completed assert_exactly_once(iceberg_ids(node, iceberg_table), [1, 2]) assert_one_snapshot_per_task(node, mt_table, iceberg_table) @@ -138,7 +140,7 @@ def test_commit_that_landed_is_not_committed_again(cluster): def test_killed_task_is_retried(cluster, source_engine): """The killed task never commits; the file its part export writes afterwards is not in the table. - It was killed before exporting its part, so the retry does not check it.""" + The retry commits its part under the commit id of the killed task.""" node = cluster.instances["replica1"] mt_table, iceberg_table = make_tables(node, source_engine) @@ -153,7 +155,7 @@ def test_killed_task_is_retried(cluster, source_engine): wait_for_partitions_exported(node, mt_table, ["2020"]) completed = completed_ttl_tasks(node, mt_table) - assert len(completed) == 1 and completed[0]["retry_of"] == [], (killed, ttl_tasks(node, mt_table)) + assert len(completed) == 1 and completed[0]["commit_id"] == killed, (killed, ttl_tasks(node, mt_table)) assert_exactly_once(iceberg_ids(node, iceberg_table), [1]) assert_one_snapshot_per_task(node, mt_table, iceberg_table) @@ -216,11 +218,10 @@ def test_dropped_destination(cluster, source_engine): def test_partition_failing_its_check_does_not_block_the_others(cluster, source_engine): """A group whose rows fall into several days of a day-partitioned Iceberg table is not exported, - writes nothing, and does not count against `ttl_export_max_concurrent_groups`.""" + writes nothing, and the other partitions are exported.""" node = cluster.instances["replica1"] mt_table, iceberg_table = make_tables( node, source_engine, columns="id Int64, t DateTime", source_partition_by="toYYYYMM(t)", spec="toRelativeDayNum(t)", - settings={"ttl_export_max_concurrent_groups": 1}, ) node.query(f"INSERT INTO {mt_table} VALUES (1, '2020-01-01 10:00:00'), (2, '2020-01-02 10:00:00')") diff --git a/tests/integration/test_export_ttl/test_iceberg_commits.py b/tests/integration/test_export_ttl/test_iceberg_commits.py index 2605db4c1e20..3f1957e56e0c 100644 --- a/tests/integration/test_export_ttl/test_iceberg_commits.py +++ b/tests/integration/test_export_ttl/test_iceberg_commits.py @@ -70,27 +70,9 @@ def test_group_is_one_snapshot(cluster, source_engine): assert_exactly_once(iceberg_ids(node, iceberg_table), range(3)) -def test_groups_limited_in_size_are_separate_snapshots(cluster, source_engine): - node = cluster.instances["replica1"] - mt_table, iceberg_table = make_tables(node, source_engine, settings={"ttl_export_max_parts_per_group": 2}) - snapshots = iceberg_snapshots(node, iceberg_table) - - node.query(f"SYSTEM STOP MERGES {mt_table}") - node.query(f"SYSTEM STOP MOVES {mt_table}") - for i in range(5): - node.query(f"INSERT INTO {mt_table} VALUES ({i}, 2020, {DUE})") - node.query(f"SYSTEM START MOVES {mt_table}") - - wait_until(lambda: len(completed_ttl_tasks(node, mt_table)) == 3, 120, "The partition was not exported in three groups") - wait_for_partitions_exported(node, mt_table, ["2020"]) - assert sorted(len(task["parts"]) for task in ttl_tasks(node, mt_table)) == [1, 2, 2] - assert iceberg_snapshots(node, iceberg_table) == snapshots + 3 - assert_exactly_once(iceberg_ids(node, iceberg_table), range(5)) - - def test_concurrent_groups_commit_to_one_table(cluster, source_engine): node = cluster.instances["replica1"] - mt_table, iceberg_table = make_tables(node, source_engine, settings={"ttl_export_max_concurrent_groups": 4}) + mt_table, iceberg_table = make_tables(node, source_engine) snapshots = iceberg_snapshots(node, iceberg_table) node.query(f"SYSTEM STOP MOVES {mt_table}") diff --git a/tests/integration/test_export_ttl/test_merge_fence.py b/tests/integration/test_export_ttl/test_merge_fence.py index d8e024868e80..f7911d34d4ca 100644 --- a/tests/integration/test_export_ttl/test_merge_fence.py +++ b/tests/integration/test_export_ttl/test_merge_fence.py @@ -11,6 +11,7 @@ assert_never_merged, assert_one_snapshot_per_task, completed_ttl_tasks, + completed_ttl_tasks_with_files, create_iceberg, create_source, group_in_flight, @@ -147,9 +148,11 @@ def test_exported_parts_merge_with_each_other(cluster, source_engine): assert_one_snapshot_per_task(node, mt_table, iceberg_table) -def test_eligible_parts_merge_while_the_group_is_collected(cluster, source_engine): +def test_eligible_parts_merge_before_they_are_shipped(cluster, source_engine): + """Due parts that are not claimed merge as usual, and the merged part is shipped.""" node = cluster.instances["replica1"] - mt_table, iceberg_table = make_tables(node, source_engine, settings={"ttl_export_batch_window_seconds": 8, "ttl_export_batch_max_delay_seconds": 600}) + mt_table, iceberg_table = make_tables(node, source_engine) + node.query(f"SYSTEM STOP MOVES {mt_table}") node.query(f"SYSTEM STOP MERGES {mt_table}") node.query(f"INSERT INTO {mt_table} VALUES (1, 2020, {DUE})") node.query(f"INSERT INTO {mt_table} VALUES (2, 2020, {DUE})") @@ -158,6 +161,7 @@ def test_eligible_parts_merge_while_the_group_is_collected(cluster, source_engin merged = active_parts(node, mt_table) assert len(merged) == 1 + node.query(f"SYSTEM START MOVES {mt_table}") wait_for_partitions_exported(node, mt_table, ["2020"]) tasks = ttl_tasks(node, mt_table) assert len(tasks) == 1 and tasks[0]["parts"] == merged, tasks @@ -221,9 +225,10 @@ def test_mutation_keeps_parts_exported(cluster, source_engine): def test_parts_without_rows_are_recorded_as_exported(cluster, source_engine): - """A part that a mutation emptied, kept by `remove_empty_parts = 0`, has nothing to export, but its - group records it as exported: otherwise it would stay apart from the exported parts around it, and - the partition could never be merged into one part. A group of such parts only has no task.""" + """A part that a mutation emptied, kept by `remove_empty_parts = 0`, has nothing to export, but it + is exported with its group, writing nothing, so it is recorded as exported: otherwise it would stay + apart from the exported parts around it, and the partition could never be merged into one part. + A group of such parts only is a task that commits nothing.""" node = cluster.instances["replica1"] # No merge is assigned until the end, so the emptied part stays between the others. mt_table, iceberg_table = make_tables( @@ -240,13 +245,16 @@ def test_parts_without_rows_are_recorded_as_exported(cluster, source_engine): wait_for_partitions_exported(node, mt_table, ["2020"], exported=3) tasks = ttl_tasks(node, mt_table) - assert len(tasks) == 1 and len(tasks[0]["parts"]) == 2, tasks + assert len(tasks) == 1 and len(tasks[0]["parts"]) == 3, tasks assert_exactly_once(iceberg_ids(node, iceberg_table), [1, 3]) node.query(f"INSERT INTO {mt_table} VALUES (4, 2020, {NOT_DUE})") node.query(f"ALTER TABLE {mt_table} DELETE WHERE id = 4", settings={"mutations_sync": 2}) wait_for_partitions_exported(node, mt_table, ["2020"], exported=4) - assert len(ttl_tasks(node, mt_table)) == 1 + tasks = ttl_tasks(node, mt_table) + assert sorted(len(task["parts"]) for task in tasks) == [1, 3], tasks + assert all(task["status"] == "COMPLETED" for task in tasks), tasks + assert completed_ttl_tasks_with_files(node, mt_table) == [task["transaction_id"] for task in tasks if len(task["parts"]) == 3] node.query(f"ALTER TABLE {mt_table} RESET SETTING max_bytes_to_merge_at_max_space_in_pool") node.query(f"OPTIMIZE TABLE {mt_table} PARTITION ID '2020' FINAL") diff --git a/tests/integration/test_export_ttl/test_replication.py b/tests/integration/test_export_ttl/test_replication.py index af147659cbb6..76f0ed56a07a 100644 --- a/tests/integration/test_export_ttl/test_replication.py +++ b/tests/integration/test_export_ttl/test_replication.py @@ -12,7 +12,6 @@ completed_ttl_tasks, create_iceberg, create_source, - first_eligible_time, iceberg_ids, pending_ttl_tasks, scheduler_holder, @@ -29,9 +28,9 @@ CLUSTER_INSTANCES = ["replica1", "replica2"] # One replica of a `ReplicatedMergeTree` table schedules the groups of the `EXPORT` TTL. Every -# replica tracks the batching windows of its parts and shows the same `system.ttl_exports`, except -# for the error of acting on a partition, and another replica takes over where the previous one left -# off. Every replica has the same Iceberg destination. +# replica shows the same `system.ttl_exports`, except for the error of acting on a partition, and +# another replica takes over where the previous one left off. Every replica has the same Iceberg +# destination. def make_replicated_tables(replicas, columns=COLUMNS, partition_by="year", spec="year", settings=None): @@ -142,38 +141,75 @@ def test_index_snapshot_is_cached(cluster): assert time.time() - start < 60, f"The index snapshot is refreshed while idle: {before} -> {after} on {replica.name}" -def test_failover_resumes_the_batch(cluster): - """Every replica tracks the batching window of the parts it has, so the replica that takes over - continues it instead of starting it again, and nothing is exported twice.""" +def keeper_requests(node, query): + """The result of *query* on *node*, and the number of Keeper requests it made.""" + query_id = f"keeper_requests_{unique_suffix()}" + result = node.query(query, query_id=query_id) + node.query("SYSTEM FLUSH LOGS query_log") + requests = int(node.query( + f"SELECT ProfileEvents['ZooKeeperTransactions'] FROM system.query_log" + f" WHERE query_id = '{query_id}' AND type = 'QueryFinish'" + ).strip()) + return result, requests + + +def test_rows_are_shown_without_reading_keeper(cluster): + """Every replica follows the export index and the scheduler lock with Keeper watches, so + `system.ttl_exports` reads nothing from Keeper, on the replica that schedules and on the others, + and a table attached again learns the scheduler that took over meanwhile.""" replicas = [cluster.instances["replica1"], cluster.instances["replica2"]] - mt_table, iceberg_table = make_replicated_tables( - replicas, settings={"ttl_export_batch_window_seconds": 60, "ttl_export_batch_max_delay_seconds": 600} - ) + mt_table, iceberg_table = make_replicated_tables(replicas) + holder, other = holder_and_other(replicas, mt_table) + + holder.query(f"INSERT INTO {mt_table} VALUES (1, 2020, {DUE})") + sync(replicas, mt_table) + rows = wait_for_same_ttl_rows(replicas, mt_table, lambda rows: rows.get("2020", {}).get("exported_parts") == 1) + assert rows["2020"]["scheduler_replica"] == holder.name, rows + for replica in replicas: + result, requests = keeper_requests(replica, f"SELECT * FROM system.ttl_exports WHERE table = '{mt_table}'") + assert result and requests == 0, f"{requests} Keeper requests on {replica.name}" + + holder.query(f"DETACH TABLE {mt_table}") + try: + wait_until(lambda: scheduler_holder(other, mt_table) == other.name, 60, "The other replica did not take over") + finally: + holder.query(f"ATTACH TABLE {mt_table}") + + def shows_the_new_scheduler(): + rows = ttl_rows(holder, mt_table) + return rows if rows.get("2020", {}).get("scheduler_replica") == other.name else None + + rows = wait_until(shows_the_new_scheduler, 60, "The attached table does not show the replica that took over") + assert rows["2020"]["exported_parts"] == 1, rows + _, requests = keeper_requests(holder, f"SELECT * FROM system.ttl_exports WHERE table = '{mt_table}'") + assert requests == 0, f"{requests} Keeper requests on {holder.name}" + assert_exactly_once(iceberg_ids(holder, iceberg_table), [1]) + + +def test_failover_exports_what_the_scheduler_left(cluster): + """The replica that takes over exports the due parts that the previous scheduler did not, and + nothing is exported twice.""" + replicas = [cluster.instances["replica1"], cluster.instances["replica2"]] + mt_table, iceberg_table = make_replicated_tables(replicas) holder, other = holder_and_other(replicas, mt_table) + holder.query(f"SYSTEM STOP MOVES {mt_table}") holder.query(f"INSERT INTO {mt_table} VALUES (1, 2020, {DUE})") sync(replicas, mt_table) - first_eligible = wait_until(lambda: first_eligible_time(other, mt_table, "2020"), 60, "The other replica does not track the batch") - # The replicas saw the part at their own checks, after it was fetched. - assert abs(first_eligible - first_eligible_time(holder, mt_table, "2020")) <= 5 + time.sleep(3) + assert ttl_tasks(other, mt_table) == [] holder.stop_clickhouse(kill=True) try: wait_until(lambda: scheduler_holder(other, mt_table) == other.name, 120, "The other replica did not take over") - wait_until(lambda: ttl_rows(other, mt_table).get("2020", {}).get("scheduler_replica") == other.name, 60, - "The rows do not show the new scheduler") - assert first_eligible_time(other, mt_table, "2020") == first_eligible, "The batch was restarted by the new scheduler" - assert ttl_tasks(other, mt_table) == [] + wait_for_partitions_exported(other, mt_table, ["2020"], timeout=120, exported=1) + assert len(completed_ttl_tasks(other, mt_table)) == 1 finally: holder.start_clickhouse() - other.query(f"INSERT INTO {mt_table} VALUES (2, 2020, {DUE})") - other.query(f"ALTER TABLE {mt_table} MODIFY SETTING ttl_export_batch_window_seconds = 0") sync(replicas, mt_table) - wait_for_partitions_exported(other, mt_table, ["2020"], timeout=120) - assert len(completed_ttl_tasks(other, mt_table)) == 1 for replica in replicas: - assert_exactly_once(iceberg_ids(replica, iceberg_table), [1, 2]) + assert_exactly_once(iceberg_ids(replica, iceberg_table), [1]) assert_one_snapshot_per_task(other, mt_table, iceberg_table) diff --git a/tests/integration/test_export_ttl/test_rolling_exports.py b/tests/integration/test_export_ttl/test_rolling_exports.py index 561f07c6791a..3b19edd24b1e 100644 --- a/tests/integration/test_export_ttl/test_rolling_exports.py +++ b/tests/integration/test_export_ttl/test_rolling_exports.py @@ -30,12 +30,9 @@ TTL_SECONDS = 10 INSERT_SECONDS = 60 # How long after it is due a row may take to be read from the destination when parts are not merged: -# the maximum batch delay, a group in flight and the next group, with room for sanitizer builds. The -# rows due in the first `INSERT_SECONDS - MAX_LAG_SECONDS` seconds are thus exported while rows keep -# coming. +# the check period, a group in flight and the next group, with room for sanitizer builds. The rows due +# in the first `INSERT_SECONDS - MAX_LAG_SECONDS` seconds are thus exported while rows keep coming. MAX_LAG_SECONDS = 40 -# Parts keep becoming due, so a group is shipped by the window or, while it does not close, by the maximum delay. -BATCH = {"ttl_export_batch_window_seconds": 3, "ttl_export_batch_max_delay_seconds": 6} NO_MERGES = {"max_bytes_to_merge_at_max_space_in_pool": 1} @@ -46,7 +43,7 @@ def make_tables(nodes, engine, settings=None): for node in nodes: create_source( node, mt_table, COLUMNS, "customer_id", f"t + INTERVAL {TTL_SECONDS} SECOND EXPORT TO TABLE {iceberg_table}", - engine=engine, replica_name=node.name, settings={**BATCH, **(settings or {})}, + engine=engine, replica_name=node.name, settings=settings, ) return mt_table, iceberg_table diff --git a/tests/integration/test_export_ttl/test_scheduling.py b/tests/integration/test_export_ttl/test_scheduling.py index fe4e5314c8ae..f844582b5ca6 100644 --- a/tests/integration/test_export_ttl/test_scheduling.py +++ b/tests/integration/test_export_ttl/test_scheduling.py @@ -12,12 +12,10 @@ completed_ttl_tasks, create_iceberg, create_source, - failpoint, - first_eligible_time, + group_in_flight, iceberg_ids, iceberg_orphan_files, iceberg_snapshots, - pending_ttl_tasks, snapshot_refreshes, ttl_rows, ttl_tasks, @@ -27,9 +25,9 @@ CLUSTER_INSTANCES = ["replica1"] -# When the `EXPORT` TTL ships groups of parts to an Iceberg destination: on the checks after their -# parts are due, batched by the window, the maximum delay and the size threshold, limited in size and -# concurrency, one snapshot per group, and never twice however many checks go by. +# When the `EXPORT` TTL ships groups of parts to an Iceberg destination: every check ships the parts +# that are due by then, one group per partition, a partition has at most one group being exported, +# every group is one snapshot, and no part is exported twice however many checks go by. def due_in(seconds): @@ -75,167 +73,65 @@ def test_exports_due_parts_once(cluster, source_engine): # The part that is not due stays in the source only. row = ttl_rows(node, mt_table)["2021"] assert row["eligible_parts"] == 0 and row["exported_parts"] == 0, row - assert [task["retry_of"] for task in ttl_tasks(node, mt_table)] == [[], []] + assert all(task["commit_id"] == task["transaction_id"] for task in ttl_tasks(node, mt_table)), ttl_tasks(node, mt_table) -def test_part_becoming_due_wakes_the_scheduler(cluster, source_engine): - """A check computes when the next part becomes due, and the scheduler wakes up then instead of - waiting for the check period.""" +def test_due_parts_ship_on_the_next_check(cluster, source_engine): + """The check period is the batch interval: a check ships every part that is due by then, one + group per partition, and a part that is not due yet waits for a later check.""" node = cluster.instances["replica1"] - # The batch window is disabled so the wake at the due time exports the part. The default window - # would hold it for another minute, past the bound that distinguishes this wake from the check period. - mt_table, iceberg_table = make_tables( - node, source_engine, - settings={"ttl_export_check_period_seconds": 120, "ttl_export_batch_window_seconds": 0, "ttl_export_batch_max_delay_seconds": 0}, - ) - - node.query(f"INSERT INTO {mt_table} VALUES (1, 2020, {due_in(12)})") - inserted = time.time() - # The table starts with a check, which sees the part that is not due yet. - node.query(f"DETACH TABLE {mt_table}") - node.query(f"ATTACH TABLE {mt_table}") - - while time.time() - inserted < 9: - assert ttl_tasks(node, mt_table) == [], "The part was exported before it was due" - time.sleep(1) - - # That check already shows the partition as idle, so wait until the part is recorded as exported. - wait_for_partitions_exported(node, mt_table, ["2020"], timeout=40, exported=1) - assert time.time() - inserted < 40, "The part was exported only by a periodic check" - assert_exactly_once(iceberg_ids(node, iceberg_table), [1]) - assert_one_snapshot_per_task(node, mt_table, iceberg_table) - - -def test_batching_window_spans_checks(cluster, source_engine): - """Parts that keep becoming due within the window are collected over several checks into one - group, which is committed as one snapshot once no new part came for the window.""" - node = cluster.instances["replica1"] - mt_table, iceberg_table = make_tables( - node, source_engine, - settings={"ttl_export_batch_window_seconds": 45, "ttl_export_batch_max_delay_seconds": 600, "ttl_export_batch_min_bytes": 0}, - ) - - first_eligible = set() - for i in range(5): - node.query(f"INSERT INTO {mt_table} VALUES ({i}, 2020, {DUE})") - assert ttl_tasks(node, mt_table) == [], "A group was shipped while parts kept coming" - first_eligible.add(first_eligible_time(node, mt_table, "2020")) - first_eligible.discard(0) - - assert len(first_eligible) == 1, f"The first eligible time of the group changed across checks: {first_eligible}" - - wait_for_partitions_exported(node, mt_table, ["2020"], timeout=120) - tasks = ttl_tasks(node, mt_table) - assert len(tasks) == 1 and len(tasks[0]["parts"]) == 5, tasks - assert_exactly_once(iceberg_ids(node, iceberg_table), range(5)) - assert_one_snapshot_per_task(node, mt_table, iceberg_table) - - -def test_maximum_delay_ships_while_parts_keep_coming(cluster, source_engine): - node = cluster.instances["replica1"] - mt_table, iceberg_table = make_tables( - node, source_engine, - settings={"ttl_export_batch_window_seconds": 15, "ttl_export_batch_max_delay_seconds": 25, "ttl_export_batch_min_bytes": 0}, - ) - - start = time.time() - shipped_at = None - i = 0 - while time.time() - start < 45: - node.query(f"INSERT INTO {mt_table} VALUES ({i}, 2020, {DUE})") - i += 1 - if shipped_at is None and ttl_tasks(node, mt_table): - shipped_at = time.time() - start - - assert shipped_at is not None, "No group was shipped although parts waited longer than the maximum delay" - assert shipped_at >= 20, f"A group was shipped after {shipped_at:.1f} s, before the maximum delay" - - wait_for_partitions_exported(node, mt_table, ["2020"], timeout=90) - assert_exactly_once(iceberg_ids(node, iceberg_table), range(i)) - assert_one_snapshot_per_task(node, mt_table, iceberg_table) - - -def test_size_threshold_ships_before_the_window(cluster, source_engine): - node = cluster.instances["replica1"] - mt_table, iceberg_table = make_tables( - node, source_engine, - settings={"ttl_export_batch_window_seconds": 600, "ttl_export_batch_max_delay_seconds": 600, "ttl_export_batch_min_bytes": 1}, - ) - node.query(f"INSERT INTO {mt_table} VALUES (1, 2020, {DUE})") - wait_for_partitions_exported(node, mt_table, ["2020"], timeout=30) - assert_exactly_once(iceberg_ids(node, iceberg_table), [1]) - assert_one_snapshot_per_task(node, mt_table, iceberg_table) - - -def test_group_size_limits_over_successive_checks(cluster, source_engine): - """A group is limited to `ttl_export_max_parts_per_group` parts; the rest of the partition is - shipped by the next groups, one at a time, each its own snapshot.""" - node = cluster.instances["replica1"] - mt_table, iceberg_table = make_tables( - node, source_engine, - settings={"ttl_export_batch_window_seconds": 15, "ttl_export_batch_max_delay_seconds": 600, - "ttl_export_batch_min_bytes": 0, "ttl_export_max_parts_per_group": 2}, - ) + period = 5 + mt_table, iceberg_table = make_tables(node, source_engine, settings={"ttl_export_check_period_seconds": period}) + # Paused, so that no check runs between the inserts. + node.query(f"SYSTEM STOP MOVES {mt_table}") for i in range(3): node.query(f"INSERT INTO {mt_table} VALUES ({i}, 2020, {DUE})") + node.query(f"INSERT INTO {mt_table} VALUES (3, 2021, {DUE})") + node.query(f"INSERT INTO {mt_table} VALUES (4, 2020, {due_in(4 * period)})") + inserted = time.time() + node.query(f"SYSTEM START MOVES {mt_table}") - # Within the window nothing is exported. - time.sleep(4) - assert ttl_tasks(node, mt_table) == [] - assert ttl_rows(node, mt_table)["2020"]["eligible_parts"] == 3 - - in_flight = [] - def settled(): - in_flight.append(pending_ttl_tasks(node, mt_table)) - return len(completed_ttl_tasks(node, mt_table)) == 2 - wait_until(settled, 120, "The partition was not exported in two groups") - assert max(in_flight) <= 1, f"Two groups of one partition were in flight at once: {in_flight}" - - assert sorted(len(task["parts"]) for task in ttl_tasks(node, mt_table)) == [1, 2] - assert_exactly_once(iceberg_ids(node, iceberg_table), range(3)) - assert_one_snapshot_per_task(node, mt_table, iceberg_table) + def started(): + tasks = ttl_tasks(node, mt_table) + return len(tasks) == 2 and tasks + tasks = wait_until(started, 3 * period, "The due parts were not shipped by the next check") + assert sorted((task["partition_id"], len(task["parts"])) for task in tasks) == [("2020", 3), ("2021", 1)], tasks -def test_group_bytes_limit_ships_parts_one_by_one(cluster, source_engine): - node = cluster.instances["replica1"] - mt_table, iceberg_table = make_tables(node, source_engine, settings={"ttl_export_max_bytes_per_group": 1}) - node.query(f"SYSTEM STOP MOVES {mt_table}") - for i in range(3): - node.query(f"INSERT INTO {mt_table} VALUES ({i}, 2020, {DUE})") - node.query(f"SYSTEM START MOVES {mt_table}") + while time.time() - inserted < 3 * period: + assert len(ttl_tasks(node, mt_table)) == 2, "The part was exported before it was due" + time.sleep(1) - wait_until(lambda: len(completed_ttl_tasks(node, mt_table)) == 3, 120, "The parts were not exported one by one") - assert [len(task["parts"]) for task in ttl_tasks(node, mt_table)] == [1, 1, 1] - assert_exactly_once(iceberg_ids(node, iceberg_table), range(3)) + wait_until(lambda: len(completed_ttl_tasks(node, mt_table)) == 3, 60, "The part was not exported once it was due") + wait_for_partitions_exported(node, mt_table, ["2020", "2021"]) + assert len(ttl_tasks(node, mt_table)[-1]["parts"]) == 1 + assert_exactly_once(iceberg_ids(node, iceberg_table), range(5)) assert_one_snapshot_per_task(node, mt_table, iceberg_table) -def test_concurrent_groups_are_capped(cluster, source_engine): - """At most `ttl_export_max_concurrent_groups` groups of the table are in flight, whatever the - number of partitions, and their commits to the one Iceberg table do not get lost.""" +def test_one_group_per_partition_at_a_time(cluster, source_engine): + """A partition has at most one group being exported: the parts that become due meanwhile wait + for it, and the next check after it committed ships them together.""" node = cluster.instances["replica1"] - mt_table, iceberg_table = make_tables( - node, source_engine, - settings={"ttl_export_max_concurrent_groups": 2, "ttl_export_settings_profile": "ttl_export_quick_retry"}, - ) + mt_table, iceberg_table = make_tables(node, source_engine) - with failpoint([node], "export_part_retryable_throw"): - node.query(f"INSERT INTO {mt_table} VALUES (1, 2020, {DUE}), (2, 2021, {DUE}), (3, 2022, {DUE}), (4, 2023, {DUE})") - wait_until(lambda: pending_ttl_tasks(node, mt_table) == 2, 60, "Two groups were not started") - samples = [] - for _ in range(8): - samples.append(pending_ttl_tasks(node, mt_table)) - time.sleep(0.5) - assert max(samples) == 2, f"More groups than the limit were in flight: {samples}" - assert len(ttl_tasks(node, mt_table)) == 2, "A group was started while the limit was reached" - assert iceberg_ids(node, iceberg_table) == [] - - wait_for_partitions_exported(node, mt_table, ["2020", "2021", "2022", "2023"], timeout=120) - assert len(completed_ttl_tasks(node, mt_table)) == 4 - assert_exactly_once(iceberg_ids(node, iceberg_table), [1, 2, 3, 4]) + with group_in_flight(node) as wait_paused: + node.query(f"INSERT INTO {mt_table} VALUES (1, 2020, {DUE})") + wait_paused() + node.query(f"INSERT INTO {mt_table} VALUES (2, 2020, {DUE})") + node.query(f"INSERT INTO {mt_table} VALUES (3, 2020, {DUE})") + # Several checks go by while the group is in flight. + time.sleep(4) + tasks = ttl_tasks(node, mt_table) + assert [(task["status"], len(task["parts"])) for task in tasks] == [("PENDING", 1)], tasks + assert ttl_rows(node, mt_table)["2020"]["eligible_parts"] == 2 + + wait_until(lambda: len(completed_ttl_tasks(node, mt_table)) == 2, 60, "The parts that became due meanwhile were not exported") + wait_for_partitions_exported(node, mt_table, ["2020"]) + assert [len(task["parts"]) for task in ttl_tasks(node, mt_table)] == [1, 2] + assert_exactly_once(iceberg_ids(node, iceberg_table), [1, 2, 3]) assert_one_snapshot_per_task(node, mt_table, iceberg_table) - assert assert_iceberg_files_partitioned(node, iceberg_table, "year", "year") == {2020, 2021, 2022, 2023} def test_idle_checks_change_nothing(cluster, source_engine): @@ -274,21 +170,6 @@ def test_stop_moves_pauses_and_start_moves_resumes(cluster, source_engine): assert_one_snapshot_per_task(node, mt_table, iceberg_table) -def test_modified_settings_apply_on_the_next_check(cluster, source_engine): - node = cluster.instances["replica1"] - mt_table, iceberg_table = make_tables( - node, source_engine, settings={"ttl_export_batch_window_seconds": 600, "ttl_export_batch_max_delay_seconds": 600} - ) - node.query(f"INSERT INTO {mt_table} VALUES (1, 2020, {DUE})") - time.sleep(4) - assert ttl_tasks(node, mt_table) == [] - - node.query(f"ALTER TABLE {mt_table} MODIFY SETTING ttl_export_batch_window_seconds = 0, ttl_export_batch_max_delay_seconds = 0") - wait_for_partitions_exported(node, mt_table, ["2020"], timeout=30) - assert_exactly_once(iceberg_ids(node, iceberg_table), [1]) - assert_one_snapshot_per_task(node, mt_table, iceberg_table) - - def test_plain_commit_is_recorded_without_waiting_for_the_period(cluster): """On a plain `MergeTree` table the task commits first, and the index records the parts as exported afterwards. The scheduler is woken by the commit, so it does not wait for the next check.""" diff --git a/tests/queries/0_stateless/export_ttl.lib b/tests/queries/0_stateless/export_ttl.lib index 00ce4e6171d2..4df89f4cfcb7 100644 --- a/tests/queries/0_stateless/export_ttl.lib +++ b/tests/queries/0_stateless/export_ttl.lib @@ -51,7 +51,7 @@ function run_export_ttl_test() CREATE TABLE $mt_table (id UInt64, year UInt16, t DateTime) ENGINE = $source_engine PARTITION BY year ORDER BY id TTL t + INTERVAL 1 DAY EXPORT TO TABLE $s3_table - SETTINGS ttl_export_check_period_seconds = 1, ttl_export_batch_window_seconds = 0, ttl_export_batch_max_delay_seconds = 0" + SETTINGS ttl_export_check_period_seconds = 1" # Paused, so both parts of 2020 are exported in one group, and not merged before. ${CLICKHOUSE_CLIENT} --query "SYSTEM STOP MOVES $mt_table"