From 6c07482bb4791ae2610b06fc8ac153c2223451bb Mon Sep 17 00:00:00 2001 From: Andrey Zvonov <32552679+zvonand@users.noreply.github.com> Date: Wed, 6 May 2026 19:10:04 +0200 Subject: [PATCH 01/43] Cherry-pick of https://github.com/Altinity/ClickHouse/pull/1718 with unresolved conflict markers (resolution in next commit) --- Original cherry-pick message follows: Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3 Antalya 26.3: apassos-3: combined port of 12 PRs # Conflicts: # ci/jobs/scripts/integration_tests_configs.py # contrib/openssl # src/Common/ErrorCodes.cpp # src/Common/FailPoint.cpp # src/Common/ProfileEvents.cpp # src/Common/setThreadName.h # src/Core/Settings.cpp # src/Core/SettingsEnums.cpp # src/Core/SettingsEnums.h # src/Disks/DiskObjectStorage/ObjectStorages/IObjectStorage.h # src/Functions/generateSnowflakeID.cpp # src/Interpreters/DDLWorker.cpp # src/Parsers/ASTAlterQuery.cpp # src/Parsers/ASTSystemQuery.cpp # src/Parsers/ParserAlterQuery.cpp # src/Storages/IPartitionStrategy.cpp # src/Storages/IPartitionStrategy.h # src/Storages/MergeTree/IMergeTreeDataPart.cpp # src/Storages/MergeTree/MergeTreeData.cpp # src/Storages/MergeTree/MergeTreeData.h # src/Storages/ObjectStorage/DataLakes/IDataLakeMetadata.h # src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergMetadata.cpp # src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergMetadata.h # src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergWrites.cpp # src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergWrites.h # src/Storages/ObjectStorage/DataLakes/Iceberg/MultipleFileWriter.cpp # src/Storages/ObjectStorage/DataLakes/Iceberg/MultipleFileWriter.h # src/Storages/ObjectStorage/DataLakes/Iceberg/Utils.cpp # src/Storages/ObjectStorage/StorageObjectStorage.cpp # src/Storages/ObjectStorage/StorageObjectStorage.h # src/Storages/ObjectStorage/StorageObjectStorageCluster.cpp # src/Storages/ObjectStorage/StorageObjectStorageConfiguration.h # src/Storages/ObjectStorage/StorageObjectStorageSink.h # src/Storages/ObjectStorage/StorageObjectStorageSource.cpp # src/Storages/ObjectStorage/StorageObjectStorageSource.h # src/Storages/StorageMergeTree.cpp # src/Storages/StorageReplicatedMergeTree.cpp # src/Storages/StorageReplicatedMergeTree.h # src/Storages/System/StorageSystemMerges.cpp # src/Storages/System/attachSystemTables.cpp # tests/queries/0_stateless/03413_experimental_settings_cannot_be_enabled_by_default.sql # tests/queries/0_stateless/03745_system_background_schedule_pool.reference --- ci/jobs/scripts/integration_tests_configs.py | 7 + docs/en/antalya/part_export.md | 331 ++++ docs/en/antalya/partition_export.md | 211 +++ docs/en/operations/system-tables/exports.md | 56 + programs/server/Server.cpp | 11 + src/Access/Common/AccessType.h | 3 + src/CMakeLists.txt | 1 + src/Common/CurrentMetrics.cpp | 1 + src/Common/ErrorCodes.cpp | 8 + src/Common/FailPoint.cpp | 8 + src/Common/ProfileEvents.cpp | 34 + src/Common/TTLCachePolicy.h | 4 +- src/Common/setThreadName.h | 4 + src/Core/ServerSettings.cpp | 7 +- src/Core/Settings.cpp | 61 + src/Core/Settings.h | 1 + src/Core/SettingsChangesHistory.cpp | 22 +- src/Core/SettingsEnums.cpp | 5 + src/Core/SettingsEnums.h | 12 + src/Databases/DatabaseReplicated.cpp | 2 +- .../AzureBlobStorage/AzureObjectStorage.h | 2 + .../ObjectStorages/IObjectStorage.h | 4 + .../ObjectStorages/S3/S3ObjectStorage.h | 2 + src/Functions/generateSnowflakeID.cpp | 9 + src/Functions/generateSnowflakeID.h | 2 + src/Interpreters/Context.cpp | 24 + src/Interpreters/Context.h | 6 + src/Interpreters/DDLWorker.cpp | 4 + src/Interpreters/InterpreterAlterQuery.cpp | 14 + .../InterpreterKillQueryQuery.cpp | 86 + src/Interpreters/InterpreterSystemQuery.cpp | 9 +- src/Interpreters/PartLog.cpp | 11 +- src/Interpreters/PartLog.h | 2 + src/Interpreters/executeDDLQueryOnCluster.cpp | 2 + src/Parsers/ASTAlterQuery.cpp | 60 + src/Parsers/ASTAlterQuery.h | 7 + src/Parsers/ASTKillQueryQuery.cpp | 3 + src/Parsers/ASTKillQueryQuery.h | 1 + src/Parsers/ASTSystemQuery.cpp | 4 + src/Parsers/ASTSystemQuery.h | 1 + src/Parsers/CommonParsers.h | 2 + src/Parsers/ParserAlterQuery.cpp | 67 + src/Parsers/ParserKillQueryQuery.cpp | 3 + .../Cache/ObjectStorageListObjectsCache.cpp | 226 +++ .../Cache/ObjectStorageListObjectsCache.h | 80 + ...test_object_storage_list_objects_cache.cpp | 244 +++ src/Storages/ColumnsDescription.cpp | 10 +- src/Storages/ColumnsDescription.h | 1 + ...portReplicatedMergeTreePartitionManifest.h | 219 +++ ...ortReplicatedMergeTreePartitionTaskEntry.h | 78 + src/Storages/IPartitionStrategy.cpp | 35 +- src/Storages/IPartitionStrategy.h | 13 +- src/Storages/IStorage.cpp | 5 + src/Storages/IStorage.h | 50 + .../MergeTree/BackgroundJobsAssignee.cpp | 4 + .../MergeTree/BackgroundJobsAssignee.h | 2 + src/Storages/MergeTree/ExportList.cpp | 74 + src/Storages/MergeTree/ExportList.h | 96 ++ .../ExportPartFromPartitionExportTask.cpp | 75 + .../ExportPartFromPartitionExportTask.h | 36 + src/Storages/MergeTree/ExportPartTask.cpp | 414 +++++ src/Storages/MergeTree/ExportPartTask.h | 34 + .../ExportPartitionManifestUpdatingTask.cpp | 883 ++++++++++ .../ExportPartitionManifestUpdatingTask.h | 49 + .../ExportPartitionTaskScheduler.cpp | 549 ++++++ .../MergeTree/ExportPartitionTaskScheduler.h | 66 + .../MergeTree/ExportPartitionUtils.cpp | 498 ++++++ src/Storages/MergeTree/ExportPartitionUtils.h | 98 ++ src/Storages/MergeTree/IMergeTreeDataPart.cpp | 37 + src/Storages/MergeTree/IMergeTreeDataPart.h | 2 + .../MergeTree/MergeTreeBackgroundExecutor.cpp | 6 + .../MergeTree/MergeTreeBackgroundExecutor.h | 2 + src/Storages/MergeTree/MergeTreeData.cpp | 382 ++++- src/Storages/MergeTree/MergeTreeData.h | 51 +- .../MergeTree/MergeTreeExportManifest.h | 50 + .../MergeTree/MergeTreePartExportManifest.h | 98 ++ .../MergeTree/MergeTreePartExportStatus.h | 20 + src/Storages/MergeTree/MergeTreePartition.cpp | 16 + src/Storages/MergeTree/MergeTreePartition.h | 2 + .../MergeTree/MergeTreeSequentialSource.cpp | 4 + .../MergeTree/MergeTreeSequentialSource.h | 1 + .../ReplicatedMergeTreeRestartingThread.cpp | 15 + .../tests/gtest_export_partition_ordering.cpp | 75 + .../DataLakes/IDataLakeMetadata.h | 37 + .../DataLakes/Iceberg/AvroSchema.h | 38 + .../DataLakes/Iceberg/Constant.h | 1 + .../DataLakes/Iceberg/IcebergDataFileEntry.h | 50 + .../DataLakes/Iceberg/IcebergMetadata.cpp | 557 ++++++ .../DataLakes/Iceberg/IcebergMetadata.h | 65 + .../DataLakes/Iceberg/IcebergWrites.cpp | 438 ++++- .../DataLakes/Iceberg/IcebergWrites.h | 86 + .../DataLakes/Iceberg/MetadataGenerator.cpp | 2 +- .../DataLakes/Iceberg/MultipleFileWriter.cpp | 51 +- .../DataLakes/Iceberg/MultipleFileWriter.h | 26 +- .../ObjectStorage/DataLakes/Iceberg/Utils.cpp | 10 + .../ObjectStorage/DataLakes/Iceberg/Utils.h | 11 + .../MultiFileStorageObjectStorageSink.cpp | 153 ++ .../MultiFileStorageObjectStorageSink.h | 57 + .../ObjectStorageFilePathGenerator.h | 83 + .../ObjectStorage/StorageObjectStorage.cpp | 134 +- .../ObjectStorage/StorageObjectStorage.h | 25 + .../StorageObjectStorageCluster.cpp | 11 + .../StorageObjectStorageConfiguration.cpp | 41 +- .../StorageObjectStorageConfiguration.h | 14 + .../StorageObjectStorageSink.cpp | 15 +- .../ObjectStorage/StorageObjectStorageSink.h | 8 + .../StorageObjectStorageSource.cpp | 71 +- .../StorageObjectStorageSource.h | 36 +- src/Storages/PartitionCommands.cpp | 35 + src/Storages/PartitionCommands.h | 8 +- src/Storages/PartitionedSink.cpp | 4 +- src/Storages/PartitionedSink.h | 12 +- src/Storages/StorageFile.cpp | 20 +- src/Storages/StorageMergeTree.cpp | 19 +- src/Storages/StorageMergeTree.h | 2 - src/Storages/StorageReplicatedMergeTree.cpp | 575 ++++++- src/Storages/StorageReplicatedMergeTree.h | 49 +- src/Storages/StorageURL.cpp | 12 +- src/Storages/System/StorageSystemExports.cpp | 71 + src/Storages/System/StorageSystemExports.h | 25 + src/Storages/System/StorageSystemMerges.cpp | 4 + ...torageSystemReplicatedPartitionExports.cpp | 151 ++ .../StorageSystemReplicatedPartitionExports.h | 43 + src/Storages/System/attachSystemTables.cpp | 19 +- ...perimental_export_merge_tree_partition.xml | 3 + tests/config/install.sh | 1 + .../helpers/export_partition_helpers.py | 201 +++ .../helpers/iceberg_export_stats.py | 179 ++ .../configs/config.d/metadata_log.xml | 7 + .../test.py | 481 ++++++ .../__init__.py | 0 .../configs/named_collections.xml | 9 + .../test.py | 314 ++++ .../__init__.py | 0 .../allow_experimental_export_partition.xml | 3 + .../configs/config.d/metadata_log.xml | 7 + .../configs/users.d/profile.xml | 9 + .../test.py | 934 ++++++++++ .../__init__.py | 0 .../allow_experimental_export_partition.xml | 3 + .../disable_experimental_export_partition.xml | 3 + .../configs/macros_shard1_replica1.xml | 6 + .../configs/macros_shard2_replica1.xml | 6 + .../configs/named_collections.xml | 9 + .../configs/users.d/profile.xml | 8 + .../test.py | 1508 +++++++++++++++++ .../config.d/allow_export_partition.xml | 3 + .../users.d/allow_export_partition.xml | 7 + .../test_export_partition_iceberg.py | 912 ++++++++++ .../test_export_partition_iceberg_catalog.py | 532 ++++++ .../01271_show_privileges.reference | 3 + ...21_system_zookeeper_unrestricted.reference | 2 + ...stem_zookeeper_unrestricted_like.reference | 2 + ...bject_storage_list_objects_cache.reference | 103 ++ ...3377_object_storage_list_objects_cache.sql | 115 ++ ..._settings_cannot_be_enabled_by_default.sql | 9 + ...572_export_merge_tree_part_basic.reference | 35 + .../03572_export_merge_tree_part_basic.sh | 85 + ..._part_limits_and_table_functions.reference | 20 + ...ge_tree_part_limits_and_table_functions.sh | 132 ++ ..._merge_tree_part_special_columns.reference | 39 + ..._export_merge_tree_part_special_columns.sh | 154 ++ ...ee_part_to_object_storage_simple.reference | 0 ...rge_tree_part_to_object_storage_simple.sql | 39 + ...erge_tree_part_to_object_storage.reference | 16 + ...cated_merge_tree_part_to_object_storage.sh | 44 + ...ee_part_to_object_storage_simple.reference | 0 ...rge_tree_part_to_object_storage_simple.sql | 22 + ...3604_export_merge_tree_partition.reference | 31 + .../03604_export_merge_tree_partition.sh | 57 + ...merge_tree_part_filename_pattern.reference | 16 + ...export_merge_tree_part_filename_pattern.sh | 49 + ..._system_background_schedule_pool.reference | 4 + 173 files changed, 14789 insertions(+), 116 deletions(-) create mode 100644 docs/en/antalya/part_export.md create mode 100644 docs/en/antalya/partition_export.md create mode 100644 docs/en/operations/system-tables/exports.md create mode 100644 src/Storages/Cache/ObjectStorageListObjectsCache.cpp create mode 100644 src/Storages/Cache/ObjectStorageListObjectsCache.h create mode 100644 src/Storages/Cache/tests/gtest_object_storage_list_objects_cache.cpp create mode 100644 src/Storages/ExportReplicatedMergeTreePartitionManifest.h create mode 100644 src/Storages/ExportReplicatedMergeTreePartitionTaskEntry.h create mode 100644 src/Storages/MergeTree/ExportList.cpp create mode 100644 src/Storages/MergeTree/ExportList.h create mode 100644 src/Storages/MergeTree/ExportPartFromPartitionExportTask.cpp create mode 100644 src/Storages/MergeTree/ExportPartFromPartitionExportTask.h create mode 100644 src/Storages/MergeTree/ExportPartTask.cpp create mode 100644 src/Storages/MergeTree/ExportPartTask.h create mode 100644 src/Storages/MergeTree/ExportPartitionManifestUpdatingTask.cpp create mode 100644 src/Storages/MergeTree/ExportPartitionManifestUpdatingTask.h create mode 100644 src/Storages/MergeTree/ExportPartitionTaskScheduler.cpp create mode 100644 src/Storages/MergeTree/ExportPartitionTaskScheduler.h create mode 100644 src/Storages/MergeTree/ExportPartitionUtils.cpp create mode 100644 src/Storages/MergeTree/ExportPartitionUtils.h create mode 100644 src/Storages/MergeTree/MergeTreeExportManifest.h create mode 100644 src/Storages/MergeTree/MergeTreePartExportManifest.h create mode 100644 src/Storages/MergeTree/MergeTreePartExportStatus.h create mode 100644 src/Storages/MergeTree/tests/gtest_export_partition_ordering.cpp create mode 100644 src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergDataFileEntry.h create mode 100644 src/Storages/ObjectStorage/MultiFileStorageObjectStorageSink.cpp create mode 100644 src/Storages/ObjectStorage/MultiFileStorageObjectStorageSink.h create mode 100644 src/Storages/ObjectStorage/ObjectStorageFilePathGenerator.h create mode 100644 src/Storages/System/StorageSystemExports.cpp create mode 100644 src/Storages/System/StorageSystemExports.h create mode 100644 src/Storages/System/StorageSystemReplicatedPartitionExports.cpp create mode 100644 src/Storages/System/StorageSystemReplicatedPartitionExports.h create mode 100644 tests/config/config.d/allow_experimental_export_merge_tree_partition.xml create mode 100644 tests/integration/helpers/export_partition_helpers.py create mode 100644 tests/integration/helpers/iceberg_export_stats.py create mode 100644 tests/integration/test_export_merge_tree_part_to_iceberg/configs/config.d/metadata_log.xml create mode 100644 tests/integration/test_export_merge_tree_part_to_iceberg/test.py create mode 100644 tests/integration/test_export_merge_tree_part_to_object_storage/__init__.py create mode 100644 tests/integration/test_export_merge_tree_part_to_object_storage/configs/named_collections.xml create mode 100644 tests/integration/test_export_merge_tree_part_to_object_storage/test.py create mode 100644 tests/integration/test_export_replicated_mt_partition_to_iceberg/__init__.py create mode 100644 tests/integration/test_export_replicated_mt_partition_to_iceberg/configs/allow_experimental_export_partition.xml create mode 100644 tests/integration/test_export_replicated_mt_partition_to_iceberg/configs/config.d/metadata_log.xml create mode 100644 tests/integration/test_export_replicated_mt_partition_to_iceberg/configs/users.d/profile.xml create mode 100644 tests/integration/test_export_replicated_mt_partition_to_iceberg/test.py create mode 100644 tests/integration/test_export_replicated_mt_partition_to_object_storage/__init__.py create mode 100644 tests/integration/test_export_replicated_mt_partition_to_object_storage/configs/allow_experimental_export_partition.xml create mode 100644 tests/integration/test_export_replicated_mt_partition_to_object_storage/configs/disable_experimental_export_partition.xml create mode 100644 tests/integration/test_export_replicated_mt_partition_to_object_storage/configs/macros_shard1_replica1.xml create mode 100644 tests/integration/test_export_replicated_mt_partition_to_object_storage/configs/macros_shard2_replica1.xml create mode 100644 tests/integration/test_export_replicated_mt_partition_to_object_storage/configs/named_collections.xml create mode 100644 tests/integration/test_export_replicated_mt_partition_to_object_storage/configs/users.d/profile.xml create mode 100644 tests/integration/test_export_replicated_mt_partition_to_object_storage/test.py create mode 100644 tests/integration/test_storage_iceberg_with_spark/configs/config.d/allow_export_partition.xml create mode 100644 tests/integration/test_storage_iceberg_with_spark/configs/users.d/allow_export_partition.xml create mode 100644 tests/integration/test_storage_iceberg_with_spark/test_export_partition_iceberg.py create mode 100644 tests/integration/test_storage_iceberg_with_spark/test_export_partition_iceberg_catalog.py create mode 100644 tests/queries/0_stateless/03377_object_storage_list_objects_cache.reference create mode 100644 tests/queries/0_stateless/03377_object_storage_list_objects_cache.sql create mode 100644 tests/queries/0_stateless/03572_export_merge_tree_part_basic.reference create mode 100755 tests/queries/0_stateless/03572_export_merge_tree_part_basic.sh create mode 100644 tests/queries/0_stateless/03572_export_merge_tree_part_limits_and_table_functions.reference create mode 100755 tests/queries/0_stateless/03572_export_merge_tree_part_limits_and_table_functions.sh create mode 100644 tests/queries/0_stateless/03572_export_merge_tree_part_special_columns.reference create mode 100755 tests/queries/0_stateless/03572_export_merge_tree_part_special_columns.sh create mode 100644 tests/queries/0_stateless/03572_export_merge_tree_part_to_object_storage_simple.reference create mode 100644 tests/queries/0_stateless/03572_export_merge_tree_part_to_object_storage_simple.sql create mode 100644 tests/queries/0_stateless/03572_export_replicated_merge_tree_part_to_object_storage.reference create mode 100755 tests/queries/0_stateless/03572_export_replicated_merge_tree_part_to_object_storage.sh create mode 100644 tests/queries/0_stateless/03572_export_replicated_merge_tree_part_to_object_storage_simple.reference create mode 100644 tests/queries/0_stateless/03572_export_replicated_merge_tree_part_to_object_storage_simple.sql create mode 100644 tests/queries/0_stateless/03604_export_merge_tree_partition.reference create mode 100755 tests/queries/0_stateless/03604_export_merge_tree_partition.sh create mode 100644 tests/queries/0_stateless/03608_export_merge_tree_part_filename_pattern.reference create mode 100755 tests/queries/0_stateless/03608_export_merge_tree_part_filename_pattern.sh diff --git a/ci/jobs/scripts/integration_tests_configs.py b/ci/jobs/scripts/integration_tests_configs.py index b3076f467d6c..105b989fd06f 100644 --- a/ci/jobs/scripts/integration_tests_configs.py +++ b/ci/jobs/scripts/integration_tests_configs.py @@ -58,6 +58,13 @@ class TC: True, "pins azurite to fixed host port 10000 (Spark emulator mode); concurrent --dist=each workers collide on bind", ), +<<<<<<< HEAD +======= + TC("test_storage_iceberg_no_spark/", True, "no idea why i'm sequential"), + TC("test_storage_iceberg_with_spark_cache/", True, "no idea why i'm sequential"), + TC("test_storage_iceberg_concurrent/", True, "no idea why i'm sequential"), + TC("test_export_replicated_mt_partition_to_object_storage/", True, "ZooKeeper can't handle too many parallel requests"), +>>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) ] IMAGES_ENV = { diff --git a/docs/en/antalya/part_export.md b/docs/en/antalya/part_export.md new file mode 100644 index 000000000000..03f9f479991b --- /dev/null +++ b/docs/en/antalya/part_export.md @@ -0,0 +1,331 @@ +# ALTER TABLE EXPORT PART + +## Overview + +The `ALTER TABLE EXPORT PART` command exports individual MergeTree data parts to object storage (S3, Azure Blob Storage, etc.) or data lakes like Apache Iceberg tables (with and without catalogs), typically in Parquet format. + +**Key Characteristics:** +- **Experimental feature** - must be enabled via `allow_experimental_export_merge_tree_part` setting +- **Asynchronous** - executes in the background, returns immediately +- **Ephemeral** - no automatic retry mechanism; manual retry required on failure +- **Idempotent** - safe to re-export the same part (skips by default if file exists) +- **Preserves sort order** from the source table + +### On Apache Iceberg storage exports: + +Each MergeTree part will become a separate file (or more depending on `max_bytes` and `max_rows` settings) following the engine naming convention. Once the part has been exported, new snapshots / manifest files are generated and the data is committed using the Apache Iceberg commit mechanism. + +### On plain object storage exports: + +A commit file is shipped to the same destination directory containing all data files exported within that transaction. + +## Syntax + +```sql +ALTER TABLE [database.]table_name +EXPORT PART 'part_name' +TO TABLE [destination_database.]destination_table +SETTINGS allow_experimental_export_merge_tree_part = 1 + [, setting_name = value, ...] +``` + +## Syntax with table function + +```sql +ALTER TABLE [database.]table_name +EXPORT PART 'part_name' +TO TABLE FUNCTION s3(s3_conn, filename='table_function', partition_strategy...) +SETTINGS allow_experimental_export_merge_tree_part = 1 + [, setting_name = value, ...] +``` + +### Parameters + +- **`table_name`**: The source MergeTree table containing the part to export +- **`part_name`**: The exact name of the data part to export (e.g., `'2020_1_1_0'`, `'all_1_1_0'`) +- **`destination_table`**: The target table for the export (typically an S3, Azure, or other object storage table) + +## Requirements + +Source and destination tables must be 100% compatible: + +1. **Identical schemas** - same columns, types, and order +2. **Matching partition keys** - partition expressions must be identical + +In case a table function is used as the destination, the schema can be omitted and it will be inferred from the source table. + +## Settings + +### `allow_experimental_export_merge_tree_part` (Required) + +- **Type**: `Bool` +- **Default**: `false` +- **Description**: Must be set to `true` to enable the experimental feature. + +### `export_merge_tree_part_overwrite_file_if_exists` (Optional) + +- **Type**: `Bool` +- **Default**: `false` +- **Description**: If set to `true`, it will overwrite the file. Otherwise, fails with exception. + +### `export_merge_tree_part_max_bytes_per_file` (Optional) + +- **Type**: `UInt64` +- **Default**: `0` +- **Description**: Maximum number of bytes to write to a single file when exporting a merge tree part. 0 means no limit. This is not a hard limit, and it highly depends on the output format granularity and input source chunk size. Using this might break idempotency, use it with care. + +### `export_merge_tree_part_max_rows_per_file` (Optional) + +- **Type**: `UInt64` +- **Default**: `0` +- **Description**: Maximum number of rows to write to a single file when exporting a merge tree part. 0 means no limit. This is not a hard limit, and it highly depends on the output format granularity and input source chunk size. Using this might break idempotency, use it with care. + +#### `export_merge_tree_part_file_already_exists_policy` (Optional) + +- **Type**: `MergeTreePartExportFileAlreadyExistsPolicy` +- **Default**: `skip` +- **Description**: Policy for handling files that already exist during export. Possible values: + - `skip` - Skip the file if it already exists + - `error` - Throw an error if the file already exists + - `overwrite` - Overwrite the file + +### `export_merge_tree_part_throw_on_pending_mutations` (Optional) + +- **Type**: `bool` +- **Default**: `true` +- **Description**: If set to true, throws if pending mutations exists for a given part. Note that by default mutations are applied to all parts, which means that if a mutation in practice would only affetct part/partition x, all the other parts/partition will throw upon export. The exception is when the `IN PARTITION` clause was used in the mutation command. Note the `IN PARTITION` clause is not properly implemented for plain MergeTree tables. + +### `export_merge_tree_part_throw_on_pending_patch_parts` (Optional) + +- **Type**: `bool` +- **Default**: `true` +- **Description**: If set to true, throws if pending patch parts exists for a given part. Note that by default mutations are applied to all parts, which means that if a mutation in practice would only affetct part/partition x, all the other parts/partition will throw upon export. The exception is when the `IN PARTITION` clause was used in the mutation command. Note the `IN PARTITION` clause is not properly implemented for plain MergeTree tables. + +### `export_merge_tree_part_filename_pattern` (Optional) + +- **Type**: `String` +- **Default**: `{part_name}_{checksum}` +- **Description**: Pattern for the filename of the exported merge tree part. The `part_name` and `checksum` are calculated and replaced on the fly. Additional macros are supported. + + +## Examples + +### Basic Export to S3 + +```sql +-- Create source and destination tables +CREATE TABLE mt_table (id UInt64, year UInt16) +ENGINE = MergeTree() PARTITION BY year ORDER BY tuple(); + +CREATE TABLE s3_table (id UInt64, year UInt16) +ENGINE = S3(s3_conn, filename='data', format=Parquet, partition_strategy='hive') +PARTITION BY year; + +-- Insert and export +INSERT INTO mt_table VALUES (1, 2020), (2, 2020), (3, 2021); + +ALTER TABLE mt_table EXPORT PART '2020_1_1_0' TO TABLE s3_table +SETTINGS allow_experimental_export_merge_tree_part = 1; + +ALTER TABLE mt_table EXPORT PART '2021_2_2_0' TO TABLE s3_table +SETTINGS allow_experimental_export_merge_tree_part = 1; +``` + +### Table function export + +```sql +-- Create source and destination tables +CREATE TABLE mt_table (id UInt64, year UInt16) +ENGINE = MergeTree() PARTITION BY year ORDER BY tuple(); + +-- Insert and export +INSERT INTO mt_table VALUES (1, 2020), (2, 2020), (3, 2021); + +ALTER TABLE mt_table EXPORT PART '2020_1_1_0' TO TABLE FUNCTION s3(s3_conn, filename='table_function', format=Parquet, partition_strategy='hive') PARTITION BY year +SETTINGS allow_experimental_export_merge_tree_part = 1; +``` + +## Monitoring + +### Active Exports + +Active exports can be found in the `system.exports` table. As of now, it only shows currently executing exports. It will not show pending or finished exports. + +```sql +arthur :) select * from system.exports; + +SELECT * +FROM system.exports + +Query id: 2026718c-d249-4208-891b-a271f1f93407 + +Row 1: +────── +source_database: default +source_table: source_mt_table +destination_database: default +destination_table: destination_table +create_time: 2025-11-19 09:09:11 +part_name: 20251016-365_1_1_0 +destination_file_paths: ['table_root/eventDate=2025-10-16/retention=365/20251016-365_1_1_0_17B2F6CD5D3C18E787C07AE3DAF16EB1.1.parquet'] +elapsed: 2.04845441 +rows_read: 1138688 -- 1.14 million +total_rows_to_read: 550961374 -- 550.96 million +total_size_bytes_compressed: 37619147120 -- 37.62 billion +total_size_bytes_uncompressed: 138166213721 -- 138.17 billion +bytes_read_uncompressed: 316892925 -- 316.89 million +memory_usage: 596006095 -- 596.01 million +peak_memory_usage: 601239033 -- 601.24 million +``` + +### Export History + +You can query succeeded or failed exports in `system.part_log`. For now, it only keeps track of completion events (either success or fails). + +```sql +arthur :) select * from system.part_log where event_type='ExportPart' and table = 'replicated_source' order by event_time desc limit 1; + +SELECT * +FROM system.part_log +WHERE (event_type = 'ExportPart') AND (`table` = 'replicated_source') +ORDER BY event_time DESC +LIMIT 1 + +Query id: ae1c1cd3-c20e-4f20-8b82-ed1f6af0237f + +Row 1: +────── +hostname: arthur +query_id: +event_type: ExportPart +merge_reason: NotAMerge +merge_algorithm: Undecided +event_date: 2025-11-19 +event_time: 2025-11-19 09:08:31 +event_time_microseconds: 2025-11-19 09:08:31.974701 +duration_ms: 4 +database: default +table: replicated_source +table_uuid: 78471c67-24f4-4398-9df5-ad0a6c3daf41 +part_name: 2021_0_0_0 +partition_id: 2021 +partition: 2021 +part_type: Compact +disk_name: default +path_on_disk: +remote_file_paths ['year=2021/2021_0_0_0_78C704B133D41CB0EF64DD2A9ED3B6BA.1.parquet'] +rows: 1 +size_in_bytes: 272 +merged_from: ['2021_0_0_0'] +bytes_uncompressed: 86 +read_rows: 1 +read_bytes: 6 +peak_memory_usage: 22 +error: 0 +exception: +ProfileEvents: {} +``` + +### Profile Events + +- `PartsExports` - Successful exports +- `PartsExportFailures` - Failed exports +- `PartsExportDuplicated` - Number of part exports that failed because target already exists. +- `PartsExportTotalMilliseconds` - Total time + +### Split large files + +```sql +alter table big_table export part '2025_0_32_3' to table replicated_big_destination SETTINGS export_merge_tree_part_max_bytes_per_file=10000000, output_format_parquet_row_group_size_bytes=5000000; + +arthur :) select * from system.exports; + +SELECT * +FROM system.exports + +Query id: d78d9ce5-cfbc-4957-b7dd-bc8129811634 + +Row 1: +────── +source_database: default +source_table: big_table +destination_database: default +destination_table: replicated_big_destination +create_time: 2025-12-15 13:12:48 +part_name: 2025_0_32_3 +destination_file_paths: ['replicated_big/year=2025/2025_0_32_3_E439C23833C39C6E5104F6F4D1048BE7.1.parquet','replicated_big/year=2025/2025_0_32_3_E439C23833C39C6E5104F6F4D1048BE7.2.parquet','replicated_big/year=2025/2025_0_32_3_E439C23833C39C6E5104F6F4D1048BE7.3.parquet','replicated_big/year=2025/2025_0_32_3_E439C23833C39C6E5104F6F4D1048BE7.4.parquet'] +elapsed: 14.360427274 +rows_read: 10256384 -- 10.26 million +total_rows_to_read: 10485760 -- 10.49 million +total_size_bytes_compressed: 83779395 -- 83.78 million +total_size_bytes_uncompressed: 10611691600 -- 10.61 billion +bytes_read_uncompressed: 10440998912 -- 10.44 billion +memory_usage: 89795477 -- 89.80 million +peak_memory_usage: 107362133 -- 107.36 million + +1 row in set. Elapsed: 0.014 sec. + +arthur :) select * from system.part_log where event_type = 'ExportPart' order by event_time desc limit 1 format Vertical; + +SELECT * +FROM system.part_log +WHERE event_type = 'ExportPart' +ORDER BY event_time DESC +LIMIT 1 +FORMAT Vertical + +Query id: 95128b01-b751-4726-8e3e-320728ac6af7 + +Row 1: +────── +hostname: arthur +query_id: +event_type: ExportPart +merge_reason: NotAMerge +merge_algorithm: Undecided +event_date: 2025-12-15 +event_time: 2025-12-15 13:13:03 +event_time_microseconds: 2025-12-15 13:13:03.197492 +duration_ms: 14673 +database: default +table: big_table +table_uuid: a3eeeea0-295c-41a3-84ef-6b5463dbbe8c +part_name: 2025_0_32_3 +partition_id: 2025 +partition: 2025 +part_type: Wide +disk_name: default +path_on_disk: ./store/a3e/a3eeeea0-295c-41a3-84ef-6b5463dbbe8c/2025_0_32_3/ +remote_file_paths: ['replicated_big/year=2025/2025_0_32_3_E439C23833C39C6E5104F6F4D1048BE7.1.parquet','replicated_big/year=2025/2025_0_32_3_E439C23833C39C6E5104F6F4D1048BE7.2.parquet','replicated_big/year=2025/2025_0_32_3_E439C23833C39C6E5104F6F4D1048BE7.3.parquet','replicated_big/year=2025/2025_0_32_3_E439C23833C39C6E5104F6F4D1048BE7.4.parquet'] +rows: 10485760 -- 10.49 million +size_in_bytes: 83779395 -- 83.78 million +merged_from: ['2025_0_32_3'] +bytes_uncompressed: 10611691600 -- 10.61 billion +read_rows: 10485760 -- 10.49 million +read_bytes: 10674503680 -- 10.67 billion +peak_memory_usage: 107362133 -- 107.36 million +error: 0 +exception: +ProfileEvents: {} + +1 row in set. Elapsed: 0.044 sec. + +arthur :) select _path, formatReadableSize(_size) as _size from s3(s3_conn, filename='**', format=One); + +SELECT + _path, + formatReadableSize(_size) AS _size +FROM s3(s3_conn, filename = '**', format = One) + +Query id: c48ae709-f590-4d1b-8158-191f8d628966 + + ┌─_path────────────────────────────────────────────────────────────────────────────────┬─_size─────┐ +1. │ test/replicated_big/year=2025/2025_0_32_3_E439C23833C39C6E5104F6F4D1048BE7.1.parquet │ 17.36 MiB │ +2. │ test/replicated_big/year=2025/2025_0_32_3_E439C23833C39C6E5104F6F4D1048BE7.2.parquet │ 17.32 MiB │ +3. │ test/replicated_big/year=2025/2025_0_32_3_E439C23833C39C6E5104F6F4D1048BE7.4.parquet │ 5.04 MiB │ +4. │ test/replicated_big/year=2025/2025_0_32_3_E439C23833C39C6E5104F6F4D1048BE7.3.parquet │ 17.40 MiB │ +5. │ test/replicated_big/year=2025/commit_2025_0_32_3_E439C23833C39C6E5104F6F4D1048BE7 │ 320.00 B │ + └──────────────────────────────────────────────────────────────────────────────────────┴───────────┘ + +5 rows in set. Elapsed: 0.072 sec. +``` diff --git a/docs/en/antalya/partition_export.md b/docs/en/antalya/partition_export.md new file mode 100644 index 000000000000..975915859482 --- /dev/null +++ b/docs/en/antalya/partition_export.md @@ -0,0 +1,211 @@ +# ALTER TABLE EXPORT PARTITION + +## Overview + +The `ALTER TABLE EXPORT PARTITION` command exports entire partitions from Replicated*MergeTree tables to object storage (S3, Azure Blob Storage, etc.) or data lakes like Apache Iceberg tables (with and without catalogs), typically in Parquet format. This feature coordinates export part operations across all replicas using ZooKeeper. + +The set of parts that are exported is based on the list of parts the replica that received the export command sees. The other replicas will assist in the export process if they have those parts locally. Otherwise they will ignore it. + +The partition export tasks can be observed through `system.replicated_partition_exports`. Querying this table results in a query to ZooKeeper, so it must be used with care. Individual part export progress can be observed as usual through `system.exports`. + +The same partition can not be exported to the same destination more than once. There are two ways to override this behavior: either by setting the `export_merge_tree_partition_force_export` setting or waiting for the task to expire. + +The export task can be killed by issuing the kill command: `KILL EXPORT PARTITION `. + +The task is persistent - it should be resumed after crashes, failures and etc. + +### On Apache Iceberg storage exports: + +Each MergeTree part will become a separate file (or more depending on `max_bytes` and `max_rows` settings) following the engine naming convention. Once all parts have been exported, new snapshots / manifest files are generated and the data is comitted using the Apache Iceberg commit mechanism. + +The manifest file produced by the commit contains a summary field `clickhouse.export-partition-transaction-id` that stores the transaction id. This field is used to implement idempotency and avoid data duplication. Some Apache Iceberg storage managers employ old manifests cleanup, ClickHouse does not. + +**IMPORTANT**: In case the storage is managed by a 3rd party application that cleans up old manifest files, it is important that the TTL of such files are greater than the timeout of export partition tasks. If it is not configured in such a way, it is possible to accidentally duplicate data in the extremely rare case a ClickHouse node is the only node working on a given export task, commits the data to Iceberg, crashes before marking the task as done and only boots up after the manifest cleanup has deleted the commit manifest. In such scenario, ClickHouse would attempt to commit those files again producing duplicates. The task timeout on ClickHouse side is controlled by the setting `export_merge_tree_partition_task_timeout_seconds`. + +The Iceberg manifest files contain statistics about the data. Exporting a merge tree partition is a non ephemeral long running task, in which nodes can be turned off and turned on. This means the stats of individual files need to be persisted somewhere in order to produce the final manifest. This is implemented through sidecars. Each data file exported will contain a "sibling" sidecar file named `_clickhouse_export_part_sidecar.avro`. ClickHouse does not clean up these files, and they can be safely deleted once the data is comitted. + +### On plain object storage exports: + +Each MergeTree part will become a separate file with the following name convention: `//_.`. To ensure atomicity, a commit file containing the relative paths of all exported parts is also shipped. A data file should only be considered part of the dataset if a commit file references it. The commit file will be named using the following convention: `/commit__`. + +## Syntax + +```sql +ALTER TABLE [database.]table_name +EXPORT PARTITION ID 'partition_id' +TO TABLE [destination_database.]destination_table +[SETTINGS setting_name = value, ...] +``` + +### Parameters + +- **`table_name`**: The source Replicated*MergeTree table containing the partition to export +- **`partition_id`**: The partition identifier to export (e.g., `'2020'`, `'2021'`) +- **`destination_table`**: The target table for the export (typically an S3, Azure, or other object storage table) + +## Settings + +### Server Settings + +#### `allow_experimental_export_merge_tree_partition` (Required) + +- **Type**: `Bool` +- **Default**: `false` +- **Description**: Enable export replicated merge tree partition feature. It is experimental and not yet ready for production use. + +### Query Settings + +#### `export_merge_tree_partition_force_export` (Optional) + +- **Type**: `Bool` +- **Default**: `false` +- **Description**: Ignore existing partition export and overwrite the ZooKeeper entry. Allows re-exporting a partition to the same destination before the manifest expires. **IMPORTANT:** this is dangerous because it can lead to duplicated data, use it with caution. + +#### `export_merge_tree_partition_max_retries` (Optional) + +- **Type**: `UInt64` +- **Default**: `3` +- **Description**: Maximum number of retries for exporting a merge tree part in an export partition task. If it exceeds, the entire task fails. + +#### `export_merge_tree_partition_manifest_ttl` (Optional) + +- **Type**: `UInt64` +- **Default**: `180` (seconds) +- **Description**: Determines how long the manifest will live in ZooKeeper. It prevents the same partition from being exported twice to the same destination. This setting does not affect or delete in-progress tasks; it only cleans up completed ones. + +#### `export_merge_tree_part_file_already_exists_policy` (Optional) + +- **Type**: `MergeTreePartExportFileAlreadyExistsPolicy` +- **Default**: `skip` +- **Description**: Policy for handling files that already exist during export. Possible values: + - `skip` - Skip the file if it already exists + - `error` - Throw an error if the file already exists + - `overwrite` - Overwrite the file + +### `export_merge_tree_part_throw_on_pending_mutations` (Optional) + +- **Type**: `bool` +- **Default**: `true` +- **Description**: If set to true, throws if pending mutations exists for a given part. Note that by default mutations are applied to all parts, which means that if a mutation in practice would only affetct part/partition x, all the other parts/partition will throw upon export. The exception is when the `IN PARTITION` clause was used in the mutation command. Note the `IN PARTITION` clause is not properly implemented for plain MergeTree tables. + +### `export_merge_tree_part_throw_on_pending_patch_parts` (Optional) + +- **Type**: `bool` +- **Default**: `true` +- **Description**: If set to true, throws if pending patch parts exists for a given part. Note that by default mutations are applied to all parts, which means that if a mutation in practice would only affetct part/partition x, all the other parts/partition will throw upon export. The exception is when the `IN PARTITION` clause was used in the mutation command. Note the `IN PARTITION` clause is not properly implemented for plain MergeTree tables. + +### `export_merge_tree_part_filename_pattern` (Optional) + +- **Type**: `String` +- **Default**: `{part_name}_{checksum}` +- **Description**: Pattern for the filename of the exported merge tree part. The `part_name` and `checksum` are calculated and replaced on the fly. Additional macros are supported. + +### `export_merge_tree_partition_task_timeout_seconds` (Optional) + +- **Type**: `UInt64` +- **Default**: `3600` +- **Description**: The timeout is measured from the manifest's create_time. Set to 0 to disable the timeout. +When the timeout is exceeded the task transitions to KILLED (same terminal state as `KILL QUERY ... EXPORT PARTITION`), and `last_exception` is populated with a timeout reason. + +Notes: +- Enforcement is best-effort: actual kill latency is bounded by one manifest-updater poll cycle (~30s) plus ZooKeeper watch propagation. +- Since both this timeout and `export_merge_tree_partition_manifest_ttl` are measured from `create_time`, keep `export_merge_tree_partition_manifest_ttl` greater than `export_merge_tree_partition_task_timeout_seconds` if you want the KILLED entry to remain visible in `system.replicated_partition_exports` after the timeout fires. + +## Examples + +### Basic Export to S3 + +```sql +CREATE TABLE rmt_table (id UInt64, year UInt16) +ENGINE = ReplicatedMergeTree('/clickhouse/tables/{database}/rmt_table', 'replica1') +PARTITION BY year ORDER BY tuple(); + +CREATE TABLE s3_table (id UInt64, year UInt16) +ENGINE = S3(s3_conn, filename='data', format=Parquet, partition_strategy='hive') +PARTITION BY year; + +INSERT INTO rmt_table VALUES (1, 2020), (2, 2020), (3, 2020), (4, 2021); + +ALTER TABLE rmt_table EXPORT PARTITION ID '2020' TO TABLE s3_table; + +## Killing Exports + +You can cancel in-progress partition exports using the `KILL EXPORT PARTITION` command: + +```sql +KILL EXPORT PARTITION +WHERE partition_id = '2020' + AND source_table = 'rmt_table' + AND destination_table = 's3_table' +``` + +The `WHERE` clause filters exports from the `system.replicated_partition_exports` table. You can use any columns from that table in the filter. + +## Monitoring + +### Active and Completed Exports + +Monitor partition exports using the `system.replicated_partition_exports` table: + +```sql +arthur :) select * from system.replicated_partition_exports Format Vertical; + +SELECT * +FROM system.replicated_partition_exports +FORMAT Vertical + +Query id: 9efc271a-a501-44d1-834f-bc4d20156164 + +Row 1: +────── +source_database: default +source_table: replicated_source +destination_database: default +destination_table: replicated_destination +create_time: 2025-11-21 18:21:51 +partition_id: 2022 +transaction_id: 7397746091717128192 +source_replica: r1 +parts: ['2022_0_0_0','2022_1_1_0','2022_2_2_0'] +parts_count: 3 +parts_to_do: 0 +status: COMPLETED +exception_replica: +last_exception: +exception_part: +exception_count: 0 + +Row 2: +────── +source_database: default +source_table: replicated_source +destination_database: default +destination_table: replicated_destination +create_time: 2025-11-21 18:20:35 +partition_id: 2021 +transaction_id: 7397745772618674176 +source_replica: r1 +parts: ['2021_0_0_0'] +parts_count: 1 +parts_to_do: 0 +status: COMPLETED +exception_replica: +last_exception: +exception_part: +exception_count: 0 + +2 rows in set. Elapsed: 0.019 sec. + +arthur :) +``` + +Status values include: +- `PENDING` - Export is queued / in progress +- `COMPLETED` - Export finished successfully +- `FAILED` - Export failed +- `KILLED` - Export was cancelled + +## Related Features + +- [ALTER TABLE EXPORT PART](/docs/en/engines/table-engines/mergetree-family/part_export.md) - Export individual parts (non-replicated) + diff --git a/docs/en/operations/system-tables/exports.md b/docs/en/operations/system-tables/exports.md new file mode 100644 index 000000000000..e26514364008 --- /dev/null +++ b/docs/en/operations/system-tables/exports.md @@ -0,0 +1,56 @@ +--- +description: 'System table containing information about in progress merge tree part exports' +keywords: ['system table', 'exports', 'merge tree', 'part'] +slug: /operations/system-tables/exports +title: 'system.exports' +--- + +Contains information about in progress merge tree part exports + +Columns: + +- `source_database` ([String](/docs/en/sql-reference/data-types/string.md)) — Name of the source database. +- `source_table` ([String](/docs/en/sql-reference/data-types/string.md)) — Name of the source table. +- `destination_database` ([String](/docs/en/sql-reference/data-types/string.md)) — Name of the destination database. +- `destination_table` ([String](/docs/en/sql-reference/data-types/string.md)) — Name of the destination table. +- `create_time` ([DateTime](/docs/en/sql-reference/data-types/datetime.md)) — Date and time when the export command was received in the server. +- `part_name` ([String](/docs/en/sql-reference/data-types/string.md)) — Name of the part. +- `destination_file_path` ([String](/docs/en/sql-reference/data-types/string.md)) — File path relative to where the part is being exported to. +- `elapsed` ([Float64](/docs/en/sql-reference/data-types/float.md)) — The time elapsed (in seconds) since the export started. +- `rows_read` ([UInt64](/docs/en/sql-reference/data-types/int-uint.md)) — The number of rows read from the exported part. +- `total_rows_to_read` ([UInt64](/docs/en/sql-reference/data-types/int-uint.md)) — The total number of rows to read from the exported part. +- `total_size_bytes_compressed` ([UInt64](/docs/en/sql-reference/data-types/int-uint.md)) — The total size of the compressed data in the exported part. +- `total_size_bytes_uncompressed` ([UInt64](/docs/en/sql-reference/data-types/int-uint.md)) — The total size of the uncompressed data in the exported part. +- `bytes_read_uncompressed` ([UInt64](/docs/en/sql-reference/data-types/int-uint.md)) — The number of uncompressed bytes read from the exported part. +- `memory_usage` ([UInt64](/docs/en/sql-reference/data-types/int-uint.md)) — Current memory usage in bytes for the export operation. +- `peak_memory_usage` ([UInt64](/docs/en/sql-reference/data-types/int-uint.md)) — Peak memory usage in bytes during the export operation. + +**Example** + +```sql +arthur :) select * from system.exports; + +SELECT * +FROM system.exports + +Query id: 2026718c-d249-4208-891b-a271f1f93407 + +Row 1: +────── +source_database: default +source_table: source_mt_table +destination_database: default +destination_table: destination_table +create_time: 2025-11-19 09:09:11 +part_name: 20251016-365_1_1_0 +destination_file_path: table_root/eventDate=2025-10-16/retention=365/20251016-365_1_1_0_17B2F6CD5D3C18E787C07AE3DAF16EB1.parquet +elapsed: 2.04845441 +rows_read: 1138688 -- 1.14 million +total_rows_to_read: 550961374 -- 550.96 million +total_size_bytes_compressed: 37619147120 -- 37.62 billion +total_size_bytes_uncompressed: 138166213721 -- 138.17 billion +bytes_read_uncompressed: 316892925 -- 316.89 million +memory_usage: 596006095 -- 596.01 million +peak_memory_usage: 601239033 -- 601.24 million +``` + diff --git a/programs/server/Server.cpp b/programs/server/Server.cpp index 9260a3c9d2dc..449db39ee6ac 100644 --- a/programs/server/Server.cpp +++ b/programs/server/Server.cpp @@ -94,6 +94,7 @@ #include #include #include +#include #include #include #include @@ -462,6 +463,9 @@ namespace ServerSetting extern const ServerSettingsString hdfs_libhdfs3_conf; extern const ServerSettingsString config_file; extern const ServerSettingsString users_to_ignore_early_memory_limit_check; + extern const ServerSettingsUInt64 object_storage_list_objects_cache_ttl; + extern const ServerSettingsUInt64 object_storage_list_objects_cache_size; + extern const ServerSettingsUInt64 object_storage_list_objects_cache_max_entries; } namespace ErrorCodes @@ -472,6 +476,9 @@ namespace ErrorCodes namespace FileCacheSetting { extern const FileCacheSettingsBool load_metadata_asynchronously; + extern const ServerSettingsUInt64 object_storage_list_objects_cache_size; + extern const ServerSettingsUInt64 object_storage_list_objects_cache_max_entries; + extern const ServerSettingsUInt64 object_storage_list_objects_cache_ttl; } } @@ -3114,6 +3121,10 @@ try /// try set up encryption. There are some errors in config, error will be printed and server wouldn't start. CompressionCodecEncrypted::Configuration::instance().load(config(), "encryption_codecs"); + ObjectStorageListObjectsCache::instance().setMaxSizeInBytes(server_settings[ServerSetting::object_storage_list_objects_cache_size]); + ObjectStorageListObjectsCache::instance().setMaxCount(server_settings[ServerSetting::object_storage_list_objects_cache_max_entries]); + ObjectStorageListObjectsCache::instance().setTTL(server_settings[ServerSetting::object_storage_list_objects_cache_ttl]); + auto replicas_reconnector = ReplicasReconnector::init(global_context); /// Set current database name before loading tables and databases because diff --git a/src/Access/Common/AccessType.h b/src/Access/Common/AccessType.h index 38766a2ed170..3b50e562c3a7 100644 --- a/src/Access/Common/AccessType.h +++ b/src/Access/Common/AccessType.h @@ -213,6 +213,8 @@ enum class AccessType : uint8_t M(ALTER_REWRITE_PARTS, "REWRITE PARTS", TABLE, ALTER_TABLE) /* allows to execute ALTER REWRITE PARTS */\ M(ALTER_SETTINGS, "ALTER SETTING, ALTER MODIFY SETTING, MODIFY SETTING, RESET SETTING", TABLE, ALTER_TABLE) /* allows to execute ALTER MODIFY SETTING */\ M(ALTER_MOVE_PARTITION, "ALTER MOVE PART, MOVE PARTITION, MOVE PART", TABLE, ALTER_TABLE) \ + M(ALTER_EXPORT_PART, "ALTER EXPORT PART, EXPORT PART", TABLE, ALTER_TABLE) \ + M(ALTER_EXPORT_PARTITION, "ALTER EXPORT PARTITION, EXPORT PARTITION", TABLE, ALTER_TABLE) \ M(ALTER_FETCH_PARTITION, "ALTER FETCH PART, FETCH PARTITION", TABLE, ALTER_TABLE) \ M(ALTER_FREEZE_PARTITION, "FREEZE PARTITION, UNFREEZE", TABLE, ALTER_TABLE) \ M(ALTER_UNLOCK_SNAPSHOT, "UNLOCK SNAPSHOT", TABLE, ALTER_TABLE) \ @@ -337,6 +339,7 @@ enum class AccessType : uint8_t M(SYSTEM_DROP_SCHEMA_CACHE, "SYSTEM CLEAR SCHEMA CACHE, SYSTEM DROP SCHEMA CACHE, DROP SCHEMA CACHE", GLOBAL, SYSTEM_DROP_CACHE) \ M(SYSTEM_DROP_FORMAT_SCHEMA_CACHE, "SYSTEM CLEAR FORMAT SCHEMA CACHE, SYSTEM DROP FORMAT SCHEMA CACHE, DROP FORMAT SCHEMA CACHE", GLOBAL, SYSTEM_DROP_CACHE) \ M(SYSTEM_DROP_S3_CLIENT_CACHE, "SYSTEM CLEAR S3 CLIENT CACHE, SYSTEM DROP S3 CLIENT, DROP S3 CLIENT CACHE", GLOBAL, SYSTEM_DROP_CACHE) \ + M(SYSTEM_DROP_OBJECT_STORAGE_LIST_OBJECTS_CACHE, "SYSTEM DROP OBJECT STORAGE LIST OBJECTS CACHE", GLOBAL, SYSTEM_DROP_CACHE) \ M(SYSTEM_DROP_CACHE, "DROP CACHE", GROUP, SYSTEM) \ M(SYSTEM_RELOAD_CONFIG, "RELOAD CONFIG", GLOBAL, SYSTEM_RELOAD) \ M(SYSTEM_RELOAD_USERS, "RELOAD USERS", GLOBAL, SYSTEM_RELOAD) \ diff --git a/src/CMakeLists.txt b/src/CMakeLists.txt index 7a9602cdbb29..17ae45393ed7 100644 --- a/src/CMakeLists.txt +++ b/src/CMakeLists.txt @@ -149,6 +149,7 @@ add_headers_and_sources(dbms Storages/ObjectStorage/Azure) add_headers_and_sources(dbms Storages/ObjectStorage/S3) add_headers_and_sources(dbms Storages/ObjectStorage/HDFS) add_headers_and_sources(dbms Storages/ObjectStorage/Local) +add_headers_and_sources(dbms Storages/ObjectStorage/MergeTree) add_headers_and_sources(dbms Storages/ObjectStorage/DataLakes) add_headers_and_sources(dbms Storages/ObjectStorage/DataLakes/Common) add_headers_and_sources(dbms Storages/ObjectStorage/DataLakes/Iceberg) diff --git a/src/Common/CurrentMetrics.cpp b/src/Common/CurrentMetrics.cpp index cdb399e249ce..0c28da23d6fe 100644 --- a/src/Common/CurrentMetrics.cpp +++ b/src/Common/CurrentMetrics.cpp @@ -12,6 +12,7 @@ M(Merge, "Number of executing background merges") \ M(MergeParts, "Number of source parts participating in current background merges") \ M(Move, "Number of currently executing moves") \ + M(Export, "Number of currently executing exports") \ M(PartMutation, "Number of mutations (ALTER DELETE/UPDATE)") \ M(ReplicatedFetch, "Number of data parts being fetched from replica") \ M(ReplicatedSend, "Number of data parts being sent to replicas") \ diff --git a/src/Common/ErrorCodes.cpp b/src/Common/ErrorCodes.cpp index 07ec3fff851f..a0f3b90d7109 100644 --- a/src/Common/ErrorCodes.cpp +++ b/src/Common/ErrorCodes.cpp @@ -671,10 +671,14 @@ M(1002, UNKNOWN_EXCEPTION) \ M(1003, SSH_EXCEPTION) \ M(1004, STARTUP_SCRIPTS_ERROR) \ +<<<<<<< HEAD M(1005, STALE_VERSION) \ M(1006, INVALID_CURSOR_LOOKUP) \ M(1007, ILLEGAL_STREAM) \ M(1008, TEMPORARY_DATA_NOT_IN_CACHE) \ +======= + M(1005, PENDING_MUTATIONS_NOT_ALLOWED) \ +>>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) /* See END */ #ifdef APPLY_FOR_EXTERNAL_ERROR_CODES @@ -691,7 +695,11 @@ namespace ErrorCodes APPLY_FOR_ERROR_CODES(M) #undef M +<<<<<<< HEAD constexpr ErrorCode END = 1008; +======= + constexpr ErrorCode END = 1005; +>>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) ErrorPairHolder values[END + 1]{}; struct ErrorCodesNames diff --git a/src/Common/FailPoint.cpp b/src/Common/FailPoint.cpp index 71b6b731d871..af7415cc19af 100644 --- a/src/Common/FailPoint.cpp +++ b/src/Common/FailPoint.cpp @@ -162,7 +162,15 @@ static struct InitFiu ONCE(write_file_operation_fail_on_read) \ REGULAR(slowdown_parallel_replicas_local_plan_read) \ ONCE(iceberg_writes_cleanup) \ +<<<<<<< HEAD REGULAR(storage_cluster_read_sleep) \ +======= + ONCE(iceberg_writes_non_retry_cleanup) \ + ONCE(iceberg_writes_post_publish_throw) \ + ONCE(iceberg_export_after_commit_before_zk_completed) \ + REGULAR(export_partition_commit_always_throw) \ + ONCE(export_partition_status_change_throw) \ +>>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) ONCE(backup_add_empty_memory_table) \ PAUSEABLE_ONCE(backup_pause_on_start) \ PAUSEABLE_ONCE(restore_pause_on_start) \ diff --git a/src/Common/ProfileEvents.cpp b/src/Common/ProfileEvents.cpp index 7f35ef869e92..6d868b24bb3c 100644 --- a/src/Common/ProfileEvents.cpp +++ b/src/Common/ProfileEvents.cpp @@ -39,6 +39,10 @@ M(FailedInitialQuery, "Number of failed initial queries.", ValueType::Number) \ M(FailedInitialSelectQuery, "Same as FailedInitialQuery, but only for SELECT queries.", ValueType::Number) \ M(FailedQuery, "Number of total failed queries, both internal and user queries.", ValueType::Number) \ + M(PartsExports, "Number of successful part exports.", ValueType::Number) \ + M(PartsExportFailures, "Number of failed part exports.", ValueType::Number) \ + M(PartsExportDuplicated, "Number of part exports that failed because target already exists.", ValueType::Number) \ + M(PartsExportTotalMilliseconds, "Total time spent on part export operations.", ValueType::Milliseconds) \ M(FailedSelectQuery, "Same as FailedQuery, but only for SELECT queries.", ValueType::Number) \ M(FailedInsertQuery, "Same as FailedQuery, but only for INSERT queries.", ValueType::Number) \ M(FailedAsyncInsertQuery, "Number of failed ASYNC INSERT queries.", ValueType::Number) \ @@ -237,10 +241,15 @@ M(MergesThrottlerSleepMicroseconds, "Total time a query was sleeping to conform 'max_merges_bandwidth_for_server' throttling.", ValueType::Microseconds) \ M(MutationsThrottlerBytes, "Bytes passed through 'max_mutations_bandwidth_for_server' throttler.", ValueType::Bytes) \ M(MutationsThrottlerSleepMicroseconds, "Total time a query was sleeping to conform 'max_mutations_bandwidth_for_server' throttling.", ValueType::Microseconds) \ +<<<<<<< HEAD M(UserThrottlerBytes, "Bytes passed through 'max_network_bandwidth_for_user' throttler.", ValueType::Bytes) \ M(UserThrottlerSleepMicroseconds, "Total time a query was sleeping to conform 'max_network_bandwidth_for_user' throttling.", ValueType::Microseconds) \ M(AllUsersThrottlerBytes, "Bytes passed through 'max_network_bandwidth_for_all_users' throttler.", ValueType::Bytes) \ M(AllUsersThrottlerSleepMicroseconds, "Total time a query was sleeping to conform 'max_network_bandwidth_for_all_users' throttling.", ValueType::Microseconds) \ +======= + M(ExportsThrottlerBytes, "Bytes passed through 'max_exports_bandwidth_for_server' throttler.", ValueType::Bytes) \ + M(ExportsThrottlerSleepMicroseconds, "Total time a query was sleeping to conform 'max_exports_bandwidth_for_server' throttling.", ValueType::Microseconds) \ +>>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) M(QueryRemoteReadThrottlerBytes, "Bytes passed through 'max_remote_read_network_bandwidth' throttler.", ValueType::Bytes) \ M(QueryRemoteReadThrottlerSleepMicroseconds, "Total time a query was sleeping to conform 'max_remote_read_network_bandwidth' throttling.", ValueType::Microseconds) \ M(ReaderExecutorSourceRequests, "Number of source-side requests opened by ReaderExecutor (excludes live-buffer reuses).", ValueType::Number) \ @@ -364,6 +373,18 @@ M(ZooKeeperBytesSent, "Number of bytes send over network while communicating with ZooKeeper.", ValueType::Bytes) \ M(ZooKeeperBytesReceived, "Number of bytes received over network while communicating with ZooKeeper.", ValueType::Bytes) \ \ + M(ExportPartitionZooKeeperRequests, "Total number of ZooKeeper requests made by the export partition feature.", ValueType::Number) \ + M(ExportPartitionZooKeeperGet, "Number of 'get' requests to ZooKeeper made by the export partition feature.", ValueType::Number) \ + M(ExportPartitionZooKeeperGetChildren, "Number of 'getChildren' requests to ZooKeeper made by the export partition feature.", ValueType::Number) \ + M(ExportPartitionZooKeeperGetChildrenWatch, "Number of 'getChildrenWatch' requests to ZooKeeper made by the export partition feature.", ValueType::Number) \ + M(ExportPartitionZooKeeperGetWatch, "Number of 'getWatch' requests to ZooKeeper made by the export partition feature.", ValueType::Number) \ + M(ExportPartitionZooKeeperCreate, "Number of 'create' requests to ZooKeeper made by the export partition feature.", ValueType::Number) \ + M(ExportPartitionZooKeeperSet, "Number of 'set' requests to ZooKeeper made by the export partition feature.", ValueType::Number) \ + M(ExportPartitionZooKeeperRemove, "Number of 'remove' requests to ZooKeeper made by the export partition feature.", ValueType::Number) \ + M(ExportPartitionZooKeeperRemoveRecursive, "Number of 'removeRecursive' requests to ZooKeeper made by the export partition feature.", ValueType::Number) \ + M(ExportPartitionZooKeeperMulti, "Number of 'multi' requests to ZooKeeper made by the export partition feature.", ValueType::Number) \ + M(ExportPartitionZooKeeperExists, "Number of 'exists' requests to ZooKeeper made by the export partition feature.", ValueType::Number) \ + \ M(DistributedConnectionTries, "Total count of distributed connection attempts.", ValueType::Number) \ M(DistributedConnectionUsable, "Total count of successful distributed connections to a usable server (with required table, but maybe stale).", ValueType::Number) \ M(DistributedConnectionFailTry, "Total count when distributed connection fails with retry.", ValueType::Number) \ @@ -1495,6 +1516,7 @@ The server successfully detected this situation and will download merged part fr M(RuntimeFilterRowsPassed, "Number of rows that passed (not filtered out by) JOIN Runtime Filters", ValueType::Number) \ M(RuntimeFilterRowsSkipped, "Number of rows in blocks that were skipped by JOIN Runtime Filters", ValueType::Number) \ \ +<<<<<<< HEAD M(JoinBuildPostProcessingMicroseconds, "Elapsed time of post-processing steps after building the right JOIN side.", ValueType::Microseconds) \ \ M(AIInputTokens, "Total prompt tokens consumed across all AI function calls in the query.", ValueType::Number) \ @@ -1503,6 +1525,18 @@ The server successfully detected this situation and will download merged part fr M(AIRowsProcessed, "Number of rows that received an AI result.", ValueType::Number) \ M(AIRowsSkipped, "Number of rows that received a default value due to quota or error.", ValueType::Number) \ \ +======= + M(ObjectStorageClusterSentToMatchedReplica, "Number of tasks in ObjectStorageCluster request sent to matched replica.", ValueType::Number) \ + M(ObjectStorageClusterSentToNonMatchedReplica, "Number of tasks in ObjectStorageCluster request sent to non-matched replica.", ValueType::Number) \ + M(ObjectStorageClusterProcessedTasks, "Number of processed tasks in ObjectStorageCluster request.", ValueType::Number) \ + M(ObjectStorageClusterWaitingMicroseconds, "Time of waiting for tasks in ObjectStorageCluster request.", ValueType::Microseconds) \ + \ + M(ObjectStorageListObjectsCacheHits, "Number of times object storage list objects operation hit the cache.", ValueType::Number) \ + M(ObjectStorageListObjectsCacheMisses, "Number of times object storage list objects operation miss the cache.", ValueType::Number) \ + M(ObjectStorageListObjectsCacheExactMatchHits, "Number of times object storage list objects operation hit the cache with an exact match.", ValueType::Number) \ + M(ObjectStorageListObjectsCachePrefixMatchHits, "Number of times object storage list objects operation miss the cache using prefix matching.", ValueType::Number) \ + +>>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) #ifdef APPLY_FOR_EXTERNAL_EVENTS #define APPLY_FOR_EVENTS(M) APPLY_FOR_BUILTIN_EVENTS(M) APPLY_FOR_EXTERNAL_EVENTS(M) diff --git a/src/Common/TTLCachePolicy.h b/src/Common/TTLCachePolicy.h index d481d0290330..1f894cb35bdb 100644 --- a/src/Common/TTLCachePolicy.h +++ b/src/Common/TTLCachePolicy.h @@ -278,10 +278,10 @@ class TTLCachePolicy : public ICachePolicy; Cache cache; - +private: /// TODO To speed up removal of stale entries, we could also add another container sorted on expiry times which maps keys to iterators /// into the cache. To insert an entry, add it to the cache + add the iterator to the sorted container. To remove stale entries, do a /// binary search on the sorted container and erase all left of the found key. diff --git a/src/Common/setThreadName.h b/src/Common/setThreadName.h index 437b5c7f184c..9992fd9c86de 100644 --- a/src/Common/setThreadName.h +++ b/src/Common/setThreadName.h @@ -169,7 +169,11 @@ namespace DB M(ZOOKEEPER_SEND, "ZooKeeperSend") \ M(BLOB_KILLER_TASK, "BlobKillerTask") \ M(BLOB_COPIER_TASK, "BlobCopierTask") \ +<<<<<<< HEAD M(DISK_OBJECT_STORAGE_COPY, "DiskObjStCopy") \ +======= + M(EXPORT_PART, "ExportPart") \ +>>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) enum class ThreadName : uint8_t diff --git a/src/Core/ServerSettings.cpp b/src/Core/ServerSettings.cpp index 1f6eb7e54653..556eef34b50e 100644 --- a/src/Core/ServerSettings.cpp +++ b/src/Core/ServerSettings.cpp @@ -166,6 +166,7 @@ namespace DECLARE(UInt64, max_unexpected_parts_loading_thread_pool_size, 8, R"(The number of threads to load inactive set of data parts (Unexpected ones) at startup.)", 0) \ DECLARE(UInt64, max_parts_cleaning_thread_pool_size, 128, R"(The number of threads for concurrent removal of inactive data parts.)", 0) \ DECLARE(UInt64, max_mutations_bandwidth_for_server, 0, R"(The maximum read speed of all mutations on server in bytes per second. Zero means unlimited.)", 0) \ + DECLARE(UInt64, max_exports_bandwidth_for_server, 0, R"(The maximum read speed of all exports on server in bytes per second. Zero means unlimited.)", 0) \ DECLARE(UInt64, max_merges_bandwidth_for_server, 0, R"(The maximum read speed of all merges on server in bytes per second. Zero means unlimited.)", 0) \ DECLARE(UInt64, max_replicated_fetches_network_bandwidth_for_server, 0, R"(The maximum speed of data exchange over the network in bytes per second for replicated fetches. Zero means unlimited.)", 0) \ DECLARE(UInt64, max_replicated_sends_network_bandwidth_for_server, 0, R"(The maximum speed of data exchange over the network in bytes per second for replicated sends. Zero means unlimited.)", 0) \ @@ -1657,7 +1658,11 @@ The policy on how to perform a scheduling of CPU slots specified by `concurrent_ ```xml 1 ``` - )", 0) + )", 0) \ + DECLARE(UInt64, object_storage_list_objects_cache_size, 500000000, "Maximum size of ObjectStorage list objects cache in bytes. Zero means disabled.", 0) \ + DECLARE(UInt64, object_storage_list_objects_cache_max_entries, 1000, "Maximum size of ObjectStorage list objects cache in entries. Zero means disabled.", 0) \ + DECLARE(UInt64, object_storage_list_objects_cache_ttl, 3600, "Time to live of records in ObjectStorage list objects cache in seconds. Zero means unlimited", 0) \ + DECLARE(Bool, allow_experimental_export_merge_tree_partition, false, "Enable export replicated merge tree partition feature. It is experimental and not yet ready for production use.", 0) /// Settings with a path are server settings with at least one layer of nesting that have a fixed structure (no lists, lists, enumerations, repetitions, ...). #define LIST_OF_SERVER_SETTINGS_WITH_PATH(DECLARE, ALIAS) \ diff --git a/src/Core/Settings.cpp b/src/Core/Settings.cpp index 99ed9098617a..8bf8969f286a 100644 --- a/src/Core/Settings.cpp +++ b/src/Core/Settings.cpp @@ -7970,6 +7970,7 @@ Use `allow_nullable_tuple_in_extracted_subcolumns` to control whether extracted )", BETA, enable_nullable_tuple_type) \ DECLARE(UInt64, archive_adaptive_buffer_max_size_bytes, 8 * DBMS_DEFAULT_BUFFER_SIZE, R"( Limits the maximum size of the adaptive buffer used when writing to archive files (for example, tar archives)", 0) \ +<<<<<<< HEAD DECLARE(UInt64, shared_merge_tree_sequential_consistency_initial_parts_update_backoff_ms, 50, R"( Initial backoff in milliseconds for parts update when using `select_sequential_consistency` with `SharedMergeTree`. Only available in ClickHouse Cloud. )", 0) \ @@ -7998,6 +7999,60 @@ Enable converting the hash table to a flat array for joins when the key is a sin )", 0) \ DECLARE(UInt64, query_plan_min_columns_for_join_lazy_indexing, 3, R"( Control the minimum number of payload columns from the left side required for enabling lazy indexing optimization in JOIN. 0 means the optimization is disabled. +======= + DECLARE(Bool, export_merge_tree_part_overwrite_file_if_exists, false, R"( +Overwrite file if it already exists when exporting a merge tree part +)", 0) \ + DECLARE(Bool, export_merge_tree_partition_force_export, false, R"( +Ignore existing partition export and overwrite the zookeeper entry +)", 0) \ + DECLARE(UInt64, export_merge_tree_partition_max_retries, 3, R"( +Maximum number of retries for exporting a merge tree part in an export partition task +)", 0) \ + DECLARE(UInt64, export_merge_tree_partition_manifest_ttl, 86400, R"( +Determines how long the manifest will live in ZooKeeper. It prevents the same partition from being exported twice to the same destination. +This setting does not affect / delete in progress tasks. It'll only cleanup the completed ones. +)", 0) \ + DECLARE(UInt64, export_merge_tree_partition_task_timeout_seconds, 3600, R"( +Maximum wall-clock duration (in seconds) an export partition task is allowed to remain in the PENDING state before it is auto-killed by the background cleanup loop. +The timeout is measured from the manifest's create_time. Set to 0 to disable the timeout. +When the timeout is exceeded the task transitions to KILLED (same terminal state as `KILL QUERY ... EXPORT PARTITION`), and `last_exception` is populated with a timeout reason. + +Notes: +- Enforcement is best-effort: actual kill latency is bounded by one manifest-updater poll cycle (~30s) plus ZooKeeper watch propagation. +- Since both this timeout and `export_merge_tree_partition_manifest_ttl` are measured from `create_time`, keep `export_merge_tree_partition_manifest_ttl` greater than `export_merge_tree_partition_task_timeout_seconds` if you want the KILLED entry to remain visible in `system.replicated_partition_exports` after the timeout fires. +)", 0) \ + DECLARE(MergeTreePartExportFileAlreadyExistsPolicy, export_merge_tree_part_file_already_exists_policy, MergeTreePartExportFileAlreadyExistsPolicy::skip, R"( +Possible values: +- skip - Skip the file if it already exists. +- error - Throw an error if the file already exists. +- overwrite - Overwrite the file. +)", 0) \ + DECLARE(UInt64, export_merge_tree_part_max_bytes_per_file, 0, R"( +Maximum number of bytes to write to a single file when exporting a merge tree part. 0 means no limit. +This is not a hard limit, and it highly depends on the output format granularity and input source chunk size. +)", 0) \ + DECLARE(UInt64, export_merge_tree_part_max_rows_per_file, 0, R"( +Maximum number of rows to write to a single file when exporting a merge tree part. 0 means no limit. +This is not a hard limit, and it highly depends on the output format granularity and input source chunk size. +)", 0) \ + DECLARE(Bool, export_merge_tree_part_throw_on_pending_mutations, true, R"( +Throw an error if there are pending mutations when exporting a merge tree part. +)", 0) \ + DECLARE(Bool, export_merge_tree_part_throw_on_pending_patch_parts, true, R"( +Throw an error if there are pending patch parts when exporting a merge tree part. +)", 0) \ + DECLARE(Bool, export_merge_tree_partition_lock_inside_the_task, false, R"( +Only lock a part when the task is already running. This might help with busy waiting where the scheduler locks a part, but the task ends in the pending list. +On the other hand, there is a chance once the task executes that part has already been locked by another replica and the task will simply early exit. +)", 0) \ + DECLARE(Bool, export_merge_tree_partition_system_table_prefer_remote_information, false, R"( +Controls whether the system.replicated_partition_exports will prefer to query ZooKeeper to get the most up to date information or use the local information. +Querying ZooKeeper is expensive, and only available if the ZooKeeper feature flag MULTI_READ is enabled. +)", 0) \ + DECLARE(String, export_merge_tree_part_filename_pattern, "{part_name}_{checksum}", R"( +Pattern for the filename of the exported merge tree part. The `part_name` and `checksum` are calculated and replaced on the fly. Additional macros are supported. +>>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) )", 0) \ \ /* ####################################################### */ \ @@ -8205,6 +8260,9 @@ Write full paths (including s3://) into iceberg metadata files. )", EXPERIMENTAL) \ DECLARE(String, iceberg_metadata_compression_method, "", R"( Method to compress `.metadata.json` file. +)", EXPERIMENTAL) \ + DECLARE(Bool, use_object_storage_list_objects_cache, false, R"( +Cache the list of objects returned by list objects calls in object storage )", EXPERIMENTAL) \ DECLARE(Bool, make_distributed_plan, false, R"( Make distributed query plan. @@ -8283,6 +8341,9 @@ Rewrite expressions like 'x IN subquery' to JOIN. This might be useful for optim DECLARE_WITH_ALIAS(Bool, allow_experimental_time_series_aggregate_functions, false, R"( Experimental timeSeries* aggregate functions for Prometheus-like timeseries resampling, rate, delta calculation. )", EXPERIMENTAL, allow_experimental_ts_to_grid_aggregate_function) \ + DECLARE(Bool, allow_experimental_export_merge_tree_part, true, R"( +Experimental export merge tree part. +)", EXPERIMENTAL) \ \ DECLARE(String, promql_database, "", R"( Specifies the database name used by the 'promql' dialect. Empty string means the current database. diff --git a/src/Core/Settings.h b/src/Core/Settings.h index 08620f364d07..49a067390657 100644 --- a/src/Core/Settings.h +++ b/src/Core/Settings.h @@ -85,6 +85,7 @@ class WriteBuffer; M(CLASS_NAME, LogsLevel) \ M(CLASS_NAME, Map) \ M(CLASS_NAME, MaxThreads) \ + M(CLASS_NAME, MergeTreePartExportFileAlreadyExistsPolicy) \ M(CLASS_NAME, Milliseconds) \ M(CLASS_NAME, MsgPackUUIDRepresentation) \ M(CLASS_NAME, MySQLDataTypesSupport) \ diff --git a/src/Core/SettingsChangesHistory.cpp b/src/Core/SettingsChangesHistory.cpp index 8c1dc4774a75..b7df09208b95 100644 --- a/src/Core/SettingsChangesHistory.cpp +++ b/src/Core/SettingsChangesHistory.cpp @@ -269,11 +269,13 @@ const VersionToSettingsChangesMap & getSettingsChangesHistory() { // {"iceberg_partition_timezone", "", "", "New setting."}, // {"s3_propagate_credentials_to_other_storages", false, false, "New setting"}, - // {"export_merge_tree_part_filename_pattern", "", "{part_name}_{checksum}", "New setting"}, + {"export_merge_tree_part_filename_pattern", "", "{part_name}_{checksum}", "New setting"}, // {"use_parquet_metadata_cache", false, true, "Enables cache of parquet file metadata."}, // {"input_format_parquet_use_metadata_cache", true, false, "Obsolete. No-op"}, // https://github.com/Altinity/ClickHouse/pull/586 // {"object_storage_remote_initiator_cluster", "", "", "New setting."}, // {"iceberg_metadata_staleness_ms", 0, 0, "New setting allowing using cached metadata version at READ operations to prevent fetching from remote catalog"}, + {"export_merge_tree_partition_task_timeout_seconds", 0, 3600, "New setting to control the timeout for export partition tasks."}, + {"export_merge_tree_partition_manifest_ttl", 180, 86400, "Reasonable default for real usage"}, }); addSettingsChanges(settings_changes_history, "26.1", { @@ -485,8 +487,26 @@ const VersionToSettingsChangesMap & getSettingsChangesHistory() // {"export_merge_tree_partition_system_table_prefer_remote_information", true, true, "New setting."}, // {"export_merge_tree_part_throw_on_pending_mutations", true, true, "New setting."}, // {"export_merge_tree_part_throw_on_pending_patch_parts", true, true, "New setting."}, + {"lock_object_storage_task_distribution_ms", 500, 500, "New setting."}, + {"allow_retries_in_cluster_requests", false, false, "New setting"}, + {"allow_experimental_export_merge_tree_part", false, true, "Turned ON by default for Antalya."}, + {"export_merge_tree_part_overwrite_file_if_exists", false, false, "New setting."}, + {"export_merge_tree_partition_force_export", false, false, "New setting."}, + {"export_merge_tree_partition_max_retries", 3, 3, "New setting."}, + {"export_merge_tree_partition_manifest_ttl", 180, 180, "New setting."}, + {"export_merge_tree_part_file_already_exists_policy", "skip", "skip", "New setting."}, + {"hybrid_table_auto_cast_columns", true, true, "New setting to automatically cast Hybrid table columns when segments disagree on types. Default enabled."}, + {"allow_experimental_hybrid_table", false, false, "Added new setting to allow the Hybrid table engine."}, + {"enable_alias_marker", true, true, "New setting."}, + {"export_merge_tree_part_max_bytes_per_file", 0, 0, "New setting."}, + {"export_merge_tree_part_max_rows_per_file", 0, 0, "New setting."}, + {"export_merge_tree_partition_lock_inside_the_task", false, false, "New setting."}, + {"export_merge_tree_partition_system_table_prefer_remote_information", true, true, "New setting."}, + {"export_merge_tree_part_throw_on_pending_mutations", true, true, "New setting."}, + {"export_merge_tree_part_throw_on_pending_patch_parts", true, true, "New setting."}, // {"object_storage_cluster", "", "", "Antalya: New setting"}, // {"object_storage_max_nodes", 0, 0, "Antalya: New setting"}, + {"use_object_storage_list_objects_cache", false, false, "New setting."}, }); addSettingsChanges(settings_changes_history, "25.8", { diff --git a/src/Core/SettingsEnums.cpp b/src/Core/SettingsEnums.cpp index 0ff8daf7faa0..56ab66303116 100644 --- a/src/Core/SettingsEnums.cpp +++ b/src/Core/SettingsEnums.cpp @@ -504,6 +504,7 @@ IMPLEMENT_SETTING_ENUM(JemallocProfileFormat, ErrorCodes::BAD_ARGUMENTS, {"symbolized", JemallocProfileFormat::Symbolized}, {"collapsed", JemallocProfileFormat::Collapsed}}) +<<<<<<< HEAD IMPLEMENT_SETTING_ENUM(S3UriStyle, ErrorCodes::BAD_ARGUMENTS, {{"auto", S3UriStyle::AUTO}, {"path", S3UriStyle::PATH}, @@ -514,4 +515,8 @@ IMPLEMENT_SETTING_ENUM( ErrorCodes::BAD_ARGUMENTS, {{"wildcard", FileLikeEngineDefaultPartitionStrategy::WILDCARD}, {"hive", FileLikeEngineDefaultPartitionStrategy::HIVE}}) +======= +IMPLEMENT_SETTING_AUTO_ENUM(MergeTreePartExportFileAlreadyExistsPolicy, ErrorCodes::BAD_ARGUMENTS); + +>>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) } diff --git a/src/Core/SettingsEnums.h b/src/Core/SettingsEnums.h index 09d175e72680..8ea761671db3 100644 --- a/src/Core/SettingsEnums.h +++ b/src/Core/SettingsEnums.h @@ -584,6 +584,7 @@ enum class JemallocProfileFormat : uint8_t DECLARE_SETTING_ENUM(JemallocProfileFormat) +<<<<<<< HEAD enum class S3UriStyle : uint8_t { AUTO, @@ -599,5 +600,16 @@ enum class FileLikeEngineDefaultPartitionStrategy : uint8_t HIVE, }; DECLARE_SETTING_ENUM(FileLikeEngineDefaultPartitionStrategy) +======= +enum class MergeTreePartExportFileAlreadyExistsPolicy : uint8_t +{ + skip, + error, + overwrite, +}; + +DECLARE_SETTING_ENUM(MergeTreePartExportFileAlreadyExistsPolicy) + +>>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) } diff --git a/src/Databases/DatabaseReplicated.cpp b/src/Databases/DatabaseReplicated.cpp index 0706f279e6eb..a292a12e33cf 100644 --- a/src/Databases/DatabaseReplicated.cpp +++ b/src/Databases/DatabaseReplicated.cpp @@ -2574,7 +2574,7 @@ bool DatabaseReplicated::shouldReplicateQuery(const ContextPtr & query_context, if (const auto * alter = query_ptr->as()) { if (alter->isAttachAlter() || alter->isFetchAlter() || alter->isDropPartitionAlter() || alter->isFreezeAlter() - || alter->isUnlockSnapshot()) + || alter->isUnlockSnapshot() || alter->isExportPartOrExportPartitionAlter()) return false; // Allowed ALTER operation on KeeperMap still should be replicated diff --git a/src/Disks/DiskObjectStorage/ObjectStorages/AzureBlobStorage/AzureObjectStorage.h b/src/Disks/DiskObjectStorage/ObjectStorages/AzureBlobStorage/AzureObjectStorage.h index 88adfe903284..88420b30472f 100644 --- a/src/Disks/DiskObjectStorage/ObjectStorages/AzureBlobStorage/AzureObjectStorage.h +++ b/src/Disks/DiskObjectStorage/ObjectStorages/AzureBlobStorage/AzureObjectStorage.h @@ -36,6 +36,8 @@ class AzureObjectStorage : public IObjectStorage const String & description_, const String & common_key_prefix_); + bool supportsListObjectsCache() override { return true; } + void listObjects(const std::string & path, RelativePathsWithMetadata & children, size_t max_keys) const override; /// Sanitizer build may crash with max_keys=1; this looks like a false positive. diff --git a/src/Disks/DiskObjectStorage/ObjectStorages/IObjectStorage.h b/src/Disks/DiskObjectStorage/ObjectStorages/IObjectStorage.h index 41b16b95e50b..97fd4fe3c50a 100644 --- a/src/Disks/DiskObjectStorage/ObjectStorages/IObjectStorage.h +++ b/src/Disks/DiskObjectStorage/ObjectStorages/IObjectStorage.h @@ -370,9 +370,13 @@ class IObjectStorage } #endif +<<<<<<< HEAD /// Returns the inner (unwrapped) object storage for decorator types such as `CachedObjectStorage`. /// Returns nullptr for non-decorator types, meaning this storage is already the base. virtual ObjectStoragePtr getUnderlying() { return nullptr; } +======= + virtual bool supportsListObjectsCache() { return false; } +>>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) }; using ObjectStoragePtr = std::shared_ptr; diff --git a/src/Disks/DiskObjectStorage/ObjectStorages/S3/S3ObjectStorage.h b/src/Disks/DiskObjectStorage/ObjectStorages/S3/S3ObjectStorage.h index 1760153a1219..141a81a3985f 100644 --- a/src/Disks/DiskObjectStorage/ObjectStorages/S3/S3ObjectStorage.h +++ b/src/Disks/DiskObjectStorage/ObjectStorages/S3/S3ObjectStorage.h @@ -70,6 +70,8 @@ class S3ObjectStorage : public IObjectStorage ObjectStorageType getType() const override { return ObjectStorageType::S3; } + bool supportsListObjectsCache() override { return true; } + bool exists(const StoredObject & object) const override; std::unique_ptr readObject( /// NOLINT diff --git a/src/Functions/generateSnowflakeID.cpp b/src/Functions/generateSnowflakeID.cpp index 136d3398fd3c..d1c2446493a9 100644 --- a/src/Functions/generateSnowflakeID.cpp +++ b/src/Functions/generateSnowflakeID.cpp @@ -155,7 +155,16 @@ uint64_t generateSnowflakeID() return fromSnowflakeId(snowflake_id); } +<<<<<<< HEAD class FunctionGenerateSnowflakeID final : public IFunction +======= +std::string generateSnowflakeIDString() +{ + return std::to_string(generateSnowflakeID()); +} + +class FunctionGenerateSnowflakeID : public IFunction +>>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) { public: static constexpr auto name = "generateSnowflakeID"; diff --git a/src/Functions/generateSnowflakeID.h b/src/Functions/generateSnowflakeID.h index 38fa684a9b4b..4fc173dcf1be 100644 --- a/src/Functions/generateSnowflakeID.h +++ b/src/Functions/generateSnowflakeID.h @@ -7,4 +7,6 @@ namespace DB uint64_t generateSnowflakeID(); +std::string generateSnowflakeIDString(); + } diff --git a/src/Interpreters/Context.cpp b/src/Interpreters/Context.cpp index 1edce5d06b45..de3e38e81083 100644 --- a/src/Interpreters/Context.cpp +++ b/src/Interpreters/Context.cpp @@ -47,6 +47,7 @@ #include #include #include +#include #include #include #include @@ -176,6 +177,8 @@ namespace ProfileEvents extern const Event BackupThrottlerSleepMicroseconds; extern const Event MergesThrottlerBytes; extern const Event MergesThrottlerSleepMicroseconds; + extern const Event ExportsThrottlerBytes; + extern const Event ExportsThrottlerSleepMicroseconds; extern const Event MutationsThrottlerBytes; extern const Event MutationsThrottlerSleepMicroseconds; extern const Event QueryLocalReadThrottlerBytes; @@ -387,6 +390,7 @@ namespace ServerSetting extern const ServerSettingsUInt64 max_local_write_bandwidth_for_server; extern const ServerSettingsUInt64 max_merges_bandwidth_for_server; extern const ServerSettingsUInt64 max_mutations_bandwidth_for_server; + extern const ServerSettingsUInt64 max_exports_bandwidth_for_server; extern const ServerSettingsUInt64 max_remote_read_network_bandwidth_for_server; extern const ServerSettingsUInt64 max_remote_write_network_bandwidth_for_server; extern const ServerSettingsUInt64 max_replicated_fetches_network_bandwidth_for_server; @@ -593,6 +597,7 @@ struct ContextSharedPart : boost::noncopyable GlobalOvercommitTracker global_overcommit_tracker; MergeList merge_list; /// The list of executable merge (for (Replicated)?MergeTree) MovesList moves_list; /// The list of executing moves (for (Replicated)?MergeTree) + ExportsList exports_list; /// The list of executing exports (for (Replicated)?MergeTree) ReplicatedFetchList replicated_fetch_list; RefreshSet refresh_set; /// The list of active refreshes (for MaterializedView) ConfigurationPtr users_config TSA_GUARDED_BY(mutex); /// Config with the users, profiles and quotas sections. @@ -639,6 +644,8 @@ struct ContextSharedPart : boost::noncopyable mutable ThrottlerPtr distributed_cache_read_throttler; /// A server-wide throttler for distributed cache read mutable ThrottlerPtr distributed_cache_write_throttler; /// A server-wide throttler for distributed cache write + mutable ThrottlerPtr exports_throttler; /// A server-wide throttler for exports + MultiVersion macros; /// Substitutions extracted from config. std::unique_ptr ddl_worker TSA_GUARDED_BY(mutex); /// Process ddl commands from zk. LoadTaskPtr ddl_worker_startup_task; /// To postpone `ddl_worker->startup()` after all tables startup @@ -1228,6 +1235,9 @@ struct ContextSharedPart : boost::noncopyable if (auto bandwidth = server_settings[ServerSetting::max_merges_bandwidth_for_server]) merges_throttler = std::make_shared(bandwidth, ProfileEvents::MergesThrottlerBytes, ProfileEvents::MergesThrottlerSleepMicroseconds); + + if (auto bandwidth = server_settings[ServerSetting::max_exports_bandwidth_for_server]) + exports_throttler = std::make_shared(bandwidth, ProfileEvents::ExportsThrottlerBytes, ProfileEvents::ExportsThrottlerSleepMicroseconds); } }; @@ -1400,6 +1410,8 @@ MergeList & Context::getMergeList() { return shared->merge_list; } const MergeList & Context::getMergeList() const { return shared->merge_list; } MovesList & Context::getMovesList() { return shared->moves_list; } const MovesList & Context::getMovesList() const { return shared->moves_list; } +ExportsList & Context::getExportsList() { return shared->exports_list; } +const ExportsList & Context::getExportsList() const { return shared->exports_list; } ReplicatedFetchList & Context::getReplicatedFetchList() { return shared->replicated_fetch_list; } const ReplicatedFetchList & Context::getReplicatedFetchList() const { return shared->replicated_fetch_list; } RefreshSet & Context::getRefreshSet() { return shared->refresh_set; } @@ -3599,6 +3611,13 @@ void Context::makeQueryContextForMutate(const MergeTreeSettings & merge_tree_set = merge_tree_settings[MergeTreeSetting::mutation_workload].value.empty() ? getMutationWorkload() : merge_tree_settings[MergeTreeSetting::mutation_workload]; } +void Context::makeQueryContextForExportPart() +{ + makeQueryContext(); + classifier.reset(); // It is assumed that there are no active queries running using this classifier, otherwise this will lead to crashes + // Export part operations don't have a specific workload setting, so we leave the default workload +} + void Context::makeSessionContext() { session_context = shared_from_this(); @@ -5187,6 +5206,11 @@ ThrottlerPtr Context::getDistributedCacheWriteThrottler() const return shared->distributed_cache_write_throttler; } +ThrottlerPtr Context::getExportsThrottler() const +{ + return shared->exports_throttler; +} + void Context::reloadRemoteThrottlerConfig(size_t read_bandwidth, size_t write_bandwidth) const { if (read_bandwidth) diff --git a/src/Interpreters/Context.h b/src/Interpreters/Context.h index e939d24e0e86..3c21cb4e262d 100644 --- a/src/Interpreters/Context.h +++ b/src/Interpreters/Context.h @@ -96,6 +96,7 @@ class AsynchronousMetrics; class BackgroundSchedulePool; class MergeList; class MovesList; +class ExportsList; class ReplicatedFetchList; class RefreshSet; class Cluster; @@ -1313,6 +1314,7 @@ class Context: public ContextData, public std::enable_shared_from_this void makeQueryContext(); void makeQueryContextForMerge(const MergeTreeSettings & merge_tree_settings); void makeQueryContextForMutate(const MergeTreeSettings & merge_tree_settings); + void makeQueryContextForExportPart(); void makeSessionContext(); void makeGlobalContext(); void makeBackgroundContext(const Poco::Util::AbstractConfiguration & config); @@ -1349,6 +1351,9 @@ class Context: public ContextData, public std::enable_shared_from_this MovesList & getMovesList(); const MovesList & getMovesList() const; + ExportsList & getExportsList(); + const ExportsList & getExportsList() const; + ReplicatedFetchList & getReplicatedFetchList(); const ReplicatedFetchList & getReplicatedFetchList() const; @@ -1942,6 +1947,7 @@ class Context: public ContextData, public std::enable_shared_from_this ThrottlerPtr getMutationsThrottler() const; ThrottlerPtr getMergesThrottler() const; + ThrottlerPtr getExportsThrottler() const; ThrottlerPtr getDistributedCacheReadThrottler() const; ThrottlerPtr getDistributedCacheWriteThrottler() const; diff --git a/src/Interpreters/DDLWorker.cpp b/src/Interpreters/DDLWorker.cpp index 4c3729c99218..c120873fbd0e 100644 --- a/src/Interpreters/DDLWorker.cpp +++ b/src/Interpreters/DDLWorker.cpp @@ -840,7 +840,11 @@ bool DDLWorker::taskShouldBeExecutedOnLeader(const ASTPtr & ast_ddl, const Stora alter->isUnlockSnapshot() || alter->isMovePartitionToDiskOrVolumeAlter() || alter->isCommentAlter() || +<<<<<<< HEAD alter->isSettingsOrCommentAlter()) +======= + alter->isExportPartOrExportPartitionAlter()) +>>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) return false; } diff --git a/src/Interpreters/InterpreterAlterQuery.cpp b/src/Interpreters/InterpreterAlterQuery.cpp index 7a6975d726eb..669130febe8d 100644 --- a/src/Interpreters/InterpreterAlterQuery.cpp +++ b/src/Interpreters/InterpreterAlterQuery.cpp @@ -756,6 +756,20 @@ AccessRightsElements InterpreterAlterQuery::getRequiredAccessForCommand(const AS required_access.emplace_back(AccessType::ALTER_DELETE | AccessType::INSERT, database, table); break; } + case ASTAlterCommand::EXPORT_PART: + { + required_access.emplace_back(AccessType::ALTER_EXPORT_PART, database, table); + /// For table functions, access control is handled by the table function itself + if (!command.to_table_function) + required_access.emplace_back(AccessType::INSERT, command.to_database, command.to_table); + break; + } + case ASTAlterCommand::EXPORT_PARTITION: + { + required_access.emplace_back(AccessType::ALTER_EXPORT_PARTITION, database, table); + required_access.emplace_back(AccessType::INSERT, command.to_database, command.to_table); + break; + } case ASTAlterCommand::FETCH_PARTITION: { required_access.emplace_back(AccessType::ALTER_FETCH_PARTITION, database, table); diff --git a/src/Interpreters/InterpreterKillQueryQuery.cpp b/src/Interpreters/InterpreterKillQueryQuery.cpp index 8c09beb6bb26..76c96586705c 100644 --- a/src/Interpreters/InterpreterKillQueryQuery.cpp +++ b/src/Interpreters/InterpreterKillQueryQuery.cpp @@ -19,6 +19,7 @@ #include #include #include +#include #include #include #include @@ -37,11 +38,17 @@ namespace Setting extern const SettingsUInt64 max_parser_depth; } +namespace ServerSetting +{ + extern const ServerSettingsBool allow_experimental_export_merge_tree_partition; +} + namespace ErrorCodes { extern const int LOGICAL_ERROR; extern const int ACCESS_DENIED; extern const int NOT_IMPLEMENTED; + extern const int SUPPORT_IS_DISABLED; } @@ -250,6 +257,82 @@ BlockIO InterpreterKillQueryQuery::execute() break; } + case ASTKillQueryQuery::Type::ExportPartition: + { + if (!getContext()->getServerSettings()[ServerSetting::allow_experimental_export_merge_tree_partition]) + { + throw Exception(ErrorCodes::SUPPORT_IS_DISABLED, + "Exporting merge tree partition is experimental. Set the server setting `allow_experimental_export_merge_tree_partition` to enable it"); + } + + Block exports_block = getSelectResult( + "source_database, source_table, transaction_id, destination_database, destination_table, partition_id", + "system.replicated_partition_exports"); + if (exports_block.empty()) + return res_io; + + const ColumnString & src_db_col = typeid_cast(*exports_block.getByName("source_database").column); + const ColumnString & src_table_col = typeid_cast(*exports_block.getByName("source_table").column); + const ColumnString & dst_db_col = typeid_cast(*exports_block.getByName("destination_database").column); + const ColumnString & dst_table_col = typeid_cast(*exports_block.getByName("destination_table").column); + const ColumnString & tx_col = typeid_cast(*exports_block.getByName("transaction_id").column); + + auto header = exports_block.cloneEmpty(); + header.insert(0, {ColumnString::create(), std::make_shared(), "kill_status"}); + + MutableColumns res_columns = header.cloneEmptyColumns(); + AccessRightsElements required_access_rights; + auto access = getContext()->getAccess(); + bool access_denied = false; + + for (size_t i = 0; i < exports_block.rows(); ++i) + { + const auto src_database = src_db_col.getDataAt(i); + const auto src_table = src_table_col.getDataAt(i); + const auto dst_database = dst_db_col.getDataAt(i); + const auto dst_table = dst_table_col.getDataAt(i); + + const auto table_id = StorageID{std::string{src_database}, std::string{src_table}}; + const auto transaction_id = tx_col.getDataAt(i); + + CancellationCode code = CancellationCode::Unknown; + if (!query.test) + { + auto storage = DatabaseCatalog::instance().tryGetTable(table_id, getContext()); + if (!storage) + code = CancellationCode::NotFound; + else + { + ASTAlterCommand alter_command{}; + alter_command.type = ASTAlterCommand::EXPORT_PARTITION; + alter_command.move_destination_type = DataDestinationType::TABLE; + alter_command.from_database = src_database; + alter_command.from_table = src_table; + alter_command.to_database = dst_database; + alter_command.to_table = dst_table; + + required_access_rights = InterpreterAlterQuery::getRequiredAccessForCommand( + alter_command, table_id.database_name, table_id.table_name); + if (!access->isGranted(required_access_rights)) + { + access_denied = true; + continue; + } + code = storage->killExportPartition(std::string{transaction_id}); + } + } + + insertResultRow(i, code, exports_block, header, res_columns); + } + + if (res_columns[0]->empty() && access_denied) + throw Exception(ErrorCodes::ACCESS_DENIED, "Not allowed to kill export partition. " + "To execute this query, it's necessary to have the grant {}", required_access_rights.toString()); + + res_io.pipeline = QueryPipeline(Pipe(std::make_shared(std::make_shared(header.cloneWithColumns(std::move(res_columns)))))); + + break; + } case ASTKillQueryQuery::Type::Mutation: { Block mutations_block = getSelectResult("database, table, mutation_id, command", "system.mutations"); @@ -474,6 +557,9 @@ AccessRightsElements InterpreterKillQueryQuery::getRequiredAccessForDDLOnCluster | AccessType::ALTER_MATERIALIZE_TTL | AccessType::ALTER_REWRITE_PARTS ); + /// todo arthur think about this + else if (query.type == ASTKillQueryQuery::Type::ExportPartition) + required_access.emplace_back(AccessType::ALTER_EXPORT_PARTITION); return required_access; } diff --git a/src/Interpreters/InterpreterSystemQuery.cpp b/src/Interpreters/InterpreterSystemQuery.cpp index 8220a1a82db6..9fe898ae0c7f 100644 --- a/src/Interpreters/InterpreterSystemQuery.cpp +++ b/src/Interpreters/InterpreterSystemQuery.cpp @@ -55,6 +55,7 @@ #include #include #include +#include #include #include #include @@ -568,7 +569,12 @@ BlockIO InterpreterSystemQuery::execute() #else throw Exception(ErrorCodes::SUPPORT_IS_DISABLED, "The server was compiled without the support for AWS S3"); #endif - + case Type::DROP_OBJECT_STORAGE_LIST_OBJECTS_CACHE: + { + getContext()->checkAccess(AccessType::SYSTEM_DROP_OBJECT_STORAGE_LIST_OBJECTS_CACHE); + ObjectStorageListObjectsCache::instance().clear(); + break; + } case Type::CLEAR_FILESYSTEM_CACHE: { getContext()->checkAccess(AccessType::SYSTEM_DROP_FILESYSTEM_CACHE); @@ -2495,6 +2501,7 @@ AccessRightsElements InterpreterSystemQuery::getRequiredAccessForDDLOnCluster() case Type::CLEAR_SCHEMA_CACHE: case Type::CLEAR_FORMAT_SCHEMA_CACHE: case Type::CLEAR_S3_CLIENT_CACHE: + case Type::DROP_OBJECT_STORAGE_LIST_OBJECTS_CACHE: { required_access.emplace_back(AccessType::SYSTEM_DROP_CACHE); break; diff --git a/src/Interpreters/PartLog.cpp b/src/Interpreters/PartLog.cpp index 2fe5660d9dfc..2084e0aa2262 100644 --- a/src/Interpreters/PartLog.cpp +++ b/src/Interpreters/PartLog.cpp @@ -72,6 +72,7 @@ ColumnsDescription PartLogElement::getColumnsDescription() {"MovePart", static_cast(MOVE_PART)}, {"MergePartsStart", static_cast(MERGE_PARTS_START)}, {"MutatePartStart", static_cast(MUTATE_PART_START)}, + {"ExportPart", static_cast(EXPORT_PART)}, } ); @@ -113,7 +114,8 @@ ColumnsDescription PartLogElement::getColumnsDescription() "RemovePart — Removing or detaching a data part using [DETACH PARTITION](/sql-reference/statements/alter/partition#detach-partitionpart)." "MutatePartStart — Mutating of a data part has started, " "MutatePart — Mutating of a data part has finished, " - "MovePart — Moving the data part from the one disk to another one."}, + "MovePart — Moving the data part from the one disk to another one." + "ExportPart — Exporting the data part from a MergeTree table into a target table that represents external storage (e.g., object storage or a data lake).."}, {"merge_reason", std::move(merge_reason_datatype), "The reason for the event with type MERGE_PARTS. Can have one of the following values: " "NotAMerge — The current event has the type other than MERGE_PARTS, " @@ -137,6 +139,7 @@ ColumnsDescription PartLogElement::getColumnsDescription() {"part_storage_type", std::make_shared(), "The type of DataPartStorage. Possible values: Packed - all files are stored in a single blob, Full - a blob per file."}, {"disk_name", std::make_shared(), "The disk name data part lies on."}, {"path_on_disk", std::make_shared(), "Absolute path to the folder with data part files."}, + {"remote_file_paths", std::make_shared(std::make_shared()), "In case of an export operation to remote storages, the file paths a given export generated"}, {"rows", std::make_shared(), "The number of rows in the data part."}, {"size_in_bytes", std::make_shared(), "Size of the data part on disk in bytes."}, @@ -197,6 +200,12 @@ void PartLogElement::appendToBlock(MutableColumns & columns) const columns[i++]->insert(disk_name); columns[i++]->insert(path_on_disk); + Array remote_file_paths_array; + remote_file_paths_array.reserve(remote_file_paths.size()); + for (const auto & remote_file_path : remote_file_paths) + remote_file_paths_array.push_back(remote_file_path); + columns[i++]->insert(remote_file_paths_array); + columns[i++]->insert(rows); columns[i++]->insert(bytes_compressed_on_disk); diff --git a/src/Interpreters/PartLog.h b/src/Interpreters/PartLog.h index 8625173f00ec..7e3c662b26c8 100644 --- a/src/Interpreters/PartLog.h +++ b/src/Interpreters/PartLog.h @@ -30,6 +30,7 @@ struct PartLogElement MOVE_PART = 6, MERGE_PARTS_START = 7, MUTATE_PART_START = 8, + EXPORT_PART = 9, }; /// Copy of MergeAlgorithm since values are written to disk. @@ -73,6 +74,7 @@ struct PartLogElement String disk_name; String path_on_disk; Strings deduplication_block_ids; + std::vector remote_file_paths; MergeTreeDataPartFormat part_format; diff --git a/src/Interpreters/executeDDLQueryOnCluster.cpp b/src/Interpreters/executeDDLQueryOnCluster.cpp index 6c141c102798..a1f76f89cfdc 100644 --- a/src/Interpreters/executeDDLQueryOnCluster.cpp +++ b/src/Interpreters/executeDDLQueryOnCluster.cpp @@ -57,6 +57,8 @@ bool isSupportedAlterTypeForOnClusterDDLQuery(int type) ASTAlterCommand::ATTACH_PARTITION, /// Usually followed by ATTACH PARTITION ASTAlterCommand::FETCH_PARTITION, + /// Data operation that should be executed locally on each replica + ASTAlterCommand::EXPORT_PART, /// Logical error ASTAlterCommand::NO_TYPE, }; diff --git a/src/Parsers/ASTAlterQuery.cpp b/src/Parsers/ASTAlterQuery.cpp index a94993b77ac1..dfa92c029ae2 100644 --- a/src/Parsers/ASTAlterQuery.cpp +++ b/src/Parsers/ASTAlterQuery.cpp @@ -68,10 +68,17 @@ ASTPtr ASTAlterCommand::clone() const res->rename_to = res->children.emplace_back(rename_to->clone()).get(); if (execute_args) res->execute_args = res->children.emplace_back(execute_args->clone()).get(); +<<<<<<< HEAD if (add_enum_values) res->add_enum_values = res->children.emplace_back(add_enum_values->clone()); if (refresh) res->refresh = res->children.emplace_back(refresh->clone()).get(); +======= + if (to_table_function) + res->to_table_function = res->children.emplace_back(to_table_function->clone()).get(); + if (partition_by_expr) + res->partition_by_expr = res->children.emplace_back(partition_by_expr->clone()).get(); +>>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) return res; } @@ -375,6 +382,49 @@ void ASTAlterCommand::formatImpl(WriteBuffer & ostr, const FormatSettings & sett ostr << quoteString(move_destination_name); } } + else if (type == ASTAlterCommand::EXPORT_PART) + { + ostr << "EXPORT PART "; + partition->format(ostr, settings, state, frame); + ostr << " TO "; + switch (move_destination_type) + { + case DataDestinationType::TABLE: + ostr << "TABLE "; + if (to_table_function) + { + ostr << "FUNCTION "; + to_table_function->format(ostr, settings, state, frame); + if (partition_by_expr) + { + ostr << " PARTITION BY "; + partition_by_expr->format(ostr, settings, state, frame); + } + } + else + { + if (!to_database.empty()) + ostr << backQuoteIfNeed(to_database) << "."; + + ostr << backQuoteIfNeed(to_table); + } + return; + default: + break; + } + + } + else if (type == ASTAlterCommand::EXPORT_PARTITION) + { + ostr << "EXPORT PARTITION "; + partition->format(ostr, settings, state, frame); + ostr << " TO TABLE "; + if (!to_database.empty()) + { + ostr << backQuoteIfNeed(to_database) << "."; + } + ostr << backQuoteIfNeed(to_table); + } else if (type == ASTAlterCommand::REPLACE_PARTITION) { ostr << (replace ? "REPLACE" : "ATTACH") << " PARTITION " @@ -605,7 +655,12 @@ void ASTAlterCommand::forEachPointerToChild(std::function>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) } @@ -736,6 +791,11 @@ bool ASTAlterQuery::isMovePartitionToDiskOrVolumeAlter() const return false; } +bool ASTAlterQuery::isExportPartOrExportPartitionAlter() const +{ + return isOneCommandTypeOnly(ASTAlterCommand::EXPORT_PART) || isOneCommandTypeOnly(ASTAlterCommand::EXPORT_PARTITION); +} + /** Get the text that identifies this element. */ String ASTAlterQuery::getID(char delim) const diff --git a/src/Parsers/ASTAlterQuery.h b/src/Parsers/ASTAlterQuery.h index fa05033ad9e4..f813a1d01209 100644 --- a/src/Parsers/ASTAlterQuery.h +++ b/src/Parsers/ASTAlterQuery.h @@ -72,6 +72,8 @@ class ASTAlterCommand : public IAST FREEZE_ALL, UNFREEZE_PARTITION, UNFREEZE_ALL, + EXPORT_PART, + EXPORT_PARTITION, DELETE, UPDATE, @@ -224,6 +226,9 @@ class ASTAlterCommand : public IAST /// MOVE PARTITION partition TO TABLE db.table String to_database; String to_table; + /// EXPORT PART/PARTITION to TABLE FUNCTION (e.g., s3()) + IAST * to_table_function = nullptr; + IAST * partition_by_expr = nullptr; String snapshot_name; IAST * snapshot_desc{}; @@ -276,6 +281,8 @@ class ASTAlterQuery : public ASTQueryWithTableAndOutput, public ASTQueryWithOnCl bool isMovePartitionToDiskOrVolumeAlter() const; + bool isExportPartOrExportPartitionAlter() const; + bool isCommentAlter() const; /// Every command modifies settings or comments: any mix of MODIFY SETTING / diff --git a/src/Parsers/ASTKillQueryQuery.cpp b/src/Parsers/ASTKillQueryQuery.cpp index 0334b78d559e..9911e60b5ed9 100644 --- a/src/Parsers/ASTKillQueryQuery.cpp +++ b/src/Parsers/ASTKillQueryQuery.cpp @@ -27,6 +27,9 @@ void ASTKillQueryQuery::formatQueryImpl(WriteBuffer & ostr, const FormatSettings case Type::Transaction: ostr << "TRANSACTION"; break; + case Type::ExportPartition: + ostr << "EXPORT PARTITION"; + break; } formatOnCluster(ostr, settings); diff --git a/src/Parsers/ASTKillQueryQuery.h b/src/Parsers/ASTKillQueryQuery.h index 73fa72adb332..101c83c566f7 100644 --- a/src/Parsers/ASTKillQueryQuery.h +++ b/src/Parsers/ASTKillQueryQuery.h @@ -13,6 +13,7 @@ class ASTKillQueryQuery : public ASTQueryWithOutput, public ASTQueryWithOnCluste { Query, /// KILL QUERY Mutation, /// KILL MUTATION + ExportPartition, /// KILL EXPORT_PARTITION PartMoveToShard, /// KILL PART_MOVE_TO_SHARD Transaction, /// KILL TRANSACTION }; diff --git a/src/Parsers/ASTSystemQuery.cpp b/src/Parsers/ASTSystemQuery.cpp index c4ee97348899..7834ce541ec2 100644 --- a/src/Parsers/ASTSystemQuery.cpp +++ b/src/Parsers/ASTSystemQuery.cpp @@ -603,7 +603,11 @@ void ASTSystemQuery::formatImpl(WriteBuffer & ostr, const FormatSettings & setti case Type::CLEAR_S3_CLIENT_CACHE: case Type::CLEAR_ICEBERG_METADATA_CACHE: case Type::CLEAR_PARQUET_METADATA_CACHE: +<<<<<<< HEAD case Type::CLEAR_AVRO_SCHEMA_CACHE: +======= + case Type::DROP_OBJECT_STORAGE_LIST_OBJECTS_CACHE: +>>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) case Type::RESET_COVERAGE: case Type::RESTART_REPLICAS: case Type::JEMALLOC_PURGE: diff --git a/src/Parsers/ASTSystemQuery.h b/src/Parsers/ASTSystemQuery.h index adf76d898d42..4705af5c089e 100644 --- a/src/Parsers/ASTSystemQuery.h +++ b/src/Parsers/ASTSystemQuery.h @@ -52,6 +52,7 @@ class ASTSystemQuery : public IAST, public ASTQueryWithOnCluster CLEAR_FORMAT_SCHEMA_CACHE, CLEAR_AVRO_SCHEMA_CACHE, CLEAR_S3_CLIENT_CACHE, + DROP_OBJECT_STORAGE_LIST_OBJECTS_CACHE, STOP_LISTEN, START_LISTEN, RESTART_REPLICAS, diff --git a/src/Parsers/CommonParsers.h b/src/Parsers/CommonParsers.h index 707a85f581ce..62e21d63de61 100644 --- a/src/Parsers/CommonParsers.h +++ b/src/Parsers/CommonParsers.h @@ -368,6 +368,8 @@ namespace DB MR_MACROS(MONTHS, "MONTHS") \ MR_MACROS(MOVE_PART, "MOVE PART") \ MR_MACROS(MOVE_PARTITION, "MOVE PARTITION") \ + MR_MACROS(EXPORT_PART, "EXPORT PART") \ + MR_MACROS(EXPORT_PARTITION, "EXPORT PARTITION") \ MR_MACROS(MOVE, "MOVE") \ MR_MACROS(MS, "MS") \ MR_MACROS(MUTATION, "MUTATION") \ diff --git a/src/Parsers/ParserAlterQuery.cpp b/src/Parsers/ParserAlterQuery.cpp index e2bd193b1a2d..fa0e2f93361e 100644 --- a/src/Parsers/ParserAlterQuery.cpp +++ b/src/Parsers/ParserAlterQuery.cpp @@ -85,6 +85,8 @@ bool ParserAlterCommand::parseImpl(Pos & pos, ASTPtr & node, Expected & expected ParserKeyword s_forget_partition(Keyword::FORGET_PARTITION); ParserKeyword s_move_partition(Keyword::MOVE_PARTITION); ParserKeyword s_move_part(Keyword::MOVE_PART); + ParserKeyword s_export_part(Keyword::EXPORT_PART); + ParserKeyword s_export_partition(Keyword::EXPORT_PARTITION); ParserKeyword s_drop_detached_partition(Keyword::DROP_DETACHED_PARTITION); ParserKeyword s_drop_detached_part(Keyword::DROP_DETACHED_PART); ParserKeyword s_fetch_partition(Keyword::FETCH_PARTITION); @@ -94,6 +96,7 @@ bool ParserAlterCommand::parseImpl(Pos & pos, ASTPtr & node, Expected & expected ParserKeyword s_unfreeze(Keyword::UNFREEZE); ParserKeyword s_unlock_snapshot(Keyword::UNLOCK_SNAPSHOT); ParserKeyword s_partition(Keyword::PARTITION); + ParserKeyword s_partition_by(Keyword::PARTITION_BY); ParserKeyword s_first(Keyword::FIRST); ParserKeyword s_after(Keyword::AFTER); @@ -108,6 +111,7 @@ bool ParserAlterCommand::parseImpl(Pos & pos, ASTPtr & node, Expected & expected ParserKeyword s_to_volume(Keyword::TO_VOLUME); ParserKeyword s_to_table(Keyword::TO_TABLE); ParserKeyword s_to_shard(Keyword::TO_SHARD); + ParserKeyword s_function(Keyword::FUNCTION); ParserKeyword s_delete(Keyword::DELETE); ParserKeyword s_update(Keyword::UPDATE); @@ -182,7 +186,12 @@ bool ParserAlterCommand::parseImpl(Pos & pos, ASTPtr & node, Expected & expected ASTPtr command_rename_to; ASTPtr command_sql_security; ASTPtr command_snapshot_desc; +<<<<<<< HEAD ASTPtr command_refresh; +======= + ASTPtr export_table_function; + ASTPtr export_table_function_partition_by_expr; +>>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) if (with_round_bracket) { @@ -547,6 +556,57 @@ bool ParserAlterCommand::parseImpl(Pos & pos, ASTPtr & node, Expected & expected command->move_destination_name = ast_space_name->as().value.safeGet(); } + else if (s_export_part.ignore(pos, expected)) + { + if (!parser_string_and_substituion.parse(pos, command_partition, expected)) + return false; + + command->type = ASTAlterCommand::EXPORT_PART; + command->part = true; + + if (!s_to_table.ignore(pos, expected)) + { + return false; + } + + if (s_function.ignore(pos, expected)) + { + ParserFunction table_function_parser(/*allow_function_parameters=*/true, /*is_table_function=*/true); + + if (!table_function_parser.parse(pos, export_table_function, expected)) + return false; + + if (s_partition_by.ignore(pos, expected)) + if (!parser_exp_elem.parse(pos, export_table_function_partition_by_expr, expected)) + return false; + + command->to_table_function = export_table_function.get(); + command->partition_by_expr = export_table_function_partition_by_expr.get(); + command->move_destination_type = DataDestinationType::TABLE; + } + else + { + if (!parseDatabaseAndTableName(pos, expected, command->to_database, command->to_table)) + return false; + command->move_destination_type = DataDestinationType::TABLE; + } + } + else if (s_export_partition.ignore(pos, expected)) + { + if (!parser_partition.parse(pos, command_partition, expected)) + return false; + + command->type = ASTAlterCommand::EXPORT_PARTITION; + + if (!s_to_table.ignore(pos, expected)) + { + return false; + } + + if (!parseDatabaseAndTableName(pos, expected, command->to_database, command->to_table)) + return false; + command->move_destination_type = DataDestinationType::TABLE; + } else if (s_move_partition.ignore(pos, expected)) { if (!parser_partition.parse(pos, command_partition, expected)) @@ -1145,8 +1205,15 @@ bool ParserAlterCommand::parseImpl(Pos & pos, ASTPtr & node, Expected & expected command->rename_to = command->children.emplace_back(std::move(command_rename_to)).get(); if (command_snapshot_desc) command->snapshot_desc = command->children.emplace_back(std::move(command_snapshot_desc)).get(); +<<<<<<< HEAD if (command_refresh) command->refresh = command->children.emplace_back(std::move(command_refresh)).get(); +======= + if (export_table_function) + command->to_table_function = command->children.emplace_back(std::move(export_table_function)).get(); + if (export_table_function_partition_by_expr) + command->partition_by_expr = command->children.emplace_back(std::move(export_table_function_partition_by_expr)).get(); +>>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) return true; } diff --git a/src/Parsers/ParserKillQueryQuery.cpp b/src/Parsers/ParserKillQueryQuery.cpp index 97e58566af67..99f2d6fd2d64 100644 --- a/src/Parsers/ParserKillQueryQuery.cpp +++ b/src/Parsers/ParserKillQueryQuery.cpp @@ -17,6 +17,7 @@ bool ParserKillQueryQuery::parseImpl(Pos & pos, ASTPtr & node, Expected & expect ParserKeyword p_kill{Keyword::KILL}; ParserKeyword p_query{Keyword::QUERY}; ParserKeyword p_mutation{Keyword::MUTATION}; + ParserKeyword p_export_partition{Keyword::EXPORT_PARTITION}; ParserKeyword p_part_move_to_shard{Keyword::PART_MOVE_TO_SHARD}; ParserKeyword p_transaction{Keyword::TRANSACTION}; ParserKeyword p_on{Keyword::ON}; @@ -33,6 +34,8 @@ bool ParserKillQueryQuery::parseImpl(Pos & pos, ASTPtr & node, Expected & expect query->type = ASTKillQueryQuery::Type::Query; else if (p_mutation.ignore(pos, expected)) query->type = ASTKillQueryQuery::Type::Mutation; + else if (p_export_partition.ignore(pos, expected)) + query->type = ASTKillQueryQuery::Type::ExportPartition; else if (p_part_move_to_shard.ignore(pos, expected)) query->type = ASTKillQueryQuery::Type::PartMoveToShard; else if (p_transaction.ignore(pos, expected)) diff --git a/src/Storages/Cache/ObjectStorageListObjectsCache.cpp b/src/Storages/Cache/ObjectStorageListObjectsCache.cpp new file mode 100644 index 000000000000..78b73381c843 --- /dev/null +++ b/src/Storages/Cache/ObjectStorageListObjectsCache.cpp @@ -0,0 +1,226 @@ +#include +#include +#include +#include + +namespace ProfileEvents +{ +extern const Event ObjectStorageListObjectsCacheHits; +extern const Event ObjectStorageListObjectsCacheMisses; +extern const Event ObjectStorageListObjectsCacheExactMatchHits; +extern const Event ObjectStorageListObjectsCachePrefixMatchHits; +} + +namespace DB +{ + +template +class ObjectStorageListObjectsCachePolicy : public TTLCachePolicy +{ +public: + using BasePolicy = TTLCachePolicy; + using typename BasePolicy::MappedPtr; + using typename BasePolicy::KeyMapped; + using BasePolicy::cache; + + ObjectStorageListObjectsCachePolicy() + : BasePolicy(CurrentMetrics::end(), CurrentMetrics::end(), std::make_unique()) + { + } + + std::optional getWithKey(const Key & key) override + { + if (const auto it = cache.find(key); it != cache.end()) + { + if (!IsStaleFunction()(it->first)) + { + return std::make_optional({it->first, it->second}); + } + // found a stale entry, remove it but don't return. We still want to perform the prefix matching search + BasePolicy::remove(it->first); + } + + if (const auto it = findBestMatchingPrefixAndRemoveExpiredEntries(key); it != cache.end()) + { + return std::make_optional({it->first, it->second}); + } + + return std::nullopt; + } + +private: + auto findBestMatchingPrefixAndRemoveExpiredEntries(Key key) + { + while (!key.prefix.empty()) + { + if (const auto it = cache.find(key); it != cache.end()) + { + if (IsStaleFunction()(it->first)) + { + BasePolicy::remove(it->first); + } + else + { + return it; + } + } + + key.prefix.pop_back(); + } + + return cache.end(); + } +}; + +ObjectStorageListObjectsCache::Key::Key( + const String & storage_description_, + const String & bucket_, + const String & prefix_, + bool with_tags_, + const std::chrono::steady_clock::time_point & expires_at_, + std::optional user_id_) + : storage_description(storage_description_), bucket(bucket_), prefix(prefix_), with_tags(with_tags_), expires_at(expires_at_), user_id(user_id_) {} + +bool ObjectStorageListObjectsCache::Key::operator==(const Key & other) const +{ + return storage_description == other.storage_description && bucket == other.bucket && prefix == other.prefix && with_tags == other.with_tags; +} + +size_t ObjectStorageListObjectsCache::KeyHasher::operator()(const Key & key) const +{ + std::size_t seed = 0; + + boost::hash_combine(seed, key.storage_description); + boost::hash_combine(seed, key.bucket); + boost::hash_combine(seed, key.prefix); + boost::hash_combine(seed, key.with_tags); + + return seed; +} + +bool ObjectStorageListObjectsCache::IsStale::operator()(const Key & key) const +{ + return key.expires_at < std::chrono::steady_clock::now(); +} + +size_t ObjectStorageListObjectsCache::WeightFunction::operator()(const Value & value) const +{ + std::size_t weight = 0; + + for (const auto & object : value) + { + const auto object_metadata = object->getObjectMetadata(); + + weight += object->getPath().capacity() + sizeof(object_metadata); + + // variable size + if (object_metadata) + { + weight += object_metadata->etag.capacity(); + weight += object_metadata->attributes.size() * (sizeof(std::string) * 2); + + for (const auto & [k, v] : object_metadata->attributes) + { + weight += k.capacity() + v.capacity(); + } + + for (const auto & [k, v] : object_metadata->tags) + { + weight += k.capacity() + v.capacity(); + } + } + } + + return weight; +} + +ObjectStorageListObjectsCache::ObjectStorageListObjectsCache() + : cache(std::make_unique>()) +{ +} + +void ObjectStorageListObjectsCache::set( + const Key & key, + const std::shared_ptr & value) +{ + auto key_with_ttl = key; + + if (ttl_in_seconds == 0) + { + key_with_ttl.expires_at = std::chrono::steady_clock::time_point::max(); + } + else + { + key_with_ttl.expires_at = std::chrono::steady_clock::now() + std::chrono::seconds(ttl_in_seconds); + } + + cache.set(key_with_ttl, value); +} + +void ObjectStorageListObjectsCache::clear() +{ + cache.clear(); +} + +std::optional ObjectStorageListObjectsCache::get(const Key & key, bool filter_by_prefix) +{ + const auto pair = cache.getWithKey(key); + + if (!pair) + { + ProfileEvents::increment(ProfileEvents::ObjectStorageListObjectsCacheMisses); + return {}; + } + + ProfileEvents::increment(ProfileEvents::ObjectStorageListObjectsCacheHits); + + if (pair->key == key) + { + ProfileEvents::increment(ProfileEvents::ObjectStorageListObjectsCacheExactMatchHits); + return *pair->mapped; + } + + ProfileEvents::increment(ProfileEvents::ObjectStorageListObjectsCachePrefixMatchHits); + + if (!filter_by_prefix) + { + return *pair->mapped; + } + + Value filtered_objects; + + filtered_objects.reserve(pair->mapped->size()); + + for (const auto & object : *pair->mapped) + { + if (object->getPath().starts_with(key.prefix)) + { + filtered_objects.push_back(object); + } + } + + return filtered_objects; +} + +void ObjectStorageListObjectsCache::setMaxSizeInBytes(std::size_t size_in_bytes_) +{ + cache.setMaxSizeInBytes(size_in_bytes_); +} + +void ObjectStorageListObjectsCache::setMaxCount(std::size_t count) +{ + cache.setMaxCount(count); +} + +void ObjectStorageListObjectsCache::setTTL(std::size_t ttl_in_seconds_) +{ + ttl_in_seconds = ttl_in_seconds_; +} + +ObjectStorageListObjectsCache & ObjectStorageListObjectsCache::instance() +{ + static ObjectStorageListObjectsCache instance; + return instance; +} + +} diff --git a/src/Storages/Cache/ObjectStorageListObjectsCache.h b/src/Storages/Cache/ObjectStorageListObjectsCache.h new file mode 100644 index 000000000000..ab1eec7efb4e --- /dev/null +++ b/src/Storages/Cache/ObjectStorageListObjectsCache.h @@ -0,0 +1,80 @@ +#pragma once + +#include +#include +#include +#include + +namespace DB +{ + +class ObjectStorageListObjectsCache +{ + friend class ObjectStorageListObjectsCacheTest; +public: + ObjectStorageListObjectsCache(const ObjectStorageListObjectsCache &) = delete; + ObjectStorageListObjectsCache(ObjectStorageListObjectsCache &&) noexcept = delete; + + ObjectStorageListObjectsCache& operator=(const ObjectStorageListObjectsCache &) = delete; + ObjectStorageListObjectsCache& operator=(ObjectStorageListObjectsCache &&) noexcept = delete; + + static ObjectStorageListObjectsCache & instance(); + + struct Key + { + Key( + const String & storage_description_, + const String & bucket_, + const String & prefix_, + bool with_tags_, + const std::chrono::steady_clock::time_point & expires_at_ = std::chrono::steady_clock::now(), + std::optional user_id_ = std::nullopt); + + std::string storage_description; + std::string bucket; + std::string prefix; + bool with_tags; + std::chrono::steady_clock::time_point expires_at; + std::optional user_id; + + bool operator==(const Key & other) const; + }; + + using Value = ObjectInfos; + struct KeyHasher + { + size_t operator()(const Key & key) const; + }; + + struct IsStale + { + bool operator()(const Key & key) const; + }; + + struct WeightFunction + { + size_t operator()(const Value & value) const; + }; + + using Cache = CacheBase; + + void set( + const Key & key, + const std::shared_ptr & value); + + std::optional get(const Key & key, bool filter_by_prefix = true); + + void clear(); + + void setMaxSizeInBytes(std::size_t size_in_bytes_); + void setMaxCount(std::size_t count); + void setTTL(std::size_t ttl_in_seconds_); + +private: + ObjectStorageListObjectsCache(); + + Cache cache; + size_t ttl_in_seconds {0}; +}; + +} diff --git a/src/Storages/Cache/tests/gtest_object_storage_list_objects_cache.cpp b/src/Storages/Cache/tests/gtest_object_storage_list_objects_cache.cpp new file mode 100644 index 000000000000..8eb45d520c96 --- /dev/null +++ b/src/Storages/Cache/tests/gtest_object_storage_list_objects_cache.cpp @@ -0,0 +1,244 @@ +#include +#include +#include +#include + +namespace DB +{ + +class ObjectStorageListObjectsCacheTest : public ::testing::Test +{ +protected: + void SetUp() override + { + cache = std::unique_ptr(new ObjectStorageListObjectsCache()); + cache->setTTL(3); + cache->setMaxCount(100); + cache->setMaxSizeInBytes(1000000); + } + + std::unique_ptr cache; + static ObjectStorageListObjectsCache::Key default_key; + + static std::shared_ptr createTestValue(const std::vector& paths) + { + auto value = std::make_shared(); + for (const auto & path : paths) + { + value->push_back(std::make_shared(path)); + } + return value; + } +}; + +ObjectStorageListObjectsCache::Key ObjectStorageListObjectsCacheTest::default_key {"default", "test-bucket", "test-prefix/", false}; + +TEST_F(ObjectStorageListObjectsCacheTest, BasicSetAndGet) +{ + cache->clear(); + auto value = createTestValue({"test-prefix/file1.txt", "test-prefix/file2.txt"}); + + cache->set(default_key, value); + + auto result = cache->get(default_key).value(); + + ASSERT_EQ(result.size(), 2); + EXPECT_EQ(result[0]->getPath(), "test-prefix/file1.txt"); + EXPECT_EQ(result[1]->getPath(), "test-prefix/file2.txt"); +} + +TEST_F(ObjectStorageListObjectsCacheTest, CacheMiss) +{ + cache->clear(); + + EXPECT_FALSE(cache->get(default_key)); +} + +TEST_F(ObjectStorageListObjectsCacheTest, ClearCache) +{ + cache->clear(); + auto value = createTestValue({"test-prefix/file1.txt", "test-prefix/file2.txt"}); + + cache->set(default_key, value); + cache->clear(); + + EXPECT_FALSE(cache->get(default_key)); +} + +TEST_F(ObjectStorageListObjectsCacheTest, PrefixMatching) +{ + cache->clear(); + + auto short_prefix_key = default_key; + short_prefix_key.prefix = "parent/"; + + auto mid_prefix_key = default_key; + mid_prefix_key.prefix = "parent/child/"; + + auto long_prefix_key = default_key; + long_prefix_key.prefix = "parent/child/grandchild/"; + + auto value = createTestValue( + { + "parent/child/grandchild/file1.txt", + "parent/child/grandchild/file2.txt"}); + + cache->set(mid_prefix_key, value); + + auto result1 = cache->get(mid_prefix_key).value(); + EXPECT_EQ(result1.size(), 2); + + auto result2 = cache->get(long_prefix_key).value(); + EXPECT_EQ(result2.size(), 2); + + EXPECT_FALSE(cache->get(short_prefix_key)); +} + +TEST_F(ObjectStorageListObjectsCacheTest, PrefixFiltering) +{ + cache->clear(); + + auto key_with_short_prefix = default_key; + key_with_short_prefix.prefix = "parent/"; + + auto key_with_mid_prefix = default_key; + key_with_mid_prefix.prefix = "parent/child1/"; + + auto value = createTestValue({ + "parent/file1.txt", + "parent/child1/file2.txt", + "parent/child2/file3.txt" + }); + + cache->set(key_with_short_prefix, value); + + auto result = cache->get(key_with_mid_prefix, true).value(); + EXPECT_EQ(result.size(), 1); + EXPECT_EQ(result[0]->getPath(), "parent/child1/file2.txt"); +} + +TEST_F(ObjectStorageListObjectsCacheTest, TTLExpiration) +{ + cache->clear(); + auto value = createTestValue({"test-prefix/file1.txt"}); + + cache->set(default_key, value); + + // Verify we can get it immediately + auto result1 = cache->get(default_key).value(); + EXPECT_EQ(result1.size(), 1); + + std::this_thread::sleep_for(std::chrono::seconds(4)); + + EXPECT_FALSE(cache->get(default_key)); +} + +TEST_F(ObjectStorageListObjectsCacheTest, TTLUnlimited) +{ + cache->clear(); + cache->setTTL(0); // 0 means unlimited + auto value = createTestValue({"test-prefix/file1.txt"}); + + cache->set(default_key, value); + + // Verify we can get it immediately + auto result1 = cache->get(default_key).value(); + EXPECT_EQ(result1.size(), 1); + + // Sleep for a reasonable amount (longer than the default 3 second TTL from SetUp) + std::this_thread::sleep_for(std::chrono::seconds(5)); + + // Should still be available since TTL is unlimited + auto result2 = cache->get(default_key).value(); + EXPECT_EQ(result2.size(), 1); + EXPECT_EQ(result2[0]->getPath(), "test-prefix/file1.txt"); +} + +TEST_F(ObjectStorageListObjectsCacheTest, TTLSwitchFromUnlimitedToFinite) +{ + cache->clear(); + cache->setTTL(0); // Start with unlimited + auto value1 = createTestValue({"test-prefix/file1.txt"}); + auto key1 = default_key; + key1.prefix = "unlimited/"; + + cache->set(key1, value1); + + // Switch to finite TTL and add another entry + cache->setTTL(1); + auto value2 = createTestValue({"test-prefix/file2.txt"}); + auto key2 = default_key; + key2.prefix = "finite/"; + + cache->set(key2, value2); + + // Verify both are available immediately + EXPECT_TRUE(cache->get(key1).has_value()); + EXPECT_TRUE(cache->get(key2).has_value()); + + // Wait for finite TTL entry to expire + std::this_thread::sleep_for(std::chrono::seconds(2)); + + // Unlimited entry should still be there, finite should be gone + EXPECT_TRUE(cache->get(key1).has_value()); + EXPECT_FALSE(cache->get(key2).has_value()); +} + +TEST_F(ObjectStorageListObjectsCacheTest, BestPrefixMatch) +{ + cache->clear(); + + auto short_prefix_key = default_key; + short_prefix_key.prefix = "a/b/"; + + auto mid_prefix_key = default_key; + mid_prefix_key.prefix = "a/b/c/"; + + auto long_prefix_key = default_key; + long_prefix_key.prefix = "a/b/c/d/"; + + auto short_prefix = createTestValue({"a/b/c/d/file1.txt", "a/b/c/file1.txt", "a/b/file2.txt"}); + auto mid_prefix = createTestValue({"a/b/c/d/file1.txt", "a/b/c/file1.txt"}); + + cache->set(short_prefix_key, short_prefix); + cache->set(mid_prefix_key, mid_prefix); + + // should pick mid_prefix, which has size 2. filter_by_prefix=false so we can assert by size + auto result = cache->get(long_prefix_key, false).value(); + EXPECT_EQ(result.size(), 2u); +} + +TEST_F(ObjectStorageListObjectsCacheTest, WithTags) +{ + cache->clear(); + + auto key_with_tags = default_key; + key_with_tags.with_tags = true; + + auto value_with_tags = createTestValue({"test.txt"}); + + cache->set(key_with_tags, value_with_tags); + + /// we have set with tags, we should be able to retrieve it + auto result_with_tags = cache->get(key_with_tags).value(); + EXPECT_EQ(result_with_tags.size(), 1u); + EXPECT_EQ(result_with_tags[0]->getPath(), "test.txt"); + + /// querying by a key without tags should return nothing + auto result_without_tags = cache->get(default_key); + EXPECT_FALSE(result_without_tags.has_value()); + + cache->clear(); + + cache->set(default_key, value_with_tags); + + /// querying by a key with tags should return nothing + EXPECT_FALSE(cache->get(key_with_tags).has_value()); + + /// querying by a key without tags should return the value + auto result_without_tags_2 = cache->get(default_key).value(); + EXPECT_EQ(result_without_tags_2.size(), 1u); + EXPECT_EQ(result_without_tags_2[0]->getPath(), "test.txt"); +} + +} diff --git a/src/Storages/ColumnsDescription.cpp b/src/Storages/ColumnsDescription.cpp index 42a1ac41886c..55a4b8c5979e 100644 --- a/src/Storages/ColumnsDescription.cpp +++ b/src/Storages/ColumnsDescription.cpp @@ -544,6 +544,15 @@ NamesAndTypesList ColumnsDescription::getInsertable() const return ret; } +NamesAndTypesList ColumnsDescription::getReadable() const +{ + NamesAndTypesList ret; + for (const auto & col : columns) + if (col.default_desc.kind != ColumnDefaultKind::Ephemeral) + ret.emplace_back(col.name, col.type); + return ret; +} + NamesAndTypesList ColumnsDescription::getMaterialized() const { NamesAndTypesList ret; @@ -944,7 +953,6 @@ std::optional ColumnsDescription::getDefault(const String & colum return {}; } - bool ColumnsDescription::hasCompressionCodec(const String & column_name) const { const auto it = columns.get<1>().find(column_name); diff --git a/src/Storages/ColumnsDescription.h b/src/Storages/ColumnsDescription.h index 152c2912a60e..5741d12c1863 100644 --- a/src/Storages/ColumnsDescription.h +++ b/src/Storages/ColumnsDescription.h @@ -172,6 +172,7 @@ class ColumnsDescription : public IHints<> NamesAndTypesList getOrdinary() const; NamesAndTypesList getMaterialized() const; NamesAndTypesList getInsertable() const; /// ordinary + ephemeral + NamesAndTypesList getReadable() const; /// ordinary + materialized + aliases (no ephemeral) NamesAndTypesList getAliases() const; NamesAndTypesList getEphemeral() const; NamesAndTypesList getAllPhysical() const; /// ordinary + materialized. diff --git a/src/Storages/ExportReplicatedMergeTreePartitionManifest.h b/src/Storages/ExportReplicatedMergeTreePartitionManifest.h new file mode 100644 index 000000000000..f1c96b120a28 --- /dev/null +++ b/src/Storages/ExportReplicatedMergeTreePartitionManifest.h @@ -0,0 +1,219 @@ +#pragma once + +#include +#include +#include +#include +#include +#include +#include + +namespace DB +{ + +struct ExportReplicatedMergeTreePartitionProcessingPartEntry +{ + + enum class Status + { + PENDING, + COMPLETED, + FAILED + }; + + String part_name; + Status status; + size_t retry_count; + String finished_by; + + std::string toJsonString() const + { + Poco::JSON::Object json; + + json.set("part_name", part_name); + json.set("status", String(magic_enum::enum_name(status))); + json.set("retry_count", retry_count); + json.set("finished_by", finished_by); + std::ostringstream oss; // STYLE_CHECK_ALLOW_STD_STRING_STREAM + oss.exceptions(std::ios::failbit); + Poco::JSON::Stringifier::stringify(json, oss); + + return oss.str(); + } + + static ExportReplicatedMergeTreePartitionProcessingPartEntry fromJsonString(const std::string & json_string) + { + Poco::JSON::Parser parser; + auto json = parser.parse(json_string).extract(); + chassert(json); + + ExportReplicatedMergeTreePartitionProcessingPartEntry entry; + + entry.part_name = json->getValue("part_name"); + entry.status = magic_enum::enum_cast(json->getValue("status")).value(); + entry.retry_count = json->getValue("retry_count"); + if (json->has("finished_by")) + { + entry.finished_by = json->getValue("finished_by"); + } + return entry; + } +}; + +struct ExportReplicatedMergeTreePartitionProcessedPartEntry +{ + String part_name; + std::vector paths_in_destination; + String finished_by; + + std::string toJsonString() const + { + Poco::JSON::Object json; + json.set("part_name", part_name); + json.set("paths_in_destination", paths_in_destination); + json.set("finished_by", finished_by); + std::ostringstream oss; // STYLE_CHECK_ALLOW_STD_STRING_STREAM + oss.exceptions(std::ios::failbit); + Poco::JSON::Stringifier::stringify(json, oss); + return oss.str(); + } + + static ExportReplicatedMergeTreePartitionProcessedPartEntry fromJsonString(const std::string & json_string) + { + Poco::JSON::Parser parser; + auto json = parser.parse(json_string).extract(); + chassert(json); + + ExportReplicatedMergeTreePartitionProcessedPartEntry entry; + + entry.part_name = json->getValue("part_name"); + + const auto paths_in_destination_array = json->getArray("paths_in_destination"); + for (size_t i = 0; i < paths_in_destination_array->size(); ++i) + entry.paths_in_destination.emplace_back(paths_in_destination_array->getElement(static_cast(i))); + + entry.finished_by = json->getValue("finished_by"); + + return entry; + } +}; + +struct ExportReplicatedMergeTreePartitionManifest +{ + String transaction_id; + String query_id; + String partition_id; + String destination_database; + String destination_table; + String source_replica; + size_t number_of_parts; + std::vector parts; + time_t create_time; + size_t max_retries; + size_t ttl_seconds; + size_t task_timeout_seconds; + size_t max_threads; + bool parallel_formatting; + bool parquet_parallel_encoding; + size_t max_bytes_per_file; + size_t max_rows_per_file; + MergeTreePartExportManifest::FileAlreadyExistsPolicy file_already_exists_policy; + String filename_pattern; + bool lock_inside_the_task; /// todo temporary + bool write_full_path_in_iceberg_metadata = false; + String iceberg_metadata_json; + + std::string toJsonString() const + { + Poco::JSON::Object json; + json.set("transaction_id", transaction_id); + json.set("query_id", query_id); + json.set("partition_id", partition_id); + json.set("destination_database", destination_database); + json.set("destination_table", destination_table); + json.set("source_replica", source_replica); + json.set("number_of_parts", number_of_parts); + + if (!iceberg_metadata_json.empty()) + { + json.set("iceberg_metadata_json", iceberg_metadata_json); + } + + Poco::JSON::Array::Ptr parts_array = new Poco::JSON::Array(); + for (const auto & part : parts) + parts_array->add(part); + json.set("parts", parts_array); + json.set("parallel_formatting", parallel_formatting); + json.set("max_threads", max_threads); + json.set("parquet_parallel_encoding", parquet_parallel_encoding); + json.set("max_bytes_per_file", max_bytes_per_file); + json.set("max_rows_per_file", max_rows_per_file); + json.set("file_already_exists_policy", String(magic_enum::enum_name(file_already_exists_policy))); + json.set("filename_pattern", filename_pattern); + json.set("create_time", create_time); + json.set("max_retries", max_retries); + json.set("ttl_seconds", ttl_seconds); + json.set("task_timeout_seconds", task_timeout_seconds); + json.set("lock_inside_the_task", lock_inside_the_task); + json.set("write_full_path_in_iceberg_metadata", write_full_path_in_iceberg_metadata); + std::ostringstream oss; // STYLE_CHECK_ALLOW_STD_STRING_STREAM + oss.exceptions(std::ios::failbit); + Poco::JSON::Stringifier::stringify(json, oss); + return oss.str(); + } + + static ExportReplicatedMergeTreePartitionManifest fromJsonString(const std::string & json_string) + { + Poco::JSON::Parser parser; + auto json = parser.parse(json_string).extract(); + chassert(json); + + ExportReplicatedMergeTreePartitionManifest manifest; + manifest.transaction_id = json->getValue("transaction_id"); + manifest.query_id = json->getValue("query_id"); + manifest.partition_id = json->getValue("partition_id"); + manifest.destination_database = json->getValue("destination_database"); + manifest.destination_table = json->getValue("destination_table"); + manifest.source_replica = json->getValue("source_replica"); + manifest.number_of_parts = json->getValue("number_of_parts"); + manifest.max_retries = json->getValue("max_retries"); + + if (json->has("iceberg_metadata_json")) + { + manifest.iceberg_metadata_json = json->getValue("iceberg_metadata_json"); + } + + auto parts_array = json->getArray("parts"); + for (size_t i = 0; i < parts_array->size(); ++i) + manifest.parts.push_back(parts_array->getElement(static_cast(i))); + + manifest.create_time = json->getValue("create_time"); + manifest.ttl_seconds = json->getValue("ttl_seconds"); + manifest.task_timeout_seconds = json->getValue("task_timeout_seconds"); + manifest.max_threads = json->getValue("max_threads"); + manifest.parallel_formatting = json->getValue("parallel_formatting"); + manifest.parquet_parallel_encoding = json->getValue("parquet_parallel_encoding"); + manifest.max_bytes_per_file = json->getValue("max_bytes_per_file"); + manifest.max_rows_per_file = json->getValue("max_rows_per_file"); + manifest.filename_pattern = json->getValue("filename_pattern"); + + if (json->has("file_already_exists_policy")) + { + const auto file_already_exists_policy = magic_enum::enum_cast(json->getValue("file_already_exists_policy")); + if (file_already_exists_policy) + { + manifest.file_already_exists_policy = file_already_exists_policy.value(); + } + + /// what to do if it's not a valid value? + } + + manifest.lock_inside_the_task = json->getValue("lock_inside_the_task"); + + manifest.write_full_path_in_iceberg_metadata = json->getValue("write_full_path_in_iceberg_metadata"); + + return manifest; + } +}; + +} diff --git a/src/Storages/ExportReplicatedMergeTreePartitionTaskEntry.h b/src/Storages/ExportReplicatedMergeTreePartitionTaskEntry.h new file mode 100644 index 000000000000..e62f7de99bed --- /dev/null +++ b/src/Storages/ExportReplicatedMergeTreePartitionTaskEntry.h @@ -0,0 +1,78 @@ +#pragma once + +#include +#include +#include "Core/QualifiedTableName.h" +#include +#include +#include +#include + +namespace DB +{ +struct ExportReplicatedMergeTreePartitionTaskEntry +{ + using DataPartPtr = std::shared_ptr; + ExportReplicatedMergeTreePartitionManifest manifest; + + enum class Status + { + PENDING, + COMPLETED, + FAILED, + KILLED + }; + + /// Allows us to skip completed / failed entries during scheduling + mutable Status status; + + /// References to the parts that should be exported + /// This is used to prevent the parts from being deleted before finishing the export operation + /// It does not mean this replica will export all the parts + /// There is also a chance this replica does not contain a given part and it is totally ok. + mutable std::vector part_references; + + std::string getCompositeKey() const + { + const auto qualified_table_name = QualifiedTableName {manifest.destination_database, manifest.destination_table}; + return manifest.partition_id + "_" + qualified_table_name.getFullName(); + } + + std::string getTransactionId() const + { + return manifest.transaction_id; + } + + /// Get create_time for sorted iteration + time_t getCreateTime() const + { + return manifest.create_time; + } +}; + +struct ExportPartitionTaskEntryTagByCompositeKey {}; +struct ExportPartitionTaskEntryTagByCreateTime {}; +struct ExportPartitionTaskEntryTagByTransactionId {}; + +// Multi-index container for export partition task entries +// - Index 0 (TagByCompositeKey): hashed_unique on composite key for O(1) lookup +// - Index 1 (TagByCreateTime): ordered_non_unique on create_time for sorted iteration +using ExportPartitionTaskEntriesContainer = boost::multi_index_container< + ExportReplicatedMergeTreePartitionTaskEntry, + boost::multi_index::indexed_by< + boost::multi_index::hashed_unique< + boost::multi_index::tag, + boost::multi_index::const_mem_fun + >, + boost::multi_index::ordered_non_unique< + boost::multi_index::tag, + boost::multi_index::const_mem_fun + >, + boost::multi_index::hashed_unique< + boost::multi_index::tag, + boost::multi_index::const_mem_fun + > + > +>; + +} diff --git a/src/Storages/IPartitionStrategy.cpp b/src/Storages/IPartitionStrategy.cpp index 079e5f07ecea..e301c977bb16 100644 --- a/src/Storages/IPartitionStrategy.cpp +++ b/src/Storages/IPartitionStrategy.cpp @@ -103,8 +103,11 @@ namespace const IPartitionStrategy & partition_strategy, BuildAST && build_ast) { +<<<<<<< HEAD /// The cache write happens in the cacheDeterministicActions function, which is called from the constructor of the partition strategy. /// If the actions are not deterministic, it will not be cached. +======= +>>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) if (cached_result) return *cached_result; @@ -312,17 +315,15 @@ ColumnPtr WildcardPartitionStrategy::computePartitionKey(const Chunk & chunk) co return block_with_partition_by_expr.getByName(actions_with_column.column_name).column; } -std::string WildcardPartitionStrategy::getPathForRead( - const std::string & prefix) +ColumnPtr WildcardPartitionStrategy::computePartitionKey(Block & block) const { - return prefix; -} + auto actions_with_column = getCachedOrBuildActions( + cached_result, + *this, + [&] { return buildToStringPartitionAST(partition_key_description.definition_ast); }); -std::string WildcardPartitionStrategy::getPathForWrite( - const std::string & prefix, - const std::string & partition_key) -{ - return PartitionedSink::replaceWildcards(prefix, partition_key); + actions_with_column.actions->execute(block); + return block.getByName(actions_with_column.column_name).column; } HiveStylePartitionStrategy::HiveStylePartitionStrategy( @@ -350,8 +351,9 @@ HiveStylePartitionStrategy::HiveStylePartitionStrategy( cacheDeterministicActions(cached_result, actions_with_column); } -std::string HiveStylePartitionStrategy::getPathForRead(const std::string & prefix) +ColumnPtr HiveStylePartitionStrategy::computePartitionKey(const Chunk & chunk) const { +<<<<<<< HEAD return prefix + "**." + Poco::toLower(file_format); } @@ -387,6 +389,8 @@ std::string HiveStylePartitionStrategy::getPathForWrite( ColumnPtr HiveStylePartitionStrategy::computePartitionKey(const Chunk & chunk) const { +======= +>>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) auto actions_with_column = getCachedOrBuildActions( cached_result, *this, @@ -399,6 +403,17 @@ ColumnPtr HiveStylePartitionStrategy::computePartitionKey(const Chunk & chunk) c return block_with_partition_by_expr.getByName(actions_with_column.column_name).column; } +ColumnPtr HiveStylePartitionStrategy::computePartitionKey(Block & block) const +{ + auto actions_with_column = getCachedOrBuildActions( + cached_result, + *this, + [&] { return buildHivePartitionAST(partition_key_description.definition_ast, getPartitionColumns()); }); + + actions_with_column.actions->execute(block); + return block.getByName(actions_with_column.column_name).column; +} + ColumnRawPtrs HiveStylePartitionStrategy::getFormatChunkColumns(const Chunk & chunk) { ColumnRawPtrs result; diff --git a/src/Storages/IPartitionStrategy.h b/src/Storages/IPartitionStrategy.h index b2899b0e4d0a..987b892054a8 100644 --- a/src/Storages/IPartitionStrategy.h +++ b/src/Storages/IPartitionStrategy.h @@ -29,8 +29,7 @@ struct IPartitionStrategy virtual ColumnPtr computePartitionKey(const Chunk & chunk) const = 0; - virtual std::string getPathForRead(const std::string & prefix) = 0; - virtual std::string getPathForWrite(const std::string & prefix, const std::string & partition_key) = 0; + virtual ColumnPtr computePartitionKey(Block & block) const = 0; virtual ColumnRawPtrs getFormatChunkColumns(const Chunk & chunk) { @@ -93,8 +92,13 @@ struct WildcardPartitionStrategy : IPartitionStrategy WildcardPartitionStrategy(KeyDescription partition_key_description_, const Block & sample_block_, ContextPtr context_); ColumnPtr computePartitionKey(const Chunk & chunk) const override; +<<<<<<< HEAD std::string getPathForRead(const std::string & prefix) override; std::string getPathForWrite(const std::string & prefix, const std::string & partition_key) override; +======= + + ColumnPtr computePartitionKey(Block & block) const override; +>>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) }; /* @@ -112,8 +116,13 @@ struct HiveStylePartitionStrategy : IPartitionStrategy bool partition_columns_in_data_file_); ColumnPtr computePartitionKey(const Chunk & chunk) const override; +<<<<<<< HEAD std::string getPathForRead(const std::string & prefix) override; std::string getPathForWrite(const std::string & prefix, const std::string & partition_key) override; +======= + + ColumnPtr computePartitionKey(Block & block) const override; +>>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) ColumnRawPtrs getFormatChunkColumns(const Chunk & chunk) override; Block getFormatHeader() override; diff --git a/src/Storages/IStorage.cpp b/src/Storages/IStorage.cpp index 6922bdd44255..81bd9f1148c6 100644 --- a/src/Storages/IStorage.cpp +++ b/src/Storages/IStorage.cpp @@ -311,6 +311,11 @@ CancellationCode IStorage::killPartMoveToShard(const UUID & /*task_uuid*/) throw Exception(ErrorCodes::NOT_IMPLEMENTED, "Part moves between shards are not supported by storage {}", getName()); } +CancellationCode IStorage::killExportPartition(const String & /*transaction_id*/) +{ + throw Exception(ErrorCodes::NOT_IMPLEMENTED, "Export partition is not supported by storage {}", getName()); +} + StorageID IStorage::getStorageID() const { std::lock_guard lock(id_mutex); diff --git a/src/Storages/IStorage.h b/src/Storages/IStorage.h index e3a4396159a2..9c02ade588df 100644 --- a/src/Storages/IStorage.h +++ b/src/Storages/IStorage.h @@ -1,5 +1,6 @@ #pragma once +#include #include #include #include @@ -19,6 +20,7 @@ #include #include #include +#include #include #include @@ -436,6 +438,51 @@ class IStorage : public std::enable_shared_from_this, public TypePromo ContextPtr /*context*/, bool /*async_insert*/); + virtual bool supportsImport(ContextPtr) const + { + return false; + } + + /* +It is currently only implemented in StorageObjectStorage. + It is meant to be used to import merge tree data parts into object storage. It is similar to the write API, + but it won't re-partition the data and should allow the filename to be set by the caller. + */ + virtual SinkToStoragePtr import( + const std::string & /* file_name */, + Block & /* block_with_partition_values */, + const std::function & /* new_file_path_callback */, + bool /* overwrite_if_exists */, + std::size_t /* max_bytes_per_file */, + std::size_t /* max_rows_per_file */, + const std::optional & /* iceberg_metadata_json_string */, + const std::optional & /* format_settings */, + ContextPtr /* context */) + { + throw Exception(ErrorCodes::NOT_IMPLEMENTED, "Import is not implemented for storage {}", getName()); + } + + struct IcebergCommitExportPartitionArguments + { + std::string metadata_json_string; + /// Partition column values (after transforms). Callers are responsible for + /// populating this: the partition-export path parses them from the persisted + /// JSON string, while the direct EXPORT PART path reads them from the part's + /// partition key. + std::vector partition_values; + }; + + virtual void commitExportPartitionTransaction( + const String & /* transaction_id */, + const String & /* partition_id */, + const Strings & /* exported_paths */, + const IcebergCommitExportPartitionArguments & /* iceberg_commit_export_partition_arguments */, + ContextPtr /* local_context */) + { + throw Exception(ErrorCodes::NOT_IMPLEMENTED, "commitExportPartitionTransaction is not implemented for storage type {}", getName()); + } + + /** Writes the data to a table in distributed manner. * It is supposed that implementation looks into SELECT part of the query and executes distributed * INSERT SELECT if it is possible with current storage as a receiver and query SELECT part as a producer. @@ -548,6 +595,9 @@ class IStorage : public std::enable_shared_from_this, public TypePromo virtual void setMutationCSN(const String & /*mutation_id*/, UInt64 /*csn*/); + /// Cancel a replicated partition export by transaction id. + virtual CancellationCode killExportPartition(const String & /*transaction_id*/); + /// Cancel a part move to shard. virtual CancellationCode killPartMoveToShard(const UUID & /*task_uuid*/); diff --git a/src/Storages/MergeTree/BackgroundJobsAssignee.cpp b/src/Storages/MergeTree/BackgroundJobsAssignee.cpp index 093983b2ced0..4669a3977289 100644 --- a/src/Storages/MergeTree/BackgroundJobsAssignee.cpp +++ b/src/Storages/MergeTree/BackgroundJobsAssignee.cpp @@ -100,6 +100,10 @@ bool BackgroundJobsAssignee::scheduleCommonTask(ExecutableTaskPtr common_task, b return schedule_res; } +std::size_t BackgroundJobsAssignee::getAvailableMoveExecutors() const +{ + return getContext()->getMovesExecutor()->getAvailableSlots(); +} String BackgroundJobsAssignee::toString(Type type) { diff --git a/src/Storages/MergeTree/BackgroundJobsAssignee.h b/src/Storages/MergeTree/BackgroundJobsAssignee.h index 25d3806ba4d2..3a50ad85cfd7 100644 --- a/src/Storages/MergeTree/BackgroundJobsAssignee.h +++ b/src/Storages/MergeTree/BackgroundJobsAssignee.h @@ -74,6 +74,8 @@ class BackgroundJobsAssignee : public WithContext bool scheduleMoveTask(ExecutableTaskPtr move_task); bool scheduleCommonTask(ExecutableTaskPtr common_task, bool need_trigger); + std::size_t getAvailableMoveExecutors() const; + /// Just call finish ~BackgroundJobsAssignee(); diff --git a/src/Storages/MergeTree/ExportList.cpp b/src/Storages/MergeTree/ExportList.cpp new file mode 100644 index 000000000000..018c1f091ef9 --- /dev/null +++ b/src/Storages/MergeTree/ExportList.cpp @@ -0,0 +1,74 @@ +#include + +namespace DB +{ + +ExportsListElement::ExportsListElement( + const StorageID & source_table_id_, + const StorageID & destination_table_id_, + UInt64 part_size_, + const String & part_name_, + const std::vector & destination_file_paths_, + UInt64 total_rows_to_read_, + UInt64 total_size_bytes_compressed_, + UInt64 total_size_bytes_uncompressed_, + time_t create_time_, + const String & query_id_, + const ContextPtr & context) +: source_table_id(source_table_id_) +, destination_table_id(destination_table_id_) +, part_size(part_size_) +, part_name(part_name_) +, destination_file_paths(destination_file_paths_) +, total_rows_to_read(total_rows_to_read_) +, total_size_bytes_compressed(total_size_bytes_compressed_) +, total_size_bytes_uncompressed(total_size_bytes_uncompressed_) +, create_time(create_time_) +, query_id(query_id_) +{ + thread_group = ThreadGroup::createForMergeMutate(context); +} + +ExportsListElement::~ExportsListElement() +{ + background_memory_tracker.adjustOnBackgroundTaskEnd(&thread_group->memory_tracker); +} + +ExportInfo ExportsListElement::getInfo() const +{ + ExportInfo res; + res.source_database = source_table_id.database_name; + res.source_table = source_table_id.table_name; + res.destination_database = destination_table_id.database_name; + res.destination_table = destination_table_id.table_name; + res.part_name = part_name; + + { + std::shared_lock lock(destination_file_paths_mutex); + res.destination_file_paths = destination_file_paths; + } + + res.rows_read = rows_read.load(std::memory_order_relaxed); + res.total_rows_to_read = total_rows_to_read; + res.total_size_bytes_compressed = total_size_bytes_compressed; + res.total_size_bytes_uncompressed = total_size_bytes_uncompressed; + res.bytes_read_uncompressed = bytes_read_uncompressed.load(std::memory_order_relaxed); + res.memory_usage = getMemoryUsage(); + res.peak_memory_usage = getPeakMemoryUsage(); + res.create_time = create_time; + res.elapsed = watch.elapsedSeconds(); + res.query_id = query_id; + return res; +} + +UInt64 ExportsListElement::getMemoryUsage() const +{ + return thread_group->memory_tracker.get(); +} + +UInt64 ExportsListElement::getPeakMemoryUsage() const +{ + return thread_group->memory_tracker.getPeak(); +} + +} diff --git a/src/Storages/MergeTree/ExportList.h b/src/Storages/MergeTree/ExportList.h new file mode 100644 index 000000000000..4a02826dfe44 --- /dev/null +++ b/src/Storages/MergeTree/ExportList.h @@ -0,0 +1,96 @@ +#pragma once + +#include +#include +#include +#include +#include +#include +#include +#include + +namespace CurrentMetrics +{ + extern const Metric Export; +} + +namespace DB +{ + +struct ExportInfo +{ + String source_database; + String source_table; + String destination_database; + String destination_table; + String part_name; + std::vector destination_file_paths; + UInt64 rows_read; + UInt64 total_rows_to_read; + UInt64 total_size_bytes_compressed; + UInt64 total_size_bytes_uncompressed; + UInt64 bytes_read_uncompressed; + UInt64 memory_usage; + UInt64 peak_memory_usage; + time_t create_time = 0; + Float64 elapsed; + String query_id; +}; + +struct ExportsListElement : private boost::noncopyable +{ + const StorageID source_table_id; + const StorageID destination_table_id; + const UInt64 part_size; + const String part_name; + + /// see destination_file_paths_mutex + std::vector destination_file_paths; + std::atomic rows_read {0}; + UInt64 total_rows_to_read {0}; + UInt64 total_size_bytes_compressed {0}; + UInt64 total_size_bytes_uncompressed {0}; + std::atomic bytes_read_uncompressed {0}; + time_t create_time {0}; + String query_id; + + Stopwatch watch; + ThreadGroupPtr thread_group; + mutable std::shared_mutex destination_file_paths_mutex; + + ExportsListElement( + const StorageID & source_table_id_, + const StorageID & destination_table_id_, + UInt64 part_size_, + const String & part_name_, + const std::vector & destination_file_paths_, + UInt64 total_rows_to_read_, + UInt64 total_size_bytes_compressed_, + UInt64 total_size_bytes_uncompressed_, + time_t create_time_, + const String & query_id_, + const ContextPtr & context); + + ~ExportsListElement(); + + ExportInfo getInfo() const; + + UInt64 getMemoryUsage() const; + UInt64 getPeakMemoryUsage() const; +}; + + +class ExportsList final : public BackgroundProcessList +{ +private: + using Parent = BackgroundProcessList; + +public: + ExportsList() + : Parent(CurrentMetrics::Export) + {} +}; + +using ExportsListEntry = BackgroundProcessListEntry; + +} diff --git a/src/Storages/MergeTree/ExportPartFromPartitionExportTask.cpp b/src/Storages/MergeTree/ExportPartFromPartitionExportTask.cpp new file mode 100644 index 000000000000..2a3343e73f13 --- /dev/null +++ b/src/Storages/MergeTree/ExportPartFromPartitionExportTask.cpp @@ -0,0 +1,75 @@ +#include +#include +#include + +namespace ProfileEvents +{ + extern const Event ExportPartitionZooKeeperRequests; + extern const Event ExportPartitionZooKeeperGetChildren; + extern const Event ExportPartitionZooKeeperCreate; +} +namespace DB +{ + +ExportPartFromPartitionExportTask::ExportPartFromPartitionExportTask( + StorageReplicatedMergeTree & storage_, + const std::string & key_, + const MergeTreePartExportManifest & manifest_) + : storage(storage_), + key(key_), + manifest(manifest_) +{ + export_part_task = std::make_shared(storage, manifest); +} + +bool ExportPartFromPartitionExportTask::executeStep() +{ + /// Runs on a MergeTreeBackgroundExecutor thread, so it does not inherit any component set by the scheduling task. + auto component_guard = Coordination::setCurrentComponent("ExportPartFromPartitionExportTask::executeStep"); + + const auto zk = storage.getZooKeeper(); + const auto part_name = manifest.data_part->name; + + LOG_INFO(storage.log, "ExportPartFromPartitionExportTask: Attempting to lock part: {}", part_name); + + ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRequests); + ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperCreate); + if (Coordination::Error::ZOK == zk->tryCreate(fs::path(storage.zookeeper_path) / "exports" / key / "locks" / part_name, storage.replica_name, zkutil::CreateMode::Ephemeral)) + { + LOG_INFO(storage.log, "ExportPartFromPartitionExportTask: Locked part: {}", part_name); + export_part_task->executeStep(); + return false; + } + + std::lock_guard inner_lock(storage.export_manifests_mutex); + storage.export_manifests.erase(manifest); + + LOG_INFO(storage.log, "ExportPartFromPartitionExportTask: Failed to lock part {}, skipping", part_name); + return false; +} + +void ExportPartFromPartitionExportTask::cancel() noexcept +{ + export_part_task->cancel(); +} + +void ExportPartFromPartitionExportTask::onCompleted() +{ + export_part_task->onCompleted(); +} + +StorageID ExportPartFromPartitionExportTask::getStorageID() const +{ + return export_part_task->getStorageID(); +} + +Priority ExportPartFromPartitionExportTask::getPriority() const +{ + return export_part_task->getPriority(); +} + +String ExportPartFromPartitionExportTask::getQueryId() const +{ + return export_part_task->getQueryId(); +} +} diff --git a/src/Storages/MergeTree/ExportPartFromPartitionExportTask.h b/src/Storages/MergeTree/ExportPartFromPartitionExportTask.h new file mode 100644 index 000000000000..e170b22b470d --- /dev/null +++ b/src/Storages/MergeTree/ExportPartFromPartitionExportTask.h @@ -0,0 +1,36 @@ +#pragma once + +#include +#include +#include +#include + +namespace DB +{ + +/* + Decorator around the ExportPartTask to lock the part inside the task +*/ +class ExportPartFromPartitionExportTask : public IExecutableTask +{ +public: + explicit ExportPartFromPartitionExportTask( + StorageReplicatedMergeTree & storage_, + const std::string & key_, + const MergeTreePartExportManifest & manifest_); + bool executeStep() override; + void onCompleted() override; + StorageID getStorageID() const override; + Priority getPriority() const override; + String getQueryId() const override; + + void cancel() noexcept override; + +private: + StorageReplicatedMergeTree & storage; + std::string key; + MergeTreePartExportManifest manifest; + std::shared_ptr export_part_task; +}; + +} diff --git a/src/Storages/MergeTree/ExportPartTask.cpp b/src/Storages/MergeTree/ExportPartTask.cpp new file mode 100644 index 000000000000..1014b9df7422 --- /dev/null +++ b/src/Storages/MergeTree/ExportPartTask.cpp @@ -0,0 +1,414 @@ +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include "Common/setThreadName.h" +#include +#include +#include +#include +#include +#include +#include +#include + +namespace ProfileEvents +{ + extern const Event PartsExportDuplicated; + extern const Event PartsExportFailures; + extern const Event PartsExports; + extern const Event PartsExportTotalMilliseconds; +} + +namespace DB +{ + +namespace ErrorCodes +{ + extern const int UNKNOWN_TABLE; + extern const int FILE_ALREADY_EXISTS; + extern const int LOGICAL_ERROR; + extern const int QUERY_WAS_CANCELLED; +} + +namespace Setting +{ + extern const SettingsUInt64 min_bytes_to_use_direct_io; + extern const SettingsUInt64 export_merge_tree_part_max_bytes_per_file; + extern const SettingsUInt64 export_merge_tree_part_max_rows_per_file; + extern const SettingsBool allow_experimental_analyzer; + extern const SettingsString export_merge_tree_part_filename_pattern; +} + +namespace +{ + void materializeSpecialColumns( + const SharedHeader & header, + const StorageMetadataPtr & storage_metadata, + const ContextPtr & local_context, + QueryPlan & plan_for_part + ) + { + const auto readable_columns = storage_metadata->getColumns().getReadable(); + + // Enable all experimental settings for default expressions + // (same pattern as in IMergeTreeReader::evaluateMissingDefaults) + auto context_for_defaults = Context::createCopy(local_context); + enableAllExperimentalSettings(context_for_defaults); + + /// Copy the behavior of `IMergeTreeReader`, see https://github.com/ClickHouse/ClickHouse/blob/c45224e3f0a6dd9a9217e5d75723f378ffe0a86a/src/Storages/MergeTree/IMergeTreeReader.cpp#L215 + context_for_defaults->setSetting("enable_analyzer", local_context->getSettingsRef()[Setting::allow_experimental_analyzer].value); + + auto defaults_dag = evaluateMissingDefaults( + *header, + readable_columns, + storage_metadata->getColumns(), + context_for_defaults); + + if (defaults_dag) + { + ActionsDAG base_dag(header->getColumnsWithTypeAndName()); + + /// `evaluateMissingDefaults` has a new analyzer path since https://github.com/ClickHouse/ClickHouse/pull/87585 + /// which returns a DAG that does not contain all columns. We need to merge it with the base DAG to get all columns. + auto merged = ActionsDAG::merge(std::move(base_dag), std::move(*defaults_dag)); + + /// Ensure columns are in the correct order matching readable_columns + merged.removeUnusedActions(readable_columns.getNames(), false); + merged.addMaterializingOutputActions(/*materialize_sparse=*/ false); + + auto expression_step = std::make_unique( + header, + std::move(merged)); + expression_step->setStepDescription("Compute alias and default expressions for export"); + plan_for_part.addStep(std::move(expression_step)); + } + } + + String buildDestinationFilename( + const MergeTreePartExportManifest & manifest, + const StorageID & storage_id, + const ContextPtr & local_context) + { + auto filename = manifest.settings[Setting::export_merge_tree_part_filename_pattern].value; + + boost::replace_all(filename, "{part_name}", manifest.data_part->name); + boost::replace_all(filename, "{checksum}", manifest.data_part->checksums.getTotalChecksumHex()); + + Macros::MacroExpansionInfo macro_info; + macro_info.table_id = storage_id; + + if (auto database = DatabaseCatalog::instance().tryGetDatabase(storage_id.database_name)) + { + if (const auto replicated = dynamic_cast(database.get())) + { + macro_info.shard = replicated->getShardName(); + macro_info.replica = replicated->getReplicaName(); + } + } + + filename = local_context->getMacros()->expand(filename, macro_info); + + return filename; + } +} + +ExportPartTask::ExportPartTask(MergeTreeData & storage_, const MergeTreePartExportManifest & manifest_) + : storage(storage_), + manifest(manifest_) +{ +} + +const MergeTreePartExportManifest & ExportPartTask::getManifest() const +{ + return manifest; +} + +bool ExportPartTask::executeStep() +{ + auto local_context = Context::createCopy(storage.getContext()); + local_context->makeQueryContextForExportPart(); + local_context->setCurrentQueryId(manifest.query_id); + local_context->setSettings(manifest.settings); + + const auto & metadata_snapshot = manifest.metadata_snapshot; + + /// Read only physical columns from the part + const auto columns_to_read = metadata_snapshot->getColumns().getNamesOfPhysical(); + + MergeTreeSequentialSourceType read_type = MergeTreeSequentialSourceType::Export; + + Block block_with_partition_values; + if (metadata_snapshot->hasPartitionKey()) + { + /// todo arthur do I need to init minmax_idx? + block_with_partition_values = manifest.data_part->minmax_idx->getBlock(storage); + } + + const auto & destination_storage = manifest.destination_storage_ptr; + const auto destination_storage_id = destination_storage->getStorageID(); + + auto exports_list_entry = storage.getContext()->getExportsList().insert( + getStorageID(), + destination_storage_id, + manifest.data_part->getBytesOnDisk(), + manifest.data_part->name, + std::vector{}, + manifest.data_part->rows_count, + manifest.data_part->getBytesOnDisk(), + manifest.data_part->getBytesUncompressedOnDisk(), + manifest.create_time, + manifest.query_id, + local_context); + + SinkToStoragePtr sink; + + const auto new_file_path_callback = [&exports_list_entry](const std::string & file_path) + { + std::unique_lock lock((*exports_list_entry)->destination_file_paths_mutex); + (*exports_list_entry)->destination_file_paths.push_back(file_path); + }; + + try + { + const auto filename = buildDestinationFilename(manifest, storage.getStorageID(), local_context); + + sink = destination_storage->import( + filename, + block_with_partition_values, + new_file_path_callback, + manifest.file_already_exists_policy == MergeTreePartExportManifest::FileAlreadyExistsPolicy::overwrite, + manifest.settings[Setting::export_merge_tree_part_max_bytes_per_file], + manifest.settings[Setting::export_merge_tree_part_max_rows_per_file], + manifest.iceberg_metadata_json, + getFormatSettings(local_context), + local_context); + + bool apply_deleted_mask = true; + bool read_with_direct_io = local_context->getSettingsRef()[Setting::min_bytes_to_use_direct_io] > manifest.data_part->getBytesOnDisk(); + bool prefetch = false; + + MergeTreeData::IMutationsSnapshot::Params mutations_snapshot_params + { + .metadata_version = metadata_snapshot->getMetadataVersion(), + .min_part_metadata_version = manifest.data_part->getMetadataVersion() + }; + + auto mutations_snapshot = storage.getMutationsSnapshot(mutations_snapshot_params); + auto alter_conversions = MergeTreeData::getAlterConversionsForPart( + manifest.data_part, + mutations_snapshot, + local_context); + + QueryPlan plan_for_part; + + createReadFromPartStep( + read_type, + plan_for_part, + storage, + storage.getStorageSnapshot(metadata_snapshot, local_context), + RangesInDataPart(manifest.data_part), + alter_conversions, + nullptr, + columns_to_read, + nullptr, + apply_deleted_mask, + std::nullopt, + read_with_direct_io, + prefetch, + local_context, + getLogger("ExportPartition")); + + ThreadGroupSwitcher switcher((*exports_list_entry)->thread_group, ThreadName::EXPORT_PART); + + /// We need to support exporting materialized and alias columns to object storage. For some reason, object storage engines don't support them. + /// This is a hack that materializes the columns before the export so they can be exported to tables that have matching columns + materializeSpecialColumns(plan_for_part.getCurrentHeader(), metadata_snapshot, local_context, plan_for_part); + + QueryPlanOptimizationSettings optimization_settings(local_context); + auto pipeline_settings = BuildQueryPipelineSettings(local_context); + auto builder = plan_for_part.buildQueryPipeline(optimization_settings, pipeline_settings); + + builder->setProgressCallback([&exports_list_entry](const Progress & progress) + { + (*exports_list_entry)->bytes_read_uncompressed += progress.read_bytes; + (*exports_list_entry)->rows_read += progress.read_rows; + }); + + pipeline = QueryPipelineBuilder::getPipeline(std::move(*builder)); + + pipeline.complete(sink); + + CompletedPipelineExecutor exec(pipeline); + + auto is_cancelled_callback = [this]() + { + return isCancelled(); + }; + + exec.setCancelCallback(is_cancelled_callback, 100); + + if (isCancelled()) + { + throw Exception(ErrorCodes::QUERY_WAS_CANCELLED, "Export part was cancelled"); + } + + exec.execute(); + + if (isCancelled()) + { + throw Exception(ErrorCodes::QUERY_WAS_CANCELLED, "Export part was cancelled"); + } + + /// For the direct EXPORT PART → Iceberg path there is no deferred-commit callback + /// (the partition-export path provides one that writes to ZooKeeper). + /// Commit the Iceberg metadata inline here so the rows become visible immediately. + if (destination_storage->isDataLake() && !manifest.completion_callback) + { + IStorage::IcebergCommitExportPartitionArguments iceberg_args; + iceberg_args.metadata_json_string = manifest.iceberg_metadata_json; + iceberg_args.partition_values = manifest.data_part->partition.value; + + destination_storage->commitExportPartitionTransaction( + manifest.transaction_id, + manifest.data_part->info.getPartitionId(), + (*exports_list_entry)->destination_file_paths, + iceberg_args, + local_context); + } + + std::lock_guard inner_lock(storage.export_manifests_mutex); + storage.writePartLog( + PartLogElement::Type::EXPORT_PART, + {}, + (*exports_list_entry)->watch.elapsed(), + manifest.data_part->name, + manifest.data_part, + {manifest.data_part}, + nullptr, + nullptr, + {}, + exports_list_entry.get()); + + storage.export_manifests.erase(manifest); + + ProfileEvents::increment(ProfileEvents::PartsExports); + ProfileEvents::increment(ProfileEvents::PartsExportTotalMilliseconds, (*exports_list_entry)->watch.elapsedMilliseconds()); + + if (manifest.completion_callback) + manifest.completion_callback(MergeTreePartExportManifest::CompletionCallbackResult::createSuccess((*exports_list_entry)->destination_file_paths)); + } + catch (const Exception & e) + { + /// If an exception is thrown before the pipeline is started, the sink will not be canceled and might leave buffers open. + /// Cancel it manually to ensure the buffers are closed. + if (sink) + { + sink->cancel(); + } + + if (e.code() == ErrorCodes::FILE_ALREADY_EXISTS) + { + ProfileEvents::increment(ProfileEvents::PartsExportDuplicated); + + /// File already exists and the policy is NO_OP, treat it as success. + if (manifest.file_already_exists_policy == MergeTreePartExportManifest::FileAlreadyExistsPolicy::skip) + { + storage.writePartLog( + PartLogElement::Type::EXPORT_PART, + {}, + (*exports_list_entry)->watch.elapsed(), + manifest.data_part->name, + manifest.data_part, + {manifest.data_part}, + nullptr, + nullptr, + {}, + exports_list_entry.get()); + + std::lock_guard inner_lock(storage.export_manifests_mutex); + storage.export_manifests.erase(manifest); + + ProfileEvents::increment(ProfileEvents::PartsExports); + ProfileEvents::increment(ProfileEvents::PartsExportTotalMilliseconds, (*exports_list_entry)->watch.elapsedMilliseconds()); + + if (manifest.completion_callback) + { + manifest.completion_callback(MergeTreePartExportManifest::CompletionCallbackResult::createSuccess((*exports_list_entry)->destination_file_paths)); + } + + return false; + } + } + + ProfileEvents::increment(ProfileEvents::PartsExportFailures); + + storage.writePartLog( + PartLogElement::Type::EXPORT_PART, + ExecutionStatus::fromCurrentException("", true), + (*exports_list_entry)->watch.elapsed(), + manifest.data_part->name, + manifest.data_part, + {manifest.data_part}, + nullptr, + nullptr, + {}, + exports_list_entry.get()); + + std::lock_guard inner_lock(storage.export_manifests_mutex); + storage.export_manifests.erase(manifest); + + if (manifest.completion_callback) + manifest.completion_callback(MergeTreePartExportManifest::CompletionCallbackResult::createFailure(e)); + return false; + } + + return false; +} + +void ExportPartTask::cancel() noexcept +{ + LOG_INFO(getLogger("ExportPartTask"), "Export part {} task cancel() method called", manifest.data_part->name); + cancel_requested.store(true); + pipeline.cancel(); +} + +bool ExportPartTask::isCancelled() const +{ + return cancel_requested.load() || storage.parts_mover.moves_blocker.isCancelled(); +} + +void ExportPartTask::onCompleted() +{ +} + +StorageID ExportPartTask::getStorageID() const +{ + return storage.getStorageID(); +} + +Priority ExportPartTask::getPriority() const +{ + return Priority{}; +} + +String ExportPartTask::getQueryId() const +{ + return manifest.query_id; +} + +} diff --git a/src/Storages/MergeTree/ExportPartTask.h b/src/Storages/MergeTree/ExportPartTask.h new file mode 100644 index 000000000000..a3f1635c4902 --- /dev/null +++ b/src/Storages/MergeTree/ExportPartTask.h @@ -0,0 +1,34 @@ +#pragma once + +#include +#include +#include + +namespace DB +{ + +class ExportPartTask : public IExecutableTask +{ +public: + explicit ExportPartTask( + MergeTreeData & storage_, + const MergeTreePartExportManifest & manifest_); + bool executeStep() override; + void onCompleted() override; + StorageID getStorageID() const override; + Priority getPriority() const override; + String getQueryId() const override; + const MergeTreePartExportManifest & getManifest() const; + + void cancel() noexcept override; + +private: + MergeTreeData & storage; + MergeTreePartExportManifest manifest; + QueryPipeline pipeline; + std::atomic cancel_requested = false; + + bool isCancelled() const; +}; + +} diff --git a/src/Storages/MergeTree/ExportPartitionManifestUpdatingTask.cpp b/src/Storages/MergeTree/ExportPartitionManifestUpdatingTask.cpp new file mode 100644 index 000000000000..0ed8d1033135 --- /dev/null +++ b/src/Storages/MergeTree/ExportPartitionManifestUpdatingTask.cpp @@ -0,0 +1,883 @@ +#include +#include +#include +#include "Storages/MergeTree/ExportPartitionUtils.h" +#include "Common/logger_useful.h" +#include +#include +#include +#include +#include +#include + +namespace ProfileEvents +{ + extern const Event ExportPartitionZooKeeperRequests; + extern const Event ExportPartitionZooKeeperGet; + extern const Event ExportPartitionZooKeeperGetChildren; + extern const Event ExportPartitionZooKeeperGetChildrenWatch; + extern const Event ExportPartitionZooKeeperGetWatch; + extern const Event ExportPartitionZooKeeperRemoveRecursive; + extern const Event ExportPartitionZooKeeperMulti; +} + +namespace DB +{ + +namespace ErrorCodes +{ + extern const int FAULT_INJECTED; +} + +namespace FailPoints +{ + extern const char export_partition_status_change_throw[]; +} + +namespace +{ + /* + Remove expired entries and fix non-committed exports that have already exported all parts. + + Return values: + - true: the cleanup was successful, the entry is removed from the entries_by_key container and the function returns true. Proceed to the next entry. + - false: the cleanup was not successful, the entry is not removed from the entries_by_key container and the function returns false. + */ + bool tryCleanup( + const zkutil::ZooKeeperPtr & zk, + const std::string & entry_path, + const LoggerPtr & log, + const ContextPtr & storage_context, + StorageReplicatedMergeTree & storage, + const std::string & key, + const ExportReplicatedMergeTreePartitionManifest & metadata, + const time_t now, + const bool is_pending, + auto & entries_by_key + ) + { + bool has_expired = metadata.create_time < now - static_cast(metadata.ttl_seconds); + + bool task_timed_out = is_pending + && metadata.task_timeout_seconds > 0 + && metadata.create_time + static_cast(metadata.task_timeout_seconds) < now; + + if (has_expired && !is_pending) + { + zk->tryRemoveRecursive(fs::path(entry_path)); + ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRequests); + ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRemoveRecursive); + auto it = entries_by_key.find(key); + if (it != entries_by_key.end()) + entries_by_key.erase(it); + LOG_INFO(log, "ExportPartition Manifest Updating Task: Removed {}: expired", key); + + return true; + } + else if (task_timed_out) + { + const std::string status_path = fs::path(entry_path) / "status"; + + Coordination::Stat status_stat; + std::string status_string; + + ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRequests); + ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperGet); + if (!zk->tryGet(status_path, status_string, &status_stat)) + { + LOG_INFO(log, "ExportPartition Manifest Updating Task: Failed to read status for {} while enforcing task timeout, skipping", entry_path); + return false; + } + + const auto current_status = magic_enum::enum_cast(status_string); + if (!current_status || *current_status != ExportReplicatedMergeTreePartitionTaskEntry::Status::PENDING) + { + LOG_INFO(log, "ExportPartition Manifest Updating Task: Task {} is not PENDING, can't set to KILLED, skipping", entry_path); + return false; + } + + const auto timeout_message = fmt::format( + "Export partition task timed out: exceeded export_merge_tree_partition_task_timeout_seconds={} (created at {}, now {})", + metadata.task_timeout_seconds, metadata.create_time, now); + + const auto killed_name = String(magic_enum::enum_name(ExportReplicatedMergeTreePartitionTaskEntry::Status::KILLED)); + + Coordination::Requests ops; + ExportPartitionUtils::appendExceptionOps( + ops, zk, fs::path(entry_path), storage.getReplicaName(), + /*part_name=*/"", timeout_message, log); + + ops.emplace_back(zkutil::makeSetRequest(status_path, killed_name, status_stat.version)); + + ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRequests); + ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperMulti); + + Coordination::Responses responses; + const auto rc = zk->tryMulti(ops, responses); + + if (rc == Coordination::Error::ZOK) + { + LOG_WARNING(log, + "ExportPartition Manifest Updating Task: task {} exceeded task_timeout_seconds={}s, " + "transitioned PENDING -> KILLED (atomic with exception record)", + entry_path, metadata.task_timeout_seconds); + } + else + { + /// ZBADVERSION (status changed), ZNODEEXISTS (lazy-create race with the scheduler), + /// counter race, or ZNONODE (entry concurrently removed). In all cases the batch + /// was rolled back atomically and the task will be re-evaluated on the next poll. + LOG_INFO(log, + "ExportPartition Manifest Updating Task: atomic kill for {} failed (rc={}); " + "status was concurrently updated or a ZK op conflicted, will retry on next poll", + entry_path, rc); + } + + /// Return false so the entry remains in entries_by_key; the status watch will drive + /// handleStatusChanges -> killExportPart on every replica, mirroring user-initiated KILL. + return false; + } + else if (is_pending) + { + auto context = ExportPartitionUtils::getContextCopyWithTaskSettings(storage_context, metadata); + + ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRequests); + ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperGetChildren); + std::vector parts_in_processing_or_pending; + if (Coordination::Error::ZOK != zk->tryGetChildren(fs::path(entry_path) / "processing", parts_in_processing_or_pending)) + { + + LOG_INFO(log, "ExportPartition Manifest Updating Task: Failed to get parts in processing or pending, skipping"); + return false; + } + + if (parts_in_processing_or_pending.empty()) + { + LOG_INFO(log, "ExportPartition Manifest Updating Task: Cleanup found PENDING for {} with all parts exported, try to fix it by committing the export", entry_path); + + const auto destination_storage_id = StorageID(QualifiedTableName {metadata.destination_database, metadata.destination_table}); + const auto destination_storage = DatabaseCatalog::instance().tryGetTable(destination_storage_id, context); + if (!destination_storage) + { + LOG_INFO(log, "ExportPartition Manifest Updating Task: Failed to reconstruct destination storage: {}, skipping", destination_storage_id.getNameForLogs()); + return false; + } + + /// it sounds like a replica exported the last part, but was not able to commit the export. Try to fix it + try + { + ExportPartitionUtils::commit(metadata, destination_storage, zk, log, entry_path, context, storage); + } + catch (const Exception & e) + { + LOG_WARNING(log, + "ExportPartition Manifest Updating Task: " + "Caught exception while committing export for {}: {}", + entry_path, e.message()); + + /// Bump commit-attempts counter; transition to FAILED once the budget is exhausted. + /// This is the primary retry path for the commit phase — handlePartExportSuccess + /// only fires once (on the last part's completion); subsequent retries come from here. + const bool became_failed = ExportPartitionUtils::handleCommitFailure( + zk, + entry_path, + metadata.max_retries, + log); + + if (became_failed) + { + LOG_WARNING(log, + "ExportPartition Manifest Updating Task: " + "Commit for {} transitioned to FAILED after exhausting max_retries={}", + entry_path, metadata.max_retries); + } + + /// Return false so the next poll re-enters the cleanup path: + /// - if FAILED: status != PENDING on re-read, cleanup is a no-op + /// until the entry expires (handled by the first tryCleanup branch). + /// - if still PENDING: next poll increments the counter again. + return false; + } + + return true; + } + } + + return false; + } +} + +ExportPartitionManifestUpdatingTask::ExportPartitionManifestUpdatingTask(StorageReplicatedMergeTree & storage_) + : storage(storage_) +{ +} + +std::vector ExportPartitionManifestUpdatingTask::getPartitionExportsInfo() const +{ + std::vector infos; + const auto zk = storage.getZooKeeper(); + + const auto exports_path = fs::path(storage.zookeeper_path) / "exports"; + + ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRequests); + ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperGetChildren); + + std::vector children; + if (Coordination::Error::ZOK != zk->tryGetChildren(exports_path, children)) + { + LOG_INFO(storage.log, "Failed to get children from exports path, returning empty export info list"); + return infos; + } + + if (children.empty()) + return infos; + + /// Batch all metadata.json, status gets, and getChildren operations in a single multi request + Coordination::Requests requests; + requests.reserve(children.size() * 4); // metadata, status, processing, exceptions_per_replica + + // Track response indices for each child + struct ChildResponseIndices + { + size_t metadata_idx; + size_t status_idx; + size_t processing_idx; + size_t exceptions_per_replica_idx; + }; + std::vector response_indices; + response_indices.reserve(children.size()); + + for (const auto & child : children) + { + const auto export_partition_path = fs::path(exports_path) / child; + + ChildResponseIndices indices; + indices.metadata_idx = requests.size(); + requests.push_back(zkutil::makeGetRequest(export_partition_path / "metadata.json")); + + indices.status_idx = requests.size(); + requests.push_back(zkutil::makeGetRequest(export_partition_path / "status")); + + indices.processing_idx = requests.size(); + requests.push_back(zkutil::makeListRequest(export_partition_path / "processing")); + + indices.exceptions_per_replica_idx = requests.size(); + requests.push_back(zkutil::makeListRequest(export_partition_path / "exceptions_per_replica")); + + response_indices.push_back(indices); + } + + ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRequests); + ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperMulti); + + Coordination::Responses responses; + Coordination::Error code = zk->tryMulti(requests, responses); + + if (code != Coordination::Error::ZOK) + { + LOG_INFO(storage.log, "Failed to execute multi request for export partition info, error: {}", code); + return infos; + } + + // Helper to extract GetResponse data + auto getGetResponseData = [&responses](size_t idx) -> std::pair + { + if (idx >= responses.size()) + return {Coordination::Error::ZRUNTIMEINCONSISTENCY, ""}; + + const auto * get_response = dynamic_cast(responses[idx].get()); + if (!get_response) + return {Coordination::Error::ZRUNTIMEINCONSISTENCY, ""}; + + return {get_response->error, get_response->data}; + }; + + // Helper to extract ListResponse data + auto getListResponseData = [&responses](size_t idx) -> std::pair + { + if (idx >= responses.size()) + return {Coordination::Error::ZRUNTIMEINCONSISTENCY, Strings{}}; + + const auto * list_response = dynamic_cast(responses[idx].get()); + if (!list_response) + return {Coordination::Error::ZRUNTIMEINCONSISTENCY, Strings{}}; + + return {list_response->error, list_response->names}; + }; + + // Create response wrappers matching the MultiTryGetResponse/MultiTryGetChildrenResponse interface + struct ResponseWrapper + { + Coordination::Error error; + std::string data; + Strings names; + + ResponseWrapper(Coordination::Error err, const std::string & d, const Strings & n) + : error(err), data(d), names(n) {} + }; + + std::vector metadata_responses_wrapper; + std::vector status_responses_wrapper; + std::vector processing_responses_wrapper; + std::vector exceptions_per_replica_responses_wrapper; + + metadata_responses_wrapper.reserve(children.size()); + status_responses_wrapper.reserve(children.size()); + processing_responses_wrapper.reserve(children.size()); + exceptions_per_replica_responses_wrapper.reserve(children.size()); + + for (size_t child_idx = 0; child_idx < children.size(); ++child_idx) + { + const auto & indices = response_indices[child_idx]; + + // Extract metadata response + auto [metadata_error, metadata_data] = getGetResponseData(indices.metadata_idx); + metadata_responses_wrapper.emplace_back(metadata_error, metadata_data, Strings{}); + + // Extract status response + auto [status_error, status_data] = getGetResponseData(indices.status_idx); + status_responses_wrapper.emplace_back(status_error, status_data, Strings{}); + + // Extract processing response + auto [processing_error, processing_names] = getListResponseData(indices.processing_idx); + processing_responses_wrapper.emplace_back(processing_error, "", processing_names); + + // Extract exceptions_per_replica response + auto [exceptions_error, exceptions_names] = getListResponseData(indices.exceptions_per_replica_idx); + exceptions_per_replica_responses_wrapper.emplace_back(exceptions_error, "", exceptions_names); + } + + // Use wrapper vectors directly - they match the interface expected by the code below + auto & metadata_responses = metadata_responses_wrapper; + auto & status_responses = status_responses_wrapper; + auto & processing_responses = processing_responses_wrapper; + auto & exceptions_per_replica_responses = exceptions_per_replica_responses_wrapper; + + /// Collect all exception replica paths for batching + struct ExceptionReplicaPath + { + size_t child_idx; + std::string replica; + std::string count_path; + std::string exception_path; + std::string part_path; + }; + + std::vector exception_replica_paths; + for (size_t child_idx = 0; child_idx < children.size(); ++child_idx) + { + const auto & child = children[child_idx]; + const auto export_partition_path = fs::path(exports_path) / child; + /// Check if we got valid responses + if (metadata_responses[child_idx].error != Coordination::Error::ZOK) + { + LOG_INFO(storage.log, "Skipping {}: missing metadata.json", child); + continue; + } + if (status_responses[child_idx].error != Coordination::Error::ZOK) + { + LOG_INFO(storage.log, "Skipping {}: missing status", child); + continue; + } + if (processing_responses[child_idx].error != Coordination::Error::ZOK) + { + LOG_INFO(storage.log, "Skipping {}: missing processing parts", child); + continue; + } + if (exceptions_per_replica_responses[child_idx].error != Coordination::Error::ZOK) + { + LOG_INFO(storage.log, "Skipping {}: missing exceptions_per_replica", export_partition_path); + continue; + } + const auto exceptions_per_replica_path = export_partition_path / "exceptions_per_replica"; + const auto & exception_replicas = exceptions_per_replica_responses[child_idx].names; + for (const auto & replica : exception_replicas) + { + const auto last_exception_path = exceptions_per_replica_path / replica / "last_exception"; + exception_replica_paths.push_back({ + child_idx, + replica, + (exceptions_per_replica_path / replica / "count").string(), + (last_exception_path / "exception").string(), + (last_exception_path / "part").string() + }); + } + } + /// Batch get all exception data in a single multi request + std::map>> exception_data_by_child; + + if (!exception_replica_paths.empty()) + { + Coordination::Requests exception_requests; + exception_requests.reserve(exception_replica_paths.size() * 3); // count, exception, part for each + + // Track response indices for each exception replica path + struct ExceptionResponseIndices + { + size_t count_idx; + size_t exception_idx; + size_t part_idx; + }; + std::vector exception_response_indices; + exception_response_indices.reserve(exception_replica_paths.size()); + + for (const auto & erp : exception_replica_paths) + { + ExceptionResponseIndices indices; + indices.count_idx = exception_requests.size(); + exception_requests.push_back(zkutil::makeGetRequest(erp.count_path)); + + indices.exception_idx = exception_requests.size(); + exception_requests.push_back(zkutil::makeGetRequest(erp.exception_path)); + + indices.part_idx = exception_requests.size(); + exception_requests.push_back(zkutil::makeGetRequest(erp.part_path)); + + exception_response_indices.push_back(indices); + } + + // Execute single multi request for all exception data + ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRequests); + ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperMulti); + + Coordination::Responses exception_responses; + Coordination::Error exception_code = zk->tryMulti(exception_requests, exception_responses); + + if (exception_code != Coordination::Error::ZOK) + { + LOG_INFO(storage.log, "Failed to execute multi request for exception data, error: {}", exception_code); + } + else + { + // Parse exception responses + for (size_t exception_path_idx = 0; exception_path_idx < exception_replica_paths.size(); ++exception_path_idx) + { + const auto & erp = exception_replica_paths[exception_path_idx]; + const auto & indices = exception_response_indices[exception_path_idx]; + + std::string count_str; + std::string exception_str; + std::string part_str; + + // Extract count response + if (indices.count_idx < exception_responses.size()) + { + const auto * count_response = dynamic_cast(exception_responses[indices.count_idx].get()); + if (count_response && count_response->error == Coordination::Error::ZOK) + count_str = count_response->data; + } + + // Extract exception response + if (indices.exception_idx < exception_responses.size()) + { + const auto * exception_response = dynamic_cast(exception_responses[indices.exception_idx].get()); + if (exception_response && exception_response->error == Coordination::Error::ZOK) + exception_str = exception_response->data; + } + + // Extract part response + if (indices.part_idx < exception_responses.size()) + { + const auto * part_response = dynamic_cast(exception_responses[indices.part_idx].get()); + if (part_response && part_response->error == Coordination::Error::ZOK) + part_str = part_response->data; + } + + exception_data_by_child[erp.child_idx].emplace_back(erp.replica, count_str, exception_str, part_str); + } + } + } + + /// Build the result + for (size_t child_idx = 0; child_idx < children.size(); ++child_idx) + { + /// Skip if we already determined this child is invalid + if (metadata_responses[child_idx].error != Coordination::Error::ZOK + || status_responses[child_idx].error != Coordination::Error::ZOK + || processing_responses[child_idx].error != Coordination::Error::ZOK + || exceptions_per_replica_responses[child_idx].error != Coordination::Error::ZOK) + { + continue; + } + + ReplicatedPartitionExportInfo info; + const auto metadata_json = metadata_responses[child_idx].data; + const auto status = status_responses[child_idx].data; + const auto processing_parts = processing_responses[child_idx].names; + const auto parts_to_do = processing_parts.size(); + std::string exception_replica; + std::string last_exception; + std::string exception_part; + std::size_t exception_count = 0; + /// Process exception data + auto exception_data_it = exception_data_by_child.find(child_idx); + if (exception_data_it != exception_data_by_child.end()) + { + for (const auto & [replica, count_str, exception_str, part_str] : exception_data_it->second) + { + if (!count_str.empty()) + { + exception_count += parse(count_str); + } + if (last_exception.empty() && !exception_str.empty() && !part_str.empty()) + { + exception_replica = replica; + last_exception = exception_str; + exception_part = part_str; + } + } + } + + const auto metadata = ExportReplicatedMergeTreePartitionManifest::fromJsonString(metadata_json); + + info.destination_database = metadata.destination_database; + info.destination_table = metadata.destination_table; + info.partition_id = metadata.partition_id; + info.transaction_id = metadata.transaction_id; + info.query_id = metadata.query_id; + info.create_time = metadata.create_time; + info.source_replica = metadata.source_replica; + info.parts_count = metadata.number_of_parts; + info.parts_to_do = parts_to_do; + info.parts = metadata.parts; + info.status = status; + info.exception_replica = exception_replica; + info.last_exception = last_exception; + info.exception_part = exception_part; + info.exception_count = exception_count; + infos.emplace_back(std::move(info)); + } + + return infos; +} + +std::vector ExportPartitionManifestUpdatingTask::getPartitionExportsInfoLocal() const +{ + std::lock_guard lock(storage.export_merge_tree_partition_mutex); + + std::vector infos; + + for (const auto & entry : storage.export_merge_tree_partition_task_entries_by_key) + { + ReplicatedPartitionExportInfo info; + + info.destination_database = entry.manifest.destination_database; + info.destination_table = entry.manifest.destination_table; + info.partition_id = entry.manifest.partition_id; + info.transaction_id = entry.manifest.transaction_id; + info.query_id = entry.manifest.query_id; + info.create_time = entry.manifest.create_time; + info.source_replica = entry.manifest.source_replica; + info.parts_count = entry.manifest.number_of_parts; + info.parts_to_do = entry.manifest.parts.size(); + info.parts = entry.manifest.parts; + info.status = magic_enum::enum_name(entry.status); + + infos.emplace_back(std::move(info)); + } + + return infos; +} + +void ExportPartitionManifestUpdatingTask::poll() +{ + std::lock_guard lock(storage.export_merge_tree_partition_mutex); + + LOG_INFO(storage.log, "ExportPartition Manifest Updating Task: Polling for new entries for table {}. Current number of entries: {}", storage.getStorageID().getNameForLogs(), storage.export_merge_tree_partition_task_entries_by_key.size()); + + auto zk = storage.getZooKeeper(); + + const std::string exports_path = fs::path(storage.zookeeper_path) / "exports"; + const std::string cleanup_lock_path = fs::path(storage.zookeeper_path) / "exports_cleanup_lock"; + + auto cleanup_lock = zkutil::EphemeralNodeHolder::tryCreate(cleanup_lock_path, *zk, storage.replica_name); + if (cleanup_lock) + { + LOG_INFO(storage.log, "ExportPartition Manifest Updating Task: Cleanup lock acquired, will remove stale entries"); + } + + ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRequests); + ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperGetChildrenWatch); + + Coordination::Stat stat; + const auto children = zk->getChildrenWatch(exports_path, &stat, storage.export_merge_tree_partition_watch_callback); + const std::unordered_set zk_children(children.begin(), children.end()); + + const auto now = time(nullptr); + + auto & entries_by_key = storage.export_merge_tree_partition_task_entries_by_key; + + /// Load new entries + /// If we have the cleanup lock, also remove stale entries from zk and local + /// Upload dangling commit files if any + for (const auto & key : zk_children) + { + const std::string entry_path = fs::path(exports_path) / key; + + ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRequests); + ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperGet); + std::string metadata_json; + if (!zk->tryGet(fs::path(entry_path) / "metadata.json", metadata_json)) + { + LOG_INFO(storage.log, "ExportPartition Manifest Updating Task: Skipping {}: missing metadata.json", key); + continue; + } + + const auto metadata = ExportReplicatedMergeTreePartitionManifest::fromJsonString(metadata_json); + + const auto local_entry = entries_by_key.find(key); + + /// If the zk entry has been replaced with export_merge_tree_partition_force_export, checking only for the export key is not enough + /// we need to make sure it is the same transaction id. If it is not, it needs to be replaced. + bool has_local_entry_and_is_up_to_date = local_entry != entries_by_key.end() + && local_entry->manifest.transaction_id == metadata.transaction_id; + + /// If the entry is up to date and we don't have the cleanup lock, early exit, nothing to be done. + if (!cleanup_lock && has_local_entry_and_is_up_to_date) + continue; + + std::weak_ptr weak_manifest_updater = storage.export_merge_tree_partition_manifest_updater; + + auto status_watch_callback = std::make_shared([weak_manifest_updater, key](const Coordination::WatchResponse &) + { + /// If the table is dropped but the watch is not removed, we need to prevent use after free + /// below code assumes that if manifest updater is still alive, the status handling task is also alive + if (auto manifest_updater = weak_manifest_updater.lock()) + { + manifest_updater->addStatusChange(key); + manifest_updater->storage.export_merge_tree_partition_status_handling_task->schedule(); + } + }); + + ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRequests); + ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperGetWatch); + std::string status_string; + if (!zk->tryGetWatch(fs::path(entry_path) / "status", status_string, nullptr, status_watch_callback)) + { + LOG_INFO(storage.log, "ExportPartition Manifest Updating Task: Skipping {}: missing status", key); + continue; + } + + const auto status = magic_enum::enum_cast(status_string); + if (!status) + { + LOG_INFO(storage.log, "ExportPartition Manifest Updating Task: Invalid status {} for task {}, skipping", status_string, key); + continue; + } + + /// if we have the cleanup lock, try to cleanup + /// if we successfully cleaned it up, early exit + if (cleanup_lock) + { + bool cleanup_successful = tryCleanup( + zk, + entry_path, + storage.log.load(), + storage.getContext(), + storage, + key, + metadata, + now, + *status == ExportReplicatedMergeTreePartitionTaskEntry::Status::PENDING, + entries_by_key); + + if (cleanup_successful) + continue; + } + + if (has_local_entry_and_is_up_to_date) + { + LOG_INFO(storage.log, "ExportPartition Manifest Updating Task: Skipping {}: already exists", key); + continue; + } + + addTask(metadata, *status, key, entries_by_key); + } + + /// Remove entries that were deleted by someone else + removeStaleEntries(zk_children, entries_by_key); + + LOG_INFO(storage.log, "ExportPartition Manifest Updating task: finished polling for new entries. Number of entries: {}", entries_by_key.size()); + + storage.export_merge_tree_partition_select_task->schedule(); +} + +void ExportPartitionManifestUpdatingTask::addTask( + const ExportReplicatedMergeTreePartitionManifest & metadata, + ExportReplicatedMergeTreePartitionTaskEntry::Status status, + const std::string & key, + auto & entries_by_key +) +{ + std::vector part_references; + + /// If the status is PENDING, we grab references to the data parts to prevent them from being deleted from the disk + /// Otherwise, the operation has already been completed and there is no need to keep the data parts alive + /// You might also ask: why bother adding tasks that have already been completed (i.e, status != PENDING)? + /// The reason is the `replicated_partition_exports` table in the local only mode might miss entries if they are not added here. + if (status == ExportReplicatedMergeTreePartitionTaskEntry::Status::PENDING) + { + for (const auto & part_name : metadata.parts) + { + if (const auto part = storage.getPartIfExists(part_name, {MergeTreeDataPartState::Active, MergeTreeDataPartState::Outdated})) + { + part_references.push_back(part); + } + } + } + + /// Insert or update entry. The multi_index container automatically maintains both indexes. + auto entry = ExportReplicatedMergeTreePartitionTaskEntry {metadata, status, std::move(part_references)}; + auto it = entries_by_key.find(key); + if (it != entries_by_key.end()) + entries_by_key.replace(it, entry); + else + entries_by_key.insert(entry); +} + +void ExportPartitionManifestUpdatingTask::removeStaleEntries( + const std::unordered_set & zk_children, + auto & entries_by_key +) +{ + for (auto it = entries_by_key.begin(); it != entries_by_key.end();) + { + const auto & key = it->getCompositeKey(); + if (zk_children.contains(key)) + { + ++it; + continue; + } + + const auto & transaction_id = it->manifest.transaction_id; + LOG_INFO(storage.log, "ExportPartition Manifest Updating Task: Export task {} was deleted, calling killExportPartition for transaction {}", key, transaction_id); + + try + { + storage.killExportPart(transaction_id); + } + catch (...) + { + tryLogCurrentException(storage.log, __PRETTY_FUNCTION__); + } + + it = entries_by_key.erase(it); + } +} + +void ExportPartitionManifestUpdatingTask::addStatusChange(const std::string & key) +{ + std::lock_guard lock(status_changes_mutex); + status_changes.emplace(key); +} + +void ExportPartitionManifestUpdatingTask::handleStatusChanges() +{ + /// copy the events to a local queue to avoid holding the status_changes_mutex while also holding export_merge_tree_partition_mutex + std::queue local_status_changes; + { + std::lock_guard lock(status_changes_mutex); + std::swap(status_changes, local_status_changes); + } + + try + { + std::lock_guard task_entries_lock(storage.export_merge_tree_partition_mutex); + auto zk = storage.getZooKeeper(); + + LOG_INFO(storage.log, "ExportPartition Manifest Updating task: handling status changes. Number of status changes: {}", local_status_changes.size()); + + while (!local_status_changes.empty()) + { + const auto & key = local_status_changes.front(); + LOG_INFO(storage.log, "ExportPartition Manifest Updating task: handling status change for task {}", key); + + fiu_do_on(FailPoints::export_partition_status_change_throw, + { + throw Exception(ErrorCodes::FAULT_INJECTED, + "Failpoint: simulating exception during status change handling for key {}", key); + }); + + auto it = storage.export_merge_tree_partition_task_entries_by_key.find(key); + if (it == storage.export_merge_tree_partition_task_entries_by_key.end()) + { + local_status_changes.pop(); + continue; + } + + ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRequests); + ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperGet); + /// get new status from zk + std::string new_status_string; + if (!zk->tryGet(fs::path(storage.zookeeper_path) / "exports" / key / "status", new_status_string)) + { + LOG_INFO(storage.log, "ExportPartition Manifest Updating Task: Failed to get new status for task {}, skipping", key); + local_status_changes.pop(); + continue; + } + + const auto new_status = magic_enum::enum_cast(new_status_string); + if (!new_status) + { + LOG_INFO(storage.log, "ExportPartition Manifest Updating Task: Invalid status {} for task {}, skipping", new_status_string, key); + local_status_changes.pop(); + continue; + } + + LOG_INFO(storage.log, "ExportPartition Manifest Updating task: status changed for task {}. New status: {}", key, magic_enum::enum_name(*new_status).data()); + + /// If status changed to KILLED, cancel local export operations + if (*new_status == ExportReplicatedMergeTreePartitionTaskEntry::Status::KILLED) + { + try + { + LOG_INFO(storage.log, "ExportPartition Manifest Updating task: killing export partition for task {}", key); + storage.killExportPart(it->manifest.transaction_id); + } + catch (...) + { + tryLogCurrentException(storage.log, __PRETTY_FUNCTION__); + } + } + + it->status = *new_status; + + if (it->status != ExportReplicatedMergeTreePartitionTaskEntry::Status::PENDING) + { + /// we no longer need to keep the data parts alive + it->part_references.clear(); + } + + local_status_changes.pop(); + } + } + catch (...) + { + tryLogCurrentException(storage.log, __PRETTY_FUNCTION__); + + LOG_INFO(storage.log, "ExportPartition Manifest Updating task: exception thrown while handling status changes, enqueuing remaining status changes back to the status_changes queue. Number of remaining status changes: {}", local_status_changes.size()); + + std::lock_guard lock(status_changes_mutex); + + /// It is possible that an exception is thrown while handling the status. In this scenario + /// we need to enqueue the remaining status changes back to the status_changes queue not to lose them. + /// The other solution to this problem would be to ignore it and schedule a poll - maybe it is simpler? + if (!local_status_changes.empty()) + { + // Prepend remaining items before any newly-arrived items + while (!status_changes.empty()) + { + local_status_changes.push(std::move(status_changes.front())); + status_changes.pop(); + } + + std::swap(status_changes, local_status_changes); + } + + LOG_INFO(storage.log, "ExportPartition Manifest Updating task: The new number of pending status after enqueueing unprocessed ones is {}", status_changes.size()); + + throw; + } +} + +} diff --git a/src/Storages/MergeTree/ExportPartitionManifestUpdatingTask.h b/src/Storages/MergeTree/ExportPartitionManifestUpdatingTask.h new file mode 100644 index 000000000000..855ecc334c09 --- /dev/null +++ b/src/Storages/MergeTree/ExportPartitionManifestUpdatingTask.h @@ -0,0 +1,49 @@ +#pragma once + +#include +#include +#include +#include +#include +#include +namespace DB +{ + +class StorageReplicatedMergeTree; +struct ExportReplicatedMergeTreePartitionManifest; + +class ExportPartitionManifestUpdatingTask +{ +public: + ExportPartitionManifestUpdatingTask(StorageReplicatedMergeTree & storage); + + void poll(); + + void handleStatusChanges(); + + void addStatusChange(const std::string & key); + + std::vector getPartitionExportsInfo() const; + + std::vector getPartitionExportsInfoLocal() const; + +private: + StorageReplicatedMergeTree & storage; + + void addTask( + const ExportReplicatedMergeTreePartitionManifest & metadata, + ExportReplicatedMergeTreePartitionTaskEntry::Status status, + const std::string & key, + auto & entries_by_key + ); + + void removeStaleEntries( + const std::unordered_set & zk_children, + auto & entries_by_key + ); + + std::mutex status_changes_mutex; + std::queue status_changes; +}; + +} diff --git a/src/Storages/MergeTree/ExportPartitionTaskScheduler.cpp b/src/Storages/MergeTree/ExportPartitionTaskScheduler.cpp new file mode 100644 index 000000000000..a77e2894b4a1 --- /dev/null +++ b/src/Storages/MergeTree/ExportPartitionTaskScheduler.cpp @@ -0,0 +1,549 @@ +#include +#include +#include +#include +#include +#include +#include +#include +#include "Storages/MergeTree/ExportPartitionUtils.h" +#include "Storages/MergeTree/MergeTreePartExportManifest.h" +#include "Storages/MergeTree/ExportPartFromPartitionExportTask.h" +#include "Formats/FormatFactory.h" +#include + +namespace ProfileEvents +{ + extern const Event ExportPartitionZooKeeperRequests; + extern const Event ExportPartitionZooKeeperGet; + extern const Event ExportPartitionZooKeeperGetChildren; + extern const Event ExportPartitionZooKeeperCreate; + extern const Event ExportPartitionZooKeeperSet; + extern const Event ExportPartitionZooKeeperRemove; + extern const Event ExportPartitionZooKeeperMulti; +} + + +namespace DB +{ + +namespace Setting +{ + extern const SettingsMergeTreePartExportFileAlreadyExistsPolicy export_merge_tree_part_file_already_exists_policy; +} + +namespace ErrorCodes +{ + extern const int QUERY_WAS_CANCELLED; + extern const int LOGICAL_ERROR; +} + +ExportPartitionTaskScheduler::ExportPartitionTaskScheduler(StorageReplicatedMergeTree & storage_) + : storage(storage_) +{ +} + +void ExportPartitionTaskScheduler::run() +{ + const auto available_move_executors = storage.background_moves_assignee.getAvailableMoveExecutors(); + + /// this is subject to TOCTOU - but for now we choose to live with it. + if (available_move_executors == 0) + { + LOG_INFO(storage.log, "ExportPartition scheduler task: No available move executors, skipping"); + return; + } + + LOG_INFO(storage.log, "ExportPartition scheduler task: Available move executors: {}", available_move_executors); + + std::size_t scheduled_exports_count = 0; + + const uint32_t seed = uint32_t(std::hash{}(storage.replica_name)) ^ uint32_t(scheduled_exports_count); + pcg64_fast rng(seed); + + std::lock_guard lock(storage.export_merge_tree_partition_mutex); + + auto zk = storage.getZooKeeper(); + + // Iterate sorted by create_time + for (auto & entry : storage.export_merge_tree_partition_task_entries_by_create_time) + { + if (scheduled_exports_count >= available_move_executors) + { + LOG_INFO(storage.log, "ExportPartition scheduler task: Scheduled exports count is greater than available move executors, skipping"); + break; + } + + const auto & manifest = entry.manifest; + const auto key = entry.getCompositeKey(); + const auto database = storage.getContext()->resolveDatabase(manifest.destination_database); + const auto & table = manifest.destination_table; + + /// No need to query zk for status if the local one is not PENDING + if (entry.status != ExportReplicatedMergeTreePartitionTaskEntry::Status::PENDING) + { + LOG_INFO(storage.log, "ExportPartition scheduler task: Skipping... Local status is {}", magic_enum::enum_name(entry.status).data()); + continue; + } + + const auto destination_storage_id = StorageID(QualifiedTableName {database, table}); + + const auto destination_storage = DatabaseCatalog::instance().tryGetTable(destination_storage_id, storage.getContext()); + + if (!destination_storage) + { + LOG_INFO(storage.log, "ExportPartition scheduler task: Failed to reconstruct destination storage: {}, skipping", destination_storage_id.getNameForLogs()); + continue; + } + + ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRequests); + ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperGet); + std::string status_in_zk_string; + if (!zk->tryGet(fs::path(storage.zookeeper_path) / "exports" / key / "status", status_in_zk_string)) + { + LOG_INFO(storage.log, "ExportPartition scheduler task: Failed to get status, skipping"); + continue; + } + + const auto status_in_zk = magic_enum::enum_cast(status_in_zk_string); + + if (!status_in_zk) + { + LOG_INFO(storage.log, "ExportPartition scheduler task: Failed to get status from zk, skipping"); + continue; + } + + if (status_in_zk.value() != ExportReplicatedMergeTreePartitionTaskEntry::Status::PENDING) + { + entry.status = status_in_zk.value(); + LOG_INFO(storage.log, "ExportPartition scheduler task: Skipping {}... Status from zk is {}", key, magic_enum::enum_name(entry.status).data()); + continue; + } + + ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRequests); + ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperGetChildren); + std::vector parts_in_processing_or_pending; + + if (Coordination::Error::ZOK != zk->tryGetChildren(fs::path(storage.zookeeper_path) / "exports" / key / "processing", parts_in_processing_or_pending)) + { + LOG_INFO(storage.log, "ExportPartition scheduler task: Failed to get parts in processing or pending, skipping"); + continue; + } + + + if (parts_in_processing_or_pending.empty()) + { + LOG_INFO(storage.log, "ExportPartition scheduler task: No parts in processing or pending, skipping"); + continue; + } + + /// shuffle the parts to reduce the risk of lock collisions + std::shuffle(parts_in_processing_or_pending.begin(), parts_in_processing_or_pending.end(), rng); + + ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRequests); + ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperGetChildren); + std::vector locked_parts; + + if (Coordination::Error::ZOK != zk->tryGetChildren(fs::path(storage.zookeeper_path) / "exports" / key / "locks", locked_parts)) + { + LOG_INFO(storage.log, "ExportPartition scheduler task: Failed to get locked parts, skipping"); + continue; + } + + std::unordered_set locked_parts_set(locked_parts.begin(), locked_parts.end()); + + for (const auto & zk_part_name : parts_in_processing_or_pending) + { + if (scheduled_exports_count >= available_move_executors) + { + LOG_INFO(storage.log, "ExportPartition scheduler task: Scheduled exports count is greater than available move executors, skipping"); + break; + } + + if (locked_parts_set.contains(zk_part_name)) + { + LOG_INFO(storage.log, "ExportPartition scheduler task: Part {} is locked, skipping", zk_part_name); + continue; + } + + const auto part = storage.getPartIfExists(zk_part_name, {MergeTreeDataPartState::Active, MergeTreeDataPartState::Outdated}); + if (!part) + { + LOG_INFO(storage.log, "ExportPartition scheduler task: Part {} not found locally, skipping", zk_part_name); + continue; + } + + LOG_INFO(storage.log, "ExportPartition scheduler task: Scheduling part export: {}", zk_part_name); + + auto context = ExportPartitionUtils::getContextCopyWithTaskSettings(storage.getContext(), manifest); + + /// todo arthur this code path does not perform all the validations a simple part export does because we are not calling exportPartToTable directly. + /// the schema and everything else has been validated when the export partition task was created, but nothing prevents the destination table from being + /// recreated with a new schema before the export task is scheduled. + if (manifest.lock_inside_the_task) + { + LOG_INFO(storage.log, "ExportPartition scheduler task: Locking part export inside the task"); + std::lock_guard part_export_lock(storage.export_manifests_mutex); + + MergeTreePartExportManifest part_export_manifest( + destination_storage, + part, + manifest.transaction_id, + manifest.query_id, + context->getSettingsRef()[Setting::export_merge_tree_part_file_already_exists_policy].value, + context->getSettingsCopy(), + storage.getInMemoryMetadataPtr(), + manifest.iceberg_metadata_json, + [this, key, zk_part_name, manifest, destination_storage] + (MergeTreePartExportManifest::CompletionCallbackResult result) + { + handlePartExportCompletion(key, zk_part_name, manifest, destination_storage, result); + }); + + part_export_manifest.task = std::make_shared(storage, key, part_export_manifest); + + /// todo arthur this might conflict with the standalone export part. what to do in this case? + if (!storage.export_manifests.emplace(part_export_manifest).second) + { + LOG_INFO(storage.log, "ExportPartition scheduler task: Part {} is already being exported, skipping", zk_part_name); + continue; + } + + if (!storage.background_moves_assignee.scheduleMoveTask(part_export_manifest.task)) + { + storage.export_manifests.erase(part_export_manifest); + LOG_INFO(storage.log, "ExportPartition scheduler task: Failed to schedule export part task, skipping"); + return; + } + + scheduled_exports_count++; + } + else + { + try + { + LOG_INFO(storage.log, "ExportPartition scheduler task: Exporting part to table"); + + LOG_INFO(storage.log, "ExportPartition scheduler task: Attempting to lock part: {}", zk_part_name); + + ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRequests); + ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperCreate); + if (Coordination::Error::ZOK != zk->tryCreate(fs::path(storage.zookeeper_path) / "exports" / key / "locks" / zk_part_name, storage.replica_name, zkutil::CreateMode::Ephemeral)) + { + LOG_INFO(storage.log, "ExportPartition scheduler task: Failed to lock part {}, skipping", zk_part_name); + continue; + } + + LOG_INFO(storage.log, "ExportPartition scheduler task: Locked part: {}", zk_part_name); + + storage.exportPartToTable( + part->name, + destination_storage_id, + manifest.transaction_id, + context, + manifest.iceberg_metadata_json, + /*allow_outdated_parts*/ true, + [this, key, zk_part_name, manifest, destination_storage] + (MergeTreePartExportManifest::CompletionCallbackResult result) + { + handlePartExportCompletion(key, zk_part_name, manifest, destination_storage, result); + }); + + scheduled_exports_count++; + } + catch (const Exception &) + { + tryLogCurrentException(__PRETTY_FUNCTION__); + ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRequests); + ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRemove); + zk->tryRemove(fs::path(storage.zookeeper_path) / "exports" / key / "locks" / zk_part_name); + /// we should not increment retry_count because the node might just be full + } + } + + } + } +} + +void ExportPartitionTaskScheduler::handlePartExportCompletion( + const std::string & export_key, + const std::string & part_name, + const ExportReplicatedMergeTreePartitionManifest & manifest, + const StoragePtr & destination_storage, + const MergeTreePartExportManifest::CompletionCallbackResult & result) +{ + /// Invoked from MergeTreeBackgroundExecutor threads, so the component is not inherited from selectPartsToExport. + auto component_guard = Coordination::setCurrentComponent("ExportPartitionTaskScheduler::handlePartExportCompletion"); + + const auto export_path = fs::path(storage.zookeeper_path) / "exports" / export_key; + const auto processing_parts_path = export_path / "processing"; + const auto processed_part_path = export_path / "processed" / part_name; + const auto zk = storage.getZooKeeper(); + + if (result.success) + { + handlePartExportSuccess(manifest, destination_storage, processing_parts_path, processed_part_path, part_name, export_path, zk, result.relative_paths_in_destination_storage); + } + else + { + handlePartExportFailure(processing_parts_path, part_name, export_path, zk, result.exception, manifest.max_retries); + } +} + +void ExportPartitionTaskScheduler::handlePartExportSuccess( + const ExportReplicatedMergeTreePartitionManifest & manifest, + const StoragePtr & destination_storage, + const std::filesystem::path & processing_parts_path, + const std::filesystem::path & processed_part_path, + const std::string & part_name, + const std::filesystem::path & export_path, + const zkutil::ZooKeeperPtr & zk, + const std::vector & relative_paths_in_destination_storage +) +{ + LOG_INFO(storage.log, "ExportPartition scheduler task: Part {} exported successfully, paths size: {}", part_name, relative_paths_in_destination_storage.size()); + + for (const auto & relative_path_in_destination_storage : relative_paths_in_destination_storage) + { + LOG_INFO(storage.log, "ExportPartition scheduler task: {}", relative_path_in_destination_storage); + } + + if (!tryToMovePartToProcessed(export_path, processing_parts_path, processed_part_path, part_name, relative_paths_in_destination_storage, zk)) + { + LOG_INFO(storage.log, "ExportPartition scheduler task: Failed to move part to processed, will not commit export partition"); + return; + } + + LOG_INFO(storage.log, "ExportPartition scheduler task: Marked part export {} as completed", part_name); + + if (!areAllPartsProcessed(export_path, zk)) + { + return; + } + + LOG_INFO(storage.log, "ExportPartition scheduler task: All parts are processed, will try to commit export partition"); + + try + { + auto context = ExportPartitionUtils::getContextCopyWithTaskSettings(storage.getContext(), manifest); + ExportPartitionUtils::commit(manifest, destination_storage, zk, storage.log.load(), export_path, context, storage); + } + catch (const Exception & e) + { + LOG_INFO(storage.log, "ExportPartition scheduler task: Caught exception while committing export partition, {}", e.message()); + + /// Bump commit-attempts counter; transition to FAILED once the budget is exhausted. + /// Prevents the task from remaining stuck in PENDING if commit() fails persistently + /// (e.g. schema/spec mismatch, prolonged destination outage). + const bool became_failed = ExportPartitionUtils::handleCommitFailure( + zk, + export_path, + manifest.max_retries, + storage.log.load()); + + if (became_failed) + { + LOG_WARNING(storage.log, + "ExportPartition scheduler task: Commit for {} transitioned to FAILED after exhausting max_retries={}", + export_path.string(), manifest.max_retries); + } + } +} + +void ExportPartitionTaskScheduler::handlePartExportFailure( + const std::filesystem::path & processing_parts_path, + const std::string & part_name, + const std::filesystem::path & export_path, + const zkutil::ZooKeeperPtr & zk, + const std::optional & exception, + size_t max_retries +) +{ + LOG_INFO(storage.log, "ExportPartition scheduler task: Part {} export failed", part_name); + + if (!exception) + { + throw Exception(ErrorCodes::LOGICAL_ERROR, "ExportPartition scheduler task: No exception provided for error handling. Sounds like a bug"); + } + + Coordination::Stat locked_by_stat; + std::string locked_by; + + ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRequests); + ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperGet); + if (!zk->tryGet(export_path / "locks" / part_name, locked_by, &locked_by_stat)) + { + LOG_INFO(storage.log, "ExportPartition scheduler task: Part {} is not locked by any replica, will not increment error counts", part_name); + return; + } + + if (locked_by != storage.replica_name) + { + LOG_INFO(storage.log, "ExportPartition scheduler task: Part {} is locked by another replica, will not increment error counts", part_name); + return; + } + + /// Early exit if the query was cancelled - no need to increment error counts + if (exception->code() == ErrorCodes::QUERY_WAS_CANCELLED) + { + /// Releasing the lock is important because a query can be cancelled due to SYSTEM STOP MOVES. If this is the case, + /// other replicas should still be able to export this individual part. That's why there is a retry loop here. + /// It is very unlikely this will be a problem in practice. The lock is ephemeral, which means it is automatically released + /// if ClickHouse loses connection to ZooKeeper + std::size_t retry_count = 0; + static constexpr std::size_t max_lock_release_retries = 3; + while (retry_count < max_lock_release_retries) + { + ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRequests); + ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRemove); + + const auto removal_code = zk->tryRemove(export_path / "locks" / part_name, locked_by_stat.version); + + if (Coordination::Error::ZOK == removal_code) + { + break; + } + + if (Coordination::Error::ZBADVERSION == removal_code) + { + LOG_INFO(storage.log, "ExportPartition scheduler task: Part {} lock version mismatch, will not increment error counts", part_name); + break; + } + + retry_count++; + } + + LOG_INFO(storage.log, "ExportPartition scheduler task: Part {} export was cancelled, skipping error handling", part_name); + return; + } + + Coordination::Requests ops; + + const auto processing_part_path = processing_parts_path / part_name; + + ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRequests); + ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperGet); + std::string processing_part_string; + + if (!zk->tryGet(processing_part_path, processing_part_string)) + { + LOG_INFO(storage.log, "ExportPartition scheduler task: Failed to get processing part, will not increment error counts"); + return; + } + + /// todo arthur could this have been cached? + auto processing_part_entry = ExportReplicatedMergeTreePartitionProcessingPartEntry::fromJsonString(processing_part_string); + + processing_part_entry.retry_count++; + + ops.emplace_back(zkutil::makeRemoveRequest(export_path / "locks" / part_name, locked_by_stat.version)); + ops.emplace_back(zkutil::makeSetRequest(processing_part_path, processing_part_entry.toJsonString(), -1)); + + LOG_INFO(storage.log, "ExportPartition scheduler task: Updating processing part entry for part {}, retry count: {}, max retries: {}", part_name, processing_part_entry.retry_count, max_retries); + + if (processing_part_entry.retry_count >= max_retries) + { + /// just set status in processing_part_path and finished_by + processing_part_entry.status = ExportReplicatedMergeTreePartitionProcessingPartEntry::Status::FAILED; + processing_part_entry.finished_by = storage.replica_name; + + ops.emplace_back(zkutil::makeSetRequest(export_path / "status", String(magic_enum::enum_name(ExportReplicatedMergeTreePartitionTaskEntry::Status::FAILED)).data(), -1)); + LOG_INFO(storage.log, "ExportPartition scheduler task: Retry count limit exceeded for part {}, will try to fail the entire task", part_name); + } + else + { + LOG_INFO(storage.log, "ExportPartition scheduler task: Retry count limit not exceeded for part {}, will increment retry count", part_name); + } + + ExportPartitionUtils::appendExceptionOps( + ops, zk, export_path, storage.replica_name, part_name, + exception->message(), storage.log.load()); + + ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRequests); + ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperMulti); + Coordination::Responses responses; + if (Coordination::Error::ZOK != zk->tryMulti(ops, responses)) + { + LOG_INFO(storage.log, "ExportPartition scheduler task: All failure mechanism failed, will not try to update it"); + return; + } + + LOG_INFO(storage.log, "ExportPartition scheduler task: Successfully updated exception counters for part {}", part_name); +} + +bool ExportPartitionTaskScheduler::tryToMovePartToProcessed( + const std::filesystem::path & export_path, + const std::filesystem::path & processing_parts_path, + const std::filesystem::path & processed_part_path, + const std::string & part_name, + const std::vector & relative_paths_in_destination_storage, + const zkutil::ZooKeeperPtr & zk +) +{ + Coordination::Stat locked_by_stat; + std::string locked_by; + + ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRequests); + ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperGet); + if (!zk->tryGet(export_path / "locks" / part_name, locked_by, &locked_by_stat)) + { + LOG_INFO(storage.log, "ExportPartition scheduler task: Part {} is not locked by any replica, will not commit or set it as completed", part_name); + return false; + } + + /// Is this a good idea? what if the file we just pushed to s3 ends up triggering an exception in the replica that actually locks the part and it does not commit? + /// I guess we should not throw if file already exists for export partition, hard coded. + if (locked_by != storage.replica_name) + { + LOG_INFO(storage.log, "ExportPartition scheduler task: Part {} is locked by another replica, will not commit or set it as completed", part_name); + return false; + } + + Coordination::Requests requests; + + ExportReplicatedMergeTreePartitionProcessedPartEntry processed_part_entry; + processed_part_entry.part_name = part_name; + processed_part_entry.paths_in_destination = relative_paths_in_destination_storage; + processed_part_entry.finished_by = storage.replica_name; + + requests.emplace_back(zkutil::makeRemoveRequest(processing_parts_path / part_name, -1)); + requests.emplace_back(zkutil::makeCreateRequest(processed_part_path, processed_part_entry.toJsonString(), zkutil::CreateMode::Persistent)); + requests.emplace_back(zkutil::makeRemoveRequest(export_path / "locks" / part_name, locked_by_stat.version)); + + ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRequests); + ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperMulti); + Coordination::Responses responses; + if (Coordination::Error::ZOK != zk->tryMulti(requests, responses)) + { + + /// todo arthur remember what to do here + LOG_INFO(storage.log, "ExportPartition scheduler task: Failed to update export path, skipping"); + return false; + } + + return true; +} + +bool ExportPartitionTaskScheduler::areAllPartsProcessed( + const std::filesystem::path & export_path, + const zkutil::ZooKeeperPtr & zk) +{ + ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRequests); + ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperGetChildren); + Strings parts_in_processing_or_pending; + if (Coordination::Error::ZOK != zk->tryGetChildren(export_path / "processing", parts_in_processing_or_pending)) + { + LOG_INFO(storage.log, "ExportPartition scheduler task: Failed to get parts in processing or pending, will not try to commit export partition"); + return false; + } + + if (!parts_in_processing_or_pending.empty()) + { + LOG_INFO(storage.log, "ExportPartition scheduler task: There are still parts in processing or pending, will not try to commit export partition"); + return false; + } + + return true; +} + +} diff --git a/src/Storages/MergeTree/ExportPartitionTaskScheduler.h b/src/Storages/MergeTree/ExportPartitionTaskScheduler.h new file mode 100644 index 000000000000..29a41fde1cb9 --- /dev/null +++ b/src/Storages/MergeTree/ExportPartitionTaskScheduler.h @@ -0,0 +1,66 @@ +#pragma once + +#include +#include + +namespace DB +{ + +class Exception; +class StorageReplicatedMergeTree; + +struct ExportReplicatedMergeTreePartitionManifest; + +/// todo arthur remember to add check(lock, version) when updating stuff because maybe if we believe we have the lock, we might not actually have it +class ExportPartitionTaskScheduler +{ +public: + ExportPartitionTaskScheduler(StorageReplicatedMergeTree & storage); + + void run(); +private: + StorageReplicatedMergeTree & storage; + + /// todo arthur maybe it is invalid to grab the manifst here + void handlePartExportCompletion( + const std::string & export_key, + const std::string & part_name, + const ExportReplicatedMergeTreePartitionManifest & manifest, + const StoragePtr & destination_storage, + const MergeTreePartExportManifest::CompletionCallbackResult & result); + + void handlePartExportSuccess( + const ExportReplicatedMergeTreePartitionManifest & manifest, + const StoragePtr & destination_storage, + const std::filesystem::path & processing_parts_path, + const std::filesystem::path & processed_part_path, + const std::string & part_name, + const std::filesystem::path & export_path, + const zkutil::ZooKeeperPtr & zk, + const std::vector & relative_paths_in_destination_storage + ); + + void handlePartExportFailure( + const std::filesystem::path & processing_parts_path, + const std::string & part_name, + const std::filesystem::path & export_path, + const zkutil::ZooKeeperPtr & zk, + const std::optional & exception, + size_t max_retries); + + bool tryToMovePartToProcessed( + const std::filesystem::path & export_path, + const std::filesystem::path & processing_parts_path, + const std::filesystem::path & processed_part_path, + const std::string & part_name, + const std::vector & relative_paths_in_destination_storage, + const zkutil::ZooKeeperPtr & zk + ); + + bool areAllPartsProcessed( + const std::filesystem::path & export_path, + const zkutil::ZooKeeperPtr & zk + ); +}; + +} diff --git a/src/Storages/MergeTree/ExportPartitionUtils.cpp b/src/Storages/MergeTree/ExportPartitionUtils.cpp new file mode 100644 index 000000000000..02069754bd3d --- /dev/null +++ b/src/Storages/MergeTree/ExportPartitionUtils.cpp @@ -0,0 +1,498 @@ +#include +#include +#include +#include +#include +#include "Storages/ExportReplicatedMergeTreePartitionManifest.h" +#include "Storages/ExportReplicatedMergeTreePartitionTaskEntry.h" +#include +#include +#include +#include + +#if USE_AVRO +#include +#include +#endif + +namespace ProfileEvents +{ + extern const Event ExportPartitionZooKeeperRequests; + extern const Event ExportPartitionZooKeeperGet; + extern const Event ExportPartitionZooKeeperGetChildren; + extern const Event ExportPartitionZooKeeperSet; + extern const Event ExportPartitionZooKeeperMulti; + extern const Event ExportPartitionZooKeeperExists; +} + +namespace DB +{ + +namespace ErrorCodes +{ + extern const int FAULT_INJECTED; + extern const int BAD_ARGUMENTS; + extern const int NO_SUCH_DATA_PART; + extern const int CORRUPTED_DATA; + extern const int NETWORK_ERROR; +} + +namespace FailPoints +{ + extern const char iceberg_export_after_commit_before_zk_completed[]; + extern const char export_partition_commit_always_throw[]; +} + +namespace fs = std::filesystem; + +namespace ExportPartitionUtils +{ + std::vector getPartitionValuesForIcebergCommit( + MergeTreeData & storage, const String & partition_id) + { + auto lock = storage.readLockParts(); + const auto parts = storage.getDataPartsVectorInPartitionForInternalUsage( + MergeTreeDataPartState::Active, partition_id, lock); + + /// todo arthur: bad arguments for now, pick a better one + if (parts.empty()) + throw Exception(ErrorCodes::NO_SUCH_DATA_PART, + "Cannot find active part for partition_id '{}' to derive Iceberg partition " + "values. Edge case: the partition may have been dropped after export started, " + "or this replica has not yet received any part for this partition. " + "The commit will be retried.", + partition_id); + return parts.front()->partition.value; + } + + ContextPtr getContextCopyWithTaskSettings(const ContextPtr & context, const ExportReplicatedMergeTreePartitionManifest & manifest) + { + auto context_copy = Context::createCopy(context); + context_copy->makeQueryContextForExportPart(); + context_copy->setCurrentQueryId(manifest.query_id); + context_copy->setSetting("output_format_parallel_formatting", manifest.parallel_formatting); + context_copy->setSetting("output_format_parquet_parallel_encoding", manifest.parquet_parallel_encoding); + context_copy->setSetting("max_threads", manifest.max_threads); + context_copy->setSetting("export_merge_tree_part_file_already_exists_policy", String(magic_enum::enum_name(manifest.file_already_exists_policy))); + context_copy->setSetting("export_merge_tree_part_max_bytes_per_file", manifest.max_bytes_per_file); + context_copy->setSetting("export_merge_tree_part_max_rows_per_file", manifest.max_rows_per_file); + context_copy->setSetting("iceberg_insert_max_bytes_in_data_file", manifest.max_bytes_per_file); + context_copy->setSetting("iceberg_insert_max_rows_in_data_file", manifest.max_rows_per_file); + + /// always skip pending mutations and patch parts because we already validated the parts during query processing + context_copy->setSetting("export_merge_tree_part_throw_on_pending_mutations", false); + context_copy->setSetting("export_merge_tree_part_throw_on_pending_patch_parts", false); + + context_copy->setSetting("export_merge_tree_part_filename_pattern", manifest.filename_pattern); + context_copy->setSetting("write_full_path_in_iceberg_metadata", manifest.write_full_path_in_iceberg_metadata); + + return context_copy; + } + + /// Collect all the exported paths from the processed parts + /// If multiRead is supported by the keeper implementation, it is done in a single request + /// Otherwise, multiple async requests are sent + std::vector getExportedPaths(const LoggerPtr & log, const zkutil::ZooKeeperPtr & zk, const std::string & export_path) + { + std::vector exported_paths; + + LOG_INFO(log, "ExportPartition: Getting exported paths for {}", export_path); + + const auto processed_parts_path = fs::path(export_path) / "processed"; + + ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRequests); + ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperGetChildren); + std::vector processed_parts; + if (Coordination::Error::ZOK != zk->tryGetChildren(processed_parts_path, processed_parts)) + { + /// todo arthur do something here + LOG_INFO(log, "ExportPartition: Failed to get parts children, exiting"); + return {}; + } + + std::vector get_paths; + + for (const auto & processed_part : processed_parts) + { + get_paths.emplace_back(processed_parts_path / processed_part); + } + + auto responses = zk->tryGet(get_paths); + ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRequests); + ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperGet, get_paths.size()); + + responses.waitForResponses(); + + for (size_t i = 0; i < responses.size(); ++i) + { + if (responses[i].error != Coordination::Error::ZOK) + { + /// todo arthur what to do in this case? + /// It could be that zk is corrupt, in that case we should fail the task + /// but it can also be some temporary network issue? not sure + LOG_INFO(log, "ExportPartition: Failed to get exported path, exiting"); + return {}; + } + + const auto processed_part_entry = ExportReplicatedMergeTreePartitionProcessedPartEntry::fromJsonString(responses[i].data); + + for (const auto & path_in_destination : processed_part_entry.paths_in_destination) + { + exported_paths.emplace_back(path_in_destination); + } + } + + return exported_paths; + } + + void commit( + const ExportReplicatedMergeTreePartitionManifest & manifest, + const StoragePtr & destination_storage, + const zkutil::ZooKeeperPtr & zk, + const LoggerPtr & log, + const std::string & entry_path, + const ContextPtr & context_in, + MergeTreeData & source_storage) + { + auto context = Context::createCopy(context_in); + context->setSetting("write_full_path_in_iceberg_metadata", manifest.write_full_path_in_iceberg_metadata); + + /// Failpoint used by integration tests to force persistent commit failure and exercise + /// the commit-attempts budget / FAILED state transition. + fiu_do_on(FailPoints::export_partition_commit_always_throw, + { + throw Exception(ErrorCodes::FAULT_INJECTED, + "Failpoint: export_partition_commit_always_throw"); + }); + + const auto exported_paths = ExportPartitionUtils::getExportedPaths(log, zk, entry_path); + + if (exported_paths.empty()) + { + throw Exception(ErrorCodes::CORRUPTED_DATA, "ExportPartition: No exported paths found, will not commit export. This might be a bug"); + } + + //// not checking for an exact match because a single part might generate multiple files + if (exported_paths.size() < manifest.parts.size()) + { + throw Exception(ErrorCodes::CORRUPTED_DATA, "ExportPartition: Reached the commit phase, but exported paths size is less than the number of parts, will not commit export. This might be a bug"); + } + + IStorage::IcebergCommitExportPartitionArguments iceberg_args; + + if (!manifest.iceberg_metadata_json.empty()) + { + iceberg_args.metadata_json_string = manifest.iceberg_metadata_json; + if (source_storage.getInMemoryMetadataPtr()->hasPartitionKey()) + iceberg_args.partition_values = + getPartitionValuesForIcebergCommit(source_storage, manifest.partition_id); + } + + destination_storage->commitExportPartitionTransaction(manifest.transaction_id, manifest.partition_id, exported_paths, iceberg_args, context); + + /// Failpoint to simulate a crash after the Iceberg commit succeeds but before + /// ZooKeeper is updated to COMPLETED. Used by idempotency integration tests. + fiu_do_on(FailPoints::iceberg_export_after_commit_before_zk_completed, + { + LOG_INFO(log, "Failpoint: simulating crash after Iceberg commit, before ZK COMPLETED"); + std::this_thread::sleep_for(std::chrono::seconds(10)); + throw Exception(ErrorCodes::FAULT_INJECTED, + "Failpoint: simulating crash after Iceberg commit, before ZK COMPLETED"); + }); + + LOG_INFO(log, "ExportPartition: Committed export, mark as completed"); + ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRequests); + ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperSet); + if (Coordination::Error::ZOK == zk->trySet(fs::path(entry_path) / "status", String(magic_enum::enum_name(ExportReplicatedMergeTreePartitionTaskEntry::Status::COMPLETED)).data(), -1)) + { + LOG_INFO(log, "ExportPartition: Marked export as completed"); + } + else + { + throw Exception(ErrorCodes::NETWORK_ERROR, "ExportPartition: Failed to mark export as completed, will not try to fix it"); + } + } + + bool handleCommitFailure( + const zkutil::ZooKeeperPtr & zk, + const std::string & entry_path, + size_t max_attempts, + const LoggerPtr & log) + { + const std::string status_path = fs::path(entry_path) / "status"; + + /// Read /status together with its stat so we can (a) bail early if another + /// replica has already moved the task out of PENDING and (b) use a + /// version-checked Set later to avoid clobbering a concurrent write + /// (e.g. a racing successful commit that marked the task COMPLETED between + /// our read and our tryMulti). + Coordination::Stat status_stat; + std::string current_status; + + ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRequests); + ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperGet); + if (!zk->tryGet(status_path, current_status, &status_stat)) + { + /// Task was removed (TTL cleanup or force-overwrite). Nothing to do. + LOG_INFO(log, "ExportPartition: /status missing for {}, skipping commit-failure bookkeeping", entry_path); + return false; + } + + const auto status = magic_enum::enum_cast(current_status); + if (!status) + { + LOG_INFO(log, "ExportPartition: Invalid status {} for task {}, skipping commit-failure bookkeeping", current_status, entry_path); + return false; + } + + if (status != ExportReplicatedMergeTreePartitionTaskEntry::Status::PENDING) + { + /// Another replica already reached a terminal state (COMPLETED or FAILED). + /// Do NOT overwrite — a successful commit by a peer must win. + LOG_INFO(log, + "ExportPartition: /status for {} is {} (not PENDING), skipping commit-failure bookkeeping", + entry_path, current_status); + return false; + } + + Coordination::Requests ops; + + /// Bump the global commit_attempts counter (shared across replicas). + /// Non-atomic get+set(-1), matching exceptions_per_replica/count semantics. + /// Under a race, two replicas may see the same value and write the same +1, + /// under-counting by one. FAILED then fires one retry later than the threshold, + /// which is acceptable (we always converge to FAILED, never "never"). + const std::string commit_attempts_path = fs::path(entry_path) / "commit_attempts"; + + size_t attempts = 0; + std::string attempts_string; + + ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRequests); + ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperGet); + if (zk->tryGet(commit_attempts_path, attempts_string)) + { + try + { + attempts = parse(attempts_string); + } + catch (...) + { + LOG_WARNING(log, "ExportPartition: commit_attempts value '{}' at {} is not a valid integer, treating as 0", attempts_string, commit_attempts_path); + attempts = 0; + } + + attempts += 1; + ops.emplace_back(zkutil::makeSetRequest(commit_attempts_path, std::to_string(attempts), -1)); + } + else + { + attempts = 1; + ops.emplace_back(zkutil::makeCreateRequest(commit_attempts_path, "1", zkutil::CreateMode::Persistent)); + } + + /// Transition to FAILED if the commit budget is exhausted. + /// Uses the same setting as per-part retries (manifest.max_retries) per user decision. + /// Version-checked Set: if /status has changed since we read it (e.g. a peer's + /// commit() succeeded and wrote COMPLETED), the whole multi aborts with + /// ZBADVERSION and we safely do nothing — the winning terminal state stands. + const bool exhausted = attempts >= max_attempts; + if (exhausted) + { + ops.emplace_back(zkutil::makeSetRequest( + status_path, + String(magic_enum::enum_name(ExportReplicatedMergeTreePartitionTaskEntry::Status::FAILED)).data(), + status_stat.version)); + } + + ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRequests); + ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperMulti); + Coordination::Responses responses; + const auto rc = zk->tryMulti(ops, responses); + if (rc != Coordination::Error::ZOK) + { + /// Any error here (ZBADVERSION on /status race or counter race, ZNODEEXISTS on + /// lazy-create race, ZNONODE if someone removed the task concurrently) is + /// non-fatal: the next attempt re-reads /status and either skips (terminal + /// state won) or retries the bookkeeping. Worst case we delay FAILED by one + /// poll cycle, which matches the best-effort property of the existing counters. + LOG_INFO(log, "ExportPartition: Failed to persist commit failure bookkeeping for {}: {}", entry_path, rc); + return false; + } + + LOG_INFO(log, + "ExportPartition: Commit failure recorded for {} (attempt {}/{}){}", + entry_path, attempts, max_attempts, + exhausted ? ", task transitioned to FAILED" : ""); + + return exhausted; + } + + void appendExceptionOps( + Coordination::Requests & ops, + const zkutil::ZooKeeperPtr & zk, + const std::filesystem::path & entry_path, + const std::string & replica_name, + const std::string & part_name, + const std::string & exception_message, + const LoggerPtr & log) + { + const auto exceptions_per_replica_path = entry_path / "exceptions_per_replica" / replica_name; + const auto count_path = exceptions_per_replica_path / "count"; + const auto last_exception_path = exceptions_per_replica_path / "last_exception"; + + ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRequests); + ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperExists); + if (zk->exists(exceptions_per_replica_path)) + { + LOG_INFO(log, "ExportPartition: Exceptions per replica path exists, no need to create it"); + std::string num_exceptions_string; + + ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRequests); + ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperGet); + if (zk->tryGet(count_path, num_exceptions_string)) + { + const auto num_exceptions = parse(num_exceptions_string) + 1; + ops.emplace_back(zkutil::makeSetRequest(count_path, std::to_string(num_exceptions), -1)); + } + else + { + /// TODO maybe we should find a better way to handle this case, not urgent + LOG_INFO(log, "ExportPartition: Failed to get number of exceptions, will not increment it"); + } + + ops.emplace_back(zkutil::makeSetRequest(last_exception_path / "part", part_name, -1)); + ops.emplace_back(zkutil::makeSetRequest(last_exception_path / "exception", exception_message, -1)); + } + else + { + LOG_INFO(log, "ExportPartition: Exceptions per replica path does not exist, will create it"); + ops.emplace_back(zkutil::makeCreateRequest(exceptions_per_replica_path, "", zkutil::CreateMode::Persistent)); + ops.emplace_back(zkutil::makeCreateRequest(count_path, "1", zkutil::CreateMode::Persistent)); + ops.emplace_back(zkutil::makeCreateRequest(last_exception_path, "", zkutil::CreateMode::Persistent)); + ops.emplace_back(zkutil::makeCreateRequest(last_exception_path / "part", part_name, zkutil::CreateMode::Persistent)); + ops.emplace_back(zkutil::makeCreateRequest(last_exception_path / "exception", exception_message, zkutil::CreateMode::Persistent)); + } + } + +#if USE_AVRO + void verifyIcebergPartitionCompatibility( + const Poco::JSON::Object::Ptr & metadata_object, + const ASTPtr & partition_key_ast) + { + const auto original_schema_id = metadata_object->getValue(Iceberg::f_current_schema_id); + const auto partition_spec_id = metadata_object->getValue(Iceberg::f_default_spec_id); + + Poco::JSON::Object::Ptr current_schema_json; + { + const auto schemas = metadata_object->getArray(Iceberg::f_schemas); + for (size_t i = 0; i < schemas->size(); ++i) + { + auto s = schemas->getObject(static_cast(i)); + if (s->getValue(Iceberg::f_schema_id) == static_cast(original_schema_id)) + { + current_schema_json = s; + break; + } + } + } + + Poco::JSON::Object::Ptr partition_spec_json; + { + const auto specs = metadata_object->getArray(Iceberg::f_partition_specs); + for (size_t i = 0; i < specs->size(); ++i) + { + auto s = specs->getObject(static_cast(i)); + if (s->getValue(Iceberg::f_spec_id) == partition_spec_id) + { + partition_spec_json = s; + break; + } + } + } + + if (!current_schema_json || !partition_spec_json) + return; + + /// Build column_name → Iceberg source-id from the destination schema (and the inverse). + std::unordered_map column_name_to_source_id; + std::unordered_map source_id_to_column_name; + { + const auto schema_fields = current_schema_json->getArray(Iceberg::f_fields); + for (size_t i = 0; i < schema_fields->size(); ++i) + { + auto f = schema_fields->getObject(static_cast(i)); + const auto col_name = f->getValue(Iceberg::f_name); + const auto source_id = f->getValue(Iceberg::f_id); + column_name_to_source_id[col_name] = source_id; + source_id_to_column_name[source_id] = col_name; + } + } + + auto source_id_to_name = [&](Int32 id) -> String + { + auto it = source_id_to_column_name.find(id); + return it != source_id_to_column_name.end() ? it->second : fmt::format("", id); + }; + + /// Convert the MergeTree PARTITION BY AST into the equivalent Iceberg spec. + Poco::JSON::Array::Ptr expected_fields; + try + { + const auto expected_spec = Iceberg::getPartitionSpec( + partition_key_ast, column_name_to_source_id).first; + expected_fields = expected_spec->getArray(Iceberg::f_fields); + } + catch (const Exception & e) + { + throw Exception(ErrorCodes::BAD_ARGUMENTS, + "Cannot export partition to Iceberg table: the source MergeTree partition " + "key cannot be represented as an Iceberg partition spec: {}", e.message()); + } + + const auto actual_fields = partition_spec_json->getArray(Iceberg::f_fields); + const size_t expected_size = expected_fields ? expected_fields->size() : 0; + const size_t actual_size = actual_fields ? actual_fields->size() : 0; + + if (expected_size != actual_size) + throw Exception(ErrorCodes::BAD_ARGUMENTS, + "Cannot export partition to Iceberg table: partition scheme mismatch. " + "Source MergeTree has {} partition field(s), destination Iceberg table has {}.", + expected_size, actual_size); + + for (size_t i = 0; i < expected_size; ++i) + { + auto ef = expected_fields->getObject(static_cast(i)); + auto af = actual_fields->getObject(static_cast(i)); + + const auto expected_source_id = ef->getValue(Iceberg::f_source_id); + const auto actual_source_id = af->getValue(Iceberg::f_source_id); + const auto expected_transform = ef->getValue(Iceberg::f_transform); + const auto actual_transform = af->getValue(Iceberg::f_transform); + + /// Normalize both transform names through parseTransformAndArgument so that + /// equivalent aliases ("day"/"days", "hour"/"hours", "year"/"years", etc.) + /// produced by different writers (ClickHouse vs Spark/Trino) compare equal. + /// Comparison is on {function_name, argument}; time_zone is writer-specific + /// and not part of the partition spec identity. + const auto expected_canonical = Iceberg::parseTransformAndArgument(expected_transform); + const auto actual_canonical = Iceberg::parseTransformAndArgument(actual_transform); + const bool transforms_match = + (expected_canonical && actual_canonical) + ? (expected_canonical->transform_name == actual_canonical->transform_name + && expected_canonical->argument == actual_canonical->argument) + : (expected_transform == actual_transform); + + if (expected_source_id != actual_source_id || !transforms_match) + throw Exception(ErrorCodes::BAD_ARGUMENTS, + "Cannot export partition to Iceberg table: partition field {} mismatch. " + "Source MergeTree maps to column '{}' (source_id={}) transform='{}', " + "but destination Iceberg has column '{}' (source_id={}) transform='{}'.", + i, + source_id_to_name(expected_source_id), expected_source_id, expected_transform, + source_id_to_name(actual_source_id), actual_source_id, actual_transform); + } + } +#endif +} + +} diff --git a/src/Storages/MergeTree/ExportPartitionUtils.h b/src/Storages/MergeTree/ExportPartitionUtils.h new file mode 100644 index 000000000000..411f3b5224be --- /dev/null +++ b/src/Storages/MergeTree/ExportPartitionUtils.h @@ -0,0 +1,98 @@ +#pragma once + +#include +#include +#include +#include +#include +#include +#include "Storages/IStorage.h" +#include + +#if USE_AVRO +#include +#include +#endif + +namespace DB +{ + +class MergeTreeData; +struct ExportReplicatedMergeTreePartitionManifest; + +namespace ExportPartitionUtils +{ + std::vector getExportedPaths(const LoggerPtr & log, const zkutil::ZooKeeperPtr & zk, const std::string & export_path); + + ContextPtr getContextCopyWithTaskSettings(const ContextPtr & context, const ExportReplicatedMergeTreePartitionManifest & manifest); + + /// Returns the partition key values for the given partition_id by reading from + /// the first active local part. Throws LOGICAL_ERROR if no such part is found. + /// + /// Edge case: if the partition was dropped after export started, or this replica + /// has not yet received any part for this partition (extreme replication lag on a + /// recovery path), no active part will be found and the commit will fail. The task + /// will be retried on the next poll cycle or picked up by a different replica. + std::vector getPartitionValuesForIcebergCommit( + MergeTreeData & storage, const String & partition_id); + + void commit( + const ExportReplicatedMergeTreePartitionManifest & manifest, + const StoragePtr & destination_storage, + const zkutil::ZooKeeperPtr & zk, + const LoggerPtr & log, + const std::string & entry_path, + const ContextPtr & context, + MergeTreeData & source_storage + ); + + /// Handles a commit-phase failure for a replicated partition export: + /// - increments /commit_attempts (lazy-created) + /// - sets /status to FAILED once attempts >= max_attempts + /// + /// The counter is a best-effort, non-atomic get+set(-1), matching + /// exceptions_per_replica/count. Concurrent failing commits may under-count by one + /// (FAILED may fire one retry later than the threshold), which is acceptable. + /// + /// `replica_name` and `exception` are currently unused and reserved for future + /// integration with per-replica diagnostics. + /// + /// Returns true if this call transitioned the task to FAILED. + bool handleCommitFailure( + const zkutil::ZooKeeperPtr & zk, + const std::string & entry_path, + size_t max_attempts, + const LoggerPtr & log); + + /// Appends ZK ops to `ops` that record a per-replica exception under + /// /exceptions_per_replica//last_exception/{exception,part} + /// and increment /exceptions_per_replica//count, + /// creating the subtree if absent. + /// + /// The count increment is non-atomic (synchronous tryGet + set with version -1). + /// Concurrent failing writers may under-count by one, which is accepted in this + /// subsystem and matches the pre-existing behaviour. + /// + /// Intended to be combined with additional ops (for example a version-guarded + /// status set) and executed as a single `tryMulti` so the exception record and + /// the accompanying state transition commit atomically. + void appendExceptionOps( + Coordination::Requests & ops, + const zkutil::ZooKeeperPtr & zk, + const std::filesystem::path & entry_path, + const std::string & replica_name, + const std::string & part_name, + const std::string & exception_message, + const LoggerPtr & log); + +#if USE_AVRO + /// Verifies that the source MergeTree partition key is compatible with the + /// destination Iceberg partition spec by comparing field source-ids and + /// transforms in order. Throws BAD_ARGUMENTS if they do not match. + void verifyIcebergPartitionCompatibility( + const Poco::JSON::Object::Ptr & metadata_object, + const ASTPtr & partition_key_ast); +#endif +} + +} diff --git a/src/Storages/MergeTree/IMergeTreeDataPart.cpp b/src/Storages/MergeTree/IMergeTreeDataPart.cpp index 5a29ab7a5f91..597df9fce602 100644 --- a/src/Storages/MergeTree/IMergeTreeDataPart.cpp +++ b/src/Storages/MergeTree/IMergeTreeDataPart.cpp @@ -345,6 +345,7 @@ String IMergeTreeDataPart::MinMaxIndex::getFileColumnName(const String & column_ return stream_name; } +<<<<<<< HEAD IMergeTreeDataPart::MinMaxIndexPtr IMergeTreeDataPart::getMinMaxIndex() const { std::lock_guard lock(minmax_idx_mutex); @@ -369,6 +370,42 @@ void IMergeTreeDataPart::setMinMaxIndex(MinMaxIndexPtr minmax_index) const { std::lock_guard lock(minmax_idx_mutex); minmax_idx = std::move(minmax_index); +======= +Block IMergeTreeDataPart::MinMaxIndex::getBlock(const MergeTreeData & data) const +{ + if (!initialized) + throw Exception(ErrorCodes::LOGICAL_ERROR, "Attempt to get block from uninitialized MinMax index."); + + Block block; + + const auto metadata_snapshot = data.getInMemoryMetadataPtr(); + const auto & partition_key = metadata_snapshot->getPartitionKey(); + + const auto minmax_column_names = data.getMinMaxColumnsNames(partition_key); + const auto minmax_column_types = data.getMinMaxColumnsTypes(partition_key); + const auto minmax_idx_size = minmax_column_types.size(); + + for (size_t i = 0; i < minmax_idx_size; ++i) + { + const auto & data_type = minmax_column_types[i]; + const auto & column_name = minmax_column_names[i]; + + const auto column = data_type->createColumn(); + + auto range = hyperrectangle.at(i); + range.shrinkToIncludedIfPossible(); + + const auto & min_val = range.left; + const auto & max_val = range.right; + + column->insert(min_val); + column->insert(max_val); + + block.insert(ColumnWithTypeAndName(column->getPtr(), data_type, column_name)); + } + + return block; +>>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) } void IMergeTreeDataPart::incrementStateMetric(MergeTreeDataPartState state_) const diff --git a/src/Storages/MergeTree/IMergeTreeDataPart.h b/src/Storages/MergeTree/IMergeTreeDataPart.h index bc9ffb8b5035..d5ec94ac6b04 100644 --- a/src/Storages/MergeTree/IMergeTreeDataPart.h +++ b/src/Storages/MergeTree/IMergeTreeDataPart.h @@ -388,6 +388,8 @@ class IMergeTreeDataPart : public std::enable_shared_from_this; diff --git a/src/Storages/MergeTree/MergeTreeBackgroundExecutor.cpp b/src/Storages/MergeTree/MergeTreeBackgroundExecutor.cpp index 5b24487f1a72..a0f3aea97a33 100644 --- a/src/Storages/MergeTree/MergeTreeBackgroundExecutor.cpp +++ b/src/Storages/MergeTree/MergeTreeBackgroundExecutor.cpp @@ -151,6 +151,12 @@ size_t MergeTreeBackgroundExecutor::getMaxTasksCount() const return max_tasks_count.load(std::memory_order_relaxed); } +template +size_t MergeTreeBackgroundExecutor::getAvailableSlots() const +{ + return getMaxTasksCount() - CurrentMetrics::values[metric].load(std::memory_order_relaxed); +} + template bool MergeTreeBackgroundExecutor::trySchedule(ExecutableTaskPtr task) { diff --git a/src/Storages/MergeTree/MergeTreeBackgroundExecutor.h b/src/Storages/MergeTree/MergeTreeBackgroundExecutor.h index e7db0523ea27..8b93a5d4799e 100644 --- a/src/Storages/MergeTree/MergeTreeBackgroundExecutor.h +++ b/src/Storages/MergeTree/MergeTreeBackgroundExecutor.h @@ -330,6 +330,8 @@ class MergeTreeBackgroundExecutor final : boost::noncopyable /// can lead only to some postponing, not logical error. size_t getMaxTasksCount() const; + size_t getAvailableSlots() const; + bool trySchedule(ExecutableTaskPtr task); void removeTasksCorrespondingToStorage(StorageID id); void wait(); diff --git a/src/Storages/MergeTree/MergeTreeData.cpp b/src/Storages/MergeTree/MergeTreeData.cpp index 0d46d17862db..5bff1d0b2763 100644 --- a/src/Storages/MergeTree/MergeTreeData.cpp +++ b/src/Storages/MergeTree/MergeTreeData.cpp @@ -8,7 +8,9 @@ #include #include +#include #include +#include #include #include #include @@ -17,8 +19,34 @@ #include #include #include +<<<<<<< HEAD #include #include +======= +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +>>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) #include #include #include @@ -36,6 +64,7 @@ #include #include #include +#include #include #include #include @@ -72,7 +101,27 @@ #include #include #include +<<<<<<< HEAD #include +======= +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +>>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) #include #include #include @@ -82,6 +131,7 @@ #include #include #include +<<<<<<< HEAD #include #include #include @@ -122,6 +172,9 @@ #include #include #include +======= +#include +>>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) #include @@ -140,6 +193,7 @@ #include #include #include +#include #include #include @@ -186,6 +240,10 @@ namespace ProfileEvents extern const Event RestorePartsSkippedFiles; extern const Event RestorePartsSkippedBytes; extern const Event LoadedStatisticsMicroseconds; + extern const Event PartsExports; + extern const Event PartsExportTotalMilliseconds; + extern const Event PartsExportFailures; + extern const Event PartsExportDuplicated; } namespace CurrentMetrics @@ -237,7 +295,18 @@ namespace Setting extern const SettingsBool use_statistics; extern const SettingsBool use_statistics_cache; extern const SettingsBool use_partition_pruning; +<<<<<<< HEAD extern const SettingsBool use_skip_indexes; +======= + extern const SettingsBool allow_experimental_export_merge_tree_part; + extern const SettingsUInt64 min_bytes_to_use_direct_io; + extern const SettingsMergeTreePartExportFileAlreadyExistsPolicy export_merge_tree_part_file_already_exists_policy; + extern const SettingsBool output_format_parallel_formatting; + extern const SettingsBool output_format_parquet_parallel_encoding; + extern const SettingsBool export_merge_tree_part_throw_on_pending_mutations; + extern const SettingsBool export_merge_tree_part_throw_on_pending_patch_parts; + extern const SettingsBool allow_insert_into_iceberg; +>>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) } namespace MergeTreeSetting @@ -374,6 +443,7 @@ namespace ErrorCodes extern const int CANNOT_FORGET_PARTITION; extern const int DATA_TYPE_CANNOT_BE_USED_IN_KEY; extern const int TOO_LARGE_LIGHTWEIGHT_UPDATES; +<<<<<<< HEAD extern const int FAULT_INJECTED; } @@ -383,6 +453,11 @@ namespace FailPoints /// transient error (e.g. temporary disk unavailability). Used to test that the refresh task /// reschedules itself after such an error instead of stopping permanently. extern const char merge_tree_refresh_parts_throw_once[]; +======= + extern const int UNKNOWN_TABLE; + extern const int FILE_ALREADY_EXISTS; + extern const int PENDING_MUTATIONS_NOT_ALLOWED; +>>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) } static String getPartNameFromAST(const ASTPtr & partition) @@ -5211,8 +5286,6 @@ void MergeTreeData::changeSettings( { if (new_settings) { - bool has_storage_policy_changed = false; - const auto & new_changes = new_settings->as().changes; StoragePolicyPtr new_storage_policy = nullptr; @@ -5251,8 +5324,6 @@ void MergeTreeData::changeSettings( disk->createDirectories(fs::path(relative_data_path) / DETACHED_DIR_NAME); } /// FIXME how would that be done while reloading configuration??? - - has_storage_policy_changed = true; } } } @@ -5299,9 +5370,6 @@ void MergeTreeData::changeSettings( setInMemoryMetadata(new_metadata); - if (has_storage_policy_changed) - startBackgroundMovesIfNeeded(); - if (has_refresh_statistics_interval_changed) { startStatisticsCache(); @@ -7002,6 +7070,250 @@ void MergeTreeData::movePartitionToTable(const PartitionCommand & command, Conte movePartitionToTable(dest_storage, command.partition, query_context); } +void MergeTreeData::exportPartToTable(const PartitionCommand & command, ContextPtr query_context) +{ + if (!query_context->getSettingsRef()[Setting::allow_experimental_export_merge_tree_part]) + { + throw Exception(ErrorCodes::SUPPORT_IS_DISABLED, + "Exporting merge tree part is experimental. Set `allow_experimental_export_merge_tree_part` to enable it"); + } + + const auto part_name = command.partition->as().value.safeGet(); + + if (!command.to_table_function) + { + const auto database_name = query_context->resolveDatabase(command.to_database); + exportPartToTable(part_name, StorageID{database_name, command.to_table}, generateSnowflakeIDString(), query_context); + + return; + } + + auto table_function_ast = command.to_table_function; + auto table_function_ptr = TableFunctionFactory::instance().get(command.to_table_function, query_context); + + if (table_function_ptr->needStructureHint()) + { + const auto source_metadata_ptr = getInMemoryMetadataPtr(); + + /// Grab only the readable columns from the source metadata to skip ephemeral columns + const auto readable_columns = ColumnsDescription(source_metadata_ptr->getColumns().getReadable()); + table_function_ptr->setStructureHint(readable_columns); + } + + if (command.partition_by_expr) + { + table_function_ptr->setPartitionBy(command.partition_by_expr); + } + + auto dest_storage = table_function_ptr->execute( + table_function_ast, + query_context, + table_function_ptr->getName(), + /* cached_columns */ {}, + /* use_global_context */ false, + /* is_insert_query */ true); + + if (!dest_storage) + { + throw Exception(ErrorCodes::BAD_ARGUMENTS, "Failed to reconstruct destination storage"); + } + + exportPartToTable(part_name, dest_storage, generateSnowflakeIDString(), query_context); +} + +void MergeTreeData::exportPartToTable( + const std::string & part_name, + const StorageID & destination_storage_id, + const String & transaction_id, + ContextPtr query_context, + const std::optional & iceberg_metadata_json, + bool allow_outdated_parts, + std::function completion_callback) +{ + auto dest_storage = DatabaseCatalog::instance().getTable(destination_storage_id, query_context); + + if (destination_storage_id == this->getStorageID()) + { + throw Exception(ErrorCodes::BAD_ARGUMENTS, "Exporting to the same table is not allowed"); + } + + exportPartToTable(part_name, dest_storage, transaction_id, query_context, iceberg_metadata_json, allow_outdated_parts, completion_callback); +} + +void MergeTreeData::exportPartToTable( + const std::string & part_name, + const StoragePtr & dest_storage, + const String & transaction_id, + ContextPtr query_context, + const std::optional & iceberg_metadata_json_, + bool allow_outdated_parts, + std::function completion_callback) +{ + if (!dest_storage->supportsImport(query_context)) + throw Exception(ErrorCodes::NOT_IMPLEMENTED, "Destination storage {} does not support MergeTree parts or uses unsupported partitioning", dest_storage->getName()); + + auto query_to_string = [] (const ASTPtr & ast) + { + return ast ? ast->formatWithSecretsOneLine() : ""; + }; + + auto source_metadata_ptr = getInMemoryMetadataPtr(); + auto destination_metadata_ptr = dest_storage->getInMemoryMetadataPtr(); + + std::string iceberg_metadata_json; + + if (dest_storage->isDataLake()) + { + if (!query_context->getSettingsRef()[Setting::allow_insert_into_iceberg]) + { + throw Exception(ErrorCodes::SUPPORT_IS_DISABLED, + "Iceberg writes are experimental. " + "To allow its usage, enable the setting allow_experimental_insert_into_iceberg"); + } + +#if USE_AVRO + if (iceberg_metadata_json_) + { + iceberg_metadata_json = *iceberg_metadata_json_; + } + else + { + auto * object_storage = dynamic_cast(dest_storage.get()); + + /// in theory this should never happen, but just in case + if (!object_storage) + { + throw Exception(ErrorCodes::BAD_ARGUMENTS, "Destination storage {} is not a StorageObjectStorage", dest_storage->getName()); + } + + auto * iceberg_metadata = dynamic_cast(object_storage->getExternalMetadata(query_context)); + if (!iceberg_metadata) + { + throw Exception(ErrorCodes::BAD_ARGUMENTS, "Destination storage {} is a data lake but not an iceberg table", dest_storage->getName()); + } + + const auto metadata_object = iceberg_metadata->getMetadataJSON(query_context); + + std::ostringstream oss; + metadata_object->stringify(oss); + iceberg_metadata_json = oss.str(); + + ExportPartitionUtils::verifyIcebergPartitionCompatibility(metadata_object, source_metadata_ptr->getPartitionKeyAST()); + } +#else + (void)iceberg_metadata_json_; + throw Exception(ErrorCodes::NOT_IMPLEMENTED, "Data lake export requires Avro support"); +#endif + } + + const auto & source_columns = source_metadata_ptr->getColumns(); + + const auto & destination_columns = destination_metadata_ptr->getColumns(); + + /// compare all source readable columns with all destination insertable columns + /// this allows us to skip ephemeral columns + if (source_columns.getReadable().sizeOfDifference(destination_columns.getInsertable())) + throw Exception(ErrorCodes::INCOMPATIBLE_COLUMNS, "Tables have different structure"); + + /// for data lakes this check is performed differently. It is a bit more complex as we need to convert the iceberg partition spec + /// to the MergeTree partition spec and compare the two. + if (!dest_storage->isDataLake()) + { + if (query_to_string(source_metadata_ptr->getPartitionKeyAST()) != query_to_string(destination_metadata_ptr->getPartitionKeyAST())) + throw Exception(ErrorCodes::BAD_ARGUMENTS, "Tables have different partition key"); + } + + auto part = getPartIfExists(part_name, {MergeTreeDataPartState::Active, MergeTreeDataPartState::Outdated}); + + if (!part) + throw Exception(ErrorCodes::NO_SUCH_DATA_PART, "No such data part '{}' to export in table '{}'", + part_name, getStorageID().getFullTableName()); + + if (part->getState() == MergeTreeDataPartState::Outdated && !allow_outdated_parts) + throw Exception( + ErrorCodes::BAD_ARGUMENTS, + "Part {} is in the outdated state and cannot be exported", + part_name); + + const bool throw_on_pending_mutations = query_context->getSettingsRef()[Setting::export_merge_tree_part_throw_on_pending_mutations]; + const bool throw_on_pending_patch_parts = query_context->getSettingsRef()[Setting::export_merge_tree_part_throw_on_pending_patch_parts]; + + MergeTreeData::IMutationsSnapshot::Params mutations_snapshot_params + { + .metadata_version = source_metadata_ptr->getMetadataVersion(), + .min_part_metadata_version = part->getMetadataVersion(), + .need_data_mutations = throw_on_pending_mutations, + .need_alter_mutations = throw_on_pending_mutations || throw_on_pending_patch_parts, + .need_patch_parts = throw_on_pending_patch_parts, + }; + + const auto mutations_snapshot = getMutationsSnapshot(mutations_snapshot_params); + + const auto alter_conversions = getAlterConversionsForPart(part, mutations_snapshot, query_context); + + /// re-check `throw_on_pending_mutations` because `pending_mutations` might have been filled due to `throw_on_pending_patch_parts` + if (throw_on_pending_mutations && alter_conversions->hasMutations()) + { + throw Exception(ErrorCodes::PENDING_MUTATIONS_NOT_ALLOWED, + "Part {} can not be exported because there are pending mutations. Either wait for the mutations to be applied or set `export_merge_tree_part_throw_on_pending_mutations` to false", + part_name); + } + + if (alter_conversions->hasPatches()) + { + throw Exception(ErrorCodes::PENDING_MUTATIONS_NOT_ALLOWED, + "Part {} can not be exported because there are pending patch parts. Either wait for the patch parts to be applied or set `export_merge_tree_part_throw_on_pending_patch_parts` to false", + part_name); + } + + { + MergeTreePartExportManifest manifest( + dest_storage, + part, + transaction_id, + query_context->getCurrentQueryId(), + query_context->getSettingsRef()[Setting::export_merge_tree_part_file_already_exists_policy].value, + query_context->getSettingsCopy(), + source_metadata_ptr, + iceberg_metadata_json, + completion_callback); + + std::lock_guard lock(export_manifests_mutex); + + manifest.task = std::make_shared(*this, manifest); + + if (!export_manifests.emplace(manifest).second) + { + throw Exception(ErrorCodes::ABORTED, "Data part '{}' is already being exported", + part_name); + } + + if (!background_moves_assignee.scheduleMoveTask(manifest.task)) + { + export_manifests.erase(manifest); + throw Exception(ErrorCodes::ABORTED, "Failed to schedule export part task for data part '{}'. Background executor is busy", + part_name); + } + } +} + +void MergeTreeData::killExportPart(const String & transaction_id) +{ + std::lock_guard lock(export_manifests_mutex); + + std::erase_if(export_manifests, [&](const auto & manifest) + { + if (manifest.transaction_id == transaction_id) + { + if (manifest.task) + manifest.task->cancel(); + + return true; + } + return false; + }); +} + void MergeTreeData::movePartitionToShard(const ASTPtr & /*partition*/, bool /*move_part*/, const String & /*to*/, ContextPtr /*query_context*/) { throw Exception(ErrorCodes::NOT_IMPLEMENTED, "MOVE PARTITION TO SHARD is not supported by storage {}", getName()); @@ -7060,6 +7372,17 @@ Pipe MergeTreeData::alterPartition( } } break; + case PartitionCommand::EXPORT_PART: + { + exportPartToTable(command, query_context); + break; + } + + case PartitionCommand::EXPORT_PARTITION: + { + exportPartitionToTable(command, query_context); + break; + } case PartitionCommand::DROP_DETACHED_PARTITION: dropDetached(command.partition, command.part, query_context); @@ -9675,6 +9998,33 @@ std::pair MergeTreeData::cloneAn return std::make_pair(dst_data_part, std::move(temporary_directory_lock)); } +std::vector MergeTreeData::getExportsStatus() const +{ + std::lock_guard lock(export_manifests_mutex); + std::vector result; + + auto source_database = getStorageID().database_name; + auto source_table = getStorageID().table_name; + + for (const auto & manifest : export_manifests) + { + MergeTreeExportStatus status; + + status.source_database = source_database; + status.source_table = source_table; + const auto destination_storage_id = manifest.destination_storage_ptr->getStorageID(); + status.destination_database = destination_storage_id.database_name; + status.destination_table = destination_storage_id.table_name; + status.create_time = manifest.create_time; + status.part_name = manifest.data_part->name; + + result.emplace_back(std::move(status)); + } + + return result; +} + + bool MergeTreeData::canUseAdaptiveGranularity() const { const auto settings = getSettings(); @@ -9991,7 +10341,11 @@ void MergeTreeData::writePartLog( const MergeListEntry * merge_entry, std::shared_ptr profile_counters, const Strings & mutation_ids, +<<<<<<< HEAD const std::map & projections_duration_ms) +======= + const ExportsListEntry * exports_entry) +>>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) try { auto table_id = getStorageID(); @@ -10059,6 +10413,16 @@ try part_log_elem.rows = (*merge_entry)->rows_written; part_log_elem.peak_memory_usage = (*merge_entry)->getMemoryTracker().getPeak(); } + else if (exports_entry) + { + part_log_elem.rows_read = (*exports_entry)->rows_read; + part_log_elem.bytes_read_uncompressed = (*exports_entry)->bytes_read_uncompressed; + part_log_elem.peak_memory_usage = (*exports_entry)->getPeakMemoryUsage(); + part_log_elem.query_id = (*exports_entry)->query_id; + + /// no need to lock because at this point no one is writing to the destination file paths + part_log_elem.remote_file_paths = (*exports_entry)->destination_file_paths; + } if (profile_counters) { @@ -10343,6 +10707,10 @@ bool MergeTreeData::canUsePolymorphicParts() const return canUsePolymorphicParts(*getSettings(), unused); } +void MergeTreeData::startBackgroundMoves() +{ + background_moves_assignee.start(); +} void MergeTreeData::checkDropOrRenameCommandDoesntAffectInProgressMutations( const AlterCommand & command, const std::map & unfinished_mutations, ContextPtr local_context) const diff --git a/src/Storages/MergeTree/MergeTreeData.h b/src/Storages/MergeTree/MergeTreeData.h index 826f4eead8f5..a5d36ce83375 100644 --- a/src/Storages/MergeTree/MergeTreeData.h +++ b/src/Storages/MergeTree/MergeTreeData.h @@ -19,6 +19,7 @@ #include #include #include +#include #include #include #include @@ -38,6 +39,8 @@ #include #include #include +#include +#include #include #include @@ -1073,6 +1076,33 @@ class MergeTreeData : public WithMutableContext, public IStorage, public IBackgr /// Moves partition to specified Table void movePartitionToTable(const PartitionCommand & command, ContextPtr query_context); + void exportPartToTable(const PartitionCommand & command, ContextPtr query_context); + + void exportPartToTable( + const std::string & part_name, + const StoragePtr & destination_storage, + const String & transaction_id, + ContextPtr query_context, + const std::optional & iceberg_metadata_json = std::nullopt, + bool allow_outdated_parts = false, + std::function completion_callback = {}); + + void exportPartToTable( + const std::string & part_name, + const StorageID & destination_storage_id, + const String & transaction_id, + ContextPtr query_context, + const std::optional & iceberg_metadata_json = std::nullopt, + bool allow_outdated_parts = false, + std::function completion_callback = {}); + + void killExportPart(const String & transaction_id); + + virtual void exportPartitionToTable(const PartitionCommand &, ContextPtr) + { + throw Exception(ErrorCodes::NOT_IMPLEMENTED, "EXPORT PARTITION is not implemented for engine {}", getName()); + } + /// Checks that Partition could be dropped right now /// Otherwise - throws an exception with detailed information. /// We do not use mutex because it is not very important that the size could change during the operation. @@ -1164,6 +1194,7 @@ class MergeTreeData : public WithMutableContext, public IStorage, public IBackgr const WriteSettings & write_settings); virtual std::vector getMutationsStatus() const = 0; + std::vector getExportsStatus() const; /// Returns true if table can create new parts with adaptive granularity /// Has additional constraint in replicated version @@ -1365,8 +1396,14 @@ class MergeTreeData : public WithMutableContext, public IStorage, public IBackgr /// Mutex for currently_moving_parts mutable std::mutex moving_parts_mutex; +<<<<<<< HEAD /// Used for streaming queries registration. mutable StreamSubscriptionManager subscription_manager; +======= + mutable std::mutex export_manifests_mutex; + + std::set export_manifests; +>>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) PinnedPartUUIDsPtr getPinnedPartUUIDs() const; @@ -1464,10 +1501,15 @@ class MergeTreeData : public WithMutableContext, public IStorage, public IBackgr friend class IPartMetadataManager; friend class IMergedBlockOutputStream; // for access to log friend struct DataPartsLock; // for access to shared_parts_list/shared_ranges_in_parts +<<<<<<< HEAD friend class VersionMetadata; // for access to log friend class VersionMetadataOnDisk; // for access to log friend class VersionMetadataOnKeeper; // for access to log friend class MutationsState; // for access to log +======= + friend class ExportPartTask; + friend class ExportPartFromPartitionExportTask; +>>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) bool require_part_metadata; @@ -1516,6 +1558,8 @@ class MergeTreeData : public WithMutableContext, public IStorage, public IBackgr size_t getColumnsDescriptionsCacheSize() const; protected: + void startBackgroundMoves(); + /// Engine-specific methods BrokenPartCallback broken_part_callback; @@ -1787,8 +1831,13 @@ class MergeTreeData : public WithMutableContext, public IStorage, public IBackgr const DataPartsVector & source_parts, const MergeListEntry * merge_entry, std::shared_ptr profile_counters, +<<<<<<< HEAD const Strings & mutation_ids, const std::map & projections_duration_ms); +======= + const Strings & mutation_ids = {}, + const ExportsListEntry * exports_entry = nullptr); +>>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) /// If part is assigned to merge or mutation (possibly replicated) /// Should be overridden by children, because they can have different @@ -2012,8 +2061,6 @@ class MergeTreeData : public WithMutableContext, public IStorage, public IBackgr bool canUsePolymorphicParts(const MergeTreeSettings & settings, String & out_reason) const; - virtual void startBackgroundMovesIfNeeded() = 0; - bool allow_nullable_key = false; void addPartContributionToDataVolume(const DataPartPtr & part); diff --git a/src/Storages/MergeTree/MergeTreeExportManifest.h b/src/Storages/MergeTree/MergeTreeExportManifest.h new file mode 100644 index 000000000000..05506ecb004a --- /dev/null +++ b/src/Storages/MergeTree/MergeTreeExportManifest.h @@ -0,0 +1,50 @@ +#include +#include + +namespace DB +{ + +struct MergeTreeExportManifest +{ + using DataPartPtr = std::shared_ptr; + + + MergeTreeExportManifest( + const StorageID & destination_storage_id_, + const DataPartPtr & data_part_, + bool overwrite_file_if_exists_, + const FormatSettings & format_settings_) + : destination_storage_id(destination_storage_id_), + data_part(data_part_), + overwrite_file_if_exists(overwrite_file_if_exists_), + format_settings(format_settings_), + create_time(time(nullptr)) {} + + StorageID destination_storage_id; + DataPartPtr data_part; + bool overwrite_file_if_exists; + FormatSettings format_settings; + + time_t create_time; + mutable bool in_progress = false; + + bool operator<(const MergeTreeExportManifest & rhs) const + { + // Lexicographic comparison: first compare destination storage, then part name + auto lhs_storage = destination_storage_id.getQualifiedName(); + auto rhs_storage = rhs.destination_storage_id.getQualifiedName(); + + if (lhs_storage != rhs_storage) + return lhs_storage < rhs_storage; + + return data_part->name < rhs.data_part->name; + } + + bool operator==(const MergeTreeExportManifest & rhs) const + { + return destination_storage_id.getQualifiedName() == rhs.destination_storage_id.getQualifiedName() + && data_part->name == rhs.data_part->name; + } +}; + +} diff --git a/src/Storages/MergeTree/MergeTreePartExportManifest.h b/src/Storages/MergeTree/MergeTreePartExportManifest.h new file mode 100644 index 000000000000..08d73febf968 --- /dev/null +++ b/src/Storages/MergeTree/MergeTreePartExportManifest.h @@ -0,0 +1,98 @@ +#pragma once + +#include +#include +#include +#include +#include +#include +#include + +namespace DB +{ + +class Exception; + +class IExecutableTask; + +struct MergeTreePartExportManifest +{ + using FileAlreadyExistsPolicy = MergeTreePartExportFileAlreadyExistsPolicy; + + using DataPartPtr = std::shared_ptr; + + struct CompletionCallbackResult + { + private: + CompletionCallbackResult(bool success_, const std::vector & relative_paths_in_destination_storage_, std::optional exception_) + : success(success_), relative_paths_in_destination_storage(relative_paths_in_destination_storage_), exception(std::move(exception_)) {} + public: + + static CompletionCallbackResult createSuccess(const std::vector & relative_paths_in_destination_storage_) + { + return CompletionCallbackResult(true, relative_paths_in_destination_storage_, std::nullopt); + } + + static CompletionCallbackResult createFailure(Exception exception_) + { + return CompletionCallbackResult(false, {}, std::move(exception_)); + } + + bool success = false; + std::vector relative_paths_in_destination_storage; + std::optional exception; + }; + + MergeTreePartExportManifest( + const StoragePtr destination_storage_ptr_, + const DataPartPtr & data_part_, + const String & transaction_id_, + const String & query_id_, + FileAlreadyExistsPolicy file_already_exists_policy_, + const Settings & settings_, + const StorageMetadataPtr & metadata_snapshot_, + const String & iceberg_metadata_json_, + std::function completion_callback_ = {}) + : destination_storage_ptr(destination_storage_ptr_), + data_part(data_part_), + transaction_id(transaction_id_), + query_id(query_id_), + file_already_exists_policy(file_already_exists_policy_), + settings(settings_), + metadata_snapshot(metadata_snapshot_), + iceberg_metadata_json(iceberg_metadata_json_), + completion_callback(completion_callback_), + create_time(time(nullptr)) {} + + StoragePtr destination_storage_ptr; + DataPartPtr data_part; + /// Used for killing the export. + String transaction_id; + String query_id; + FileAlreadyExistsPolicy file_already_exists_policy; + Settings settings; + + /// Metadata snapshot captured at the time of query validation to prevent race conditions with mutations + /// Otherwise the export could fail if the schema changes between validation and execution + StorageMetadataPtr metadata_snapshot; + + String iceberg_metadata_json; + + std::function completion_callback; + + time_t create_time; + /// Required to cancel export tasks + mutable std::shared_ptr task = nullptr; + + bool operator<(const MergeTreePartExportManifest & rhs) const + { + return data_part->name < rhs.data_part->name; + } + + bool operator==(const MergeTreePartExportManifest & rhs) const + { + return data_part->name == rhs.data_part->name; + } +}; + +} diff --git a/src/Storages/MergeTree/MergeTreePartExportStatus.h b/src/Storages/MergeTree/MergeTreePartExportStatus.h new file mode 100644 index 000000000000..e71a2f15e6ed --- /dev/null +++ b/src/Storages/MergeTree/MergeTreePartExportStatus.h @@ -0,0 +1,20 @@ +#pragma once + +#include +#include + + +namespace DB +{ + +struct MergeTreeExportStatus +{ + String source_database; + String source_table; + String destination_database; + String destination_table; + time_t create_time = 0; + std::string part_name; +}; + +} diff --git a/src/Storages/MergeTree/MergeTreePartition.cpp b/src/Storages/MergeTree/MergeTreePartition.cpp index 12dfce7728a8..341e9d1a55e3 100644 --- a/src/Storages/MergeTree/MergeTreePartition.cpp +++ b/src/Storages/MergeTree/MergeTreePartition.cpp @@ -507,6 +507,22 @@ void MergeTreePartition::create(const StorageMetadataPtr & metadata_snapshot, Bl } } +Block MergeTreePartition::getBlockWithPartitionValues(const NamesAndTypesList & partition_columns) const +{ + chassert(partition_columns.size() == value.size()); + + Block result; + + std::size_t i = 0; + for (const auto & partition_column : partition_columns) + { + auto column = partition_column.type->createColumnConst(1, value[i++]); + result.insert({column, partition_column.type, partition_column.name}); + } + + return result; +} + NamesAndTypesList MergeTreePartition::executePartitionByExpression(const StorageMetadataPtr & metadata_snapshot, Block & block, ContextPtr context) { auto adjusted_partition_key = adjustPartitionKey(metadata_snapshot, context); diff --git a/src/Storages/MergeTree/MergeTreePartition.h b/src/Storages/MergeTree/MergeTreePartition.h index 17936fc78e31..964a7bcdc0ae 100644 --- a/src/Storages/MergeTree/MergeTreePartition.h +++ b/src/Storages/MergeTree/MergeTreePartition.h @@ -60,6 +60,8 @@ struct MergeTreePartition void create(const StorageMetadataPtr & metadata_snapshot, Block block, size_t row, ContextPtr context); + Block getBlockWithPartitionValues(const NamesAndTypesList & partition_columns) const; + /// Adjust partition key and execute its expression on block. Return sample block according to used expression. static NamesAndTypesList executePartitionByExpression(const StorageMetadataPtr & metadata_snapshot, Block & block, ContextPtr context); diff --git a/src/Storages/MergeTree/MergeTreeSequentialSource.cpp b/src/Storages/MergeTree/MergeTreeSequentialSource.cpp index 3b02eb002daa..8a4ffc4ca673 100644 --- a/src/Storages/MergeTree/MergeTreeSequentialSource.cpp +++ b/src/Storages/MergeTree/MergeTreeSequentialSource.cpp @@ -169,6 +169,10 @@ MergeTreeSequentialSource::MergeTreeSequentialSource( addThrottler(read_settings.remote_throttler, context->getMergesThrottler()); addThrottler(read_settings.local_throttler, context->getMergesThrottler()); break; + case Export: + addThrottler(read_settings.local_throttler, context->getExportsThrottler()); + addThrottler(read_settings.remote_throttler, context->getExportsThrottler()); + break; } MergeTreeReadTask::Extras extras = diff --git a/src/Storages/MergeTree/MergeTreeSequentialSource.h b/src/Storages/MergeTree/MergeTreeSequentialSource.h index abba230d9e79..a858adf33bb5 100644 --- a/src/Storages/MergeTree/MergeTreeSequentialSource.h +++ b/src/Storages/MergeTree/MergeTreeSequentialSource.h @@ -15,6 +15,7 @@ enum MergeTreeSequentialSourceType { Mutation, Merge, + Export, }; /// Create stream for reading single part from MergeTree. diff --git a/src/Storages/MergeTree/ReplicatedMergeTreeRestartingThread.cpp b/src/Storages/MergeTree/ReplicatedMergeTreeRestartingThread.cpp index 99867ac0c711..44ecfeec1176 100644 --- a/src/Storages/MergeTree/ReplicatedMergeTreeRestartingThread.cpp +++ b/src/Storages/MergeTree/ReplicatedMergeTreeRestartingThread.cpp @@ -13,6 +13,7 @@ #include #include #include +#include namespace CurrentMetrics @@ -34,6 +35,11 @@ namespace MergeTreeSetting extern const MergeTreeSettingsSeconds zookeeper_session_expiration_check_period; } +namespace ServerSetting +{ + extern const ServerSettingsBool allow_experimental_export_merge_tree_partition; +} + namespace ErrorCodes { extern const int REPLICA_IS_ALREADY_ACTIVE; @@ -180,6 +186,14 @@ bool ReplicatedMergeTreeRestartingThread::runImpl() storage.mutations_updating_task->activateAndSchedule(); storage.mutations_finalizing_task->activateAndSchedule(); storage.merge_selecting_task->activateAndSchedule(); + + if (storage.getContext()->getServerSettings()[ServerSetting::allow_experimental_export_merge_tree_partition]) + { + storage.export_merge_tree_partition_updating_task->activateAndSchedule(); + storage.export_merge_tree_partition_select_task->activateAndSchedule(); + storage.export_merge_tree_partition_status_handling_task->activateAndSchedule(); + } + storage.cleanup_thread.start(); storage.async_block_ids_cache.start(); storage.part_check_thread.start(); @@ -187,6 +201,7 @@ bool ReplicatedMergeTreeRestartingThread::runImpl() if (storage.getContext()->getServerSettings()[ServerSetting::insert_deduplication_version].value != InsertDeduplicationVersions::OLD_SEPARATE_HASHES) storage.deduplication_hashes_cache.start(); + LOG_DEBUG(log, "Table started successfully"); return true; } diff --git a/src/Storages/MergeTree/tests/gtest_export_partition_ordering.cpp b/src/Storages/MergeTree/tests/gtest_export_partition_ordering.cpp new file mode 100644 index 000000000000..c9e3ffd9eef9 --- /dev/null +++ b/src/Storages/MergeTree/tests/gtest_export_partition_ordering.cpp @@ -0,0 +1,75 @@ +#include +#include + +namespace DB +{ + +class ExportPartitionOrderingTest : public ::testing::Test +{ +protected: + ExportPartitionTaskEntriesContainer container; + ExportPartitionTaskEntriesContainer::index::type & by_key; + ExportPartitionTaskEntriesContainer::index::type & by_create_time; + + ExportPartitionOrderingTest() + : by_key(container.get()) + , by_create_time(container.get()) + { + } +}; + +TEST_F(ExportPartitionOrderingTest, IterationOrderMatchesCreateTime) +{ + time_t base_time = 1000; + + ExportReplicatedMergeTreePartitionManifest manifest1; + manifest1.partition_id = "2020"; + manifest1.destination_database = "db1"; + manifest1.destination_table = "table1"; + manifest1.transaction_id = "tx1"; + manifest1.create_time = base_time + 300; // Latest + + ExportReplicatedMergeTreePartitionManifest manifest2; + manifest2.partition_id = "2021"; + manifest2.destination_database = "db1"; + manifest2.destination_table = "table1"; + manifest2.transaction_id = "tx2"; + manifest2.create_time = base_time + 100; // Middle + + ExportReplicatedMergeTreePartitionManifest manifest3; + manifest3.partition_id = "2022"; + manifest3.destination_database = "db1"; + manifest3.destination_table = "table1"; + manifest3.transaction_id = "tx3"; + manifest3.create_time = base_time; // Oldest + + ExportReplicatedMergeTreePartitionTaskEntry entry1{manifest1, ExportReplicatedMergeTreePartitionTaskEntry::Status::PENDING, {}}; + ExportReplicatedMergeTreePartitionTaskEntry entry2{manifest2, ExportReplicatedMergeTreePartitionTaskEntry::Status::PENDING, {}}; + ExportReplicatedMergeTreePartitionTaskEntry entry3{manifest3, ExportReplicatedMergeTreePartitionTaskEntry::Status::PENDING, {}}; + + // Insert in reverse order + by_key.insert(entry1); + by_key.insert(entry2); + by_key.insert(entry3); + + // Verify iteration order matches create_time (ascending) + auto it = by_create_time.begin(); + ASSERT_NE(it, by_create_time.end()); + EXPECT_EQ(it->manifest.partition_id, "2022"); // Oldest first + EXPECT_EQ(it->manifest.create_time, base_time); + + ++it; + ASSERT_NE(it, by_create_time.end()); + EXPECT_EQ(it->manifest.partition_id, "2021"); + EXPECT_EQ(it->manifest.create_time, base_time + 100); + + ++it; + ASSERT_NE(it, by_create_time.end()); + EXPECT_EQ(it->manifest.partition_id, "2020"); + EXPECT_EQ(it->manifest.create_time, base_time + 300); + + ++it; + EXPECT_EQ(it, by_create_time.end()); +} + +} diff --git a/src/Storages/ObjectStorage/DataLakes/IDataLakeMetadata.h b/src/Storages/ObjectStorage/DataLakes/IDataLakeMetadata.h index 389232fb1d16..3ce4f9f887ae 100644 --- a/src/Storages/ObjectStorage/DataLakes/IDataLakeMetadata.h +++ b/src/Storages/ObjectStorage/DataLakes/IDataLakeMetadata.h @@ -3,8 +3,14 @@ #include #include +#include #include #include +<<<<<<< HEAD +======= +#include +#include +>>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) #include #include #include @@ -135,6 +141,37 @@ class IDataLakeMetadata : boost::noncopyable throwNotImplemented("write"); } + virtual bool supportsImport(ContextPtr) const + { + return false; + } + + virtual SinkToStoragePtr import( + std::shared_ptr /* catalog */, + const std::function & /* new_file_path_callback */, + SharedHeader /* sample_block */, + const std::string & /* iceberg_metadata_json_string */, + const std::optional & /* format_settings_ */, + ContextPtr /* context */) + { + throwNotImplemented("import"); + } + + virtual void commitExportPartitionTransaction( + std::shared_ptr /* catalog */, + const StorageID & /* table_id */, + const String & /* transaction_id */, + Int64 /* original_schema_id */, + Int64 /* partition_spec_id */, + const std::vector & /* partition_values */, + SharedHeader /* sample_block */, + const std::vector & /* data_file_paths */, + StorageObjectStorageConfigurationPtr /* configuration */, + ContextPtr /* context */) + { + throwNotImplemented("commitExportPartitionTransaction"); + } + virtual bool optimize( const StorageMetadataPtr & /*metadata_snapshot*/, ContextPtr /*context*/, const std::optional & /*format_settings*/) { diff --git a/src/Storages/ObjectStorage/DataLakes/Iceberg/AvroSchema.h b/src/Storages/ObjectStorage/DataLakes/Iceberg/AvroSchema.h index 4e70988735b3..97c832760a11 100644 --- a/src/Storages/ObjectStorage/DataLakes/Iceberg/AvroSchema.h +++ b/src/Storages/ObjectStorage/DataLakes/Iceberg/AvroSchema.h @@ -581,4 +581,42 @@ static constexpr const char * manifest_entry_v2_schema = R"( } )"; +/// Schema for the per-data-file sidecar Avro files written alongside every data file +/// during import/export. The sidecar carries the row count and byte size that cannot +/// be cheaply inferred from the data file itself without a full scan. +static constexpr const char * data_file_sidecar_schema = R"( +{ + "type": "record", + "name": "data_file_metadata", + "fields": [ + {"name": "record_count", "type": "long"}, + {"name": "file_size_in_bytes", "type": "long"}, + { + "name": "column_sizes", + "type": {"type": "array", "items": {"type": "record", "name": "cs_entry", + "fields": [{"name": "key", "type": "int"}, {"name": "value", "type": "long"}]}}, + "default": [] + }, + { + "name": "null_value_counts", + "type": {"type": "array", "items": {"type": "record", "name": "nvc_entry", + "fields": [{"name": "key", "type": "int"}, {"name": "value", "type": "long"}]}}, + "default": [] + }, + { + "name": "lower_bounds", + "type": {"type": "array", "items": {"type": "record", "name": "lb_entry", + "fields": [{"name": "key", "type": "int"}, {"name": "value", "type": "bytes"}]}}, + "default": [] + }, + { + "name": "upper_bounds", + "type": {"type": "array", "items": {"type": "record", "name": "ub_entry", + "fields": [{"name": "key", "type": "int"}, {"name": "value", "type": "bytes"}]}}, + "default": [] + } + ] +} +)"; + } diff --git a/src/Storages/ObjectStorage/DataLakes/Iceberg/Constant.h b/src/Storages/ObjectStorage/DataLakes/Iceberg/Constant.h index 6495b22f7cf0..505e1f02f12a 100644 --- a/src/Storages/ObjectStorage/DataLakes/Iceberg/Constant.h +++ b/src/Storages/ObjectStorage/DataLakes/Iceberg/Constant.h @@ -163,6 +163,7 @@ DEFINE_ICEBERG_FIELD_ALIAS(max_ref_age_ms, history.expire.max-ref-age-ms); DEFINE_ICEBERG_FIELD_ALIAS(ref_min_snapshots_to_keep, min-snapshots-to-keep); DEFINE_ICEBERG_FIELD_ALIAS(ref_max_snapshot_age_ms, max-snapshot-age-ms); DEFINE_ICEBERG_FIELD_ALIAS(ref_max_ref_age_ms, max-ref-age-ms); +DEFINE_ICEBERG_FIELD_ALIAS(clickhouse_export_partition_transaction_id, clickhouse.export-partition-transaction-id); /// These are compound fields like `data_file.file_path`, we use prefix 'c_' to distinguish them. DEFINE_ICEBERG_FIELD_COMPOUND(data_file, file_path); DEFINE_ICEBERG_FIELD_COMPOUND(data_file, file_format); diff --git a/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergDataFileEntry.h b/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergDataFileEntry.h new file mode 100644 index 000000000000..61fa7be9005b --- /dev/null +++ b/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergDataFileEntry.h @@ -0,0 +1,50 @@ +#pragma once + +#include "config.h" + +#if USE_AVRO + +#include +#include +#include +#include + +namespace DB +{ + +/// Column-level statistics for a single Iceberg data file stored in Iceberg wire format. +/// Bounds are pre-serialized to bytes so the struct can be persisted to sidecar Avro files +/// and used directly at manifest-commit time without requiring the original ClickHouse +/// DataFileStatistics or a live Block schema. +struct IcebergSerializedFileStats +{ + Int64 record_count = 0; + Int64 file_size_in_bytes = 0; + + /// field_id → compressed byte size of column in the file + std::vector> column_sizes; + /// field_id → number of null values in the file + std::vector> null_value_counts; + /// field_id → Iceberg-serialized lower bound (same binary format as manifest) + std::vector>> lower_bounds; + /// field_id → Iceberg-serialized upper bound (same binary format as manifest) + std::vector>> upper_bounds; +}; + +/// One entry describing a data file that will be registered in an Iceberg manifest. +/// Carries per-file statistics so that each manifest entry gets accurate metadata +/// (column sizes, null counts, min/max bounds, record count, file size). +struct IcebergDataFileEntry +{ + String path; + Int64 record_count = 0; + Int64 file_size_in_bytes = 0; + + /// Per-file column statistics (null counts, min/max bounds, column sizes). + /// Pass std::nullopt when statistics are not available or not yet computed. + std::optional statistics; +}; + +} + +#endif diff --git a/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergMetadata.cpp b/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergMetadata.cpp index 82adf64ab122..28158abdec65 100644 --- a/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergMetadata.cpp +++ b/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergMetadata.cpp @@ -54,6 +54,7 @@ #include #include +#include #include #include #include @@ -78,9 +79,13 @@ #include #include #include +#include #include +<<<<<<< HEAD #include +======= +>>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) #include #include #include @@ -137,6 +142,8 @@ extern const SettingsBool allow_experimental_expire_snapshots; extern const SettingsBool iceberg_delete_data_on_drop; } +static constexpr size_t MAX_TRANSACTION_RETRIES = 100; + namespace { String dumpMetadataObjectToString(const Poco::JSON::Object::Ptr & metadata_object) @@ -145,6 +152,51 @@ String dumpMetadataObjectToString(const Poco::JSON::Object::Ptr & metadata_objec Poco::JSON::Stringifier::stringify(metadata_object, oss); return removeEscapedSlashes(oss.str()); } + +/// Check if a previous attempt already committed this transaction the snapshot +/// (with our transaction_id embedded in its summary) is still present in the snapshots array +/// unless an external engine ran expireSnapshots in the meantime. If found, skip re-committing. +bool isExportPartitionTransactionAlreadyCommitted(const Poco::JSON::Object::Ptr & metadata, const String & transaction_id) +{ + const auto throw_error = [&](const std::string & missing_field_name) + { + throw Exception( + ErrorCodes::ICEBERG_SPECIFICATION_VIOLATION, + "No {} found in metadata for iceberg file while trying to commit export partition transaction", + missing_field_name); + }; + + const auto snapshots = metadata->getArray(Iceberg::f_snapshots); + + if (!snapshots) + { + throw_error(Iceberg::f_snapshots); + } + + for (size_t i = 0; i < snapshots->size(); ++i) + { + const auto snap = snapshots->getObject(static_cast(i)); + const auto summary = snap->getObject(Iceberg::f_summary); + + if (!summary) + { + throw_error(Iceberg::f_summary); + } + + if (summary->has(Iceberg::f_clickhouse_export_partition_transaction_id)) + { + const auto tid = summary->getValue(Iceberg::f_clickhouse_export_partition_transaction_id); + + if (tid == transaction_id) + { + return true; + } + } + } + + return false; +} + } @@ -1365,6 +1417,511 @@ KeyDescription IcebergMetadata::getSortingKey(ContextPtr local_context, TableSta return result; } +SinkToStoragePtr IcebergMetadata::import( + std::shared_ptr catalog, + const std::function & new_file_path_callback, + SharedHeader sample_block, + const std::string & iceberg_metadata_json_string, + const std::optional & format_settings, + ContextPtr context) +{ + Poco::JSON::Parser parser; /// For some reason base/base/JSON.h can not parse this json file + Poco::Dynamic::Var json = parser.parse(iceberg_metadata_json_string); + Poco::JSON::Object::Ptr metadata_json = json.extract(); + + return std::make_shared( + catalog, persistent_components, metadata_json, object_storage, + context, format_settings, write_format, sample_block, data_lake_settings, new_file_path_callback); +} + +namespace FailPoints +{ + extern const char iceberg_writes_cleanup[]; + extern const char iceberg_writes_non_retry_cleanup[]; + extern const char iceberg_writes_post_publish_throw[]; +} + +namespace +{ + +/// Find the partition spec object with the given spec-id inside a metadata JSON document. +/// Throws BAD_ARGUMENTS if the spec is not found (indicates metadata/spec-id mismatch). +Poco::JSON::Object::Ptr lookupPartitionSpec(const Poco::JSON::Object::Ptr & meta, Int64 spec_id) +{ + auto specs = meta->getArray(Iceberg::f_partition_specs); + for (size_t i = 0; i < specs->size(); ++i) + { + auto spec = specs->getObject(static_cast(i)); + if (spec->getValue(Iceberg::f_spec_id) == spec_id) + return spec; + } + throw Exception(ErrorCodes::BAD_ARGUMENTS, + "Partition spec with id {} not found in table metadata", spec_id); +} + +Poco::JSON::Object::Ptr lookupSchema(const Poco::JSON::Object::Ptr & meta, Int64 schema_id) +{ + auto schemas = meta->getArray(Iceberg::f_schemas); + for (size_t i = 0; i < schemas->size(); ++i) + { + auto schema = schemas->getObject(static_cast(i)); + if (schema->getValue(Iceberg::f_schema_id) == schema_id) + return schema; + } + + throw Exception(ErrorCodes::BAD_ARGUMENTS, + "Schema with id {} not found in table metadata", schema_id); +} + +} + +bool IcebergMetadata::commitImportPartitionTransactionImpl( + FileNamesGenerator & filename_generator, + Poco::JSON::Object::Ptr & metadata, + Poco::JSON::Object::Ptr & partition_spec, + const String & transaction_id, + Int64 original_schema_id, + Int64 partition_spec_id, + const std::vector & partition_values, + const std::vector & partition_columns, + const std::vector & partition_types, + SharedHeader sample_block, + const std::vector & data_file_paths, + const std::vector & per_file_stats, + Int64 total_data_files, + Int64 total_rows, + Int64 total_chunks_size, + std::shared_ptr catalog, + const StorageID & table_id, + const String & blob_storage_type_name, + const String & blob_storage_namespace_name, + ContextPtr context) +{ + /// this check also exists here because the metadata might have been updated upon retry attempts. + if (isExportPartitionTransactionAlreadyCommitted(metadata, transaction_id)) + { + LOG_INFO(log, + "Export transaction {} already committed, skipping re-commit", + transaction_id); + return true; + } + + CompressionMethod metadata_compression_method = persistent_components.metadata_compression_method; + + auto [metadata_name, storage_metadata_name] = filename_generator.generateMetadataName(); + + Int64 parent_snapshot = -1; + if (metadata->has(Iceberg::f_current_snapshot_id)) + parent_snapshot = metadata->getValue(Iceberg::f_current_snapshot_id); + + auto [new_snapshot, manifest_list_name, storage_manifest_list_name] = MetadataGenerator(metadata).generateNextMetadata( + filename_generator, metadata_name, parent_snapshot, total_data_files, total_rows, total_chunks_size, total_data_files, /* added_delete_files */0, /* num_deleted_rows */0); + + /// Embed the stable transaction identifier in the snapshot summary so that a retry after crash + /// can detect the commit already happened by scanning the live snapshots array, without extra S3 + /// files. The field is a ClickHouse extension; Spark/Flink readers ignore unknown summary keys. + new_snapshot->getObject(Iceberg::f_summary)->set( + Iceberg::f_clickhouse_export_partition_transaction_id, transaction_id); + + String manifest_entry_name; + String storage_manifest_entry_name; + Int64 manifest_lengths = 0; + + /// Tracks whether the snapshot has become visible to readers. + /// For the file-based layout that happens as soon as writeMetadataFileAndVersionHint + /// succeeds; for a catalog layout it happens when catalog->updateMetadata succeeds. + /// Once published, the manifest entry / manifest list are referenced by the live + /// snapshot and must NOT be deleted by the outer failure cleanup, otherwise the + /// already-published snapshot becomes unreadable. + bool published = false; + + auto cleanup = [&](bool retry_because_of_metadata_conflict) + { + /// We can't cleanup the data files upon retry even if retry_because_of_metadata_conflict == false + /// because this replica or some other replica might attempt to commit the same transaction later + /// todo arthur: in the future, we should consider failing the entire task if retry_because_of_metadata_conflict = true + + object_storage->removeObjectIfExists(StoredObject(storage_manifest_entry_name)); + object_storage->removeObjectIfExists(StoredObject(storage_manifest_list_name)); + + if (retry_because_of_metadata_conflict) + { + MetadataFileWithInfo latest_metadata_file_info; + if (catalog && catalog->isTransactional()) + { + const auto & [namespace_name, table_name] = DataLake::parseTableName(table_id.getTableName()); + DataLake::TableMetadata table_metadata = DataLake::TableMetadata().withLocation().withDataLakeSpecificProperties(); + catalog->getTableMetadata(namespace_name, table_name, table_metadata); + + auto table_specific_properties = table_metadata.getDataLakeSpecificProperties(); + if (!table_specific_properties.has_value() || table_specific_properties->iceberg_metadata_file_location.empty()) + throw Exception(ErrorCodes::LOGICAL_ERROR, "Catalog didn't return iceberg metadata location for table {}.{}", namespace_name, table_name); + + String metadata_path = table_metadata.getMetadataLocation(table_specific_properties->iceberg_metadata_file_location); + if (!metadata_path.starts_with(persistent_components.table_path)) + metadata_path = std::filesystem::path(persistent_components.table_path) / metadata_path; + latest_metadata_file_info = Iceberg::getMetadataFileAndVersion(metadata_path); + } + else + { + latest_metadata_file_info = getLatestOrExplicitMetadataFileAndVersion( + object_storage, + persistent_components.table_path, + data_lake_settings, + persistent_components.metadata_cache, + context, + getLogger("IcebergWrites").get(), + persistent_components.table_uuid, + true); + } + + auto [last_version, metadata_path, compression_method] = latest_metadata_file_info; + + LOG_DEBUG(log, "Rereading metadata file {} with version {}", metadata_path, last_version); + + metadata_compression_method = compression_method; + filename_generator.setVersion(last_version + 1); + + metadata = getMetadataJSONObject( + metadata_path, + object_storage, + persistent_components.metadata_cache, + context, + getLogger("IcebergMetadata"), + compression_method, + persistent_components.table_uuid); + + /// For the export path the schema and partition spec are fixed at the start of the + /// operation (saved in ZooKeeper). If either changed we must fail immediately — + /// the caller has to restart the export from scratch. + const auto new_schema_id = metadata->getValue(Iceberg::f_current_schema_id); + if (new_schema_id != original_schema_id) + throw Exception(ErrorCodes::NOT_IMPLEMENTED, + "Table schema changed during export (expected schema {}, got {}). Restart the export operation.", + original_schema_id, new_schema_id); + + const Int64 new_partition_spec_id = metadata->getValue(Iceberg::f_default_spec_id); + if (new_partition_spec_id != partition_spec_id) + throw Exception(ErrorCodes::NOT_IMPLEMENTED, + "Partition spec changed during export (expected spec {}, got {}). Restart the export operation.", + partition_spec_id, new_partition_spec_id); + + partition_spec = lookupPartitionSpec(metadata, partition_spec_id); + + /// partition_values, partition_columns, partition_types, and + /// data_file_paths are all fixed from the saved state — no update needed. + } + }; + + try + { + { + auto result = filename_generator.generateManifestEntryName(); + manifest_entry_name = result.path_in_metadata; + storage_manifest_entry_name = result.path_in_storage; + } + + auto buffer_manifest_entry = object_storage->writeObject( + StoredObject(storage_manifest_entry_name), WriteMode::Rewrite, std::nullopt, DBMS_DEFAULT_BUFFER_SIZE, context->getWriteSettings()); + + try + { + fiu_do_on(FailPoints::iceberg_writes_non_retry_cleanup, + { + throw Exception(ErrorCodes::BAD_ARGUMENTS, "Failpoint for cleanup enabled"); + }); + + generateManifestFile( + metadata, + partition_columns, + partition_values, + partition_types, + data_file_paths, + std::nullopt, /// per_file_stats is filled, no need for the generic aggregate + sample_block, + new_snapshot, + write_format, + partition_spec, + partition_spec_id, + *buffer_manifest_entry, + Iceberg::FileContentType::DATA, + per_file_stats); + buffer_manifest_entry->finalize(); + manifest_lengths += buffer_manifest_entry->count(); + } + catch (...) + { + cleanup(false); + throw; + } + + { + auto buffer_manifest_list = object_storage->writeObject( + StoredObject(storage_manifest_list_name), WriteMode::Rewrite, std::nullopt, DBMS_DEFAULT_BUFFER_SIZE, context->getWriteSettings()); + + try + { + generateManifestList( + filename_generator, metadata, object_storage, context, {manifest_entry_name}, new_snapshot, manifest_lengths, *buffer_manifest_list, Iceberg::FileContentType::DATA, true); + buffer_manifest_list->finalize(); + } + catch (...) + { + cleanup(false); + throw; + } + } + + { + std::ostringstream oss; // STYLE_CHECK_ALLOW_STD_STRING_STREAM + Poco::JSON::Stringifier::stringify(metadata, oss, 4); + std::string json_representation = removeEscapedSlashes(oss.str()); + + LOG_DEBUG(log, "Writing new metadata file {}", storage_metadata_name); + auto hint = filename_generator.generateVersionHint(); + if (!writeMetadataFileAndVersionHint( + storage_metadata_name, + json_representation, + hint.path_in_storage, + storage_metadata_name, + object_storage, + context, + metadata_compression_method, + data_lake_settings[DataLakeStorageSetting::iceberg_use_version_hint])) + { + LOG_DEBUG(log, "Failed to write metadata {}, retrying", storage_metadata_name); + cleanup(true); + return false; + } + + LOG_DEBUG(log, "Metadata file {} written", storage_metadata_name); + + if (catalog) + { + String catalog_filename = metadata_name; + if (!catalog_filename.starts_with(blob_storage_type_name)) + catalog_filename = blob_storage_type_name + "://" + blob_storage_namespace_name + "/" + metadata_name; + + const auto & [namespace_name, table_name] = DataLake::parseTableName(table_id.getTableName()); + if (!catalog->updateMetadata(namespace_name, table_name, catalog_filename, new_snapshot)) + { + cleanup(true); + return false; + } + + /// Catalog has accepted the commit - the new snapshot is now live and references + /// storage_manifest_entry_name / storage_manifest_list_name. From here on, any + /// failure must NOT delete those files. + published = true; + } + else + { + /// File-based layout: the snapshot becomes visible via the metadata file and + /// version hint that were just written above. From here on, any failure must + /// NOT delete manifest entry / manifest list. + published = true; + } + } + + /// Fault-injection hook that simulates an exception in the trailing post-publish + /// region (e.g. failure in metadata-cache invalidation). Must be placed AFTER + /// `published = true` to exercise the exception-safety guard in the outer catch. + fiu_do_on(FailPoints::iceberg_writes_post_publish_throw, + { + throw Exception(ErrorCodes::BAD_ARGUMENTS, "Failpoint iceberg_writes_post_publish_throw enabled"); + }); + + if (persistent_components.metadata_cache) + { + /// If there's an active metadata cache + /// We can't just cache 'our' written version as latest, because it could've been overwritten by a concurrent catalog update + /// This is why, we are safely invalidating the cache, and the very next reader will get the most up-to-date latest version + persistent_components.metadata_cache->remove(persistent_components.table_path); + if (persistent_components.table_uuid) + persistent_components.metadata_cache->remove(*persistent_components.table_uuid); + } + } + catch (...) + { + if (published) + { + /// Commit has already become visible to readers. The failure is in trailing + /// post-publish work (e.g. metadata-cache invalidation). Running cleanup() + /// here would delete manifest files referenced by the published snapshot + /// and corrupt it. Log and swallow - any transient state (stale cache) + /// is self-healing on subsequent reads. + tryLogCurrentException(log, + "Post-publish work failed after Iceberg snapshot was committed; " + "skipping manifest cleanup to preserve published snapshot"); + return true; + } + + LOG_ERROR(log, "Failed to commit import partition transaction: {}", getCurrentExceptionMessage(false)); + cleanup(false); + throw; + } + + return true; +} + +void IcebergMetadata::commitExportPartitionTransaction( + std::shared_ptr catalog, + const StorageID & table_id, + const String & transaction_id, + Int64 original_schema_id, + Int64 partition_spec_id, + const std::vector & partition_values, + SharedHeader sample_block, + const std::vector & data_file_paths, + StorageObjectStorageConfigurationPtr configuration, + ContextPtr context) +{ + + MetadataFileWithInfo updated_metadata_file_info = getLatestOrExplicitMetadataFileAndVersion( + object_storage, + persistent_components.table_path, + data_lake_settings, + persistent_components.metadata_cache, + context, + getLogger("IcebergMetadata").get(), + persistent_components.table_uuid, + true); + + /// Latest metadata is ALWAYS necessary to commit - but we abort in case schema or partition spec changed + Poco::JSON::Object::Ptr metadata = getMetadataJSONObject( + updated_metadata_file_info.path, + object_storage, + persistent_components.metadata_cache, + context, + getLogger("IcebergMetadata"), + updated_metadata_file_info.compression_method, + persistent_components.table_uuid); + + if (isExportPartitionTransactionAlreadyCommitted(metadata, transaction_id)) + { + LOG_INFO(log, + "Export transaction {} already committed, skipping re-commit", + transaction_id); + return; + } + + /// Fail fast if the table schema or partition spec changed between export-start and commit. + /// The exported data files and partition values were produced against the original spec; + const auto latest_schema_id = metadata->getValue(Iceberg::f_current_schema_id); + if (latest_schema_id != original_schema_id) + throw Exception(ErrorCodes::NOT_IMPLEMENTED, + "Table schema changed before export could commit (expected schema {}, got {}). " + "Restart the export operation.", + original_schema_id, latest_schema_id); + + const auto latest_spec_id = metadata->getValue(Iceberg::f_default_spec_id); + if (latest_spec_id != partition_spec_id) + throw Exception(ErrorCodes::NOT_IMPLEMENTED, + "Partition spec changed before export could commit (expected spec {}, got {}). " + "Restart the export operation.", + partition_spec_id, latest_spec_id); + + /// Derive partition_columns and partition_types from the schema and partition spec. + /// The IDs are validated equal above so derivation from the latest metadata yields + /// the same result as from the original ZK-pinned snapshot. + + const auto schema = lookupSchema(metadata, original_schema_id); + + auto partition_spec = lookupPartitionSpec(metadata, partition_spec_id); + + ChunkPartitioner partitioner(partition_spec->getArray(Iceberg::f_fields), schema, context, sample_block); + + const auto partition_columns = partitioner.getColumns(); + const auto partition_types = partitioner.getResultTypes(); + + const auto metadata_compression_method = persistent_components.metadata_compression_method; + auto config_path = persistent_components.table_path; + if (config_path.empty() || config_path.back() != '/') + config_path += "/"; + if (!config_path.starts_with('/')) + config_path = '/' + config_path; + + FileNamesGenerator filename_generator; + if (!context->getSettingsRef()[Setting::write_full_path_in_iceberg_metadata]) + { + filename_generator = FileNamesGenerator( + config_path, config_path, (catalog != nullptr && catalog->isTransactional()), metadata_compression_method, write_format); + } + else + { + auto bucket = metadata->getValue(Iceberg::f_location); + if (bucket.empty() || bucket.back() != '/') + bucket += "/"; + filename_generator = FileNamesGenerator( + bucket, config_path, (catalog != nullptr && catalog->isTransactional()), metadata_compression_method, write_format); + } + filename_generator.setVersion(updated_metadata_file_info.version + 1); + + /// Load per-file sidecar stats, necessary to populate the manifest file stats. + std::vector per_file_stats; + const Int64 total_data_files = static_cast(data_file_paths.size()); + Int64 total_rows = 0; + Int64 total_chunks_size = 0; + per_file_stats.reserve(data_file_paths.size()); + for (const auto & path : data_file_paths) + { + const auto sidecar_path = getIcebergExportPartSidecarStoragePath(path); + auto stats = readDataFileSidecar(sidecar_path, object_storage, context); + total_rows += stats.record_count; + total_chunks_size += stats.file_size_in_bytes; + + per_file_stats.push_back(std::move(stats)); + } + + size_t attempt = 0; + while (attempt < MAX_TRANSACTION_RETRIES) + { + if (commitImportPartitionTransactionImpl( + filename_generator, + metadata, + partition_spec, + transaction_id, + original_schema_id, + partition_spec_id, + partition_values, + partition_columns, + partition_types, + sample_block, + data_file_paths, + per_file_stats, + total_data_files, + total_rows, + total_chunks_size, + catalog, + table_id, + configuration->getTypeName(), + configuration->getNamespace(), + context)) + { + return; + } + + ++attempt; + } + + throw Exception(ErrorCodes::NOT_IMPLEMENTED, + "Failed to commit export partition transaction after {} attempts due to repeated metadata conflicts.", + attempt); +} + +Poco::JSON::Object::Ptr IcebergMetadata::getMetadataJSON(ContextPtr local_context) const +{ + auto [actual_data_snapshot, actual_table_state_snapshot] = getRelevantState(local_context); + return getMetadataJSONObject( + actual_table_state_snapshot.metadata_file_path, + object_storage, + persistent_components.metadata_cache, + local_context, + log, + persistent_components.metadata_compression_method, + persistent_components.table_uuid); +} + } #endif diff --git a/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergMetadata.h b/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergMetadata.h index ee7b86b47791..8b81a9ab9e10 100644 --- a/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergMetadata.h +++ b/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergMetadata.h @@ -1,4 +1,6 @@ #pragma once +#include "Storages/ObjectStorage/DataLakes/Iceberg/ChunkPartitioner.h" +#include "Storages/ObjectStorage/DataLakes/Iceberg/FileNamesGenerator.h" #include "config.h" #if USE_AVRO @@ -22,6 +24,7 @@ #include #include +#include #include #include #include @@ -137,6 +140,38 @@ class IcebergMetadata : public IDataLakeMetadata ContextPtr context, std::shared_ptr catalog) override; + bool supportsImport(ContextPtr) const override { return true; } + + SinkToStoragePtr import( + std::shared_ptr catalog, + const std::function & new_file_path_callback, + SharedHeader sample_block, + const std::string & iceberg_metadata_json_string, + const std::optional & format_settings, + ContextPtr context) override; + + /// Commit an export-partition transaction. All parameters that are saved in ZooKeeper at the + /// start of the export operation (schema_id, partition_spec_id, partition_values, + /// partition_columns, partition_types) must be provided by the caller. + /// The partition spec object is derived from the metadata using partition_spec_id. + /// If the live metadata has diverged (schema or partition spec changed) the call throws + /// immediately — the caller must restart from scratch. + /// + /// data_file_paths contains the metadata-path for each exported data file (as recorded in + /// ZooKeeper). For every path a co-located sidecar Avro file (same path, ".avro" extension) + /// must exist in the object storage; it supplies record_count and file_size_in_bytes. + void commitExportPartitionTransaction( + std::shared_ptr catalog, + const StorageID & table_id, + const String & transaction_id, + Int64 original_schema_id, + Int64 partition_spec_id, + const std::vector & partition_values, + SharedHeader sample_block, + const std::vector & data_file_paths, + StorageObjectStorageConfigurationPtr configuration, + ContextPtr context) override; + CompressionMethod getCompressionMethod() const { return persistent_components.metadata_compression_method; } bool optimize(const StorageMetadataPtr & metadata_snapshot, ContextPtr context, const std::optional & format_settings) override; @@ -179,6 +214,14 @@ class IcebergMetadata : public IDataLakeMetadata void drop(ContextPtr context) override; +<<<<<<< HEAD +======= + std::optional partitionKey(ContextPtr) const override; + std::optional sortingKey(ContextPtr) const override; + + Poco::JSON::Object::Ptr getMetadataJSON(ContextPtr local_context) const; + +>>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) private: static Iceberg::PersistentTableComponents initializePersistentTableComponents( ObjectStoragePtr object_storage, @@ -198,6 +241,28 @@ class IcebergMetadata : public IDataLakeMetadata Iceberg::IcebergDataSnapshotPtr getRelevantDataSnapshotFromTableStateSnapshot(Iceberg::TableStateSnapshot table_state_snapshot, ContextPtr local_context) const; + bool commitImportPartitionTransactionImpl( + FileNamesGenerator & filename_generator, + Poco::JSON::Object::Ptr & metadata, + Poco::JSON::Object::Ptr & partition_spec, + const String & transaction_id, + Int64 original_schema_id, + Int64 partition_spec_id, + const std::vector & partition_values, + const std::vector & partition_columns, + const std::vector & partition_types, + SharedHeader sample_block, + const std::vector & data_file_paths, + const std::vector & per_file_stats, + Int64 total_data_files, + Int64 total_rows, + Int64 total_chunks_size, + std::shared_ptr catalog, + const StorageID & table_id, + const String & blob_storage_type_name, + const String & blob_storage_namespace_name, + ContextPtr context); + LoggerPtr log; const ObjectStoragePtr object_storage; const DB::Iceberg::PersistentTableComponents persistent_components; diff --git a/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergWrites.cpp b/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergWrites.cpp index 9529764f1fdd..d9c9cf1e755d 100644 --- a/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergWrites.cpp +++ b/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergWrites.cpp @@ -196,6 +196,16 @@ bool canWriteStatistics( } +String getIcebergExportPartSidecarStoragePath(const String & data_file_storage_path) +{ + static constexpr auto postfix = "_clickhouse_export_part_sidecar.avro"; + auto dot_pos = data_file_storage_path.rfind('.'); + auto slash_pos = data_file_storage_path.rfind('/'); + if (dot_pos != String::npos && (slash_pos == String::npos || dot_pos > slash_pos)) + return data_file_storage_path.substr(0, dot_pos) + postfix; + return data_file_storage_path + postfix; +} + String removeEscapedSlashes(const String & json_str) { size_t pos = json_str.find("\\/"); @@ -219,7 +229,165 @@ String removeEscapedSlashes(const String & json_str) return result; } +<<<<<<< HEAD static void extendSchemaForPartitions( +======= +IcebergSerializedFileStats readDataFileSidecar( + const String & sidecar_storage_path, + const ObjectStoragePtr & object_storage, + const ContextPtr & context) +{ + auto buf = object_storage->readObject(StoredObject(sidecar_storage_path), context->getReadSettings()); + auto input_stream = std::make_unique(*buf); + avro::DataFileReader reader(std::move(input_stream)); + + avro::GenericDatum datum(reader.readerSchema()); + if (!reader.read(datum)) + throw Exception( + ErrorCodes::BAD_ARGUMENTS, + "Data file sidecar '{}' contains no records", + sidecar_storage_path); + + const auto & record = datum.value(); + IcebergSerializedFileStats result; + result.record_count = record.field("record_count").value(); + result.file_size_in_bytes = record.field("file_size_in_bytes").value(); + + auto read_long_map = [&](const std::string & name, std::vector> & out) + { + const auto & arr = record.field(name).value().value(); + for (const auto & item : arr) + { + const auto & r = item.value(); + out.emplace_back(r.field("key").value(), r.field("value").value()); + } + }; + + auto read_bytes_map = [&](const std::string & name, std::vector>> & out) + { + const auto & arr = record.field(name).value().value(); + for (const auto & item : arr) + { + const auto & r = item.value(); + out.emplace_back(r.field("key").value(), r.field("value").value>()); + } + }; + + read_long_map("column_sizes", result.column_sizes); + read_long_map("null_value_counts", result.null_value_counts); + read_bytes_map("lower_bounds", result.lower_bounds); + read_bytes_map("upper_bounds", result.upper_bounds); + + return result; +} + +void writeDataFileSidecar( + const String & data_file_storage_path, + const IcebergSerializedFileStats & stats, + const ObjectStoragePtr & object_storage, + const ContextPtr & context) +{ + const String sidecar_path = getIcebergExportPartSidecarStoragePath(data_file_storage_path); + auto buf = object_storage->writeObject( + StoredObject(sidecar_path), WriteMode::Rewrite, std::nullopt, DBMS_DEFAULT_BUFFER_SIZE, context->getWriteSettings()); + + { + auto schema = avro::compileJsonSchemaFromString(data_file_sidecar_schema); + auto adapter = std::make_unique(*buf); + avro::DataFileWriter writer(std::move(adapter), schema); + + avro::GenericDatum datum(schema.root()); + avro::GenericRecord & rec = datum.value(); + rec.field("record_count") = avro::GenericDatum(stats.record_count); + rec.field("file_size_in_bytes") = avro::GenericDatum(stats.file_size_in_bytes); + + auto write_long_map = [&](const std::string & name, const std::vector> & entries) + { + auto & field = rec.field(name); + auto & arr = field.value(); + auto schema_element = arr.schema()->leafAt(0); + for (const auto & [k, v] : entries) + { + avro::GenericDatum item(schema_element); + auto & item_rec = item.value(); + item_rec.field("key") = avro::GenericDatum(k); + item_rec.field("value") = avro::GenericDatum(v); + arr.value().push_back(item); + } + }; + + auto write_bytes_map = [&](const std::string & name, const std::vector>> & entries) + { + auto & field = rec.field(name); + auto & arr = field.value(); + auto schema_element = arr.schema()->leafAt(0); + for (const auto & [k, v] : entries) + { + avro::GenericDatum item(schema_element); + auto & item_rec = item.value(); + item_rec.field("key") = avro::GenericDatum(k); + item_rec.field("value") = avro::GenericDatum(v); + arr.value().push_back(item); + } + }; + + write_long_map("column_sizes", stats.column_sizes); + write_long_map("null_value_counts", stats.null_value_counts); + write_bytes_map("lower_bounds", stats.lower_bounds); + write_bytes_map("upper_bounds", stats.upper_bounds); + + writer.write(datum); + writer.flush(); + // writer destructor writes the Avro end-of-file sync marker + } + + buf->finalize(); +} + +/// vibe coded - needs extra attention +IcebergSerializedFileStats serializeDataFileStats( + const DataFileStatistics & stats, + SharedHeader sample_block, + Int64 record_count, + Int64 file_size_in_bytes) +{ + IcebergSerializedFileStats result; + result.record_count = record_count; + result.file_size_in_bytes = file_size_in_bytes; + + for (const auto & [field_id, sz] : stats.getColumnSizes()) + result.column_sizes.emplace_back(static_cast(field_id), static_cast(sz)); + + for (const auto & [field_id, cnt] : stats.getNullCounts()) + result.null_value_counts.emplace_back(static_cast(field_id), static_cast(cnt)); + + std::unordered_map field_id_to_col_idx; + { + auto field_ids = stats.getFieldIds(); + for (size_t i = 0; i < field_ids.size(); ++i) + field_id_to_col_idx[field_ids[i]] = i; + } + + auto serialize_bounds = [&](const std::vector> & bounds, + std::vector>> & out) + { + if (!canWriteStatistics(bounds, field_id_to_col_idx, sample_block)) + return; + for (const auto & [field_id, value] : bounds) + { + auto bytes = dumpFieldToBytes(value, sample_block->getDataTypes()[field_id_to_col_idx.at(field_id)]); + out.emplace_back(static_cast(field_id), std::move(bytes)); + } + }; + + serialize_bounds(stats.getLowerBounds(), result.lower_bounds); + serialize_bounds(stats.getUpperBounds(), result.upper_bounds); + + return result; +} + +void extendSchemaForPartitions( +>>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) String & schema, const std::vector & partition_columns, const std::vector & partition_types) @@ -263,7 +431,11 @@ void generateManifestFile( Int64 partition_spec_id, WriteBuffer & buf, Iceberg::FileContentType content_type, +<<<<<<< HEAD std::optional user_defined_sequence_number) +======= + const std::vector & per_file_stats) +>>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) { Int32 version = metadata->getValue(Iceberg::f_format_version); String schema_representation; @@ -296,7 +468,10 @@ void generateManifestFile( Poco::JSON::Stringifier::stringify(partition_spec->getArray(Iceberg::f_fields), oss_partition_spec); writer.setMetadata(Iceberg::f_partition_spec, oss_partition_spec.str()); writer.setMetadata(Iceberg::f_partition_spec_id, std::to_string(partition_spec_id)); +<<<<<<< HEAD writer.setMetadata(Iceberg::f_format_version, std::to_string(version)); +======= +>>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) for (size_t file_idx = 0; file_idx < data_file_names.size(); ++file_idx) { const auto & data_file_name = data_file_names[file_idx]; @@ -341,31 +516,63 @@ void generateManifestFile( data_file.field(Iceberg::f_file_path) = avro::GenericDatum(data_file_name.serialize()); data_file.field(Iceberg::f_file_format) = avro::GenericDatum(format); - if (data_file_statistics) + /// vibe coded - needs extra attention + /// Export path: per-file serialized stats override everything (record count, file size, + /// and all column statistics). Existing insert/mutation paths use the aggregate path below. + if (!per_file_stats.empty() && file_idx < per_file_stats.size()) { - auto set_fields = [&]( - const std::vector> & statistics, const std::string & field_name, U && dump_function) + const auto & pf = per_file_stats[file_idx]; + + auto write_long_map = [&](const std::vector> & entries, const String & field_name) { - auto & data_file_record = data_file.field(field_name); - data_file_record.selectBranch(1); - auto & record_values = data_file_record.value(); - auto schema_element = record_values.schema()->leafAt(0); - for (const auto & [field_id, value] : statistics) + if (entries.empty()) + return; + auto & field = data_file.field(field_name); + field.selectBranch(1); + auto & arr = field.value(); + auto schema_element = arr.schema()->leafAt(0); + for (const auto & [k, v] : entries) { +<<<<<<< HEAD avro::GenericDatum record_datum(schema_element); auto & record = record_datum.value(); record.field(Iceberg::f_key) = static_cast(field_id); record.field(Iceberg::f_value) = dump_function(field_id, value); record_values.value().push_back(record_datum); +======= + avro::GenericDatum item(schema_element); + auto & item_rec = item.value(); + item_rec.field(Iceberg::f_key) = avro::GenericDatum(k); + item_rec.field(Iceberg::f_value) = avro::GenericDatum(v); + arr.value().push_back(item); +>>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) } }; - auto statistics = data_file_statistics->getColumnSizes(); - set_fields(statistics, Iceberg::f_column_sizes, [](size_t, size_t value) { return static_cast(value); }); + auto write_bytes_map = [&](const std::vector>> & entries, const String & field_name) + { + if (entries.empty()) + return; + auto & field = data_file.field(field_name); + field.selectBranch(1); + auto & arr = field.value(); + auto schema_element = arr.schema()->leafAt(0); + for (const auto & [k, v] : entries) + { + avro::GenericDatum item(schema_element); + auto & item_rec = item.value(); + item_rec.field(Iceberg::f_key) = avro::GenericDatum(k); + item_rec.field(Iceberg::f_value) = avro::GenericDatum(v); + arr.value().push_back(item); + } + }; - statistics = data_file_statistics->getNullCounts(); - set_fields(statistics, Iceberg::f_null_value_counts, [](size_t, size_t value) { return static_cast(value); }); + write_long_map(pf.column_sizes, Iceberg::f_column_sizes); + write_long_map(pf.null_value_counts, Iceberg::f_null_value_counts); + write_bytes_map(pf.lower_bounds, Iceberg::f_lower_bounds); + write_bytes_map(pf.upper_bounds, Iceberg::f_upper_bounds); +<<<<<<< HEAD std::unordered_map field_id_to_column_index; auto field_ids = data_file_statistics->getFieldIds(); for (size_t i = 0; i < field_ids.size(); ++i) @@ -387,6 +594,68 @@ void generateManifestFile( } data_file.field(Iceberg::f_record_count) = avro::GenericDatum(static_cast(data_file_row_counts[file_idx])); data_file.field(Iceberg::f_file_size_in_bytes) = avro::GenericDatum(static_cast(data_file_byte_counts[file_idx])); +======= + data_file.field(Iceberg::f_record_count) = avro::GenericDatum(pf.record_count); + data_file.field(Iceberg::f_file_size_in_bytes) = avro::GenericDatum(pf.file_size_in_bytes); + } + else + { + /// Regular INSERT / mutation path: aggregate column statistics applied to every file. + if (data_file_statistics) + { + auto set_fields = [&]( + const std::vector> & statistics, const std::string & field_name, U && dump_function) + { + auto & data_file_record = data_file.field(field_name); + data_file_record.selectBranch(1); + auto & record_values = data_file_record.value(); + auto schema_element = record_values.schema()->leafAt(0); + for (const auto & [field_id, value] : statistics) + { + avro::GenericDatum record_datum(schema_element); + auto & record = record_datum.value(); + record.field(Iceberg::f_key) = static_cast(field_id); + record.field(Iceberg::f_value) = dump_function(field_id, value); + record_values.value().push_back(record_datum); + } + }; + + auto statistics = data_file_statistics->getColumnSizes(); + set_fields(statistics, Iceberg::f_column_sizes, [](size_t, size_t value) { return static_cast(value); }); + + statistics = data_file_statistics->getNullCounts(); + set_fields(statistics, Iceberg::f_null_value_counts, [](size_t, size_t value) { return static_cast(value); }); + + std::unordered_map field_id_to_column_index; + auto field_ids = data_file_statistics->getFieldIds(); + for (size_t i = 0; i < field_ids.size(); ++i) + field_id_to_column_index[field_ids[i]] = i; + + auto dump_fields = [&](size_t field_id, Field value) + { return dumpFieldToBytes(value, sample_block->getDataTypes()[field_id_to_column_index.at(field_id)]); }; + + auto lower_statistics = data_file_statistics->getLowerBounds(); + if (canWriteStatistics(lower_statistics, field_id_to_column_index, sample_block)) + set_fields(lower_statistics, Iceberg::f_lower_bounds, dump_fields); + auto upper_statistics = data_file_statistics->getUpperBounds(); + if (canWriteStatistics(upper_statistics, field_id_to_column_index, sample_block)) + set_fields(upper_statistics, Iceberg::f_upper_bounds, dump_fields); + } + + /// Record count and file size from the snapshot summary (aggregate for all files). + auto summary = new_snapshot->getObject(Iceberg::f_summary); + if (summary->has(Iceberg::f_added_records)) + { + data_file.field(Iceberg::f_record_count) = avro::GenericDatum(summary->getValue(Iceberg::f_added_records)); + data_file.field(Iceberg::f_file_size_in_bytes) = avro::GenericDatum(summary->getValue(Iceberg::f_added_files_size)); + } + else + { + data_file.field(Iceberg::f_record_count) = avro::GenericDatum(summary->getValue(Iceberg::f_added_position_deletes)); + data_file.field(Iceberg::f_file_size_in_bytes) = avro::GenericDatum(summary->getValue(Iceberg::f_added_files_size)); + } + } +>>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) avro::GenericRecord & partition_record = data_file.field("partition").value(); for (size_t i = 0; i < partition_columns.size(); ++i) { @@ -1108,6 +1377,151 @@ bool IcebergStorageSink::initializeMetadata() return true; } +IcebergImportSink::IcebergImportSink( + std::shared_ptr catalog_, + const Iceberg::PersistentTableComponents & persistent_table_components_, + Poco::JSON::Object::Ptr metadata_json_, + ObjectStoragePtr object_storage_, + ContextPtr context_, + std::optional format_settings_, + const String & write_format_, + SharedHeader sample_block_, + const DataLakeStorageSettings & data_lake_settings_, + std::function new_file_path_callback_) + : SinkToStorage(sample_block_) + , catalog(catalog_) + , persistent_table_components(persistent_table_components_) + , metadata_json(metadata_json_) + , object_storage(object_storage_) + , context(context_) + , format_settings(format_settings_) + , write_format(write_format_) + , sample_block(sample_block_) + , data_lake_settings(data_lake_settings_) + , new_file_path_callback(std::move(new_file_path_callback_)) +{ + const auto current_schema_id = metadata_json->getValue(Iceberg::f_current_schema_id); + const auto schemas = metadata_json->getArray(Iceberg::f_schemas); + + for (size_t i = 0; i < schemas->size(); ++i) + { + if (schemas->getObject(static_cast(i))->getValue(Iceberg::f_schema_id) == current_schema_id) + { + current_schema = schemas->getObject(static_cast(i)); + break; + } + } + + const auto metadata_compression_method = persistent_table_components.metadata_compression_method; + + auto config_path = persistent_table_components.table_path; + if (config_path.empty() || config_path.back() != '/') + config_path += "/"; + if (!config_path.starts_with('/')) + config_path = '/' + config_path; + + if (!context_->getSettingsRef()[Setting::write_full_path_in_iceberg_metadata]) + { + filename_generator = FileNamesGenerator( + config_path, config_path, (catalog != nullptr && catalog->isTransactional()), metadata_compression_method, write_format); + } + else + { + auto bucket = metadata_json->getValue(Iceberg::f_location); + if (bucket.empty() || bucket.back() != '/') + bucket += "/"; + filename_generator = FileNamesGenerator( + bucket, config_path, (catalog != nullptr && catalog->isTransactional()), metadata_compression_method, write_format); + } + + const auto [last_version, unused_meta_path, unused_compression] = getLatestOrExplicitMetadataFileAndVersion( + object_storage, + persistent_table_components.table_path, + data_lake_settings, + persistent_table_components.metadata_cache, + context_, + getLogger("IcebergWrites").get(), + persistent_table_components.table_uuid); + (void)unused_meta_path; + (void)unused_compression; + + filename_generator.setVersion(last_version + 1); + + writer = std::make_unique( + context->getSettingsRef()[Setting::iceberg_insert_max_rows_in_data_file], + context->getSettingsRef()[Setting::iceberg_insert_max_bytes_in_data_file], + current_schema->getArray(Iceberg::f_fields), + filename_generator, + object_storage, + context, + format_settings, + write_format, + sample_block, + new_file_path_callback); +} + +IcebergImportSink::~IcebergImportSink() +{ + cancelBuffers(); +} + +void IcebergImportSink::consume(Chunk & chunk) +{ + if (isCancelled()) + return; + + writer->consume(chunk); +} + +void IcebergImportSink::onFinish() +{ + if (isCancelled()) + { + cancelBuffers(); + return; + } + + finalizeBuffers(); + + for (const auto & entry : writer->getDataFileEntries()) + { + IcebergSerializedFileStats serialized_stats; + if (entry.statistics) + { + serialized_stats = serializeDataFileStats(*entry.statistics, sample_block, entry.record_count, entry.file_size_in_bytes); + } + else + { + serialized_stats.record_count = entry.record_count; + serialized_stats.file_size_in_bytes = entry.file_size_in_bytes; + } + + writeDataFileSidecar(entry.path, serialized_stats, object_storage, context); + } + + releaseBuffers(); +} + +void IcebergImportSink::onException(std::exception_ptr /* exception */) +{ + cancelBuffers(); +} + +void IcebergImportSink::finalizeBuffers() +{ + writer->finalize(); +} + +void IcebergImportSink::releaseBuffers() +{ + writer->release(); +} + +void IcebergImportSink::cancelBuffers() +{ + writer->cancel(); +} + } // NOLINTEND(clang-analyzer-core.uninitialized.UndefReturn) diff --git a/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergWrites.h b/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergWrites.h index f25c77baef8d..3f342fbc5e01 100644 --- a/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergWrites.h +++ b/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergWrites.h @@ -24,6 +24,7 @@ #include #include #include +#include #include #include @@ -44,6 +45,37 @@ namespace DB String removeEscapedSlashes(const String & json_str); +/// Read a data-file sidecar and return its contents in Iceberg wire format. +/// The returned struct carries the row count, byte size, and per-column statistics. +IcebergSerializedFileStats readDataFileSidecar( + const String & sidecar_storage_path, + const ObjectStoragePtr & object_storage, + const ContextPtr & context); + +/// Write a sidecar Avro file alongside a data file. +/// All six fields are written; empty stat vectors are valid when statistics are unavailable. +void writeDataFileSidecar( + const String & data_file_storage_path, + const IcebergSerializedFileStats & stats, + const ObjectStoragePtr & object_storage, + const ContextPtr & context); + +/// Convert in-memory DataFileStatistics (ClickHouse-internal) to the Iceberg wire format. +/// Bounds are serialized to bytes using the same encoding used in the manifest file, +/// so the result can be stored in sidecar Avro files and used at commit time on any node. +IcebergSerializedFileStats serializeDataFileStats( + const DataFileStatistics & stats, + SharedHeader sample_block, + Int64 record_count, + Int64 file_size_in_bytes); + +/// Generate an Iceberg manifest file for a set of data files. +/// +/// \param data_file_statistics Aggregate column statistics applied to every file (regular +/// INSERT and mutation paths). Ignored when \p per_file_stats is non-empty. +/// \param per_file_stats Per-file pre-serialized statistics (export-commit path). +/// When non-empty each entry overrides both the record count / file size AND the column +/// statistics for the corresponding file. Leave empty to preserve the existing behaviour. void generateManifestFile( Poco::JSON::Object::Ptr metadata, const std::vector & partition_columns, @@ -60,7 +92,11 @@ void generateManifestFile( Int64 partition_spec_id, WriteBuffer & buf, Iceberg::FileContentType content_type, +<<<<<<< HEAD std::optional user_defined_sequence_number = std::nullopt); +======= + const std::vector & per_file_stats = {}); +>>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) void generateManifestList( const Iceberg::IcebergPathResolver & path_resolver, @@ -74,7 +110,13 @@ void generateManifestList( Iceberg::FileContentType content_type, bool use_previous_snapshots = true); +<<<<<<< HEAD class IcebergStorageSink final : public SinkToStorage +======= +std::string getIcebergExportPartSidecarStoragePath(const String & data_file_storage_path); + +class IcebergStorageSink : public SinkToStorage +>>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) { public: IcebergStorageSink( @@ -132,6 +174,50 @@ class IcebergStorageSink final : public SinkToStorage }; +class IcebergImportSink : public SinkToStorage +{ +public: + IcebergImportSink( + std::shared_ptr catalog_, + const Iceberg::PersistentTableComponents & persistent_table_components_, + Poco::JSON::Object::Ptr metadata_json_, + ObjectStoragePtr object_storage_, + ContextPtr context_, + std::optional format_settings_, + const String & write_format_, + SharedHeader sample_block_, + const DataLakeStorageSettings & data_lake_settings_, + std::function new_file_path_callback_ = {}); + + ~IcebergImportSink() override; + + String getName() const override { return "IcebergImportSink"; } + + void consume(Chunk & chunk) override; + + void onFinish() override; + void onException(std::exception_ptr exception) override; + +private: + void finalizeBuffers(); + void releaseBuffers(); + void cancelBuffers(); + + std::shared_ptr catalog; + const Iceberg::PersistentTableComponents & persistent_table_components; + Poco::JSON::Object::Ptr metadata_json; + Poco::JSON::Object::Ptr current_schema; + FileNamesGenerator filename_generator; + ObjectStoragePtr object_storage; + ContextPtr context; + std::optional format_settings; + const String& write_format; + SharedHeader sample_block; + std::unique_ptr writer; + const DataLakeStorageSettings & data_lake_settings; + std::function new_file_path_callback; +}; + } #endif diff --git a/src/Storages/ObjectStorage/DataLakes/Iceberg/MetadataGenerator.cpp b/src/Storages/ObjectStorage/DataLakes/Iceberg/MetadataGenerator.cpp index 85f5127c21c4..103872caf28a 100644 --- a/src/Storages/ObjectStorage/DataLakes/Iceberg/MetadataGenerator.cpp +++ b/src/Storages/ObjectStorage/DataLakes/Iceberg/MetadataGenerator.cpp @@ -194,7 +194,7 @@ MetadataGenerator::NextMetadataResult MetadataGenerator::generateNextMetadata( metadata_object->set(Iceberg::f_current_snapshot_id, snapshot_id); if (!metadata_object->has(Iceberg::f_refs)) - metadata_object->set(Iceberg::f_refs, new Poco::JSON::Object); + metadata_object->set(Iceberg::f_refs, Poco::JSON::Object::Ptr(new Poco::JSON::Object)); if (!metadata_object->getObject(Iceberg::f_refs)->has(Iceberg::f_main)) { diff --git a/src/Storages/ObjectStorage/DataLakes/Iceberg/MultipleFileWriter.cpp b/src/Storages/ObjectStorage/DataLakes/Iceberg/MultipleFileWriter.cpp index 54908e2eda08..1f440ae26709 100644 --- a/src/Storages/ObjectStorage/DataLakes/Iceberg/MultipleFileWriter.cpp +++ b/src/Storages/ObjectStorage/DataLakes/Iceberg/MultipleFileWriter.cpp @@ -22,12 +22,18 @@ MultipleFileWriter::MultipleFileWriter( ContextPtr context_, const std::optional & format_settings_, const String & write_format_, - SharedHeader sample_block_) + SharedHeader sample_block_, + std::function new_file_path_callback_) : max_data_file_num_rows(max_data_file_num_rows_) , max_data_file_num_bytes(max_data_file_num_bytes_) +<<<<<<< HEAD , schema(schema_) , stats(schema_) , column_mapper(std::make_shared()) +======= + , aggregate_stats(schema) + , current_file_stats(schema) +>>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) , filename_generator(filename_generator_) , path_resolver(path_resolver_) , object_storage(object_storage_) @@ -35,6 +41,8 @@ MultipleFileWriter::MultipleFileWriter( , format_settings(format_settings_) , write_format(std::move(write_format_)) , sample_block(sample_block_) + , schema_fields_json(schema) + , new_file_path_callback(std::move(new_file_path_callback_)) { column_mapper->setStorageColumnEncoding(Iceberg::IcebergSchemaProcessor::traverseSchema(schema_)); } @@ -42,7 +50,10 @@ MultipleFileWriter::MultipleFileWriter( void MultipleFileWriter::startNewFile() { if (buffer) + { finalize(); + current_file_stats = DataFileStatistics(schema_fields_json); + } current_file_stats = std::make_shared(schema); current_file_num_rows = 0; @@ -50,7 +61,14 @@ void MultipleFileWriter::startNewFile() auto metadata_path = filename_generator.generateDataFileName(); auto storage_path = path_resolver.resolve(metadata_path); +<<<<<<< HEAD data_file_names.push_back(metadata_path); +======= + data_file_names.push_back(filename.path_in_storage); + if (new_file_path_callback) + new_file_path_callback(filename.path_in_storage); + +>>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) buffer = object_storage->writeObject( StoredObject(storage_path), WriteMode::Rewrite, std::nullopt, DBMS_DEFAULT_BUFFER_SIZE, context->getWriteSettings()); @@ -75,8 +93,13 @@ void MultipleFileWriter::consume(const Chunk & chunk) output_format->flush(); *current_file_num_rows += chunk.getNumRows(); *current_file_num_bytes += chunk.bytes(); +<<<<<<< HEAD stats.update(chunk); current_file_stats->update(chunk); +======= + aggregate_stats.update(chunk); + current_file_stats.update(chunk); +>>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) } void MultipleFileWriter::finalize() @@ -84,6 +107,7 @@ void MultipleFileWriter::finalize() output_format->flush(); output_format->finalize(); buffer->finalize(); +<<<<<<< HEAD auto buffer_bytes = buffer->count(); UInt64 file_bytes = 0; if (buffer_bytes > 0) @@ -104,6 +128,31 @@ void MultipleFileWriter::finalize() completed_file_stats.push_back(std::move(current_file_stats)); data_file_byte_counts.push_back(file_bytes); data_file_row_counts.push_back(current_file_num_rows.value_or(0)); +======= + const UInt64 file_bytes = buffer->count(); + total_bytes += file_bytes; + per_file_record_counts.push_back(static_cast(*current_file_num_rows)); + per_file_byte_sizes.push_back(static_cast(file_bytes)); + per_file_stats_list.push_back(current_file_stats); +} + +std::vector MultipleFileWriter::getDataFileEntries() const +{ + chassert(data_file_names.size() == per_file_record_counts.size()); + chassert(data_file_names.size() == per_file_stats_list.size()); + + std::vector entries; + entries.reserve(data_file_names.size()); + + for (size_t i = 0; i < data_file_names.size(); ++i) + entries.emplace_back( + data_file_names[i], + per_file_record_counts[i], + per_file_byte_sizes[i], + per_file_stats_list[i]); + + return entries; +>>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) } void MultipleFileWriter::release() diff --git a/src/Storages/ObjectStorage/DataLakes/Iceberg/MultipleFileWriter.h b/src/Storages/ObjectStorage/DataLakes/Iceberg/MultipleFileWriter.h index 59bd9142139a..f807aad7f094 100644 --- a/src/Storages/ObjectStorage/DataLakes/Iceberg/MultipleFileWriter.h +++ b/src/Storages/ObjectStorage/DataLakes/Iceberg/MultipleFileWriter.h @@ -5,6 +5,7 @@ #include #include #include +#include namespace DB { @@ -24,7 +25,8 @@ class MultipleFileWriter ContextPtr context_, const std::optional & format_settings_, const String & write_format_, - SharedHeader sample_block_); + SharedHeader sample_block_, + std::function new_file_path_callback_ = {}); void consume(const Chunk & chunk); void startNewFile(); @@ -52,17 +54,25 @@ class MultipleFileWriter const DataFileStatistics & getResultStatistics() const { - return stats; + return aggregate_stats; } +<<<<<<< HEAD const std::vector & getPerFileStatistics() const { return completed_file_stats; } +======= + /// Returns one entry per written data file, with the accurate row count, byte size, + /// and per-file column statistics collected during finalization. + /// Must be called only after finalize(). + std::vector getDataFileEntries() const; +>>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) private: UInt64 max_data_file_num_rows; UInt64 max_data_file_num_bytes; +<<<<<<< HEAD Poco::JSON::Array::Ptr schema; DataFileStatistics stats; DataFileStatisticsPtr current_file_stats; @@ -76,6 +86,16 @@ class MultipleFileWriter std::vector data_file_names; std::vector data_file_row_counts; std::vector data_file_byte_counts; +======= + DataFileStatistics aggregate_stats; /// accumulates across all files + DataFileStatistics current_file_stats; /// accumulates for the current file only + std::optional current_file_num_rows = std::nullopt; + std::optional current_file_num_bytes = std::nullopt; + std::vector data_file_names; + std::vector per_file_record_counts; + std::vector per_file_byte_sizes; + std::vector per_file_stats_list; +>>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) std::unique_ptr buffer; OutputFormatPtr output_format; FileNamesGenerator & filename_generator; @@ -86,6 +106,8 @@ class MultipleFileWriter const String& write_format; SharedHeader sample_block; UInt64 total_bytes = 0; + Poco::JSON::Array::Ptr schema_fields_json; + std::function new_file_path_callback; }; #endif diff --git a/src/Storages/ObjectStorage/DataLakes/Iceberg/Utils.cpp b/src/Storages/ObjectStorage/DataLakes/Iceberg/Utils.cpp index f7a3f164ac1d..14c6b51fd361 100644 --- a/src/Storages/ObjectStorage/DataLakes/Iceberg/Utils.cpp +++ b/src/Storages/ObjectStorage/DataLakes/Iceberg/Utils.cpp @@ -162,7 +162,11 @@ static bool isTemporaryMetadataFile(const String & file_name) return Poco::UUID{}.tryParse(substring); } +<<<<<<< HEAD static MetadataFileWithInfo getMetadataFileAndVersion(const std::string & path) +======= +Iceberg::MetadataFileWithInfo getMetadataFileAndVersion(const std::string & path) +>>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) { String file_name = std::filesystem::path(path).filename(); if (isTemporaryMetadataFile(file_name)) @@ -509,6 +513,8 @@ std::pair getIcebergType(DataTypePtr type, Int32 & ite { switch (type->getTypeId()) { + case TypeIndex::UInt16: + case TypeIndex::Int16: case TypeIndex::UInt32: case TypeIndex::Int32: return {"int", true}; @@ -602,10 +608,14 @@ Poco::Dynamic::Var getAvroType(DataTypePtr type) { switch (type->getTypeId()) { +<<<<<<< HEAD case TypeIndex::UInt8: case TypeIndex::Int8: case TypeIndex::UInt16: case TypeIndex::Int16: +======= + case TypeIndex::UInt16: +>>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) case TypeIndex::UInt32: case TypeIndex::Int32: case TypeIndex::Date: diff --git a/src/Storages/ObjectStorage/DataLakes/Iceberg/Utils.h b/src/Storages/ObjectStorage/DataLakes/Iceberg/Utils.h index 43d2c040ad59..cf3ebbff2eea 100644 --- a/src/Storages/ObjectStorage/DataLakes/Iceberg/Utils.h +++ b/src/Storages/ObjectStorage/DataLakes/Iceberg/Utils.h @@ -19,6 +19,7 @@ #include #include #include +#include #include #include #include @@ -31,6 +32,8 @@ class GenericDatum; namespace DB::Iceberg { +Iceberg::MetadataFileWithInfo getMetadataFileAndVersion(const std::string & path); + void writeMessageToFile( const String & data, const String & filename, @@ -75,6 +78,14 @@ Poco::JSON::Object::Ptr getMetadataJSONObject( std::pair getIcebergType(DataTypePtr type, Int32 & iter); Poco::Dynamic::Var getAvroType(DataTypePtr type); +/// Converts a ClickHouse PARTITION BY AST into the corresponding Iceberg partition-spec JSON object. +/// column_name_to_source_id maps each column name to the Iceberg field-id from the table schema. +/// The returned Int32 is the last partition-field-id allocated (useful for tracking the id counter). +/// Throws if the AST contains expressions that cannot be represented as Iceberg transforms. +std::pair getPartitionSpec( + ASTPtr partition_by, + const std::unordered_map & column_name_to_source_id); + /// Spec: https://iceberg.apache.org/spec/?h=metadata.json#table-metadata-fields std::pair createEmptyMetadataFile( String path_location, diff --git a/src/Storages/ObjectStorage/MultiFileStorageObjectStorageSink.cpp b/src/Storages/ObjectStorage/MultiFileStorageObjectStorageSink.cpp new file mode 100644 index 000000000000..f46985b9a52f --- /dev/null +++ b/src/Storages/ObjectStorage/MultiFileStorageObjectStorageSink.cpp @@ -0,0 +1,153 @@ +#include +#include +#include +#include + +namespace DB +{ + +namespace ErrorCodes +{ + extern const int FILE_ALREADY_EXISTS; +} + +MultiFileStorageObjectStorageSink::MultiFileStorageObjectStorageSink( + const std::string & base_path_, + const String & transaction_id_, + ObjectStoragePtr object_storage_, + StorageObjectStorageConfigurationPtr configuration_, + std::size_t max_bytes_per_file_, + std::size_t max_rows_per_file_, + bool overwrite_if_exists_, + const std::function & new_file_path_callback_, + const std::optional & format_settings_, + SharedHeader sample_block_, + ContextPtr context_) + : SinkToStorage(sample_block_), + base_path(base_path_), + transaction_id(transaction_id_), + object_storage(object_storage_), + configuration(configuration_), + max_bytes_per_file(max_bytes_per_file_), + max_rows_per_file(max_rows_per_file_), + overwrite_if_exists(overwrite_if_exists_), + new_file_path_callback(new_file_path_callback_), + format_settings(format_settings_), + sample_block(sample_block_), + context(context_) +{ + current_sink = createNewSink(); +} + +MultiFileStorageObjectStorageSink::~MultiFileStorageObjectStorageSink() +{ + if (isCancelled()) + current_sink->cancel(); +} + +/// Adds a counter that represents file index to the file path. +/// Example: +/// Input is `table_root/year=2025/month=12/day=12/file.parquet` +/// Output is `table_root/year=2025/month=12/day=12/file.1.parquet` +std::string MultiFileStorageObjectStorageSink::generateNewFilePath() +{ + const auto file_format = Poco::toLower(configuration->format); + const auto index_string = std::to_string(file_paths.size() + 1); + std::size_t pos = base_path.rfind(file_format); + + /// normal case - path ends with the file format + if (pos != std::string::npos) + { + const auto path_without_extension = base_path.substr(0, pos); + const auto file_format_extension = "." + file_format; + + return path_without_extension + index_string + file_format_extension; + } + + /// if no extension is found, just append the index - I am not even sure this is possible + return base_path + "." + index_string; +} + +std::shared_ptr MultiFileStorageObjectStorageSink::createNewSink() +{ + auto new_path = generateNewFilePath(); + + /// todo + /// sounds like bad design, but callers might decide to ignore the exception, and if we throw it before the callback + /// they will not be able to grab the file path. + /// maybe I should consider moving the file already exists policy in here? + new_file_path_callback(new_path); + + file_paths.emplace_back(std::move(new_path)); + + if (!overwrite_if_exists && object_storage->exists(StoredObject(file_paths.back()))) + { + throw Exception(ErrorCodes::FILE_ALREADY_EXISTS, "File {} already exists", file_paths.back()); + } + + return std::make_shared( + file_paths.back(), + object_storage, + format_settings, + sample_block, + context, + configuration->format, + configuration->compression_method); +} + +void MultiFileStorageObjectStorageSink::consume(Chunk & chunk) +{ + if (isCancelled()) + { + current_sink->cancel(); + return; + } + + const auto written_bytes = current_sink->getWrittenBytes(); + + const bool exceeded_bytes_limit = max_bytes_per_file && written_bytes >= max_bytes_per_file; + const bool exceeded_rows_limit = max_rows_per_file && current_sink_written_rows >= max_rows_per_file; + + if (exceeded_bytes_limit || exceeded_rows_limit) + { + current_sink->onFinish(); + current_sink = createNewSink(); + current_sink_written_rows = 0; + } + + current_sink->consume(chunk); + current_sink_written_rows += chunk.getNumRows(); +} + +void MultiFileStorageObjectStorageSink::onFinish() +{ + current_sink->onFinish(); + commit(); +} + +void MultiFileStorageObjectStorageSink::commit() +{ + /// the commit file path should be in the same directory as the data files + const auto commit_file_path = fs::path(base_path).parent_path() / ("commit_" + transaction_id); + + if (!overwrite_if_exists && object_storage->exists(StoredObject(commit_file_path))) + { + throw Exception(ErrorCodes::FILE_ALREADY_EXISTS, "Commit file {} already exists, aborting {} export", commit_file_path, transaction_id); + } + + auto out = object_storage->writeObject( + StoredObject(commit_file_path), + WriteMode::Rewrite, /* attributes= */ + {}, DBMS_DEFAULT_BUFFER_SIZE, + context->getWriteSettings()); + + for (const auto & p : file_paths) + { + out->write(p.data(), p.size()); + out->write("\n", 1); + } + + out->finalize(); +} + +} diff --git a/src/Storages/ObjectStorage/MultiFileStorageObjectStorageSink.h b/src/Storages/ObjectStorage/MultiFileStorageObjectStorageSink.h new file mode 100644 index 000000000000..51f6b8094232 --- /dev/null +++ b/src/Storages/ObjectStorage/MultiFileStorageObjectStorageSink.h @@ -0,0 +1,57 @@ +#pragma once + +#include + +namespace DB +{ + +/// This is useful when the data is too large to fit into a single file. +/// It will create a new file when the current file exceeds the max bytes or max rows limit. +/// Ships a commit file including the list of data files to make it transactional +class MultiFileStorageObjectStorageSink : public SinkToStorage +{ +public: + MultiFileStorageObjectStorageSink( + const std::string & base_path_, + const String & transaction_id_, + ObjectStoragePtr object_storage_, + StorageObjectStorageConfigurationPtr configuration_, + std::size_t max_bytes_per_file_, + std::size_t max_rows_per_file_, + bool overwrite_if_exists_, + const std::function & new_file_path_callback_, + const std::optional & format_settings_, + SharedHeader sample_block_, + ContextPtr context_); + + ~MultiFileStorageObjectStorageSink() override; + + void consume(Chunk & chunk) override; + + void onFinish() override; + + String getName() const override { return "MultiFileStorageObjectStorageSink"; } + +private: + const std::string base_path; + const String transaction_id; + ObjectStoragePtr object_storage; + StorageObjectStorageConfigurationPtr configuration; + std::size_t max_bytes_per_file; + std::size_t max_rows_per_file; + bool overwrite_if_exists; + std::function new_file_path_callback; + const std::optional format_settings; + SharedHeader sample_block; + ContextPtr context; + + std::vector file_paths; + std::shared_ptr current_sink; + std::size_t current_sink_written_rows = 0; + + std::string generateNewFilePath(); + std::shared_ptr createNewSink(); + void commit(); +}; + +} diff --git a/src/Storages/ObjectStorage/ObjectStorageFilePathGenerator.h b/src/Storages/ObjectStorage/ObjectStorageFilePathGenerator.h new file mode 100644 index 000000000000..a1f21dc502d5 --- /dev/null +++ b/src/Storages/ObjectStorage/ObjectStorageFilePathGenerator.h @@ -0,0 +1,83 @@ +#pragma once + +#include +#include +#include +#include +#include + +namespace DB +{ + struct ObjectStorageFilePathGenerator + { + virtual ~ObjectStorageFilePathGenerator() = default; + std::string getPathForWrite(const std::string & partition_id) const { + return getPathForWrite(partition_id, ""); + } + virtual std::string getPathForWrite(const std::string & partition_id, const std::string & /* file_name_override */) const = 0; + virtual std::string getPathForRead() const = 0; + }; + + struct ObjectStorageWildcardFilePathGenerator : ObjectStorageFilePathGenerator + { + static constexpr const char * FILE_WILDCARD = "{_file}"; + explicit ObjectStorageWildcardFilePathGenerator(const std::string & raw_path_) : raw_path(raw_path_) {} + + using ObjectStorageFilePathGenerator::getPathForWrite; // Bring base class overloads into scope + std::string getPathForWrite(const std::string & partition_id, const std::string & file_name_override) const override + { + const auto partition_replaced_path = PartitionedSink::replaceWildcards(raw_path, partition_id); + const auto final_path = boost::replace_all_copy(partition_replaced_path, FILE_WILDCARD, file_name_override); + return final_path; + } + + std::string getPathForRead() const override + { + return raw_path; + } + + private: + std::string raw_path; + + }; + + struct ObjectStorageAppendFilePathGenerator : ObjectStorageFilePathGenerator + { + explicit ObjectStorageAppendFilePathGenerator( + const std::string & raw_path_, + const std::string & file_format_) + : raw_path(raw_path_), file_format(Poco::toLower(file_format_)){} + + using ObjectStorageFilePathGenerator::getPathForWrite; // Bring base class overloads into scope + std::string getPathForWrite(const std::string & partition_id, const std::string & file_name_override) const override + { + std::string result; + + result += raw_path; + + if (!result.empty() && result.back() != '/') + { + result += "/"; + } + + /// Not adding '/' because buildExpressionHive() always adds a trailing '/' + result += partition_id; + + const auto file_name = file_name_override.empty() ? std::to_string(generateSnowflakeID()) : file_name_override; + + result += file_name + "." + file_format; + + return result; + } + + std::string getPathForRead() const override + { + return raw_path + "**." + file_format; + } + + private: + std::string raw_path; + std::string file_format; + }; + +} diff --git a/src/Storages/ObjectStorage/StorageObjectStorage.cpp b/src/Storages/ObjectStorage/StorageObjectStorage.cpp index ef86fd71bb40..20caac158ad6 100644 --- a/src/Storages/ObjectStorage/StorageObjectStorage.cpp +++ b/src/Storages/ObjectStorage/StorageObjectStorage.cpp @@ -1,4 +1,5 @@ #include +#include #include #include @@ -32,11 +33,16 @@ #include #include #include +#include #include #include #include #include #include +#include +#include +#include +#include namespace DB @@ -56,6 +62,7 @@ namespace ErrorCodes extern const int NOT_IMPLEMENTED; extern const int INCORRECT_DATA; extern const int BAD_ARGUMENTS; + extern const int FILE_ALREADY_EXISTS; } String StorageObjectStorage::getPathSample(ContextPtr context) @@ -615,7 +622,8 @@ SinkToStoragePtr StorageObjectStorage::write( if (configuration->partition_strategy) { - return std::make_shared(object_storage, configuration, format_settings, sample_block, local_context); + auto sink_creator = std::make_shared(object_storage, configuration, format_settings, sample_block, local_context); + return std::make_shared(configuration->partition_strategy, sink_creator, local_context, sample_block); } auto paths = configuration->getPaths(); @@ -648,6 +656,127 @@ bool StorageObjectStorage::optimize( return configuration->optimize(object_storage, metadata_snapshot, context, format_settings); } +bool StorageObjectStorage::supportsImport(ContextPtr local_context) const +{ + if (isDataLake()) + { + configuration->lazyInitializeIfNeeded(object_storage, local_context); + return configuration->getExternalMetadata()->supportsImport(local_context); + } + + if (!configuration->partition_strategy) + return false; + + if (configuration->partition_strategy_type == PartitionStrategyFactory::StrategyType::WILDCARD) + return configuration->getRawPath().hasExportFilenameWildcard(); + + return configuration->partition_strategy_type == PartitionStrategyFactory::StrategyType::HIVE; +} + +SinkToStoragePtr StorageObjectStorage::import( + const std::string & file_name, + Block & block_with_partition_values, + const std::function & new_file_path_callback, + bool overwrite_if_exists, + std::size_t max_bytes_per_file, + std::size_t max_rows_per_file, + const std::optional & iceberg_metadata_json_string, + const std::optional & format_settings_, + ContextPtr local_context) +{ + if (isDataLake()) + { + configuration->lazyInitializeIfNeeded(object_storage, local_context); + return configuration->getExternalMetadata()->import( + catalog, + new_file_path_callback, + std::make_shared(getInMemoryMetadataPtr()->getSampleBlock()), + *iceberg_metadata_json_string, + format_settings_ ? format_settings_ : format_settings, + local_context); + } + + std::string partition_key; + + if (configuration->partition_strategy) + { + const auto column_with_partition_key = configuration->partition_strategy->computePartitionKey(block_with_partition_values); + + if (!column_with_partition_key->empty()) + { + partition_key = column_with_partition_key->getDataAt(0); + } + } + + const auto base_path = configuration->getPathForWrite(partition_key, file_name).path; + + return std::make_shared( + base_path, + /* transaction_id= */ file_name, /// not pretty, but the sink needs some sort of id to generate the commit file name. Using the source part name should be enough + object_storage, + configuration, + max_bytes_per_file, + max_rows_per_file, + overwrite_if_exists, + new_file_path_callback, + format_settings_ ? format_settings_ : format_settings, + std::make_shared(getInMemoryMetadataPtr()->getSampleBlock()), + local_context); +} + +void StorageObjectStorage::commitExportPartitionTransaction( + const String & transaction_id, + const String & partition_id, + const Strings & exported_paths, + const IcebergCommitExportPartitionArguments & iceberg_commit_export_partition_arguments, + ContextPtr local_context) +{ + if (isDataLake()) + { + /// Parse the Iceberg metadata snapshot (stored in ZooKeeper at export-start time) only to + /// extract the schema-id and partition-spec-id that were current when the export began. + /// partition_columns and partition_types are derived inside commitExportPartitionTransaction + /// from the same JSON, so only partition_values need to be carried here. + Poco::JSON::Parser iceberg_parser; + Poco::JSON::Object::Ptr iceberg_metadata = + iceberg_parser.parse(iceberg_commit_export_partition_arguments.metadata_json_string).extract(); + + const auto original_schema_id = iceberg_metadata->getValue(Iceberg::f_current_schema_id); + const auto partition_spec_id = iceberg_metadata->getValue(Iceberg::f_default_spec_id); + + configuration->lazyInitializeIfNeeded(object_storage, local_context); + configuration->getExternalMetadata()->commitExportPartitionTransaction( + catalog, + storage_id, + transaction_id, + original_schema_id, + partition_spec_id, + iceberg_commit_export_partition_arguments.partition_values, + std::make_shared(getInMemoryMetadataPtr()->getSampleBlock()), + exported_paths, + configuration, + local_context); + return; + } + + const String commit_object = configuration->getRawPath().path + "/commit_" + partition_id + "_" + transaction_id; + + /// if file already exists, nothing to be done + if (object_storage->exists(StoredObject(commit_object))) + { + LOG_DEBUG(getLogger("StorageObjectStorage"), "Commit file already exists, nothing to be done: {}", commit_object); + return; + } + + auto out = object_storage->writeObject(StoredObject(commit_object), WriteMode::Rewrite, /* attributes= */ {}, DBMS_DEFAULT_BUFFER_SIZE, local_context->getWriteSettings()); + for (const auto & p : exported_paths) + { + out->write(p.data(), p.size()); + out->write("\n", 1); + } + out->finalize(); +} + void StorageObjectStorage::truncate( const ASTPtr & /* query */, const StorageMetadataPtr & /* metadata_snapshot */, @@ -857,6 +986,7 @@ void StorageObjectStorage::checkAlterIsPossible(const AlterCommands & commands, configuration->checkAlterIsPossible(object_storage, context, commands); } +<<<<<<< HEAD void StorageObjectStorage::startup() { if (configuration->isBackgroundExecutable()) @@ -877,4 +1007,6 @@ bool StorageObjectStorage::scheduleDataProcessingJob(BackgroundJobsAssignee & as return configuration->scheduleDataProcessingJob(assignee, *this); } +======= +>>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) } diff --git a/src/Storages/ObjectStorage/StorageObjectStorage.h b/src/Storages/ObjectStorage/StorageObjectStorage.h index 6b8ac470f8ec..fc4dd867de63 100644 --- a/src/Storages/ObjectStorage/StorageObjectStorage.h +++ b/src/Storages/ObjectStorage/StorageObjectStorage.h @@ -7,6 +7,11 @@ #include #include #include +<<<<<<< HEAD +======= +#include +#include "Storages/ObjectStorage/ObjectStorageFilePathGenerator.h" +>>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) #include #include #include @@ -79,6 +84,26 @@ class StorageObjectStorage : public IStorage, public IBackgroundOperation ContextPtr context, bool async_insert) override; + bool supportsImport(ContextPtr) const override; + + SinkToStoragePtr import( + const std::string & /* file_name */, + Block & /* block_with_partition_values */, + const std::function & new_file_path_callback, + bool /* overwrite_if_exists */, + std::size_t /* max_bytes_per_file */, + std::size_t /* max_rows_per_file */, + const std::optional & /* iceberg_metadata_json_string */, + const std::optional & /* format_settings_ */, + ContextPtr /* context */) override; + + void commitExportPartitionTransaction( + const String & transaction_id, + const String & partition_id, + const Strings & exported_paths, + const IcebergCommitExportPartitionArguments & iceberg_commit_export_partition_arguments, + ContextPtr local_context) override; + void truncate( const ASTPtr & query, const StorageMetadataPtr & metadata_snapshot, diff --git a/src/Storages/ObjectStorage/StorageObjectStorageCluster.cpp b/src/Storages/ObjectStorage/StorageObjectStorageCluster.cpp index aeeb118e405f..4f71616447e9 100644 --- a/src/Storages/ObjectStorage/StorageObjectStorageCluster.cpp +++ b/src/Storages/ObjectStorage/StorageObjectStorageCluster.cpp @@ -1,3 +1,4 @@ +#include #include #include @@ -126,7 +127,17 @@ StorageObjectStorageCluster::StorageObjectStorageCluster( } metadata.setConstraints(constraints_); +<<<<<<< HEAD metadata.setVirtuals(VirtualColumnUtils::getVirtualsForFileLikeStorage( +======= + + if (configuration->partition_strategy) + { + metadata.partition_key = configuration->partition_strategy->getPartitionKeyDescription(); + } + + setVirtuals(VirtualColumnUtils::getVirtualsForFileLikeStorage( +>>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) metadata.columns, context_, /* format_settings */std::nullopt, diff --git a/src/Storages/ObjectStorage/StorageObjectStorageConfiguration.cpp b/src/Storages/ObjectStorage/StorageObjectStorageConfiguration.cpp index e65d0eaa78af..9e33987cae0e 100644 --- a/src/Storages/ObjectStorage/StorageObjectStorageConfiguration.cpp +++ b/src/Storages/ObjectStorage/StorageObjectStorageConfiguration.cpp @@ -153,10 +153,20 @@ void StorageObjectStorageConfiguration::initialize( else FormatFactory::instance().checkFormatName(configuration_to_initialize.format); - /// It might be changed on `StorageObjectStorageConfiguration::initPartitionStrategy` + if (configuration_to_initialize.partition_strategy_type == PartitionStrategyFactory::StrategyType::HIVE) + { + configuration_to_initialize.file_path_generator = std::make_shared( + configuration_to_initialize.getRawPath().path, + configuration_to_initialize.format); + } + else + { + configuration_to_initialize.file_path_generator = std::make_shared(configuration_to_initialize.getRawPath().path); + } + /// We shouldn't set path for disk setup because path prefix is already set in used object_storage. if (disk_name.empty()) - configuration_to_initialize.read_path = configuration_to_initialize.getRawPath(); + configuration_to_initialize.read_path = configuration_to_initialize.file_path_generator->getPathForRead(); configuration_to_initialize.initialized = true; } @@ -180,6 +190,12 @@ void StorageObjectStorageConfiguration::setSchemaHash(const String & hash) boost::replace_all(path.path, SCHEMA_HASH_WILDCARD, schema_hash); setRawPath(path); setPaths({path}); + + /// `file_path_generator` was constructed before `setSchemaHash` ran and still + /// holds a copy of the raw path with the unreplaced `{_schema_hash}` placeholder. + /// `_schema_hash` is rejected for hive partitioning earlier, so the wildcard + /// generator is the only valid variant here. + file_path_generator = std::make_shared(path.path); } void StorageObjectStorageConfiguration::initPartitionStrategy(ASTPtr partition_by, const ColumnsDescription & columns, ContextPtr context) @@ -226,7 +242,6 @@ void StorageObjectStorageConfiguration::initPartitionStrategy(ASTPtr partition_b if (partition_strategy) { - read_path = partition_strategy->getPathForRead(getRawPath().path); LOG_DEBUG(getLogger("StorageObjectStorageConfiguration"), "Initialized partition strategy {}", magic_enum::enum_name(partition_strategy_type)); } } @@ -238,17 +253,12 @@ const StorageObjectStorageConfiguration::Path & StorageObjectStorageConfiguratio StorageObjectStorageConfiguration::Path StorageObjectStorageConfiguration::getPathForWrite(const std::string & partition_id) const { - auto raw_path = getRawPath(); - - if (!schema_hash.empty()) - boost::replace_all(raw_path.path, SCHEMA_HASH_WILDCARD, schema_hash); - - if (!partition_strategy) - { - return raw_path; - } + return getPathForWrite(partition_id, /* filename_override */ ""); +} - return Path {partition_strategy->getPathForWrite(raw_path.path, partition_id)}; +StorageObjectStorageConfiguration::Path StorageObjectStorageConfiguration::getPathForWrite(const std::string & partition_id, const std::string & filename_override) const +{ + return Path {file_path_generator->getPathForWrite(partition_id, filename_override)}; } bool StorageObjectStorageConfiguration::Path::hasPartitionWildcard() const @@ -257,6 +267,11 @@ bool StorageObjectStorageConfiguration::Path::hasPartitionWildcard() const return path.find(PARTITION_ID_WILDCARD) != String::npos; } +bool StorageObjectStorageConfiguration::Path::hasExportFilenameWildcard() const +{ + return path.find(ObjectStorageWildcardFilePathGenerator::FILE_WILDCARD) != String::npos; +} + bool StorageObjectStorageConfiguration::Path::hasSchemaHashWildcard() const { return path.find(StorageObjectStorageConfiguration::SCHEMA_HASH_WILDCARD) != String::npos; diff --git a/src/Storages/ObjectStorage/StorageObjectStorageConfiguration.h b/src/Storages/ObjectStorage/StorageObjectStorageConfiguration.h index aebf4baff2c0..fbb0e7f5c923 100644 --- a/src/Storages/ObjectStorage/StorageObjectStorageConfiguration.h +++ b/src/Storages/ObjectStorage/StorageObjectStorageConfiguration.h @@ -18,7 +18,11 @@ #include #include #include +<<<<<<< HEAD #include +======= +#include +>>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) namespace DB { @@ -79,6 +83,7 @@ class StorageObjectStorageConfiguration bool hasPartitionWildcard() const; bool hasSchemaHashWildcard() const; bool hasGlobsIgnorePlaceholders() const; + bool hasExportFilenameWildcard() const; bool hasGlobs() const; std::string cutGlobs(bool supports_partial_prefix) const; }; @@ -111,8 +116,10 @@ class StorageObjectStorageConfiguration virtual const String & getRawURI() const = 0; const Path & getPathForRead() const; + // Path used for writing, it should not be globbed and might contain a partition key Path getPathForWrite(const std::string & partition_id = "") const; + Path getPathForWrite(const std::string & partition_id, const std::string & filename_override) const; void setPathForRead(const Path & path) { @@ -314,15 +321,20 @@ class StorageObjectStorageConfiguration String format = "auto"; String compression_method = "auto"; String structure = "auto"; + PartitionStrategyFactory::StrategyType partition_strategy_type = PartitionStrategyFactory::StrategyType::NONE; + std::shared_ptr partition_strategy; /// Whether partition column values are contained in the actual data. /// And alternative is with hive partitioning, when they are contained in file path. bool partition_columns_in_data_file = true; +<<<<<<< HEAD /// Tracks whether `partition_columns_in_data_file` was explicitly provided by the user. /// When false, `initPartitionStrategy` recomputes the default once the effective strategy is known /// (which may have been chosen implicitly via `file_like_engine_default_partition_strategy`). bool partition_columns_in_data_file_was_set = false; std::shared_ptr partition_strategy; +======= +>>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) protected: void initializeFromParsedArguments(const StorageParsedArguments & parsed_arguments); @@ -342,6 +354,8 @@ class StorageObjectStorageConfiguration // Path used for reading, by default it is the same as `getRawPath` // When using `partition_strategy=hive`, a recursive reading pattern will be appended `'table_root/**.parquet' Path read_path; + + std::shared_ptr file_path_generator; }; using StorageObjectStorageConfigurationPtr = std::shared_ptr; diff --git a/src/Storages/ObjectStorage/StorageObjectStorageSink.cpp b/src/Storages/ObjectStorage/StorageObjectStorageSink.cpp index 0c688bf15004..0669c4e4524d 100644 --- a/src/Storages/ObjectStorage/StorageObjectStorageSink.cpp +++ b/src/Storages/ObjectStorage/StorageObjectStorageSink.cpp @@ -136,14 +136,20 @@ size_t StorageObjectStorageSink::getFileSize() const return *result_file_size; } +size_t StorageObjectStorageSink::getWrittenBytes() const +{ + if (!write_buf) + throw Exception(ErrorCodes::LOGICAL_ERROR, "Buffer must be initialized before requesting written bytes"); + return write_buf->count(); +} + PartitionedStorageObjectStorageSink::PartitionedStorageObjectStorageSink( ObjectStoragePtr object_storage_, StorageObjectStorageConfigurationPtr configuration_, std::optional format_settings_, SharedHeader sample_block_, ContextPtr context_) - : PartitionedSink(configuration_->partition_strategy, context_, sample_block_) - , object_storage(object_storage_) + : object_storage(object_storage_) , configuration(configuration_) , query_settings(configuration_->getQuerySettings(context_)) , format_settings(format_settings_) @@ -177,10 +183,11 @@ SinkPtr PartitionedStorageObjectStorageSink::createSinkForPartition(const String file_path, object_storage, format_settings, - std::make_shared(partition_strategy->getFormatHeader()), + std::make_shared(configuration->partition_strategy->getFormatHeader()), context, configuration->format, - configuration->compression_method); + configuration->compression_method + ); } } diff --git a/src/Storages/ObjectStorage/StorageObjectStorageSink.h b/src/Storages/ObjectStorage/StorageObjectStorageSink.h index 86f11a2e0f27..7f0732a3476d 100644 --- a/src/Storages/ObjectStorage/StorageObjectStorageSink.h +++ b/src/Storages/ObjectStorage/StorageObjectStorageSink.h @@ -8,6 +8,8 @@ namespace DB { class StorageObjectStorageSink final : public SinkToStorage { +friend class StorageObjectStorageImporterSink; + public: StorageObjectStorageSink( const std::string & path_, @@ -28,6 +30,8 @@ class StorageObjectStorageSink final : public SinkToStorage const String & getPath() const { return path; } + size_t getWrittenBytes() const; + size_t getFileSize() const; private: @@ -42,7 +46,11 @@ class StorageObjectStorageSink final : public SinkToStorage void cancelBuffers(); }; +<<<<<<< HEAD class PartitionedStorageObjectStorageSink final : public PartitionedSink +======= +class PartitionedStorageObjectStorageSink : public PartitionedSink::SinkCreator +>>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) { public: PartitionedStorageObjectStorageSink( diff --git a/src/Storages/ObjectStorage/StorageObjectStorageSource.cpp b/src/Storages/ObjectStorage/StorageObjectStorageSource.cpp index d004243b3131..1ea2ab8b4ea3 100644 --- a/src/Storages/ObjectStorage/StorageObjectStorageSource.cpp +++ b/src/Storages/ObjectStorage/StorageObjectStorageSource.cpp @@ -95,6 +95,12 @@ namespace Setting extern const SettingsBool table_engine_read_through_distributed_cache; extern const SettingsUInt64 s3_path_filter_limit; extern const SettingsBool use_parquet_metadata_cache; +<<<<<<< HEAD +======= + extern const SettingsBool input_format_parquet_use_native_reader_v3; + extern const SettingsBool allow_experimental_iceberg_read_optimization; + extern const SettingsBool use_object_storage_list_objects_cache; +>>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) } namespace ErrorCodes @@ -244,18 +250,52 @@ std::shared_ptr StorageObjectStorageSource::createFileIterator( // If paths contains a value, validate the extracted paths and use the key-based iterator // (even if the result is empty, indicating no scanning is required). if (!paths) + { + std::shared_ptr object_iterator = nullptr; + std::unique_ptr cache_ptr = nullptr; + + if (local_context->getSettingsRef()[Setting::use_object_storage_list_objects_cache] && object_storage->supportsListObjectsCache()) + { + auto & cache = ObjectStorageListObjectsCache::instance(); + ObjectStorageListObjectsCache::Key cache_key {object_storage->getDescription(), configuration->getNamespace(), configuration->getRawPath().cutGlobs(configuration->supportsPartialPathPrefix()), with_tags}; + + if (auto objects_info = cache.get(cache_key, /*filter_by_prefix=*/ false)) + { + /// suboptimal because of the recent upstream changes to the ObjectInfo structure + /// re-think this with more time and see if there is a more optimized approach + RelativePathsWithMetadata relative_path_with_metadata; + relative_path_with_metadata.reserve(objects_info->size()); + + for (const auto & object_info : *objects_info) + { + relative_path_with_metadata.emplace_back(std::make_shared(object_info->getPath(), object_info->getObjectMetadata())); + } + + object_iterator = std::make_shared(std::move(relative_path_with_metadata)); + } + else + { + cache_ptr = std::make_unique(cache, cache_key); + object_iterator = object_storage->iterate(configuration->getRawPath().cutGlobs(configuration->supportsPartialPathPrefix()), query_settings.list_object_keys_size, with_tags, std::nullopt); + } + } + else + { + object_iterator = object_storage->iterate(configuration->getRawPath().cutGlobs(configuration->supportsPartialPathPrefix()), query_settings.list_object_keys_size, with_tags, std::nullopt); + } + iterator = std::make_unique( - object_storage, + object_iterator, configuration, predicate, virtual_columns, hive_columns, local_context, is_archive ? nullptr : read_keys, - query_settings.list_object_keys_size, query_settings.throw_on_zero_files_match, - with_tags, - file_progress_callback); + file_progress_callback, + std::move(cache_ptr)); + } else { // Validate that extracted paths match the glob pattern to prevent scanning unallowed data @@ -1292,19 +1332,18 @@ std::unique_ptr createReadBuffer( } StorageObjectStorageSource::GlobIterator::GlobIterator( - ObjectStoragePtr object_storage_, - StorageObjectStorageConfigurationPtr configuration_, + const ObjectStorageIteratorPtr & object_storage_iterator_, + ConfigurationPtr configuration_, const ActionsDAG::Node * predicate, const NamesAndTypesList & virtual_columns_, const NamesAndTypesList & hive_columns_, ContextPtr context_, ObjectInfos * read_keys_, - size_t list_object_keys_size, bool throw_on_zero_files_match_, - bool with_tags, - std::function file_progress_callback_) + std::function file_progress_callback_, + std::unique_ptr list_cache_) : WithContext(context_) - , object_storage(object_storage_) + , object_storage_iterator(object_storage_iterator_) , configuration(configuration_) , virtual_columns(virtual_columns_) , hive_columns(hive_columns_) @@ -1313,6 +1352,7 @@ StorageObjectStorageSource::GlobIterator::GlobIterator( , read_keys(read_keys_) , local_context(context_) , file_progress_callback(file_progress_callback_) + , list_cache(std::move(list_cache_)) { const auto & reading_path = configuration->getPathForRead(); if (reading_path.hasGlobs()) @@ -1320,8 +1360,6 @@ StorageObjectStorageSource::GlobIterator::GlobIterator( const auto & key_with_globs = reading_path; const auto key_prefix = reading_path.cutGlobs(configuration->supportsPartialPathPrefix()); - object_storage_iterator = object_storage->iterate(key_prefix, list_object_keys_size, with_tags, std::nullopt); - matcher = std::make_unique(makeRegexpPatternFromGlobs(key_with_globs.path)); if (!matcher->ok()) { @@ -1387,6 +1425,10 @@ ObjectInfoPtr StorageObjectStorageSource::GlobIterator::nextUnlocked(size_t /* p auto result = object_storage_iterator->getCurrentBatchAndScheduleNext(); if (!result.has_value()) { + if (list_cache) + { + list_cache->set(std::move(object_list)); + } is_finished = true; LOG_DEBUG(log, "Listing finished: total_listed={}, glob_filtered={}, predicate_filtered={}", total_listed, total_glob_filtered, total_predicate_filtered); @@ -1405,6 +1447,11 @@ ObjectInfoPtr StorageObjectStorageSource::GlobIterator::nextUnlocked(size_t /* p listed_in_batch = new_batch.size(); + if (list_cache) + { + object_list.insert(object_list.end(), new_batch.begin(), new_batch.end()); + } + for (auto it = new_batch.begin(); it != new_batch.end();) { if (!recursive && !re2::RE2::FullMatch((*it)->getPath(), *matcher)) diff --git a/src/Storages/ObjectStorage/StorageObjectStorageSource.h b/src/Storages/ObjectStorage/StorageObjectStorageSource.h index 3e72bbd15b08..ab145708d8c6 100644 --- a/src/Storages/ObjectStorage/StorageObjectStorageSource.h +++ b/src/Storages/ObjectStorage/StorageObjectStorageSource.h @@ -12,6 +12,11 @@ #include #include #include +<<<<<<< HEAD +======= +#include + +>>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) namespace DB { @@ -183,18 +188,33 @@ class StorageObjectStorageSource::ReadTaskIterator : public IObjectIterator, pri class StorageObjectStorageSource::GlobIterator : public IObjectIterator, WithContext { public: + struct ListObjectsCacheWithKey + { + ListObjectsCacheWithKey(ObjectStorageListObjectsCache & cache_, const ObjectStorageListObjectsCache::Key & key_) : cache(cache_), key(key_) {} + + void set(ObjectStorageListObjectsCache::Value && value) const + { + cache.set(key, std::make_shared(std::move(value))); + } + + private: + ObjectStorageListObjectsCache & cache; + ObjectStorageListObjectsCache::Key key; + }; + + using ConfigurationPtr = std::shared_ptr; + GlobIterator( - ObjectStoragePtr object_storage_, - StorageObjectStorageConfigurationPtr configuration_, + const ObjectStorageIteratorPtr & object_storage_iterator_, + ConfigurationPtr configuration_, const ActionsDAG::Node * predicate, const NamesAndTypesList & virtual_columns_, const NamesAndTypesList & hive_columns_, ContextPtr context_, ObjectInfos * read_keys_, - size_t list_object_keys_size, bool throw_on_zero_files_match_, - bool with_tags, - std::function file_progress_callback_ = {}); + std::function file_progress_callback_ = {}, + std::unique_ptr list_cache_ = nullptr); ~GlobIterator() override = default; @@ -207,7 +227,7 @@ class StorageObjectStorageSource::GlobIterator : public IObjectIterator, WithCon void createFilterAST(const String & any_key); void fillBufferForKey(const std::string & uri_key); - const ObjectStoragePtr object_storage; + ObjectStorageIteratorPtr object_storage_iterator; const StorageObjectStorageConfigurationPtr configuration; const NamesAndTypesList virtual_columns; const NamesAndTypesList hive_columns; @@ -219,7 +239,6 @@ class StorageObjectStorageSource::GlobIterator : public IObjectIterator, WithCon ObjectInfos object_infos; ObjectInfos * read_keys; ExpressionActionsPtr filter_expr; - ObjectStorageIteratorPtr object_storage_iterator; bool recursive{false}; std::vector expanded_keys; std::vector::iterator expanded_keys_iter; @@ -236,6 +255,9 @@ class StorageObjectStorageSource::GlobIterator : public IObjectIterator, WithCon size_t total_listed = 0; size_t total_glob_filtered = 0; size_t total_predicate_filtered = 0; + + std::unique_ptr list_cache; + ObjectInfos object_list; }; class StorageObjectStorageSource::KeysIterator : public IObjectIterator diff --git a/src/Storages/PartitionCommands.cpp b/src/Storages/PartitionCommands.cpp index 593f4d895aad..c4d038f925c1 100644 --- a/src/Storages/PartitionCommands.cpp +++ b/src/Storages/PartitionCommands.cpp @@ -131,6 +131,37 @@ std::optional PartitionCommand::parse(const ASTAlterCommand * res.with_name = command_ast->with_name; return res; } + if (command_ast->type == ASTAlterCommand::EXPORT_PART) + { + PartitionCommand res; + res.type = EXPORT_PART; + res.partition = command_ast->partition->clone(); + res.part = command_ast->part; + res.to_database = command_ast->to_database; + res.to_table = command_ast->to_table; + if (command_ast->to_table_function) + { + res.to_table_function = command_ast->to_table_function->ptr(); + if (command_ast->partition_by_expr) + res.partition_by_expr = command_ast->partition_by_expr->clone(); + } + return res; + } + if (command_ast->type == ASTAlterCommand::EXPORT_PARTITION) + { + PartitionCommand res; + res.type = EXPORT_PARTITION; + res.partition = command_ast->partition->clone(); + res.to_database = command_ast->to_database; + res.to_table = command_ast->to_table; + if (command_ast->to_table_function) + { + res.to_table_function = command_ast->to_table_function->ptr(); + if (command_ast->partition_by_expr) + res.partition_by_expr = command_ast->partition_by_expr->clone(); + } + return res; + } return {}; } @@ -172,6 +203,10 @@ std::string PartitionCommand::typeToString() const return "UNFREEZE ALL"; case PartitionCommand::Type::REPLACE_PARTITION: return "REPLACE PARTITION"; + case PartitionCommand::Type::EXPORT_PART: + return "EXPORT PART"; + case PartitionCommand::Type::EXPORT_PARTITION: + return "EXPORT PARTITION"; default: throw Exception(ErrorCodes::LOGICAL_ERROR, "Uninitialized partition command"); } diff --git a/src/Storages/PartitionCommands.h b/src/Storages/PartitionCommands.h index 399a4b64c478..9c476d3c8e4f 100644 --- a/src/Storages/PartitionCommands.h +++ b/src/Storages/PartitionCommands.h @@ -32,6 +32,8 @@ struct PartitionCommand UNFREEZE_ALL_PARTITIONS, UNFREEZE_PARTITION, REPLACE_PARTITION, + EXPORT_PART, + EXPORT_PARTITION, }; Type type = UNKNOWN; @@ -49,10 +51,14 @@ struct PartitionCommand String from_table; bool replace = true; - /// For MOVE PARTITION + /// For MOVE PARTITION and EXPORT PART and EXPORT PARTITION String to_database; String to_table; + /// For EXPORT PART and EXPORT PARTITION with table functions + ASTPtr to_table_function; + ASTPtr partition_by_expr; + /// For FETCH PARTITION - path in ZK to the shard, from which to download the partition. String from_path; diff --git a/src/Storages/PartitionedSink.cpp b/src/Storages/PartitionedSink.cpp index 8af531716d4e..fbc41f6be4d3 100644 --- a/src/Storages/PartitionedSink.cpp +++ b/src/Storages/PartitionedSink.cpp @@ -26,10 +26,12 @@ namespace ErrorCodes PartitionedSink::PartitionedSink( std::shared_ptr partition_strategy_, + std::shared_ptr sink_creator_, ContextPtr context_, SharedHeader source_header_) : SinkToStorage(source_header_) , partition_strategy(partition_strategy_) + , sink_creator(sink_creator_) , context(context_) , source_header(source_header_) { @@ -41,7 +43,7 @@ SinkPtr PartitionedSink::getSinkForPartitionKey(std::string_view partition_key) auto it = partition_id_to_sink.find(partition_key); if (it == partition_id_to_sink.end()) { - auto sink = createSinkForPartition(std::string{partition_key}); + auto sink = sink_creator->createSinkForPartition(std::string{partition_key}); std::tie(it, std::ignore) = partition_id_to_sink.emplace(partition_key, sink); } diff --git a/src/Storages/PartitionedSink.h b/src/Storages/PartitionedSink.h index af2b88baad27..1f612076db92 100644 --- a/src/Storages/PartitionedSink.h +++ b/src/Storages/PartitionedSink.h @@ -16,10 +16,17 @@ namespace DB class PartitionedSink : public SinkToStorage { public: + struct SinkCreator + { + virtual ~SinkCreator() = default; + virtual SinkPtr createSinkForPartition(const String & partition_id) = 0; + }; + static constexpr auto PARTITION_ID_WILDCARD = "{_partition_id}"; PartitionedSink( std::shared_ptr partition_strategy_, + std::shared_ptr sink_creator_, ContextPtr context_, SharedHeader source_header_); @@ -33,16 +40,15 @@ class PartitionedSink : public SinkToStorage void onFinish() override; - virtual SinkPtr createSinkForPartition(const String & partition_id) = 0; - static void validatePartitionKey(const String & str, bool allow_slash); static String replaceWildcards(const String & haystack, const String & partition_id); + protected: std::shared_ptr partition_strategy; - private: + std::shared_ptr sink_creator; ContextPtr context; SharedHeader source_header; diff --git a/src/Storages/StorageFile.cpp b/src/Storages/StorageFile.cpp index 643bef2f593e..b5bd1f50ac5e 100644 --- a/src/Storages/StorageFile.cpp +++ b/src/Storages/StorageFile.cpp @@ -2213,7 +2213,7 @@ class StorageFileSink final : public SinkToStorage, WithContext std::unique_lock lock; }; -class PartitionedStorageFileSink : public PartitionedSink +class PartitionedStorageFileSink : public PartitionedSink::SinkCreator { public: PartitionedStorageFileSink( @@ -2228,7 +2228,7 @@ class PartitionedStorageFileSink : public PartitionedSink const String format_name_, ContextPtr context_, int flags_) - : PartitionedSink(partition_strategy_, context_, std::make_shared(metadata_snapshot_->getSampleBlock())) + : partition_strategy(partition_strategy_) , path(path_) , metadata_snapshot(metadata_snapshot_) , table_name_for_log(table_name_for_log_) @@ -2244,11 +2244,12 @@ class PartitionedStorageFileSink : public PartitionedSink SinkPtr createSinkForPartition(const String & partition_id) override { - std::string filepath = partition_strategy->getPathForWrite(path, partition_id); + const auto file_path_generator = std::make_shared(path); + std::string filepath = file_path_generator->getPathForWrite(partition_id); fs::create_directories(fs::path(filepath).parent_path()); - validatePartitionKey(filepath, true); + PartitionedSink::validatePartitionKey(filepath, true); checkCreationIsAllowed(context, context->getUserFilesPath(), filepath, /*can_be_directory=*/ true); return std::make_shared( metadata_snapshot, @@ -2265,6 +2266,7 @@ class PartitionedStorageFileSink : public PartitionedSink } private: + std::shared_ptr partition_strategy; const String path; StorageMetadataPtr metadata_snapshot; String table_name_for_log; @@ -2316,7 +2318,7 @@ SinkToStoragePtr StorageFile::write( has_wildcards, /* partition_columns_in_data_file */true); - return std::make_shared( + auto sink_creator = std::make_shared( partition_strategy, metadata_snapshot, getStorageID().getNameForLogs(), @@ -2328,6 +2330,13 @@ SinkToStoragePtr StorageFile::write( format_name, context, flags); + + return std::make_shared( + partition_strategy, + sink_creator, + context, + std::make_shared(metadata_snapshot->getSampleBlock()) + ); } String path; @@ -2353,6 +2362,7 @@ SinkToStoragePtr StorageFile::write( String new_path; do { + new_path = path.substr(0, pos) + "." + std::to_string(index) + (pos == std::string::npos ? "" : path.substr(pos)); ++index; } diff --git a/src/Storages/StorageMergeTree.cpp b/src/Storages/StorageMergeTree.cpp index b58b4e0dd1f6..3916130d1963 100644 --- a/src/Storages/StorageMergeTree.cpp +++ b/src/Storages/StorageMergeTree.cpp @@ -145,7 +145,11 @@ namespace ErrorCodes extern const int TOO_MANY_PARTS; extern const int PART_IS_LOCKED; extern const int PART_IS_TEMPORARILY_LOCKED; +<<<<<<< HEAD extern const int FAULT_INJECTED; +======= + extern const int INCOMPATIBLE_COLUMNS; +>>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) } namespace ActionLocks @@ -245,8 +249,12 @@ void StorageMergeTree::startup() { cleanup_thread.start(); background_operations_assignee.start(); +<<<<<<< HEAD background_streaming_assignee.start(); startBackgroundMovesIfNeeded(); +======= + startBackgroundMoves(); +>>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) startOutdatedAndUnexpectedDataPartsLoadingTask(); startStatisticsCache(); } @@ -311,6 +319,11 @@ void StorageMergeTree::shutdown(bool) if (deduplication_log) deduplication_log->shutdown(); + + { + std::lock_guard lock(export_manifests_mutex); + export_manifests.clear(); + } } @@ -3518,12 +3531,6 @@ MutationCounters StorageMergeTree::getMutationCounters() const return mutation_counters; } -void StorageMergeTree::startBackgroundMovesIfNeeded() -{ - if (areBackgroundMovesNeeded()) - background_moves_assignee.start(); -} - std::unique_ptr StorageMergeTree::getDefaultSettings() const { return std::make_unique(getContext()->getMergeTreeSettings()); diff --git a/src/Storages/StorageMergeTree.h b/src/Storages/StorageMergeTree.h index e3e823f1bae7..15db160702c3 100644 --- a/src/Storages/StorageMergeTree.h +++ b/src/Storages/StorageMergeTree.h @@ -316,8 +316,6 @@ class StorageMergeTree final : public MergeTreeData std::unique_ptr fillNewPartName(MutableDataPartPtr & part, DataPartsLock & lock); std::unique_ptr fillNewPartNameAndResetLevel(MutableDataPartPtr & part, DataPartsLock & lock); - void startBackgroundMovesIfNeeded() override; - BackupEntries backupMutations(UInt64 version, const String & data_path_in_backup) const; /// Attaches restored parts to the storage. diff --git a/src/Storages/StorageReplicatedMergeTree.cpp b/src/Storages/StorageReplicatedMergeTree.cpp index ad52354eb849..db58ba6cb776 100644 --- a/src/Storages/StorageReplicatedMergeTree.cpp +++ b/src/Storages/StorageReplicatedMergeTree.cpp @@ -7,6 +7,7 @@ #include #include +#include "Common/ZooKeeper/IKeeper.h" #include #include #include @@ -29,6 +30,7 @@ #include #include +#include #include #include #include @@ -72,12 +74,17 @@ #include #include #include +#include +#include +#include #include #include +#include #include #include #include #include +#include #include #include @@ -126,6 +133,14 @@ #include #include +#include "Functions/generateSnowflakeID.h" +#include "Interpreters/StorageID.h" +#include "QueryPipeline/QueryPlanResourceHolder.h" +#include "Storages/ExportReplicatedMergeTreePartitionManifest.h" +#include "Storages/ExportReplicatedMergeTreePartitionTaskEntry.h" +#include +#include +#include #include #include @@ -157,9 +172,21 @@ namespace ProfileEvents extern const Event ReplicaPartialShutdown; extern const Event ReplicatedCoveredPartsInZooKeeperOnStart; extern const Event MergesRejectedByMemoryLimit; +<<<<<<< HEAD extern const Event ZooKeeperWatchTriggeredReplicatedMergeTreeLeaderElection; extern const Event ZooKeeperWatchTriggeredReplicatedMergeTreeReplicaSync; extern const Event ZooKeeperWatchTriggeredReplicatedMergeTreeMutations; +======= + extern const Event ExportPartitionZooKeeperRequests; + extern const Event ExportPartitionZooKeeperGet; + extern const Event ExportPartitionZooKeeperGetChildren; + extern const Event ExportPartitionZooKeeperCreate; + extern const Event ExportPartitionZooKeeperSet; + extern const Event ExportPartitionZooKeeperRemove; + extern const Event ExportPartitionZooKeeperRemoveRecursive; + extern const Event ExportPartitionZooKeeperMulti; + extern const Event ExportPartitionZooKeeperExists; +>>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) } namespace CurrentMetrics @@ -197,6 +224,25 @@ namespace Setting extern const SettingsInt64 replication_wait_for_inactive_replica_timeout; extern const SettingsUInt64 select_sequential_consistency; extern const SettingsBool update_sequential_consistency; + extern const SettingsBool allow_experimental_export_merge_tree_part; + extern const SettingsBool export_merge_tree_partition_force_export; + extern const SettingsUInt64 export_merge_tree_partition_max_retries; + extern const SettingsUInt64 export_merge_tree_partition_manifest_ttl; + extern const SettingsUInt64 export_merge_tree_partition_task_timeout_seconds; + extern const SettingsBool output_format_parallel_formatting; + extern const SettingsBool output_format_parquet_parallel_encoding; + extern const SettingsMaxThreads max_threads; + extern const SettingsMergeTreePartExportFileAlreadyExistsPolicy export_merge_tree_part_file_already_exists_policy; + extern const SettingsUInt64 export_merge_tree_part_max_bytes_per_file; + extern const SettingsUInt64 export_merge_tree_part_max_rows_per_file; + extern const SettingsBool export_merge_tree_part_throw_on_pending_mutations; + extern const SettingsBool export_merge_tree_part_throw_on_pending_patch_parts; + extern const SettingsBool export_merge_tree_partition_lock_inside_the_task; + extern const SettingsString export_merge_tree_part_filename_pattern; + extern const SettingsBool write_full_path_in_iceberg_metadata; + extern const SettingsBool allow_insert_into_iceberg; + extern const SettingsUInt64 iceberg_insert_max_bytes_in_data_file; + extern const SettingsUInt64 iceberg_insert_max_rows_in_data_file; } @@ -310,6 +356,13 @@ namespace ErrorCodes extern const int FAULT_INJECTED; extern const int CANNOT_FORGET_PARTITION; extern const int TIMEOUT_EXCEEDED; + extern const int INVALID_SETTING_VALUE; + extern const int PENDING_MUTATIONS_NOT_ALLOWED; +} + +namespace ServerSetting +{ + extern const ServerSettingsBool allow_experimental_export_merge_tree_partition; } namespace ActionLocks @@ -444,6 +497,9 @@ StorageReplicatedMergeTree::StorageReplicatedMergeTree( , merge_strategy_picker(*this) , queue(*this, merge_strategy_picker) , fetcher(*this) + , export_merge_tree_partition_task_entries_by_key(export_merge_tree_partition_task_entries.get()) + , export_merge_tree_partition_task_entries_by_transaction_id(export_merge_tree_partition_task_entries.get()) + , export_merge_tree_partition_task_entries_by_create_time(export_merge_tree_partition_task_entries.get()) , cleanup_thread(*this) , deduplication_hashes_cache(*this, "deduplication_hashes") , async_block_ids_cache(*this, "async_blocks") @@ -495,6 +551,31 @@ StorageReplicatedMergeTree::StorageReplicatedMergeTree( /// Will be activated by restarting thread. mutations_finalizing_task->deactivate(); + if (getContext()->getServerSettings()[ServerSetting::allow_experimental_export_merge_tree_partition]) + { + export_merge_tree_partition_manifest_updater = std::make_shared(*this); + + export_merge_tree_partition_task_scheduler = std::make_shared(*this); + + export_merge_tree_partition_updating_task = getContext()->getSchedulePool().createTask( + getStorageID(), getStorageID().getFullTableName() + " (StorageReplicatedMergeTree::export_merge_tree_partition_updating_task)", [this] { exportMergeTreePartitionUpdatingTask(); }); + + export_merge_tree_partition_updating_task->deactivate(); + + export_merge_tree_partition_status_handling_task = getContext()->getSchedulePool().createTask( + getStorageID(), getStorageID().getFullTableName() + " (StorageReplicatedMergeTree::export_merge_tree_partition_status_handling_task)", [this] { exportMergeTreePartitionStatusHandlingTask(); }); + + export_merge_tree_partition_status_handling_task->deactivate(); + + export_merge_tree_partition_watch_callback = export_merge_tree_partition_updating_task->getWatchCallback(); + + export_merge_tree_partition_select_task = getContext()->getSchedulePool().createTask( + getStorageID(), getStorageID().getFullTableName() + " (StorageReplicatedMergeTree::export_merge_tree_partition_select_task)", [this] { selectPartsToExport(); }); + + export_merge_tree_partition_select_task->deactivate(); + } + + bool has_zookeeper = getContext()->hasZooKeeper() || getContext()->hasAuxiliaryZooKeeper(zookeeper_info.zookeeper_name); auto component_guard = Coordination::setCurrentComponent("StorageReplicatedMergeTree::StorageReplicatedMergeTree"); if (has_zookeeper) @@ -949,6 +1030,7 @@ void StorageReplicatedMergeTree::createNewZooKeeperNodesAttempt() const futures.push_back(zookeeper->asyncTryCreateNoThrow(zookeeper_path + "/quorum/last_part", String(), zkutil::CreateMode::Persistent)); futures.push_back(zookeeper->asyncTryCreateNoThrow(zookeeper_path + "/quorum/failed_parts", String(), zkutil::CreateMode::Persistent)); futures.push_back(zookeeper->asyncTryCreateNoThrow(zookeeper_path + "/mutations", String(), zkutil::CreateMode::Persistent)); + futures.push_back(zookeeper->asyncTryCreateNoThrow(zookeeper_path + "/exports", String(), zkutil::CreateMode::Persistent)); futures.push_back(zookeeper->asyncTryCreateNoThrow(zookeeper_path + "/quorum/parallel", String(), zkutil::CreateMode::Persistent)); @@ -1121,6 +1203,8 @@ bool StorageReplicatedMergeTree::createTableIfNotExistsAttempt(const StorageMeta zkutil::CreateMode::Persistent)); ops.emplace_back(zkutil::makeCreateRequest(zookeeper_path + "/mutations", "", zkutil::CreateMode::Persistent)); + ops.emplace_back(zkutil::makeCreateRequest(zookeeper_path + "/exports", "", + zkutil::CreateMode::Persistent)); /// And create first replica atomically. See also "createReplica" method that is used to create not the first replicas. @@ -4624,6 +4708,94 @@ void StorageReplicatedMergeTree::mutationsFinalizingTask() } } +void StorageReplicatedMergeTree::exportMergeTreePartitionUpdatingTask() +{ + auto component_guard = Coordination::setCurrentComponent("StorageReplicatedMergeTree::exportMergeTreePartitionUpdatingTask"); + try + { + export_merge_tree_partition_manifest_updater->poll(); + } + catch (const Coordination::Exception & e) + { + tryLogCurrentException(log, __PRETTY_FUNCTION__); + if (e.code == Coordination::Error::ZSESSIONEXPIRED) + { + LOG_DEBUG(log, "Export partition updating task: ZooKeeper session expired, waking up restarting thread"); + restarting_thread.wakeup(); + return; + } + } + catch (...) + { + tryLogCurrentException(log, __PRETTY_FUNCTION__); + } + + export_merge_tree_partition_updating_task->scheduleAfter(30 * 1000); +} + +void StorageReplicatedMergeTree::selectPartsToExport() +{ + auto component_guard = Coordination::setCurrentComponent("StorageReplicatedMergeTree::selectPartsToExport"); + try + { + if (parts_mover.moves_blocker.isCancelled()) + { + LOG_INFO(log, "Export partition select task: Moves are blocked, skipping"); + } + else + { + export_merge_tree_partition_task_scheduler->run(); + } + } + catch (...) + { + tryLogCurrentException(log, __PRETTY_FUNCTION__); + } + + export_merge_tree_partition_select_task->scheduleAfter(1000 * 5); +} + +void StorageReplicatedMergeTree::exportMergeTreePartitionStatusHandlingTask() +{ + auto component_guard = Coordination::setCurrentComponent("StorageReplicatedMergeTree::exportMergeTreePartitionStatusHandlingTask"); + try + { + export_merge_tree_partition_manifest_updater->handleStatusChanges(); + } + catch (const Coordination::Exception & e) + { + tryLogCurrentException(log, __PRETTY_FUNCTION__); + if (e.code == Coordination::Error::ZSESSIONEXPIRED) + { + restarting_thread.wakeup(); + } + else + { + /// if an exception is thrown, we might have unprocessed status changes, so we need to schedule the task again + export_merge_tree_partition_status_handling_task->scheduleAfter(5000); + } + + return; + } + catch (...) + { + tryLogCurrentException(log, __PRETTY_FUNCTION__); + export_merge_tree_partition_status_handling_task->scheduleAfter(5000); + } +} + +std::vector StorageReplicatedMergeTree::getPartitionExportsInfo(bool prefer_remote_information) const +{ + /// Called from a query thread (system.replicated_partition_exports), which does not have a component set. + auto component_guard = Coordination::setCurrentComponent("StorageReplicatedMergeTree::getPartitionExportsInfo"); + + if (prefer_remote_information && getZooKeeper()->isFeatureEnabled(DB::KeeperFeatureFlag::MULTI_READ)) + { + return export_merge_tree_partition_manifest_updater->getPartitionExportsInfo(); + } + + return export_merge_tree_partition_manifest_updater->getPartitionExportsInfoLocal(); +} StorageReplicatedMergeTree::CreateMergeEntryResult StorageReplicatedMergeTree::createLogEntryToMergeParts( zkutil::ZooKeeperPtr & zookeeper, @@ -5927,7 +6099,7 @@ void StorageReplicatedMergeTree::startupImpl(bool from_attach_thread, const ZooK restarting_thread.start(true); }); - startBackgroundMovesIfNeeded(); + startBackgroundMoves(); part_moves_between_shards_orchestrator.start(); @@ -6026,6 +6198,13 @@ void StorageReplicatedMergeTree::partialShutdown() mutations_updating_task->deactivate(); mutations_finalizing_task->deactivate(); + if (getContext()->getServerSettings()[ServerSetting::allow_experimental_export_merge_tree_partition]) + { + export_merge_tree_partition_updating_task->deactivate(); + export_merge_tree_partition_select_task->deactivate(); + export_merge_tree_partition_status_handling_task->deactivate(); + } + cleanup_thread.stop(); deduplication_hashes_cache.stop(); async_block_ids_cache.stop(); @@ -6099,6 +6278,17 @@ void StorageReplicatedMergeTree::shutdown(bool) /// Wait for all of them std::lock_guard lock(data_parts_exchange_ptr->rwlock); } + + { + std::lock_guard lock(export_merge_tree_partition_mutex); + export_merge_tree_partition_task_entries.clear(); + } + + { + std::lock_guard lock(export_manifests_mutex); + export_manifests.clear(); + } + LOG_TRACE(log, "Shutdown finished"); } @@ -8307,6 +8497,296 @@ void StorageReplicatedMergeTree::fetchPartition( LOG_TRACE(log, "Fetch took {} sec. ({} tries)", watch.elapsedSeconds(), try_no); } +void StorageReplicatedMergeTree::exportPartitionToTable(const PartitionCommand & command, ContextPtr query_context) +{ + auto component_guard = Coordination::setCurrentComponent("StorageReplicatedMergeTree::exportPartitionToTable"); + if (!query_context->getServerSettings()[ServerSetting::allow_experimental_export_merge_tree_partition]) + { + throw Exception(ErrorCodes::SUPPORT_IS_DISABLED, + "Exporting merge tree partition is experimental. Set the server setting `allow_experimental_export_merge_tree_partition` to enable it (on all replicas).\n" + "If you are exporting to an Apache Iceberg table, you also need to enable the setting `allow_experimental_insert_into_iceberg` on all replicas. The same goes for `allow_experimental_export_merge_tree_part`"); + } + + const auto dest_database = query_context->resolveDatabase(command.to_database); + const auto dest_table = command.to_table; + const auto dest_storage_id = StorageID(dest_database, dest_table); + auto dest_storage = DatabaseCatalog::instance().getTable({dest_database, dest_table}, query_context); + + if (dest_storage->getStorageID() == this->getStorageID()) + { + throw Exception(ErrorCodes::BAD_ARGUMENTS, "Exporting to the same table is not allowed"); + } + + if (!dest_storage->supportsImport(query_context)) + throw Exception(ErrorCodes::NOT_IMPLEMENTED, "Destination storage {} does not support MergeTree parts or uses unsupported partitioning", dest_storage->getName()); + + auto query_to_string = [] (const ASTPtr & ast) + { + return ast ? ast->formatWithSecretsOneLine() : ""; + }; + + auto src_snapshot = getInMemoryMetadataPtr(); + auto destination_snapshot = dest_storage->getInMemoryMetadataPtr(); + + /// compare all source readable columns with all destination insertable columns + /// this allows us to skip ephemeral columns + if (src_snapshot->getColumns().getReadable().sizeOfDifference(destination_snapshot->getColumns().getInsertable())) + throw Exception(ErrorCodes::INCOMPATIBLE_COLUMNS, "Tables have different structure"); + + /// for data lakes this check is performed later. It is a bit more complex as we need to convert the iceberg partition spec + /// to the MergeTree partition spec and compare the two. + if (!dest_storage->isDataLake()) + { + if (query_to_string(src_snapshot->getPartitionKeyAST()) != query_to_string(destination_snapshot->getPartitionKeyAST())) + throw Exception(ErrorCodes::BAD_ARGUMENTS, "Tables have different partition key"); + } + + + zkutil::ZooKeeperPtr zookeeper = getZooKeeperAndAssertNotReadonly(); + + const String partition_id = getPartitionIDFromQuery(command.partition, query_context); + + const auto exports_path = fs::path(zookeeper_path) / "exports"; + + const auto export_key = partition_id + "_" + dest_storage_id.getQualifiedName().getFullName(); + + const auto partition_exports_path = fs::path(exports_path) / export_key; + + /// check if entry already exists + ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRequests); + ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperExists); + if (zookeeper->exists(partition_exports_path)) + { + LOG_INFO(log, "Export with key {} is already exported or it is being exported. Checking if it has expired so that we can overwrite it", export_key); + + bool has_expired = false; + + ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRequests); + ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperExists); + if (zookeeper->exists(fs::path(partition_exports_path) / "metadata.json")) + { + std::string metadata_json; + ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRequests); + ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperGet); + if (zookeeper->tryGet(fs::path(partition_exports_path) / "metadata.json", metadata_json)) + { + const auto manifest = ExportReplicatedMergeTreePartitionManifest::fromJsonString(metadata_json); + + const auto now = time(nullptr); + const auto expiration_time = manifest.create_time + manifest.ttl_seconds; + + LOG_INFO(log, "Export with key {} has expiration time {}, now is {}", export_key, expiration_time, now); + + if (static_cast(expiration_time) < now) + { + has_expired = true; + } + } + } + + if (!has_expired && !query_context->getSettingsRef()[Setting::export_merge_tree_partition_force_export]) + { + throw Exception(ErrorCodes::BAD_ARGUMENTS, "Export with key {} already exported or it is being exported, and it has not expired. Set `export_merge_tree_partition_force_export` to overwrite it.", export_key); + } + + LOG_INFO(log, "Overwriting export with key {}", export_key); + + /// Not putting in ops (same transaction) because we can't construct a "tryRemoveRecursive" request. + /// It is possible that the zk being used does not support RemoveRecursive requests. + /// It is ok for this to be non transactional. Worst case scenario an on-going export is going to be killed and a new task won't be scheduled. + ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRequests); + ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRemoveRecursive); + zookeeper->tryRemoveRecursive(partition_exports_path); + } + + Coordination::Requests ops; + + ops.emplace_back(zkutil::makeCreateRequest(partition_exports_path, "", zkutil::CreateMode::Persistent)); + + DataPartsVector parts; + + { + auto data_parts_lock = lockParts(); + parts = getDataPartsVectorInPartitionForInternalUsage(MergeTreeDataPartState::Active, partition_id, data_parts_lock); + } + + if (parts.empty()) + { + throw Exception(ErrorCodes::BAD_ARGUMENTS, "Partition {} doesn't exist", partition_id); + } + + const bool throw_on_pending_mutations = query_context->getSettingsRef()[Setting::export_merge_tree_part_throw_on_pending_mutations]; + const bool throw_on_pending_patch_parts = query_context->getSettingsRef()[Setting::export_merge_tree_part_throw_on_pending_patch_parts]; + + MergeTreeData::IMutationsSnapshot::Params mutations_snapshot_params + { + .metadata_version = getInMemoryMetadataPtr()->getMetadataVersion(), + .min_part_metadata_version = MergeTreeData::getMinMetadataVersion(parts), + .need_data_mutations = throw_on_pending_mutations, + .need_alter_mutations = throw_on_pending_mutations || throw_on_pending_patch_parts, + .need_patch_parts = throw_on_pending_patch_parts, + }; + + const auto mutations_snapshot = getMutationsSnapshot(mutations_snapshot_params); + + std::vector part_names; + for (const auto & part : parts) + { + const auto alter_conversions = getAlterConversionsForPart(part, mutations_snapshot, query_context); + + /// re-check `throw_on_pending_mutations` because `pending_mutations` might have been filled due to `throw_on_pending_patch_parts` + if (alter_conversions->hasMutations() && throw_on_pending_mutations) + { + throw Exception(ErrorCodes::PENDING_MUTATIONS_NOT_ALLOWED, + "Partition {} can not be exported because the part {} has pending mutations. Either wait for the mutations to be applied or set `export_merge_tree_part_throw_on_pending_mutations` to false", + partition_id, + part->name); + } + + if (alter_conversions->hasPatches()) + { + throw Exception(ErrorCodes::PENDING_MUTATIONS_NOT_ALLOWED, + "Partition {} can not be exported because the part {} has pending patch parts. Either wait for the patch parts to be applied or set `export_merge_tree_part_throw_on_pending_patch_parts` to false", + partition_id, + part->name); + } + + part_names.push_back(part->name); + } + + /// TODO arthur somehow check if the list of parts is updated "enough" + + ExportReplicatedMergeTreePartitionManifest manifest; + + manifest.transaction_id = generateSnowflakeIDString(); + manifest.query_id = query_context->getCurrentQueryId(); + manifest.partition_id = partition_id; + manifest.destination_database = dest_database; + manifest.destination_table = dest_table; + manifest.source_replica = replica_name; + manifest.number_of_parts = part_names.size(); + manifest.parts = part_names; + manifest.create_time = time(nullptr); + manifest.max_retries = query_context->getSettingsRef()[Setting::export_merge_tree_partition_max_retries]; + manifest.ttl_seconds = query_context->getSettingsRef()[Setting::export_merge_tree_partition_manifest_ttl]; + manifest.task_timeout_seconds = query_context->getSettingsRef()[Setting::export_merge_tree_partition_task_timeout_seconds]; + manifest.max_threads = query_context->getSettingsRef()[Setting::max_threads]; + manifest.parallel_formatting = query_context->getSettingsRef()[Setting::output_format_parallel_formatting]; + manifest.parquet_parallel_encoding = query_context->getSettingsRef()[Setting::output_format_parquet_parallel_encoding]; + manifest.max_bytes_per_file = query_context->getSettingsRef()[Setting::export_merge_tree_part_max_bytes_per_file]; + manifest.max_rows_per_file = query_context->getSettingsRef()[Setting::export_merge_tree_part_max_rows_per_file]; + manifest.lock_inside_the_task = query_context->getSettingsRef()[Setting::export_merge_tree_partition_lock_inside_the_task]; + + manifest.file_already_exists_policy = query_context->getSettingsRef()[Setting::export_merge_tree_part_file_already_exists_policy].value; + manifest.filename_pattern = query_context->getSettingsRef()[Setting::export_merge_tree_part_filename_pattern].value; + manifest.write_full_path_in_iceberg_metadata = query_context->getSettingsRef()[Setting::write_full_path_in_iceberg_metadata]; + + if (dest_storage->isDataLake()) + { +#if USE_AVRO + auto * object_storage = dynamic_cast(dest_storage.get()); + + /// in theory this should never happen, but just in case + if (!object_storage) + { + throw Exception(ErrorCodes::BAD_ARGUMENTS, "Destination storage {} is not a StorageObjectStorage", dest_storage->getName()); + } + + auto * iceberg_metadata = dynamic_cast(object_storage->getExternalMetadata(query_context)); + if (!iceberg_metadata) + { + throw Exception(ErrorCodes::BAD_ARGUMENTS, "Destination storage {} is a data lake but not an iceberg table", dest_storage->getName()); + } + + if (!query_context->getSettingsRef()[Setting::allow_insert_into_iceberg]) + { + throw Exception(ErrorCodes::SUPPORT_IS_DISABLED, + "Iceberg writes are experimental. " + "To allow its usage, enable the setting allow_experimental_insert_into_iceberg (on all replicas). The same goes for `allow_experimental_export_merge_tree_partition` and `allow_experimental_export_merge_tree_part`"); + } + + const auto metadata_object = iceberg_metadata->getMetadataJSON(query_context); + + ExportPartitionUtils::verifyIcebergPartitionCompatibility( + metadata_object, src_snapshot->getPartitionKeyAST()); + + std::ostringstream oss; // STYLE_CHECK_ALLOW_STD_STRING_STREAM + oss.exceptions(std::ios::failbit); + metadata_object->stringify(oss); + manifest.iceberg_metadata_json = oss.str(); + + manifest.max_bytes_per_file = query_context->getSettingsRef()[Setting::iceberg_insert_max_bytes_in_data_file]; + manifest.max_rows_per_file = query_context->getSettingsRef()[Setting::iceberg_insert_max_rows_in_data_file]; + +#else + throw Exception(ErrorCodes::NOT_IMPLEMENTED, "Data lake export requires Avro support"); +#endif + } + + ops.emplace_back(zkutil::makeCreateRequest( + fs::path(partition_exports_path) / "metadata.json", + manifest.toJsonString(), + zkutil::CreateMode::Persistent)); + + ops.emplace_back(zkutil::makeCreateRequest( + fs::path(partition_exports_path) / "exceptions_per_replica", + "", + zkutil::CreateMode::Persistent)); + + ops.emplace_back(zkutil::makeCreateRequest( + fs::path(partition_exports_path) / "processing", + "", + zkutil::CreateMode::Persistent)); + + for (const auto & part : part_names) + { + ExportReplicatedMergeTreePartitionProcessingPartEntry entry; + entry.status = ExportReplicatedMergeTreePartitionProcessingPartEntry::Status::PENDING; + entry.part_name = part; + entry.retry_count = 0; + + ops.emplace_back(zkutil::makeCreateRequest( + fs::path(partition_exports_path) / "processing" / part, + entry.toJsonString(), + zkutil::CreateMode::Persistent)); + } + + ops.emplace_back(zkutil::makeCreateRequest( + fs::path(partition_exports_path) / "processed", + "", + zkutil::CreateMode::Persistent)); + + ops.emplace_back(zkutil::makeCreateRequest( + fs::path(partition_exports_path) / "locks", + "", + zkutil::CreateMode::Persistent)); + + /// status: IN_PROGRESS, COMPLETED, FAILED + ops.emplace_back(zkutil::makeCreateRequest( + fs::path(partition_exports_path) / "status", + "PENDING", + zkutil::CreateMode::Persistent)); + + ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRequests); + ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperMulti); + Coordination::Responses responses; + Coordination::Error code = zookeeper->tryMulti(ops, responses); + + if (code != Coordination::Error::ZOK) + { + if (code == Coordination::Error::ZNODEEXISTS + && zkutil::getFailedOpIndex(code, responses) == 0) + { + /// Lost the race on the root export node. Current code already + /// validated (exists / expired / force) — so this is *always* a race. + throw Exception(ErrorCodes::BAD_ARGUMENTS, + "Export with key {} was created concurrently by another replica. Retry if needed", + export_key); + } + throw zkutil::KeeperException::fromPath(code, partition_exports_path); + } +} + void StorageReplicatedMergeTree::forgetPartition(const ASTPtr & partition, ContextPtr query_context) { @@ -9761,6 +10241,92 @@ CancellationCode StorageReplicatedMergeTree::killPartMoveToShard(const UUID & ta return part_moves_between_shards_orchestrator.killPartMoveToShard(task_uuid); } +CancellationCode StorageReplicatedMergeTree::killExportPartition(const String & transaction_id) +{ + /// Called from a query thread (KILL EXPORT PARTITION via InterpreterKillQueryQuery), which does not have a component set. + auto component_guard = Coordination::setCurrentComponent("StorageReplicatedMergeTree::killExportPartition"); + + auto try_set_status_to_killed = [this](const zkutil::ZooKeeperPtr & zk, const std::string & status_path) + { + Coordination::Stat stat; + std::string status_from_zk_string; + + if (!zk->tryGet(status_path, status_from_zk_string, &stat)) + { + /// found entry locally, but not in zk. It might have been deleted by another replica and we did not have time to update the local entry. + LOG_INFO(log, "Export partition task not found in zk, can not cancel it"); + return CancellationCode::CancelCannotBeSent; + } + + const auto status_from_zk = magic_enum::enum_cast(status_from_zk_string); + + if (!status_from_zk) + { + LOG_INFO(log, "Export partition task status is invalid, can not cancel it"); + return CancellationCode::CancelCannotBeSent; + } + + if (status_from_zk.value() != ExportReplicatedMergeTreePartitionTaskEntry::Status::PENDING) + { + LOG_INFO(log, "Export partition task is {}, can not cancel it", String(magic_enum::enum_name(status_from_zk.value()))); + return CancellationCode::CancelCannotBeSent; + } + + if (zk->trySet(status_path, String(magic_enum::enum_name(ExportReplicatedMergeTreePartitionTaskEntry::Status::KILLED)), stat.version) != Coordination::Error::ZOK) + { + LOG_INFO(log, "Status has been updated while trying to kill the export partition task, can not cancel it"); + return CancellationCode::CancelCannotBeSent; + } + + return CancellationCode::CancelSent; + }; + + std::lock_guard lock(export_merge_tree_partition_mutex); + + const auto zk = getZooKeeper(); + + /// if we have the entry locally, no need to list from zk. we can save some requests. + const auto & entry = export_merge_tree_partition_task_entries_by_transaction_id.find(transaction_id); + if (entry != export_merge_tree_partition_task_entries_by_transaction_id.end()) + { + LOG_INFO(log, "Export partition task found locally, trying to cancel it"); + /// found locally, no need to get children on zk + if (entry->status != ExportReplicatedMergeTreePartitionTaskEntry::Status::PENDING) + { + LOG_INFO(log, "Export partition task is not pending, can not cancel it"); + return CancellationCode::CancelCannotBeSent; + } + + return try_set_status_to_killed(zk, fs::path(zookeeper_path) / "exports" / entry->getCompositeKey() / "status"); + } + else + { + LOG_INFO(log, "Export partition task not found locally, trying to find it on zk"); + /// for some reason, we don't have the entry locally. ls on zk to find the entry + const auto exports_path = fs::path(zookeeper_path) / "exports"; + + const auto export_keys = zk->getChildren(exports_path); + String export_key_to_be_cancelled; + + for (const auto & export_key : export_keys) + { + std::string metadata_json; + if (!zk->tryGet(fs::path(exports_path) / export_key / "metadata.json", metadata_json)) + continue; + const auto manifest = ExportReplicatedMergeTreePartitionManifest::fromJsonString(metadata_json); + if (manifest.transaction_id == transaction_id) + { + LOG_INFO(log, "Export partition task found on zk, trying to cancel it"); + return try_set_status_to_killed(zk, fs::path(exports_path) / export_key / "status"); + } + } + } + + LOG_INFO(log, "Export partition task not found, can not cancel it"); + + return CancellationCode::NotFound; +} + void StorageReplicatedMergeTree::getCommitPartOps( Coordination::Requests & ops, const DataPartPtr & part, @@ -10321,13 +10887,6 @@ MutationCounters StorageReplicatedMergeTree::getMutationCounters() const return queue.getMutationCounters(); } -void StorageReplicatedMergeTree::startBackgroundMovesIfNeeded() -{ - if (areBackgroundMovesNeeded()) - background_moves_assignee.start(); -} - - std::unique_ptr StorageReplicatedMergeTree::getDefaultSettings() const { return std::make_unique(getContext()->getReplicatedMergeTreeSettings()); diff --git a/src/Storages/StorageReplicatedMergeTree.h b/src/Storages/StorageReplicatedMergeTree.h index 45e216ecd705..c35434749595 100644 --- a/src/Storages/StorageReplicatedMergeTree.h +++ b/src/Storages/StorageReplicatedMergeTree.h @@ -1,5 +1,10 @@ #pragma once +<<<<<<< HEAD +======= +#include +#include +>>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) #include #include #include @@ -10,6 +15,9 @@ #include #include #include +#include +#include +#include #include #include #include @@ -95,6 +103,8 @@ namespace DB class ZooKeeperWithFaultInjection; using ZooKeeperWithFaultInjectionPtr = std::shared_ptr; +struct ReplicatedPartitionExportInfo; + class StorageReplicatedMergeTree final : public MergeTreeData { public: @@ -371,6 +381,8 @@ class StorageReplicatedMergeTree final : public MergeTreeData using ShutdownDeadline = std::chrono::time_point; void waitForUniquePartsToBeFetchedByOtherReplicas(ShutdownDeadline shutdown_deadline); + std::vector getPartitionExportsInfo(bool prefer_remote_information) const; + private: std::atomic_bool are_restoring_replica {false}; @@ -395,6 +407,9 @@ class StorageReplicatedMergeTree final : public MergeTreeData friend class MergeFromLogEntryTask; friend class MutateFromLogEntryTask; friend class ReplicatedMergeMutateTaskBase; + friend class ExportPartitionManifestUpdatingTask; + friend class ExportPartitionTaskScheduler; + friend class ExportPartFromPartitionExportTask; using MergeStrategyPicker = ReplicatedMergeTreeMergeStrategyPicker; using LogEntry = ReplicatedMergeTreeLogEntry; @@ -507,6 +522,26 @@ class StorageReplicatedMergeTree final : public MergeTreeData /// A task that marks finished mutations as done. BackgroundSchedulePoolTaskHolder mutations_finalizing_task; + BackgroundSchedulePoolTaskHolder export_merge_tree_partition_updating_task; + + /// mostly handle kill operations + BackgroundSchedulePoolTaskHolder export_merge_tree_partition_status_handling_task; + std::shared_ptr export_merge_tree_partition_manifest_updater; + + std::shared_ptr export_merge_tree_partition_task_scheduler; + + Coordination::WatchCallbackPtr export_merge_tree_partition_watch_callback; + + std::mutex export_merge_tree_partition_mutex; + + BackgroundSchedulePoolTaskHolder export_merge_tree_partition_select_task; + + ExportPartitionTaskEntriesContainer export_merge_tree_partition_task_entries; + + // Convenience references to indexes + ExportPartitionTaskEntriesContainer::index::type & export_merge_tree_partition_task_entries_by_key; + ExportPartitionTaskEntriesContainer::index::type & export_merge_tree_partition_task_entries_by_transaction_id; + ExportPartitionTaskEntriesContainer::index::type & export_merge_tree_partition_task_entries_by_create_time; /// A thread that removes old parts, log entries, and blocks. ReplicatedMergeTreeCleanupThread cleanup_thread; @@ -739,6 +774,14 @@ class StorageReplicatedMergeTree final : public MergeTreeData /// Checks if some mutations are done and marks them as done. void mutationsFinalizingTask(); + void selectPartsToExport(); + + /// update in-memory list of partition exports + void exportMergeTreePartitionUpdatingTask(); + + /// handle status changes for export partition tasks + void exportMergeTreePartitionStatusHandlingTask(); + /** Write the selected parts to merge into the log, * Call when merge_selecting_mutex is locked. * Returns false if any part is not in ZK. @@ -927,6 +970,7 @@ class StorageReplicatedMergeTree final : public MergeTreeData void movePartitionToTable(const StoragePtr & dest_table, const ASTPtr & partition, ContextPtr query_context) override; void movePartitionToShard(const ASTPtr & partition, bool move_part, const String & to, ContextPtr query_context) override; CancellationCode killPartMoveToShard(const UUID & task_uuid) override; + CancellationCode killExportPartition(const String & transaction_id) override; void fetchPartition( const ASTPtr & partition, const StorageMetadataPtr & metadata_snapshot, @@ -934,7 +978,8 @@ class StorageReplicatedMergeTree final : public MergeTreeData bool fetch_part, ContextPtr query_context) override; void forgetPartition(const ASTPtr & partition, ContextPtr query_context) override; - + + void exportPartitionToTable(const PartitionCommand &, ContextPtr) override; /// NOTE: there are no guarantees for concurrent merges. Dropping part can /// be concurrently merged into some covering part and dropPart will do @@ -966,8 +1011,6 @@ class StorageReplicatedMergeTree final : public MergeTreeData MutationsSnapshotPtr getMutationsSnapshot(const IMutationsSnapshot::Params & params) const override; - void startBackgroundMovesIfNeeded() override; - /// Attaches restored parts to the storage. void attachRestoredParts(MutableDataPartsVector && parts, const std::optional & zookeeper_retries_info) override; diff --git a/src/Storages/StorageURL.cpp b/src/Storages/StorageURL.cpp index 9c061ab0cf80..c8107f659a81 100644 --- a/src/Storages/StorageURL.cpp +++ b/src/Storages/StorageURL.cpp @@ -752,7 +752,7 @@ void StorageURLSink::cancelBuffers() write_buf->cancel(); } -class PartitionedStorageURLSink : public PartitionedSink +class PartitionedStorageURLSink : public PartitionedSink::SinkCreator { public: PartitionedStorageURLSink( @@ -766,7 +766,7 @@ class PartitionedStorageURLSink : public PartitionedSink const CompressionMethod compression_method_, const HTTPHeaderEntries & headers_, const String & http_method_) - : PartitionedSink(partition_strategy_, context_, std::make_shared(sample_block_)) + : partition_strategy(partition_strategy_) , uri(uri_) , format(format_) , format_settings(format_settings_) @@ -781,7 +781,8 @@ class PartitionedStorageURLSink : public PartitionedSink SinkPtr createSinkForPartition(const String & partition_id) override { - std::string partition_path = partition_strategy->getPathForWrite(uri, partition_id); + const auto file_path_generator = std::make_shared(uri); + std::string partition_path = file_path_generator->getPathForWrite(partition_id); context->getRemoteHostFilter().checkURL(Poco::URI(partition_path)); return std::make_shared( @@ -789,6 +790,7 @@ class PartitionedStorageURLSink : public PartitionedSink } private: + std::shared_ptr partition_strategy; const String uri; const String format; const std::optional format_settings; @@ -1470,7 +1472,7 @@ SinkToStoragePtr IStorageURLBase::write(const ASTPtr & query, const StorageMetad has_wildcards, /* partition_columns_in_data_file */true); - return std::make_shared( + auto sink_creator = std::make_shared( partition_strategy, uri, format_name, @@ -1481,6 +1483,8 @@ SinkToStoragePtr IStorageURLBase::write(const ASTPtr & query, const StorageMetad compression_method, headers, http_method); + + return std::make_shared(partition_strategy, sink_creator, context, std::make_shared(metadata_snapshot->getSampleBlock())); } return std::make_shared( diff --git a/src/Storages/System/StorageSystemExports.cpp b/src/Storages/System/StorageSystemExports.cpp new file mode 100644 index 000000000000..1bac86870712 --- /dev/null +++ b/src/Storages/System/StorageSystemExports.cpp @@ -0,0 +1,71 @@ +#include +#include +#include +#include +#include +#include +#include +#include +#include + +namespace DB +{ + +ColumnsDescription StorageSystemExports::getColumnsDescription() +{ + return ColumnsDescription + { + {"source_database", std::make_shared(), "Name of the source database."}, + {"source_table", std::make_shared(), "Name of the source table."}, + {"destination_database", std::make_shared(), "Name of the destination database."}, + {"destination_table", std::make_shared(), "Name of the destination table."}, + {"create_time", std::make_shared(), "Date and time when the export command was received in the server."}, + {"part_name", std::make_shared(), "Name of the part"}, + {"query_id", std::make_shared(), "Query ID of the export operation."}, + {"destination_file_paths", std::make_shared(std::make_shared()), "File paths where the part is being exported."}, + {"elapsed", std::make_shared(), "The time elapsed (in seconds) since the export started."}, + {"rows_read", std::make_shared(), "The number of rows read from the exported part."}, + {"total_rows_to_read", std::make_shared(), "The total number of rows to read from the exported part."}, + {"total_size_bytes_compressed", std::make_shared(), "The total size of the compressed data in the exported part."}, + {"total_size_bytes_uncompressed", std::make_shared(), "The total size of the uncompressed data in the exported part."}, + {"bytes_read_uncompressed", std::make_shared(), "The number of uncompressed bytes read from the exported part."}, + {"memory_usage", std::make_shared(), "Current memory usage in bytes for the export operation."}, + {"peak_memory_usage", std::make_shared(), "Peak memory usage in bytes during the export operation."}, + }; +} + +void StorageSystemExports::fillData(MutableColumns & res_columns, ContextPtr context, const ActionsDAG::Node *, std::vector) const +{ + const auto access = context->getAccess(); + const bool check_access_for_tables = !access->isGranted(AccessType::SHOW_TABLES); + + for (const auto & export_info : context->getExportsList().get()) + { + if (check_access_for_tables && !access->isGranted(AccessType::SHOW_TABLES, export_info.source_database, export_info.source_table)) + continue; + + size_t i = 0; + res_columns[i++]->insert(export_info.source_database); + res_columns[i++]->insert(export_info.source_table); + res_columns[i++]->insert(export_info.destination_database); + res_columns[i++]->insert(export_info.destination_table); + res_columns[i++]->insert(export_info.create_time); + res_columns[i++]->insert(export_info.part_name); + res_columns[i++]->insert(export_info.query_id); + Array destination_file_paths_array; + destination_file_paths_array.reserve(export_info.destination_file_paths.size()); + for (const auto & file_path : export_info.destination_file_paths) + destination_file_paths_array.push_back(file_path); + res_columns[i++]->insert(destination_file_paths_array); + res_columns[i++]->insert(export_info.elapsed); + res_columns[i++]->insert(export_info.rows_read); + res_columns[i++]->insert(export_info.total_rows_to_read); + res_columns[i++]->insert(export_info.total_size_bytes_compressed); + res_columns[i++]->insert(export_info.total_size_bytes_uncompressed); + res_columns[i++]->insert(export_info.bytes_read_uncompressed); + res_columns[i++]->insert(export_info.memory_usage); + res_columns[i++]->insert(export_info.peak_memory_usage); + } +} + +} diff --git a/src/Storages/System/StorageSystemExports.h b/src/Storages/System/StorageSystemExports.h new file mode 100644 index 000000000000..e13fbfa26aaa --- /dev/null +++ b/src/Storages/System/StorageSystemExports.h @@ -0,0 +1,25 @@ +#pragma once + +#include + + +namespace DB +{ + +class Context; + + +class StorageSystemExports final : public IStorageSystemOneBlock +{ +public: + std::string getName() const override { return "SystemExports"; } + + static ColumnsDescription getColumnsDescription(); + +protected: + using IStorageSystemOneBlock::IStorageSystemOneBlock; + + void fillData(MutableColumns & res_columns, ContextPtr context, const ActionsDAG::Node *, std::vector) const override; +}; + +} diff --git a/src/Storages/System/StorageSystemMerges.cpp b/src/Storages/System/StorageSystemMerges.cpp index e152bcdf8a19..dcb09713bd3c 100644 --- a/src/Storages/System/StorageSystemMerges.cpp +++ b/src/Storages/System/StorageSystemMerges.cpp @@ -1,10 +1,14 @@ +<<<<<<< HEAD #include #include #include #include #include #include +======= +>>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) #include +#include #include #include diff --git a/src/Storages/System/StorageSystemReplicatedPartitionExports.cpp b/src/Storages/System/StorageSystemReplicatedPartitionExports.cpp new file mode 100644 index 000000000000..e088e4f77214 --- /dev/null +++ b/src/Storages/System/StorageSystemReplicatedPartitionExports.cpp @@ -0,0 +1,151 @@ +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include "Columns/ColumnString.h" +#include "Storages/VirtualColumnUtils.h" +#include + + +namespace DB +{ + +namespace Setting +{ + extern const SettingsBool export_merge_tree_partition_system_table_prefer_remote_information; +} + +ColumnsDescription StorageSystemReplicatedPartitionExports::getColumnsDescription() +{ + return ColumnsDescription + { + {"source_database", std::make_shared(), "Name of the source database."}, + {"source_table", std::make_shared(), "Name of the source table."}, + {"destination_database", std::make_shared(), "Name of the destination database."}, + {"destination_table", std::make_shared(), "Name of the destination table."}, + {"create_time", std::make_shared(), "Date and time when the export command was submitted"}, + {"partition_id", std::make_shared(), "ID of the partition"}, + {"transaction_id", std::make_shared(), "ID of the transaction."}, + {"query_id", std::make_shared(), "Query ID of the export operation."}, + {"source_replica", std::make_shared(), "Name of the source replica."}, + {"parts", std::make_shared(std::make_shared()), "List of part names to be exported."}, + {"parts_count", std::make_shared(), "Number of parts in the export."}, + {"parts_to_do", std::make_shared(), "Number of parts pending to be exported."}, + {"status", std::make_shared(), "Status of the export."}, + {"exception_replica", std::make_shared(), "Replica that caused the last exception"}, + {"last_exception", std::make_shared(), "Last exception message of any part (not necessarily the last global exception)"}, + {"exception_part", std::make_shared(), "Part that caused the last exception"}, + {"exception_count", std::make_shared(), "Number of global exceptions"}, + }; +} + +void StorageSystemReplicatedPartitionExports::fillData(MutableColumns & res_columns, ContextPtr context, const ActionsDAG::Node * predicate, std::vector) const +{ + const auto access = context->getAccess(); + const bool check_access_for_databases = !access->isGranted(AccessType::SHOW_TABLES); + + std::map> replicated_merge_tree_tables; + for (const auto & db : DatabaseCatalog::instance().getDatabases(GetDatabasesOptions{.with_datalake_catalogs = false})) + { + /// skip data lakes + if (db.second->isExternal()) + continue; + + const bool check_access_for_tables = check_access_for_databases && !access->isGranted(AccessType::SHOW_TABLES, db.first); + + for (auto iterator = db.second->getTablesIterator(context); iterator->isValid(); iterator->next()) + { + const auto & table = iterator->table(); + if (!table) + continue; + + StorageReplicatedMergeTree * table_replicated = dynamic_cast(table.get()); + if (!table_replicated) + continue; + + if (check_access_for_tables && !access->isGranted(AccessType::SHOW_TABLES, db.first, iterator->name())) + continue; + + replicated_merge_tree_tables[db.first][iterator->name()] = table; + } + } + + MutableColumnPtr col_database_mut = ColumnString::create(); + MutableColumnPtr col_table_mut = ColumnString::create(); + + for (auto & db : replicated_merge_tree_tables) + { + for (auto & table : db.second) + { + col_database_mut->insert(db.first); + col_table_mut->insert(table.first); + } + } + + ColumnPtr col_database = std::move(col_database_mut); + ColumnPtr col_table = std::move(col_table_mut); + + /// Determine what tables are needed by the conditions in the query. + { + Block filtered_block + { + { col_database, std::make_shared(), "database" }, + { col_table, std::make_shared(), "table" }, + }; + + VirtualColumnUtils::filterBlockWithPredicate(predicate, filtered_block, context); + + if (!filtered_block.rows()) + return; + + col_database = filtered_block.getByName("database").column; + col_table = filtered_block.getByName("table").column; + } + + for (size_t i_storage = 0; i_storage < col_database->size(); ++i_storage) + { + const auto database = (*col_database)[i_storage].safeGet(); + const auto table = (*col_table)[i_storage].safeGet(); + + std::vector partition_exports_info; + { + const IStorage * storage = replicated_merge_tree_tables[database][table].get(); + if (const auto * replicated_merge_tree = dynamic_cast(storage)) + partition_exports_info = replicated_merge_tree->getPartitionExportsInfo(context->getSettingsRef()[Setting::export_merge_tree_partition_system_table_prefer_remote_information]); + } + + for (const ReplicatedPartitionExportInfo & info : partition_exports_info) + { + std::size_t i = 0; + res_columns[i++]->insert(database); + res_columns[i++]->insert(table); + res_columns[i++]->insert(info.destination_database); + res_columns[i++]->insert(info.destination_table); + res_columns[i++]->insert(info.create_time); + res_columns[i++]->insert(info.partition_id); + res_columns[i++]->insert(info.transaction_id); + res_columns[i++]->insert(info.query_id); + res_columns[i++]->insert(info.source_replica); + Array parts_array; + parts_array.reserve(info.parts.size()); + for (const auto & part : info.parts) + parts_array.push_back(part); + res_columns[i++]->insert(parts_array); + res_columns[i++]->insert(info.parts_count); + res_columns[i++]->insert(info.parts_to_do); + res_columns[i++]->insert(info.status); + res_columns[i++]->insert(info.exception_replica); + res_columns[i++]->insert(info.last_exception); + res_columns[i++]->insert(info.exception_part); + res_columns[i++]->insert(info.exception_count); + } + } +} + +} diff --git a/src/Storages/System/StorageSystemReplicatedPartitionExports.h b/src/Storages/System/StorageSystemReplicatedPartitionExports.h new file mode 100644 index 000000000000..15eb54f38c0e --- /dev/null +++ b/src/Storages/System/StorageSystemReplicatedPartitionExports.h @@ -0,0 +1,43 @@ +#pragma once + +#include + +namespace DB +{ + +class Context; + +struct ReplicatedPartitionExportInfo +{ + String destination_database; + String destination_table; + String partition_id; + String transaction_id; + String query_id; + time_t create_time; + String source_replica; + size_t parts_count; + size_t parts_to_do; + std::vector parts; + String status; + std::string exception_replica; + std::string last_exception; + std::string exception_part; + size_t exception_count = 0; +}; + +class StorageSystemReplicatedPartitionExports final : public IStorageSystemOneBlock +{ +public: + + std::string getName() const override { return "SystemReplicatedPartitionExports"; } + + static ColumnsDescription getColumnsDescription(); + +protected: + using IStorageSystemOneBlock::IStorageSystemOneBlock; + + void fillData(MutableColumns & res_columns, ContextPtr context, const ActionsDAG::Node *, std::vector) const override; +}; + +} diff --git a/src/Storages/System/attachSystemTables.cpp b/src/Storages/System/attachSystemTables.cpp index ab37f168d51e..aa7c12a06a05 100644 --- a/src/Storages/System/attachSystemTables.cpp +++ b/src/Storages/System/attachSystemTables.cpp @@ -1,11 +1,12 @@ #include +#include #include "config.h" #include #include #include #include - +#include #include #include #include @@ -40,6 +41,7 @@ #include #include #include +#include #include #include #include @@ -134,6 +136,7 @@ # include #endif #include +#include #include @@ -159,7 +162,16 @@ namespace DB { +<<<<<<< HEAD void attachSystemTablesServer(ContextPtr context, IDatabase & system_database, bool has_zookeeper, [[maybe_unused]] bool has_keeper_server) +======= +namespace ServerSetting +{ + extern const ServerSettingsBool allow_experimental_export_merge_tree_partition; +} + +void attachSystemTablesServer(ContextPtr context, IDatabase & system_database, bool has_zookeeper) +>>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) { auto component_guard = Coordination::setCurrentComponent("attachSystemTablesServer"); attachNoDescription(context, system_database, "one", "This table contains a single row with a single dummy UInt8 column containing the value 0. Used when the table is not specified explicitly, for example in queries like `SELECT 1`."); @@ -253,6 +265,11 @@ void attachSystemTablesServer(ContextPtr context, IDatabase & system_database, b attach(context, system_database, "dimensional_metrics", "Contains dimensional metrics, which have multiple dimensions (labels) to provide more granular information. For example, counting failed merges by their error code. This table is always up to date."); attach(context, system_database, "merges", "Contains a list of merges currently executing merges of MergeTree tables and their progress. Each merge operation is represented by a single row."); attach(context, system_database, "moves", "Contains information about in-progress data part moves of MergeTree tables. Each data part movement is represented by a single row."); + attach(context, system_database, "exports", "Contains a list of exports currently executing exports of MergeTree tables and their progress. Each export operation is represented by a single row."); + if (context->getServerSettings()[ServerSetting::allow_experimental_export_merge_tree_partition]) + { + attach(context, system_database, "replicated_partition_exports", "Contains a list of partition exports of ReplicatedMergeTree tables and their progress. Each export operation is represented by a single row."); + } attach(context, system_database, "mutations", "Contains a list of mutations and their progress. Each mutation command is represented by a single row."); attachNoDescription(context, system_database, "replicas", "Contains information and status of all table replicas on current server. Each replica is represented by a single row."); attachNoDescription(context, system_database, "database_replicas", "Contains information and status of all database replicas on current server. Each database replica is represented by a single row."); diff --git a/tests/config/config.d/allow_experimental_export_merge_tree_partition.xml b/tests/config/config.d/allow_experimental_export_merge_tree_partition.xml new file mode 100644 index 000000000000..514cd710836a --- /dev/null +++ b/tests/config/config.d/allow_experimental_export_merge_tree_partition.xml @@ -0,0 +1,3 @@ + + 1 + diff --git a/tests/config/install.sh b/tests/config/install.sh index 359cd9cfd6f8..7f2be2bae858 100755 --- a/tests/config/install.sh +++ b/tests/config/install.sh @@ -101,6 +101,7 @@ ln -sf $SRC_PATH/config.d/predicate_statistics_log.xml $DEST_SERVER_PATH/config. ln -sf $SRC_PATH/config.d/custom_settings_prefixes.xml $DEST_SERVER_PATH/config.d/ ln -sf $SRC_PATH/config.d/database_catalog_drop_table_concurrency.xml $DEST_SERVER_PATH/config.d/ ln -sf $SRC_PATH/config.d/enable_access_control_improvements.xml $DEST_SERVER_PATH/config.d/ +ln -sf $SRC_PATH/config.d/allow_experimental_export_merge_tree_partition.xml $DEST_SERVER_PATH/config.d/ ln -sf $SRC_PATH/config.d/macros.xml $DEST_SERVER_PATH/config.d/ ln -sf $SRC_PATH/config.d/secure_ports.xml $DEST_SERVER_PATH/config.d/ ln -sf $SRC_PATH/config.d/clusters.xml $DEST_SERVER_PATH/config.d/ diff --git a/tests/integration/helpers/export_partition_helpers.py b/tests/integration/helpers/export_partition_helpers.py new file mode 100644 index 000000000000..d6bf78df0998 --- /dev/null +++ b/tests/integration/helpers/export_partition_helpers.py @@ -0,0 +1,201 @@ +""" +Shared helpers for export-partition and export-part integration tests. + +Centralises wait-polling, table creation, and partition helpers that were +previously duplicated across multiple test modules. +""" + +import time +import uuid + + +MINIO_USER = "minio" +MINIO_PASS = "ClickHouse_Minio_P@ssw0rd" + + +def wait_for_export_status( + node, + source_table, + dest_table, + partition_id, + expected_status="COMPLETED", + timeout=60, + poll_interval=0.5, +): + """Poll system.replicated_partition_exports until status matches. + + *dest_table* may be ``None`` to skip filtering by destination table + (useful for catalog-based tests where the destination is a database-qualified path). + """ + start_time = time.time() + last_status = None + while time.time() - start_time < timeout: + dest_filter = ( + f" AND destination_table = '{dest_table}'" if dest_table else "" + ) + status = node.query( + f"SELECT status FROM system.replicated_partition_exports" + f" WHERE source_table = '{source_table}'" + f"{dest_filter}" + f" AND partition_id = '{partition_id}'" + ).strip() + + last_status = status + if status and status == expected_status: + return status + + time.sleep(poll_interval) + + raise TimeoutError( + f"Export status did not reach '{expected_status}' within {timeout}s. " + f"Last status: '{last_status}'" + ) + + +def wait_for_export_to_start( + node, + source_table, + dest_table, + partition_id, + timeout=10, + poll_interval=0.2, +): + """Poll until at least one row exists in system.replicated_partition_exports.""" + start_time = time.time() + while time.time() - start_time < timeout: + count = node.query( + f"SELECT count() FROM system.replicated_partition_exports" + f" WHERE source_table = '{source_table}'" + f" AND destination_table = '{dest_table}'" + f" AND partition_id = '{partition_id}'" + ).strip() + + if count != "0": + return True + + time.sleep(poll_interval) + + raise TimeoutError( + f"Export of partition {partition_id!r} did not start within {timeout}s." + ) + + +def wait_for_exception_count( + node, + source_table, + dest_table, + partition_id, + min_exception_count=1, + timeout=30, + poll_interval=0.5, +): + """Wait for exception_count to reach at least *min_exception_count*.""" + start_time = time.time() + last_exception_count = None + while time.time() - start_time < timeout: + exception_count_str = node.query( + f"SELECT exception_count FROM system.replicated_partition_exports" + f" WHERE source_table = '{source_table}'" + f" AND destination_table = '{dest_table}'" + f" AND partition_id = '{partition_id}'" + f" SETTINGS export_merge_tree_partition_system_table_prefer_remote_information = 1" + ).strip() + + if exception_count_str: + exception_count = int(exception_count_str) + last_exception_count = exception_count + if exception_count >= min_exception_count: + return exception_count + + time.sleep(poll_interval) + + raise TimeoutError( + f"Exception count did not reach {min_exception_count} within {timeout}s. " + f"Last exception_count: {last_exception_count if last_exception_count is not None else 'N/A'}" + ) + + +# -- block-number settings are needed for patch parts support +_BLOCK_SETTINGS = ( + "enable_block_number_column = 1, enable_block_offset_column = 1" +) + + +def make_rmt( + node, + name, + columns, + partition_by, + replica_name="r1", + order_by="tuple()", +): + """Create a ReplicatedMergeTree table with block-number settings.""" + node.query( + f""" + CREATE TABLE {name} ({columns}) + ENGINE = ReplicatedMergeTree('/clickhouse/tables/{name}', '{replica_name}') + PARTITION BY {partition_by} + ORDER BY {order_by} + SETTINGS {_BLOCK_SETTINGS} + """ + ) + + +def make_mt( + node, + name, + columns, + partition_by, + order_by="tuple()", +): + """Create a MergeTree table with block-number settings.""" + node.query( + f""" + CREATE TABLE {name} ({columns}) + ENGINE = MergeTree() + PARTITION BY {partition_by} + ORDER BY {order_by} + SETTINGS {_BLOCK_SETTINGS} + """ + ) + + +def make_iceberg_s3( + node, + name, + columns, + partition_by="", + url=None, + s3_retry_attempts=3, + if_not_exists=False, +): + """Create an IcebergS3 table at a MinIO prefix. + + *url* defaults to ``http://minio1:9001/root/data/{name}/``. + """ + if url is None: + url = f"http://minio1:9001/root/data/{name}/" + ine = "IF NOT EXISTS " if if_not_exists else "" + pclause = f"PARTITION BY {partition_by}" if partition_by else "" + node.query( + f""" + CREATE TABLE {ine}{name} ({columns}) + ENGINE = IcebergS3('{url}', '{MINIO_USER}', '{MINIO_PASS}') + {pclause} + SETTINGS s3_retry_attempts = {s3_retry_attempts} + """ + ) + + +def first_partition_id(node, table): + """Return the partition_id of the first active part of *table*.""" + return node.query( + f"SELECT partition_id FROM system.parts" + f" WHERE database = currentDatabase() AND table = '{table}' AND active" + f" ORDER BY name LIMIT 1" + ).strip() + + +def unique_suffix(): + """Return a UUID with hyphens replaced by underscores, suitable for table names.""" + return str(uuid.uuid4()).replace("-", "_") diff --git a/tests/integration/helpers/iceberg_export_stats.py b/tests/integration/helpers/iceberg_export_stats.py new file mode 100644 index 000000000000..45a997922099 --- /dev/null +++ b/tests/integration/helpers/iceberg_export_stats.py @@ -0,0 +1,179 @@ +"""Shared helpers for verifying Iceberg per-file column statistics produced by +``EXPORT PART`` / ``EXPORT PARTITION``. + +Both the MergeTree and ReplicatedMergeTree export test modules drive the same +schema and expected stats shape (see ``assert_exported_stats``), so the +assertions, the manifest-entry reader, and the small byte/int decoders live +here instead of being duplicated in each test module. + +The only ClickHouse-side prerequisite is that ``system.iceberg_metadata_log`` +is enabled on the node: point the test cluster at +``configs/config.d/metadata_log.xml`` (shipped next to each test) and run the +probing SELECT with ``SETTINGS iceberg_metadata_log_level = 'manifest_file_entry'``. +""" + +import json + +from helpers.iceberg_utils import get_bound_for_column + + +# Iceberg assigns field ids positionally (starting at 1) to the non-partition +# columns in declaration order; partition-source columns share the same ids and +# partition transform outputs live in a separate 1000+ namespace. For the +# schema used by the two export-stats tests (id Int32, name String, +# tag Nullable(String), year Int32) this yields the mapping below. +STATS_FIELD_IDS = {"id": 1, "name": 2, "tag": 3} + + +def decode_int_bound(raw): + """Decode a JSON-serialized Iceberg integer bound (little-endian signed bytes). + + ClickHouse's Iceberg writer currently dumps integer bounds using the + underlying ``Field`` storage width (8 bytes for Int32/Int64/Date/...). Some + writers produce the spec-correct 4-byte Int32 encoding. Accept both. + """ + data = raw.encode("latin-1") + assert len(data) in (4, 8), f"Unexpected bound width {len(data)}: {raw!r}" + return int.from_bytes(data, "little", signed=True) + + +def get_int_for_column(m, column_id): + """Look up a value for ``column_id`` in an Iceberg integer map serialized as + either a dict ``{str(column_id): value}`` or a list of ``{key, value}`` records. + + Unlike :func:`helpers.iceberg_utils.get_bound_for_column`, this helper does + not try to unescape the value, so it works for numeric columns like + ``column_sizes`` and ``null_value_counts`` where the raw value is an int + (or a quoted int64 string). + """ + if m is None: + return None + value = None + if isinstance(m, dict): + value = m.get(str(column_id)) + elif isinstance(m, list): + for item in m: + if isinstance(item, dict) and item.get("key") == column_id: + value = item.get("value") + break + if value is None: + return None + return int(value) if isinstance(value, str) else value + + +def fetch_manifest_entries(node, query_id): + """Read JSON manifest-file entries emitted for ``query_id`` into + ``system.iceberg_metadata_log``. + + The outer ``FORMAT JSONEachRow`` is required: the default TSV format + escapes backslashes in the ``content`` string, which would double-encode + the inner ``\\uXXXX`` sequences coming from Iceberg bytes-bounds. + """ + node.query("SYSTEM FLUSH LOGS") + raw = node.query( + f""" + SELECT DISTINCT content + FROM system.iceberg_metadata_log + WHERE query_id = '{query_id}' + AND content_type = 'ManifestFileEntry' + AND content != '' + FORMAT JSONEachRow + """ + ) + entries = [] + for line in raw.strip().split("\n"): + if not line: + continue + outer = json.loads(line) + content = outer.get("content") + if content: + entries.append(json.loads(content)) + return entries + + +def assert_exported_stats(entries): + """Assert that at least one manifest entry describes the exported 2020 data file. + + Expected shape (three rows, one NULL in ``tag``): + + * ``record_count = 3`` + * ``file_size_in_bytes > 0`` + * ``column_sizes[id|name|tag] > 0`` + * ``null_value_counts = {id: 0, name: 0, tag: 1}`` + * ``lower_bounds = {id: 1, name: "aaa", tag: "x"}`` + * ``upper_bounds = {id: 3, name: "zzz", tag: "y"}`` + """ + assert entries, "No ManifestFileEntry rows recorded in system.iceberg_metadata_log" + + id_fid = STATS_FIELD_IDS["id"] + name_fid = STATS_FIELD_IDS["name"] + tag_fid = STATS_FIELD_IDS["tag"] + + matched = False + for entry in entries: + data_file = entry.get("data_file") or {} + # Skip manifest entries that explicitly mark themselves as deletes; data + # entries either omit `content` (v1 manifest) or set it to 0. + if data_file.get("content", 0) not in (0, None): + continue + + record_count = data_file.get("record_count") + if record_count != 3: + continue + + file_size = data_file.get("file_size_in_bytes") + assert file_size and file_size > 0, ( + f"Expected positive file_size_in_bytes, got {file_size!r}" + ) + + for field in ("id", "name", "tag"): + fid = STATS_FIELD_IDS[field] + size = get_int_for_column(data_file.get("column_sizes"), fid) + assert size is not None, ( + f"column_sizes missing entry for field_id={fid} ({field})" + ) + assert size > 0, f"column_sizes[{field}] expected > 0, got {size!r}" + + null_counts = data_file.get("null_value_counts") + assert get_int_for_column(null_counts, id_fid) == 0, ( + f"Expected 0 nulls in id, got null_value_counts={null_counts!r}" + ) + assert get_int_for_column(null_counts, name_fid) == 0, ( + f"Expected 0 nulls in name, got null_value_counts={null_counts!r}" + ) + assert get_int_for_column(null_counts, tag_fid) == 1, ( + f"Expected 1 null in tag (one NULL was inserted), " + f"got null_value_counts={null_counts!r}" + ) + + lower = data_file.get("lower_bounds") + upper = data_file.get("upper_bounds") + + assert decode_int_bound(get_bound_for_column(lower, id_fid)) == 1, ( + f"lower_bounds[id] expected 1, got {get_bound_for_column(lower, id_fid)!r}" + ) + assert decode_int_bound(get_bound_for_column(upper, id_fid)) == 3, ( + f"upper_bounds[id] expected 3, got {get_bound_for_column(upper, id_fid)!r}" + ) + + assert get_bound_for_column(lower, name_fid) == "aaa", ( + f"lower_bounds[name] expected 'aaa', got {get_bound_for_column(lower, name_fid)!r}" + ) + assert get_bound_for_column(upper, name_fid) == "zzz", ( + f"upper_bounds[name] expected 'zzz', got {get_bound_for_column(upper, name_fid)!r}" + ) + + assert get_bound_for_column(lower, tag_fid) == "x", ( + f"lower_bounds[tag] expected 'x' (nulls are skipped), got {get_bound_for_column(lower, tag_fid)!r}" + ) + assert get_bound_for_column(upper, tag_fid) == "y", ( + f"upper_bounds[tag] expected 'y' (nulls are skipped), got {get_bound_for_column(upper, tag_fid)!r}" + ) + + matched = True + break + + assert matched, ( + f"No data-file manifest entry with record_count=3 was found. " + f"Parsed {len(entries)} entr(y|ies) but none matched." + ) diff --git a/tests/integration/test_export_merge_tree_part_to_iceberg/configs/config.d/metadata_log.xml b/tests/integration/test_export_merge_tree_part_to_iceberg/configs/config.d/metadata_log.xml new file mode 100644 index 000000000000..c1fece21745c --- /dev/null +++ b/tests/integration/test_export_merge_tree_part_to_iceberg/configs/config.d/metadata_log.xml @@ -0,0 +1,7 @@ + + + system + iceberg_metadata_log
+ 10 +
+
diff --git a/tests/integration/test_export_merge_tree_part_to_iceberg/test.py b/tests/integration/test_export_merge_tree_part_to_iceberg/test.py new file mode 100644 index 000000000000..b371727a08cd --- /dev/null +++ b/tests/integration/test_export_merge_tree_part_to_iceberg/test.py @@ -0,0 +1,481 @@ +""" +Integration tests for EXPORT PART to an IcebergS3 destination. + +These tests cover the data-movement path from a plain MergeTree table to an +IcebergS3 table using the single-part export operation: + + ALTER TABLE EXPORT PART '' TO TABLE + +Coverage: + test_export_part_basic_to_iceberg – simple (id, year) schema; data + part_log checks + test_export_part_all_iceberg_types – schema covering all major Iceberg data types + test_export_multiple_parts_to_iceberg – two parts from different partitions land together + test_export_part_with_year_transform_partition – toYearNumSinceEpoch() partition expression + test_export_part_with_bucket_partition – icebergBucket(N, col) partition expression + test_export_part_partition_key_mismatch_is_rejected – mismatched partition spec rejected synchronously +""" + +import logging +import time + +import pytest + +from helpers.cluster import ClickHouseCluster +from helpers.export_partition_helpers import ( + first_partition_id, + make_iceberg_s3, + make_mt, + unique_suffix, +) +from helpers.iceberg_export_stats import ( + assert_exported_stats, + fetch_manifest_entries, +) + + +# --------------------------------------------------------------------------- +# Cluster fixture +# --------------------------------------------------------------------------- + + +@pytest.fixture(scope="module") +def cluster(): + try: + cluster = ClickHouseCluster(__file__) + cluster.add_instance( + "node1", + main_configs=["configs/config.d/metadata_log.xml"], + with_minio=True, + ) + logging.info("Starting cluster...") + cluster.start() + yield cluster + finally: + cluster.shutdown() + + +# --------------------------------------------------------------------------- +# Helpers +# --------------------------------------------------------------------------- + + +def get_part(node, table: str, partition_id: str) -> str: + """Return the name of the first active part of *table* in *partition_id*.""" + return node.query( + f"SELECT name FROM system.parts " + f"WHERE database = currentDatabase() AND table = '{table}' " + f"AND partition_id = '{partition_id}' AND active " + f"ORDER BY name LIMIT 1" + ).strip() + + +def export_part(node, table: str, part: str, dest: str) -> None: + node.query( + f"ALTER TABLE {table} EXPORT PART '{part}' TO TABLE {dest} " + f"SETTINGS allow_experimental_export_merge_tree_part = 1, " + f"allow_experimental_insert_into_iceberg = 1" + ) + + +def wait_for_export_part( + node, + table: str, + part: str, + timeout: int = 60, + poll_interval: float = 0.5, +) -> None: + """Poll system.part_log until an ExportPart event appears for *part*.""" + deadline = time.time() + timeout + while time.time() < deadline: + node.query("SYSTEM FLUSH LOGS") + count = node.query( + f"SELECT count() FROM system.part_log " + f"WHERE event_type = 'ExportPart' " + f"AND database = currentDatabase() " + f"AND table = '{table}' " + f"AND part_name = '{part}'" + ).strip() + if count != "0": + return + time.sleep(poll_interval) + raise TimeoutError( + f"ExportPart event for part {part!r} in table {table!r} " + f"did not appear in system.part_log within {timeout}s" + ) + + +def assert_part_log(node, table: str, part: str) -> None: + """Assert that system.part_log contains at least one ExportPart entry.""" + log_count = int( + node.query( + f"SELECT count() FROM system.part_log " + f"WHERE event_type = 'ExportPart' " + f"AND database = currentDatabase() " + f"AND table = '{table}' " + f"AND part_name = '{part}'" + ).strip() + ) + assert log_count >= 1, ( + f"Expected at least one ExportPart entry in system.part_log " + f"for part {part!r} in table {table!r}, found {log_count}" + ) + + +# --------------------------------------------------------------------------- +# Tests +# --------------------------------------------------------------------------- + + +def test_export_part_basic_to_iceberg(cluster): + """ + Basic happy path: export a single MergeTree part to an IcebergS3 table and + verify the row count, the content, and the system.part_log ExportPart entry. + """ + node = cluster.instances["node1"] + sfx = unique_suffix() + mt = f"mt_basic_{sfx}" + iceberg = f"iceberg_basic_{sfx}" + + make_mt(node, mt, "id Int32, year Int32", "year") + make_iceberg_s3(node, iceberg, "id Int32, year Int32", "year") + + node.query(f"INSERT INTO {mt} VALUES (1, 2020), (2, 2020), (3, 2020), (4, 2021)") + + part_2020 = get_part(node, mt, "2020") + export_part(node, mt, part_2020, iceberg) + wait_for_export_part(node, mt, part_2020) + + count = int(node.query(f"SELECT count() FROM {iceberg}").strip()) + assert count == 3, f"Expected 3 rows in Iceberg table after export, got {count}" + + result = node.query(f"SELECT id, year FROM {iceberg} ORDER BY id").strip() + assert result == "1\t2020\n2\t2020\n3\t2020", f"Unexpected exported data:\n{result}" + + assert_part_log(node, mt, part_2020) + + node.query(f"DROP TABLE IF EXISTS {mt} SYNC") + node.query(f"DROP TABLE IF EXISTS {iceberg}") + + +def test_export_part_all_iceberg_types(cluster): + """ + Export a part whose schema covers every ClickHouse type that getIcebergType() + in Utils.cpp maps to an Iceberg primitive (see the switch statement): + + Iceberg type ClickHouse column + ------------- -------------------------- + int id Int32 + long big_val Int64 + float f32 Float32 + double f64 Float64 + date event_dt Date + timestamp ts DateTime64(6) (DateTime / DateTime64 both → "timestamp") + string name String + uuid uid_val UUID + + Types not in the switch (Bool/UInt8, FixedString, Decimal) are intentionally + excluded — they throw BAD_ARGUMENTS from getIcebergType(). + + Verifies that every column round-trips correctly through the Iceberg layer + and that system.part_log records the ExportPart event. + """ + node = cluster.instances["node1"] + sfx = unique_suffix() + mt = f"mt_types_{sfx}" + iceberg = f"iceberg_types_{sfx}" + + columns = ( + "id Int32, " + "big_val Int64, " + "f32 Float32, " + "f64 Float64, " + "event_dt Date, " + "ts DateTime64(6), " + "name String, " + "uid_val UUID, " + "year Int32" + ) + + make_mt(node, mt, columns, "year", order_by="id") + make_iceberg_s3(node, iceberg, columns, "year") + + node.query( + f""" + INSERT INTO {mt} (id, big_val, f32, f64, event_dt, ts, name, uid_val, year) + VALUES ( + 1, + 9999999999999, + 3.14, + 2.718281828459045, + '2024-01-15', + '2024-01-15 12:30:45.123456', + 'hello iceberg', + '550e8400-e29b-41d4-a716-446655440000', + 2024 + ) + """ + ) + + part = get_part(node, mt, "2024") + export_part(node, mt, part, iceberg) + wait_for_export_part(node, mt, part) + + count = int(node.query(f"SELECT count() FROM {iceberg}").strip()) + assert count == 1, f"Expected 1 row in Iceberg table, got {count}" + + row = node.query( + f"SELECT id, big_val, name, year FROM {iceberg}" + ).strip() + assert "1" in row, f"id column missing/wrong: {row}" + assert "9999999999999" in row, f"big_val column missing/wrong: {row}" + assert "hello iceberg" in row, f"name column missing/wrong: {row}" + assert "2024" in row, f"year column missing/wrong: {row}" + + # Verify date round-trip + date_result = node.query(f"SELECT toString(event_dt) FROM {iceberg}").strip() + assert date_result == "2024-01-15", f"Date round-trip failed: {date_result!r}" + + # Verify timestamp round-trip (date component is sufficient; exact time format varies) + ts_result = node.query(f"SELECT ts FROM {iceberg}").strip() + assert "2024-01-15" in ts_result, f"Timestamp date component missing: {ts_result!r}" + + # Verify UUID round-trip + uid_result = node.query(f"SELECT toString(uid_val) FROM {iceberg}").strip() + assert uid_result == "550e8400-e29b-41d4-a716-446655440000", ( + f"UUID round-trip failed: {uid_result!r}" + ) + + assert_part_log(node, mt, part) + + node.query(f"DROP TABLE IF EXISTS {mt} SYNC") + node.query(f"DROP TABLE IF EXISTS {iceberg}") + + +def test_export_multiple_parts_to_iceberg(cluster): + """ + Export parts from two different partitions to the same Iceberg table and + verify that both land correctly without overwriting each other. + system.part_log must contain one ExportPart entry per exported part. + """ + node = cluster.instances["node1"] + sfx = unique_suffix() + mt = f"mt_multi_{sfx}" + iceberg = f"iceberg_multi_{sfx}" + + make_mt(node, mt, "id Int32, year Int32", "year") + make_iceberg_s3(node, iceberg, "id Int32, year Int32", "year") + + # Each INSERT creates a separate part per partition + node.query(f"INSERT INTO {mt} VALUES (1, 2020), (2, 2020)") + node.query(f"INSERT INTO {mt} VALUES (10, 2021), (11, 2021), (12, 2021)") + + part_2020 = get_part(node, mt, "2020") + part_2021 = get_part(node, mt, "2021") + + export_part(node, mt, part_2020, iceberg) + export_part(node, mt, part_2021, iceberg) + + wait_for_export_part(node, mt, part_2020) + wait_for_export_part(node, mt, part_2021) + + total = int(node.query(f"SELECT count() FROM {iceberg}").strip()) + assert total == 5, f"Expected 5 rows total (2+3), got {total}" + + count_2020 = int(node.query(f"SELECT count() FROM {iceberg} WHERE year = 2020").strip()) + count_2021 = int(node.query(f"SELECT count() FROM {iceberg} WHERE year = 2021").strip()) + assert count_2020 == 2, f"Expected 2 rows for year=2020, got {count_2020}" + assert count_2021 == 3, f"Expected 3 rows for year=2021, got {count_2021}" + + result_2020 = node.query( + f"SELECT id FROM {iceberg} WHERE year = 2020 ORDER BY id" + ).strip() + assert result_2020 == "1\n2", f"Unexpected 2020 rows: {result_2020}" + + result_2021 = node.query( + f"SELECT id FROM {iceberg} WHERE year = 2021 ORDER BY id" + ).strip() + assert result_2021 == "10\n11\n12", f"Unexpected 2021 rows: {result_2021}" + + assert_part_log(node, mt, part_2020) + assert_part_log(node, mt, part_2021) + + node.query(f"DROP TABLE IF EXISTS {mt} SYNC") + node.query(f"DROP TABLE IF EXISTS {iceberg}") + + +def test_export_part_with_year_transform_partition(cluster): + """ + Export a part from a MergeTree table partitioned by toYearNumSinceEpoch(event_date) + to an Iceberg table with the matching year-transform spec. + + Verifies that the Iceberg year-transform partition expression is accepted + and that all rows survive the round-trip intact. + """ + node = cluster.instances["node1"] + sfx = unique_suffix() + mt = f"mt_year_tf_{sfx}" + iceberg = f"iceberg_year_tf_{sfx}" + + cols = "id Int64, event_date Date" + partition_by = "toYearNumSinceEpoch(event_date)" + + make_mt(node, mt, cols, partition_by, order_by="id") + make_iceberg_s3(node, iceberg, cols, partition_by) + + node.query( + f"INSERT INTO {mt} VALUES " + f"(1, '2023-03-15'), (2, '2023-11-01'), (3, '2023-06-30')" + ) + + pid = first_partition_id(node, mt) + part = get_part(node, mt, pid) + + export_part(node, mt, part, iceberg) + wait_for_export_part(node, mt, part) + + count = int(node.query(f"SELECT count() FROM {iceberg}").strip()) + assert count == 3, f"Expected 3 rows, got {count}" + + result = node.query( + f"SELECT id, toString(event_date) FROM {iceberg} ORDER BY id" + ).strip() + assert "1\t2023-03-15" in result, f"Row 1 missing or incorrect:\n{result}" + assert "2\t2023-11-01" in result, f"Row 2 missing or incorrect:\n{result}" + assert "3\t2023-06-30" in result, f"Row 3 missing or incorrect:\n{result}" + + assert_part_log(node, mt, part) + + node.query(f"DROP TABLE IF EXISTS {mt} SYNC") + node.query(f"DROP TABLE IF EXISTS {iceberg}") + + +def test_export_part_partition_key_mismatch_is_rejected(cluster): + """ + EXPORT PART must synchronously reject (BAD_ARGUMENTS) when the source + MergeTree partition key does not match the destination Iceberg partition + spec. Export does not repartition data, so the two specs must agree on + every field (same source column by Iceberg field-id and same transform, + in the same order). + + Failing case: MergeTree PARTITION BY year, Iceberg PARTITION BY id. + The part must NOT land in the Iceberg table. + """ + node = cluster.instances["node1"] + sfx = unique_suffix() + mt = f"mt_pkey_mismatch_{sfx}" + iceberg = f"iceberg_pkey_mismatch_{sfx}" + + make_mt(node, mt, "id Int32, year Int32", "year") + make_iceberg_s3(node, iceberg, "id Int32, year Int32", "id") + + node.query(f"INSERT INTO {mt} VALUES (1, 2020), (2, 2020), (3, 2020)") + + part_2020 = get_part(node, mt, "2020") + + error = node.query_and_get_error( + f"ALTER TABLE {mt} EXPORT PART '{part_2020}' TO TABLE {iceberg} " + f"SETTINGS allow_experimental_export_merge_tree_part = 1, " + f"allow_experimental_insert_into_iceberg = 1" + ) + assert "BAD_ARGUMENTS" in error, ( + f"Expected BAD_ARGUMENTS for partition key mismatch, got: {error!r}" + ) + + count = int(node.query(f"SELECT count() FROM {iceberg}").strip()) + assert count == 0, ( + f"Expected 0 rows in Iceberg table after rejected export, got {count}" + ) + + node.query(f"DROP TABLE IF EXISTS {mt} SYNC") + node.query(f"DROP TABLE IF EXISTS {iceberg}") + + +def test_export_part_with_bucket_partition(cluster): + """ + Export a part from a MergeTree table partitioned by icebergBucket(8, user_id) + to a matching Iceberg table. + + Verifies that the bucket partition expression is accepted for EXPORT PART and + that data lands correctly in the Iceberg bucket partition. + """ + node = cluster.instances["node1"] + sfx = unique_suffix() + mt = f"mt_bucket_{sfx}" + iceberg = f"iceberg_bucket_{sfx}" + + cols = "id Int64, user_id Int64, value String" + partition_by = "icebergBucket(8, user_id)" + + make_mt(node, mt, cols, partition_by) + make_iceberg_s3(node, iceberg, cols, partition_by) + + # Both rows go to the same bucket (user_id=42 → bucket 2 for N=8) + node.query(f"INSERT INTO {mt} VALUES (1, 42, 'hello'), (2, 42, 'world')") + + pid = first_partition_id(node, mt) + part = get_part(node, mt, pid) + + export_part(node, mt, part, iceberg) + wait_for_export_part(node, mt, part) + + count = int(node.query(f"SELECT count() FROM {iceberg}").strip()) + assert count == 2, f"Expected 2 rows in Iceberg table, got {count}" + + result = node.query( + f"SELECT id, user_id, value FROM {iceberg} ORDER BY id" + ).strip() + assert "1\t42\thello" in result, f"Row 1 missing or incorrect:\n{result}" + assert "2\t42\tworld" in result, f"Row 2 missing or incorrect:\n{result}" + + assert_part_log(node, mt, part) + + node.query(f"DROP TABLE IF EXISTS {mt} SYNC") + node.query(f"DROP TABLE IF EXISTS {iceberg}") + + +def test_export_part_writes_column_statistics(cluster): + """ + Export a MergeTree part that contains one NULL and verify that the resulting + Iceberg manifest entry carries accurate per-file column statistics: + record_count, file_size_in_bytes, column_sizes, null_value_counts, + and lower/upper bounds (Int32 + String + Nullable(String) mix). + """ + node = cluster.instances["node1"] + sfx = unique_suffix() + mt = f"mt_stats_{sfx}" + iceberg = f"iceberg_stats_{sfx}" + + columns = "id Int32, name String, tag Nullable(String), year Int32" + + make_mt(node, mt, columns, "year", order_by="id") + make_iceberg_s3(node, iceberg, columns, "year") + + node.query( + f""" + INSERT INTO {mt} (id, name, tag, year) VALUES + (1, 'aaa', 'x', 2020), + (2, 'mmm', NULL, 2020), + (3, 'zzz', 'y', 2020), + (4, 'kkk', 'z', 2021) + """ + ) + + part_2020 = get_part(node, mt, "2020") + export_part(node, mt, part_2020, iceberg) + wait_for_export_part(node, mt, part_2020) + + count = int(node.query(f"SELECT count() FROM {iceberg}").strip()) + assert count == 3, f"Expected 3 rows after export, got {count}" + + query_id = f"stats_part_{sfx}" + node.query( + f"SELECT * FROM {iceberg} ORDER BY id", + query_id=query_id, + settings={"iceberg_metadata_log_level": "manifest_file_entry"}, + ) + + entries = fetch_manifest_entries(node, query_id) + assert_exported_stats(entries) + + node.query(f"DROP TABLE IF EXISTS {mt} SYNC") + node.query(f"DROP TABLE IF EXISTS {iceberg}") diff --git a/tests/integration/test_export_merge_tree_part_to_object_storage/__init__.py b/tests/integration/test_export_merge_tree_part_to_object_storage/__init__.py new file mode 100644 index 000000000000..e69de29bb2d1 diff --git a/tests/integration/test_export_merge_tree_part_to_object_storage/configs/named_collections.xml b/tests/integration/test_export_merge_tree_part_to_object_storage/configs/named_collections.xml new file mode 100644 index 000000000000..d46920b7ba88 --- /dev/null +++ b/tests/integration/test_export_merge_tree_part_to_object_storage/configs/named_collections.xml @@ -0,0 +1,9 @@ + + + + http://minio1:9001/root/data + minio + ClickHouse_Minio_P@ssw0rd + + + \ No newline at end of file diff --git a/tests/integration/test_export_merge_tree_part_to_object_storage/test.py b/tests/integration/test_export_merge_tree_part_to_object_storage/test.py new file mode 100644 index 000000000000..b8c15c26275f --- /dev/null +++ b/tests/integration/test_export_merge_tree_part_to_object_storage/test.py @@ -0,0 +1,314 @@ +import logging +import time +import uuid + +import pytest + +from helpers.cluster import ClickHouseCluster +from helpers.network import PartitionManager + + +def skip_if_remote_database_disk_enabled(cluster): + """Skip test if any instance in the cluster has remote database disk enabled. + + Tests that block MinIO cannot run when remote database disk is enabled, + as the database metadata is stored on MinIO and blocking it would break the database. + """ + for instance in cluster.instances.values(): + if instance.with_remote_database_disk: + pytest.skip("Test cannot run with remote database disk enabled (db disk), as it blocks MinIO which stores database metadata") + + +@pytest.fixture(scope="module") +def cluster(): + try: + cluster = ClickHouseCluster(__file__) + cluster.add_instance( + "node1", + main_configs=["configs/named_collections.xml"], + with_minio=True, + ) + logging.info("Starting cluster...") + cluster.start() + yield cluster + finally: + cluster.shutdown() + + +def create_s3_table(node, s3_table): + node.query(f"CREATE TABLE {s3_table} (id UInt64, year UInt16) ENGINE = S3(s3_conn, filename='{s3_table}', format=Parquet, partition_strategy='hive') PARTITION BY year") + + +def create_tables_and_insert_data(node, mt_table, s3_table): + # enable_block_number_column and enable_block_offset_column are needed for patch parts support + node.query(f"CREATE TABLE {mt_table} (id UInt64, year UInt16) ENGINE = MergeTree() PARTITION BY year ORDER BY tuple() SETTINGS enable_block_number_column = 1, enable_block_offset_column = 1") + node.query(f"INSERT INTO {mt_table} VALUES (1, 2020), (2, 2020), (3, 2020), (4, 2021)") + + create_s3_table(node, s3_table) + + +def test_drop_column_during_export_snapshot(cluster): + skip_if_remote_database_disk_enabled(cluster) + node = cluster.instances["node1"] + + postfix = str(uuid.uuid4()).replace("-", "_") + + mt_table = f"mutations_snapshot_mt_table_{postfix}" + s3_table = f"mutations_snapshot_s3_table_{postfix}" + + create_tables_and_insert_data(node, mt_table, s3_table) + + # Block traffic to/from MinIO to force upload errors and retries, following existing S3 tests style + minio_ip = cluster.minio_ip + minio_port = cluster.minio_port + + # Ensure export sees a consistent snapshot at start time even if we mutate the source later + with PartitionManager() as pm: + # Block responses from MinIO (source_port matches MinIO service) + pm_rule_reject_responses = { + "instance": node, + "destination": node.ip_address, + "protocol": "tcp", + "source_port": minio_port, + "action": "REJECT --reject-with tcp-reset", + } + pm.add_rule(pm_rule_reject_responses) + + # Block requests to MinIO (destination: MinIO, destination_port: minio_port) + pm_rule_reject_requests = { + "instance": node, + "destination": minio_ip, + "protocol": "tcp", + "destination_port": minio_port, + "action": "REJECT --reject-with tcp-reset", + } + pm.add_rule(pm_rule_reject_requests) + + # Start export of 2020 + node.query( + f"ALTER TABLE {mt_table} EXPORT PART '2020_1_1_0' TO TABLE {s3_table};" + ) + + # Drop a column that is required for the export + node.query(f"ALTER TABLE {mt_table} DROP COLUMN id") + + time.sleep(3) + # assert the mutation has been applied AND the data has not been exported yet + assert "Unknown expression identifier `id`" in node.query_and_get_error(f"SELECT id FROM {mt_table}"), "Column id is not removed" + + # Wait for export to finish and then verify destination still reflects the original snapshot (3 rows) + time.sleep(5) + assert node.query(f"SELECT count() FROM {s3_table} WHERE id >= 0") == '3\n', "Export did not preserve snapshot at start time after source mutation" + + +def test_add_column_during_export(cluster): + skip_if_remote_database_disk_enabled(cluster) + node = cluster.instances["node1"] + + postfix = str(uuid.uuid4()).replace("-", "_") + + mt_table = f"add_column_during_export_mt_table_{postfix}" + s3_table = f"add_column_during_export_s3_table_{postfix}" + + create_tables_and_insert_data(node, mt_table, s3_table) + + # Block traffic to/from MinIO to force upload errors and retries, following existing S3 tests style + minio_ip = cluster.minio_ip + minio_port = cluster.minio_port + + # Ensure export sees a consistent snapshot at start time even if we mutate the source later + with PartitionManager() as pm: + # Block responses from MinIO (source_port matches MinIO service) + pm_rule_reject_responses = { + "instance": node, + "destination": node.ip_address, + "protocol": "tcp", + "source_port": minio_port, + "action": "REJECT --reject-with tcp-reset", + } + pm.add_rule(pm_rule_reject_responses) + + # Block requests to MinIO (destination: MinIO, destination_port: minio_port) + pm_rule_reject_requests = { + "instance": node, + "destination": minio_ip, + "protocol": "tcp", + "destination_port": minio_port, + "action": "REJECT --reject-with tcp-reset", + } + pm.add_rule(pm_rule_reject_requests) + + # Start export of 2020 + node.query( + f"ALTER TABLE {mt_table} EXPORT PART '2020_1_1_0' TO TABLE {s3_table};" + ) + + node.query(f"ALTER TABLE {mt_table} ADD COLUMN id2 UInt64") + + time.sleep(3) + + # assert the mutation has been applied AND the data has not been exported yet + assert node.query(f"SELECT count(id2) FROM {mt_table}") == '4\n', "Column id2 is not added" + + # Wait for export to finish and then verify destination still reflects the original snapshot (3 rows) + time.sleep(5) + assert node.query(f"SELECT count() FROM {s3_table} WHERE id >= 0") == '3\n', "Export did not preserve snapshot at start time after source mutation" + assert "Unknown expression identifier `id2`" in node.query_and_get_error(f"SELECT id2 FROM {s3_table}"), "Column id2 is present in the exported data" + + +def test_pending_mutations_throw_before_export(cluster): + """Test that pending mutations before export throw an error with default settings.""" + node = cluster.instances["node1"] + + postfix = str(uuid.uuid4()).replace("-", "_") + + mt_table = f"pending_mutations_throw_mt_table_{postfix}" + s3_table = f"pending_mutations_throw_s3_table_{postfix}" + + create_tables_and_insert_data(node, mt_table, s3_table) + + node.query(f"SYSTEM STOP MERGES {mt_table}") + + node.query(f"ALTER TABLE {mt_table} UPDATE id = id + 100 WHERE year = 2020") + + mutations = node.query(f"SELECT count() FROM system.mutations WHERE table = '{mt_table}' AND is_done = 0") + assert mutations.strip() != '0', "Mutation should be pending" + + error = node.query_and_get_error( + f"ALTER TABLE {mt_table} EXPORT PART '2020_1_1_0' TO TABLE {s3_table} SETTINGS export_merge_tree_part_throw_on_pending_mutations=true" + ) + + assert "PENDING_MUTATIONS_NOT_ALLOWED" in error, f"Expected error about pending mutations, got: {error}" + + +def test_pending_mutations_skip_before_export(cluster): + """Test that pending mutations before export are skipped with throw_on_pending_mutations=false.""" + node = cluster.instances["node1"] + + postfix = str(uuid.uuid4()).replace("-", "_") + + mt_table = f"pending_mutations_skip_mt_table_{postfix}" + s3_table = f"pending_mutations_skip_s3_table_{postfix}" + + create_tables_and_insert_data(node, mt_table, s3_table) + + node.query(f"SYSTEM STOP MERGES {mt_table}") + + node.query(f"ALTER TABLE {mt_table} UPDATE id = id + 100 WHERE year = 2020") + + mutations = node.query(f"SELECT count() FROM system.mutations WHERE table = '{mt_table}' AND is_done = 0") + assert mutations.strip() != '0', "Mutation should be pending" + + node.query( + f"ALTER TABLE {mt_table} EXPORT PART '2020_1_1_0' TO TABLE {s3_table} " + f"SETTINGS export_merge_tree_part_throw_on_pending_mutations=false" + ) + + time.sleep(5) + + result = node.query(f"SELECT id FROM {s3_table} WHERE year = 2020 ORDER BY id") + assert "101" not in result and "102" not in result and "103" not in result, \ + "Export should contain original data before mutation" + assert "1\n2\n3" in result, "Export should contain original data" + + +def test_data_mutations_after_export_started(cluster): + """Test that mutations applied after export starts don't affect the exported data.""" + skip_if_remote_database_disk_enabled(cluster) + node = cluster.instances["node1"] + + postfix = str(uuid.uuid4()).replace("-", "_") + + mt_table = f"mutations_after_export_mt_table_{postfix}" + s3_table = f"mutations_after_export_s3_table_{postfix}" + + create_tables_and_insert_data(node, mt_table, s3_table) + + # Block traffic to MinIO to delay export + minio_ip = cluster.minio_ip + minio_port = cluster.minio_port + + with PartitionManager() as pm: + pm_rule_reject_responses = { + "instance": node, + "destination": node.ip_address, + "protocol": "tcp", + "source_port": minio_port, + "action": "REJECT --reject-with tcp-reset", + } + pm.add_rule(pm_rule_reject_responses) + + pm_rule_reject_requests = { + "instance": node, + "destination": minio_ip, + "protocol": "tcp", + "destination_port": minio_port, + "action": "REJECT --reject-with tcp-reset", + } + pm.add_rule(pm_rule_reject_requests) + + node.query( + f"ALTER TABLE {mt_table} EXPORT PART '2020_1_1_0' TO TABLE {s3_table} " + f"SETTINGS export_merge_tree_part_throw_on_pending_mutations=true" + ) + + node.query(f"ALTER TABLE {mt_table} UPDATE id = id + 100 WHERE year = 2020") + + time.sleep(5) + + result = node.query(f"SELECT id FROM {s3_table} WHERE year = 2020 ORDER BY id") + assert "1\n2\n3" in result, "Export should contain original data before mutation" + assert "101" not in result, "Export should not contain mutated data" + + +def test_pending_patch_parts_throw_before_export(cluster): + """Test that pending patch parts before export throw an error with default settings.""" + node = cluster.instances["node1"] + + postfix = str(uuid.uuid4()).replace("-", "_") + + mt_table = f"pending_patches_throw_mt_table_{postfix}" + s3_table = f"pending_patches_throw_s3_table_{postfix}" + + create_tables_and_insert_data(node, mt_table, s3_table) + + node.query(f"SYSTEM STOP MERGES {mt_table}") + + node.query(f"UPDATE {mt_table} SET id = id + 100 WHERE year = 2020") + + error = node.query_and_get_error( + f"ALTER TABLE {mt_table} EXPORT PART '2020_1_1_0' TO TABLE {s3_table}" + ) + + node.query(f"DROP TABLE {mt_table}") + + assert "PENDING_MUTATIONS_NOT_ALLOWED" in error or "pending patch parts" in error.lower(), \ + f"Expected error about pending patch parts, got: {error}" + + +def test_pending_patch_parts_skip_before_export(cluster): + """Test that pending patch parts before export are skipped with throw_on_pending_patch_parts=false.""" + node = cluster.instances["node1"] + + postfix = str(uuid.uuid4()).replace("-", "_") + + mt_table = f"pending_patches_skip_mt_table_{postfix}" + s3_table = f"pending_patches_skip_s3_table_{postfix}" + + create_tables_and_insert_data(node, mt_table, s3_table) + + node.query(f"SYSTEM STOP MERGES {mt_table}") + + node.query(f"UPDATE {mt_table} SET id = id + 100 WHERE year = 2020") + + node.query( + f"ALTER TABLE {mt_table} EXPORT PART '2020_1_1_0' TO TABLE {s3_table} " + f"SETTINGS export_merge_tree_part_throw_on_pending_patch_parts=false" + ) + + time.sleep(5) + + result = node.query(f"SELECT id FROM {s3_table} WHERE year = 2020 ORDER BY id") + assert "1\n2\n3" in result, "Export should contain original data before patch" + + node.query(f"DROP TABLE {mt_table}") diff --git a/tests/integration/test_export_replicated_mt_partition_to_iceberg/__init__.py b/tests/integration/test_export_replicated_mt_partition_to_iceberg/__init__.py new file mode 100644 index 000000000000..e69de29bb2d1 diff --git a/tests/integration/test_export_replicated_mt_partition_to_iceberg/configs/allow_experimental_export_partition.xml b/tests/integration/test_export_replicated_mt_partition_to_iceberg/configs/allow_experimental_export_partition.xml new file mode 100644 index 000000000000..514cd710836a --- /dev/null +++ b/tests/integration/test_export_replicated_mt_partition_to_iceberg/configs/allow_experimental_export_partition.xml @@ -0,0 +1,3 @@ + + 1 + diff --git a/tests/integration/test_export_replicated_mt_partition_to_iceberg/configs/config.d/metadata_log.xml b/tests/integration/test_export_replicated_mt_partition_to_iceberg/configs/config.d/metadata_log.xml new file mode 100644 index 000000000000..c1fece21745c --- /dev/null +++ b/tests/integration/test_export_replicated_mt_partition_to_iceberg/configs/config.d/metadata_log.xml @@ -0,0 +1,7 @@ + + + system + iceberg_metadata_log
+ 10 +
+
diff --git a/tests/integration/test_export_replicated_mt_partition_to_iceberg/configs/users.d/profile.xml b/tests/integration/test_export_replicated_mt_partition_to_iceberg/configs/users.d/profile.xml new file mode 100644 index 000000000000..1b50dfbdd310 --- /dev/null +++ b/tests/integration/test_export_replicated_mt_partition_to_iceberg/configs/users.d/profile.xml @@ -0,0 +1,9 @@ + + + + 3 + 1 + + + + diff --git a/tests/integration/test_export_replicated_mt_partition_to_iceberg/test.py b/tests/integration/test_export_replicated_mt_partition_to_iceberg/test.py new file mode 100644 index 000000000000..af6ca75bafd7 --- /dev/null +++ b/tests/integration/test_export_replicated_mt_partition_to_iceberg/test.py @@ -0,0 +1,934 @@ +import io +import json +import logging +import re +import time + +import pytest +from avro.datafile import DataFileReader +from avro.io import DatumReader + +from helpers.cluster import ClickHouseCluster +from helpers.export_partition_helpers import ( + first_partition_id, + make_iceberg_s3, + make_rmt, + unique_suffix, + wait_for_export_status, + wait_for_export_to_start, +) +from helpers.iceberg_export_stats import ( + assert_exported_stats, + fetch_manifest_entries, +) +from helpers.network import PartitionManager + + +@pytest.fixture(scope="module") +def cluster(): + try: + cluster = ClickHouseCluster(__file__) + cluster.add_instance( + "replica1", + main_configs=[ + "configs/allow_experimental_export_partition.xml", + "configs/config.d/metadata_log.xml", + ], + user_configs=["configs/users.d/profile.xml"], + with_minio=True, + stay_alive=True, + with_zookeeper=True, + keeper_required_feature_flags=["multi_read"], + ) + cluster.add_instance( + "replica2", + main_configs=[ + "configs/allow_experimental_export_partition.xml", + "configs/config.d/metadata_log.xml", + ], + user_configs=["configs/users.d/profile.xml"], + with_minio=True, + stay_alive=True, + with_zookeeper=True, + keeper_required_feature_flags=["multi_read"], + ) + logging.info("Starting cluster...") + cluster.start() + yield cluster + finally: + cluster.shutdown() + + +@pytest.fixture(autouse=True) +def drop_tables_after_test(cluster): + """Drop all tables in the default database after every test. + + Without this, ReplicatedMergeTree tables from completed tests remain alive and keep + running ZooKeeper background threads. With many tables alive simultaneously the + ZooKeeper session becomes overwhelmed and subsequent tests start seeing + operation-timeout / session-expired errors. + """ + yield + for instance_name, instance in cluster.instances.items(): + try: + tables_str = instance.query( + "SELECT name FROM system.tables WHERE database = 'default' FORMAT TabSeparated" + ).strip() + if not tables_str: + continue + for table in tables_str.split("\n"): + table = table.strip() + if table: + instance.query(f"DROP TABLE IF EXISTS default.`{table}` SYNC") + except Exception as e: + logging.warning( + f"drop_tables_after_test: cleanup failed on {instance_name}: {e}" + ) + + +# --------------------------------------------------------------------------- +# Helpers +# --------------------------------------------------------------------------- + +def create_replicated_mt(node, mt_table: str, replica_name: str): + make_rmt(node, mt_table, "id Int64, year Int32", "year", + replica_name=replica_name) + + +def create_iceberg_s3_table(node, iceberg_table: str, if_not_exists: bool = False, + s3_retry_attempts: int = 3): + """Create (or attach to an existing) IcebergS3 table at a per-test MinIO prefix.""" + make_iceberg_s3( + node, iceberg_table, "id Int64, year Int32", + partition_by="year", if_not_exists=if_not_exists, + s3_retry_attempts=s3_retry_attempts, + ) + + +def setup_tables(cluster, mt_table: str, iceberg_table: str, nodes: list | None = None, + s3_retry_attempts: int = 3): + """ + Create the ReplicatedMergeTree table on the given nodes, insert data on the first + node, wait for replication, then create the Iceberg destination table on each node. + + The Iceberg table is created on the first node (which initialises the S3 metadata). + Subsequent nodes attach to the same path with IF NOT EXISTS. + + `nodes` defaults to ["replica1", "replica2"]. + """ + if nodes is None: + nodes = ["replica1", "replica2"] + + instances = [cluster.instances[n] for n in nodes] + primary = instances[0] + + for i, instance in enumerate(instances): + create_replicated_mt(instance, mt_table, nodes[i]) + + primary.query(f"INSERT INTO {mt_table} VALUES (1, 2020), (2, 2020), (3, 2020), (4, 2021)") + for instance in instances[1:]: + instance.query(f"SYSTEM SYNC REPLICA {mt_table}") + + create_iceberg_s3_table(primary, iceberg_table, s3_retry_attempts=s3_retry_attempts) + for instance in instances[1:]: + create_iceberg_s3_table(instance, iceberg_table, if_not_exists=True, + s3_retry_attempts=s3_retry_attempts) + + +# --------------------------------------------------------------------------- +# Tests +# --------------------------------------------------------------------------- + +def test_export_partition_to_iceberg(cluster): + """ + Basic happy path: export a single partition and verify row count and content. + """ + node = cluster.instances["replica1"] + + uid = unique_suffix() + mt_table = f"mt_{uid}" + iceberg_table = f"iceberg_{uid}" + + setup_tables(cluster, mt_table, iceberg_table, nodes=["replica1"]) + + node.query( + f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {iceberg_table}" + ) + wait_for_export_status(node, mt_table, iceberg_table, "2020", "COMPLETED") + + count = int(node.query(f"SELECT count() FROM {iceberg_table}").strip()) + assert count == 3, f"Expected 3 rows in Iceberg table after export, got {count}" + + result = node.query(f"SELECT id, year FROM {iceberg_table} ORDER BY id").strip() + assert result == "1\t2020\n2\t2020\n3\t2020", ( + f"Unexpected data in Iceberg table:\n{result}" + ) + + +def test_export_two_partitions_to_iceberg(cluster): + """ + Export two partitions in a single ALTER TABLE statement and verify that both + land in the Iceberg table with correct row counts. + """ + node = cluster.instances["replica1"] + + uid = unique_suffix() + mt_table = f"mt_{uid}" + iceberg_table = f"iceberg_{uid}" + + setup_tables(cluster, mt_table, iceberg_table, nodes=["replica1"]) + + node.query( + f""" + ALTER TABLE {mt_table} + EXPORT PARTITION ID '2020' TO TABLE {iceberg_table}, + EXPORT PARTITION ID '2021' TO TABLE {iceberg_table} + """ + ) + + wait_for_export_status(node, mt_table, iceberg_table, "2020", "COMPLETED") + wait_for_export_status(node, mt_table, iceberg_table, "2021", "COMPLETED") + + count_2020 = int(node.query(f"SELECT count() FROM {iceberg_table} WHERE year = 2020").strip()) + count_2021 = int(node.query(f"SELECT count() FROM {iceberg_table} WHERE year = 2021").strip()) + + assert count_2020 == 3, f"Expected 3 rows for year=2020, got {count_2020}" + assert count_2021 == 1, f"Expected 1 row for year=2021, got {count_2021}" + + +def test_failure_is_logged_in_system_table(cluster): + """ + When S3 is unreachable the export must be marked FAILED in + system.replicated_partition_exports with a non-zero exception_count. + """ + node = cluster.instances["replica1"] + minio_ip = cluster.minio_ip + minio_port = cluster.minio_port + + uid = unique_suffix() + mt_table = f"mt_{uid}" + iceberg_table = f"iceberg_{uid}" + + setup_tables(cluster, mt_table, iceberg_table, nodes=["replica1"], + s3_retry_attempts=1) + + node.query(f"SYSTEM STOP MOVES {mt_table}") + + node.query(f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {iceberg_table} SETTINGS export_merge_tree_partition_max_retries = 1") + + with PartitionManager() as pm: + pm.add_rule({ + "instance": node, + "destination": node.ip_address, + "protocol": "tcp", + "source_port": minio_port, + "action": "REJECT --reject-with tcp-reset", + }) + pm.add_rule({ + "instance": node, + "destination": minio_ip, + "protocol": "tcp", + "destination_port": minio_port, + "action": "REJECT --reject-with tcp-reset", + }) + + node.query(f"SYSTEM START MOVES {mt_table}") + + wait_for_export_status(node, mt_table, iceberg_table, "2020", "FAILED", timeout=60) + + status = node.query( + f""" + SELECT status FROM system.replicated_partition_exports + WHERE source_table = '{mt_table}' + AND destination_table = '{iceberg_table}' + AND partition_id = '2020' + SETTINGS export_merge_tree_partition_system_table_prefer_remote_information = 1 + """ + ).strip() + assert status == "FAILED", f"Expected FAILED status, got: {status!r}" + + exception_count = int(node.query( + f""" + SELECT any(exception_count) FROM system.replicated_partition_exports + WHERE source_table = '{mt_table}' + AND destination_table = '{iceberg_table}' + AND partition_id = '2020' + SETTINGS export_merge_tree_partition_system_table_prefer_remote_information = 1 + """ + ).strip()) + assert exception_count > 0, "Expected non-zero exception_count in system.replicated_partition_exports" + + +def test_inject_short_living_failures(cluster): + """ + Transient S3 failures must not prevent the export from completing: after the + network is restored the export should retry and eventually land COMPLETED. + """ + node = cluster.instances["replica1"] + minio_ip = cluster.minio_ip + minio_port = cluster.minio_port + + uid = unique_suffix() + mt_table = f"mt_{uid}" + iceberg_table = f"iceberg_{uid}" + + setup_tables(cluster, mt_table, iceberg_table, nodes=["replica1"], + s3_retry_attempts=1) + + node.query(f"SYSTEM STOP MOVES {mt_table}") + + node.query(f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {iceberg_table} SETTINGS export_merge_tree_partition_max_retries = 100") + + with PartitionManager() as pm: + pm.add_rule({ + "instance": node, + "destination": node.ip_address, + "protocol": "tcp", + "source_port": minio_port, + "action": "REJECT --reject-with tcp-reset", + }) + pm.add_rule({ + "instance": node, + "destination": minio_ip, + "protocol": "tcp", + "destination_port": minio_port, + "action": "REJECT --reject-with tcp-reset", + }) + + node.query(f"SYSTEM START MOVES {mt_table}") + + # Let at least one retry happen before restoring the network. + time.sleep(15) + + wait_for_export_status(node, mt_table, iceberg_table, "2020", "COMPLETED") + + count = int(node.query(f"SELECT count() FROM {iceberg_table} WHERE year = 2020").strip()) + assert count == 3, f"Expected 3 rows after retry, got {count}" + + status = node.query( + f""" + SELECT status FROM system.replicated_partition_exports + WHERE source_table = '{mt_table}' + AND destination_table = '{iceberg_table}' + AND partition_id = '2020' + """ + ).strip() + assert status == "COMPLETED", f"Expected COMPLETED in system table, got: {status!r}" + + exception_count = int(node.query( + f""" + SELECT exception_count FROM system.replicated_partition_exports + WHERE source_table = '{mt_table}' + AND destination_table = '{iceberg_table}' + AND partition_id = '2020' + SETTINGS export_merge_tree_partition_system_table_prefer_remote_information = 1 + """ + ).strip()) + assert exception_count >= 1, "Expected at least one transient exception to be recorded" + + +def test_export_partition_scheduler_skipped_when_moves_stopped(cluster): + """ + Verify that selectPartsToExport() skips the scheduler entirely when moves + are stopped (moves_blocker guard at the top of the function). + + No ZK locks are acquired and no background tasks are submitted, so the + Iceberg table must remain empty across multiple scheduler cycles. Once moves + are re-enabled the export completes and rows appear in the Iceberg table. + """ + node = cluster.instances["replica1"] + + uid = unique_suffix() + mt_table = f"mt_{uid}" + iceberg_table = f"iceberg_{uid}" + + setup_tables(cluster, mt_table, iceberg_table, nodes=["replica1"]) + + node.query(f"SYSTEM STOP MOVES {mt_table}") + + node.query( + f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {iceberg_table}" + ) + + wait_for_export_to_start(node, mt_table, iceberg_table, "2020") + + # Wait for several scheduler cycles (each fires every 5 s). + # If the guard is absent the scheduler would run and rows would appear in the Iceberg table. + time.sleep(12) + + status = node.query( + f"SELECT status FROM system.replicated_partition_exports" + f" WHERE source_table = '{mt_table}' AND destination_table = '{iceberg_table}'" + f" AND partition_id = '2020'" + ).strip() + + assert status == "PENDING", f"Expected PENDING while moves are stopped, got '{status}'" + + count = int(node.query(f"SELECT count() FROM {iceberg_table} WHERE year = 2020").strip()) + assert count == 0, f"Expected 0 rows in Iceberg table while scheduler is skipped, got {count}" + + node.query(f"SYSTEM START MOVES {mt_table}") + + wait_for_export_status(node, mt_table, iceberg_table, "2020", "COMPLETED") + + count = int(node.query(f"SELECT count() FROM {iceberg_table} WHERE year = 2020").strip()) + assert count == 3, f"Expected 3 rows in Iceberg table after export completed, got {count}" + + +def test_export_partition_resumes_after_stop_moves(cluster): + """ + Verify that SYSTEM STOP MOVES before EXPORT PARTITION does not permanently + orphan the ZooKeeper part lock for Iceberg destinations. + + When moves are stopped the scheduler still picks parts up and submits them to + the background executor, but ExportPartTask::isCancelled() returns true (via + moves_blocker), causing QUERY_WAS_CANCELLED before any data is written. The + fix in handlePartExportFailure must release the ZK lock so the part is retried + once moves are restarted. + """ + node = cluster.instances["replica1"] + + uid = unique_suffix() + mt_table = f"mt_{uid}" + iceberg_table = f"iceberg_{uid}" + + setup_tables(cluster, mt_table, iceberg_table, nodes=["replica1"]) + + node.query(f"SYSTEM STOP MOVES {mt_table}") + + node.query( + f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {iceberg_table}" + f" SETTINGS export_merge_tree_partition_max_retries = 50" + ) + + wait_for_export_to_start(node, mt_table, iceberg_table, "2020") + + # Give the scheduler enough time to attempt (and cancel) the part task at least once. + time.sleep(5) + + status = node.query( + f"SELECT status FROM system.replicated_partition_exports" + f" WHERE source_table = '{mt_table}' AND destination_table = '{iceberg_table}'" + f" AND partition_id = '2020'" + ).strip() + assert status == "PENDING", f"Expected PENDING while moves are stopped, got '{status}'" + + count = int(node.query(f"SELECT count() FROM {iceberg_table} WHERE year = 2020").strip()) + assert count == 0, f"Expected 0 rows in Iceberg table while moves are stopped, got {count}" + + node.query(f"SYSTEM START MOVES {mt_table}") + + wait_for_export_status(node, mt_table, iceberg_table, "2020", "COMPLETED") + + count = int(node.query(f"SELECT count() FROM {iceberg_table} WHERE year = 2020").strip()) + assert count == 3, f"Expected 3 rows in Iceberg table after export completed, got {count}" + + +def test_export_partition_resumes_after_stop_moves_during_export(cluster): + """ + Verify that SYSTEM STOP MOVES issued while an Iceberg export is actively + retrying (S3 blocked) does not permanently orphan the ZooKeeper part lock. + """ + node = cluster.instances["replica1"] + minio_ip = cluster.minio_ip + minio_port = cluster.minio_port + + uid = unique_suffix() + mt_table = f"mt_{uid}" + iceberg_table = f"iceberg_{uid}" + + setup_tables(cluster, mt_table, iceberg_table, nodes=["replica1"]) + + node.query(f"SYSTEM STOP MOVES {mt_table}") + + node.query( + f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {iceberg_table}" + f" SETTINGS export_merge_tree_partition_max_retries = 50") + + wait_for_export_to_start(node, mt_table, iceberg_table, "2020") + + with PartitionManager() as pm: + pm.add_rule({ + "instance": node, + "destination": node.ip_address, + "protocol": "tcp", + "source_port": minio_port, + "action": "REJECT --reject-with tcp-reset", + }) + pm.add_rule({ + "instance": node, + "destination": minio_ip, + "protocol": "tcp", + "destination_port": minio_port, + "action": "REJECT --reject-with tcp-reset", + }) + + node.query(f"SYSTEM STOP MOVES {mt_table}") + + time.sleep(3) + + status = node.query( + f"SELECT status FROM system.replicated_partition_exports" + f" WHERE source_table = '{mt_table}' AND destination_table = '{iceberg_table}'" + f" AND partition_id = '2020'" + ).strip() + assert status == "PENDING", ( + f"Expected PENDING while moves are stopped and S3 is blocked, got '{status}'" + ) + + node.query(f"SYSTEM START MOVES {mt_table}") + + # MinIO is now unblocked; the next scheduler cycle should succeed. + wait_for_export_status(node, mt_table, iceberg_table, "2020", "COMPLETED") + + count = int(node.query(f"SELECT count() FROM {iceberg_table} WHERE year = 2020").strip()) + assert count == 3, f"Expected 3 rows in Iceberg table after export completed, got {count}" + + +def test_partition_transform_compatibility_accepted(cluster): + """ + Verify that EXPORT PARTITION is accepted (no BAD_ARGUMENTS) for every + supported transform when the MergeTree and Iceberg partition specs match. + + Cases covered: + 1. Compound identity (year, region) + 2. Year transform – toYearNumSinceEpoch(event_date) + 3. Month transform – toMonthNumSinceEpoch(event_date) + 4. truncate[4] – icebergTruncate(4, category) + 5. bucket[8] – icebergBucket(8, user_id) + 6. Compound mixed – (toYearNumSinceEpoch(event_date), icebergBucket(16, user_id)) + """ + node = cluster.instances["replica1"] + uid = unique_suffix() + + def check_accepted(mt, iceberg, description): + pid = first_partition_id(node, mt) + node.query( + f"ALTER TABLE {mt} EXPORT PARTITION ID '{pid}' TO TABLE {iceberg}" + ) + + # 1. Compound identity: (year, region) + cols = "id Int64, year Int32, region String" + t = f"mt_acc_1_{uid}"; i = f"iceberg_acc_1_{uid}" + make_rmt(node, t, cols, "(year, region)") + node.query(f"INSERT INTO {t} VALUES (1, 2023, 'EU')") + make_iceberg_s3(node, i, cols, "(year, region)") + check_accepted(t, i, "compound identity (year, region)") + + # 2. Year transform + cols = "id Int64, event_date Date" + t = f"mt_acc_2_{uid}"; i = f"iceberg_acc_2_{uid}" + make_rmt(node, t, cols, "toYearNumSinceEpoch(event_date)") + node.query(f"INSERT INTO {t} VALUES (1, '2020-06-15')") + make_iceberg_s3(node, i, cols, "toYearNumSinceEpoch(event_date)") + check_accepted(t, i, "year transform") + + # 3. Month transform + cols = "id Int64, event_date Date" + t = f"mt_acc_3_{uid}"; i = f"iceberg_acc_3_{uid}" + make_rmt(node, t, cols, "toMonthNumSinceEpoch(event_date)") + node.query(f"INSERT INTO {t} VALUES (1, '2020-06-15')") + make_iceberg_s3(node, i, cols, "toMonthNumSinceEpoch(event_date)") + check_accepted(t, i, "month transform") + + # 4. truncate[4] + cols = "id Int64, category String" + t = f"mt_acc_4_{uid}"; i = f"iceberg_acc_4_{uid}" + make_rmt(node, t, cols, "icebergTruncate(4, category)") + node.query(f"INSERT INTO {t} VALUES (1, 'clickhouse')") + make_iceberg_s3(node, i, cols, "icebergTruncate(4, category)") + check_accepted(t, i, "truncate[4]") + + # 5. bucket[8] + cols = "id Int64, user_id Int64" + t = f"mt_acc_5_{uid}"; i = f"iceberg_acc_5_{uid}" + make_rmt(node, t, cols, "icebergBucket(8, user_id)") + node.query(f"INSERT INTO {t} VALUES (1, 42)") + make_iceberg_s3(node, i, cols, "icebergBucket(8, user_id)") + check_accepted(t, i, "bucket[8]") + + # 6. Compound mixed: year(event_date) + bucket[16](user_id) + cols = "id Int64, event_date Date, user_id Int64" + t = f"mt_acc_6_{uid}"; i = f"iceberg_acc_6_{uid}" + make_rmt(node, t, cols, "(toYearNumSinceEpoch(event_date), icebergBucket(16, user_id))") + node.query(f"INSERT INTO {t} VALUES (1, '2021-03-01', 99)") + make_iceberg_s3(node, i, cols, "(toYearNumSinceEpoch(event_date), icebergBucket(16, user_id))") + check_accepted(t, i, "compound year+bucket[16]") + + +def test_partition_transform_compatibility_rejected(cluster): + """ + Verify that mismatched partition specs are rejected with BAD_ARGUMENTS. + + Cases covered: + 1. Compound field order reversed: MergeTree (year, region) vs Iceberg (region, year) + 2. Transform mismatch on same column: year-transform vs identity + 3. Bucket count mismatch: bucket[8] vs bucket[16] + 4. Truncate width mismatch: truncate[4] vs truncate[8] + 5. Field-count mismatch: 2-field MergeTree vs 1-field Iceberg + 6. Unsupported MergeTree expression (intDiv — not an Iceberg transform) + """ + node = cluster.instances["replica1"] + uid = unique_suffix() + + def assert_rejected(mt, iceberg, description): + # The compatibility check fires synchronously; any partition ID works here. + pid = first_partition_id(node, mt) + error = node.query_and_get_error( + f"ALTER TABLE {mt} EXPORT PARTITION ID '{pid}' TO TABLE {iceberg}" + ) + assert "BAD_ARGUMENTS" in error, ( + f"[{description}] Expected BAD_ARGUMENTS, got: {error!r}" + ) + + # 1. Compound field order reversed + cols = "id Int64, year Int32, region String" + t = f"mt_rej_1_{uid}"; i = f"iceberg_rej_1_{uid}" + make_rmt(node, t, cols, "(year, region)") + node.query(f"INSERT INTO {t} VALUES (1, 2020, 'EU')") + make_iceberg_s3(node, i, cols, "(region, year)") + assert_rejected(t, i, "compound field order reversed") + + # 2. Transform mismatch: MergeTree year-transform, Iceberg identity on same Date col + cols = "id Int64, event_date Date" + t = f"mt_rej_2_{uid}"; i = f"iceberg_rej_2_{uid}" + make_rmt(node, t, cols, "toYearNumSinceEpoch(event_date)") + node.query(f"INSERT INTO {t} VALUES (1, '2020-01-01')") + make_iceberg_s3(node, i, cols, "event_date") # identity, not year-transform + assert_rejected(t, i, "year-transform vs identity on same column") + + # 3. Bucket count mismatch: bucket[8] vs bucket[16] + cols = "id Int64, user_id Int64" + t = f"mt_rej_3_{uid}"; i = f"iceberg_rej_3_{uid}" + make_rmt(node, t, cols, "icebergBucket(8, user_id)") + node.query(f"INSERT INTO {t} VALUES (1, 42)") + make_iceberg_s3(node, i, cols, "icebergBucket(16, user_id)") + assert_rejected(t, i, "bucket[8] vs bucket[16]") + + # 4. Truncate width mismatch: truncate[4] vs truncate[8] + cols = "id Int64, category String" + t = f"mt_rej_4_{uid}"; i = f"iceberg_rej_4_{uid}" + make_rmt(node, t, cols, "icebergTruncate(4, category)") + node.query(f"INSERT INTO {t} VALUES (1, 'clickhouse')") + make_iceberg_s3(node, i, cols, "icebergTruncate(8, category)") + assert_rejected(t, i, "truncate[4] vs truncate[8]") + + # 5. Field-count mismatch: MergeTree has 2 fields, Iceberg has 1 + cols = "id Int64, year Int32, region String" + t = f"mt_rej_5_{uid}"; i = f"iceberg_rej_5_{uid}" + make_rmt(node, t, cols, "(year, region)") + node.query(f"INSERT INTO {t} VALUES (1, 2020, 'EU')") + make_iceberg_s3(node, i, cols, "year") + assert_rejected(t, i, "2-field MergeTree vs 1-field Iceberg") + + # 6. Unsupported MergeTree expression: intDiv(year, 100) is not an Iceberg transform + cols = "id Int64, year Int32" + t = f"mt_rej_6_{uid}"; i = f"iceberg_rej_6_{uid}" + make_rmt(node, t, cols, "intDiv(year, 100)") + node.query(f"INSERT INTO {t} VALUES (1, 2020)") + make_iceberg_s3(node, i, cols, "year") + assert_rejected(t, i, "unsupported MergeTree expression intDiv") + + +def test_partition_key_compatibility_check(cluster): + """ + Verify that EXPORT PARTITION throws BAD_ARGUMENTS synchronously when the + MergeTree partition key does not match the Iceberg table's partition spec, + and is accepted without error when the keys match. + + Three cases: + 1. Column mismatch – MergeTree PARTITION BY year, Iceberg PARTITION BY id + 2. Count mismatch – MergeTree PARTITION BY year, Iceberg unpartitioned + 3. Matching keys – both PARTITION BY year (must be accepted) + """ + node = cluster.instances["replica1"] + + uid = unique_suffix() + mt_table = f"mt_{uid}" + + create_replicated_mt(node, mt_table, "replica1") + node.query(f"INSERT INTO {mt_table} VALUES (1, 2020), (2, 2020), (3, 2021)") + node.query(f"SYSTEM SYNC REPLICA {mt_table}") + + # --- Case 1: Iceberg partitioned by 'id' but MergeTree by 'year' --- + iceberg_col_mismatch = f"iceberg_col_mismatch_{uid}" + node.query( + f""" + CREATE TABLE {iceberg_col_mismatch} + (id Int64, year Int32) + ENGINE = IcebergS3( + 'http://minio1:9001/root/data/{iceberg_col_mismatch}/', + 'minio', + 'ClickHouse_Minio_P@ssw0rd' + ) + PARTITION BY id SETTINGS s3_retry_attempts = 3 + """ + ) + error = node.query_and_get_error( + f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {iceberg_col_mismatch}" + ) + assert "BAD_ARGUMENTS" in error, ( + f"Expected BAD_ARGUMENTS for partition column mismatch, got: {error!r}" + ) + + # --- Case 2: Iceberg unpartitioned but MergeTree PARTITION BY year --- + iceberg_count_mismatch = f"iceberg_count_mismatch_{uid}" + node.query( + f""" + CREATE TABLE {iceberg_count_mismatch} + (id Int64, year Int32) + ENGINE = IcebergS3( + 'http://minio1:9001/root/data/{iceberg_count_mismatch}/', + 'minio', + 'ClickHouse_Minio_P@ssw0rd' + ) + SETTINGS s3_retry_attempts = 3 + """ + ) + error = node.query_and_get_error( + f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {iceberg_count_mismatch}" + ) + assert "BAD_ARGUMENTS" in error, ( + f"Expected BAD_ARGUMENTS for partition count mismatch, got: {error!r}" + ) + + # --- Case 3: Matching partition keys (both PARTITION BY year) --- + iceberg_match = f"iceberg_match_{uid}" + node.query( + f""" + CREATE TABLE {iceberg_match} + (id Int64, year Int32) + ENGINE = IcebergS3( + 'http://minio1:9001/root/data/{iceberg_match}/', + 'minio', + 'ClickHouse_Minio_P@ssw0rd' + ) + PARTITION BY year SETTINGS s3_retry_attempts = 3 + """ + ) + # Should not raise — the check passes so the export is accepted synchronously + node.query( + f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {iceberg_match}" + ) + + +def test_export_ttl(cluster): + """ + After a manifest TTL expires the same partition can be re-exported, and the + new data is appended to (or replaces) what is in the Iceberg table. + """ + node = cluster.instances["replica1"] + ttl_seconds = 3 + + uid = unique_suffix() + mt_table = f"mt_{uid}" + iceberg_table = f"iceberg_{uid}" + + setup_tables(cluster, mt_table, iceberg_table, nodes=["replica1"]) + + # First export. + node.query( + f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {iceberg_table} " + f"SETTINGS export_merge_tree_partition_manifest_ttl = {ttl_seconds}" + ) + + # A second export before the TTL expires must be rejected. + error = node.query_and_get_error( + f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {iceberg_table}" + ) + assert "Export with key" in error, f"Expected duplicate-export error before TTL, got: {error}" + + wait_for_export_status(node, mt_table, iceberg_table, "2020", "COMPLETED") + + count_after_first = int(node.query(f"SELECT count() FROM {iceberg_table} WHERE year = 2020").strip()) + assert count_after_first == 3, f"Expected 3 rows after first export, got {count_after_first}" + + # Wait for the manifest TTL to expire. + time.sleep(ttl_seconds * 2) + + # Second export must be accepted now. + node.query( + f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {iceberg_table}" + ) + wait_for_export_status(node, mt_table, iceberg_table, "2020", "COMPLETED") + + +def test_export_data_files_are_not_cleaned_up_on_commit_failure(cluster): + """ + Verify that the data files are not cleaned up on commit failure and the export is retried. + This is to avoid data loss. + + If the data files were cleaned up, a retry would commit a new snapshot that points to dangling references. + """ + node = cluster.instances["replica1"] + uid = unique_suffix() + mt_table = f"mt_{uid}" + iceberg_table = f"iceberg_{uid}" + setup_tables(cluster, mt_table, iceberg_table, nodes=["replica1"]) + + node.query("SYSTEM ENABLE FAILPOINT iceberg_writes_non_retry_cleanup") + + node.query(f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {iceberg_table}") + wait_for_export_status(node, mt_table, iceberg_table, "2020", "COMPLETED") + + count = int(node.query(f"SELECT count() FROM {iceberg_table} WHERE year = 2020").strip()) + assert count == 3, f"Expected 3 rows after first export, got {count}" + + +def test_post_publish_exception_preserves_snapshot(cluster): + """ + Regression test for the post-publish exception-safety bug in + commitImportPartitionTransactionImpl. + + Before the fix, any exception thrown after the Iceberg snapshot was published + (e.g. from metadata-cache invalidation) would fall through to the outer + `catch (...)` and invoke `cleanup(false)`, which unconditionally removed the + manifest entry and manifest list referenced by the just-published snapshot. + A subsequent read would then fail because the live snapshot points to deleted + files. + + The failpoint `iceberg_writes_post_publish_throw` is placed inside the + post-publish region (after both the metadata file is written and + `published = true` is set). With the fix in place: + - the commit stays durable (snapshot is readable, manifests are intact); + - the export is marked COMPLETED because the idempotency check on retry + detects that the transaction is already committed and returns success; + - all exported rows are visible through the Iceberg table. + """ + node = cluster.instances["replica1"] + uid = unique_suffix() + mt_table = f"mt_{uid}" + iceberg_table = f"iceberg_{uid}" + setup_tables(cluster, mt_table, iceberg_table, nodes=["replica1"]) + + node.query("SYSTEM ENABLE FAILPOINT iceberg_writes_post_publish_throw") + + node.query(f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {iceberg_table}") + wait_for_export_status(node, mt_table, iceberg_table, "2020", "COMPLETED") + + count = int(node.query(f"SELECT count() FROM {iceberg_table} WHERE year = 2020").strip()) + assert count == 3, ( + f"Snapshot must remain readable after a post-publish exception, " + f"expected 3 rows but got {count} (manifest files likely deleted by " + f"over-broad cleanup)" + ) + + result = node.query( + f"SELECT id, year FROM {iceberg_table} WHERE year = 2020 ORDER BY id" + ).strip() + assert result == "1\t2020\n2\t2020\n3\t2020", ( + f"Unexpected data after post-publish exception recovery:\n{result}" + ) + + +def test_export_task_timeout_kills_stuck_pending_task(cluster): + """ + Verify that export_merge_tree_partition_task_timeout_seconds auto-kills a task + that remains PENDING past the deadline, transitioning it to KILLED with a + descriptive last_exception. + + The export_partition_commit_always_throw failpoint wedges the task in the + commit retry loop (REGULAR failpoint, fires on every commit attempt). A very + large max_retries budget prevents the commit-attempts path from transitioning + to FAILED before the timeout fires, so the timeout branch in tryCleanup is + the actual mechanism under test. + """ + node = cluster.instances["replica1"] + uid = unique_suffix() + mt_table = f"mt_{uid}" + iceberg_table = f"iceberg_{uid}" + setup_tables(cluster, mt_table, iceberg_table, nodes=["replica1"]) + + node.query("SYSTEM ENABLE FAILPOINT export_partition_commit_always_throw") + + try: + node.query( + f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {iceberg_table}" + f" SETTINGS export_merge_tree_partition_task_timeout_seconds = 5," + f" export_merge_tree_partition_max_retries = 1000000," + f" export_merge_tree_partition_manifest_ttl = 3600" + ) + + # Timeout budget must cover: the 5s task timeout + one manifest-updating + # poll cycle (~30s) + watch propagation. 90s is safe. + wait_for_export_status( + node, mt_table, iceberg_table, "2020", + expected_status="KILLED", + timeout=90, + ) + + # TODO: system.replicated_partition_exports does not currently surface + # last_exception / exception_count reliably (the engine's aggregation + # from exceptions_per_replica is incomplete). Read the raw znode via + # system.zookeeper until that is fixed. + export_key = f"2020_default.{iceberg_table}" + last_exception_path = ( + f"/clickhouse/tables/{mt_table}/exports/{export_key}" + f"/exceptions_per_replica/replica1/last_exception" + ) + last_exception = node.query( + f""" + SELECT value FROM system.zookeeper + WHERE path = '{last_exception_path}' AND name = 'exception' + """ + ).strip() + assert "timed out" in last_exception, ( + f"Expected last_exception znode to mention the timeout reason, got: {last_exception!r}" + ) + finally: + node.query("SYSTEM DISABLE FAILPOINT export_partition_commit_always_throw") + + +def setup_stats_tables(node, mt_table: str, iceberg_table: str): + """Local variant of setup_tables using the wider schema with a Nullable column.""" + columns = "id Int32, name String, tag Nullable(String), year Int32" + + make_rmt( + node, mt_table, columns, "year", + order_by="id", replica_name="replica1", + ) + node.query( + f""" + INSERT INTO {mt_table} (id, name, tag, year) VALUES + (1, 'aaa', 'x', 2020), + (2, 'mmm', NULL, 2020), + (3, 'zzz', 'y', 2020), + (4, 'kkk', 'z', 2021) + """ + ) + + make_iceberg_s3(node, iceberg_table, columns, partition_by="year") + + +def test_export_partition_writes_column_statistics(cluster): + """ + Export a whole partition (EXPORT PARTITION ID '2020') that contains one NULL + and verify that the resulting Iceberg manifest entry carries accurate per-file + column statistics: record_count, file_size_in_bytes, column_sizes, + null_value_counts, and lower/upper bounds. + """ + node = cluster.instances["replica1"] + + uid = unique_suffix() + mt_table = f"mt_stats_{uid}" + iceberg_table = f"iceberg_stats_{uid}" + + setup_stats_tables(node, mt_table, iceberg_table) + + node.query( + f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {iceberg_table}" + ) + wait_for_export_status(node, mt_table, iceberg_table, "2020", "COMPLETED") + + count = int(node.query(f"SELECT count() FROM {iceberg_table}").strip()) + assert count == 3, f"Expected 3 rows in Iceberg table after export, got {count}" + + query_id = f"stats_partition_{uid}" + node.query( + f"SELECT * FROM {iceberg_table} ORDER BY id", + query_id=query_id, + settings={"iceberg_metadata_log_level": "manifest_file_entry"}, + ) + + entries = fetch_manifest_entries(node, query_id) + assert_exported_stats(entries) diff --git a/tests/integration/test_export_replicated_mt_partition_to_object_storage/__init__.py b/tests/integration/test_export_replicated_mt_partition_to_object_storage/__init__.py new file mode 100644 index 000000000000..e69de29bb2d1 diff --git a/tests/integration/test_export_replicated_mt_partition_to_object_storage/configs/allow_experimental_export_partition.xml b/tests/integration/test_export_replicated_mt_partition_to_object_storage/configs/allow_experimental_export_partition.xml new file mode 100644 index 000000000000..d931c6fb00db --- /dev/null +++ b/tests/integration/test_export_replicated_mt_partition_to_object_storage/configs/allow_experimental_export_partition.xml @@ -0,0 +1,3 @@ + + 1 + \ No newline at end of file diff --git a/tests/integration/test_export_replicated_mt_partition_to_object_storage/configs/disable_experimental_export_partition.xml b/tests/integration/test_export_replicated_mt_partition_to_object_storage/configs/disable_experimental_export_partition.xml new file mode 100644 index 000000000000..5379b8e892f0 --- /dev/null +++ b/tests/integration/test_export_replicated_mt_partition_to_object_storage/configs/disable_experimental_export_partition.xml @@ -0,0 +1,3 @@ + + 0 + \ No newline at end of file diff --git a/tests/integration/test_export_replicated_mt_partition_to_object_storage/configs/macros_shard1_replica1.xml b/tests/integration/test_export_replicated_mt_partition_to_object_storage/configs/macros_shard1_replica1.xml new file mode 100644 index 000000000000..bae1ce119255 --- /dev/null +++ b/tests/integration/test_export_replicated_mt_partition_to_object_storage/configs/macros_shard1_replica1.xml @@ -0,0 +1,6 @@ + + + shard1 + replica1 + + diff --git a/tests/integration/test_export_replicated_mt_partition_to_object_storage/configs/macros_shard2_replica1.xml b/tests/integration/test_export_replicated_mt_partition_to_object_storage/configs/macros_shard2_replica1.xml new file mode 100644 index 000000000000..fb9a587e736d --- /dev/null +++ b/tests/integration/test_export_replicated_mt_partition_to_object_storage/configs/macros_shard2_replica1.xml @@ -0,0 +1,6 @@ + + + shard2 + replica1 + + diff --git a/tests/integration/test_export_replicated_mt_partition_to_object_storage/configs/named_collections.xml b/tests/integration/test_export_replicated_mt_partition_to_object_storage/configs/named_collections.xml new file mode 100644 index 000000000000..d46920b7ba88 --- /dev/null +++ b/tests/integration/test_export_replicated_mt_partition_to_object_storage/configs/named_collections.xml @@ -0,0 +1,9 @@ + + + + http://minio1:9001/root/data + minio + ClickHouse_Minio_P@ssw0rd + + + \ No newline at end of file diff --git a/tests/integration/test_export_replicated_mt_partition_to_object_storage/configs/users.d/profile.xml b/tests/integration/test_export_replicated_mt_partition_to_object_storage/configs/users.d/profile.xml new file mode 100644 index 000000000000..518f29708929 --- /dev/null +++ b/tests/integration/test_export_replicated_mt_partition_to_object_storage/configs/users.d/profile.xml @@ -0,0 +1,8 @@ + + + + 3 + + + + diff --git a/tests/integration/test_export_replicated_mt_partition_to_object_storage/test.py b/tests/integration/test_export_replicated_mt_partition_to_object_storage/test.py new file mode 100644 index 000000000000..a5e60f81194e --- /dev/null +++ b/tests/integration/test_export_replicated_mt_partition_to_object_storage/test.py @@ -0,0 +1,1508 @@ +import logging +import time +import uuid + +import pytest + +from helpers.cluster import ClickHouseCluster +from helpers.export_partition_helpers import ( + wait_for_exception_count, + wait_for_export_status, + wait_for_export_to_start, +) +from helpers.network import PartitionManager + + + +def skip_if_remote_database_disk_enabled(cluster): + """Skip test if any instance in the cluster has remote database disk enabled. + + Tests that block MinIO cannot run when remote database disk is enabled, + as the database metadata is stored on MinIO and blocking it would break the database. + """ + for instance in cluster.instances.values(): + if instance.with_remote_database_disk: + pytest.skip("Test cannot run with remote database disk enabled (db disk), as it blocks MinIO which stores database metadata") + + +@pytest.fixture(scope="module") +def cluster(): + try: + cluster = ClickHouseCluster(__file__) + cluster.add_instance( + "replica1", + main_configs=["configs/named_collections.xml", "configs/allow_experimental_export_partition.xml"], + user_configs=["configs/users.d/profile.xml"], + with_minio=True, + stay_alive=True, + with_zookeeper=True, + keeper_required_feature_flags=["multi_read"], + ) + cluster.add_instance( + "replica2", + main_configs=["configs/named_collections.xml", "configs/allow_experimental_export_partition.xml"], + user_configs=["configs/users.d/profile.xml"], + with_minio=True, + stay_alive=True, + with_zookeeper=True, + keeper_required_feature_flags=["multi_read"], + ) + # node that does not participate in the export, but will have visibility over the s3 table + cluster.add_instance( + "watcher_node", + main_configs=["configs/named_collections.xml"], + user_configs=[], + with_minio=True, + ) + cluster.add_instance( + "replica_with_export_disabled", + main_configs=["configs/named_collections.xml", "configs/disable_experimental_export_partition.xml"], + user_configs=["configs/users.d/profile.xml"], + with_minio=True, + stay_alive=True, + with_zookeeper=True, + keeper_required_feature_flags=["multi_read"], + ) + # Sharded instances for filename pattern tests + cluster.add_instance( + "shard1_replica1", + main_configs=["configs/named_collections.xml", "configs/allow_experimental_export_partition.xml", "configs/macros_shard1_replica1.xml"], + user_configs=["configs/users.d/profile.xml"], + with_minio=True, + stay_alive=True, + with_zookeeper=True, + keeper_required_feature_flags=["multi_read"], + ) + + cluster.add_instance( + "shard2_replica1", + main_configs=["configs/named_collections.xml", "configs/allow_experimental_export_partition.xml", "configs/macros_shard2_replica1.xml"], + user_configs=["configs/users.d/profile.xml"], + with_minio=True, + stay_alive=True, + with_zookeeper=True, + keeper_required_feature_flags=["multi_read"], + ) + logging.info("Starting cluster...") + cluster.start() + yield cluster + finally: + cluster.shutdown() + + +@pytest.fixture(autouse=True) +def drop_tables_after_test(cluster): + """Drop all tables in the default database after every test. + + Without this, ReplicatedMergeTree tables from completed tests remain alive and keep + running ZooKeeper background threads (merge selector, queue log, cleanup, export manifest + updater). With many tables alive simultaneously the ZooKeeper session becomes overwhelmed + and subsequent tests start seeing operation-timeout / session-expired errors. + """ + yield + for instance_name, instance in cluster.instances.items(): + try: + tables_str = instance.query( + "SELECT name FROM system.tables WHERE database = 'default' FORMAT TabSeparated" + ).strip() + if not tables_str: + continue + for table in tables_str.split('\n'): + table = table.strip() + if table: + instance.query(f"DROP TABLE IF EXISTS default.`{table}` SYNC") + except Exception as e: + logging.warning(f"drop_tables_after_test: cleanup failed on {instance_name}: {e}") + + +def create_s3_table(node, s3_table): + node.query(f"CREATE TABLE {s3_table} (id UInt64, year UInt16) ENGINE = S3(s3_conn, filename='{s3_table}', format=Parquet, partition_strategy='hive') PARTITION BY year") + + +def create_tables_and_insert_data(node, mt_table, s3_table, replica_name): + node.query(f"DROP TABLE IF EXISTS {mt_table} SYNC") + # enable_block_number_column and enable_block_offset_column are needed for patch parts support + node.query(f"CREATE TABLE {mt_table} (id UInt64, year UInt16) ENGINE = ReplicatedMergeTree('/clickhouse/tables/{mt_table}', '{replica_name}') PARTITION BY year ORDER BY tuple() SETTINGS enable_block_number_column = 1, enable_block_offset_column = 1") + node.query(f"INSERT INTO {mt_table} VALUES (1, 2020), (2, 2020), (3, 2020), (4, 2021)") + + create_s3_table(node, s3_table) + + +def create_sharded_tables_and_insert_data(node, mt_table, s3_table, replica_name): + """Create sharded ReplicatedMergeTree table with {shard} macro in ZooKeeper path.""" + node.query(f"CREATE TABLE {mt_table} (id UInt64, year UInt16) ENGINE = ReplicatedMergeTree('/clickhouse/tables/{{shard}}/{mt_table}', '{replica_name}') PARTITION BY year ORDER BY tuple()") + node.query(f"INSERT INTO {mt_table} VALUES (1, 2020), (2, 2020), (3, 2020), (4, 2021)") + + create_s3_table(node, s3_table) + + +def test_restart_nodes_during_export(cluster): + skip_if_remote_database_disk_enabled(cluster) + node = cluster.instances["replica1"] + node2 = cluster.instances["replica2"] + watcher_node = cluster.instances["watcher_node"] + + postfix = str(uuid.uuid4()).replace("-", "_") + mt_table = f"disaster_mt_table_{postfix}" + s3_table = f"disaster_s3_table_{postfix}" + + create_tables_and_insert_data(node, mt_table, s3_table, "replica1") + create_tables_and_insert_data(node2, mt_table, s3_table, "replica2") + create_s3_table(watcher_node, s3_table) + + # Block S3/MinIO requests to keep exports alive via retry mechanism + # This allows ZooKeeper operations to proceed quickly + minio_ip = cluster.minio_ip + minio_port = cluster.minio_port + + with PartitionManager() as pm: + # Block responses from MinIO (source_port matches MinIO service) + pm_rule_reject_responses_node1 = { + "instance": node, + "destination": node.ip_address, + "protocol": "tcp", + "source_port": minio_port, + "action": "REJECT --reject-with tcp-reset", + } + pm.add_rule(pm_rule_reject_responses_node1) + + pm_rule_reject_responses_node2 = { + "instance": node2, + "destination": node2.ip_address, + "protocol": "tcp", + "source_port": minio_port, + "action": "REJECT --reject-with tcp-reset", + } + pm.add_rule(pm_rule_reject_responses_node2) + + # Block requests to MinIO (destination: MinIO, destination_port: minio_port) + pm_rule_reject_requests_node1 = { + "instance": node, + "destination": minio_ip, + "protocol": "tcp", + "destination_port": minio_port, + "action": "REJECT --reject-with tcp-reset", + } + pm.add_rule(pm_rule_reject_requests_node1) + + pm_rule_reject_requests_node2 = { + "instance": node2, + "destination": minio_ip, + "protocol": "tcp", + "destination_port": minio_port, + "action": "REJECT --reject-with tcp-reset", + } + pm.add_rule(pm_rule_reject_requests_node2) + + export_queries = f""" + ALTER TABLE {mt_table} + EXPORT PARTITION ID '2020' TO TABLE {s3_table} + SETTINGS export_merge_tree_partition_max_retries = 50; + ALTER TABLE {mt_table} + EXPORT PARTITION ID '2021' TO TABLE {s3_table} + SETTINGS export_merge_tree_partition_max_retries = 50; + """ + + node.query(export_queries) + + # wait for the exports to start + wait_for_export_to_start(node, mt_table, s3_table, "2020") + wait_for_export_to_start(node, mt_table, s3_table, "2021") + + node.stop_clickhouse(kill=True) + node2.stop_clickhouse(kill=True) + + assert watcher_node.query(f"SELECT count() FROM {s3_table} where year = 2020") == '0\n', "Partition 2020 was written to S3 during network delay crash" + + assert watcher_node.query(f"SELECT count() FROM {s3_table} where year = 2021") == '0\n', "Partition 2021 was written to S3 during network delay crash" + + # start the nodes, they should finish the export + node.start_clickhouse() + node2.start_clickhouse() + + wait_for_export_status(node, mt_table, s3_table, "2020", "COMPLETED") + wait_for_export_status(node, mt_table, s3_table, "2021", "COMPLETED") + + assert node.query(f"SELECT count() FROM {s3_table} WHERE year = 2020") != f'0\n', "Export of partition 2020 did not resume after crash" + + assert node.query(f"SELECT count() FROM {s3_table} WHERE year = 2021") != f'0\n', "Export of partition 2021 did not resume after crash" + + +@pytest.mark.parametrize( + "system_table_prefer_remote_information", ['0', '1'] +) +def test_kill_export(cluster, system_table_prefer_remote_information): + skip_if_remote_database_disk_enabled(cluster) + node = cluster.instances["replica1"] + node2 = cluster.instances["replica2"] + watcher_node = cluster.instances["watcher_node"] + + postfix = str(uuid.uuid4()).replace("-", "_") + mt_table = f"kill_export_mt_table_{system_table_prefer_remote_information}_{postfix}" + s3_table = f"kill_export_s3_table_{system_table_prefer_remote_information}_{postfix}" + + create_tables_and_insert_data(node, mt_table, s3_table, "replica1") + create_tables_and_insert_data(node2, mt_table, s3_table, "replica2") + + # Block S3/MinIO requests to keep exports alive via retry mechanism + # This allows ZooKeeper operations (KILL) to proceed quickly + minio_ip = cluster.minio_ip + minio_port = cluster.minio_port + + with PartitionManager() as pm: + # Block responses from MinIO (source_port matches MinIO service) + pm_rule_reject_responses = { + "instance": node, + "destination": node.ip_address, + "protocol": "tcp", + "source_port": minio_port, + "action": "REJECT --reject-with tcp-reset", + } + pm.add_rule(pm_rule_reject_responses) + + # Block requests to MinIO (destination: MinIO, destination_port: minio_port) + pm_rule_reject_requests = { + "instance": node, + "destination": minio_ip, + "protocol": "tcp", + "destination_port": minio_port, + "action": "REJECT --reject-with tcp-reset", + } + pm.add_rule(pm_rule_reject_requests) + + # Block responses from MinIO for node2 + pm_rule_reject_responses_node2 = { + "instance": node2, + "destination": node2.ip_address, + "protocol": "tcp", + "source_port": minio_port, + "action": "REJECT --reject-with tcp-reset", + } + pm.add_rule(pm_rule_reject_responses_node2) + + # Block requests to MinIO from node2 + pm_rule_reject_requests_node2 = { + "instance": node2, + "destination": minio_ip, + "protocol": "tcp", + "destination_port": minio_port, + "action": "REJECT --reject-with tcp-reset", + } + pm.add_rule(pm_rule_reject_requests_node2) + + export_queries = f""" + ALTER TABLE {mt_table} + EXPORT PARTITION ID '2020' TO TABLE {s3_table} + SETTINGS export_merge_tree_partition_max_retries = 50; + ALTER TABLE {mt_table} + EXPORT PARTITION ID '2021' TO TABLE {s3_table} + SETTINGS export_merge_tree_partition_max_retries = 50; + """ + + node.query(export_queries) + + # Kill only 2020 while S3 is blocked - retry mechanism keeps exports alive + # ZooKeeper operations (KILL) proceed quickly since only S3 is blocked + node.query(f"KILL EXPORT PARTITION WHERE partition_id = '2020' and source_table = '{mt_table}' and destination_table = '{s3_table}'") + + # sleep for a while to let the kill to be processed + time.sleep(2) + + # wait for 2021 to finish + wait_for_export_status(node, mt_table, s3_table, "2021", "COMPLETED") + + # checking for the commit file because maybe the data file was too fast? + assert node.query(f"SELECT count() FROM s3(s3_conn, filename='{s3_table}/commit_2020_*', format=LineAsString)") == '0\n', "Partition 2020 was written to S3, it was not killed as expected" + assert node.query(f"SELECT count() FROM s3(s3_conn, filename='{s3_table}/commit_2021_*', format=LineAsString)") != f'0\n', "Partition 2021 was not written to S3, but it should have been" + + # check system.replicated_partition_exports for the export, status should be KILLED + assert node.query(f"SELECT status FROM system.replicated_partition_exports WHERE partition_id = '2020' and source_table = '{mt_table}' and destination_table = '{s3_table}' SETTINGS export_merge_tree_partition_system_table_prefer_remote_information = {system_table_prefer_remote_information}") == 'KILLED\n', "Partition 2020 was not killed as expected" + assert node.query(f"SELECT status FROM system.replicated_partition_exports WHERE partition_id = '2021' and source_table = '{mt_table}' and destination_table = '{s3_table}' SETTINGS export_merge_tree_partition_system_table_prefer_remote_information = {system_table_prefer_remote_information}") == 'COMPLETED\n', "Partition 2021 was not completed, this is unexpected" + + # check the data did not land on s3 + assert node.query(f"SELECT count() FROM {s3_table} WHERE year = 2020") == '0\n', "Partition 2020 was written to S3, it was not killed as expected" + + +def test_kill_export_resilient_to_status_handling_failure(cluster): + """KILL EXPORT PARTITION must eventually take effect even when the first + attempt to handle the ZK status-change event throws (simulated via a ONCE + failpoint). The re-queue + reschedule mechanism retries after ~5 s and + the second attempt succeeds because the ONCE failpoint has already fired.""" + skip_if_remote_database_disk_enabled(cluster) + node = cluster.instances["replica1"] + + postfix = str(uuid.uuid4()).replace("-", "_") + mt_table = f"kill_resilient_mt_{postfix}" + s3_table = f"kill_resilient_s3_{postfix}" + + create_tables_and_insert_data(node, mt_table, s3_table, "replica1") + + minio_ip = cluster.minio_ip + minio_port = cluster.minio_port + + with PartitionManager() as pm: + pm.add_rule({ + "instance": node, + "destination": node.ip_address, + "protocol": "tcp", + "source_port": minio_port, + "action": "REJECT --reject-with tcp-reset", + }) + + pm.add_rule({ + "instance": node, + "destination": minio_ip, + "protocol": "tcp", + "destination_port": minio_port, + "action": "REJECT --reject-with tcp-reset", + }) + + node.query( + f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {s3_table}" + f" SETTINGS export_merge_tree_partition_max_retries = 50" + ) + + node.query("SYSTEM ENABLE FAILPOINT export_partition_status_change_throw") + + node.query( + f"KILL EXPORT PARTITION WHERE partition_id = '2020'" + f" AND source_table = '{mt_table}' AND destination_table = '{s3_table}'") + + # sleep for a while to let the kill to be processed + time.sleep(5) + + # The ONCE failpoint makes the first handleStatusChanges() throw. + # The catch re-queues the key and scheduleAfter(5000) arms a retry. + # Wait up to 15 s (5 s retry delay + margin) for the kill to propagate. + wait_for_export_status(node, mt_table, s3_table, "2020", "KILLED", timeout=15) + + # query the local status export_merge_tree_partition_system_table_prefer_remote_information=0 + assert ( + node.query( + f"SELECT status FROM system.replicated_partition_exports" + f" WHERE partition_id = '2020'" + f" AND source_table = '{mt_table}'" + f" AND destination_table = '{s3_table}'" + f" SETTINGS export_merge_tree_partition_system_table_prefer_remote_information = 0" + ).strip() == "KILLED" + ), "Export was not killed — status change was lost after the injected failure" + + +def test_drop_source_table_during_export(cluster): + skip_if_remote_database_disk_enabled(cluster) + node = cluster.instances["replica1"] + # node2 = cluster.instances["replica2"] + watcher_node = cluster.instances["watcher_node"] + + postfix = str(uuid.uuid4()).replace("-", "_") + mt_table = f"drop_source_table_during_export_mt_table_{postfix}" + s3_table = f"drop_source_table_during_export_s3_table_{postfix}" + + create_tables_and_insert_data(node, mt_table, s3_table, "replica1") + # create_tables_and_insert_data(node2, mt_table, s3_table, "replica2") + create_s3_table(watcher_node, s3_table) + + # Block S3/MinIO requests to keep exports alive via retry mechanism + # This allows ZooKeeper operations (KILL) to proceed quickly + minio_ip = cluster.minio_ip + minio_port = cluster.minio_port + + with PartitionManager() as pm: + # Block responses from MinIO (source_port matches MinIO service) + pm_rule_reject_responses = { + "instance": node, + "destination": node.ip_address, + "protocol": "tcp", + "source_port": minio_port, + "action": "REJECT --reject-with tcp-reset", + } + pm.add_rule(pm_rule_reject_responses) + + # Block requests to MinIO (destination: MinIO, destination_port: minio_port) + pm_rule_reject_requests = { + "instance": node, + "destination": minio_ip, + "protocol": "tcp", + "destination_port": minio_port, + "action": "REJECT --reject-with tcp-reset", + } + pm.add_rule(pm_rule_reject_requests) + + export_queries = f""" + ALTER TABLE {mt_table} + EXPORT PARTITION ID '2020' TO TABLE {s3_table} SETTINGS s3_retry_attempts = 500, export_merge_tree_partition_max_retries = 50; + ALTER TABLE {mt_table} + EXPORT PARTITION ID '2021' TO TABLE {s3_table} SETTINGS s3_retry_attempts = 500, export_merge_tree_partition_max_retries = 50; + """ + + node.query(export_queries) + + wait_for_export_status(node, mt_table, s3_table, "2020", "PENDING") + wait_for_export_status(node, mt_table, s3_table, "2021", "PENDING") + + # This should kill the background operations and drop the table + node.query(f"DROP TABLE {mt_table}") + + # Sleep some time to let the export finish (assuming it was not properly cancelled) + time.sleep(10) + + assert node.query(f"SELECT count() FROM s3(s3_conn, filename='{s3_table}/commit_*', format=LineAsString)") == '0\n', "Background operations completed even with the table dropped" + + +def test_concurrent_exports_to_different_targets(cluster): + node = cluster.instances["replica1"] + + postfix = str(uuid.uuid4()).replace("-", "_") + mt_table = f"concurrent_diff_targets_mt_table_{postfix}" + s3_table_a = f"concurrent_diff_targets_s3_a_{postfix}" + s3_table_b = f"concurrent_diff_targets_s3_b_{postfix}" + + create_tables_and_insert_data(node, mt_table, s3_table_a, "replica1") + create_s3_table(node, s3_table_b) + + # Launch two exports of the same partition to two different S3 tables concurrently + with PartitionManager() as pm: + pm.add_network_delay(node, delay_ms=1000) + + node.query( + f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {s3_table_a}" + ) + node.query( + f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {s3_table_b}" + ) + + wait_for_export_status(node, mt_table, s3_table_a, "2020", "COMPLETED") + wait_for_export_status(node, mt_table, s3_table_b, "2020", "COMPLETED") + + # Both targets should receive the same data independently + assert node.query(f"SELECT count() FROM {s3_table_a} WHERE year = 2020") == '3\n', "First target did not receive expected rows" + assert node.query(f"SELECT count() FROM {s3_table_b} WHERE year = 2020") == '3\n', "Second target did not receive expected rows" + + # And both should have a commit marker + assert node.query( + f"SELECT count() FROM s3(s3_conn, filename='{s3_table_a}/commit_2020_*', format=LineAsString)" + ) != '0\n', "Commit file missing for first target" + assert node.query( + f"SELECT count() FROM s3(s3_conn, filename='{s3_table_b}/commit_2020_*', format=LineAsString)" + ) != '0\n', "Commit file missing for second target" + + +def test_failure_is_logged_in_system_table(cluster): + skip_if_remote_database_disk_enabled(cluster) + node = cluster.instances["replica1"] + + postfix = str(uuid.uuid4()).replace("-", "_") + mt_table = f"failure_is_logged_in_system_table_mt_table_{postfix}" + s3_table = f"failure_is_logged_in_system_table_s3_table_{postfix}" + + create_tables_and_insert_data(node, mt_table, s3_table, "replica1") + + # Block traffic to/from MinIO to force upload errors and retries, following existing S3 tests style + minio_ip = cluster.minio_ip + minio_port = cluster.minio_port + + with PartitionManager() as pm: + # Block responses from MinIO (source_port matches MinIO service) + pm_rule_reject_responses = { + "instance": node, + "destination": node.ip_address, + "protocol": "tcp", + "source_port": minio_port, + "action": "REJECT --reject-with tcp-reset", + } + pm.add_rule(pm_rule_reject_responses) + + # Also block requests to MinIO (destination: MinIO, destination_port: 9001) with REJECT to fail fast + pm_rule_reject_requests = { + "instance": node, + "destination": minio_ip, + "protocol": "tcp", + "destination_port": minio_port, + "action": "REJECT --reject-with tcp-reset", + } + pm.add_rule(pm_rule_reject_requests) + + node.query( + f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {s3_table} SETTINGS export_merge_tree_partition_max_retries=1;" + ) + + # Wait so that the export fails + wait_for_export_status(node, mt_table, s3_table, "2020", "FAILED", timeout=60) + + # Network restored; verify the export is marked as FAILED in the system table + # Also verify we captured at least one exception and no commit file exists + status = node.query( + f""" + SELECT status FROM system.replicated_partition_exports + WHERE source_table = '{mt_table}' + AND destination_table = '{s3_table}' + AND partition_id = '2020' + SETTINGS export_merge_tree_partition_system_table_prefer_remote_information = 1 + """ + ) + + assert status.strip() == "FAILED", f"Expected FAILED status, got: {status!r}" + + exception_count = node.query( + f""" + SELECT any(exception_count) FROM system.replicated_partition_exports + WHERE source_table = '{mt_table}' + AND destination_table = '{s3_table}' + AND partition_id = '2020' + SETTINGS export_merge_tree_partition_system_table_prefer_remote_information = 1 + """ + ) + assert int(exception_count.strip()) > 0, "Expected non-zero exception_count in system.replicated_partition_exports" + + # No commit should have been produced for this partition + assert node.query( + f"SELECT count() FROM s3(s3_conn, filename='{s3_table}/commit_2020_*', format=LineAsString)" + ) == '0\n', "Commit file exists despite forced S3 failures" + + +def test_inject_short_living_failures(cluster): + skip_if_remote_database_disk_enabled(cluster) + node = cluster.instances["replica1"] + + postfix = str(uuid.uuid4()).replace("-", "_") + mt_table = f"inject_short_living_failures_mt_table_{postfix}" + s3_table = f"inject_short_living_failures_s3_table_{postfix}" + + create_tables_and_insert_data(node, mt_table, s3_table, "replica1") + + # Block traffic to/from MinIO to force upload errors and retries, following existing S3 tests style + minio_ip = cluster.minio_ip + minio_port = cluster.minio_port + + with PartitionManager() as pm: + # Block responses from MinIO (source_port matches MinIO service) + pm_rule_reject_responses = { + "instance": node, + "destination": node.ip_address, + "protocol": "tcp", + "source_port": minio_port, + "action": "REJECT --reject-with tcp-reset", + } + pm.add_rule(pm_rule_reject_responses) + + # Also block requests to MinIO (destination: MinIO, destination_port: 9001) with REJECT to fail fast + pm_rule_reject_requests = { + "instance": node, + "destination": minio_ip, + "protocol": "tcp", + "destination_port": minio_port, + "action": "REJECT --reject-with tcp-reset", + } + pm.add_rule(pm_rule_reject_requests) + + # set big max_retries so that the export does not fail completely + node.query( + f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {s3_table} SETTINGS export_merge_tree_partition_max_retries=100;" + ) + + # wait for at least one exception to occur, but not enough to finish the export + wait_for_exception_count(node, mt_table, s3_table, "2020", min_exception_count=1, timeout=30) + + # wait for the export to finish + wait_for_export_status(node, mt_table, s3_table, "2020", "COMPLETED") + + # Assert the export succeeded + assert node.query(f"SELECT count() FROM {s3_table} WHERE year = 2020") == '3\n', "Export did not succeed" + assert node.query(f"SELECT count() FROM s3(s3_conn, filename='{s3_table}/commit_2020_*', format=LineAsString)") == '1\n', "Export did not succeed" + + # check system.replicated_partition_exports for the export + assert node.query( + f""" + SELECT status FROM system.replicated_partition_exports + WHERE source_table = '{mt_table}' + AND destination_table = '{s3_table}' + AND partition_id = '2020' + """ + ) == "COMPLETED\n", "Export should be marked as COMPLETED" + + exception_count = node.query( + f""" + SELECT exception_count FROM system.replicated_partition_exports + WHERE source_table = '{mt_table}' + AND destination_table = '{s3_table}' + AND partition_id = '2020' + SETTINGS export_merge_tree_partition_system_table_prefer_remote_information = 1 + """ + ) + assert int(exception_count.strip()) >= 1, "Expected at least one exception" + + +def test_export_ttl(cluster): + node = cluster.instances["replica1"] + + postfix = str(uuid.uuid4()).replace("-", "_") + mt_table = f"export_ttl_mt_table_{postfix}" + s3_table = f"export_ttl_s3_table_{postfix}" + + expiration_time = 3 + + create_tables_and_insert_data(node, mt_table, s3_table, "replica1") + + # start export + node.query(f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {s3_table} SETTINGS export_merge_tree_partition_manifest_ttl={expiration_time};") + + # assert that I get an error when trying to export the same partition again, query_and_get_error + error = node.query_and_get_error(f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {s3_table};") + assert "Export with key" in error, "Expected error about expired export" + + # wait for the export to finish and for the manifest to expire + wait_for_export_status(node, mt_table, s3_table, "2020", "COMPLETED") + time.sleep(expiration_time * 2) + + # assert that the export succeeded, check the commit file + assert node.query(f"SELECT count() FROM s3(s3_conn, filename='{s3_table}/commit_2020_*', format=LineAsString)") == '1\n', "Export did not succeed" + + # start export again + node.query(f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {s3_table}") + + # wait for the export to finish + wait_for_export_status(node, mt_table, s3_table, "2020", "COMPLETED") + + # assert that the export succeeded, check the commit file + # there should be two commit files now, one for the first export and one for the second export + assert node.query(f"SELECT count() FROM s3(s3_conn, filename='{s3_table}/commit_2020_*', format=LineAsString)") == '2\n', "Export did not succeed" + + +def test_export_partition_file_already_exists_policy(cluster): + node = cluster.instances["replica1"] + + postfix = str(uuid.uuid4()).replace("-", "_") + mt_table = f"export_partition_file_already_exists_policy_mt_table_{postfix}" + s3_table = f"export_partition_file_already_exists_policy_s3_table_{postfix}" + + create_tables_and_insert_data(node, mt_table, s3_table, "replica1") + + # stop merges so part names remain stable. it is important for the test. + node.query(f"SYSTEM STOP MERGES {mt_table}") + + # Export all parts + node.query( + f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {s3_table}", + ) + + # check system.replicated_partition_exports for the export + assert node.query( + f""" + SELECT status FROM system.replicated_partition_exports + WHERE source_table = '{mt_table}' + AND destination_table = '{s3_table}' + AND partition_id = '2020' + """ + ) == "COMPLETED\n", "Export should be marked as COMPLETED" + + # wait for the exports to finish + wait_for_export_status(node, mt_table, s3_table, "2020", "COMPLETED") + + # try to export the partition + node.query( + f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {s3_table} SETTINGS export_merge_tree_partition_force_export=1" + ) + + wait_for_export_status(node, mt_table, s3_table, "2020", "COMPLETED") + + assert node.query( + f""" + SELECT count() FROM system.replicated_partition_exports + WHERE source_table = '{mt_table}' + AND destination_table = '{s3_table}' + AND partition_id = '2020' + AND status = 'COMPLETED' + """ + ) == '1\n', "Expected the export to be marked as COMPLETED" + + # overwrite policy + node.query( + f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {s3_table} SETTINGS export_merge_tree_partition_force_export=1, export_merge_tree_part_file_already_exists_policy='overwrite'" + ) + + # wait for the export to finish + wait_for_export_status(node, mt_table, s3_table, "2020", "COMPLETED") + + # check system.replicated_partition_exports for the export + # ideally we would make sure the transaction id is different, but I do not have the time to do that now + assert node.query( + f""" + SELECT count() FROM system.replicated_partition_exports + WHERE source_table = '{mt_table}' + AND destination_table = '{s3_table}' + AND partition_id = '2020' + AND status = 'COMPLETED' + """ + ) == '1\n', "Expected the export to be marked as COMPLETED" + + # last but not least, let's try with the error policy + # max retries = 1 so it fails fast + node.query( + f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {s3_table} SETTINGS export_merge_tree_partition_force_export=1, export_merge_tree_part_file_already_exists_policy='error', export_merge_tree_partition_max_retries=1", + ) + + # wait for the export to finish + wait_for_export_status(node, mt_table, s3_table, "2020", "FAILED") + + # check system.replicated_partition_exports for the export + assert node.query( + f""" + SELECT count() FROM system.replicated_partition_exports + WHERE source_table = '{mt_table}' + AND destination_table = '{s3_table}' + AND partition_id = '2020' + AND status = 'FAILED' + """ + ) == '1\n', "Expected the export to be marked as FAILED" + + +def test_export_partition_feature_is_disabled(cluster): + replica_with_export_disabled = cluster.instances["replica_with_export_disabled"] + + postfix = str(uuid.uuid4()).replace("-", "_") + mt_table = f"export_partition_feature_is_disabled_mt_table_{postfix}" + s3_table = f"export_partition_feature_is_disabled_s3_table_{postfix}" + + create_tables_and_insert_data(replica_with_export_disabled, mt_table, s3_table, "replica1") + + error = replica_with_export_disabled.query_and_get_error(f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {s3_table};") + assert "experimental" in error, "Expected error about disabled feature" + + # make sure kill operation also throws + error = replica_with_export_disabled.query_and_get_error(f"KILL EXPORT PARTITION WHERE partition_id = '2020' and source_table = '{mt_table}' and destination_table = '{s3_table}'") + assert "experimental" in error, "Expected error about disabled feature" + + +def test_export_partition_permissions(cluster): + """Test that export partition validates permissions correctly: + - User needs ALTER permission on source table + - User needs INSERT permission on destination table + """ + node = cluster.instances["replica1"] + + postfix = str(uuid.uuid4()).replace("-", "_") + mt_table = f"permissions_mt_table_{postfix}" + s3_table = f"permissions_s3_table_{postfix}" + + # Create tables as default user + create_tables_and_insert_data(node, mt_table, s3_table, "replica1") + + # Create test users with specific permissions + node.query("CREATE USER IF NOT EXISTS user_no_alter IDENTIFIED WITH no_password") + node.query("CREATE USER IF NOT EXISTS user_no_insert IDENTIFIED WITH no_password") + node.query("CREATE USER IF NOT EXISTS user_with_permissions IDENTIFIED WITH no_password") + + # Grant basic access to all users + node.query(f"GRANT SELECT ON {mt_table} TO user_no_alter") + node.query(f"GRANT SELECT ON {s3_table} TO user_no_alter") + + # user_no_insert has ALTER on source but no INSERT on destination + node.query(f"GRANT ALTER ON {mt_table} TO user_no_insert") + node.query(f"GRANT SELECT ON {s3_table} TO user_no_insert") + + # user_with_permissions has both ALTER and INSERT + node.query(f"GRANT ALTER ON {mt_table} TO user_with_permissions") + node.query(f"GRANT INSERT ON {s3_table} TO user_with_permissions") + + # Test 1: User without ALTER permission should fail + error = node.query_and_get_error( + f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {s3_table}", + user="user_no_alter" + ) + + assert "ACCESS_DENIED" in error or "Not enough privileges" in error, \ + f"Expected ACCESS_DENIED error for user without ALTER, got: {error}" + + # Test 2: User with ALTER but without INSERT permission should fail + error = node.query_and_get_error( + f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {s3_table}", + user="user_no_insert" + ) + + assert "ACCESS_DENIED" in error or "Not enough privileges" in error, \ + f"Expected ACCESS_DENIED error for user without INSERT, got: {error}" + + # Test 3: User with both ALTER and INSERT should succeed + node.query( + f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {s3_table}", + user="user_with_permissions" + ) + + # Wait for export to complete + wait_for_export_status(node, mt_table, s3_table, "2020", "COMPLETED") + + # Verify the export succeeded + result = node.query(f"SELECT count() FROM {s3_table} WHERE year = 2020") + assert result.strip() == "3", f"Expected 3 rows exported, got: {result}" + + # Verify system table shows COMPLETED status + status = node.query( + f""" + SELECT status FROM system.replicated_partition_exports + WHERE source_table = '{mt_table}' + AND destination_table = '{s3_table}' + AND partition_id = '2020' + """ + ) + assert status.strip() == "COMPLETED", f"Expected COMPLETED status, got: {status}" + + +# assert multiple exports within a single query are executed. They all share the same query id +# and previously the transaction id was the query id, which would cause problems +def test_multiple_exports_within_a_single_query(cluster): + node = cluster.instances["replica1"] + + postfix = str(uuid.uuid4()).replace("-", "_") + mt_table = f"multiple_exports_within_a_single_query_mt_table_{postfix}" + s3_table = f"multiple_exports_within_a_single_query_s3_table_{postfix}" + + create_tables_and_insert_data(node, mt_table, s3_table, "replica1") + + node.query(f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {s3_table}, EXPORT PARTITION ID '2021' TO TABLE {s3_table};") + + wait_for_export_status(node, mt_table, s3_table, "2020", "COMPLETED") + wait_for_export_status(node, mt_table, s3_table, "2021", "COMPLETED") + + # assert the exports have been executed + assert node.query(f"SELECT count() FROM {s3_table} WHERE year = 2020") == '3\n', "Export did not succeed" + assert node.query(f"SELECT count() FROM {s3_table} WHERE year = 2021") == '1\n', "Export did not succeed" + + # check system.replicated_partition_exports for the exports + assert node.query( + f""" + SELECT status FROM system.replicated_partition_exports + WHERE source_table = '{mt_table}' + AND destination_table = '{s3_table}' + AND partition_id = '2020' + """ + ) == "COMPLETED\n", "Export should be marked as COMPLETED" + + assert node.query( + f""" + SELECT status FROM system.replicated_partition_exports + WHERE source_table = '{mt_table}' + AND destination_table = '{s3_table}' + AND partition_id = '2021' + """ + ) == "COMPLETED\n", "Export should be marked as COMPLETED" + + +def test_pending_mutations_throw_before_export_partition(cluster): + """Test that pending mutations before export partition throw an error.""" + node = cluster.instances["replica1"] + + postfix = str(uuid.uuid4()).replace("-", "_") + mt_table = f"pending_mutations_throw_partition_mt_table_{postfix}" + s3_table = f"pending_mutations_throw_partition_s3_table_{postfix}" + + create_tables_and_insert_data(node, mt_table, s3_table, "replica1") + + node.query(f"SYSTEM STOP MERGES {mt_table}") + + node.query(f"ALTER TABLE {mt_table} UPDATE id = id + 100 WHERE year = 2020") + + mutations = node.query(f"SELECT count() FROM system.mutations WHERE table = '{mt_table}' AND is_done = 0") + assert mutations.strip() != '0', "Mutation should be pending" + + error = node.query_and_get_error( + f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {s3_table} " + f"SETTINGS export_merge_tree_part_throw_on_pending_mutations=true" + ) + + assert "PENDING_MUTATIONS_NOT_ALLOWED" in error, f"Expected error about pending mutations, got: {error}" + + +def test_pending_mutations_skip_before_export_partition(cluster): + """Test that pending mutations before export partition are skipped with throw_on_pending_mutations=false.""" + node = cluster.instances["replica1"] + + postfix = str(uuid.uuid4()).replace("-", "_") + mt_table = f"pending_mutations_skip_partition_mt_table_{postfix}" + s3_table = f"pending_mutations_skip_partition_s3_table_{postfix}" + + create_tables_and_insert_data(node, mt_table, s3_table, "replica1") + + node.query(f"SYSTEM STOP MERGES {mt_table}") + + node.query(f"ALTER TABLE {mt_table} UPDATE id = id + 100 WHERE year = 2020") + + mutations = node.query(f"SELECT count() FROM system.mutations WHERE table = '{mt_table}' AND is_done = 0") + assert mutations.strip() != '0', "Mutation should be pending" + + node.query( + f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {s3_table} " + f"SETTINGS export_merge_tree_part_throw_on_pending_mutations=false" + ) + + wait_for_export_status(node, mt_table, s3_table, "2020", "COMPLETED") + + result = node.query(f"SELECT id FROM {s3_table} WHERE year = 2020 ORDER BY id") + assert "101" not in result and "102" not in result and "103" not in result, \ + "Export should contain original data before mutation" + assert "1\n2\n3" in result, "Export should contain original data" + + +def test_pending_patch_parts_throw_before_export_partition(cluster): + """Test that pending patch parts before export partition throw an error with default settings.""" + node = cluster.instances["replica1"] + + postfix = str(uuid.uuid4()).replace("-", "_") + mt_table = f"pending_patches_throw_partition_mt_table_{postfix}" + s3_table = f"pending_patches_throw_partition_s3_table_{postfix}" + + create_tables_and_insert_data(node, mt_table, s3_table, "replica1") + + node.query(f"SYSTEM STOP MERGES {mt_table}") + + node.query(f"UPDATE {mt_table} SET id = id + 100 WHERE year = 2020") + + error = node.query_and_get_error( + f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {s3_table}" + ) + + node.query(f"DROP TABLE {mt_table}") + + assert "PENDING_MUTATIONS_NOT_ALLOWED" in error or "pending patch parts" in error.lower(), \ + f"Expected error about pending patch parts, got: {error}" + + +def test_pending_patch_parts_skip_before_export_partition(cluster): + """Test that pending patch parts before export partition are skipped with throw_on_pending_patch_parts=false.""" + node = cluster.instances["replica1"] + + postfix = str(uuid.uuid4()).replace("-", "_") + mt_table = f"pending_patches_skip_partition_mt_table_{postfix}" + s3_table = f"pending_patches_skip_partition_s3_table_{postfix}" + + create_tables_and_insert_data(node, mt_table, s3_table, "replica1") + + node.query(f"SYSTEM STOP MERGES {mt_table}") + + node.query(f"UPDATE {mt_table} SET id = id + 100 WHERE year = 2020") + + node.query( + f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {s3_table} " + f"SETTINGS export_merge_tree_part_throw_on_pending_patch_parts=false" + ) + + wait_for_export_status(node, mt_table, s3_table, "2020", "COMPLETED") + + result = node.query(f"SELECT id FROM {s3_table} WHERE year = 2020 ORDER BY id") + assert "1\n2\n3" in result, "Export should contain original data before patch" + + node.query(f"DROP TABLE {mt_table}") + + +def test_mutations_after_export_partition_started(cluster): + """Test that mutations applied after export partition starts don't affect the exported data.""" + skip_if_remote_database_disk_enabled(cluster) + node = cluster.instances["replica1"] + + postfix = str(uuid.uuid4()).replace("-", "_") + mt_table = f"mutations_after_export_partition_mt_table_{postfix}" + s3_table = f"mutations_after_export_partition_s3_table_{postfix}" + + create_tables_and_insert_data(node, mt_table, s3_table, "replica1") + + # Block traffic to MinIO to delay export + minio_ip = cluster.minio_ip + minio_port = cluster.minio_port + + with PartitionManager() as pm: + pm_rule_reject_responses = { + "instance": node, + "destination": node.ip_address, + "protocol": "tcp", + "source_port": minio_port, + "action": "REJECT --reject-with tcp-reset", + } + pm.add_rule(pm_rule_reject_responses) + + pm_rule_reject_requests = { + "instance": node, + "destination": minio_ip, + "protocol": "tcp", + "destination_port": minio_port, + "action": "REJECT --reject-with tcp-reset", + } + pm.add_rule(pm_rule_reject_requests) + + node.query( + f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {s3_table} " + f"SETTINGS export_merge_tree_part_throw_on_pending_mutations=true" + ) + + # Wait for export to start + wait_for_export_to_start(node, mt_table, s3_table, "2020") + + node.query(f"ALTER TABLE {mt_table} UPDATE id = id + 100 WHERE year = 2020") + + wait_for_export_status(node, mt_table, s3_table, "2020", "COMPLETED") + + result = node.query(f"SELECT id FROM {s3_table} WHERE year = 2020 ORDER BY id") + assert "1\n2\n3" in result, "Export should contain original data before mutation" + assert "101" not in result, "Export should not contain mutated data" + + +def test_patch_parts_after_export_partition_started(cluster): + """Test that patch parts created after export partition starts don't affect the exported data.""" + skip_if_remote_database_disk_enabled(cluster) + node = cluster.instances["replica1"] + + postfix = str(uuid.uuid4()).replace("-", "_") + mt_table = f"patches_after_export_partition_mt_table_{postfix}" + s3_table = f"patches_after_export_partition_s3_table_{postfix}" + + create_tables_and_insert_data(node, mt_table, s3_table, "replica1") + + # Block traffic to MinIO to delay export + minio_ip = cluster.minio_ip + minio_port = cluster.minio_port + + with PartitionManager() as pm: + pm_rule_reject_responses = { + "instance": node, + "destination": node.ip_address, + "protocol": "tcp", + "source_port": minio_port, + "action": "REJECT --reject-with tcp-reset", + } + pm.add_rule(pm_rule_reject_responses) + + pm_rule_reject_requests = { + "instance": node, + "destination": minio_ip, + "protocol": "tcp", + "destination_port": minio_port, + "action": "REJECT --reject-with tcp-reset", + } + pm.add_rule(pm_rule_reject_requests) + + node.query( + f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {s3_table}" + ) + + # Wait for export to start + wait_for_export_to_start(node, mt_table, s3_table, "2020") + + node.query(f"UPDATE {mt_table} SET id = id + 100 WHERE year = 2020") + + wait_for_export_status(node, mt_table, s3_table, "2020", "COMPLETED") + + result = node.query(f"SELECT id FROM {s3_table} WHERE year = 2020 ORDER BY id") + assert "1\n2\n3" in result, "Export should contain original data before patch" + assert "101" not in result, "Export should not contain patched data" + + node.query(f"DROP TABLE {mt_table}") + + +def test_mutation_in_partition_clause(cluster): + """Test that mutations limited to specific partitions using IN PARTITION clause + allow exports of unaffected partitions to succeed.""" + node = cluster.instances["replica1"] + + postfix = str(uuid.uuid4()).replace("-", "_") + mt_table = f"mutation_in_partition_clause_mt_table_{postfix}" + s3_table = f"mutation_in_partition_clause_s3_table_{postfix}" + + create_tables_and_insert_data(node, mt_table, s3_table, "replica1") + + node.query(f"SYSTEM STOP MERGES {mt_table}") + + # Issue a mutation that uses IN PARTITION to limit it to partition 2020 + node.query(f"ALTER TABLE {mt_table} UPDATE id = id + 100 IN PARTITION '2020' WHERE year = 2020") + + # Verify mutation is pending for 2020 + mutations = node.query( + f"SELECT count() FROM system.mutations WHERE table = '{mt_table}' AND is_done = 0" + ) + assert mutations.strip() != '0', "Mutation should be pending" + + # Export of 2020 should fail (it has pending mutations) + error = node.query_and_get_error( + f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {s3_table} " + f"SETTINGS export_merge_tree_part_throw_on_pending_mutations=true" + ) + assert "PENDING_MUTATIONS_NOT_ALLOWED" in error, f"Expected error about pending mutations for partition 2020, got: {error}" + + # Export of 2021 should succeed (no mutations affecting it) + node.query( + f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2021' TO TABLE {s3_table} " + f"SETTINGS export_merge_tree_part_throw_on_pending_mutations=true" + ) + + wait_for_export_status(node, mt_table, s3_table, "2021", "COMPLETED") + + result = node.query(f"SELECT id FROM {s3_table} WHERE year = 2021 ORDER BY id") + assert "4" in result, "Export of partition 2021 should contain original data" + + +def test_export_partition_with_mixed_computed_columns(cluster): + """Test export partition with ALIAS, MATERIALIZED, and EPHEMERAL columns.""" + node = cluster.instances["replica1"] + + postfix = str(uuid.uuid4()).replace("-", "_") + mt_table = f"mixed_computed_mt_table_{postfix}" + s3_table = f"mixed_computed_s3_table_{postfix}" + + node.query(f""" + CREATE TABLE {mt_table} ( + id UInt32, + value UInt32, + tag_input String EPHEMERAL, + doubled UInt64 ALIAS value * 2, + tripled UInt64 MATERIALIZED value * 3, + tag String DEFAULT upper(tag_input) + ) ENGINE = ReplicatedMergeTree('/clickhouse/tables/{mt_table}', 'replica1') + PARTITION BY id + ORDER BY id + SETTINGS index_granularity = 1 + """) + + # Create S3 destination table with regular columns (no EPHEMERAL) + node.query(f""" + CREATE TABLE {s3_table} ( + id UInt32, + value UInt32, + doubled UInt64, + tripled UInt64, + tag String + ) ENGINE = S3(s3_conn, filename='{s3_table}', format=Parquet, partition_strategy='hive') + PARTITION BY id + """) + + node.query(f"INSERT INTO {mt_table} (id, value, tag_input) VALUES (1, 5, 'test'), (1, 10, 'prod')") + + node.query(f"ALTER TABLE {mt_table} EXPORT PARTITION ID '1' TO TABLE {s3_table}") + + wait_for_export_status(node, mt_table, s3_table, "1", "COMPLETED") + + # Verify source data (ALIAS computed, EPHEMERAL not stored) + source_result = node.query(f"SELECT id, value, doubled, tripled, tag FROM {mt_table} ORDER BY value") + expected = "1\t5\t10\t15\tTEST\n1\t10\t20\t30\tPROD\n" + assert source_result == expected, f"Source table data mismatch. Expected:\n{expected}\nGot:\n{source_result}" + + dest_result = node.query(f"SELECT id, value, doubled, tripled, tag FROM {s3_table} ORDER BY value") + assert dest_result == expected, f"Exported data mismatch. Expected:\n{expected}\nGot:\n{dest_result}" + + status = node.query(f""" + SELECT status FROM system.replicated_partition_exports + WHERE source_table = '{mt_table}' + AND destination_table = '{s3_table}' + AND partition_id = '1' + """) + assert status.strip() == "COMPLETED", f"Expected COMPLETED status, got: {status}" + + +def test_sharded_export_partition_with_filename_pattern(cluster): + """Test that export partition with filename pattern prevents collisions in sharded setup.""" + shard1_r1 = cluster.instances["shard1_replica1"] + shard2_r1 = cluster.instances["shard2_replica1"] + watcher_node = cluster.instances["watcher_node"] + + postfix = str(uuid.uuid4()).replace("-", "_") + mt_table = f"sharded_mt_table_{postfix}" + s3_table = f"sharded_s3_table_{postfix}" + + # Create sharded tables on all shards with same partition data (same part names) + # Each shard uses different ZooKeeper path via {shard} macro + create_sharded_tables_and_insert_data(shard1_r1, mt_table, s3_table, "replica1") + create_sharded_tables_and_insert_data(shard2_r1, mt_table, s3_table, "replica1") + create_s3_table(watcher_node, s3_table) + + # Export partition from both shards with filename pattern including shard + # This should prevent filename collisions + shard1_r1.query( + f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {s3_table} " + f"SETTINGS export_merge_tree_part_filename_pattern = '{{part_name}}_{{shard}}_{{replica}}_{{checksum}}'" + ) + shard2_r1.query( + f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {s3_table} " + f"SETTINGS export_merge_tree_part_filename_pattern = '{{part_name}}_{{shard}}_{{replica}}_{{checksum}}'" + ) + + # Wait for exports to complete + wait_for_export_status(shard1_r1, mt_table, s3_table, "2020", "COMPLETED") + wait_for_export_status(shard2_r1, mt_table, s3_table, "2020", "COMPLETED") + + total_count = watcher_node.query(f"SELECT count() FROM {s3_table} WHERE year = 2020").strip() + assert total_count == "6", f"Expected 6 total rows (3 from each shard), got {total_count}" + + # Verify filenames contain shard information (check via S3 directly) + # Get all files from S3 - query from watcher_node since S3 is shared + files_shard1 = watcher_node.query( + f"SELECT _file FROM s3(s3_conn, filename='{s3_table}/**', format='One') WHERE _file LIKE '%shard1%' LIMIT 1" + ).strip() + files_shard2 = watcher_node.query( + f"SELECT _file FROM s3(s3_conn, filename='{s3_table}/**', format='One') WHERE _file LIKE '%shard2%' LIMIT 1" + ).strip() + + # Both shards should have files with their shard names + assert "shard1" in files_shard1 or files_shard1 == "", f"Expected shard1 in filenames, got: {files_shard1}" + assert "shard2" in files_shard2 or files_shard2 == "", f"Expected shard2 in filenames, got: {files_shard2}" + + +def test_export_partition_from_replicated_database_uses_db_shard_replica_macros(cluster): + """Test that {shard} and {replica} in the filename pattern are expanded from the + DatabaseReplicated identity, NOT from server config macros. + + replica1 has no / entries in its server config section. + Without the fix buildDestinationFilename() leaves macro_info.shard/replica unset, so + Macros::expand() falls through to the config-macros lookup and throws NO_ELEMENTS_IN_CONFIG. + With the fix the DatabaseReplicated shard_name / replica_name are injected into macro_info + before the expand call, and the pattern resolves correctly. + """ + + # The remote disk test suite sets the shard and replica macros in https://github.com/Altinity/ClickHouse/blob/bbabcaa96e8b7fe8f70ecd0bd4f76fb0f76f2166/tests/integration/helpers/cluster.py#L4356 + # When expanding the macros, the configured ones are preferred over the ones from the DatabaseReplicated definition. + # Therefore, this test fails. It is easier to skip it than to fix it. + skip_if_remote_database_disk_enabled(cluster) + + node = cluster.instances["replica1"] + watcher_node = cluster.instances["watcher_node"] + + postfix = str(uuid.uuid4()).replace("-", "_") + db_name = f"repdb_{postfix}" + table_name = "mt_table" + s3_table = f"s3_dbreplicated_{postfix}" + + # These values exist only in the DatabaseReplicated definition – they are NOT + # present anywhere in replica1's server config . + db_shard = "db_shard_x" + db_replica = "db_replica_y" + + node.query( + f"CREATE DATABASE {db_name} " + f"ENGINE = Replicated('/clickhouse/databases/{db_name}', '{db_shard}', '{db_replica}')") + + node.query(f""" + CREATE TABLE {db_name}.{table_name} + (id UInt64, year UInt16) + ENGINE = ReplicatedMergeTree() + PARTITION BY year ORDER BY tuple()""") + + node.query(f"INSERT INTO {db_name}.{table_name} VALUES (1, 2020), (2, 2020), (3, 2020)") + # Stop merges so part names stay stable during the test. + node.query(f"SYSTEM STOP MERGES {db_name}.{table_name}") + + node.query( + f"CREATE TABLE {s3_table} (id UInt64, year UInt16) " + f"ENGINE = S3(s3_conn, filename='{s3_table}', format=Parquet, partition_strategy='hive') " + f"PARTITION BY year") + + watcher_node.query( + f"CREATE TABLE {s3_table} (id UInt64, year UInt16) " + f"ENGINE = S3(s3_conn, filename='{s3_table}', format=Parquet, partition_strategy='hive') " + f"PARTITION BY year") + + # Export with {shard} and {replica} in the pattern. + # Before the fix: Macros::expand throws NO_ELEMENTS_IN_CONFIG because replica1 has + # no / server config macros. + # After the fix: DatabaseReplicated's shard_name/replica_name are wired into + # macro_info before the expand call, so this succeeds and produces the right names. + node.query( + f"ALTER TABLE {db_name}.{table_name} EXPORT PARTITION ID '2020' TO TABLE {s3_table} " + f"SETTINGS export_merge_tree_part_filename_pattern = " + f"'{{part_name}}_{{shard}}_{{replica}}_{{checksum}}'") + + # A FAILED status here almost certainly means the macro expansion threw + # NO_ELEMENTS_IN_CONFIG (i.e. the fix is missing or broken). + wait_for_export_status(node, table_name, s3_table, "2020", "COMPLETED") + + # Data should have landed in S3. + count = watcher_node.query(f"SELECT count() FROM {s3_table} WHERE year = 2020").strip() + assert count == "3", f"Expected 3 exported rows, got {count}" + + # The exported filename must contain the exact shard and replica names from the + # DatabaseReplicated definition, proving the fix injected them (not server config macros). + filename = watcher_node.query( + f"SELECT _file FROM s3(s3_conn, filename='{s3_table}/**/*.parquet', format='One') LIMIT 1" + ).strip() + + assert db_shard in filename, ( + f"Expected filename to contain DatabaseReplicated shard '{db_shard}', got: {filename!r}. " + "Suggests {shard} was not expanded from the DatabaseReplicated identity.") + + assert db_replica in filename, ( + f"Expected filename to contain DatabaseReplicated replica '{db_replica}', got: {filename!r}. " + "Suggests {replica} was not expanded from the DatabaseReplicated identity.") + + +def test_sharded_export_partition_default_pattern(cluster): + shard1_r1 = cluster.instances["shard1_replica1"] + shard2_r1 = cluster.instances["shard2_replica1"] + watcher_node = cluster.instances["watcher_node"] + + mt_table = "sharded_mt_table_default" + s3_table = "sharded_s3_table_default" + + # Create sharded tables with different ZooKeeper paths per shard + create_sharded_tables_and_insert_data(shard1_r1, mt_table, s3_table, "replica1") + create_sharded_tables_and_insert_data(shard2_r1, mt_table, s3_table, "replica1") + create_s3_table(watcher_node, s3_table) + + # Export with default pattern ({part_name}_{checksum}) - may cause collisions if parts have same name and the same checksum + shard1_r1.query( + f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {s3_table}" + ) + shard2_r1.query( + f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {s3_table}" + ) + + wait_for_export_status(shard1_r1, mt_table, s3_table, "2020", "COMPLETED") + wait_for_export_status(shard2_r1, mt_table, s3_table, "2020", "COMPLETED") + + # Both exports should complete (even if there are collisions, the overwrite policy handles it) + # S3 tables are shared, so query from watcher_node + total_count = watcher_node.query(f"SELECT count() FROM {s3_table} WHERE year = 2020").strip() + + # only one file with 3 rows should be present + assert int(total_count) == 3, f"Expected 3 rows, got {total_count}" + + +def test_export_partition_scheduler_skipped_when_moves_stopped(cluster): + node = cluster.instances["replica1"] + + uid = str(uuid.uuid4()).replace("-", "_") + mt_table = f"sched_skip_mt_{uid}" + s3_table = f"sched_skip_s3_{uid}" + + create_tables_and_insert_data(node, mt_table, s3_table, "replica1") + + node.query(f"SYSTEM STOP MOVES {mt_table}") + + node.query( + f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {s3_table}" + ) + + wait_for_export_to_start(node, mt_table, s3_table, "2020") + + # Wait for several scheduler cycles (each fires every 5 s). + # If the guard is missing the scheduler would run and data would land in S3. + time.sleep(10) + + status = node.query( + f"SELECT status FROM system.replicated_partition_exports" + f" WHERE source_table = '{mt_table}' AND destination_table = '{s3_table}'" + f" AND partition_id = '2020'" + ).strip() + + assert status == "PENDING", ( + f"Expected PENDING while moves are stopped, got '{status}'" + ) + + row_count = int(node.query(f"SELECT count() FROM {s3_table} WHERE year = 2020").strip()) + assert row_count == 0, ( + f"Expected 0 rows in S3 while scheduler is skipped, got {row_count}" + ) + + node.query(f"SYSTEM START MOVES {mt_table}") + + wait_for_export_status(node, mt_table, s3_table, "2020", "COMPLETED", timeout=60) + + row_count = int(node.query(f"SELECT count() FROM {s3_table} WHERE year = 2020").strip()) + assert row_count == 3, f"Expected 3 rows in S3 after export completed, got {row_count}" + + +def test_export_partition_resumes_after_stop_moves(cluster): + node = cluster.instances["replica1"] + + uid = str(uuid.uuid4()).replace("-", "_") + mt_table = f"stop_moves_before_mt_{uid}" + s3_table = f"stop_moves_before_s3_{uid}" + + create_tables_and_insert_data(node, mt_table, s3_table, "replica1") + + node.query(f"SYSTEM STOP MOVES {mt_table}") + + node.query( + f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {s3_table}" + f" SETTINGS export_merge_tree_partition_max_retries = 50" + ) + + wait_for_export_to_start(node, mt_table, s3_table, "2020") + + # Give the scheduler enough time to attempt (and cancel) the part task at + # least once, exercising the lock-release code path. + time.sleep(5) + + status = node.query( + f"SELECT status FROM system.replicated_partition_exports" + f" WHERE source_table = '{mt_table}' AND destination_table = '{s3_table}'" + f" AND partition_id = '2020'" + ).strip() + assert status == "PENDING", f"Expected PENDING while moves are stopped, got '{status}'" + + row_count = int(node.query(f"SELECT count() FROM {s3_table} WHERE year = 2020").strip()) + assert row_count == 0, f"Expected 0 rows in S3 while moves are stopped, got {row_count}" + + node.query(f"SYSTEM START MOVES {mt_table}") + + wait_for_export_status(node, mt_table, s3_table, "2020", "COMPLETED", timeout=60) + + row_count = int(node.query(f"SELECT count() FROM {s3_table} WHERE year = 2020").strip()) + assert row_count == 3, f"Expected 3 rows in S3 after export completed, got {row_count}" + + +def test_export_partition_resumes_after_stop_moves_during_export(cluster): + skip_if_remote_database_disk_enabled(cluster) + + node = cluster.instances["replica1"] + + uid = str(uuid.uuid4()).replace("-", "_") + mt_table = f"stop_moves_during_mt_{uid}" + s3_table = f"stop_moves_during_s3_{uid}" + + create_tables_and_insert_data(node, mt_table, s3_table, "replica1") + + minio_ip = cluster.minio_ip + minio_port = cluster.minio_port + + with PartitionManager() as pm: + pm.add_rule({ + "instance": node, + "destination": node.ip_address, + "protocol": "tcp", + "source_port": minio_port, + "action": "REJECT --reject-with tcp-reset", + }) + pm.add_rule({ + "instance": node, + "destination": minio_ip, + "protocol": "tcp", + "destination_port": minio_port, + "action": "REJECT --reject-with tcp-reset", + }) + + node.query( + f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {s3_table}" + f" SETTINGS export_merge_tree_partition_max_retries = 50" + ) + + wait_for_export_to_start(node, mt_table, s3_table, "2020") + + # Let the tasks start executing and failing against the blocked S3. + time.sleep(2) + + node.query(f"SYSTEM STOP MOVES {mt_table}") + + # Give the cancel callback time to fire and the lock-release path to run. + time.sleep(3) + + status = node.query( + f"SELECT status FROM system.replicated_partition_exports" + f" WHERE source_table = '{mt_table}' AND destination_table = '{s3_table}'" + f" AND partition_id = '2020'" + ).strip() + + assert status == "PENDING", ( + f"Expected PENDING while moves are stopped and S3 is blocked, got '{status}'" + ) + + node.query(f"SYSTEM START MOVES {mt_table}") + + # MinIO is now unblocked; the next scheduler cycle should succeed. + wait_for_export_status(node, mt_table, s3_table, "2020", "COMPLETED", timeout=60) + + row_count = int(node.query(f"SELECT count() FROM {s3_table} WHERE year = 2020").strip()) + assert row_count == 3, f"Expected 3 rows in S3 after export completed, got {row_count}" diff --git a/tests/integration/test_storage_iceberg_with_spark/configs/config.d/allow_export_partition.xml b/tests/integration/test_storage_iceberg_with_spark/configs/config.d/allow_export_partition.xml new file mode 100644 index 000000000000..514cd710836a --- /dev/null +++ b/tests/integration/test_storage_iceberg_with_spark/configs/config.d/allow_export_partition.xml @@ -0,0 +1,3 @@ + + 1 + diff --git a/tests/integration/test_storage_iceberg_with_spark/configs/users.d/allow_export_partition.xml b/tests/integration/test_storage_iceberg_with_spark/configs/users.d/allow_export_partition.xml new file mode 100644 index 000000000000..b129913efd18 --- /dev/null +++ b/tests/integration/test_storage_iceberg_with_spark/configs/users.d/allow_export_partition.xml @@ -0,0 +1,7 @@ + + + + 1 + + + diff --git a/tests/integration/test_storage_iceberg_with_spark/test_export_partition_iceberg.py b/tests/integration/test_storage_iceberg_with_spark/test_export_partition_iceberg.py new file mode 100644 index 000000000000..d289639eac12 --- /dev/null +++ b/tests/integration/test_storage_iceberg_with_spark/test_export_partition_iceberg.py @@ -0,0 +1,912 @@ +""" +Tests for EXPORT PARTITION to an Iceberg table that was created by Apache Spark. + +The destination Iceberg metadata — including field IDs and the partition spec — +is written by Spark, not by ClickHouse, which removes any bias from tests where +both source and destination are ClickHouse-created. + +A separate module-level fixture is used because the package-level +started_cluster_iceberg_with_spark does not include ZooKeeper (which is +required for ReplicatedMergeTree / EXPORT PARTITION). + +Transform coverage (ClickHouse → Iceberg): + identity → identity + toYearNumSinceEpoch → year + toMonthNumSinceEpoch → month + toRelativeDayNum → day + toRelativeHourNum → hour + icebergBucket(N) → bucket(N) + icebergTruncate(N) → truncate(N) + compound → multiple fields +""" + +import logging +import threading +import time +from concurrent.futures import ThreadPoolExecutor + +import pytest +import pyspark + +from helpers.cluster import ClickHouseCluster +from helpers.export_partition_helpers import ( + first_partition_id, + make_iceberg_s3, + make_rmt, + unique_suffix, + wait_for_export_status, +) +from helpers.iceberg_utils import ( + create_iceberg_table, + default_upload_directory, +) +from helpers.s3_tools import S3Uploader, prepare_s3_bucket + + +# --------------------------------------------------------------------------- +# Spark session +# --------------------------------------------------------------------------- + +def get_spark(): + builder = ( + pyspark.sql.SparkSession.builder + .appName("test_export_partition_spark_iceberg") + .config( + "spark.sql.catalog.spark_catalog", + "org.apache.iceberg.spark.SparkSessionCatalog", + ) + .config("spark.sql.catalog.local", "org.apache.iceberg.spark.SparkCatalog") + .config("spark.sql.catalog.spark_catalog.type", "hadoop") + .config( + "spark.sql.catalog.spark_catalog.warehouse", + "/var/lib/clickhouse/user_files/iceberg_data", + ) + .config( + "spark.sql.extensions", + "org.apache.iceberg.spark.extensions.IcebergSparkSessionExtensions", + ) + .master("local") + ) + return builder.getOrCreate() + + +# --------------------------------------------------------------------------- +# Cluster fixture +# --------------------------------------------------------------------------- + +@pytest.fixture(scope="module") +def export_cluster(): + try: + cluster = ClickHouseCluster(__file__, with_spark=True) + cluster.add_instance( + "node1", + main_configs=[ + "configs/config.d/named_collections.xml", + "configs/config.d/allow_export_partition.xml", + ], + user_configs=[ + "configs/users.d/allow_export_partition.xml", + ], + with_minio=True, + stay_alive=True, + with_zookeeper=True, + keeper_required_feature_flags=["multi_read"], + ) + for name in ["replica1", "replica2", "replica3"]: + cluster.add_instance( + name, + main_configs=[ + "configs/config.d/named_collections.xml", + "configs/config.d/allow_export_partition.xml", + ], + user_configs=[ + "configs/users.d/allow_export_partition.xml", + ], + stay_alive=True, + with_zookeeper=True, + keeper_required_feature_flags=["multi_read"], + ) + logging.info("Starting export_cluster...") + cluster.start() + prepare_s3_bucket(cluster) + cluster.spark_session = get_spark() + cluster.default_s3_uploader = S3Uploader(cluster.minio_client, cluster.minio_bucket) + yield cluster + finally: + cluster.shutdown() + + +@pytest.fixture(autouse=True) +def drop_tables(export_cluster): + yield + for node_name in ["node1", "replica1", "replica2", "replica3"]: + node = export_cluster.instances[node_name] + try: + tables = node.query( + "SELECT name FROM system.tables WHERE database = 'default' FORMAT TabSeparated" + ).strip() + for table in tables.splitlines(): + table = table.strip() + if table: + node.query(f"DROP TABLE IF EXISTS default.`{table}` SYNC") + except Exception as e: + logging.warning(f"drop_tables cleanup failed on {node_name}: {e}") + + +# --------------------------------------------------------------------------- +# Helpers +# --------------------------------------------------------------------------- + +def spark_iceberg(cluster, spark, iceberg_name: str, ddl: str): + """Execute a Spark DDL and upload the resulting Iceberg files to MinIO.""" + spark.sql(ddl) + default_upload_directory( + cluster, + "s3", + f"/iceberg_data/default/{iceberg_name}/", + f"/iceberg_data/default/{iceberg_name}/", + ) + + +def attach_ch_iceberg(node, iceberg_name: str, schema: str, cluster): + """ + Attach a ClickHouse IcebergS3 table to an existing Spark-written Iceberg path. + No PARTITION BY is specified — the spec is read from Spark's metadata. + """ + create_iceberg_table( + "s3", + node, + iceberg_name, + cluster, + schema=f"({schema})", + if_not_exists=True, + ) + + + +def run_accepted(export_cluster, label, spark_ddl, ch_schema, rmt_columns, rmt_partition_by, insert_values): + """ + Create a Spark-created Iceberg table, attach ClickHouse to it, create the + source RMT, export, wait, and return (node, source, iceberg, partition_id) + so the caller can do additional assertions. + """ + node = export_cluster.instances["node1"] + spark = export_cluster.spark_session + + uid = unique_suffix() + source = f"rmt_{label}_{uid}" + iceberg = f"spark_{label}_{uid}" + + spark_iceberg(export_cluster, spark, iceberg, spark_ddl.format(TABLE=iceberg)) + attach_ch_iceberg(node, iceberg, ch_schema, export_cluster) + make_rmt(node, source, rmt_columns, rmt_partition_by, order_by="id") + node.query(f"INSERT INTO {source} VALUES {insert_values}") + + pid = first_partition_id(node, source) + node.query(f"ALTER TABLE {source} EXPORT PARTITION ID '{pid}' TO TABLE {iceberg}") + wait_for_export_status(node, source, iceberg, pid) + + return node, source, iceberg, pid + + +def run_rejected(export_cluster, label, spark_ddl, ch_schema, rmt_columns, rmt_partition_by, insert_values): + """ + Create a mismatched pair and assert that EXPORT PARTITION fails with BAD_ARGUMENTS. + The check fires synchronously before any task is enqueued. + """ + node = export_cluster.instances["node1"] + spark = export_cluster.spark_session + + uid = unique_suffix() + source = f"rmt_{label}_{uid}" + iceberg = f"spark_{label}_{uid}" + + spark_iceberg(export_cluster, spark, iceberg, spark_ddl.format(TABLE=iceberg)) + attach_ch_iceberg(node, iceberg, ch_schema, export_cluster) + make_rmt(node, source, rmt_columns, rmt_partition_by, order_by="id") + node.query(f"INSERT INTO {source} VALUES {insert_values}") + + pid = first_partition_id(node, source) + error = node.query_and_get_error( + f"ALTER TABLE {source} EXPORT PARTITION ID '{pid}' TO TABLE {iceberg}" + ) + return error + + +# --------------------------------------------------------------------------- +# Replicated helpers +# --------------------------------------------------------------------------- + + +def create_iceberg_s3_table(node, iceberg_table: str, if_not_exists: bool = False): + """Create (or attach to an existing) IcebergS3 table at a per-test MinIO prefix.""" + make_iceberg_s3( + node, iceberg_table, "id Int64, year Int32", + partition_by="year", if_not_exists=if_not_exists, + ) + + +def setup_replicas(cluster, mt_table: str, iceberg_table: str, replica_names: list): + """ + Create RMT on each replica with a per-replica replica_name so all instances share + the same ZooKeeper path. Create IcebergS3 on the primary; attach with IF NOT EXISTS + on the rest. No data is inserted here — callers manage their own test data. + """ + instances = [cluster.instances[n] for n in replica_names] + primary = instances[0] + + for rname, instance in zip(replica_names, instances): + make_rmt(instance, mt_table, "id Int64, year Int32", "year", replica_name=rname) + + create_iceberg_s3_table(primary, iceberg_table) + for instance in instances[1:]: + create_iceberg_s3_table(instance, iceberg_table, if_not_exists=True) + + + +# --------------------------------------------------------------------------- +# Happy-path tests — one per transform +# --------------------------------------------------------------------------- + +def test_identity_transform(export_cluster): + """Spark identity(year) <-> PARTITION BY year.""" + node, _, iceberg, _ = run_accepted( + export_cluster, + "identity", + spark_ddl="CREATE TABLE {TABLE} (id BIGINT, year INT)" + " USING iceberg PARTITIONED BY (identity(year)) OPTIONS('format-version'='2')", + ch_schema="id Int64, year Int32", + rmt_columns="id Int64, year Int32", + rmt_partition_by="year", + insert_values="(1, 2024), (2, 2024), (3, 2024)", + ) + assert int(node.query(f"SELECT count() FROM {iceberg}").strip()) == 3 + + +def test_year_transform(export_cluster): + """Spark years(dt) <-> PARTITION BY toYearNumSinceEpoch(dt).""" + node, _, iceberg, _ = run_accepted( + export_cluster, + "year", + spark_ddl="CREATE TABLE {TABLE} (id BIGINT, dt DATE)" + " USING iceberg PARTITIONED BY (years(dt)) OPTIONS('format-version'='2')", + ch_schema="id Int64, dt Date", + rmt_columns="id Int64, dt Date", + rmt_partition_by="toYearNumSinceEpoch(dt)", + insert_values="(1, '2021-03-01'), (2, '2021-07-15'), (3, '2021-12-31')", + ) + assert int(node.query(f"SELECT count() FROM {iceberg}").strip()) == 3 + + +def test_month_transform(export_cluster): + """Spark months(dt) <-> PARTITION BY toMonthNumSinceEpoch(dt).""" + node, _, iceberg, _ = run_accepted( + export_cluster, + "month", + spark_ddl="CREATE TABLE {TABLE} (id BIGINT, dt DATE)" + " USING iceberg PARTITIONED BY (months(dt)) OPTIONS('format-version'='2')", + ch_schema="id Int64, dt Date", + rmt_columns="id Int64, dt Date", + rmt_partition_by="toMonthNumSinceEpoch(dt)", + insert_values="(1, '2020-06-01'), (2, '2020-06-15'), (3, '2020-06-30')", + ) + assert int(node.query(f"SELECT count() FROM {iceberg}").strip()) == 3 + + +def test_day_transform(export_cluster): + """Spark days(dt) <-> PARTITION BY toRelativeDayNum(dt).""" + node, _, iceberg, _ = run_accepted( + export_cluster, + "day", + spark_ddl="CREATE TABLE {TABLE} (id BIGINT, dt DATE)" + " USING iceberg PARTITIONED BY (days(dt)) OPTIONS('format-version'='2')", + ch_schema="id Int64, dt Date", + rmt_columns="id Int64, dt Date", + rmt_partition_by="toRelativeDayNum(dt)", + insert_values="(1, '2023-03-15'), (2, '2023-03-15'), (3, '2023-03-15')", + ) + assert int(node.query(f"SELECT count() FROM {iceberg}").strip()) == 3 + + +def test_hour_transform(export_cluster): + """Spark hours(ts) <-> PARTITION BY toRelativeHourNum(ts). + + Spark TIMESTAMP maps to Iceberg 'timestamp' which ClickHouse reads as DateTime64(6). + All three rows fall within the same hour so a single partition is exported. + """ + node, _, iceberg, _ = run_accepted( + export_cluster, + "hour", + spark_ddl="CREATE TABLE {TABLE} (id BIGINT, ts TIMESTAMP)" + " USING iceberg PARTITIONED BY (hours(ts)) OPTIONS('format-version'='2')", + ch_schema="id Int64, ts DateTime64(6)", + rmt_columns="id Int64, ts DateTime64(6)", + rmt_partition_by="toRelativeHourNum(ts)", + insert_values=( + "(1, '2023-03-15 10:00:00'), " + "(2, '2023-03-15 10:30:00'), " + "(3, '2023-03-15 10:59:00')" + ), + ) + assert int(node.query(f"SELECT count() FROM {iceberg}").strip()) == 3 + + +def test_bucket_transform(export_cluster): + """Spark bucket(8, user_id) <-> PARTITION BY icebergBucket(8, user_id).""" + node, _, iceberg, _ = run_accepted( + export_cluster, + "bucket", + spark_ddl="CREATE TABLE {TABLE} (id BIGINT, user_id BIGINT)" + " USING iceberg PARTITIONED BY (bucket(8, user_id)) OPTIONS('format-version'='2')", + ch_schema="id Int64, user_id Int64", + rmt_columns="id Int64, user_id Int64", + rmt_partition_by="icebergBucket(8, user_id)", + # All rows share the same user_id → same bucket → single partition. + insert_values="(1, 42), (2, 42), (3, 42)", + ) + assert int(node.query(f"SELECT count() FROM {iceberg}").strip()) == 3 + + +def test_truncate_transform(export_cluster): + """Spark truncate(4, category) <-> PARTITION BY icebergTruncate(4, category).""" + node, _, iceberg, _ = run_accepted( + export_cluster, + "truncate", + spark_ddl="CREATE TABLE {TABLE} (id BIGINT, category STRING)" + " USING iceberg PARTITIONED BY (truncate(4, category)) OPTIONS('format-version'='2')", + ch_schema="id Int64, category String", + rmt_columns="id Int64, category String", + rmt_partition_by="icebergTruncate(4, category)", + # All share the 4-char prefix 'clic' → same truncate bucket. + insert_values="(1, 'clickhouse'), (2, 'click'), (3, 'clickstream')", + ) + assert int(node.query(f"SELECT count() FROM {iceberg}").strip()) == 3 + + +def test_compound_transform(export_cluster): + """Spark (identity(year), identity(region)) <-> PARTITION BY (year, region).""" + node, _, iceberg, _ = run_accepted( + export_cluster, + "compound", + spark_ddl="CREATE TABLE {TABLE} (id BIGINT, year INT, region STRING)" + " USING iceberg PARTITIONED BY (identity(year), identity(region))" + " OPTIONS('format-version'='2')", + ch_schema="id Int64, year Int32, region String", + rmt_columns="id Int64, year Int32, region String", + rmt_partition_by="(year, region)", + insert_values="(1, 2022, 'EU'), (2, 2022, 'EU'), (3, 2022, 'EU')", + ) + assert int(node.query(f"SELECT count() FROM {iceberg}").strip()) == 3 + + +def test_identity_int64(export_cluster): + """Spark identity(user_id) on BIGINT <-> PARTITION BY user_id (Int64). + + Int64 → Avro 'long' is already handled by getAvroType(). This test covers + the identity transform on a 64-bit integer column, which is not covered by + the existing test_identity_transform (which uses Int32). + """ + node, _, iceberg, _ = run_accepted( + export_cluster, + "identity_int64", + spark_ddl="CREATE TABLE {TABLE} (id BIGINT, user_id BIGINT)" + " USING iceberg PARTITIONED BY (identity(user_id))" + " OPTIONS('format-version'='2')", + ch_schema="id Int64, user_id Int64", + rmt_columns="id Int64, user_id Int64", + rmt_partition_by="user_id", + insert_values="(1, 100), (2, 100), (3, 100)", + ) + assert int(node.query(f"SELECT count() FROM {iceberg}").strip()) == 3 + + +def test_identity_date(export_cluster): + """Spark identity(event_date) on DATE <-> PARTITION BY event_date (Date32). + + Date32 → Avro 'int' is already handled by getAvroType(). This test covers + the identity transform directly on a date column. Existing date-related tests + (test_year_transform, test_month_transform, etc.) use time-based transforms + such as years() and months(), not identity(). + """ + node, _, iceberg, _ = run_accepted( + export_cluster, + "identity_date", + spark_ddl="CREATE TABLE {TABLE} (id BIGINT, event_date DATE)" + " USING iceberg PARTITIONED BY (identity(event_date))" + " OPTIONS('format-version'='2')", + ch_schema="id Int64, event_date Date32", + rmt_columns="id Int64, event_date Date32", + rmt_partition_by="event_date", + insert_values="(1, '2024-03-15'), (2, '2024-03-15'), (3, '2024-03-15')", + ) + assert int(node.query(f"SELECT count() FROM {iceberg}").strip()) == 3 + + +def test_identity_string(export_cluster): + """Spark identity(region) on STRING <-> PARTITION BY region (String). + + String → Avro 'string' is already handled by getAvroType(). This test covers + identity on a string column as the sole partition field. The existing + test_compound_transform uses identity(region) only as part of a multi-field spec, + so a standalone string identity partition was not previously exercised end-to-end. + """ + node, _, iceberg, _ = run_accepted( + export_cluster, + "identity_str", + spark_ddl="CREATE TABLE {TABLE} (id BIGINT, region STRING)" + " USING iceberg PARTITIONED BY (identity(region))" + " OPTIONS('format-version'='2')", + ch_schema="id Int64, region String", + rmt_columns="id Int64, region String", + rmt_partition_by="region", + insert_values="(1, 'EU'), (2, 'EU'), (3, 'EU')", + ) + assert int(node.query(f"SELECT count() FROM {iceberg}").strip()) == 3 + + +def test_truncate_int64(export_cluster): + """Spark truncate(10, amount) on BIGINT <-> PARTITION BY icebergTruncate(10, amount) (Int64). + + Int64 truncate produces floor(v / 10) * 10, so all rows with amount=42 land in + partition value 40 (same partition). This is a distinct code path from truncate on + String (which trims a character prefix). The existing test_truncate_transform uses + String only, leaving the integer truncate path untested. + """ + node, _, iceberg, _ = run_accepted( + export_cluster, + "truncate_int64", + spark_ddl="CREATE TABLE {TABLE} (id BIGINT, amount BIGINT)" + " USING iceberg PARTITIONED BY (truncate(10, amount))" + " OPTIONS('format-version'='2')", + ch_schema="id Int64, amount Int64", + rmt_columns="id Int64, amount Int64", + rmt_partition_by="icebergTruncate(10, amount)", + # All rows have amount=42 → truncated partition value is 40. + insert_values="(1, 42), (2, 42), (3, 42)", + ) + assert int(node.query(f"SELECT count() FROM {iceberg}").strip()) == 3 + + +def test_bucket_string(export_cluster): + """Spark bucket(8, name) on STRING <-> PARTITION BY icebergBucket(8, name) (String). + + Bucket on strings uses Murmur3 hash of the UTF-8 bytes, a different hash path than + bucket on integers. All rows share the same name so they land in the same bucket. + The existing test_bucket_transform uses BIGINT only, leaving string bucketing untested. + """ + node, _, iceberg, _ = run_accepted( + export_cluster, + "bucket_str", + spark_ddl="CREATE TABLE {TABLE} (id BIGINT, name STRING)" + " USING iceberg PARTITIONED BY (bucket(8, name))" + " OPTIONS('format-version'='2')", + ch_schema="id Int64, name String", + rmt_columns="id Int64, name String", + rmt_partition_by="icebergBucket(8, name)", + # All rows share the same name → same Murmur3 bucket. + insert_values="(1, 'alice'), (2, 'alice'), (3, 'alice')", + ) + assert int(node.query(f"SELECT count() FROM {iceberg}").strip()) == 3 + + +def test_year_transform_timestamp(export_cluster): + """Spark years(ts) on TIMESTAMP <-> PARTITION BY toYearNumSinceEpoch(ts) (DateTime64(6)). + + DateTime64 → Avro 'long' is already handled by getAvroType(). The year transform on + TIMESTAMP follows a different branch than on DATE (long vs int in Avro). The existing + test_year_transform uses DATE only. All three rows fall within the same year. + """ + node, _, iceberg, _ = run_accepted( + export_cluster, + "year_ts", + spark_ddl="CREATE TABLE {TABLE} (id BIGINT, ts TIMESTAMP)" + " USING iceberg PARTITIONED BY (years(ts))" + " OPTIONS('format-version'='2')", + ch_schema="id Int64, ts DateTime64(6)", + rmt_columns="id Int64, ts DateTime64(6)", + rmt_partition_by="toYearNumSinceEpoch(ts)", + insert_values=( + "(1, '2023-01-15 08:00:00'), " + "(2, '2023-06-01 12:00:00'), " + "(3, '2023-12-31 23:59:59')" + ), + ) + assert int(node.query(f"SELECT count() FROM {iceberg}").strip()) == 3 + + +def test_month_transform_timestamp(export_cluster): + """Spark months(ts) on TIMESTAMP <-> PARTITION BY toMonthNumSinceEpoch(ts) (DateTime64(6)). + + Analogous to test_year_transform_timestamp but for the month transform. + All three rows fall within the same calendar month. + """ + node, _, iceberg, _ = run_accepted( + export_cluster, + "month_ts", + spark_ddl="CREATE TABLE {TABLE} (id BIGINT, ts TIMESTAMP)" + " USING iceberg PARTITIONED BY (months(ts))" + " OPTIONS('format-version'='2')", + ch_schema="id Int64, ts DateTime64(6)", + rmt_columns="id Int64, ts DateTime64(6)", + rmt_partition_by="toMonthNumSinceEpoch(ts)", + insert_values=( + "(1, '2023-06-01 00:00:00'), " + "(2, '2023-06-15 12:00:00'), " + "(3, '2023-06-30 23:59:59')" + ), + ) + assert int(node.query(f"SELECT count() FROM {iceberg}").strip()) == 3 + + +def test_day_transform_timestamp(export_cluster): + """Spark days(ts) on TIMESTAMP <-> PARTITION BY toRelativeDayNum(ts) (DateTime64(6)). + + Analogous to test_year_transform_timestamp but for the day transform. + All three rows fall within the same calendar day. + """ + node, _, iceberg, _ = run_accepted( + export_cluster, + "day_ts", + spark_ddl="CREATE TABLE {TABLE} (id BIGINT, ts TIMESTAMP)" + " USING iceberg PARTITIONED BY (days(ts))" + " OPTIONS('format-version'='2')", + ch_schema="id Int64, ts DateTime64(6)", + rmt_columns="id Int64, ts DateTime64(6)", + rmt_partition_by="toRelativeDayNum(ts)", + insert_values=( + "(1, '2023-06-15 00:00:00'), " + "(2, '2023-06-15 12:00:00'), " + "(3, '2023-06-15 23:59:59')" + ), + ) + assert int(node.query(f"SELECT count() FROM {iceberg}").strip()) == 3 + + +# --------------------------------------------------------------------------- +# Unhappy-path tests — BAD_ARGUMENTS must be raised synchronously +# --------------------------------------------------------------------------- + +def test_rejected_column_mismatch(export_cluster): + """Spark identity(year) — RMT PARTITION BY id: different column.""" + error = run_rejected( + export_cluster, + "rej_col_mismatch", + spark_ddl="CREATE TABLE {TABLE} (id BIGINT, year INT)" + " USING iceberg PARTITIONED BY (identity(year)) OPTIONS('format-version'='2')", + ch_schema="id Int64, year Int32", + rmt_columns="id Int64, year Int32", + rmt_partition_by="id", + insert_values="(1, 2024)", + ) + assert "BAD_ARGUMENTS" in error, f"Expected BAD_ARGUMENTS, got: {error!r}" + + +def test_rejected_transform_mismatch(export_cluster): + """Spark years(dt) — RMT PARTITION BY dt (identity, not year-transform).""" + error = run_rejected( + export_cluster, + "rej_xform_mismatch", + spark_ddl="CREATE TABLE {TABLE} (id BIGINT, dt DATE)" + " USING iceberg PARTITIONED BY (years(dt)) OPTIONS('format-version'='2')", + ch_schema="id Int64, dt Date", + rmt_columns="id Int64, dt Date", + rmt_partition_by="dt", + insert_values="(1, '2021-06-01')", + ) + assert "BAD_ARGUMENTS" in error, f"Expected BAD_ARGUMENTS, got: {error!r}" + + +def test_rejected_bucket_count_mismatch(export_cluster): + """Spark bucket(8, user_id) — RMT icebergBucket(16, user_id): wrong N.""" + error = run_rejected( + export_cluster, + "rej_bucket_n", + spark_ddl="CREATE TABLE {TABLE} (id BIGINT, user_id BIGINT)" + " USING iceberg PARTITIONED BY (bucket(8, user_id)) OPTIONS('format-version'='2')", + ch_schema="id Int64, user_id Int64", + rmt_columns="id Int64, user_id Int64", + rmt_partition_by="icebergBucket(16, user_id)", + insert_values="(1, 42)", + ) + assert "BAD_ARGUMENTS" in error, f"Expected BAD_ARGUMENTS, got: {error!r}" + + +def test_rejected_truncate_width_mismatch(export_cluster): + """Spark truncate(4, category) — RMT icebergTruncate(8, category): wrong width.""" + error = run_rejected( + export_cluster, + "rej_trunc_w", + spark_ddl="CREATE TABLE {TABLE} (id BIGINT, category STRING)" + " USING iceberg PARTITIONED BY (truncate(4, category)) OPTIONS('format-version'='2')", + ch_schema="id Int64, category String", + rmt_columns="id Int64, category String", + rmt_partition_by="icebergTruncate(8, category)", + insert_values="(1, 'clickhouse')", + ) + assert "BAD_ARGUMENTS" in error, f"Expected BAD_ARGUMENTS, got: {error!r}" + + +def test_rejected_field_count_mismatch(export_cluster): + """Spark 1-field identity(year) — RMT 2-field (year, region).""" + error = run_rejected( + export_cluster, + "rej_field_n", + spark_ddl="CREATE TABLE {TABLE} (id BIGINT, year INT, region STRING)" + " USING iceberg PARTITIONED BY (identity(year)) OPTIONS('format-version'='2')", + ch_schema="id Int64, year Int32, region String", + rmt_columns="id Int64, year Int32, region String", + rmt_partition_by="(year, region)", + insert_values="(1, 2024, 'EU')", + ) + assert "BAD_ARGUMENTS" in error, f"Expected BAD_ARGUMENTS, got: {error!r}" + + +def test_rejected_compound_order_reversed(export_cluster): + """Spark (identity(year), identity(region)) — RMT (region, year): reversed order.""" + error = run_rejected( + export_cluster, + "rej_compound_rev", + spark_ddl="CREATE TABLE {TABLE} (id BIGINT, year INT, region STRING)" + " USING iceberg PARTITIONED BY (identity(year), identity(region))" + " OPTIONS('format-version'='2')", + ch_schema="id Int64, year Int32, region String", + rmt_columns="id Int64, year Int32, region String", + rmt_partition_by="(region, year)", + insert_values="(1, 2024, 'EU')", + ) + assert "BAD_ARGUMENTS" in error, f"Expected BAD_ARGUMENTS, got: {error!r}" + + +def test_idempotency_after_commit_crash(export_cluster): + """ + Verify that an Iceberg export commit is idempotent when ClickHouse crashes (via + std::terminate() in a failpoint) after the Iceberg metadata is written but before + ZooKeeper is updated to COMPLETED. + Expected behaviour: + - The failpoint fires once: std::terminate() kills the process immediately after the + Iceberg commit; ZK task remains PENDING. + - ClickHouse is restarted. The scheduler picks up the PENDING task and retries the + commit. commitExportPartitionTransaction finds the transaction_id already present in + the Iceberg snapshot summary and skips re-committing. + - The task eventually reaches COMPLETED. + - The row count in the Iceberg table is exactly the number inserted (no duplicates). + """ + node = export_cluster.instances["node1"] + spark = export_cluster.spark_session + uid = unique_suffix() + source = f"rmt_{uid}" + iceberg = f"spark_{uid}" + spark_iceberg( + export_cluster, + spark, + iceberg, + f"CREATE TABLE {iceberg} (id BIGINT, year INT)" + f" USING iceberg PARTITIONED BY (identity(year)) OPTIONS('format-version'='2')", + ) + attach_ch_iceberg(node, iceberg, "id Int64, year Int32", export_cluster) + make_rmt(node, source, "id Int64, year Int32", "year") + node.query(f"INSERT INTO {source} VALUES (1, 2024), (2, 2024), (3, 2024)") + pid = first_partition_id(node, source) + # Enable the ONCE failpoint. When the background scheduler thread reaches the + # injection point (after a successful Iceberg commit), std::terminate() is called + # and the process exits immediately without setting ZK COMPLETED. + node.query("SYSTEM ENABLE FAILPOINT iceberg_export_after_commit_before_zk_completed") + node.query(f"ALTER TABLE {source} EXPORT PARTITION ID '{pid}' TO TABLE {iceberg}") + # the fail point will sleep for 10 seconds. Wait for 5 and then re-start clickhouse. + time.sleep(5) + # Restart ClickHouse. The ZK task is still PENDING; the scheduler will pick it up. + node.restart_clickhouse() + time.sleep(5) + # On restart the scheduler retries the commit. commitExportPartitionTransaction + # detects the transaction_id in the existing Iceberg snapshot summary and returns + # without re-writing any data, then sets ZK COMPLETED. + wait_for_export_status(node, source, iceberg, pid, timeout=60) + # Exactly 3 rows — no duplicates from the idempotent re-commit. + count = int(node.query(f"SELECT count() FROM {iceberg}").strip()) + assert count == 3, f"Expected 3 rows (no duplicates), got {count}" + + +def test_commit_attempts_budget_transitions_to_failed(export_cluster): + """ + Verify that the commit-attempts budget transitions a stuck task to FAILED + instead of leaving it in PENDING forever. + + Reproduction: + - Parts export successfully. + - A REGULAR failpoint (``export_partition_commit_always_throw``) makes every + ``ExportPartitionUtils::commit`` attempt throw before talking to Iceberg. + - ``ExportPartitionUtils::handleCommitFailure`` bumps ``/commit_attempts`` + on each failure and transitions ``/status`` to FAILED once the counter + reaches ``export_merge_tree_partition_max_retries``. + + Expected behaviour: + - The first attempt is made synchronously when the last part completes + (scheduler's ``handlePartExportSuccess``). + - Subsequent attempts come from the manifest-updating task's ``tryCleanup`` + path, polling every 30s. + - With max_retries=2, the task reaches FAILED within roughly one poll cycle. + - The ``commit_attempts`` znode reaches at least max_retries. + """ + node = export_cluster.instances["node1"] + spark = export_cluster.spark_session + + uid = unique_suffix() + source = f"rmt_commit_budget_{uid}" + iceberg = f"spark_commit_budget_{uid}" + + spark_iceberg( + export_cluster, + spark, + iceberg, + f"CREATE TABLE {iceberg} (id BIGINT, year INT)" + f" USING iceberg PARTITIONED BY (identity(year)) OPTIONS('format-version'='2')", + ) + attach_ch_iceberg(node, iceberg, "id Int64, year Int32", export_cluster) + make_rmt(node, source, "id Int64, year Int32", "year") + node.query(f"INSERT INTO {source} VALUES (1, 2024), (2, 2024), (3, 2024)") + pid = first_partition_id(node, source) + + # Force every commit attempt to throw. REGULAR failpoint fires on every hit, + # unlike ONCE which would only fire for the first call. + node.query("SYSTEM ENABLE FAILPOINT export_partition_commit_always_throw") + + # try block exists so we can add a finally that disables the failpoint + try: + # max_retries=2 bounds the test: one attempt from handlePartExportSuccess + # plus one from the manifest-updating task's next poll (~30s) is enough + # to exhaust the budget and flip the task to FAILED. + node.query( + f"ALTER TABLE {source} EXPORT PARTITION ID '{pid}' TO TABLE {iceberg}" + f" SETTINGS export_merge_tree_partition_max_retries = 2" + ) + + # Timeout must cover: at least one manifest-updating poll cycle (30s) + # plus slack for task scheduling and keeper RTT. + wait_for_export_status( + node, source, iceberg, pid, + expected_status="FAILED", + timeout=90, + ) + + # The commit_attempts znode must have reached (at least) max_retries — the + # counter is the direct mechanism that drove the FAILED transition. + # Locate the export's ZK root via the RMT's zookeeper_path and the + # partition_id_destination_db.destination_table export key convention. + export_key = f"{pid}_default.{iceberg}" + commit_attempts = int(node.query( + f"SELECT value FROM system.zookeeper" + f" WHERE path = '/clickhouse/tables/{source}/exports/{export_key}'" + f" AND name = 'commit_attempts'" + ).strip()) + assert commit_attempts >= 2, ( + f"Expected commit_attempts >= 2 (two commit attempts), got {commit_attempts}" + ) + finally: + node.query("SYSTEM DISABLE FAILPOINT export_partition_commit_always_throw") + + +# --------------------------------------------------------------------------- +# Replicated tests — IcebergS3, no catalog +# --------------------------------------------------------------------------- + + +def test_export_initiated_from_replica2(export_cluster): + """ + Export is initiated from replica2 (not the inserting replica). + Validates that any replica can start the export, not just the writer. + """ + uid = unique_suffix() + mt_table = f"rmt_from_replica2_{uid}" + iceberg_table = f"iceberg_from_replica2_{uid}" + + setup_replicas(export_cluster, mt_table, iceberg_table, ["replica1", "replica2"]) + + r1 = export_cluster.instances["replica1"] + r2 = export_cluster.instances["replica2"] + + r1.query(f"INSERT INTO {mt_table} VALUES (1, 2020), (2, 2020), (3, 2020)") + r2.query(f"SYSTEM SYNC REPLICA {mt_table}") + + r2.query(f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {iceberg_table}") + wait_for_export_status(r2, mt_table, iceberg_table, "2020") + + count_r1 = int(r1.query(f"SELECT count() FROM {iceberg_table}").strip()) + assert count_r1 == 3, f"Expected 3 rows from replica1, got {count_r1}" + count_r2 = int(r2.query(f"SELECT count() FROM {iceberg_table}").strip()) + assert count_r2 == 3, f"Expected 3 rows from replica2, got {count_r2}" + + +def test_concurrent_exports_different_partitions_across_replicas(export_cluster): + """ + Three replicas concurrently export distinct partitions (2020, 2021, 2022) to the + same IcebergS3 table. All three commits must succeed and the total row count must + equal the sum of all inserted rows. + """ + uid = unique_suffix() + mt_table = f"rmt_concurrent_diff_parts_{uid}" + iceberg_table = f"iceberg_concurrent_diff_parts_{uid}" + + setup_replicas( + export_cluster, mt_table, iceberg_table, + ["replica1", "replica2", "replica3"], + ) + + r1 = export_cluster.instances["replica1"] + r2 = export_cluster.instances["replica2"] + r3 = export_cluster.instances["replica3"] + + r1.query(f"INSERT INTO {mt_table} VALUES (1, 2020), (2, 2020), (3, 2020)") + r1.query(f"INSERT INTO {mt_table} VALUES (4, 2021), (5, 2021), (6, 2021)") + r1.query(f"INSERT INTO {mt_table} VALUES (7, 2022), (8, 2022), (9, 2022)") + r2.query(f"SYSTEM SYNC REPLICA {mt_table}") + r3.query(f"SYSTEM SYNC REPLICA {mt_table}") + + errors: list = [] + + def export_from(node, pid): + try: + node.query( + f"ALTER TABLE {mt_table} EXPORT PARTITION ID '{pid}' TO TABLE {iceberg_table}" + ) + wait_for_export_status(node, mt_table, iceberg_table, pid) + except Exception as exc: + errors.append(exc) + + threads = [ + threading.Thread(target=export_from, args=(r1, "2020")), + threading.Thread(target=export_from, args=(r2, "2021")), + threading.Thread(target=export_from, args=(r3, "2022")), + ] + for t in threads: + t.start() + for t in threads: + t.join() + + assert not errors, f"Export threads raised errors: {errors}" + + count = int(r1.query(f"SELECT count() FROM {iceberg_table}").strip()) + assert count == 9, f"Expected 9 rows total (3 per partition), got {count}" + + +def test_three_replica_concurrent_exports(export_cluster): + """ + ThreadPoolExecutor with 3 workers: each replica exports its own distinct partition. + All futures must complete successfully; total row count must be correct. + """ + uid = unique_suffix() + mt_table = f"rmt_three_replicas_concurrent_{uid}" + iceberg_table = f"iceberg_three_replicas_concurrent_{uid}" + + setup_replicas( + export_cluster, mt_table, iceberg_table, + ["replica1", "replica2", "replica3"], + ) + + r1 = export_cluster.instances["replica1"] + r2 = export_cluster.instances["replica2"] + r3 = export_cluster.instances["replica3"] + + r1.query(f"INSERT INTO {mt_table} VALUES (1, 2020), (2, 2020), (3, 2020)") + r1.query(f"INSERT INTO {mt_table} VALUES (4, 2021), (5, 2021), (6, 2021)") + r1.query(f"INSERT INTO {mt_table} VALUES (7, 2022), (8, 2022), (9, 2022)") + r2.query(f"SYSTEM SYNC REPLICA {mt_table}") + r3.query(f"SYSTEM SYNC REPLICA {mt_table}") + + def export_fn(node_pid): + node, pid = node_pid + node.query( + f"ALTER TABLE {mt_table} EXPORT PARTITION ID '{pid}' TO TABLE {iceberg_table}" + ) + wait_for_export_status(node, mt_table, iceberg_table, pid) + + with ThreadPoolExecutor(max_workers=3) as executor: + futures = [ + executor.submit(export_fn, (r1, "2020")), + executor.submit(export_fn, (r2, "2021")), + executor.submit(export_fn, (r3, "2022")), + ] + for fut in futures: + fut.result() + + count = int(r1.query(f"SELECT count() FROM {iceberg_table}").strip()) + assert count == 9, f"Expected 9 rows total (3 per partition), got {count}" diff --git a/tests/integration/test_storage_iceberg_with_spark/test_export_partition_iceberg_catalog.py b/tests/integration/test_storage_iceberg_with_spark/test_export_partition_iceberg_catalog.py new file mode 100644 index 000000000000..9d0cc557dc18 --- /dev/null +++ b/tests/integration/test_storage_iceberg_with_spark/test_export_partition_iceberg_catalog.py @@ -0,0 +1,532 @@ +""" +Tests for EXPORT PARTITION to a catalog-backed Iceberg table (Glue catalog via Moto). + +These tests verify that the catalog commit path (catalog->updateMetadata) is +exercised correctly for EXPORT PARTITION. A dedicated module-level cluster fixture +combines ZooKeeper (for ReplicatedMergeTree) with the Glue docker-compose stack +(Moto mock + MinIO warehouse bucket). + +Test coverage: + test_catalog_basic_export — single partition exported; catalog shows new snapshot + test_catalog_concurrent_export — two partitions exported in parallel; both commits succeed + test_catalog_idempotent_retry — crash after catalog commit; restart; no data duplication +""" + +import logging +import os +import threading +import time +import uuid + +import pytest +from pyiceberg.catalog import load_catalog +from pyiceberg.partitioning import PartitionField, PartitionSpec +from pyiceberg.schema import Schema +from pyiceberg.transforms import IdentityTransform +from pyiceberg.types import LongType, NestedField, StringType + +from helpers.cluster import ClickHouseCluster +from helpers.config_cluster import minio_access_key, minio_secret_key +from helpers.export_partition_helpers import ( + make_rmt, + wait_for_export_status, +) + + +GLUE_BASE_URL = "http://glue:3000" +GLUE_BASE_URL_LOCAL = "http://localhost:3000" +CH_CATALOG_DB = "glue_export_catalog" + + +# --------------------------------------------------------------------------- +# Cluster fixture +# --------------------------------------------------------------------------- + + +@pytest.fixture(scope="module") +def catalog_export_cluster(): + """ + Cluster with ZooKeeper (for ReplicatedMergeTree / EXPORT PARTITION) and the + Glue docker-compose stack (Moto mock + MinIO warehouse bucket). + Spark is not needed; pyiceberg handles table creation and catalog inspection. + replica1 and replica2 are additional nodes for replicated-export tests; they + share the same ZooKeeper, Glue, and MinIO containers as node1. + """ + try: + os.environ["AWS_ACCESS_KEY_ID"] = "testing" + os.environ["AWS_SECRET_ACCESS_KEY"] = "testing" + cluster = ClickHouseCluster(__file__) + for name in ["node1", "replica1", "replica2"]: + cluster.add_instance( + name, + main_configs=[ + "configs/config.d/allow_export_partition.xml", + ], + user_configs=[ + "configs/users.d/allow_export_partition.xml", + ], + stay_alive=True, + with_zookeeper=True, + keeper_required_feature_flags=["multi_read"], + with_glue_catalog=True, + ) + cluster.start() + + time.sleep(15) + + yield cluster + finally: + cluster.shutdown() + + +@pytest.fixture(autouse=True) +def cleanup_tables(catalog_export_cluster): + """Drop all default-DB tables on every node after each test.""" + yield + for node_name in ["node1", "replica1", "replica2"]: + node = catalog_export_cluster.instances[node_name] + try: + tables = node.query( + "SELECT name FROM system.tables WHERE database = 'default' FORMAT TabSeparated" + ).strip() + for tbl in tables.splitlines(): + tbl = tbl.strip() + if tbl: + node.query(f"DROP TABLE IF EXISTS default.`{tbl}` SYNC") + except Exception as exc: + logging.warning("cleanup_tables on %s: %s", node_name, exc) + + +# --------------------------------------------------------------------------- +# Helpers +# --------------------------------------------------------------------------- + + +def connect_catalog(cluster): + """ + Connect to the Moto Glue mock from the test host via localhost:3000. + MinIO is accessed via the container IP for S3 operations. + """ + minio_ip = cluster.get_instance_ip("minio") + return load_catalog( + "glue_test", + **{ + "type": "glue", + "glue.endpoint": GLUE_BASE_URL_LOCAL, + "glue.region": "us-east-1", + "s3.endpoint": f"http://{minio_ip}:9000", + "s3.access-key-id": minio_access_key, + "s3.secret-access-key": minio_secret_key, + }, + ) + + +def setup_ch_catalog_db(node, db_name: str = CH_CATALOG_DB) -> None: + """Drop-and-recreate the ClickHouse DataLakeCatalog database pointing at Glue (Moto).""" + node.query(f"DROP DATABASE IF EXISTS {db_name}") + node.query( + f""" + SET write_full_path_in_iceberg_metadata = 1; + SET allow_database_glue_catalog = 1; + CREATE DATABASE {db_name} + ENGINE = DataLakeCatalog('{GLUE_BASE_URL}', '{minio_access_key}', '{minio_secret_key}') + SETTINGS catalog_type = 'glue', + warehouse = 'test', + storage_endpoint = 'http://minio:9000/warehouse-glue', + region = 'us-east-1' + """ + ) + + +def create_catalog_rmt(node, name: str, replica_name: str = "r1") -> None: + """Create an identity(region)-partitioned ReplicatedMergeTree source table.""" + make_rmt(node, name, "id Int64, region String", "region", + replica_name=replica_name, order_by="id") + + +def partition_id_for(node, table: str, region: str) -> str: + return node.query( + f"SELECT DISTINCT partition_id FROM system.parts" + f" WHERE table = '{table}' AND active AND partition = '{region}'" + f" FORMAT TabSeparated" + ).strip() + + +def create_catalog_iceberg_table(catalog, ns: str, tbl: str) -> None: + """ + Create a simple identity(region)-partitioned Iceberg table in the catalog. + Using format-version 2 and uncompressed metadata for test simplicity. + """ + catalog.create_table( + identifier=f"{ns}.{tbl}", + schema=Schema( + NestedField(field_id=1, name="id", field_type=LongType(), required=True), + NestedField(field_id=2, name="region", field_type=StringType(), required=True), + ), + location=f"s3://warehouse-glue/data/{tbl}", + partition_spec=PartitionSpec( + PartitionField( + source_id=2, + field_id=1000, + transform=IdentityTransform(), + name="region", + ) + ), + properties={ + "write.metadata.compression-codec": "none", + "write.format.default": "parquet", + "format-version": "2", + }, + ) + + +# --------------------------------------------------------------------------- +# Replicated catalog helpers +# --------------------------------------------------------------------------- + + +def setup_catalog_replicas(cluster, source_table: str, replica_names: list) -> None: + """ + Create RMT on each named replica (each with its own replica_name so they share + the same ZK path) and set up the DataLakeCatalog database on every node. + No data is inserted here — callers manage their own test data. + """ + for rname in replica_names: + create_catalog_rmt(cluster.instances[rname], source_table, replica_name=rname) + setup_ch_catalog_db(cluster.instances[rname]) + + +# --------------------------------------------------------------------------- +# Tests +# --------------------------------------------------------------------------- + + +def test_catalog_basic_export(catalog_export_cluster): + """ + Create a catalog-registered Iceberg table via pyiceberg, export one partition + from a ReplicatedMergeTree, and verify: + - The catalog (Glue) shows a new snapshot after the export. + - SELECT via the DataLakeCatalog database returns the correct row count. + + This test exercises the catalog commit path: + IcebergMetadata::commitImportPartitionTransactionImpl + → catalog->updateMetadata(namespace, table, new_metadata_file, snapshot) + """ + node = catalog_export_cluster.instances["node1"] + catalog = connect_catalog(catalog_export_cluster) + + ns = f"ns_basic_{uuid.uuid4().hex[:8]}" + tbl = f"tbl_basic_{uuid.uuid4().hex[:8]}" + source = f"rmt_basic_{uuid.uuid4().hex[:8]}" + + catalog.create_namespace((ns,)) + create_catalog_iceberg_table(catalog, ns, tbl) + setup_ch_catalog_db(node) + create_catalog_rmt(node, source) + + node.query(f"INSERT INTO {source} VALUES (1, 'EU'), (2, 'EU'), (3, 'EU')") + + pid = partition_id_for(node, source, "EU") + dest_ch = f"`{CH_CATALOG_DB}`.`{ns}.{tbl}`" + + node.query( + f"ALTER TABLE {source} EXPORT PARTITION ID '{pid}' TO TABLE {dest_ch}", + settings={"write_full_path_in_iceberg_metadata": 1}, + ) + wait_for_export_status(node, source, None, pid) + + count = int(node.query(f"SELECT count() FROM {dest_ch}").strip()) + assert count == 3, f"Expected 3 rows, got {count}" + + iceberg_tbl = catalog.load_table(f"{ns}.{tbl}") + assert iceberg_tbl.current_snapshot() is not None, \ + "Expected at least one snapshot in Glue after the export" + + +def test_catalog_concurrent_export(catalog_export_cluster): + """ + Export two partitions concurrently to the same catalog-backed Iceberg table. + + Both commits go through catalog->updateMetadata (Glue). Both commits must + ultimately succeed. + + Verifies: + - Total row count equals total inserted (no rows lost). + - The catalog history contains at least two snapshots (one per partition). + """ + node = catalog_export_cluster.instances["node1"] + catalog = connect_catalog(catalog_export_cluster) + + ns = f"ns_concurrent_{uuid.uuid4().hex[:8]}" + tbl = f"tbl_concurrent_{uuid.uuid4().hex[:8]}" + source = f"rmt_concurrent_{uuid.uuid4().hex[:8]}" + + catalog.create_namespace((ns,)) + create_catalog_iceberg_table(catalog, ns, tbl) + setup_ch_catalog_db(node) + create_catalog_rmt(node, source) + + node.query(f"INSERT INTO {source} VALUES (1, 'EU'), (2, 'EU'), (3, 'EU')") + node.query(f"INSERT INTO {source} VALUES (4, 'US'), (5, 'US'), (6, 'US')") + + pid_eu = partition_id_for(node, source, "EU") + pid_us = partition_id_for(node, source, "US") + dest_ch = f"`{CH_CATALOG_DB}`.`{ns}.{tbl}`" + + errors: list = [] + + def export_partition(pid: str) -> None: + try: + node.query( + f"ALTER TABLE {source} EXPORT PARTITION ID '{pid}' TO TABLE {dest_ch}", + settings={"write_full_path_in_iceberg_metadata": 1}, + ) + wait_for_export_status(node, source, None, pid, timeout=120) + except Exception as exc: + errors.append(exc) + + t1 = threading.Thread(target=export_partition, args=(pid_eu,)) + t2 = threading.Thread(target=export_partition, args=(pid_us,)) + t1.start() + t2.start() + t1.join() + t2.join() + + assert not errors, f"Export threads raised errors: {errors}" + + count = int(node.query(f"SELECT count() FROM {dest_ch}").strip()) + assert count == 6, f"Expected 6 rows (3 EU + 3 US), got {count}" + + iceberg_tbl = catalog.load_table(f"{ns}.{tbl}") + history = iceberg_tbl.history() + assert len(history) >= 2, ( + f"Expected ≥2 snapshots (one per concurrent partition commit), got {len(history)}" + ) + + +def test_catalog_idempotent_retry(catalog_export_cluster): + """ + Simulate a crash after the catalog commit but before ZooKeeper is updated to + COMPLETED (via the iceberg_export_after_commit_before_zk_completed failpoint). + + After restart the scheduler retries the PENDING task. + IcebergMetadata::commitExportPartitionTransaction finds the transaction_id already + embedded in a snapshot summary field (clickhouse.export-partition-transaction-id) + and returns without re-committing. + + Verifies: + - Exactly 3 rows in the Iceberg table (no duplicates from the re-commit). + - Exactly 1 snapshot in the Glue catalog (the idempotent retry was a no-op). + """ + node = catalog_export_cluster.instances["node1"] + catalog = connect_catalog(catalog_export_cluster) + + ns = f"ns_idempotent_{uuid.uuid4().hex[:8]}" + tbl = f"tbl_idempotent_{uuid.uuid4().hex[:8]}" + source = f"rmt_idempotent_{uuid.uuid4().hex[:8]}" + + catalog.create_namespace((ns,)) + create_catalog_iceberg_table(catalog, ns, tbl) + setup_ch_catalog_db(node) + create_catalog_rmt(node, source) + + node.query(f"INSERT INTO {source} VALUES (1, 'EU'), (2, 'EU'), (3, 'EU')") + + pid = partition_id_for(node, source, "EU") + dest_ch = f"`{CH_CATALOG_DB}`.`{ns}.{tbl}`" + + # Enable the ONCE failpoint: after a successful catalog commit the process + # calls std::terminate() before writing ZK COMPLETED — simulating a hard crash. + node.query("SYSTEM ENABLE FAILPOINT iceberg_export_after_commit_before_zk_completed") + node.query( + f"ALTER TABLE {source} EXPORT PARTITION ID '{pid}' TO TABLE {dest_ch}", + settings={"write_full_path_in_iceberg_metadata": 1}, + ) + + # Give the background scheduler time to export the data files and reach the + # failpoint. The crash is immediate (std::terminate), so 10 s is generous. + time.sleep(10) + node.restart_clickhouse() + + # ClickHouse persists database metadata to disk so the DataLakeCatalog database + # survives the crash. Recreate it anyway to make the test self-contained. + setup_ch_catalog_db(node) + + # The scheduler picks up the PENDING task and retries. commitExportPartitionTransaction + # detects the transaction_id in the existing snapshot summary and skips the + # re-commit, then marks the task COMPLETED in ZooKeeper. + wait_for_export_status(node, source, None, pid, timeout=120) + + count = int(node.query(f"SELECT count() FROM {dest_ch}").strip()) + assert count == 3, f"Expected 3 rows (no duplicates from idempotent retry), got {count}" + + iceberg_tbl = catalog.load_table(f"{ns}.{tbl}") + history = iceberg_tbl.history() + assert len(history) == 1, ( + f"Expected exactly 1 snapshot (idempotent re-commit was a no-op), " + f"got {len(history)}" + ) + + +# --------------------------------------------------------------------------- +# Replicated catalog tests +# --------------------------------------------------------------------------- + + +def test_catalog_export_two_replicas_basic(catalog_export_cluster): + """ + End-to-end: export one partition from replica1 in a 2-replica setup. + Export is initiated on replica1; row count is verified from replica2 via + the DataLakeCatalog database to confirm the catalog commit was visible. + """ + catalog = connect_catalog(catalog_export_cluster) + + ns = f"ns_two_replicas_{uuid.uuid4().hex[:8]}" + tbl = f"tbl_two_replicas_{uuid.uuid4().hex[:8]}" + source = f"rmt_two_replicas_{uuid.uuid4().hex[:8]}" + + catalog.create_namespace((ns,)) + create_catalog_iceberg_table(catalog, ns, tbl) + + setup_catalog_replicas(catalog_export_cluster, source, ["replica1", "replica2"]) + + r1 = catalog_export_cluster.instances["replica1"] + r2 = catalog_export_cluster.instances["replica2"] + + r1.query(f"INSERT INTO {source} VALUES (1, 'EU'), (2, 'EU'), (3, 'EU')") + r2.query(f"SYSTEM SYNC REPLICA {source}") + + pid = partition_id_for(r1, source, "EU") + dest_ch = f"`{CH_CATALOG_DB}`.`{ns}.{tbl}`" + + r1.query( + f"ALTER TABLE {source} EXPORT PARTITION ID '{pid}' TO TABLE {dest_ch}", + settings={"write_full_path_in_iceberg_metadata": 1}, + ) + wait_for_export_status(r1, source, None, pid) + + iceberg_tbl = catalog.load_table(f"{ns}.{tbl}") + assert iceberg_tbl.current_snapshot() is not None, \ + "Expected at least one snapshot in Glue after export" + + count = int(r2.query(f"SELECT count() FROM {dest_ch}").strip()) + assert count == 3, f"Expected 3 rows from replica2 via catalog, got {count}" + + +def test_catalog_concurrent_export_from_different_replicas(catalog_export_cluster): + """ + Two replicas concurrently export different partitions (EU / US) to the same + catalog-backed Iceberg table. Both catalog commits must succeed; total row count + must equal 6 and Glue history must contain at least 2 snapshots. + """ + catalog = connect_catalog(catalog_export_cluster) + + ns = f"ns_conc_replicas_{uuid.uuid4().hex[:8]}" + tbl = f"tbl_conc_replicas_{uuid.uuid4().hex[:8]}" + source = f"rmt_conc_replicas_{uuid.uuid4().hex[:8]}" + + catalog.create_namespace((ns,)) + create_catalog_iceberg_table(catalog, ns, tbl) + + setup_catalog_replicas(catalog_export_cluster, source, ["replica1", "replica2"]) + + r1 = catalog_export_cluster.instances["replica1"] + r2 = catalog_export_cluster.instances["replica2"] + + r1.query(f"INSERT INTO {source} VALUES (1, 'EU'), (2, 'EU'), (3, 'EU')") + r1.query(f"INSERT INTO {source} VALUES (4, 'US'), (5, 'US'), (6, 'US')") + r2.query(f"SYSTEM SYNC REPLICA {source}") + + pid_eu = partition_id_for(r1, source, "EU") + pid_us = partition_id_for(r1, source, "US") + dest_ch = f"`{CH_CATALOG_DB}`.`{ns}.{tbl}`" + + errors: list = [] + + def export_partition(node, pid): + try: + node.query( + f"ALTER TABLE {source} EXPORT PARTITION ID '{pid}' TO TABLE {dest_ch}", + settings={"write_full_path_in_iceberg_metadata": 1}, + ) + wait_for_export_status(node, source, None, pid, timeout=120) + except Exception as exc: + errors.append(exc) + + t1 = threading.Thread(target=export_partition, args=(r1, pid_eu)) + t2 = threading.Thread(target=export_partition, args=(r2, pid_us)) + t1.start() + t2.start() + t1.join() + t2.join() + + assert not errors, f"Export threads raised errors: {errors}" + + count = int(r1.query(f"SELECT count() FROM {dest_ch}").strip()) + assert count == 6, f"Expected 6 rows (3 EU + 3 US), got {count}" + + iceberg_tbl = catalog.load_table(f"{ns}.{tbl}") + history = iceberg_tbl.history() + assert len(history) >= 2, ( + f"Expected ≥2 snapshots (one per concurrent partition commit), got {len(history)}" + ) + + +# TODO arthur fix: TOCTOU in export registration path. +# The exists() pre-check and the tryMulti() commit are not a single atomic ZK +# transaction. Depending on timing, the loser gets either KEEPER_EXCEPTION +# "Node exists" (both replicas race past exists() and collide at tryMulti) or +# BAD_ARGUMENTS "already exported" (the winner commits before the loser's +# exists() check). The test cannot reliably assert either error in isolation. +# def test_catalog_idempotent_same_partition_two_replicas(catalog_export_cluster): +# catalog = connect_catalog(catalog_export_cluster) +# +# ns = f"ns_{uuid.uuid4().hex[:8]}" +# tbl = f"tbl_{uuid.uuid4().hex[:8]}" +# source = f"rmt_{uuid.uuid4().hex[:8]}" +# +# catalog.create_namespace((ns,)) +# create_catalog_iceberg_table(catalog, ns, tbl) +# +# setup_catalog_replicas(catalog_export_cluster, source, ["replica1", "replica2"]) +# +# r1 = catalog_export_cluster.instances["replica1"] +# r2 = catalog_export_cluster.instances["replica2"] +# +# r1.query(f"INSERT INTO {source} VALUES (1, 'EU'), (2, 'EU'), (3, 'EU')") +# r2.query(f"SYSTEM SYNC REPLICA {source}") +# +# pid = partition_id_for(r1, source, "EU") +# dest_ch = f"`{CH_CATALOG_DB}`.`{ns}.{tbl}`" +# +# errors: list = [] +# +# def export_from(node): +# try: +# node.query( +# f"ALTER TABLE {source} EXPORT PARTITION ID '{pid}' TO TABLE {dest_ch}", +# settings={"write_full_path_in_iceberg_metadata": 1}, +# ) +# wait_for_export_status(node, source, None, pid, timeout=120) +# except Exception as exc: +# errors.append(exc) +# +# t1 = threading.Thread(target=export_from, args=(r1,)) +# t2 = threading.Thread(target=export_from, args=(r2,)) +# t1.start() +# t2.start() +# t1.join() +# t2.join() +# +# unexpected = [e for e in errors if "already exported" not in str(e)] +# assert not unexpected, f"Unexpected export errors: {unexpected}" +# +# count = int(r1.query(f"SELECT count() FROM {dest_ch}").strip()) +# assert count == 3, f"Expected 3 rows (no duplication), got {count}" +# +# iceberg_tbl = catalog.load_table(f"{ns}.{tbl}") +# history = iceberg_tbl.history() +# assert len(history) == 1, ( +# f"Expected exactly 1 snapshot (one winner, one rejected by export-key guard), " +# f"got {len(history)}" +# ) diff --git a/tests/queries/0_stateless/01271_show_privileges.reference b/tests/queries/0_stateless/01271_show_privileges.reference index cc318eca54d0..c4f4519dcdc2 100644 --- a/tests/queries/0_stateless/01271_show_privileges.reference +++ b/tests/queries/0_stateless/01271_show_privileges.reference @@ -44,6 +44,8 @@ ALTER MATERIALIZE TTL ['MATERIALIZE TTL'] TABLE ALTER TABLE ALTER REWRITE PARTS ['REWRITE PARTS'] TABLE ALTER TABLE ALTER SETTINGS ['ALTER SETTING','ALTER MODIFY SETTING','MODIFY SETTING','RESET SETTING'] TABLE ALTER TABLE ALTER MOVE PARTITION ['ALTER MOVE PART','MOVE PARTITION','MOVE PART'] TABLE ALTER TABLE +ALTER EXPORT PART ['ALTER EXPORT PART','EXPORT PART'] TABLE ALTER TABLE +ALTER EXPORT PARTITION ['ALTER EXPORT PARTITION','EXPORT PARTITION'] TABLE ALTER TABLE ALTER FETCH PARTITION ['ALTER FETCH PART','FETCH PARTITION'] TABLE ALTER TABLE ALTER FREEZE PARTITION ['FREEZE PARTITION','UNFREEZE'] TABLE ALTER TABLE ALTER UNLOCK SNAPSHOT ['UNLOCK SNAPSHOT'] TABLE ALTER TABLE @@ -147,6 +149,7 @@ SYSTEM DROP PAGE CACHE ['SYSTEM CLEAR PAGE CACHE','SYSTEM DROP PAGE CACHE','DROP SYSTEM DROP SCHEMA CACHE ['SYSTEM CLEAR SCHEMA CACHE','SYSTEM DROP SCHEMA CACHE','DROP SCHEMA CACHE'] GLOBAL SYSTEM DROP CACHE SYSTEM DROP FORMAT SCHEMA CACHE ['SYSTEM CLEAR FORMAT SCHEMA CACHE','SYSTEM DROP FORMAT SCHEMA CACHE','DROP FORMAT SCHEMA CACHE'] GLOBAL SYSTEM DROP CACHE SYSTEM DROP S3 CLIENT CACHE ['SYSTEM CLEAR S3 CLIENT CACHE','SYSTEM DROP S3 CLIENT','DROP S3 CLIENT CACHE'] GLOBAL SYSTEM DROP CACHE +SYSTEM DROP OBJECT STORAGE LIST OBJECTS CACHE ['SYSTEM DROP OBJECT STORAGE LIST OBJECTS CACHE'] GLOBAL SYSTEM DROP CACHE SYSTEM DROP CACHE ['DROP CACHE'] \N SYSTEM SYSTEM RELOAD CONFIG ['RELOAD CONFIG'] GLOBAL SYSTEM RELOAD SYSTEM RELOAD USERS ['RELOAD USERS'] GLOBAL SYSTEM RELOAD diff --git a/tests/queries/0_stateless/02221_system_zookeeper_unrestricted.reference b/tests/queries/0_stateless/02221_system_zookeeper_unrestricted.reference index bd379621b73d..2a6bc4c98833 100644 --- a/tests/queries/0_stateless/02221_system_zookeeper_unrestricted.reference +++ b/tests/queries/0_stateless/02221_system_zookeeper_unrestricted.reference @@ -20,6 +20,8 @@ creator_info creator_info deduplication_hashes deduplication_hashes +exports +exports failed_parts failed_parts flags diff --git a/tests/queries/0_stateless/02221_system_zookeeper_unrestricted_like.reference b/tests/queries/0_stateless/02221_system_zookeeper_unrestricted_like.reference index b1c76b04526e..db94343ec003 100644 --- a/tests/queries/0_stateless/02221_system_zookeeper_unrestricted_like.reference +++ b/tests/queries/0_stateless/02221_system_zookeeper_unrestricted_like.reference @@ -9,6 +9,7 @@ columns columns creator_info deduplication_hashes +exports failed_parts flags host @@ -51,6 +52,7 @@ columns columns creator_info deduplication_hashes +exports failed_parts flags host diff --git a/tests/queries/0_stateless/03377_object_storage_list_objects_cache.reference b/tests/queries/0_stateless/03377_object_storage_list_objects_cache.reference new file mode 100644 index 000000000000..76535ad25106 --- /dev/null +++ b/tests/queries/0_stateless/03377_object_storage_list_objects_cache.reference @@ -0,0 +1,103 @@ +-- { echoOn } + +-- The cached key should be `dir_`, and that includes all three files: 1, 2 and 3. Cache should return all three, but ClickHouse should filter out the third. +SELECT _path, * FROM s3(s3_conn, filename='dir_a/dir_b/t_03377_sample_{1..2}.parquet') order by id SETTINGS use_object_storage_list_objects_cache=1; +test/dir_a/dir_b/t_03377_sample_1.parquet 1 +test/dir_a/dir_b/t_03377_sample_2.parquet 2 +-- Make sure the filtering did not interfere with the cached values +SELECT _path, * FROM s3(s3_conn, filename='dir_a/dir_b/t_03377_sample_*.parquet') order by id SETTINGS use_object_storage_list_objects_cache=1; +test/dir_a/dir_b/t_03377_sample_1.parquet 1 +test/dir_a/dir_b/t_03377_sample_2.parquet 2 +test/dir_a/dir_b/t_03377_sample_3.parquet 3 +SYSTEM FLUSH LOGS; +SELECT ProfileEvents['ObjectStorageListObjectsCacheMisses'] > 0 as miss +FROM system.query_log +where log_comment = 'cold_list_cache' +AND type = 'QueryFinish' +ORDER BY event_time desc +LIMIT 1; +1 +SELECT ProfileEvents['ObjectStorageListObjectsCacheHits'] > 0 as hit +FROM system.query_log +where log_comment = 'warm_list_exact_cache' +AND type = 'QueryFinish' +ORDER BY event_time desc +LIMIT 1; +1 +SELECT ProfileEvents['ObjectStorageListObjectsCacheExactMatchHits'] > 0 as hit +FROM system.query_log +where log_comment = 'warm_list_exact_cache' +AND type = 'QueryFinish' +ORDER BY event_time desc +LIMIT 1; +1 +SELECT ProfileEvents['ObjectStorageListObjectsCachePrefixMatchHits'] > 0 as prefix_match_hit +FROM system.query_log +where log_comment = 'warm_list_exact_cache' +AND type = 'QueryFinish' +ORDER BY event_time desc +LIMIT 1; +0 +SELECT ProfileEvents['ObjectStorageListObjectsCacheHits'] > 0 as hit +FROM system.query_log +where log_comment = 'warm_list_prefix_match_cache' +AND type = 'QueryFinish' +ORDER BY event_time desc +LIMIT 1; +1 +SELECT ProfileEvents['ObjectStorageListObjectsCacheExactMatchHits'] > 0 as exact_match_hit +FROM system.query_log +where log_comment = 'warm_list_prefix_match_cache' +AND type = 'QueryFinish' +ORDER BY event_time desc +LIMIT 1; +0 +SELECT ProfileEvents['ObjectStorageListObjectsCachePrefixMatchHits'] > 0 as prefix_match_hit +FROM system.query_log +where log_comment = 'warm_list_prefix_match_cache' +AND type = 'QueryFinish' +ORDER BY event_time desc +LIMIT 1; +1 +SELECT ProfileEvents['ObjectStorageListObjectsCacheHits'] > 0 as hit +FROM system.query_log +where log_comment = 'even_shorter_prefix' +AND type = 'QueryFinish' +ORDER BY event_time desc +LIMIT 1; +0 +SELECT ProfileEvents['ObjectStorageListObjectsCacheMisses'] > 0 as miss +FROM system.query_log +where log_comment = 'even_shorter_prefix' +AND type = 'QueryFinish' +ORDER BY event_time desc +LIMIT 1; +1 +SELECT ProfileEvents['ObjectStorageListObjectsCacheHits'] > 0 as hit +FROM system.query_log +where log_comment = 'still_exact_match_after_shorter_prefix' +AND type = 'QueryFinish' +ORDER BY event_time desc +LIMIT 1; +1 +SELECT ProfileEvents['ObjectStorageListObjectsCacheExactMatchHits'] > 0 as exact_match_hit +FROM system.query_log +where log_comment = 'still_exact_match_after_shorter_prefix' +AND type = 'QueryFinish' +ORDER BY event_time desc +LIMIT 1; +1 +SELECT ProfileEvents['ObjectStorageListObjectsCacheHits'] > 0 as hit +FROM system.query_log +where log_comment = 'after_drop' +AND type = 'QueryFinish' +ORDER BY event_time desc +LIMIT 1; +0 +SELECT ProfileEvents['ObjectStorageListObjectsCacheMisses'] > 0 as miss +FROM system.query_log +where log_comment = 'after_drop' +AND type = 'QueryFinish' +ORDER BY event_time desc +LIMIT 1; +1 diff --git a/tests/queries/0_stateless/03377_object_storage_list_objects_cache.sql b/tests/queries/0_stateless/03377_object_storage_list_objects_cache.sql new file mode 100644 index 000000000000..9638faa88d23 --- /dev/null +++ b/tests/queries/0_stateless/03377_object_storage_list_objects_cache.sql @@ -0,0 +1,115 @@ +-- Tags: no-parallel, no-fasttest + +SYSTEM DROP OBJECT STORAGE LIST OBJECTS CACHE; + +INSERT INTO TABLE FUNCTION s3(s3_conn, filename='dir_a/dir_b/t_03377_sample_{_partition_id}.parquet', format='Parquet', structure='id UInt64') PARTITION BY id SETTINGS s3_truncate_on_insert=1 VALUES (1), (2), (3); + +SELECT * FROM s3(s3_conn, filename='dir_**.parquet') Format Null SETTINGS use_object_storage_list_objects_cache=1, log_comment='cold_list_cache'; +SELECT * FROM s3(s3_conn, filename='dir_**.parquet') Format Null SETTINGS use_object_storage_list_objects_cache=1, log_comment='warm_list_exact_cache'; +SELECT * FROM s3(s3_conn, filename='dir_a/dir_b**.parquet') Format Null SETTINGS use_object_storage_list_objects_cache=1, log_comment='warm_list_prefix_match_cache'; +SELECT * FROM s3(s3_conn, filename='dirr_**.parquet') Format Null SETTINGS use_object_storage_list_objects_cache=1, log_comment='warm_list_cache_miss'; -- { serverError CANNOT_EXTRACT_TABLE_STRUCTURE } +SELECT * FROM s3(s3_conn, filename='d**.parquet') Format Null SETTINGS use_object_storage_list_objects_cache=1, log_comment='even_shorter_prefix'; +SELECT * FROM s3(s3_conn, filename='dir_**.parquet') Format Null SETTINGS use_object_storage_list_objects_cache=1, log_comment='still_exact_match_after_shorter_prefix'; +SYSTEM DROP OBJECT STORAGE LIST OBJECTS CACHE; +SELECT * FROM s3(s3_conn, filename='dir_**.parquet') Format Null SETTINGS use_object_storage_list_objects_cache=1, log_comment='after_drop'; + +-- { echoOn } + +-- The cached key should be `dir_`, and that includes all three files: 1, 2 and 3. Cache should return all three, but ClickHouse should filter out the third. +SELECT _path, * FROM s3(s3_conn, filename='dir_a/dir_b/t_03377_sample_{1..2}.parquet') order by id SETTINGS use_object_storage_list_objects_cache=1; + +-- Make sure the filtering did not interfere with the cached values +SELECT _path, * FROM s3(s3_conn, filename='dir_a/dir_b/t_03377_sample_*.parquet') order by id SETTINGS use_object_storage_list_objects_cache=1; + +SYSTEM FLUSH LOGS; + +SELECT ProfileEvents['ObjectStorageListObjectsCacheMisses'] > 0 as miss +FROM system.query_log +where log_comment = 'cold_list_cache' +AND type = 'QueryFinish' +ORDER BY event_time desc +LIMIT 1; + +SELECT ProfileEvents['ObjectStorageListObjectsCacheHits'] > 0 as hit +FROM system.query_log +where log_comment = 'warm_list_exact_cache' +AND type = 'QueryFinish' +ORDER BY event_time desc +LIMIT 1; + +SELECT ProfileEvents['ObjectStorageListObjectsCacheExactMatchHits'] > 0 as hit +FROM system.query_log +where log_comment = 'warm_list_exact_cache' +AND type = 'QueryFinish' +ORDER BY event_time desc +LIMIT 1; + +SELECT ProfileEvents['ObjectStorageListObjectsCachePrefixMatchHits'] > 0 as prefix_match_hit +FROM system.query_log +where log_comment = 'warm_list_exact_cache' +AND type = 'QueryFinish' +ORDER BY event_time desc +LIMIT 1; + +SELECT ProfileEvents['ObjectStorageListObjectsCacheHits'] > 0 as hit +FROM system.query_log +where log_comment = 'warm_list_prefix_match_cache' +AND type = 'QueryFinish' +ORDER BY event_time desc +LIMIT 1; + +SELECT ProfileEvents['ObjectStorageListObjectsCacheExactMatchHits'] > 0 as exact_match_hit +FROM system.query_log +where log_comment = 'warm_list_prefix_match_cache' +AND type = 'QueryFinish' +ORDER BY event_time desc +LIMIT 1; + +SELECT ProfileEvents['ObjectStorageListObjectsCachePrefixMatchHits'] > 0 as prefix_match_hit +FROM system.query_log +where log_comment = 'warm_list_prefix_match_cache' +AND type = 'QueryFinish' +ORDER BY event_time desc +LIMIT 1; + +SELECT ProfileEvents['ObjectStorageListObjectsCacheHits'] > 0 as hit +FROM system.query_log +where log_comment = 'even_shorter_prefix' +AND type = 'QueryFinish' +ORDER BY event_time desc +LIMIT 1; + +SELECT ProfileEvents['ObjectStorageListObjectsCacheMisses'] > 0 as miss +FROM system.query_log +where log_comment = 'even_shorter_prefix' +AND type = 'QueryFinish' +ORDER BY event_time desc +LIMIT 1; + +SELECT ProfileEvents['ObjectStorageListObjectsCacheHits'] > 0 as hit +FROM system.query_log +where log_comment = 'still_exact_match_after_shorter_prefix' +AND type = 'QueryFinish' +ORDER BY event_time desc +LIMIT 1; + +SELECT ProfileEvents['ObjectStorageListObjectsCacheExactMatchHits'] > 0 as exact_match_hit +FROM system.query_log +where log_comment = 'still_exact_match_after_shorter_prefix' +AND type = 'QueryFinish' +ORDER BY event_time desc +LIMIT 1; + +SELECT ProfileEvents['ObjectStorageListObjectsCacheHits'] > 0 as hit +FROM system.query_log +where log_comment = 'after_drop' +AND type = 'QueryFinish' +ORDER BY event_time desc +LIMIT 1; + +SELECT ProfileEvents['ObjectStorageListObjectsCacheMisses'] > 0 as miss +FROM system.query_log +where log_comment = 'after_drop' +AND type = 'QueryFinish' +ORDER BY event_time desc +LIMIT 1; diff --git a/tests/queries/0_stateless/03413_experimental_settings_cannot_be_enabled_by_default.sql b/tests/queries/0_stateless/03413_experimental_settings_cannot_be_enabled_by_default.sql index 26a2c5f562d2..075cb9552fc3 100644 --- a/tests/queries/0_stateless/03413_experimental_settings_cannot_be_enabled_by_default.sql +++ b/tests/queries/0_stateless/03413_experimental_settings_cannot_be_enabled_by_default.sql @@ -4,5 +4,14 @@ -- However, some settings in the experimental tier are meant to control another experimental feature, and then they can be enabled as long as the feature itself is disabled. -- These are in the exceptions list inside NOT IN. +<<<<<<< HEAD SELECT name, value FROM system.settings WHERE tier = 'Experimental' AND type = 'Bool' AND value != '0' AND name NOT IN ('throw_on_unsupported_query_inside_transaction', 'ai_function_throw_on_error', 'ai_function_throw_on_quota_exceeded'); +======= +SELECT name, value FROM system.settings WHERE tier = 'Experimental' AND type = 'Bool' AND value != '0' AND name NOT IN ( + 'throw_on_unsupported_query_inside_transaction', +-- turned ON for Altinity Antalya builds specifically + 'allow_experimental_iceberg_read_optimization', + 'allow_experimental_export_merge_tree_part' +); +>>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) SELECT name, value FROM system.merge_tree_settings WHERE tier = 'Experimental' AND type = 'Bool' AND value != '0' AND name NOT IN ('remove_rolled_back_parts_immediately'); diff --git a/tests/queries/0_stateless/03572_export_merge_tree_part_basic.reference b/tests/queries/0_stateless/03572_export_merge_tree_part_basic.reference new file mode 100644 index 000000000000..8b08677b5d48 --- /dev/null +++ b/tests/queries/0_stateless/03572_export_merge_tree_part_basic.reference @@ -0,0 +1,35 @@ +---- Export 1: Export 2020_1_1_0 and 2021_2_2_0 +---- Export 2: Export 2022_3_3_0 and 2023_4_4_0 to wildcard table +---- Export 3: Export 2020_1_1_0 and 2021_2_2_0 to wildcard table with partition expression with function +---- Export 4: Export the same part again, it should be idempotent +---- Export 5: Export the same part again to wildcard, it should be idempotent +---- Verify Export 1: Both data parts should appear +1 2020 +2 2020 +3 2020 +4 2021 +---- Verify Export 4: Export the same part again, it should be idempotent +1 2020 +2 2020 +3 2020 +4 2021 +---- Verify: Data in roundtrip MergeTree table (should match s3_table) +1 2020 +2 2020 +3 2020 +4 2021 +---- Verify Export 2: Both data parts should appear (2022_3_3_0 and 2023_4_4_0) +5 2022 +6 2022 +7 2023 +8 2023 +---- Verify Export 5: Export the same part again, it should be idempotent +5 2022 +6 2022 +7 2023 +8 2023 +---- Verify Export 3: Both data parts should appear +1 2020 +2 2020 +3 2020 +4 2021 diff --git a/tests/queries/0_stateless/03572_export_merge_tree_part_basic.sh b/tests/queries/0_stateless/03572_export_merge_tree_part_basic.sh new file mode 100755 index 000000000000..fc5df9b541da --- /dev/null +++ b/tests/queries/0_stateless/03572_export_merge_tree_part_basic.sh @@ -0,0 +1,85 @@ +#!/usr/bin/env bash +# Tags: no-fasttest +# Tag no-fasttest: requires s3 storage + +CURDIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd) +# shellcheck source=../shell_config.sh +. "$CURDIR"/../shell_config.sh + +mt_table="mt_table_${RANDOM}" +mt_table_partition_expression_with_function="mt_table_partition_expression_with_function_${RANDOM}" +s3_table="s3_table_${RANDOM}" +s3_table_wildcard="s3_table_wildcard_${RANDOM}" +s3_table_wildcard_partition_expression_with_function="s3_table_wildcard_partition_expression_with_function_${RANDOM}" +mt_table_roundtrip="mt_table_roundtrip_${RANDOM}" + +query() { + $CLICKHOUSE_CLIENT --query "$1" +} + +query "DROP TABLE IF EXISTS $mt_table, $s3_table, $mt_table_roundtrip, $s3_table_wildcard, $s3_table_wildcard_partition_expression_with_function, $mt_table_partition_expression_with_function" + +# Create all tables +query "CREATE TABLE $mt_table (id UInt64, year UInt16) ENGINE = MergeTree() PARTITION BY year ORDER BY tuple()" +query "CREATE TABLE $s3_table (id UInt64, year UInt16) ENGINE = S3(s3_conn, filename='$s3_table', format=Parquet, partition_strategy='hive') PARTITION BY year" +query "CREATE TABLE $s3_table_wildcard (id UInt64, year UInt16) ENGINE = S3(s3_conn, filename='$s3_table_wildcard/{_partition_id}/{_file}.parquet', format=Parquet, partition_strategy='wildcard') PARTITION BY year" +query "CREATE TABLE $mt_table_partition_expression_with_function (id UInt64, year UInt16) ENGINE = MergeTree() PARTITION BY toString(year) ORDER BY tuple()" +query "CREATE TABLE $s3_table_wildcard_partition_expression_with_function (id UInt64, year UInt16) ENGINE = S3(s3_conn, filename='$s3_table_wildcard_partition_expression_with_function/{_partition_id}/{_file}.parquet', format=Parquet, partition_strategy='wildcard') PARTITION BY toString(year)" + +# Insert all data +query "INSERT INTO $mt_table VALUES (1, 2020), (2, 2020), (3, 2020), (4, 2021), (5, 2022), (6, 2022), (7, 2023), (8, 2023), (9, 2024), (10, 2024), (11, 2025), (12, 2025)" +query "INSERT INTO $mt_table_partition_expression_with_function VALUES (1, 2020), (2, 2020), (3, 2020), (4, 2021)" + +# ============================================================================ +# ALL EXPORTS HAPPEN HERE +# ============================================================================ + +echo "---- Export 1: Export 2020_1_1_0 and 2021_2_2_0" +query "ALTER TABLE $mt_table EXPORT PART '2020_1_1_0' TO TABLE $s3_table SETTINGS allow_experimental_export_merge_tree_part = 1" +query "ALTER TABLE $mt_table EXPORT PART '2021_2_2_0' TO TABLE $s3_table SETTINGS allow_experimental_export_merge_tree_part = 1" + +echo "---- Export 2: Export 2022_3_3_0 and 2023_4_4_0 to wildcard table" +query "ALTER TABLE $mt_table EXPORT PART '2022_3_3_0' TO TABLE $s3_table_wildcard SETTINGS allow_experimental_export_merge_tree_part = 1" +query "ALTER TABLE $mt_table EXPORT PART '2023_4_4_0' TO TABLE $s3_table_wildcard SETTINGS allow_experimental_export_merge_tree_part = 1" + +echo "---- Export 3: Export 2020_1_1_0 and 2021_2_2_0 to wildcard table with partition expression with function" +query "ALTER TABLE $mt_table_partition_expression_with_function EXPORT PART 'cb217c742dc7d143b61583011996a160_1_1_0' TO TABLE $s3_table_wildcard_partition_expression_with_function SETTINGS allow_experimental_export_merge_tree_part = 1" +query "ALTER TABLE $mt_table_partition_expression_with_function EXPORT PART '3be6d49ecf9749a383964bc6fab22d10_2_2_0' TO TABLE $s3_table_wildcard_partition_expression_with_function SETTINGS allow_experimental_export_merge_tree_part = 1" + +# below exports are using parts that were exported in export 1 and export 2, so we need to wait for them to complete +sleep 5 + +echo "---- Export 4: Export the same part again, it should be idempotent" +query "ALTER TABLE $mt_table EXPORT PART '2020_1_1_0' TO TABLE $s3_table SETTINGS allow_experimental_export_merge_tree_part = 1" + +echo "---- Export 5: Export the same part again to wildcard, it should be idempotent" +query "ALTER TABLE $mt_table EXPORT PART '2022_3_3_0' TO TABLE $s3_table_wildcard SETTINGS allow_experimental_export_merge_tree_part = 1" + +# ONE BIG SLEEP after all exports +sleep 15 + +# ============================================================================ +# ALL SELECTS/VERIFICATIONS HAPPEN HERE +# ============================================================================ + +echo "---- Verify Export 1: Both data parts should appear" +query "SELECT * FROM $s3_table ORDER BY id" + +echo "---- Verify Export 4: Export the same part again, it should be idempotent" +query "SELECT * FROM $s3_table ORDER BY id" + +query "CREATE TABLE $mt_table_roundtrip ENGINE = MergeTree() PARTITION BY year ORDER BY tuple() AS SELECT * FROM $s3_table" + +echo "---- Verify: Data in roundtrip MergeTree table (should match s3_table)" +query "SELECT * FROM $mt_table_roundtrip ORDER BY id" + +echo "---- Verify Export 2: Both data parts should appear (2022_3_3_0 and 2023_4_4_0)" +query "SELECT * FROM s3(s3_conn, filename='$s3_table_wildcard/**.parquet') ORDER BY id" + +echo "---- Verify Export 5: Export the same part again, it should be idempotent" +query "SELECT * FROM s3(s3_conn, filename='$s3_table_wildcard/**.parquet') ORDER BY id" + +echo "---- Verify Export 3: Both data parts should appear" +query "SELECT * FROM s3(s3_conn, filename='$s3_table_wildcard_partition_expression_with_function/**.parquet') ORDER BY id" + +query "DROP TABLE IF EXISTS $mt_table, $s3_table, $mt_table_roundtrip, $s3_table_wildcard, $s3_table_wildcard_partition_expression_with_function, $mt_table_partition_expression_with_function" diff --git a/tests/queries/0_stateless/03572_export_merge_tree_part_limits_and_table_functions.reference b/tests/queries/0_stateless/03572_export_merge_tree_part_limits_and_table_functions.reference new file mode 100644 index 000000000000..b7f1f4411bf6 --- /dev/null +++ b/tests/queries/0_stateless/03572_export_merge_tree_part_limits_and_table_functions.reference @@ -0,0 +1,20 @@ +---- Test max_bytes and max_rows per file +---- Table function with schema inheritance (no schema specified) +---- Table function with explicit compatible schema +Waiting for exports to complete (timeout: 60s)... +All exports completed. +---- Count files in big_destination_max_bytes, should be 5 (4 parquet, 1 commit) +5 +---- Count rows in big_table and big_destination_max_bytes +4194304 +4194304 +---- Count files in big_destination_max_rows, should be 5 (4 parquet, 1 commit) +5 +---- Count rows in big_table and big_destination_max_rows +4194304 +4194304 +---- Data should be exported with inherited schema +100 test1 2022 +101 test2 2022 +---- Data should be exported with explicit schema +102 test3 2023 diff --git a/tests/queries/0_stateless/03572_export_merge_tree_part_limits_and_table_functions.sh b/tests/queries/0_stateless/03572_export_merge_tree_part_limits_and_table_functions.sh new file mode 100755 index 000000000000..dff7332662d0 --- /dev/null +++ b/tests/queries/0_stateless/03572_export_merge_tree_part_limits_and_table_functions.sh @@ -0,0 +1,132 @@ +#!/usr/bin/env bash +# Tags: no-fasttest +# Tag no-fasttest: requires s3 storage + +CURDIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd) +# shellcheck source=../shell_config.sh +. "$CURDIR"/../shell_config.sh + +big_table="big_table_${RANDOM}" +big_destination_max_bytes="big_destination_max_bytes_${RANDOM}" +big_destination_max_rows="big_destination_max_rows_${RANDOM}" +tf_schema_inherit="tf_schema_inherit_${RANDOM}" +tf_schema_explicit="tf_schema_explicit_${RANDOM}" +mt_table_tf="mt_table_tf_${RANDOM}" + +query() { + local query_text="$1" + local query_id="$2" + + if [ -n "$query_id" ]; then + $CLICKHOUSE_CLIENT --query_id="$query_id" --query "$query_text" + else + $CLICKHOUSE_CLIENT --query "$query_text" + fi +} + +query "DROP TABLE IF EXISTS $big_table, $big_destination_max_bytes, $big_destination_max_rows, $mt_table_tf" + +echo "---- Test max_bytes and max_rows per file" + +# Create all tables +query "CREATE TABLE $big_table (id UInt64, data String, year UInt16) Engine=MergeTree() order by id partition by year" +query "CREATE TABLE $big_destination_max_bytes(id UInt64, data String, year UInt16) engine=S3(s3_conn, filename='$big_destination_max_bytes', partition_strategy='hive', format=Parquet) partition by year" +query "CREATE TABLE $big_destination_max_rows(id UInt64, data String, year UInt16) engine=S3(s3_conn, filename='$big_destination_max_rows', partition_strategy='hive', format=Parquet) partition by year" +query "CREATE TABLE $mt_table_tf (id UInt64, value String, year UInt16) ENGINE = MergeTree() PARTITION BY year ORDER BY tuple()" + +# Insert all data +# 4194304 is a number that came up during multiple iterations, it does not really mean anything (aside from the fact that the below numbers depend on it) +query "INSERT INTO $big_table SELECT number AS id, repeat('x', 100) AS data, 2025 AS year FROM numbers(4194304)" +query "INSERT INTO $big_table SELECT number AS id, repeat('x', 100) AS data, 2026 AS year FROM numbers(4194304)" +query "INSERT INTO $mt_table_tf VALUES (100, 'test1', 2022), (101, 'test2', 2022), (102, 'test3', 2023)" + +# make sure we have only one part +query "OPTIMIZE TABLE $big_table FINAL" + +# Get part names +big_part_max_bytes=$(query "SELECT name FROM system.parts WHERE database = currentDatabase() AND table = '$big_table' AND partition_id = '2025' AND active = 1 ORDER BY name LIMIT 1" | tr -d '\n') +big_part_max_rows=$(query "SELECT name FROM system.parts WHERE database = currentDatabase() AND table = '$big_table' AND partition_id = '2026' AND active = 1 ORDER BY name LIMIT 1" | tr -d '\n') + +# ============================================================================ +# ALL EXPORTS HAPPEN HERE +# ============================================================================ + +# Generate unique query_ids for each export to track them in part_log +export_query_id_1="export_${RANDOM}_1" +export_query_id_2="export_${RANDOM}_2" +export_query_id_3="export_${RANDOM}_3" +export_query_id_4="export_${RANDOM}_4" + +# this should generate ~4 files +query "ALTER TABLE $big_table EXPORT PART '$big_part_max_bytes' TO TABLE $big_destination_max_bytes SETTINGS allow_experimental_export_merge_tree_part = 1, export_merge_tree_part_max_bytes_per_file=3500000, output_format_parquet_row_group_size_bytes=1000000" "$export_query_id_1" +# export_merge_tree_part_max_rows_per_file = 1048576 (which is 4194304/4) to generate 4 files +query "ALTER TABLE $big_table EXPORT PART '$big_part_max_rows' TO TABLE $big_destination_max_rows SETTINGS allow_experimental_export_merge_tree_part = 1, export_merge_tree_part_max_rows_per_file=1048576" "$export_query_id_2" + +echo "---- Table function with schema inheritance (no schema specified)" +query "ALTER TABLE $mt_table_tf EXPORT PART '2022_1_1_0' TO TABLE FUNCTION s3(s3_conn, filename='$tf_schema_inherit', format='Parquet', partition_strategy='hive') PARTITION BY year SETTINGS allow_experimental_export_merge_tree_part = 1" "$export_query_id_3" + +echo "---- Table function with explicit compatible schema" +query "ALTER TABLE $mt_table_tf EXPORT PART '2023_2_2_0' TO TABLE FUNCTION s3(s3_conn, filename='$tf_schema_explicit', format='Parquet', structure='id UInt64, value String, year UInt16', partition_strategy='hive') PARTITION BY year SETTINGS allow_experimental_export_merge_tree_part = 1" "$export_query_id_4" + +# Wait for all exports to complete +wait_for_exports() { + local timeout=${1:-60} + local poll_interval=${2:-0.5} + local start_time=$(date +%s) + local elapsed=0 + + echo "Waiting for exports to complete (timeout: ${timeout}s)..." + + while [ $elapsed -lt $timeout ]; do + # Flush logs to ensure part_log entries are visible + query "SYSTEM FLUSH LOGS" > /dev/null 2>&1 || true + + # Wait for part_log entries - these are written synchronously when export completes + # Check if all expected exports have corresponding part_log entries by query_id + local completed_count=$(query "SELECT count() FROM system.part_log WHERE event_type = 'ExportPart' AND query_id IN ('$export_query_id_1', '$export_query_id_2', '$export_query_id_3', '$export_query_id_4')" | tr -d '\n') + + if [ "$completed_count" = "4" ]; then + echo "All exports completed." + return 0 + fi + + sleep $poll_interval + elapsed=$(($(date +%s) - start_time)) + done + + echo "Timeout waiting for exports to complete after ${timeout}s" + query "SYSTEM FLUSH LOGS" > /dev/null 2>&1 || true + echo "Completed exports in part_log:" + query "SELECT query_id, table, part_name, event_time FROM system.part_log WHERE event_type = 'ExportPart' AND query_id IN ('$export_query_id_1', '$export_query_id_2', '$export_query_id_3', '$export_query_id_4')" + echo "Remaining exports in system.exports:" + query "SELECT source_table, part_name, elapsed, rows_read, total_rows_to_read FROM system.exports WHERE ((source_table = '$big_table' AND part_name IN ('$big_part_max_bytes', '$big_part_max_rows')) OR (source_table = '$mt_table_tf' AND part_name IN ('2022_1_1_0', '2023_2_2_0')))" + return 1 +} + +wait_for_exports 60 + +# ============================================================================ +# ALL SELECTS/VERIFICATIONS HAPPEN HERE +# ============================================================================ + +echo "---- Count files in big_destination_max_bytes, should be 5 (4 parquet, 1 commit)" +query "SELECT count(_file) FROM s3(s3_conn, filename='$big_destination_max_bytes/**', format='One')" + +echo "---- Count rows in big_table and big_destination_max_bytes" +query "SELECT COUNT() from $big_table WHERE year = 2025" +query "SELECT COUNT() from $big_destination_max_bytes" + +echo "---- Count files in big_destination_max_rows, should be 5 (4 parquet, 1 commit)" +query "SELECT count(_file) FROM s3(s3_conn, filename='$big_destination_max_rows/**', format='One')" + +echo "---- Count rows in big_table and big_destination_max_rows" +query "SELECT COUNT() from $big_table WHERE year = 2026" +query "SELECT COUNT() from $big_destination_max_rows" + +echo "---- Data should be exported with inherited schema" +query "SELECT * FROM s3(s3_conn, filename='$tf_schema_inherit/**.parquet') ORDER BY id" + +echo "---- Data should be exported with explicit schema" +query "SELECT * FROM s3(s3_conn, filename='$tf_schema_explicit/**.parquet') ORDER BY id" + +query "DROP TABLE IF EXISTS $big_table, $big_destination_max_bytes, $big_destination_max_rows, $mt_table_tf" diff --git a/tests/queries/0_stateless/03572_export_merge_tree_part_special_columns.reference b/tests/queries/0_stateless/03572_export_merge_tree_part_special_columns.reference new file mode 100644 index 000000000000..14bcbb452591 --- /dev/null +++ b/tests/queries/0_stateless/03572_export_merge_tree_part_special_columns.reference @@ -0,0 +1,39 @@ +---- Test ALIAS columns export +---- Test MATERIALIZED columns export +---- Test EPHEMERAL column (not stored, ignored during export) +---- Test Mixed ALIAS, MATERIALIZED, and EPHEMERAL in same table +---- Test Complex Expressions in computed columns +---- Test Export to Table Function with mixed columns +---- Verify ALIAS column data in source table (arr_1 computed from arr[1]) +1 [1,2,3] 1 +1 [10,20,30] 10 +---- Verify ALIAS column data exported to S3 (should match source) +1 [1,2,3] 1 +1 [10,20,30] 10 +---- Verify MATERIALIZED column data in source table (arr_1 computed from arr[1]) +1 [1,2,3] 1 +1 [10,20,30] 10 +---- Verify MATERIALIZED column data exported to S3 (should match source) +1 [1,2,3] 1 +1 [10,20,30] 10 +---- Verify data in source +1 ALICE +1 BOB +---- Verify exported data +1 ALICE +1 BOB +---- Verify mixed columns in source table +1 5 10 15 TEST +1 10 20 30 PROD +2 15 30 45 DEV +---- Verify mixed columns exported to S3 +1 5 10 15 TEST +1 10 20 30 PROD +---- Verify mixed columns exported to S3 +2 15 30 45 DEV +---- Verify complex expressions in source table +1 alice ALICE alice-1 +1 bob BOB bob-1 +---- Verify complex expressions exported to S3 (should match source) +1 alice ALICE alice-1 +1 bob BOB bob-1 diff --git a/tests/queries/0_stateless/03572_export_merge_tree_part_special_columns.sh b/tests/queries/0_stateless/03572_export_merge_tree_part_special_columns.sh new file mode 100755 index 000000000000..0164dd70c4e0 --- /dev/null +++ b/tests/queries/0_stateless/03572_export_merge_tree_part_special_columns.sh @@ -0,0 +1,154 @@ +#!/usr/bin/env bash +# Tags: no-fasttest +# Tag no-fasttest: requires s3 storage + +CURDIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd) +# shellcheck source=../shell_config.sh +. "$CURDIR"/../shell_config.sh + +mt_alias="mt_alias_${RANDOM}" +mt_materialized="mt_materialized_${RANDOM}" +s3_alias_export="s3_alias_export_${RANDOM}" +s3_materialized_export="s3_materialized_export_${RANDOM}" +mt_mixed="mt_mixed_${RANDOM}" +s3_mixed_export="s3_mixed_export_${RANDOM}" +mt_complex_expr="mt_complex_expr_${RANDOM}" +s3_complex_expr_export="s3_complex_expr_export_${RANDOM}" +mt_ephemeral="mt_ephemeral_${RANDOM}" +s3_ephemeral_export="s3_ephemeral_export_${RANDOM}" +s3_mixed_export_table_function="s3_mixed_export_table_function_${RANDOM}" + +query() { + $CLICKHOUSE_CLIENT --query "$1" +} + +query "DROP TABLE IF EXISTS $mt_alias, $mt_materialized, $s3_alias_export, $s3_materialized_export, $mt_mixed, $s3_mixed_export, $mt_complex_expr, $s3_complex_expr_export, $mt_ephemeral, $s3_ephemeral_export" + +# Create all tables +echo "---- Test ALIAS columns export" +query "CREATE TABLE $mt_alias (a UInt32, arr Array(UInt64), arr_1 UInt64 ALIAS arr[1]) ENGINE = MergeTree() PARTITION BY a ORDER BY (a, arr[1]) SETTINGS index_granularity = 1" +query "CREATE TABLE $s3_alias_export (a UInt32, arr Array(UInt64), arr_1 UInt64) ENGINE = S3(s3_conn, filename='$s3_alias_export', format=Parquet, partition_strategy='hive') PARTITION BY a" + +echo "---- Test MATERIALIZED columns export" +query "CREATE TABLE $mt_materialized (a UInt32, arr Array(UInt64), arr_1 UInt64 MATERIALIZED arr[1]) ENGINE = MergeTree() PARTITION BY a ORDER BY (a, arr_1) SETTINGS index_granularity = 1" +query "CREATE TABLE $s3_materialized_export (a UInt32, arr Array(UInt64), arr_1 UInt64) ENGINE = S3(s3_conn, filename='$s3_materialized_export', format=Parquet, partition_strategy='hive') PARTITION BY a" + +echo "---- Test EPHEMERAL column (not stored, ignored during export)" +query "CREATE TABLE $mt_ephemeral ( + id UInt32, + name_input String EPHEMERAL, + name_upper String DEFAULT upper(name_input) +) ENGINE = MergeTree() PARTITION BY id ORDER BY id SETTINGS index_granularity = 1" + +query "CREATE TABLE $s3_ephemeral_export ( + id UInt32, + name_upper String +) ENGINE = S3(s3_conn, filename='$s3_ephemeral_export', format=Parquet, partition_strategy='hive') PARTITION BY id" + +echo "---- Test Mixed ALIAS, MATERIALIZED, and EPHEMERAL in same table" +query "CREATE TABLE $mt_mixed ( + id UInt32, + value UInt32, + tag_input String EPHEMERAL, + doubled UInt64 ALIAS value * 2, + tripled UInt64 MATERIALIZED value * 3, + tag String DEFAULT upper(tag_input) +) ENGINE = MergeTree() PARTITION BY id ORDER BY id SETTINGS index_granularity = 1" + +query "CREATE TABLE $s3_mixed_export ( + id UInt32, + value UInt32, + doubled UInt64, + tripled UInt64, + tag String +) ENGINE = S3(s3_conn, filename='$s3_mixed_export', format=Parquet, partition_strategy='hive') PARTITION BY id" + +echo "---- Test Complex Expressions in computed columns" +query "CREATE TABLE $mt_complex_expr ( + id UInt32, + name String, + upper_name String ALIAS upper(name), + concat_result String MATERIALIZED concat(name, '-', toString(id)) +) ENGINE = MergeTree() PARTITION BY id ORDER BY id SETTINGS index_granularity = 1" + +query "CREATE TABLE $s3_complex_expr_export ( + id UInt32, + name String, + upper_name String, + concat_result String +) ENGINE = S3(s3_conn, filename='$s3_complex_expr_export', format=Parquet, partition_strategy='hive') PARTITION BY id" + +# Insert all data +query "INSERT INTO $mt_alias VALUES (1, [1, 2, 3]), (1, [10, 20, 30])" +query "INSERT INTO $mt_materialized VALUES (1, [1, 2, 3]), (1, [10, 20, 30])" +query "INSERT INTO $mt_ephemeral (id, name_input) VALUES (1, 'alice'), (1, 'bob')" +query "INSERT INTO $mt_mixed (id, value, tag_input) VALUES (1, 5, 'test'), (1, 10, 'prod')" +query "INSERT INTO $mt_mixed (id, value, tag_input) VALUES (2, 15, 'dev')" +query "INSERT INTO $mt_complex_expr (id, name) VALUES (1, 'alice'), (1, 'bob')" + +# Get all part names +alias_part=$(query "SELECT name FROM system.parts WHERE database = currentDatabase() AND table = '$mt_alias' AND partition_id = '1' AND active = 1 ORDER BY name LIMIT 1" | tr -d '\n') +materialized_part=$(query "SELECT name FROM system.parts WHERE database = currentDatabase() AND table = '$mt_materialized' AND partition_id = '1' AND active = 1 ORDER BY name LIMIT 1" | tr -d '\n') +ephemeral_part=$(query "SELECT name FROM system.parts WHERE database = currentDatabase() AND table = '$mt_ephemeral' AND partition_id = '1' AND active = 1 ORDER BY name LIMIT 1" | tr -d '\n') +mixed_part=$(query "SELECT name FROM system.parts WHERE database = currentDatabase() AND table = '$mt_mixed' AND partition_id = '1' AND active = 1 ORDER BY name LIMIT 1" | tr -d '\n') +mixed_part_2=$(query "SELECT name FROM system.parts WHERE database = currentDatabase() AND table = '$mt_mixed' AND partition_id = '2' AND active = 1 ORDER BY name LIMIT 1" | tr -d '\n') +complex_expr_part=$(query "SELECT name FROM system.parts WHERE database = currentDatabase() AND table = '$mt_complex_expr' AND partition_id = '1' AND active = 1 ORDER BY name LIMIT 1" | tr -d '\n') + +# ============================================================================ +# ALL EXPORTS HAPPEN HERE +# ============================================================================ + +query "ALTER TABLE $mt_alias EXPORT PART '$alias_part' TO TABLE $s3_alias_export SETTINGS allow_experimental_export_merge_tree_part = 1" + +query "ALTER TABLE $mt_materialized EXPORT PART '$materialized_part' TO TABLE $s3_materialized_export SETTINGS allow_experimental_export_merge_tree_part = 1" + +query "ALTER TABLE $mt_ephemeral EXPORT PART '$ephemeral_part' TO TABLE $s3_ephemeral_export SETTINGS allow_experimental_export_merge_tree_part = 1" + +query "ALTER TABLE $mt_mixed EXPORT PART '$mixed_part' TO TABLE $s3_mixed_export SETTINGS allow_experimental_export_merge_tree_part = 1" + +echo "---- Test Export to Table Function with mixed columns" +query "ALTER TABLE $mt_mixed EXPORT PART '$mixed_part_2' TO TABLE FUNCTION s3(s3_conn, filename='$s3_mixed_export_table_function', format=Parquet, partition_strategy='hive') PARTITION BY id SETTINGS allow_experimental_export_merge_tree_part = 1" + +query "ALTER TABLE $mt_complex_expr EXPORT PART '$complex_expr_part' TO TABLE $s3_complex_expr_export SETTINGS allow_experimental_export_merge_tree_part = 1" + +# ONE BIG SLEEP after all exports +sleep 20 + +# ============================================================================ +# ALL SELECTS/VERIFICATIONS HAPPEN HERE +# ============================================================================ + +echo "---- Verify ALIAS column data in source table (arr_1 computed from arr[1])" +query "SELECT a, arr, arr_1 FROM $mt_alias ORDER BY arr" + +echo "---- Verify ALIAS column data exported to S3 (should match source)" +query "SELECT a, arr, arr_1 FROM $s3_alias_export ORDER BY arr" + +echo "---- Verify MATERIALIZED column data in source table (arr_1 computed from arr[1])" +query "SELECT a, arr, arr_1 FROM $mt_materialized ORDER BY arr" + +echo "---- Verify MATERIALIZED column data exported to S3 (should match source)" +query "SELECT a, arr, arr_1 FROM $s3_materialized_export ORDER BY arr" + +echo "---- Verify data in source" +query "SELECT id, name_upper FROM $mt_ephemeral ORDER BY name_upper" + +echo "---- Verify exported data" +query "SELECT id, name_upper FROM $s3_ephemeral_export ORDER BY name_upper" + +echo "---- Verify mixed columns in source table" +query "SELECT id, value, doubled, tripled, tag FROM $mt_mixed ORDER BY value" + +echo "---- Verify mixed columns exported to S3" +query "SELECT id, value, doubled, tripled, tag FROM $s3_mixed_export ORDER BY value" + +echo "---- Verify mixed columns exported to S3" +query "SELECT * FROM s3(s3_conn, filename='$s3_mixed_export_table_function/**.parquet', format=Parquet) ORDER BY value" + +echo "---- Verify complex expressions in source table" +query "SELECT id, name, upper_name, concat_result FROM $mt_complex_expr ORDER BY name" + +echo "---- Verify complex expressions exported to S3 (should match source)" +query "SELECT id, name, upper_name, concat_result FROM $s3_complex_expr_export ORDER BY name" + +query "DROP TABLE IF EXISTS $mt_alias, $mt_materialized, $s3_alias_export, $s3_materialized_export, $mt_ephemeral, $s3_ephemeral_export, $mt_mixed, $s3_mixed_export, $mt_complex_expr, $s3_complex_expr_export" diff --git a/tests/queries/0_stateless/03572_export_merge_tree_part_to_object_storage_simple.reference b/tests/queries/0_stateless/03572_export_merge_tree_part_to_object_storage_simple.reference new file mode 100644 index 000000000000..e69de29bb2d1 diff --git a/tests/queries/0_stateless/03572_export_merge_tree_part_to_object_storage_simple.sql b/tests/queries/0_stateless/03572_export_merge_tree_part_to_object_storage_simple.sql new file mode 100644 index 000000000000..7cb70af024a2 --- /dev/null +++ b/tests/queries/0_stateless/03572_export_merge_tree_part_to_object_storage_simple.sql @@ -0,0 +1,39 @@ +-- Tags: no-parallel, no-fasttest + +DROP TABLE IF EXISTS 03572_mt_table, 03572_invalid_schema_table, 03572_ephemeral_mt_table, 03572_matching_ephemeral_s3_table; + +SET allow_experimental_export_merge_tree_part=1; + +CREATE TABLE 03572_mt_table (id UInt64, year UInt16) ENGINE = MergeTree() PARTITION BY year ORDER BY tuple(); + +INSERT INTO 03572_mt_table VALUES (1, 2020); + +-- Create a table with a different partition key and export a partition to it. It should throw +CREATE TABLE 03572_invalid_schema_table (id UInt64, x UInt16) ENGINE = S3(s3_conn, filename='03572_invalid_schema_table', format='Parquet', partition_strategy='hive') PARTITION BY x; + +ALTER TABLE 03572_mt_table EXPORT PART '2020_1_1_0' TO TABLE 03572_invalid_schema_table +SETTINGS allow_experimental_export_merge_tree_part = 1; -- {serverError INCOMPATIBLE_COLUMNS} + +DROP TABLE 03572_invalid_schema_table; + +-- The only partition strategy that supports exports is hive. Wildcard should throw +CREATE TABLE 03572_invalid_schema_table (id UInt64, year UInt16) ENGINE = S3(s3_conn, filename='03572_invalid_schema_table/{_partition_id}', format='Parquet', partition_strategy='wildcard') PARTITION BY (id, year); + +ALTER TABLE 03572_mt_table EXPORT PART '2020_1_1_0' TO TABLE 03572_invalid_schema_table; -- {serverError NOT_IMPLEMENTED} + +-- Not a table function, should throw +ALTER TABLE 03572_mt_table EXPORT PART '2020_1_1_0' TO TABLE FUNCTION extractKeyValuePairs('name:ronaldo'); -- {serverError UNKNOWN_FUNCTION} + +-- It is a table function, but the engine does not support exports/imports, should throw +ALTER TABLE 03572_mt_table EXPORT PART '2020_1_1_0' TO TABLE FUNCTION url('a.parquet'); -- {serverError NOT_IMPLEMENTED} + +-- Test that destination table can not have a column that matches the source ephemeral +CREATE TABLE 03572_ephemeral_mt_table (id UInt64, year UInt16, name String EPHEMERAL) ENGINE = MergeTree() PARTITION BY year ORDER BY tuple(); + +CREATE TABLE 03572_matching_ephemeral_s3_table (id UInt64, year UInt16, name String) ENGINE = S3(s3_conn, filename='03572_matching_ephemeral_s3_table', format='Parquet', partition_strategy='hive') PARTITION BY year; + +INSERT INTO 03572_ephemeral_mt_table (id, year, name) VALUES (1, 2020, 'alice'); + +ALTER TABLE 03572_ephemeral_mt_table EXPORT PART '2020_1_1_0' TO TABLE 03572_matching_ephemeral_s3_table; -- {serverError INCOMPATIBLE_COLUMNS} + +DROP TABLE IF EXISTS 03572_mt_table, 03572_invalid_schema_table, 03572_ephemeral_mt_table, 03572_matching_ephemeral_s3_table; diff --git a/tests/queries/0_stateless/03572_export_replicated_merge_tree_part_to_object_storage.reference b/tests/queries/0_stateless/03572_export_replicated_merge_tree_part_to_object_storage.reference new file mode 100644 index 000000000000..07f1ec6376a6 --- /dev/null +++ b/tests/queries/0_stateless/03572_export_replicated_merge_tree_part_to_object_storage.reference @@ -0,0 +1,16 @@ +---- Get actual part names and export them +---- Both data parts should appear +1 2020 +2 2020 +3 2020 +4 2021 +---- Export the same part again, it should be idempotent +1 2020 +2 2020 +3 2020 +4 2021 +---- Data in roundtrip ReplicatedMergeTree table (should match s3_table) +1 2020 +2 2020 +3 2020 +4 2021 diff --git a/tests/queries/0_stateless/03572_export_replicated_merge_tree_part_to_object_storage.sh b/tests/queries/0_stateless/03572_export_replicated_merge_tree_part_to_object_storage.sh new file mode 100755 index 000000000000..a691d4bdf37a --- /dev/null +++ b/tests/queries/0_stateless/03572_export_replicated_merge_tree_part_to_object_storage.sh @@ -0,0 +1,44 @@ +#!/usr/bin/env bash +# Tags: replica, no-parallel, no-replicated-database, no-fasttest + +CURDIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd) +# shellcheck source=../shell_config.sh +. "$CURDIR"/../shell_config.sh + +rmt_table="rmt_table_${RANDOM}" +s3_table="s3_table_${RANDOM}" +rmt_table_roundtrip="rmt_table_roundtrip_${RANDOM}" + +query() { + $CLICKHOUSE_CLIENT --query "$1" +} + +query "DROP TABLE IF EXISTS $rmt_table, $s3_table, $rmt_table_roundtrip" + +query "CREATE TABLE $rmt_table (id UInt64, year UInt16) ENGINE = ReplicatedMergeTree('/clickhouse/tables/{database}/$rmt_table', 'replica1') PARTITION BY year ORDER BY tuple()" +query "CREATE TABLE $s3_table (id UInt64, year UInt16) ENGINE = S3(s3_conn, filename='$s3_table', format=Parquet, partition_strategy='hive') PARTITION BY year" + +query "INSERT INTO $rmt_table VALUES (1, 2020), (2, 2020), (3, 2020), (4, 2021)" + +echo "---- Get actual part names and export them" +part_2020=$(query "SELECT name FROM system.parts WHERE database = currentDatabase() AND table = '$rmt_table' AND partition = '2020' ORDER BY name LIMIT 1" | tr -d '\n') +part_2021=$(query "SELECT name FROM system.parts WHERE database = currentDatabase() AND table = '$rmt_table' AND partition = '2021' ORDER BY name LIMIT 1" | tr -d '\n') + +query "ALTER TABLE $rmt_table EXPORT PART '$part_2020' TO TABLE $s3_table SETTINGS allow_experimental_export_merge_tree_part = 1" +query "ALTER TABLE $rmt_table EXPORT PART '$part_2021' TO TABLE $s3_table SETTINGS allow_experimental_export_merge_tree_part = 1" + +echo "---- Both data parts should appear" +query "SELECT * FROM $s3_table ORDER BY id" + +echo "---- Export the same part again, it should be idempotent" +query "ALTER TABLE $rmt_table EXPORT PART '$part_2020' TO TABLE $s3_table SETTINGS allow_experimental_export_merge_tree_part = 1" + +query "SELECT * FROM $s3_table ORDER BY id" + +query "CREATE TABLE $rmt_table_roundtrip (id UInt64, year UInt16) ENGINE = ReplicatedMergeTree('/clickhouse/tables/{database}/$rmt_table_roundtrip', 'replica1') PARTITION BY year ORDER BY tuple()" +query "INSERT INTO $rmt_table_roundtrip SELECT * FROM $s3_table" + +echo "---- Data in roundtrip ReplicatedMergeTree table (should match s3_table)" +query "SELECT * FROM $rmt_table_roundtrip ORDER BY id" + +query "DROP TABLE IF EXISTS $rmt_table, $s3_table, $rmt_table_roundtrip" diff --git a/tests/queries/0_stateless/03572_export_replicated_merge_tree_part_to_object_storage_simple.reference b/tests/queries/0_stateless/03572_export_replicated_merge_tree_part_to_object_storage_simple.reference new file mode 100644 index 000000000000..e69de29bb2d1 diff --git a/tests/queries/0_stateless/03572_export_replicated_merge_tree_part_to_object_storage_simple.sql b/tests/queries/0_stateless/03572_export_replicated_merge_tree_part_to_object_storage_simple.sql new file mode 100644 index 000000000000..f8f23532f0a7 --- /dev/null +++ b/tests/queries/0_stateless/03572_export_replicated_merge_tree_part_to_object_storage_simple.sql @@ -0,0 +1,22 @@ +-- Tags: no-parallel, no-fasttest + +DROP TABLE IF EXISTS 03572_rmt_table, 03572_invalid_schema_table; + +CREATE TABLE 03572_rmt_table (id UInt64, year UInt16) ENGINE = ReplicatedMergeTree('/clickhouse/{database}/test_03572_rmt/03572_rmt_table', 'replica1') PARTITION BY year ORDER BY tuple(); + +INSERT INTO 03572_rmt_table VALUES (1, 2020); + +-- Create a table with a different partition key and export a partition to it. It should throw +CREATE TABLE 03572_invalid_schema_table (id UInt64, x UInt16) ENGINE = S3(s3_conn, filename='03572_invalid_schema_table', format='Parquet', partition_strategy='hive') PARTITION BY x; + +ALTER TABLE 03572_rmt_table EXPORT PART '2020_0_0_0' TO TABLE 03572_invalid_schema_table +SETTINGS allow_experimental_export_merge_tree_part = 1; -- {serverError INCOMPATIBLE_COLUMNS} + +DROP TABLE 03572_invalid_schema_table; + +-- The only partition strategy that supports exports is hive. Wildcard should throw +CREATE TABLE 03572_invalid_schema_table (id UInt64, year UInt16) ENGINE = S3(s3_conn, filename='03572_invalid_schema_table/{_partition_id}', format='Parquet', partition_strategy='wildcard') PARTITION BY (id, year); + +ALTER TABLE 03572_rmt_table EXPORT PART '2020_0_0_0' TO TABLE 03572_invalid_schema_table SETTINGS allow_experimental_export_merge_tree_part = 1; -- {serverError NOT_IMPLEMENTED} + +DROP TABLE IF EXISTS 03572_rmt_table, 03572_invalid_schema_table; diff --git a/tests/queries/0_stateless/03604_export_merge_tree_partition.reference b/tests/queries/0_stateless/03604_export_merge_tree_partition.reference new file mode 100644 index 000000000000..d48023362b99 --- /dev/null +++ b/tests/queries/0_stateless/03604_export_merge_tree_partition.reference @@ -0,0 +1,31 @@ +Select from source table +1 2020 +2 2020 +3 2020 +4 2021 +5 2021 +6 2022 +7 2022 +Select from destination table +1 2020 +2 2020 +3 2020 +4 2021 +5 2021 +Export partition 2022 +Select from destination table again +1 2020 +2 2020 +3 2020 +4 2021 +5 2021 +6 2022 +7 2022 +---- Data in roundtrip ReplicatedMergeTree table (should match s3_table) +1 2020 +2 2020 +3 2020 +4 2021 +5 2021 +6 2022 +7 2022 diff --git a/tests/queries/0_stateless/03604_export_merge_tree_partition.sh b/tests/queries/0_stateless/03604_export_merge_tree_partition.sh new file mode 100755 index 000000000000..87503112aadb --- /dev/null +++ b/tests/queries/0_stateless/03604_export_merge_tree_partition.sh @@ -0,0 +1,57 @@ +#!/usr/bin/env bash +# Tags: no-fasttest, replica, no-parallel, no-replicated-database + +CURDIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd) +# shellcheck source=../shell_config.sh +. "$CURDIR"/../shell_config.sh + +rmt_table="rmt_table_${RANDOM}" +s3_table="s3_table_${RANDOM}" +rmt_table_roundtrip="rmt_table_roundtrip_${RANDOM}" + +query() { + $CLICKHOUSE_CLIENT --query "$1" +} + +query "DROP TABLE IF EXISTS $rmt_table, $s3_table, $rmt_table_roundtrip" + +query "CREATE TABLE $rmt_table (id UInt64, year UInt16) ENGINE = ReplicatedMergeTree('/clickhouse/tables/{database}/$rmt_table', 'replica1') PARTITION BY year ORDER BY tuple()" +query "CREATE TABLE $s3_table (id UInt64, year UInt16) ENGINE = S3(s3_conn, filename='$s3_table', format=Parquet, partition_strategy='hive') PARTITION BY year" + +query "INSERT INTO $rmt_table VALUES (1, 2020), (2, 2020), (4, 2021)" + +query "INSERT INTO $rmt_table VALUES (3, 2020), (5, 2021)" + +query "INSERT INTO $rmt_table VALUES (6, 2022), (7, 2022)" + +# sync replicas +query "SYSTEM SYNC REPLICA $rmt_table" + +query "ALTER TABLE $rmt_table EXPORT PARTITION ID '2020' TO TABLE $s3_table SETTINGS allow_experimental_export_merge_tree_part = 1" + +query "ALTER TABLE $rmt_table EXPORT PARTITION ID '2021' TO TABLE $s3_table SETTINGS allow_experimental_export_merge_tree_part = 1" + +# todo poll some kind of status +sleep 15 + +echo "Select from source table" +query "SELECT * FROM $rmt_table ORDER BY id" + +echo "Select from destination table" +query "SELECT * FROM $s3_table ORDER BY id" + +echo "Export partition 2022" +query "ALTER TABLE $rmt_table EXPORT PARTITION ID '2022' TO TABLE $s3_table SETTINGS allow_experimental_export_merge_tree_part = 1" + +# todo poll some kind of status +sleep 5 + +echo "Select from destination table again" +query "SELECT * FROM $s3_table ORDER BY id" + +query "CREATE TABLE $rmt_table_roundtrip ENGINE = ReplicatedMergeTree('/clickhouse/tables/{database}/$rmt_table_roundtrip', 'replica1') PARTITION BY year ORDER BY tuple() AS SELECT * FROM $s3_table" + +echo "---- Data in roundtrip ReplicatedMergeTree table (should match s3_table)" +query "SELECT * FROM $rmt_table_roundtrip ORDER BY id" + +query "DROP TABLE IF EXISTS $rmt_table, $s3_table, $rmt_table_roundtrip" \ No newline at end of file diff --git a/tests/queries/0_stateless/03608_export_merge_tree_part_filename_pattern.reference b/tests/queries/0_stateless/03608_export_merge_tree_part_filename_pattern.reference new file mode 100644 index 000000000000..8016f5aa113e --- /dev/null +++ b/tests/queries/0_stateless/03608_export_merge_tree_part_filename_pattern.reference @@ -0,0 +1,16 @@ +---- Test: Default pattern {part_name}_{checksum} +1 2020 +2 2020 +3 2020 +---- Verify filename matches 2020_1_1_0_*.1.parquet +1 +---- Test: Custom prefix pattern +4 2021 +---- Verify filename matches myprefix_2021_2_2_0.1.parquet +1 +---- Test: Pattern with macros +1 2020 +2 2020 +3 2020 +---- Verify macros expanded (no literal braces in parquet filenames, that's the best we can do for stateless tests) +1 diff --git a/tests/queries/0_stateless/03608_export_merge_tree_part_filename_pattern.sh b/tests/queries/0_stateless/03608_export_merge_tree_part_filename_pattern.sh new file mode 100755 index 000000000000..12b47f4f2664 --- /dev/null +++ b/tests/queries/0_stateless/03608_export_merge_tree_part_filename_pattern.sh @@ -0,0 +1,49 @@ +#!/usr/bin/env bash +# Tags: no-fasttest +# Tag no-fasttest: requires s3 storage + +CURDIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd) +# shellcheck source=../shell_config.sh +. "$CURDIR"/../shell_config.sh + +R=$RANDOM +mt="mt_${R}" +dest1="fp_dest1_${R}" +dest2="fp_dest2_${R}" +dest3="fp_dest3_${R}" + +query() { + $CLICKHOUSE_CLIENT --query "$1" +} + +query "DROP TABLE IF EXISTS $mt, $dest1, $dest2, $dest3" + +query "CREATE TABLE $mt (id UInt64, year UInt16) ENGINE = MergeTree() PARTITION BY year ORDER BY tuple()" +query "INSERT INTO $mt VALUES (1, 2020), (2, 2020), (3, 2020), (4, 2021)" + +query "CREATE TABLE $dest1 (id UInt64, year UInt16) ENGINE = S3(s3_conn, filename='$dest1', format=Parquet, partition_strategy='hive') PARTITION BY year" +query "CREATE TABLE $dest2 (id UInt64, year UInt16) ENGINE = S3(s3_conn, filename='$dest2', format=Parquet, partition_strategy='hive') PARTITION BY year" +query "CREATE TABLE $dest3 (id UInt64, year UInt16) ENGINE = S3(s3_conn, filename='$dest3', format=Parquet, partition_strategy='hive') PARTITION BY year" + +echo "---- Test: Default pattern {part_name}_{checksum}" +query "ALTER TABLE $mt EXPORT PART '2020_1_1_0' TO TABLE $dest1 SETTINGS allow_experimental_export_merge_tree_part = 1, export_merge_tree_part_filename_pattern = '{part_name}_{checksum}'" +sleep 3 +query "SELECT * FROM $dest1 ORDER BY id" +echo "---- Verify filename matches 2020_1_1_0_*.1.parquet" +query "SELECT count() FROM s3(s3_conn, filename='$dest1/**/2020_1_1_0_*.1.parquet', format='One')" + +echo "---- Test: Custom prefix pattern" +query "ALTER TABLE $mt EXPORT PART '2021_2_2_0' TO TABLE $dest2 SETTINGS allow_experimental_export_merge_tree_part = 1, export_merge_tree_part_filename_pattern = 'myprefix_{part_name}'" +sleep 3 +query "SELECT * FROM $dest2 ORDER BY id" +echo "---- Verify filename matches myprefix_2021_2_2_0.1.parquet" +query "SELECT count() FROM s3(s3_conn, filename='$dest2/**/myprefix_2021_2_2_0.1.parquet', format='One')" + +echo "---- Test: Pattern with macros" +query "ALTER TABLE $mt EXPORT PART '2020_1_1_0' TO TABLE $dest3 SETTINGS allow_experimental_export_merge_tree_part = 1, export_merge_tree_part_filename_pattern = '{database}_{table}_{part_name}'" +sleep 3 +query "SELECT * FROM $dest3 ORDER BY id" +echo "---- Verify macros expanded (no literal braces in parquet filenames, that's the best we can do for stateless tests)" +query "SELECT count() = 0 FROM s3(s3_conn, filename='$dest3/**/*.1.parquet', format='One') WHERE _file LIKE '%{%'" + +query "DROP TABLE IF EXISTS $mt, $dest1, $dest2, $dest3" diff --git a/tests/queries/0_stateless/03745_system_background_schedule_pool.reference b/tests/queries/0_stateless/03745_system_background_schedule_pool.reference index d09cfc400cbb..ed65c4991867 100644 --- a/tests/queries/0_stateless/03745_system_background_schedule_pool.reference +++ b/tests/queries/0_stateless/03745_system_background_schedule_pool.reference @@ -1,6 +1,10 @@ 1 buffer_flush default test_buffer_03745 1 StorageBuffer (default.test_buffer_03745)/Bg schedule default test_merge_tree_03745 1 BackgroundJobsAssignee:DataProcessing +<<<<<<< HEAD schedule default test_merge_tree_03745 1 BackgroundJobsAssignee:Streaming +======= +schedule default test_merge_tree_03745 1 BackgroundJobsAssignee:Moving +>>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) schedule default test_merge_tree_03745 1 default.test_merge_tree_03745 (CleanupThread) distributed default test_distributed_03745 1 default.test_distributed_03745.DistributedInsertQueue.default/Bg From 7ace22a96b43b18b973dc6387423bafa65671338 Mon Sep 17 00:00:00 2001 From: Andrey Zvonov <32552679+zvonand@users.noreply.github.com> Date: Mon, 3 Aug 2026 13:45:19 +0200 Subject: [PATCH 02/43] Resolve conflicts in cherry-pick of #1718 Kept antalya-26.6 side for everything outside the source PR's scope and added the PR's new surfaces (export part/partition, list-objects cache, Iceberg export) on top. Adapted: PENDING_MUTATIONS_NOT_ALLOWED moved to error code 1009 (1005-1008 are taken on antalya-26.6) and END bumped accordingly Adapted: MergeTreeData::writePartLog keeps antalya-26.6's projections_duration_ms parameter and appends the PR's exports_entry; ExportPartTask call sites pass the extra argument Adapted: generateManifestFile keeps antalya-26.6's user_defined_sequence_number / per-file row+byte counts and appends the PR's per_file_stats; the export-commit call site boxes data file paths into Iceberg::IcebergPathFromMetadata Adapted: MultipleFileWriter::getDataFileEntries is implemented on antalya-26.6's existing per-file bookkeeping (data_file_names / data_file_row_counts / data_file_byte_counts / completed_file_stats) instead of the PR's duplicate vectors Source-PR: #1718 (https://github.com/Altinity/ClickHouse/pull/1718) --- ci/jobs/scripts/integration_tests_configs.py | 6 -- src/Common/ErrorCodes.cpp | 11 +--- src/Common/FailPoint.cpp | 3 - src/Common/ProfileEvents.cpp | 12 ---- src/Common/setThreadName.h | 3 - src/Core/Settings.cpp | 4 +- src/Core/SettingsChangesHistory.cpp | 5 -- src/Core/SettingsEnums.cpp | 4 +- src/Core/SettingsEnums.h | 4 +- .../ObjectStorages/IObjectStorage.h | 4 +- src/Functions/generateSnowflakeID.cpp | 6 +- src/Interpreters/DDLWorker.cpp | 5 +- src/Parsers/ASTAlterQuery.cpp | 6 -- src/Parsers/ASTSystemQuery.cpp | 3 - src/Parsers/ParserAlterQuery.cpp | 6 -- src/Storages/IPartitionStrategy.cpp | 41 ------------ src/Storages/IPartitionStrategy.h | 10 --- src/Storages/MergeTree/ExportPartTask.cpp | 3 + src/Storages/MergeTree/IMergeTreeDataPart.cpp | 5 +- src/Storages/MergeTree/MergeTreeData.cpp | 57 ++-------------- src/Storages/MergeTree/MergeTreeData.h | 13 +--- .../DataLakes/IDataLakeMetadata.h | 4 -- .../DataLakes/Iceberg/IcebergMetadata.cpp | 13 ++-- .../DataLakes/Iceberg/IcebergMetadata.h | 6 -- .../DataLakes/Iceberg/IcebergWrites.cpp | 66 +++---------------- .../DataLakes/Iceberg/IcebergWrites.h | 11 +--- .../DataLakes/Iceberg/MultipleFileWriter.cpp | 49 +++++--------- .../DataLakes/Iceberg/MultipleFileWriter.h | 20 +----- .../ObjectStorage/DataLakes/Iceberg/Utils.cpp | 8 --- .../ObjectStorage/StorageObjectStorage.cpp | 3 - .../ObjectStorage/StorageObjectStorage.h | 4 -- .../StorageObjectStorageCluster.cpp | 6 +- .../StorageObjectStorageConfiguration.h | 7 -- .../ObjectStorage/StorageObjectStorageSink.h | 6 +- .../StorageObjectStorageSource.cpp | 5 -- .../StorageObjectStorageSource.h | 4 -- src/Storages/StorageMergeTree.cpp | 7 -- src/Storages/StorageReplicatedMergeTree.cpp | 3 - src/Storages/StorageReplicatedMergeTree.h | 4 -- src/Storages/System/StorageSystemMerges.cpp | 4 -- src/Storages/System/attachSystemTables.cpp | 6 +- ..._settings_cannot_be_enabled_by_default.sql | 11 +--- ..._system_background_schedule_pool.reference | 5 +- 43 files changed, 63 insertions(+), 400 deletions(-) diff --git a/ci/jobs/scripts/integration_tests_configs.py b/ci/jobs/scripts/integration_tests_configs.py index 105b989fd06f..673a86f0bd72 100644 --- a/ci/jobs/scripts/integration_tests_configs.py +++ b/ci/jobs/scripts/integration_tests_configs.py @@ -58,13 +58,7 @@ class TC: True, "pins azurite to fixed host port 10000 (Spark emulator mode); concurrent --dist=each workers collide on bind", ), -<<<<<<< HEAD -======= - TC("test_storage_iceberg_no_spark/", True, "no idea why i'm sequential"), - TC("test_storage_iceberg_with_spark_cache/", True, "no idea why i'm sequential"), - TC("test_storage_iceberg_concurrent/", True, "no idea why i'm sequential"), TC("test_export_replicated_mt_partition_to_object_storage/", True, "ZooKeeper can't handle too many parallel requests"), ->>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) ] IMAGES_ENV = { diff --git a/src/Common/ErrorCodes.cpp b/src/Common/ErrorCodes.cpp index a0f3b90d7109..d6e809c6ed33 100644 --- a/src/Common/ErrorCodes.cpp +++ b/src/Common/ErrorCodes.cpp @@ -671,14 +671,11 @@ M(1002, UNKNOWN_EXCEPTION) \ M(1003, SSH_EXCEPTION) \ M(1004, STARTUP_SCRIPTS_ERROR) \ -<<<<<<< HEAD M(1005, STALE_VERSION) \ M(1006, INVALID_CURSOR_LOOKUP) \ M(1007, ILLEGAL_STREAM) \ M(1008, TEMPORARY_DATA_NOT_IN_CACHE) \ -======= - M(1005, PENDING_MUTATIONS_NOT_ALLOWED) \ ->>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) + M(1009, PENDING_MUTATIONS_NOT_ALLOWED) \ /* See END */ #ifdef APPLY_FOR_EXTERNAL_ERROR_CODES @@ -695,11 +692,7 @@ namespace ErrorCodes APPLY_FOR_ERROR_CODES(M) #undef M -<<<<<<< HEAD - constexpr ErrorCode END = 1008; -======= - constexpr ErrorCode END = 1005; ->>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) + constexpr ErrorCode END = 1009; ErrorPairHolder values[END + 1]{}; struct ErrorCodesNames diff --git a/src/Common/FailPoint.cpp b/src/Common/FailPoint.cpp index af7415cc19af..bf5ae1bea887 100644 --- a/src/Common/FailPoint.cpp +++ b/src/Common/FailPoint.cpp @@ -162,15 +162,12 @@ static struct InitFiu ONCE(write_file_operation_fail_on_read) \ REGULAR(slowdown_parallel_replicas_local_plan_read) \ ONCE(iceberg_writes_cleanup) \ -<<<<<<< HEAD REGULAR(storage_cluster_read_sleep) \ -======= ONCE(iceberg_writes_non_retry_cleanup) \ ONCE(iceberg_writes_post_publish_throw) \ ONCE(iceberg_export_after_commit_before_zk_completed) \ REGULAR(export_partition_commit_always_throw) \ ONCE(export_partition_status_change_throw) \ ->>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) ONCE(backup_add_empty_memory_table) \ PAUSEABLE_ONCE(backup_pause_on_start) \ PAUSEABLE_ONCE(restore_pause_on_start) \ diff --git a/src/Common/ProfileEvents.cpp b/src/Common/ProfileEvents.cpp index 6d868b24bb3c..8f3745c38762 100644 --- a/src/Common/ProfileEvents.cpp +++ b/src/Common/ProfileEvents.cpp @@ -241,15 +241,12 @@ M(MergesThrottlerSleepMicroseconds, "Total time a query was sleeping to conform 'max_merges_bandwidth_for_server' throttling.", ValueType::Microseconds) \ M(MutationsThrottlerBytes, "Bytes passed through 'max_mutations_bandwidth_for_server' throttler.", ValueType::Bytes) \ M(MutationsThrottlerSleepMicroseconds, "Total time a query was sleeping to conform 'max_mutations_bandwidth_for_server' throttling.", ValueType::Microseconds) \ -<<<<<<< HEAD M(UserThrottlerBytes, "Bytes passed through 'max_network_bandwidth_for_user' throttler.", ValueType::Bytes) \ M(UserThrottlerSleepMicroseconds, "Total time a query was sleeping to conform 'max_network_bandwidth_for_user' throttling.", ValueType::Microseconds) \ M(AllUsersThrottlerBytes, "Bytes passed through 'max_network_bandwidth_for_all_users' throttler.", ValueType::Bytes) \ M(AllUsersThrottlerSleepMicroseconds, "Total time a query was sleeping to conform 'max_network_bandwidth_for_all_users' throttling.", ValueType::Microseconds) \ -======= M(ExportsThrottlerBytes, "Bytes passed through 'max_exports_bandwidth_for_server' throttler.", ValueType::Bytes) \ M(ExportsThrottlerSleepMicroseconds, "Total time a query was sleeping to conform 'max_exports_bandwidth_for_server' throttling.", ValueType::Microseconds) \ ->>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) M(QueryRemoteReadThrottlerBytes, "Bytes passed through 'max_remote_read_network_bandwidth' throttler.", ValueType::Bytes) \ M(QueryRemoteReadThrottlerSleepMicroseconds, "Total time a query was sleeping to conform 'max_remote_read_network_bandwidth' throttling.", ValueType::Microseconds) \ M(ReaderExecutorSourceRequests, "Number of source-side requests opened by ReaderExecutor (excludes live-buffer reuses).", ValueType::Number) \ @@ -1516,7 +1513,6 @@ The server successfully detected this situation and will download merged part fr M(RuntimeFilterRowsPassed, "Number of rows that passed (not filtered out by) JOIN Runtime Filters", ValueType::Number) \ M(RuntimeFilterRowsSkipped, "Number of rows in blocks that were skipped by JOIN Runtime Filters", ValueType::Number) \ \ -<<<<<<< HEAD M(JoinBuildPostProcessingMicroseconds, "Elapsed time of post-processing steps after building the right JOIN side.", ValueType::Microseconds) \ \ M(AIInputTokens, "Total prompt tokens consumed across all AI function calls in the query.", ValueType::Number) \ @@ -1525,19 +1521,11 @@ The server successfully detected this situation and will download merged part fr M(AIRowsProcessed, "Number of rows that received an AI result.", ValueType::Number) \ M(AIRowsSkipped, "Number of rows that received a default value due to quota or error.", ValueType::Number) \ \ -======= - M(ObjectStorageClusterSentToMatchedReplica, "Number of tasks in ObjectStorageCluster request sent to matched replica.", ValueType::Number) \ - M(ObjectStorageClusterSentToNonMatchedReplica, "Number of tasks in ObjectStorageCluster request sent to non-matched replica.", ValueType::Number) \ - M(ObjectStorageClusterProcessedTasks, "Number of processed tasks in ObjectStorageCluster request.", ValueType::Number) \ - M(ObjectStorageClusterWaitingMicroseconds, "Time of waiting for tasks in ObjectStorageCluster request.", ValueType::Microseconds) \ - \ M(ObjectStorageListObjectsCacheHits, "Number of times object storage list objects operation hit the cache.", ValueType::Number) \ M(ObjectStorageListObjectsCacheMisses, "Number of times object storage list objects operation miss the cache.", ValueType::Number) \ M(ObjectStorageListObjectsCacheExactMatchHits, "Number of times object storage list objects operation hit the cache with an exact match.", ValueType::Number) \ M(ObjectStorageListObjectsCachePrefixMatchHits, "Number of times object storage list objects operation miss the cache using prefix matching.", ValueType::Number) \ ->>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) - #ifdef APPLY_FOR_EXTERNAL_EVENTS #define APPLY_FOR_EVENTS(M) APPLY_FOR_BUILTIN_EVENTS(M) APPLY_FOR_EXTERNAL_EVENTS(M) #else diff --git a/src/Common/setThreadName.h b/src/Common/setThreadName.h index 9992fd9c86de..11f8e4338c97 100644 --- a/src/Common/setThreadName.h +++ b/src/Common/setThreadName.h @@ -169,11 +169,8 @@ namespace DB M(ZOOKEEPER_SEND, "ZooKeeperSend") \ M(BLOB_KILLER_TASK, "BlobKillerTask") \ M(BLOB_COPIER_TASK, "BlobCopierTask") \ -<<<<<<< HEAD M(DISK_OBJECT_STORAGE_COPY, "DiskObjStCopy") \ -======= M(EXPORT_PART, "ExportPart") \ ->>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) enum class ThreadName : uint8_t diff --git a/src/Core/Settings.cpp b/src/Core/Settings.cpp index 8bf8969f286a..d17025dd1cc1 100644 --- a/src/Core/Settings.cpp +++ b/src/Core/Settings.cpp @@ -7970,7 +7970,6 @@ Use `allow_nullable_tuple_in_extracted_subcolumns` to control whether extracted )", BETA, enable_nullable_tuple_type) \ DECLARE(UInt64, archive_adaptive_buffer_max_size_bytes, 8 * DBMS_DEFAULT_BUFFER_SIZE, R"( Limits the maximum size of the adaptive buffer used when writing to archive files (for example, tar archives)", 0) \ -<<<<<<< HEAD DECLARE(UInt64, shared_merge_tree_sequential_consistency_initial_parts_update_backoff_ms, 50, R"( Initial backoff in milliseconds for parts update when using `select_sequential_consistency` with `SharedMergeTree`. Only available in ClickHouse Cloud. )", 0) \ @@ -7999,7 +7998,7 @@ Enable converting the hash table to a flat array for joins when the key is a sin )", 0) \ DECLARE(UInt64, query_plan_min_columns_for_join_lazy_indexing, 3, R"( Control the minimum number of payload columns from the left side required for enabling lazy indexing optimization in JOIN. 0 means the optimization is disabled. -======= +)", 0) \ DECLARE(Bool, export_merge_tree_part_overwrite_file_if_exists, false, R"( Overwrite file if it already exists when exporting a merge tree part )", 0) \ @@ -8052,7 +8051,6 @@ Querying ZooKeeper is expensive, and only available if the ZooKeeper feature fla )", 0) \ DECLARE(String, export_merge_tree_part_filename_pattern, "{part_name}_{checksum}", R"( Pattern for the filename of the exported merge tree part. The `part_name` and `checksum` are calculated and replaced on the fly. Additional macros are supported. ->>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) )", 0) \ \ /* ####################################################### */ \ diff --git a/src/Core/SettingsChangesHistory.cpp b/src/Core/SettingsChangesHistory.cpp index b7df09208b95..c7d722164f22 100644 --- a/src/Core/SettingsChangesHistory.cpp +++ b/src/Core/SettingsChangesHistory.cpp @@ -487,17 +487,12 @@ const VersionToSettingsChangesMap & getSettingsChangesHistory() // {"export_merge_tree_partition_system_table_prefer_remote_information", true, true, "New setting."}, // {"export_merge_tree_part_throw_on_pending_mutations", true, true, "New setting."}, // {"export_merge_tree_part_throw_on_pending_patch_parts", true, true, "New setting."}, - {"lock_object_storage_task_distribution_ms", 500, 500, "New setting."}, - {"allow_retries_in_cluster_requests", false, false, "New setting"}, {"allow_experimental_export_merge_tree_part", false, true, "Turned ON by default for Antalya."}, {"export_merge_tree_part_overwrite_file_if_exists", false, false, "New setting."}, {"export_merge_tree_partition_force_export", false, false, "New setting."}, {"export_merge_tree_partition_max_retries", 3, 3, "New setting."}, {"export_merge_tree_partition_manifest_ttl", 180, 180, "New setting."}, {"export_merge_tree_part_file_already_exists_policy", "skip", "skip", "New setting."}, - {"hybrid_table_auto_cast_columns", true, true, "New setting to automatically cast Hybrid table columns when segments disagree on types. Default enabled."}, - {"allow_experimental_hybrid_table", false, false, "Added new setting to allow the Hybrid table engine."}, - {"enable_alias_marker", true, true, "New setting."}, {"export_merge_tree_part_max_bytes_per_file", 0, 0, "New setting."}, {"export_merge_tree_part_max_rows_per_file", 0, 0, "New setting."}, {"export_merge_tree_partition_lock_inside_the_task", false, false, "New setting."}, diff --git a/src/Core/SettingsEnums.cpp b/src/Core/SettingsEnums.cpp index 56ab66303116..41e291f24820 100644 --- a/src/Core/SettingsEnums.cpp +++ b/src/Core/SettingsEnums.cpp @@ -504,7 +504,6 @@ IMPLEMENT_SETTING_ENUM(JemallocProfileFormat, ErrorCodes::BAD_ARGUMENTS, {"symbolized", JemallocProfileFormat::Symbolized}, {"collapsed", JemallocProfileFormat::Collapsed}}) -<<<<<<< HEAD IMPLEMENT_SETTING_ENUM(S3UriStyle, ErrorCodes::BAD_ARGUMENTS, {{"auto", S3UriStyle::AUTO}, {"path", S3UriStyle::PATH}, @@ -515,8 +514,7 @@ IMPLEMENT_SETTING_ENUM( ErrorCodes::BAD_ARGUMENTS, {{"wildcard", FileLikeEngineDefaultPartitionStrategy::WILDCARD}, {"hive", FileLikeEngineDefaultPartitionStrategy::HIVE}}) -======= + IMPLEMENT_SETTING_AUTO_ENUM(MergeTreePartExportFileAlreadyExistsPolicy, ErrorCodes::BAD_ARGUMENTS); ->>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) } diff --git a/src/Core/SettingsEnums.h b/src/Core/SettingsEnums.h index 8ea761671db3..841e6132020e 100644 --- a/src/Core/SettingsEnums.h +++ b/src/Core/SettingsEnums.h @@ -584,7 +584,6 @@ enum class JemallocProfileFormat : uint8_t DECLARE_SETTING_ENUM(JemallocProfileFormat) -<<<<<<< HEAD enum class S3UriStyle : uint8_t { AUTO, @@ -600,7 +599,7 @@ enum class FileLikeEngineDefaultPartitionStrategy : uint8_t HIVE, }; DECLARE_SETTING_ENUM(FileLikeEngineDefaultPartitionStrategy) -======= + enum class MergeTreePartExportFileAlreadyExistsPolicy : uint8_t { skip, @@ -610,6 +609,5 @@ enum class MergeTreePartExportFileAlreadyExistsPolicy : uint8_t DECLARE_SETTING_ENUM(MergeTreePartExportFileAlreadyExistsPolicy) ->>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) } diff --git a/src/Disks/DiskObjectStorage/ObjectStorages/IObjectStorage.h b/src/Disks/DiskObjectStorage/ObjectStorages/IObjectStorage.h index 97fd4fe3c50a..4af4b6c4bbdc 100644 --- a/src/Disks/DiskObjectStorage/ObjectStorages/IObjectStorage.h +++ b/src/Disks/DiskObjectStorage/ObjectStorages/IObjectStorage.h @@ -370,13 +370,11 @@ class IObjectStorage } #endif -<<<<<<< HEAD /// Returns the inner (unwrapped) object storage for decorator types such as `CachedObjectStorage`. /// Returns nullptr for non-decorator types, meaning this storage is already the base. virtual ObjectStoragePtr getUnderlying() { return nullptr; } -======= + virtual bool supportsListObjectsCache() { return false; } ->>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) }; using ObjectStoragePtr = std::shared_ptr; diff --git a/src/Functions/generateSnowflakeID.cpp b/src/Functions/generateSnowflakeID.cpp index d1c2446493a9..846c31986ba9 100644 --- a/src/Functions/generateSnowflakeID.cpp +++ b/src/Functions/generateSnowflakeID.cpp @@ -155,16 +155,12 @@ uint64_t generateSnowflakeID() return fromSnowflakeId(snowflake_id); } -<<<<<<< HEAD -class FunctionGenerateSnowflakeID final : public IFunction -======= std::string generateSnowflakeIDString() { return std::to_string(generateSnowflakeID()); } -class FunctionGenerateSnowflakeID : public IFunction ->>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) +class FunctionGenerateSnowflakeID final : public IFunction { public: static constexpr auto name = "generateSnowflakeID"; diff --git a/src/Interpreters/DDLWorker.cpp b/src/Interpreters/DDLWorker.cpp index c120873fbd0e..81355067db4a 100644 --- a/src/Interpreters/DDLWorker.cpp +++ b/src/Interpreters/DDLWorker.cpp @@ -840,11 +840,8 @@ bool DDLWorker::taskShouldBeExecutedOnLeader(const ASTPtr & ast_ddl, const Stora alter->isUnlockSnapshot() || alter->isMovePartitionToDiskOrVolumeAlter() || alter->isCommentAlter() || -<<<<<<< HEAD - alter->isSettingsOrCommentAlter()) -======= + alter->isSettingsOrCommentAlter() || alter->isExportPartOrExportPartitionAlter()) ->>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) return false; } diff --git a/src/Parsers/ASTAlterQuery.cpp b/src/Parsers/ASTAlterQuery.cpp index dfa92c029ae2..066b74111891 100644 --- a/src/Parsers/ASTAlterQuery.cpp +++ b/src/Parsers/ASTAlterQuery.cpp @@ -68,17 +68,14 @@ ASTPtr ASTAlterCommand::clone() const res->rename_to = res->children.emplace_back(rename_to->clone()).get(); if (execute_args) res->execute_args = res->children.emplace_back(execute_args->clone()).get(); -<<<<<<< HEAD if (add_enum_values) res->add_enum_values = res->children.emplace_back(add_enum_values->clone()); if (refresh) res->refresh = res->children.emplace_back(refresh->clone()).get(); -======= if (to_table_function) res->to_table_function = res->children.emplace_back(to_table_function->clone()).get(); if (partition_by_expr) res->partition_by_expr = res->children.emplace_back(partition_by_expr->clone()).get(); ->>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) return res; } @@ -655,12 +652,9 @@ void ASTAlterCommand::forEachPointerToChild(std::function>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) } diff --git a/src/Parsers/ASTSystemQuery.cpp b/src/Parsers/ASTSystemQuery.cpp index 7834ce541ec2..2d04acd0dbfe 100644 --- a/src/Parsers/ASTSystemQuery.cpp +++ b/src/Parsers/ASTSystemQuery.cpp @@ -603,11 +603,8 @@ void ASTSystemQuery::formatImpl(WriteBuffer & ostr, const FormatSettings & setti case Type::CLEAR_S3_CLIENT_CACHE: case Type::CLEAR_ICEBERG_METADATA_CACHE: case Type::CLEAR_PARQUET_METADATA_CACHE: -<<<<<<< HEAD case Type::CLEAR_AVRO_SCHEMA_CACHE: -======= case Type::DROP_OBJECT_STORAGE_LIST_OBJECTS_CACHE: ->>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) case Type::RESET_COVERAGE: case Type::RESTART_REPLICAS: case Type::JEMALLOC_PURGE: diff --git a/src/Parsers/ParserAlterQuery.cpp b/src/Parsers/ParserAlterQuery.cpp index fa0e2f93361e..ce382013216a 100644 --- a/src/Parsers/ParserAlterQuery.cpp +++ b/src/Parsers/ParserAlterQuery.cpp @@ -186,12 +186,9 @@ bool ParserAlterCommand::parseImpl(Pos & pos, ASTPtr & node, Expected & expected ASTPtr command_rename_to; ASTPtr command_sql_security; ASTPtr command_snapshot_desc; -<<<<<<< HEAD ASTPtr command_refresh; -======= ASTPtr export_table_function; ASTPtr export_table_function_partition_by_expr; ->>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) if (with_round_bracket) { @@ -1205,15 +1202,12 @@ bool ParserAlterCommand::parseImpl(Pos & pos, ASTPtr & node, Expected & expected command->rename_to = command->children.emplace_back(std::move(command_rename_to)).get(); if (command_snapshot_desc) command->snapshot_desc = command->children.emplace_back(std::move(command_snapshot_desc)).get(); -<<<<<<< HEAD if (command_refresh) command->refresh = command->children.emplace_back(std::move(command_refresh)).get(); -======= if (export_table_function) command->to_table_function = command->children.emplace_back(std::move(export_table_function)).get(); if (export_table_function_partition_by_expr) command->partition_by_expr = command->children.emplace_back(std::move(export_table_function_partition_by_expr)).get(); ->>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) return true; } diff --git a/src/Storages/IPartitionStrategy.cpp b/src/Storages/IPartitionStrategy.cpp index e301c977bb16..01c85f481427 100644 --- a/src/Storages/IPartitionStrategy.cpp +++ b/src/Storages/IPartitionStrategy.cpp @@ -103,11 +103,8 @@ namespace const IPartitionStrategy & partition_strategy, BuildAST && build_ast) { -<<<<<<< HEAD /// The cache write happens in the cacheDeterministicActions function, which is called from the constructor of the partition strategy. /// If the actions are not deterministic, it will not be cached. -======= ->>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) if (cached_result) return *cached_result; @@ -353,44 +350,6 @@ HiveStylePartitionStrategy::HiveStylePartitionStrategy( ColumnPtr HiveStylePartitionStrategy::computePartitionKey(const Chunk & chunk) const { -<<<<<<< HEAD - return prefix + "**." + Poco::toLower(file_format); -} - -std::string HiveStylePartitionStrategy::getPathForWrite( - const std::string & prefix, - const std::string & partition_key) -{ - std::string path; - - if (!prefix.empty()) - { - path += prefix; - if (path.back() != '/') - { - path += '/'; - } - } - - /// Not adding '/' because buildExpressionHive() always adds a trailing '/' - path += partition_key; - - /* - * File extension is toLower(format) - * This isn't ideal, but I guess multiple formats can be specified and introduced. - * So I think it is simpler to keep it this way. - * - * Or perhaps implement something like `IInputFormat::getFileExtension()` - */ - path += std::to_string(generateSnowflakeID()) + "." + Poco::toLower(file_format); - - return path; -} - -ColumnPtr HiveStylePartitionStrategy::computePartitionKey(const Chunk & chunk) const -{ -======= ->>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) auto actions_with_column = getCachedOrBuildActions( cached_result, *this, diff --git a/src/Storages/IPartitionStrategy.h b/src/Storages/IPartitionStrategy.h index 987b892054a8..1378762c6911 100644 --- a/src/Storages/IPartitionStrategy.h +++ b/src/Storages/IPartitionStrategy.h @@ -92,13 +92,8 @@ struct WildcardPartitionStrategy : IPartitionStrategy WildcardPartitionStrategy(KeyDescription partition_key_description_, const Block & sample_block_, ContextPtr context_); ColumnPtr computePartitionKey(const Chunk & chunk) const override; -<<<<<<< HEAD - std::string getPathForRead(const std::string & prefix) override; - std::string getPathForWrite(const std::string & prefix, const std::string & partition_key) override; -======= ColumnPtr computePartitionKey(Block & block) const override; ->>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) }; /* @@ -116,13 +111,8 @@ struct HiveStylePartitionStrategy : IPartitionStrategy bool partition_columns_in_data_file_); ColumnPtr computePartitionKey(const Chunk & chunk) const override; -<<<<<<< HEAD - std::string getPathForRead(const std::string & prefix) override; - std::string getPathForWrite(const std::string & prefix, const std::string & partition_key) override; -======= ColumnPtr computePartitionKey(Block & block) const override; ->>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) ColumnRawPtrs getFormatChunkColumns(const Chunk & chunk) override; Block getFormatHeader() override; diff --git a/src/Storages/MergeTree/ExportPartTask.cpp b/src/Storages/MergeTree/ExportPartTask.cpp index 1014b9df7422..bb6fc1491afa 100644 --- a/src/Storages/MergeTree/ExportPartTask.cpp +++ b/src/Storages/MergeTree/ExportPartTask.cpp @@ -302,6 +302,7 @@ bool ExportPartTask::executeStep() nullptr, nullptr, {}, + {}, exports_list_entry.get()); storage.export_manifests.erase(manifest); @@ -338,6 +339,7 @@ bool ExportPartTask::executeStep() nullptr, nullptr, {}, + {}, exports_list_entry.get()); std::lock_guard inner_lock(storage.export_manifests_mutex); @@ -367,6 +369,7 @@ bool ExportPartTask::executeStep() nullptr, nullptr, {}, + {}, exports_list_entry.get()); std::lock_guard inner_lock(storage.export_manifests_mutex); diff --git a/src/Storages/MergeTree/IMergeTreeDataPart.cpp b/src/Storages/MergeTree/IMergeTreeDataPart.cpp index 597df9fce602..06bb629de89e 100644 --- a/src/Storages/MergeTree/IMergeTreeDataPart.cpp +++ b/src/Storages/MergeTree/IMergeTreeDataPart.cpp @@ -345,7 +345,6 @@ String IMergeTreeDataPart::MinMaxIndex::getFileColumnName(const String & column_ return stream_name; } -<<<<<<< HEAD IMergeTreeDataPart::MinMaxIndexPtr IMergeTreeDataPart::getMinMaxIndex() const { std::lock_guard lock(minmax_idx_mutex); @@ -370,7 +369,8 @@ void IMergeTreeDataPart::setMinMaxIndex(MinMaxIndexPtr minmax_index) const { std::lock_guard lock(minmax_idx_mutex); minmax_idx = std::move(minmax_index); -======= +} + Block IMergeTreeDataPart::MinMaxIndex::getBlock(const MergeTreeData & data) const { if (!initialized) @@ -405,7 +405,6 @@ Block IMergeTreeDataPart::MinMaxIndex::getBlock(const MergeTreeData & data) cons } return block; ->>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) } void IMergeTreeDataPart::incrementStateMetric(MergeTreeDataPartState state_) const diff --git a/src/Storages/MergeTree/MergeTreeData.cpp b/src/Storages/MergeTree/MergeTreeData.cpp index 5bff1d0b2763..0ff8975a4cbb 100644 --- a/src/Storages/MergeTree/MergeTreeData.cpp +++ b/src/Storages/MergeTree/MergeTreeData.cpp @@ -19,34 +19,14 @@ #include #include #include -<<<<<<< HEAD #include #include -======= -#include -#include -#include -#include -#include -#include -#include -#include -#include -#include -#include -#include -#include #include #include #include #include #include #include -#include -#include -#include -#include ->>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) #include #include #include @@ -101,27 +81,10 @@ #include #include #include -<<<<<<< HEAD #include -======= -#include -#include -#include -#include -#include -#include -#include -#include #include -#include -#include -#include -#include -#include -#include #include #include ->>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) #include #include #include @@ -131,7 +94,6 @@ #include #include #include -<<<<<<< HEAD #include #include #include @@ -172,9 +134,7 @@ #include #include #include -======= #include ->>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) #include @@ -295,9 +255,7 @@ namespace Setting extern const SettingsBool use_statistics; extern const SettingsBool use_statistics_cache; extern const SettingsBool use_partition_pruning; -<<<<<<< HEAD extern const SettingsBool use_skip_indexes; -======= extern const SettingsBool allow_experimental_export_merge_tree_part; extern const SettingsUInt64 min_bytes_to_use_direct_io; extern const SettingsMergeTreePartExportFileAlreadyExistsPolicy export_merge_tree_part_file_already_exists_policy; @@ -306,7 +264,6 @@ namespace Setting extern const SettingsBool export_merge_tree_part_throw_on_pending_mutations; extern const SettingsBool export_merge_tree_part_throw_on_pending_patch_parts; extern const SettingsBool allow_insert_into_iceberg; ->>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) } namespace MergeTreeSetting @@ -443,8 +400,10 @@ namespace ErrorCodes extern const int CANNOT_FORGET_PARTITION; extern const int DATA_TYPE_CANNOT_BE_USED_IN_KEY; extern const int TOO_LARGE_LIGHTWEIGHT_UPDATES; -<<<<<<< HEAD extern const int FAULT_INJECTED; + extern const int UNKNOWN_TABLE; + extern const int FILE_ALREADY_EXISTS; + extern const int PENDING_MUTATIONS_NOT_ALLOWED; } namespace FailPoints @@ -453,11 +412,6 @@ namespace FailPoints /// transient error (e.g. temporary disk unavailability). Used to test that the refresh task /// reschedules itself after such an error instead of stopping permanently. extern const char merge_tree_refresh_parts_throw_once[]; -======= - extern const int UNKNOWN_TABLE; - extern const int FILE_ALREADY_EXISTS; - extern const int PENDING_MUTATIONS_NOT_ALLOWED; ->>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) } static String getPartNameFromAST(const ASTPtr & partition) @@ -10341,11 +10295,8 @@ void MergeTreeData::writePartLog( const MergeListEntry * merge_entry, std::shared_ptr profile_counters, const Strings & mutation_ids, -<<<<<<< HEAD - const std::map & projections_duration_ms) -======= + const std::map & projections_duration_ms, const ExportsListEntry * exports_entry) ->>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) try { auto table_id = getStorageID(); diff --git a/src/Storages/MergeTree/MergeTreeData.h b/src/Storages/MergeTree/MergeTreeData.h index a5d36ce83375..480226d94fdf 100644 --- a/src/Storages/MergeTree/MergeTreeData.h +++ b/src/Storages/MergeTree/MergeTreeData.h @@ -1396,14 +1396,12 @@ class MergeTreeData : public WithMutableContext, public IStorage, public IBackgr /// Mutex for currently_moving_parts mutable std::mutex moving_parts_mutex; -<<<<<<< HEAD /// Used for streaming queries registration. mutable StreamSubscriptionManager subscription_manager; -======= + mutable std::mutex export_manifests_mutex; std::set export_manifests; ->>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) PinnedPartUUIDsPtr getPinnedPartUUIDs() const; @@ -1501,15 +1499,12 @@ class MergeTreeData : public WithMutableContext, public IStorage, public IBackgr friend class IPartMetadataManager; friend class IMergedBlockOutputStream; // for access to log friend struct DataPartsLock; // for access to shared_parts_list/shared_ranges_in_parts -<<<<<<< HEAD friend class VersionMetadata; // for access to log friend class VersionMetadataOnDisk; // for access to log friend class VersionMetadataOnKeeper; // for access to log friend class MutationsState; // for access to log -======= friend class ExportPartTask; friend class ExportPartFromPartitionExportTask; ->>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) bool require_part_metadata; @@ -1831,13 +1826,9 @@ class MergeTreeData : public WithMutableContext, public IStorage, public IBackgr const DataPartsVector & source_parts, const MergeListEntry * merge_entry, std::shared_ptr profile_counters, -<<<<<<< HEAD const Strings & mutation_ids, - const std::map & projections_duration_ms); -======= - const Strings & mutation_ids = {}, + const std::map & projections_duration_ms, const ExportsListEntry * exports_entry = nullptr); ->>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) /// If part is assigned to merge or mutation (possibly replicated) /// Should be overridden by children, because they can have different diff --git a/src/Storages/ObjectStorage/DataLakes/IDataLakeMetadata.h b/src/Storages/ObjectStorage/DataLakes/IDataLakeMetadata.h index 3ce4f9f887ae..29fe0ceb420e 100644 --- a/src/Storages/ObjectStorage/DataLakes/IDataLakeMetadata.h +++ b/src/Storages/ObjectStorage/DataLakes/IDataLakeMetadata.h @@ -6,11 +6,7 @@ #include #include #include -<<<<<<< HEAD -======= -#include #include ->>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) #include #include #include diff --git a/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergMetadata.cpp b/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergMetadata.cpp index 28158abdec65..6292aa10bd49 100644 --- a/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergMetadata.cpp +++ b/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergMetadata.cpp @@ -81,11 +81,8 @@ #include #include #include -<<<<<<< HEAD #include -======= ->>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) #include #include #include @@ -1631,12 +1628,19 @@ bool IcebergMetadata::commitImportPartitionTransactionImpl( throw Exception(ErrorCodes::BAD_ARGUMENTS, "Failpoint for cleanup enabled"); }); + std::vector data_file_metadata_paths; + data_file_metadata_paths.reserve(data_file_paths.size()); + for (const auto & data_file_path : data_file_paths) + data_file_metadata_paths.push_back(Iceberg::IcebergPathFromMetadata::deserialize(data_file_path)); + generateManifestFile( metadata, partition_columns, partition_values, partition_types, - data_file_paths, + data_file_metadata_paths, + /* data_file_row_counts */ {}, + /* data_file_byte_counts */ {}, std::nullopt, /// per_file_stats is filled, no need for the generic aggregate sample_block, new_snapshot, @@ -1645,6 +1649,7 @@ bool IcebergMetadata::commitImportPartitionTransactionImpl( partition_spec_id, *buffer_manifest_entry, Iceberg::FileContentType::DATA, + /* user_defined_sequence_number */ std::nullopt, per_file_stats); buffer_manifest_entry->finalize(); manifest_lengths += buffer_manifest_entry->count(); diff --git a/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergMetadata.h b/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergMetadata.h index 8b81a9ab9e10..07e002ae42c9 100644 --- a/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergMetadata.h +++ b/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergMetadata.h @@ -214,14 +214,8 @@ class IcebergMetadata : public IDataLakeMetadata void drop(ContextPtr context) override; -<<<<<<< HEAD -======= - std::optional partitionKey(ContextPtr) const override; - std::optional sortingKey(ContextPtr) const override; - Poco::JSON::Object::Ptr getMetadataJSON(ContextPtr local_context) const; ->>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) private: static Iceberg::PersistentTableComponents initializePersistentTableComponents( ObjectStoragePtr object_storage, diff --git a/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergWrites.cpp b/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergWrites.cpp index d9c9cf1e755d..86420c029fbc 100644 --- a/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergWrites.cpp +++ b/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergWrites.cpp @@ -229,9 +229,6 @@ String removeEscapedSlashes(const String & json_str) return result; } -<<<<<<< HEAD -static void extendSchemaForPartitions( -======= IcebergSerializedFileStats readDataFileSidecar( const String & sidecar_storage_path, const ObjectStoragePtr & object_storage, @@ -386,8 +383,7 @@ IcebergSerializedFileStats serializeDataFileStats( return result; } -void extendSchemaForPartitions( ->>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) +static void extendSchemaForPartitions( String & schema, const std::vector & partition_columns, const std::vector & partition_types) @@ -431,11 +427,8 @@ void generateManifestFile( Int64 partition_spec_id, WriteBuffer & buf, Iceberg::FileContentType content_type, -<<<<<<< HEAD - std::optional user_defined_sequence_number) -======= + std::optional user_defined_sequence_number, const std::vector & per_file_stats) ->>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) { Int32 version = metadata->getValue(Iceberg::f_format_version); String schema_representation; @@ -468,10 +461,7 @@ void generateManifestFile( Poco::JSON::Stringifier::stringify(partition_spec->getArray(Iceberg::f_fields), oss_partition_spec); writer.setMetadata(Iceberg::f_partition_spec, oss_partition_spec.str()); writer.setMetadata(Iceberg::f_partition_spec_id, std::to_string(partition_spec_id)); -<<<<<<< HEAD writer.setMetadata(Iceberg::f_format_version, std::to_string(version)); -======= ->>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) for (size_t file_idx = 0; file_idx < data_file_names.size(); ++file_idx) { const auto & data_file_name = data_file_names[file_idx]; @@ -533,19 +523,11 @@ void generateManifestFile( auto schema_element = arr.schema()->leafAt(0); for (const auto & [k, v] : entries) { -<<<<<<< HEAD - avro::GenericDatum record_datum(schema_element); - auto & record = record_datum.value(); - record.field(Iceberg::f_key) = static_cast(field_id); - record.field(Iceberg::f_value) = dump_function(field_id, value); - record_values.value().push_back(record_datum); -======= avro::GenericDatum item(schema_element); auto & item_rec = item.value(); item_rec.field(Iceberg::f_key) = avro::GenericDatum(k); item_rec.field(Iceberg::f_value) = avro::GenericDatum(v); arr.value().push_back(item); ->>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) } }; @@ -572,29 +554,6 @@ void generateManifestFile( write_bytes_map(pf.lower_bounds, Iceberg::f_lower_bounds); write_bytes_map(pf.upper_bounds, Iceberg::f_upper_bounds); -<<<<<<< HEAD - std::unordered_map field_id_to_column_index; - auto field_ids = data_file_statistics->getFieldIds(); - for (size_t i = 0; i < field_ids.size(); ++i) - field_id_to_column_index[field_ids[i]] = i; - - auto dump_fields = [&](size_t field_id, Field value) - { return dumpFieldToBytes(value, sample_block->getDataTypes()[field_id_to_column_index.at(field_id)]); }; - - auto lower_statistics = data_file_statistics->getLowerBounds(); - if (canWriteStatistics(lower_statistics, field_id_to_column_index, sample_block)) - { - set_fields(lower_statistics, Iceberg::f_lower_bounds, dump_fields); - } - auto upper_statistics = data_file_statistics->getUpperBounds(); - if (canWriteStatistics(upper_statistics, field_id_to_column_index, sample_block)) - { - set_fields(upper_statistics, Iceberg::f_upper_bounds, dump_fields); - } - } - data_file.field(Iceberg::f_record_count) = avro::GenericDatum(static_cast(data_file_row_counts[file_idx])); - data_file.field(Iceberg::f_file_size_in_bytes) = avro::GenericDatum(static_cast(data_file_byte_counts[file_idx])); -======= data_file.field(Iceberg::f_record_count) = avro::GenericDatum(pf.record_count); data_file.field(Iceberg::f_file_size_in_bytes) = avro::GenericDatum(pf.file_size_in_bytes); } @@ -614,7 +573,7 @@ void generateManifestFile( { avro::GenericDatum record_datum(schema_element); auto & record = record_datum.value(); - record.field(Iceberg::f_key) = static_cast(field_id); + record.field(Iceberg::f_key) = static_cast(field_id); record.field(Iceberg::f_value) = dump_function(field_id, value); record_values.value().push_back(record_datum); } @@ -636,26 +595,19 @@ void generateManifestFile( auto lower_statistics = data_file_statistics->getLowerBounds(); if (canWriteStatistics(lower_statistics, field_id_to_column_index, sample_block)) + { set_fields(lower_statistics, Iceberg::f_lower_bounds, dump_fields); + } auto upper_statistics = data_file_statistics->getUpperBounds(); if (canWriteStatistics(upper_statistics, field_id_to_column_index, sample_block)) + { set_fields(upper_statistics, Iceberg::f_upper_bounds, dump_fields); + } } - /// Record count and file size from the snapshot summary (aggregate for all files). - auto summary = new_snapshot->getObject(Iceberg::f_summary); - if (summary->has(Iceberg::f_added_records)) - { - data_file.field(Iceberg::f_record_count) = avro::GenericDatum(summary->getValue(Iceberg::f_added_records)); - data_file.field(Iceberg::f_file_size_in_bytes) = avro::GenericDatum(summary->getValue(Iceberg::f_added_files_size)); - } - else - { - data_file.field(Iceberg::f_record_count) = avro::GenericDatum(summary->getValue(Iceberg::f_added_position_deletes)); - data_file.field(Iceberg::f_file_size_in_bytes) = avro::GenericDatum(summary->getValue(Iceberg::f_added_files_size)); - } + data_file.field(Iceberg::f_record_count) = avro::GenericDatum(static_cast(data_file_row_counts[file_idx])); + data_file.field(Iceberg::f_file_size_in_bytes) = avro::GenericDatum(static_cast(data_file_byte_counts[file_idx])); } ->>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) avro::GenericRecord & partition_record = data_file.field("partition").value(); for (size_t i = 0; i < partition_columns.size(); ++i) { diff --git a/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergWrites.h b/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergWrites.h index 3f342fbc5e01..858c67670660 100644 --- a/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergWrites.h +++ b/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergWrites.h @@ -92,11 +92,8 @@ void generateManifestFile( Int64 partition_spec_id, WriteBuffer & buf, Iceberg::FileContentType content_type, -<<<<<<< HEAD - std::optional user_defined_sequence_number = std::nullopt); -======= + std::optional user_defined_sequence_number = std::nullopt, const std::vector & per_file_stats = {}); ->>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) void generateManifestList( const Iceberg::IcebergPathResolver & path_resolver, @@ -110,13 +107,9 @@ void generateManifestList( Iceberg::FileContentType content_type, bool use_previous_snapshots = true); -<<<<<<< HEAD -class IcebergStorageSink final : public SinkToStorage -======= std::string getIcebergExportPartSidecarStoragePath(const String & data_file_storage_path); -class IcebergStorageSink : public SinkToStorage ->>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) +class IcebergStorageSink final : public SinkToStorage { public: IcebergStorageSink( diff --git a/src/Storages/ObjectStorage/DataLakes/Iceberg/MultipleFileWriter.cpp b/src/Storages/ObjectStorage/DataLakes/Iceberg/MultipleFileWriter.cpp index 1f440ae26709..3414bf04e4c5 100644 --- a/src/Storages/ObjectStorage/DataLakes/Iceberg/MultipleFileWriter.cpp +++ b/src/Storages/ObjectStorage/DataLakes/Iceberg/MultipleFileWriter.cpp @@ -26,14 +26,9 @@ MultipleFileWriter::MultipleFileWriter( std::function new_file_path_callback_) : max_data_file_num_rows(max_data_file_num_rows_) , max_data_file_num_bytes(max_data_file_num_bytes_) -<<<<<<< HEAD , schema(schema_) - , stats(schema_) + , aggregate_stats(schema_) , column_mapper(std::make_shared()) -======= - , aggregate_stats(schema) - , current_file_stats(schema) ->>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) , filename_generator(filename_generator_) , path_resolver(path_resolver_) , object_storage(object_storage_) @@ -41,7 +36,6 @@ MultipleFileWriter::MultipleFileWriter( , format_settings(format_settings_) , write_format(std::move(write_format_)) , sample_block(sample_block_) - , schema_fields_json(schema) , new_file_path_callback(std::move(new_file_path_callback_)) { column_mapper->setStorageColumnEncoding(Iceberg::IcebergSchemaProcessor::traverseSchema(schema_)); @@ -52,7 +46,6 @@ void MultipleFileWriter::startNewFile() if (buffer) { finalize(); - current_file_stats = DataFileStatistics(schema_fields_json); } current_file_stats = std::make_shared(schema); @@ -61,14 +54,10 @@ void MultipleFileWriter::startNewFile() auto metadata_path = filename_generator.generateDataFileName(); auto storage_path = path_resolver.resolve(metadata_path); -<<<<<<< HEAD data_file_names.push_back(metadata_path); -======= - data_file_names.push_back(filename.path_in_storage); if (new_file_path_callback) - new_file_path_callback(filename.path_in_storage); + new_file_path_callback(storage_path); ->>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) buffer = object_storage->writeObject( StoredObject(storage_path), WriteMode::Rewrite, std::nullopt, DBMS_DEFAULT_BUFFER_SIZE, context->getWriteSettings()); @@ -93,13 +82,8 @@ void MultipleFileWriter::consume(const Chunk & chunk) output_format->flush(); *current_file_num_rows += chunk.getNumRows(); *current_file_num_bytes += chunk.bytes(); -<<<<<<< HEAD - stats.update(chunk); - current_file_stats->update(chunk); -======= aggregate_stats.update(chunk); - current_file_stats.update(chunk); ->>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) + current_file_stats->update(chunk); } void MultipleFileWriter::finalize() @@ -107,7 +91,6 @@ void MultipleFileWriter::finalize() output_format->flush(); output_format->finalize(); buffer->finalize(); -<<<<<<< HEAD auto buffer_bytes = buffer->count(); UInt64 file_bytes = 0; if (buffer_bytes > 0) @@ -128,31 +111,31 @@ void MultipleFileWriter::finalize() completed_file_stats.push_back(std::move(current_file_stats)); data_file_byte_counts.push_back(file_bytes); data_file_row_counts.push_back(current_file_num_rows.value_or(0)); -======= - const UInt64 file_bytes = buffer->count(); - total_bytes += file_bytes; - per_file_record_counts.push_back(static_cast(*current_file_num_rows)); - per_file_byte_sizes.push_back(static_cast(file_bytes)); - per_file_stats_list.push_back(current_file_stats); } std::vector MultipleFileWriter::getDataFileEntries() const { - chassert(data_file_names.size() == per_file_record_counts.size()); - chassert(data_file_names.size() == per_file_stats_list.size()); + chassert(data_file_names.size() == data_file_row_counts.size()); + chassert(data_file_names.size() == data_file_byte_counts.size()); + chassert(data_file_names.size() == completed_file_stats.size()); std::vector entries; entries.reserve(data_file_names.size()); for (size_t i = 0; i < data_file_names.size(); ++i) + { + std::optional statistics; + if (completed_file_stats[i]) + statistics = *completed_file_stats[i]; + entries.emplace_back( - data_file_names[i], - per_file_record_counts[i], - per_file_byte_sizes[i], - per_file_stats_list[i]); + path_resolver.resolve(data_file_names[i]), + static_cast(data_file_row_counts[i]), + static_cast(data_file_byte_counts[i]), + std::move(statistics)); + } return entries; ->>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) } void MultipleFileWriter::release() diff --git a/src/Storages/ObjectStorage/DataLakes/Iceberg/MultipleFileWriter.h b/src/Storages/ObjectStorage/DataLakes/Iceberg/MultipleFileWriter.h index f807aad7f094..973b7e4932f3 100644 --- a/src/Storages/ObjectStorage/DataLakes/Iceberg/MultipleFileWriter.h +++ b/src/Storages/ObjectStorage/DataLakes/Iceberg/MultipleFileWriter.h @@ -57,25 +57,22 @@ class MultipleFileWriter return aggregate_stats; } -<<<<<<< HEAD const std::vector & getPerFileStatistics() const { return completed_file_stats; } -======= + /// Returns one entry per written data file, with the accurate row count, byte size, /// and per-file column statistics collected during finalization. /// Must be called only after finalize(). std::vector getDataFileEntries() const; ->>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) private: UInt64 max_data_file_num_rows; UInt64 max_data_file_num_bytes; -<<<<<<< HEAD Poco::JSON::Array::Ptr schema; - DataFileStatistics stats; - DataFileStatisticsPtr current_file_stats; + DataFileStatistics aggregate_stats; /// accumulates across all files + DataFileStatisticsPtr current_file_stats; /// accumulates for the current file only std::vector completed_file_stats; /// Pre-built ColumnMapper for `startNewFile`. Traversing the Iceberg schema is invariant /// for the lifetime of the writer, so we compute the mapping once and reuse it across @@ -86,16 +83,6 @@ class MultipleFileWriter std::vector data_file_names; std::vector data_file_row_counts; std::vector data_file_byte_counts; -======= - DataFileStatistics aggregate_stats; /// accumulates across all files - DataFileStatistics current_file_stats; /// accumulates for the current file only - std::optional current_file_num_rows = std::nullopt; - std::optional current_file_num_bytes = std::nullopt; - std::vector data_file_names; - std::vector per_file_record_counts; - std::vector per_file_byte_sizes; - std::vector per_file_stats_list; ->>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) std::unique_ptr buffer; OutputFormatPtr output_format; FileNamesGenerator & filename_generator; @@ -106,7 +93,6 @@ class MultipleFileWriter const String& write_format; SharedHeader sample_block; UInt64 total_bytes = 0; - Poco::JSON::Array::Ptr schema_fields_json; std::function new_file_path_callback; }; diff --git a/src/Storages/ObjectStorage/DataLakes/Iceberg/Utils.cpp b/src/Storages/ObjectStorage/DataLakes/Iceberg/Utils.cpp index 14c6b51fd361..72a73c026b70 100644 --- a/src/Storages/ObjectStorage/DataLakes/Iceberg/Utils.cpp +++ b/src/Storages/ObjectStorage/DataLakes/Iceberg/Utils.cpp @@ -162,11 +162,7 @@ static bool isTemporaryMetadataFile(const String & file_name) return Poco::UUID{}.tryParse(substring); } -<<<<<<< HEAD -static MetadataFileWithInfo getMetadataFileAndVersion(const std::string & path) -======= Iceberg::MetadataFileWithInfo getMetadataFileAndVersion(const std::string & path) ->>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) { String file_name = std::filesystem::path(path).filename(); if (isTemporaryMetadataFile(file_name)) @@ -608,14 +604,10 @@ Poco::Dynamic::Var getAvroType(DataTypePtr type) { switch (type->getTypeId()) { -<<<<<<< HEAD case TypeIndex::UInt8: case TypeIndex::Int8: case TypeIndex::UInt16: case TypeIndex::Int16: -======= - case TypeIndex::UInt16: ->>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) case TypeIndex::UInt32: case TypeIndex::Int32: case TypeIndex::Date: diff --git a/src/Storages/ObjectStorage/StorageObjectStorage.cpp b/src/Storages/ObjectStorage/StorageObjectStorage.cpp index 20caac158ad6..9ffa15fa654d 100644 --- a/src/Storages/ObjectStorage/StorageObjectStorage.cpp +++ b/src/Storages/ObjectStorage/StorageObjectStorage.cpp @@ -986,7 +986,6 @@ void StorageObjectStorage::checkAlterIsPossible(const AlterCommands & commands, configuration->checkAlterIsPossible(object_storage, context, commands); } -<<<<<<< HEAD void StorageObjectStorage::startup() { if (configuration->isBackgroundExecutable()) @@ -1007,6 +1006,4 @@ bool StorageObjectStorage::scheduleDataProcessingJob(BackgroundJobsAssignee & as return configuration->scheduleDataProcessingJob(assignee, *this); } -======= ->>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) } diff --git a/src/Storages/ObjectStorage/StorageObjectStorage.h b/src/Storages/ObjectStorage/StorageObjectStorage.h index fc4dd867de63..233dcf71cc66 100644 --- a/src/Storages/ObjectStorage/StorageObjectStorage.h +++ b/src/Storages/ObjectStorage/StorageObjectStorage.h @@ -7,11 +7,7 @@ #include #include #include -<<<<<<< HEAD -======= -#include #include "Storages/ObjectStorage/ObjectStorageFilePathGenerator.h" ->>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) #include #include #include diff --git a/src/Storages/ObjectStorage/StorageObjectStorageCluster.cpp b/src/Storages/ObjectStorage/StorageObjectStorageCluster.cpp index 4f71616447e9..061fcf839d0d 100644 --- a/src/Storages/ObjectStorage/StorageObjectStorageCluster.cpp +++ b/src/Storages/ObjectStorage/StorageObjectStorageCluster.cpp @@ -127,17 +127,13 @@ StorageObjectStorageCluster::StorageObjectStorageCluster( } metadata.setConstraints(constraints_); -<<<<<<< HEAD - metadata.setVirtuals(VirtualColumnUtils::getVirtualsForFileLikeStorage( -======= if (configuration->partition_strategy) { metadata.partition_key = configuration->partition_strategy->getPartitionKeyDescription(); } - setVirtuals(VirtualColumnUtils::getVirtualsForFileLikeStorage( ->>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) + metadata.setVirtuals(VirtualColumnUtils::getVirtualsForFileLikeStorage( metadata.columns, context_, /* format_settings */std::nullopt, diff --git a/src/Storages/ObjectStorage/StorageObjectStorageConfiguration.h b/src/Storages/ObjectStorage/StorageObjectStorageConfiguration.h index fbb0e7f5c923..8b9ad1c7ebeb 100644 --- a/src/Storages/ObjectStorage/StorageObjectStorageConfiguration.h +++ b/src/Storages/ObjectStorage/StorageObjectStorageConfiguration.h @@ -18,11 +18,8 @@ #include #include #include -<<<<<<< HEAD #include -======= #include ->>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) namespace DB { @@ -327,14 +324,10 @@ class StorageObjectStorageConfiguration /// Whether partition column values are contained in the actual data. /// And alternative is with hive partitioning, when they are contained in file path. bool partition_columns_in_data_file = true; -<<<<<<< HEAD /// Tracks whether `partition_columns_in_data_file` was explicitly provided by the user. /// When false, `initPartitionStrategy` recomputes the default once the effective strategy is known /// (which may have been chosen implicitly via `file_like_engine_default_partition_strategy`). bool partition_columns_in_data_file_was_set = false; - std::shared_ptr partition_strategy; -======= ->>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) protected: void initializeFromParsedArguments(const StorageParsedArguments & parsed_arguments); diff --git a/src/Storages/ObjectStorage/StorageObjectStorageSink.h b/src/Storages/ObjectStorage/StorageObjectStorageSink.h index 7f0732a3476d..0494f3f7e22e 100644 --- a/src/Storages/ObjectStorage/StorageObjectStorageSink.h +++ b/src/Storages/ObjectStorage/StorageObjectStorageSink.h @@ -46,11 +46,7 @@ friend class StorageObjectStorageImporterSink; void cancelBuffers(); }; -<<<<<<< HEAD -class PartitionedStorageObjectStorageSink final : public PartitionedSink -======= -class PartitionedStorageObjectStorageSink : public PartitionedSink::SinkCreator ->>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) +class PartitionedStorageObjectStorageSink final : public PartitionedSink::SinkCreator { public: PartitionedStorageObjectStorageSink( diff --git a/src/Storages/ObjectStorage/StorageObjectStorageSource.cpp b/src/Storages/ObjectStorage/StorageObjectStorageSource.cpp index 1ea2ab8b4ea3..088695b2ee46 100644 --- a/src/Storages/ObjectStorage/StorageObjectStorageSource.cpp +++ b/src/Storages/ObjectStorage/StorageObjectStorageSource.cpp @@ -95,12 +95,7 @@ namespace Setting extern const SettingsBool table_engine_read_through_distributed_cache; extern const SettingsUInt64 s3_path_filter_limit; extern const SettingsBool use_parquet_metadata_cache; -<<<<<<< HEAD -======= - extern const SettingsBool input_format_parquet_use_native_reader_v3; - extern const SettingsBool allow_experimental_iceberg_read_optimization; extern const SettingsBool use_object_storage_list_objects_cache; ->>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) } namespace ErrorCodes diff --git a/src/Storages/ObjectStorage/StorageObjectStorageSource.h b/src/Storages/ObjectStorage/StorageObjectStorageSource.h index ab145708d8c6..93b2ea6e0b10 100644 --- a/src/Storages/ObjectStorage/StorageObjectStorageSource.h +++ b/src/Storages/ObjectStorage/StorageObjectStorageSource.h @@ -12,12 +12,8 @@ #include #include #include -<<<<<<< HEAD -======= #include ->>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) - namespace DB { diff --git a/src/Storages/StorageMergeTree.cpp b/src/Storages/StorageMergeTree.cpp index 3916130d1963..d5af7da7e380 100644 --- a/src/Storages/StorageMergeTree.cpp +++ b/src/Storages/StorageMergeTree.cpp @@ -145,11 +145,8 @@ namespace ErrorCodes extern const int TOO_MANY_PARTS; extern const int PART_IS_LOCKED; extern const int PART_IS_TEMPORARILY_LOCKED; -<<<<<<< HEAD extern const int FAULT_INJECTED; -======= extern const int INCOMPATIBLE_COLUMNS; ->>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) } namespace ActionLocks @@ -249,12 +246,8 @@ void StorageMergeTree::startup() { cleanup_thread.start(); background_operations_assignee.start(); -<<<<<<< HEAD background_streaming_assignee.start(); - startBackgroundMovesIfNeeded(); -======= startBackgroundMoves(); ->>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) startOutdatedAndUnexpectedDataPartsLoadingTask(); startStatisticsCache(); } diff --git a/src/Storages/StorageReplicatedMergeTree.cpp b/src/Storages/StorageReplicatedMergeTree.cpp index db58ba6cb776..adf50f731fd9 100644 --- a/src/Storages/StorageReplicatedMergeTree.cpp +++ b/src/Storages/StorageReplicatedMergeTree.cpp @@ -172,11 +172,9 @@ namespace ProfileEvents extern const Event ReplicaPartialShutdown; extern const Event ReplicatedCoveredPartsInZooKeeperOnStart; extern const Event MergesRejectedByMemoryLimit; -<<<<<<< HEAD extern const Event ZooKeeperWatchTriggeredReplicatedMergeTreeLeaderElection; extern const Event ZooKeeperWatchTriggeredReplicatedMergeTreeReplicaSync; extern const Event ZooKeeperWatchTriggeredReplicatedMergeTreeMutations; -======= extern const Event ExportPartitionZooKeeperRequests; extern const Event ExportPartitionZooKeeperGet; extern const Event ExportPartitionZooKeeperGetChildren; @@ -186,7 +184,6 @@ namespace ProfileEvents extern const Event ExportPartitionZooKeeperRemoveRecursive; extern const Event ExportPartitionZooKeeperMulti; extern const Event ExportPartitionZooKeeperExists; ->>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) } namespace CurrentMetrics diff --git a/src/Storages/StorageReplicatedMergeTree.h b/src/Storages/StorageReplicatedMergeTree.h index c35434749595..58e46d3b019b 100644 --- a/src/Storages/StorageReplicatedMergeTree.h +++ b/src/Storages/StorageReplicatedMergeTree.h @@ -1,10 +1,6 @@ #pragma once -<<<<<<< HEAD -======= -#include #include ->>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) #include #include #include diff --git a/src/Storages/System/StorageSystemMerges.cpp b/src/Storages/System/StorageSystemMerges.cpp index dcb09713bd3c..e152bcdf8a19 100644 --- a/src/Storages/System/StorageSystemMerges.cpp +++ b/src/Storages/System/StorageSystemMerges.cpp @@ -1,14 +1,10 @@ -<<<<<<< HEAD #include #include #include #include #include #include -======= ->>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) #include -#include #include #include diff --git a/src/Storages/System/attachSystemTables.cpp b/src/Storages/System/attachSystemTables.cpp index aa7c12a06a05..5e37b976c611 100644 --- a/src/Storages/System/attachSystemTables.cpp +++ b/src/Storages/System/attachSystemTables.cpp @@ -162,16 +162,12 @@ namespace DB { -<<<<<<< HEAD -void attachSystemTablesServer(ContextPtr context, IDatabase & system_database, bool has_zookeeper, [[maybe_unused]] bool has_keeper_server) -======= namespace ServerSetting { extern const ServerSettingsBool allow_experimental_export_merge_tree_partition; } -void attachSystemTablesServer(ContextPtr context, IDatabase & system_database, bool has_zookeeper) ->>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) +void attachSystemTablesServer(ContextPtr context, IDatabase & system_database, bool has_zookeeper, [[maybe_unused]] bool has_keeper_server) { auto component_guard = Coordination::setCurrentComponent("attachSystemTablesServer"); attachNoDescription(context, system_database, "one", "This table contains a single row with a single dummy UInt8 column containing the value 0. Used when the table is not specified explicitly, for example in queries like `SELECT 1`."); diff --git a/tests/queries/0_stateless/03413_experimental_settings_cannot_be_enabled_by_default.sql b/tests/queries/0_stateless/03413_experimental_settings_cannot_be_enabled_by_default.sql index 075cb9552fc3..a1517d410ad1 100644 --- a/tests/queries/0_stateless/03413_experimental_settings_cannot_be_enabled_by_default.sql +++ b/tests/queries/0_stateless/03413_experimental_settings_cannot_be_enabled_by_default.sql @@ -4,14 +4,5 @@ -- However, some settings in the experimental tier are meant to control another experimental feature, and then they can be enabled as long as the feature itself is disabled. -- These are in the exceptions list inside NOT IN. -<<<<<<< HEAD -SELECT name, value FROM system.settings WHERE tier = 'Experimental' AND type = 'Bool' AND value != '0' AND name NOT IN ('throw_on_unsupported_query_inside_transaction', 'ai_function_throw_on_error', 'ai_function_throw_on_quota_exceeded'); -======= -SELECT name, value FROM system.settings WHERE tier = 'Experimental' AND type = 'Bool' AND value != '0' AND name NOT IN ( - 'throw_on_unsupported_query_inside_transaction', --- turned ON for Altinity Antalya builds specifically - 'allow_experimental_iceberg_read_optimization', - 'allow_experimental_export_merge_tree_part' -); ->>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) +SELECT name, value FROM system.settings WHERE tier = 'Experimental' AND type = 'Bool' AND value != '0' AND name NOT IN ('throw_on_unsupported_query_inside_transaction', 'ai_function_throw_on_error', 'ai_function_throw_on_quota_exceeded', 'allow_experimental_export_merge_tree_part'); SELECT name, value FROM system.merge_tree_settings WHERE tier = 'Experimental' AND type = 'Bool' AND value != '0' AND name NOT IN ('remove_rolled_back_parts_immediately'); diff --git a/tests/queries/0_stateless/03745_system_background_schedule_pool.reference b/tests/queries/0_stateless/03745_system_background_schedule_pool.reference index ed65c4991867..be78f66a355c 100644 --- a/tests/queries/0_stateless/03745_system_background_schedule_pool.reference +++ b/tests/queries/0_stateless/03745_system_background_schedule_pool.reference @@ -1,10 +1,7 @@ 1 buffer_flush default test_buffer_03745 1 StorageBuffer (default.test_buffer_03745)/Bg schedule default test_merge_tree_03745 1 BackgroundJobsAssignee:DataProcessing -<<<<<<< HEAD -schedule default test_merge_tree_03745 1 BackgroundJobsAssignee:Streaming -======= schedule default test_merge_tree_03745 1 BackgroundJobsAssignee:Moving ->>>>>>> 9a9645c97cc (Merge pull request #1718 from Altinity/feature/antalya-26.3/apassos-3) +schedule default test_merge_tree_03745 1 BackgroundJobsAssignee:Streaming schedule default test_merge_tree_03745 1 default.test_merge_tree_03745 (CleanupThread) distributed default test_distributed_03745 1 default.test_distributed_03745.DistributedInsertQueue.default/Bg From 54477fd073171d39365803b644f8d1aaf3fbe51c Mon Sep 17 00:00:00 2001 From: Andrey Zvonov <32552679+zvonand@users.noreply.github.com> Date: Wed, 6 May 2026 21:16:58 +0200 Subject: [PATCH 03/43] Cherry-pick of https://github.com/Altinity/ClickHouse/pull/1646 with unresolved conflict markers (resolution in next commit) --- Original cherry-pick message follows: Merge pull request #1646 from Altinity/frontport/antalya-26.3/fix_remote_calls 26.3 Antalya port - fixes for s3Cluster distributed calls # Conflicts: # src/Planner/Planner.cpp # src/Processors/QueryPlan/ObjectFilterStep.cpp # src/Processors/QueryPlan/ObjectFilterStep.h # src/Processors/QueryPlan/QueryPlanStepRegistry.cpp # src/Processors/QueryPlan/ReadFromRemote.cpp # src/QueryPipeline/RemoteQueryExecutor.h # src/Storages/ObjectStorage/StorageObjectStorageCluster.cpp # tests/integration/test_database_iceberg/test.py # tests/integration/test_s3_cluster/test.py --- src/Core/Settings.cpp | 16 + src/Core/Settings.h | 1 + src/Core/SettingsChangesHistory.cpp | 3 + src/Core/SettingsEnums.cpp | 5 + src/Core/SettingsEnums.h | 10 + .../ClusterProxy/executeQuery.cpp | 1 + src/Interpreters/Context.cpp | 5 +- src/Interpreters/PreparedSets.cpp | 26 +- src/Interpreters/PreparedSets.h | 7 + src/Planner/Planner.cpp | 31 ++ src/Planner/PlannerJoinTree.cpp | 4 +- src/Processors/QueryPlan/ObjectFilterStep.cpp | 7 + src/Processors/QueryPlan/ObjectFilterStep.h | 14 + .../QueryPlan/QueryPlanStepRegistry.cpp | 7 + src/Processors/QueryPlan/ReadFromRemote.cpp | 12 +- src/Processors/QueryPlan/ReadFromRemote.h | 2 + src/QueryPipeline/RemoteQueryExecutor.cpp | 11 +- src/QueryPipeline/RemoteQueryExecutor.h | 15 + src/Storages/IStorageCluster.cpp | 276 ++++++++++++- src/Storages/IStorageCluster.h | 10 + .../StorageObjectStorageCluster.cpp | 4 + .../extractTableFunctionFromSelectQuery.cpp | 32 +- .../extractTableFunctionFromSelectQuery.h | 4 + .../integration/test_database_iceberg/test.py | 149 ++++++- tests/integration/test_s3_cluster/test.py | 363 ++++++++++++++++++ .../test_cluster_joins.py | 154 ++++++++ .../test_cluster_table_function.py | 18 + .../0_stateless/02126_dist_desc.sql.j2 | 2 +- .../03550_analyzer_remote_view_columns.sql | 2 +- ...0_analyzer_distributed_global_in.reference | 2 +- .../03620_analyzer_distributed_global_in.sql | 2 +- 31 files changed, 1176 insertions(+), 19 deletions(-) create mode 100644 tests/integration/test_storage_iceberg_with_spark/test_cluster_joins.py diff --git a/src/Core/Settings.cpp b/src/Core/Settings.cpp index d17025dd1cc1..6ce6575d8d5c 100644 --- a/src/Core/Settings.cpp +++ b/src/Core/Settings.cpp @@ -2120,6 +2120,22 @@ Possible values: - `global` — Replaces the `IN`/`JOIN` query with `GLOBAL IN`/`GLOBAL JOIN.` - `allow` — Allows the use of these types of subqueries. )", IMPORTANT) \ + DECLARE(ObjectStorageClusterJoinMode, object_storage_cluster_join_mode, ObjectStorageClusterJoinMode::ALLOW, R"( +Changes the behaviour of object storage cluster function or table. + +ClickHouse applies this setting when the query contains the product of object storage cluster function or table, i.e. when the query for a object storage cluster function or table contains a non-GLOBAL subquery for the object storage cluster function or table. + +Restrictions: + +- Only applied for JOIN subqueries. +- Only if the FROM section uses a object storage cluster function or table. + +Possible values: + +- `local` — Replaces the database and table in the subquery with local ones for the destination server (shard), leaving the normal `IN`/`JOIN.` +- `global` — Unsupported for now. Replaces the `IN`/`JOIN` query with `GLOBAL IN`/`GLOBAL JOIN.` +- `allow` — Default value. Allows the use of these types of subqueries. +)", 0) \ \ DECLARE(UInt64, max_concurrent_queries_for_all_users, 0, R"( Throw exception if the value of this setting is less or equal than the current number of simultaneously processed queries. diff --git a/src/Core/Settings.h b/src/Core/Settings.h index 49a067390657..277b1dfda86d 100644 --- a/src/Core/Settings.h +++ b/src/Core/Settings.h @@ -60,6 +60,7 @@ class WriteBuffer; M(CLASS_NAME, DistributedCachePoolBehaviourOnLimit) /* Cloud only */ \ M(CLASS_NAME, DistributedDDLOutputMode) \ M(CLASS_NAME, DistributedProductMode) \ + M(CLASS_NAME, ObjectStorageClusterJoinMode) \ M(CLASS_NAME, Double) \ M(CLASS_NAME, EscapingRule) \ M(CLASS_NAME, Float) \ diff --git a/src/Core/SettingsChangesHistory.cpp b/src/Core/SettingsChangesHistory.cpp index c7d722164f22..786b2ca773bc 100644 --- a/src/Core/SettingsChangesHistory.cpp +++ b/src/Core/SettingsChangesHistory.cpp @@ -197,6 +197,9 @@ const VersionToSettingsChangesMap & getSettingsChangesHistory() {"ai_function_throw_on_quota_exceeded", true, true, "New setting"}, {"variant_throw_on_type_mismatch", true, true, "New setting to control type mismatch behavior in default Variant implementation"}, {"dynamic_throw_on_type_mismatch", true, true, "New setting to control type mismatch behavior in default Dynamic implementation"}, + addSettingsChanges(settings_changes_history, "26.3.1.20001.altinityantalya", + { + {"object_storage_cluster_join_mode", "allow", "allow", "New setting"}, }); addSettingsChanges(settings_changes_history, "26.3", { diff --git a/src/Core/SettingsEnums.cpp b/src/Core/SettingsEnums.cpp index 41e291f24820..dcb252f1d26d 100644 --- a/src/Core/SettingsEnums.cpp +++ b/src/Core/SettingsEnums.cpp @@ -99,6 +99,11 @@ IMPLEMENT_SETTING_ENUM(DistributedProductMode, ErrorCodes::UNKNOWN_DISTRIBUTED_P {"global", DistributedProductMode::GLOBAL}, {"allow", DistributedProductMode::ALLOW}}) +IMPLEMENT_SETTING_ENUM(ObjectStorageClusterJoinMode, ErrorCodes::BAD_ARGUMENTS, + {{"local", ObjectStorageClusterJoinMode::LOCAL}, + {"global", ObjectStorageClusterJoinMode::GLOBAL}, + {"allow", ObjectStorageClusterJoinMode::ALLOW}}) + IMPLEMENT_SETTING_ENUM(QueryResultCacheNondeterministicFunctionHandling, ErrorCodes::BAD_ARGUMENTS, {{"throw", QueryResultCacheNondeterministicFunctionHandling::Throw}, diff --git a/src/Core/SettingsEnums.h b/src/Core/SettingsEnums.h index 841e6132020e..1af70b226b84 100644 --- a/src/Core/SettingsEnums.h +++ b/src/Core/SettingsEnums.h @@ -165,6 +165,16 @@ enum class DistributedProductMode : uint8_t DECLARE_SETTING_ENUM(DistributedProductMode) +/// The setting for executing object storage cluster function or table JOIN sections. +enum class ObjectStorageClusterJoinMode : uint8_t +{ + LOCAL, /// Convert to local query + GLOBAL, /// Convert to global query + ALLOW /// Enable +}; + +DECLARE_SETTING_ENUM(ObjectStorageClusterJoinMode) + /// How the query result cache handles queries with non-deterministic functions, e.g. now() enum class QueryResultCacheNondeterministicFunctionHandling : uint8_t { diff --git a/src/Interpreters/ClusterProxy/executeQuery.cpp b/src/Interpreters/ClusterProxy/executeQuery.cpp index df5a73c56b4d..4d17ed131498 100644 --- a/src/Interpreters/ClusterProxy/executeQuery.cpp +++ b/src/Interpreters/ClusterProxy/executeQuery.cpp @@ -501,6 +501,7 @@ void executeQuery( std::move(unavailable_shard_tracker)); read_from_remote->setStepDescription("Read from remote replica"); + read_from_remote->setIsRemoteFunction(is_remote_function); plan->addStep(std::move(read_from_remote)); plan->addInterpreterContext(new_context); plans.emplace_back(std::move(plan)); diff --git a/src/Interpreters/Context.cpp b/src/Interpreters/Context.cpp index de3e38e81083..78af67b69f99 100644 --- a/src/Interpreters/Context.cpp +++ b/src/Interpreters/Context.cpp @@ -3461,8 +3461,11 @@ void Context::setCurrentQueryId(const String & query_id) client_info.current_query_id = query_id_to_set; - if (client_info.query_kind == ClientInfo::QueryKind::INITIAL_QUERY) + if (client_info.query_kind == ClientInfo::QueryKind::INITIAL_QUERY + && (getApplicationType() != ApplicationType::SERVER || client_info.initial_query_id.empty())) + { client_info.initial_query_id = client_info.current_query_id; + } } void Context::killCurrentQuery() const diff --git a/src/Interpreters/PreparedSets.cpp b/src/Interpreters/PreparedSets.cpp index 1ad55f36ec5c..e28c8bfd5d03 100644 --- a/src/Interpreters/PreparedSets.cpp +++ b/src/Interpreters/PreparedSets.cpp @@ -284,6 +284,12 @@ SetAndKeyPtr FutureSetFromSubquery::detachSetAndKey() } SetPtr FutureSetFromSubquery::get() const +{ + std::lock_guard lock(mutex); + return get_unsafe(); +} + +SetPtr FutureSetFromSubquery::get_unsafe() const { if (set_and_key->set != nullptr && set_and_key->set->isCreated()) return set_and_key->set; @@ -293,6 +299,7 @@ SetPtr FutureSetFromSubquery::get() const void FutureSetFromSubquery::setQueryPlan(std::unique_ptr source_) { + std::lock_guard lock(mutex); source = std::move(source_); set_and_key->set->setHeader(source->getCurrentHeader()->getColumnsWithTypeAndName()); } @@ -344,6 +351,8 @@ void FutureSetFromSubquery::buildExternalTableFromInplaceSet(StoragePtr external void FutureSetFromSubquery::setExternalTable(StoragePtr external_table_) { + std::lock_guard lock(mutex); + if (set_and_key->set->isCreated()) { if (!set_and_key->set->hasExplicitSetElements()) @@ -357,12 +366,19 @@ void FutureSetFromSubquery::setExternalTable(StoragePtr external_table_) DataTypes FutureSetFromSubquery::getTypes() const { + std::lock_guard lock(mutex); return set_and_key->set->getElementsTypes(); } FutureSet::Hash FutureSetFromSubquery::getHash() const { return hash; } std::unique_ptr FutureSetFromSubquery::build(const SizeLimits & network_transfer_limits, const PreparedSetsCachePtr & prepared_sets_cache) +{ + std::lock_guard lock(mutex); + return build_unsafe(network_transfer_limits, prepared_sets_cache); +} + +std::unique_ptr FutureSetFromSubquery::build_unsafe(const SizeLimits & network_transfer_limits, const PreparedSetsCachePtr & prepared_sets_cache) { if (set_and_key->set->isCreated()) return nullptr; @@ -384,6 +400,8 @@ std::unique_ptr FutureSetFromSubquery::build(const SizeLimits & netwo void FutureSetFromSubquery::buildSetInplace(const ContextPtr & context) { + std::lock_guard lock(mutex); + if (external_table_set) external_table_set->buildSetInplace(context); @@ -396,7 +414,7 @@ void FutureSetFromSubquery::buildSetInplace(const ContextPtr & context) SizeLimits network_transfer_limits(settings[Setting::max_rows_to_transfer], settings[Setting::max_bytes_to_transfer], settings[Setting::transfer_overflow_mode]); auto prepared_sets_cache = context->getPreparedSetsCache(); - auto plan = build(network_transfer_limits, prepared_sets_cache); + auto plan = build_unsafe(network_transfer_limits, prepared_sets_cache); if (!plan) return; @@ -422,7 +440,9 @@ SetPtr FutureSetFromSubquery::buildOrderedSetInplace(const ContextPtr & context) if (!context->getSettingsRef()[Setting::use_index_for_in_with_subqueries]) return nullptr; - if (auto set = get()) + std::lock_guard lock(mutex); + + if (auto set = get_unsafe()) { if (set->hasExplicitSetElements()) return set; @@ -449,7 +469,7 @@ SetPtr FutureSetFromSubquery::buildOrderedSetInplace(const ContextPtr & context) SizeLimits network_transfer_limits(settings[Setting::max_rows_to_transfer], settings[Setting::max_bytes_to_transfer], settings[Setting::transfer_overflow_mode]); auto prepared_sets_cache = context->getPreparedSetsCache(); - auto plan = build(network_transfer_limits, prepared_sets_cache); + auto plan = build_unsafe(network_transfer_limits, prepared_sets_cache); if (!plan) return nullptr; diff --git a/src/Interpreters/PreparedSets.h b/src/Interpreters/PreparedSets.h index 4322972e597b..6e06ef9517a2 100644 --- a/src/Interpreters/PreparedSets.h +++ b/src/Interpreters/PreparedSets.h @@ -191,6 +191,11 @@ class FutureSetFromSubquery final : public FutureSet QueryPlan * getQueryPlan() { return source.get(); } private: + SetPtr get_unsafe() const; + std::unique_ptr build_unsafe( + const SizeLimits & network_transfer_limits, + const PreparedSetsCachePtr & prepared_sets_cache); + Hash hash; ASTPtr ast; SetAndKeyPtr set_and_key; @@ -198,6 +203,8 @@ class FutureSetFromSubquery final : public FutureSet std::unique_ptr source; QueryTreeNodePtr query_tree; + + mutable std::mutex mutex; }; using FutureSetFromSubqueryPtr = std::shared_ptr; diff --git a/src/Planner/Planner.cpp b/src/Planner/Planner.cpp index d8fdc41f648e..e0304721e792 100644 --- a/src/Planner/Planner.cpp +++ b/src/Planner/Planner.cpp @@ -41,7 +41,11 @@ #include #include #include +<<<<<<< HEAD #include +======= +#include +>>>>>>> d9d3710bd9b (Merge pull request #1646 from Altinity/frontport/antalya-26.3/fix_remote_calls) #include #include @@ -59,6 +63,7 @@ #include #include #include +#include #include @@ -168,6 +173,7 @@ namespace Setting extern const SettingsBool serialize_string_in_memory_with_zero_byte; extern const SettingsString temporary_files_codec; extern const SettingsNonZeroUInt64 temporary_files_buffer_size; + extern const SettingsBool use_hive_partitioning; } namespace ServerSetting @@ -625,6 +631,21 @@ ALWAYS_INLINE void addFilterStep( query_plan.addStep(std::move(where_step)); } +template +ALWAYS_INLINE void addObjectFilterStep( + QueryPlan & query_plan, + FilterAnalysisResult & filter_analysis_result, + const char (&step_description)[size]) +{ + auto actions = std::move(filter_analysis_result.filter_actions->dag); + + auto where_step = std::make_unique(query_plan.getCurrentHeader(), + std::move(actions), + filter_analysis_result.filter_column_name); + where_step->setStepDescription(step_description); + query_plan.addStep(std::move(where_step)); +} + Aggregator::Params getAggregatorParams(const PlannerContextPtr & planner_context, const AggregationAnalysisResult & aggregation_analysis_result, const QueryAnalysisResult & query_analysis_result, @@ -2442,6 +2463,16 @@ void Planner::buildPlanForQueryNode() if (query_processing_info.isSecondStage() || query_processing_info.isFromAggregationState()) { + if (settings[Setting::use_hive_partitioning] + && !query_processing_info.isFirstStage() + && expression_analysis_result.hasWhere()) + { + if (typeid_cast(query_plan.getRootNode()->step.get())) + { + addObjectFilterStep(query_plan, expression_analysis_result.getWhere(), "WHERE"); + } + } + if (query_processing_info.isFromAggregationState()) { /// Aggregation was performed on remote shards diff --git a/src/Planner/PlannerJoinTree.cpp b/src/Planner/PlannerJoinTree.cpp index 5e6bdb620003..dd4ef0a462a1 100644 --- a/src/Planner/PlannerJoinTree.cpp +++ b/src/Planner/PlannerJoinTree.cpp @@ -1551,7 +1551,9 @@ JoinTreeQueryPlan buildQueryPlanForTableExpression(QueryTreeNodePtr table_expres /// Overall, IStorage::read -> FetchColumns returns normal column names (except Distributed, which is inconsistent) /// Interpreter::getQueryPlan -> FetchColumns returns identifiers (why?) and this the reason for the bug ^ in Distributed /// Hopefully there is no other case when we read from Distributed up to FetchColumns. - if (table_node && table_node->getStorage()->isRemote() && select_query_options.to_stage == QueryProcessingStage::FetchColumns) + if (table_node && table_node->getStorage()->isRemote()) + updated_actions_dag_outputs.push_back(output_node); + else if (table_function_node && table_function_node->getStorage()->isRemote()) updated_actions_dag_outputs.push_back(output_node); } else diff --git a/src/Processors/QueryPlan/ObjectFilterStep.cpp b/src/Processors/QueryPlan/ObjectFilterStep.cpp index cfa44162feaf..f5d2d3ee667b 100644 --- a/src/Processors/QueryPlan/ObjectFilterStep.cpp +++ b/src/Processors/QueryPlan/ObjectFilterStep.cpp @@ -15,7 +15,11 @@ namespace ErrorCodes } ObjectFilterStep::ObjectFilterStep( +<<<<<<< HEAD const SharedHeader & input_header_, +======= + SharedHeader input_header_, +>>>>>>> d9d3710bd9b (Merge pull request #1646 from Altinity/frontport/antalya-26.3/fix_remote_calls) ActionsDAG actions_dag_, String filter_column_name_) : actions_dag(std::move(actions_dag_)) @@ -55,7 +59,10 @@ std::unique_ptr ObjectFilterStep::deserialize(Deserialization & return std::make_unique(ctx.input_headers.front(), std::move(actions_dag), std::move(filter_column_name)); } +<<<<<<< HEAD void registerObjectFilterStep(QueryPlanStepRegistry & registry); +======= +>>>>>>> d9d3710bd9b (Merge pull request #1646 from Altinity/frontport/antalya-26.3/fix_remote_calls) void registerObjectFilterStep(QueryPlanStepRegistry & registry) { registry.registerStep("ObjectFilter", ObjectFilterStep::deserialize); diff --git a/src/Processors/QueryPlan/ObjectFilterStep.h b/src/Processors/QueryPlan/ObjectFilterStep.h index f0030a6624e6..f2cb3db195cb 100644 --- a/src/Processors/QueryPlan/ObjectFilterStep.h +++ b/src/Processors/QueryPlan/ObjectFilterStep.h @@ -5,31 +5,45 @@ namespace DB { +<<<<<<< HEAD /// Implements WHERE condition only to filter objects in object storage /// Difference with FilterStep is that ObjectFilterStep is added only for distributed calls /// (table functions like `s3Cluster`) and is used only to filter objects, /// not to filter data after reading, because initiator can have not this column /// In query like `SELECT count() FROM s3Cluster('cluster', ...) WHERE key=42` /// column `key` does not exist in blocks getting from cluster replicas. +======= +/// Implements WHERE operation. +>>>>>>> d9d3710bd9b (Merge pull request #1646 from Altinity/frontport/antalya-26.3/fix_remote_calls) class ObjectFilterStep : public IQueryPlanStep { public: ObjectFilterStep( +<<<<<<< HEAD const SharedHeader & input_header_, +======= + SharedHeader input_header_, +>>>>>>> d9d3710bd9b (Merge pull request #1646 from Altinity/frontport/antalya-26.3/fix_remote_calls) ActionsDAG actions_dag_, String filter_column_name_); String getName() const override { return "ObjectFilter"; } QueryPipelineBuilderPtr updatePipeline(QueryPipelineBuilders pipelines, const BuildQueryPipelineSettings & settings) override; +<<<<<<< HEAD bool hasCorrelatedExpressions() const override { return actions_dag.hasCorrelatedColumns(); } +======= +>>>>>>> d9d3710bd9b (Merge pull request #1646 from Altinity/frontport/antalya-26.3/fix_remote_calls) const ActionsDAG & getExpression() const { return actions_dag; } ActionsDAG & getExpression() { return actions_dag; } const String & getFilterColumnName() const { return filter_column_name; } void serialize(Serialization & ctx) const override; +<<<<<<< HEAD bool isSerializable() const override { return true; } +======= +>>>>>>> d9d3710bd9b (Merge pull request #1646 from Altinity/frontport/antalya-26.3/fix_remote_calls) static std::unique_ptr deserialize(Deserialization & ctx); diff --git a/src/Processors/QueryPlan/QueryPlanStepRegistry.cpp b/src/Processors/QueryPlan/QueryPlanStepRegistry.cpp index 08fcf3126283..d434c7cda4ef 100644 --- a/src/Processors/QueryPlan/QueryPlanStepRegistry.cpp +++ b/src/Processors/QueryPlan/QueryPlanStepRegistry.cpp @@ -56,6 +56,7 @@ void registerFilterStep(QueryPlanStepRegistry & registry); void registerTotalsHavingStep(QueryPlanStepRegistry & registry); void registerExtremesStep(QueryPlanStepRegistry & registry); void registerJoinStep(QueryPlanStepRegistry & registry); +<<<<<<< HEAD void registerShuffleSendStep(QueryPlanStepRegistry & registry); void registerShuffleReceiveStep(QueryPlanStepRegistry & registry); void registerGatherSendStep(QueryPlanStepRegistry & registry); @@ -63,6 +64,9 @@ void registerGatherReceiveStep(QueryPlanStepRegistry & registry); void registerBroadcastSendStep(QueryPlanStepRegistry & registry); void registerBroadcastReceiveStep(QueryPlanStepRegistry & registry); void registerReadFromMergeTreeStep(QueryPlanStepRegistry & registry); +======= +void registerObjectFilterStep(QueryPlanStepRegistry & registry); +>>>>>>> d9d3710bd9b (Merge pull request #1646 from Altinity/frontport/antalya-26.3/fix_remote_calls) void registerReadFromTableStep(QueryPlanStepRegistry & registry); void registerReadFromTableFunctionStep(QueryPlanStepRegistry & registry); @@ -109,9 +113,12 @@ void QueryPlanStepRegistry::registerPlanSteps() registerReadFromTableFunctionStep(registry); registerBuildRuntimeFilterStep(registry); registerObjectFilterStep(registry); +<<<<<<< HEAD registerReadFromStorageStep(registry); +======= +>>>>>>> d9d3710bd9b (Merge pull request #1646 from Altinity/frontport/antalya-26.3/fix_remote_calls) } } diff --git a/src/Processors/QueryPlan/ReadFromRemote.cpp b/src/Processors/QueryPlan/ReadFromRemote.cpp index 6b47e8ca4ffa..eba29d63b116 100644 --- a/src/Processors/QueryPlan/ReadFromRemote.cpp +++ b/src/Processors/QueryPlan/ReadFromRemote.cpp @@ -610,7 +610,8 @@ void ReadFromRemote::addLazyPipe( my_stage = stage, my_storage = storage, add_agg_info, add_totals, add_extremes, async_read, async_query_sending, query_tree = shard.query_tree, planner_context = shard.planner_context, - pushed_down_filters, parallel_marshalling_threads]() mutable + pushed_down_filters, parallel_marshalling_threads, + my_is_remote_function = is_remote_function]() mutable -> QueryPipelineBuilder { auto current_settings = my_context->getSettingsRef(); @@ -709,7 +710,12 @@ void ReadFromRemote::addLazyPipe( auto remote_query_executor = std::make_shared( std::move(connections), query_string, header, my_context, my_throttler, my_scalars, my_external_tables, stage_to_use, my_shard.query_plan, /*extension=*/std::nullopt, my_shard.shard_info.pool); +<<<<<<< HEAD remote_query_executor->setDistributedFanout(my_distributed_fanout); +======= + remote_query_executor->setRemoteFunction(my_is_remote_function); + remote_query_executor->setShardCount(my_shard_count); +>>>>>>> d9d3710bd9b (Merge pull request #1646 from Altinity/frontport/antalya-26.3/fix_remote_calls) auto pipe = createRemoteSourcePipe( remote_query_executor, add_agg_info, add_totals, add_extremes, async_read, async_query_sending, parallel_marshalling_threads); @@ -804,6 +810,8 @@ void ReadFromRemote::addPipe( remote_query_executor->setPoolMode(PoolMode::GET_ONE); remote_query_executor->setDistributedFanout(shards.size() * shard.shard_info.per_replica_pools.size()); remote_query_executor->setUnavailableShardTracker(unavailable_shard_tracker); + remote_query_executor->setRemoteFunction(is_remote_function); + remote_query_executor->setShardCount(shard_count); if (!table_func_ptr) remote_query_executor->setMainTable(shard.main_table ? shard.main_table : main_table); @@ -834,6 +842,8 @@ void ReadFromRemote::addPipe( remote_query_executor->setLogger(log); remote_query_executor->setDistributedFanout(shards.size()); remote_query_executor->setUnavailableShardTracker(unavailable_shard_tracker); + remote_query_executor->setRemoteFunction(is_remote_function); + remote_query_executor->setShardCount(shard_count); if (context->canUseTaskBasedParallelReplicas() || parallel_replicas_disabled) { diff --git a/src/Processors/QueryPlan/ReadFromRemote.h b/src/Processors/QueryPlan/ReadFromRemote.h index 7ebc0f69d072..2e739e032be3 100644 --- a/src/Processors/QueryPlan/ReadFromRemote.h +++ b/src/Processors/QueryPlan/ReadFromRemote.h @@ -51,6 +51,7 @@ class ReadFromRemote final : public SourceStepWithFilterBase void enableMemoryBoundMerging(); void enforceAggregationInOrder(const SortDescription & sort_description); + void setIsRemoteFunction(bool is_remote_function_ = true) { is_remote_function = is_remote_function_; } bool hasSerializedPlan() const; @@ -69,6 +70,7 @@ class ReadFromRemote final : public SourceStepWithFilterBase const String cluster_name; UnavailableShardTrackerPtr unavailable_shard_tracker; std::optional priority_func_factory; + bool is_remote_function = false; Pipes addPipes(const ClusterProxy::SelectStreamFactory::Shards & used_shards, const SharedHeader & out_header); diff --git a/src/QueryPipeline/RemoteQueryExecutor.cpp b/src/QueryPipeline/RemoteQueryExecutor.cpp index d4a9a3a9e876..c544105c657b 100644 --- a/src/QueryPipeline/RemoteQueryExecutor.cpp +++ b/src/QueryPipeline/RemoteQueryExecutor.cpp @@ -450,7 +450,16 @@ void RemoteQueryExecutor::sendQueryUnlocked(ClientInfo::QueryKind query_kind, As auto timeouts = ConnectionTimeouts::getTCPTimeoutsWithFailover(settings); ClientInfo modified_client_info = context->getClientInfo(); - modified_client_info.query_kind = query_kind; + + /// Doesn't support now "remote('1.1.1.{1,2}')"" + if (is_remote_function && (shard_count == 1)) + { + modified_client_info.setInitialQuery(); + modified_client_info.client_name = "ClickHouse server"; + modified_client_info.interface = ClientInfo::Interface::TCP; + } + else + modified_client_info.query_kind = query_kind; if (extension) modified_client_info.collaborate_with_initiator = true; diff --git a/src/QueryPipeline/RemoteQueryExecutor.h b/src/QueryPipeline/RemoteQueryExecutor.h index df5d59583d56..855b9974547f 100644 --- a/src/QueryPipeline/RemoteQueryExecutor.h +++ b/src/QueryPipeline/RemoteQueryExecutor.h @@ -215,7 +215,13 @@ class RemoteQueryExecutor void setUnavailableShardTracker(UnavailableShardTrackerPtr tracker) { unavailable_shard_tracker = std::move(tracker); } +<<<<<<< HEAD void setDistributedFanout(size_t total_connections) { distributed_fanout = total_connections; } +======= + void setRemoteFunction(bool is_remote_function_ = true) { is_remote_function = is_remote_function_; } + + void setShardCount(UInt32 shard_count_) { shard_count = shard_count_; } +>>>>>>> d9d3710bd9b (Merge pull request #1646 from Altinity/frontport/antalya-26.3/fix_remote_calls) const Block & getHeader() const { return *header; } const SharedHeader & getSharedHeader() const { return header; } @@ -309,6 +315,15 @@ class RemoteQueryExecutor bool packet_in_progress = false; #endif +<<<<<<< HEAD +======= + bool is_remote_function = false; + UInt32 shard_count = 0; + + /// Parts uuids, collected from remote replicas + std::vector duplicated_part_uuids; + +>>>>>>> d9d3710bd9b (Merge pull request #1646 from Altinity/frontport/antalya-26.3/fix_remote_calls) PoolMode pool_mode = PoolMode::GET_MANY; StorageID main_table = StorageID::createEmpty(); diff --git a/src/Storages/IStorageCluster.cpp b/src/Storages/IStorageCluster.cpp index 761191632566..90c887dd3eb4 100644 --- a/src/Storages/IStorageCluster.cpp +++ b/src/Storages/IStorageCluster.cpp @@ -19,6 +19,15 @@ #include #include #include +#include +#include +#include +#include +#include +#include +#include +#include +#include #include @@ -43,6 +52,12 @@ namespace Setting extern const SettingsBool parallel_replicas_local_plan; extern const SettingsString cluster_for_parallel_replicas; extern const SettingsNonZeroUInt64 max_parallel_replicas; + extern const SettingsObjectStorageClusterJoinMode object_storage_cluster_join_mode; +} + +namespace ErrorCodes +{ + extern const int LOGICAL_ERROR; } namespace ErrorCodes @@ -85,6 +100,207 @@ void ReadFromCluster::createExtension(const ActionsDAG::Node * predicate) getStorageSnapshot()->metadata); } +namespace +{ + +/* +Helping class to find in query tree first node of required type +*/ +class SearcherVisitor : public InDepthQueryTreeVisitorWithContext +{ +public: + using Base = InDepthQueryTreeVisitorWithContext; + using Base::Base; + + explicit SearcherVisitor(std::unordered_set types_, ContextPtr context) : Base(context), types(types_) {} + + bool needChildVisit(QueryTreeNodePtr & /*parent*/, QueryTreeNodePtr & /*child*/) + { + return getSubqueryDepth() <= 2 && !passed_node; + } + + void enterImpl(QueryTreeNodePtr & node) + { + if (passed_node) + return; + + auto node_type = node->getNodeType(); + + if (types.contains(node_type)) + passed_node = node; + } + + QueryTreeNodePtr getNode() const { return passed_node; } + +private: + std::unordered_set types; + QueryTreeNodePtr passed_node; +}; + +/* +Helping class to find all used columns with specific source +*/ +class CollectUsedColumnsForSourceVisitor : public InDepthQueryTreeVisitorWithContext +{ +public: + using Base = InDepthQueryTreeVisitorWithContext; + using Base::Base; + + explicit CollectUsedColumnsForSourceVisitor( + QueryTreeNodePtr source_, + ContextPtr context, + bool collect_columns_from_other_sources_ = false) + : Base(context) + , source(source_) + , collect_columns_from_other_sources(collect_columns_from_other_sources_) + {} + + void enterImpl(QueryTreeNodePtr & node) + { + auto node_type = node->getNodeType(); + + if (node_type != QueryTreeNodeType::COLUMN) + return; + + auto & column_node = node->as(); + auto column_source = column_node.getColumnSourceOrNull(); + if (!column_source) + return; + + if ((column_source == source) != collect_columns_from_other_sources) + { + const auto & name = column_node.getColumnName(); + if (!names.count(name)) + { + columns.emplace_back(column_node.getColumn()); + names.insert(name); + } + } + } + + const NamesAndTypes & getColumns() const { return columns; } + +private: + std::unordered_set names; + QueryTreeNodePtr source; + NamesAndTypes columns; + bool collect_columns_from_other_sources; +}; + +}; + +/* +Try to make subquery to send on nodes +Converts + + SELECT s3.c1, s3.c2, t.c3 + FROM + s3Cluster(...) AS s3 + JOIN + localtable as t + ON s3.key == t.key + +to + + SELECT s3.c1, s3.c2, s3.key + FROM + s3Cluster(...) AS s3 +*/ +void IStorageCluster::updateQueryWithJoinToSendIfNeeded( + ASTPtr & query_to_send, + QueryTreeNodePtr query_tree, + const ContextPtr & context) +{ + auto object_storage_cluster_join_mode = context->getSettingsRef()[Setting::object_storage_cluster_join_mode]; + switch (object_storage_cluster_join_mode) + { + case ObjectStorageClusterJoinMode::LOCAL: + { + auto info = getQueryTreeInfo(query_tree, context); + + if (info.has_join || info.has_cross_join || info.has_local_columns_in_where) + { + auto modified_query_tree = query_tree->clone(); + + SearcherVisitor left_table_expression_searcher({QueryTreeNodeType::TABLE, QueryTreeNodeType::TABLE_FUNCTION}, context); + left_table_expression_searcher.visit(modified_query_tree); + auto table_function_node = left_table_expression_searcher.getNode(); + if (!table_function_node) + throw Exception(ErrorCodes::LOGICAL_ERROR, "Can't find table function node"); + + QueryTreeNodePtr query_tree_distributed; + + auto & query_node = modified_query_tree->as(); + + if (info.has_join) + { + auto join_node = query_node.getJoinTree(); + query_tree_distributed = join_node->as()->getLeftTableExpression()->clone(); + } + else if (info.has_cross_join) + { + SearcherVisitor join_searcher({QueryTreeNodeType::CROSS_JOIN}, context); + join_searcher.visit(modified_query_tree); + auto cross_join_node = join_searcher.getNode(); + if (!cross_join_node) + throw Exception(ErrorCodes::LOGICAL_ERROR, "Can't find CROSS JOIN node"); + // CrossJoinNode contains vector of nodes. 0 is left expression, always exists. + query_tree_distributed = cross_join_node->as()->getTableExpressions()[0]->clone(); + } + + // Find add used columns from table function to make proper projection list + // Need to do before changing WHERE condition + CollectUsedColumnsForSourceVisitor collector(table_function_node, context); + collector.visit(modified_query_tree); + const auto & columns = collector.getColumns(); + + if (columns.empty()) + { + auto column_nodes_to_select = std::make_shared(); + column_nodes_to_select->getNodes().reserve(1); + column_nodes_to_select->getNodes().emplace_back(std::make_shared(1)); + query_node.getProjectionNode() = column_nodes_to_select; + } + else + { + query_node.resolveProjectionColumns(columns); + auto column_nodes_to_select = std::make_shared(); + column_nodes_to_select->getNodes().reserve(columns.size()); + for (auto & column : columns) + column_nodes_to_select->getNodes().emplace_back(std::make_shared(column, table_function_node)); + query_node.getProjectionNode() = column_nodes_to_select; + } + + if (info.has_local_columns_in_where) + { + if (query_node.getPrewhere()) + removeExpressionsThatDoNotDependOnTableIdentifiers(query_node.getPrewhere(), table_function_node, context); + if (query_node.getWhere()) + removeExpressionsThatDoNotDependOnTableIdentifiers(query_node.getWhere(), table_function_node, context); + } + + query_node.getOrderByNode() = std::make_shared(); + query_node.getGroupByNode() = std::make_shared(); + + if (query_tree_distributed) + { + // Left only table function to send on cluster nodes + modified_query_tree = modified_query_tree->cloneAndReplace(query_node.getJoinTree(), query_tree_distributed); + } + + query_to_send = queryNodeToDistributedSelectQuery(modified_query_tree); + } + + return; + } + case ObjectStorageClusterJoinMode::GLOBAL: + // TODO + throw Exception(ErrorCodes::NOT_IMPLEMENTED, "`Global` mode for `object_storage_cluster_join_mode` setting is unimplemented for now"); + case ObjectStorageClusterJoinMode::ALLOW: // Do nothing special + return; + } +} + /// The code executes on initiator void IStorageCluster::read( QueryPlan & query_plan, @@ -108,13 +324,15 @@ void IStorageCluster::read( SharedHeader sample_block; ASTPtr query_to_send = query_info.query; + updateQueryWithJoinToSendIfNeeded(query_to_send, query_info.query_tree, context); + if (context->getSettingsRef()[Setting::allow_experimental_analyzer]) { - sample_block = InterpreterSelectQueryAnalyzer::getSampleBlock(query_info.query, context, SelectQueryOptions(processed_stage)); + sample_block = InterpreterSelectQueryAnalyzer::getSampleBlock(query_to_send, context, SelectQueryOptions(processed_stage)); } else { - auto interpreter = InterpreterSelectQuery(query_info.query, context, SelectQueryOptions(processed_stage).analyze()); + auto interpreter = InterpreterSelectQuery(query_to_send, context, SelectQueryOptions(processed_stage).analyze()); sample_block = interpreter.getSampleBlock(); query_to_send = interpreter.getQueryInfo().query->clone(); } @@ -122,7 +340,7 @@ void IStorageCluster::read( updateQueryToSendIfNeeded(query_to_send, storage_snapshot, context); RestoreQualifiedNamesVisitor::Data data; - data.distributed_table = DatabaseAndTableWithAlias(*getTableExpression(query_info.query->as(), 0)); + data.distributed_table = DatabaseAndTableWithAlias(*getTableExpression(query_to_send->as(), 0)); data.remote_table.database = context->getCurrentDatabase(); data.remote_table.table = getName(); RestoreQualifiedNamesVisitor(data).visit(query_to_send); @@ -219,9 +437,59 @@ void ReadFromCluster::initializePipeline(QueryPipelineBuilder & pipeline, const pipeline.init(std::move(pipe)); } +IStorageCluster::QueryTreeInfo IStorageCluster::getQueryTreeInfo(QueryTreeNodePtr query_tree, ContextPtr context) +{ + QueryTreeInfo info; + + auto & query_node = query_tree->as(); + if (auto join_node = query_node.getJoinTree()) + { + if (join_node->getNodeType() == QueryTreeNodeType::JOIN) + info.has_join = true; + else if (join_node->getNodeType() == QueryTreeNodeType::CROSS_JOIN) + info.has_cross_join = true; + } + + SearcherVisitor left_table_expression_searcher({QueryTreeNodeType::TABLE, QueryTreeNodeType::TABLE_FUNCTION}, context); + left_table_expression_searcher.visit(query_tree); + auto table_function_node = left_table_expression_searcher.getNode(); + if (!table_function_node) + throw Exception(ErrorCodes::LOGICAL_ERROR, "Can't find table or table function node"); + + if (query_node.hasWhere() || query_node.hasPrewhere()) + { + CollectUsedColumnsForSourceVisitor collector_where(table_function_node, context, true); + if (query_node.hasPrewhere()) + collector_where.visit(query_node.getPrewhere()); + if (query_node.hasWhere()) + collector_where.visit(query_node.getWhere()); + + // SELECT x FROM datalake.table WHERE x IN local.table. + // Need to modify 'WHERE' on remote node if it contains columns from other sources + // because remote node might not have those sources. + if (!collector_where.getColumns().empty()) + info.has_local_columns_in_where = true; + } + + return info; +} + QueryProcessingStage::Enum IStorageCluster::getQueryProcessingStage( - ContextPtr context, QueryProcessingStage::Enum to_stage, const StorageSnapshotPtr &, SelectQueryInfo &) const + ContextPtr context, QueryProcessingStage::Enum to_stage, const StorageSnapshotPtr &, SelectQueryInfo & query_info) const { + auto object_storage_cluster_join_mode = context->getSettingsRef()[Setting::object_storage_cluster_join_mode]; + + if (object_storage_cluster_join_mode != ObjectStorageClusterJoinMode::ALLOW) + { + if (!context->getSettingsRef()[Setting::allow_experimental_analyzer]) + throw Exception(ErrorCodes::NOT_IMPLEMENTED, + "object_storage_cluster_join_mode!='allow' is not supported without allow_experimental_analyzer=true"); + + auto info = getQueryTreeInfo(query_info.query_tree, context); + if (info.has_join || info.has_cross_join || info.has_local_columns_in_where) + return QueryProcessingStage::Enum::FetchColumns; + } + /// Initiator executes query on remote node. if (context->getClientInfo().query_kind == ClientInfo::QueryKind::INITIAL_QUERY) if (to_stage >= QueryProcessingStage::Enum::WithMergeableState) diff --git a/src/Storages/IStorageCluster.h b/src/Storages/IStorageCluster.h index 43b2d690955d..ac76323a22c1 100644 --- a/src/Storages/IStorageCluster.h +++ b/src/Storages/IStorageCluster.h @@ -57,12 +57,22 @@ class IStorageCluster : public IStorage protected: virtual void updateBeforeRead(const ContextPtr &) {} virtual void updateQueryToSendIfNeeded(ASTPtr & /*query*/, const StorageSnapshotPtr & /*storage_snapshot*/, const ContextPtr & /*context*/) {} + void updateQueryWithJoinToSendIfNeeded(ASTPtr & query_to_send, QueryTreeNodePtr query_tree, const ContextPtr & context); virtual void updateConfigurationIfNeeded(ContextPtr /* context */) {} private: LoggerPtr log; String cluster_name; + + struct QueryTreeInfo + { + bool has_join = false; + bool has_cross_join = false; + bool has_local_columns_in_where = false; + }; + + static QueryTreeInfo getQueryTreeInfo(QueryTreeNodePtr query_tree, ContextPtr context); }; diff --git a/src/Storages/ObjectStorage/StorageObjectStorageCluster.cpp b/src/Storages/ObjectStorage/StorageObjectStorageCluster.cpp index 061fcf839d0d..40b00dff7341 100644 --- a/src/Storages/ObjectStorage/StorageObjectStorageCluster.cpp +++ b/src/Storages/ObjectStorage/StorageObjectStorageCluster.cpp @@ -284,7 +284,11 @@ RemoteQueryExecutor::Extension StorageObjectStorageCluster::getTaskIteratorExten local_context, predicate, filter, +<<<<<<< HEAD storage_metadata_snapshot->virtuals.getSampleBlock(VirtualsKind::All, VirtualsMaterializationPlace::Reader).getNamesAndTypesList(), +======= + getVirtualsList(), +>>>>>>> d9d3710bd9b (Merge pull request #1646 from Altinity/frontport/antalya-26.3/fix_remote_calls) hive_partition_columns_to_read_from_file_path, nullptr, local_context->getFileProgressCallback(), diff --git a/src/Storages/extractTableFunctionFromSelectQuery.cpp b/src/Storages/extractTableFunctionFromSelectQuery.cpp index 57302036c889..064f538eeae7 100644 --- a/src/Storages/extractTableFunctionFromSelectQuery.cpp +++ b/src/Storages/extractTableFunctionFromSelectQuery.cpp @@ -9,7 +9,7 @@ namespace DB { -ASTFunction * extractTableFunctionFromSelectQuery(ASTPtr & query) +ASTTableExpression * extractTableExpressionASTPtrFromSelectQuery(ASTPtr & query) { auto * select_query = query->as(); if (!select_query || !select_query->tables()) @@ -17,10 +17,36 @@ ASTFunction * extractTableFunctionFromSelectQuery(ASTPtr & query) auto * tables = select_query->tables()->as(); auto * table_expression = tables->children[0]->as()->table_expression->as(); - if (!table_expression->table_function) + return table_expression; +} + +ASTPtr extractTableFunctionASTPtrFromSelectQuery(ASTPtr & query) +{ + auto table_expression = extractTableExpressionASTPtrFromSelectQuery(query); + return table_expression ? table_expression->table_function : nullptr; +} + +ASTPtr extractTableASTPtrFromSelectQuery(ASTPtr & query) +{ + auto table_expression = extractTableExpressionASTPtrFromSelectQuery(query); + return table_expression ? table_expression->database_and_table_name : nullptr; +} + +ASTFunction * extractTableFunctionFromSelectQuery(ASTPtr & query) +{ + auto table_function_ast = extractTableFunctionASTPtrFromSelectQuery(query); + if (!table_function_ast) return nullptr; - return table_expression->table_function->as(); + return table_function_ast->as(); +} + +ASTExpressionList * extractTableFunctionArgumentsFromSelectQuery(ASTPtr & query) +{ + auto * table_function = extractTableFunctionFromSelectQuery(query); + if (!table_function) + return nullptr; + return table_function->arguments->as(); } } diff --git a/src/Storages/extractTableFunctionFromSelectQuery.h b/src/Storages/extractTableFunctionFromSelectQuery.h index c69cc7ce6c52..20cc1ae93896 100644 --- a/src/Storages/extractTableFunctionFromSelectQuery.h +++ b/src/Storages/extractTableFunctionFromSelectQuery.h @@ -6,7 +6,11 @@ namespace DB { +struct ASTTableExpression; +ASTTableExpression * extractTableExpressionASTPtrFromSelectQuery(ASTPtr & query); +ASTPtr extractTableFunctionASTPtrFromSelectQuery(ASTPtr & query); +ASTPtr extractTableASTPtrFromSelectQuery(ASTPtr & query); ASTFunction * extractTableFunctionFromSelectQuery(ASTPtr & query); } diff --git a/tests/integration/test_database_iceberg/test.py b/tests/integration/test_database_iceberg/test.py index 31ec8882a357..3f4e60d43afd 100644 --- a/tests/integration/test_database_iceberg/test.py +++ b/tests/integration/test_database_iceberg/test.py @@ -11,18 +11,24 @@ import requests import pytz from pyiceberg.catalog import load_catalog -from pyiceberg.partitioning import PartitionField, PartitionSpec +from pyiceberg.partitioning import PartitionField, PartitionSpec, UNPARTITIONED_PARTITION_SPEC from pyiceberg.schema import Schema from pyiceberg.table.sorting import SortField, SortOrder from pyiceberg.transforms import DayTransform, IdentityTransform from pyiceberg.types import ( DoubleType, +<<<<<<< HEAD +======= + LongType, + FloatType, +>>>>>>> d9d3710bd9b (Merge pull request #1646 from Altinity/frontport/antalya-26.3/fix_remote_calls) NestedField, StringType, StructType, TimestampType, TimestamptzType ) +from pyiceberg.table.sorting import UNSORTED_SORT_ORDER from helpers.cluster import ClickHouseCluster from helpers.config_cluster import minio_secret_key, minio_access_key @@ -199,6 +205,7 @@ def started_cluster(): user_configs=[], stay_alive=True, with_iceberg_catalog=True, + with_zookeeper=True, ) logging.info("Starting cluster...") @@ -1143,6 +1150,7 @@ def test_gcs(started_cluster): assert "Google cloud storage converts to S3" in str(err.value) +<<<<<<< HEAD def test_invalid_auth_header_format(started_cluster): node = started_cluster.instances["node1"] @@ -1179,11 +1187,23 @@ def test_iceberg_file_progress_callback(started_cluster): test_ref = f"test_progress_callback_{uuid.uuid4().hex[:8]}" table_name = f"{test_ref}_table" +======= +# TODO - turn on after merge alternative syntax +def _test_cluster_joins(started_cluster): + node = started_cluster.instances["node1"] + + test_ref = f"test_join_tables_{uuid.uuid4()}" + table_name = f"{test_ref}_table" + table_name_2 = f"{test_ref}_table_2" + table_name_local = f"{test_ref}_table_local" + +>>>>>>> d9d3710bd9b (Merge pull request #1646 from Altinity/frontport/antalya-26.3/fix_remote_calls) root_namespace = f"{test_ref}_namespace" catalog = load_catalog_impl(started_cluster) catalog.create_namespace(root_namespace) +<<<<<<< HEAD table = create_table( catalog, root_namespace, @@ -1238,3 +1258,130 @@ def test_iceberg_file_progress_callback(started_cluster): f"`IcebergIterator::next` did not invoke the file-progress callback " f"(regression of PR #105413 wiring)." ) +======= + schema = Schema( + NestedField( + field_id=1, + name="tag", + field_type=LongType(), + required=False + ), + NestedField( + field_id=2, + name="name", + field_type=StringType(), + required=False, + ), + ) + table = create_table(catalog, root_namespace, table_name, schema, + partition_spec=UNPARTITIONED_PARTITION_SPEC, sort_order=UNSORTED_SORT_ORDER) + data = [{"tag": 1, "name": "John"}, {"tag": 2, "name": "Jack"}] + df = pa.Table.from_pylist(data) + table.append(df) + + schema2 = Schema( + NestedField( + field_id=1, + name="id", + field_type=LongType(), + required=False + ), + NestedField( + field_id=2, + name="second_name", + field_type=StringType(), + required=False, + ), + ) + table2 = create_table(catalog, root_namespace, table_name_2, schema2, + partition_spec=UNPARTITIONED_PARTITION_SPEC, sort_order=UNSORTED_SORT_ORDER) + data = [{"id": 1, "second_name": "Dow"}, {"id": 2, "second_name": "Sparrow"}] + df = pa.Table.from_pylist(data) + table2.append(df) + + node.query(f"CREATE TABLE `{table_name_local}` (id Int64, second_name String) ENGINE = Memory()") + node.query(f"INSERT INTO `{table_name_local}` VALUES (1, 'Silver'), (2, 'Black')") + + create_clickhouse_iceberg_database(started_cluster, node, CATALOG_NAME) + + res = node.query( + f""" + SELECT t1.name,t2.second_name + FROM {CATALOG_NAME}.`{root_namespace}.{table_name}` AS t1 + JOIN {CATALOG_NAME}.`{root_namespace}.{table_name_2}` AS t2 + ON t1.tag=t2.id + WHERE t1.tag < 10 AND t2.id < 20 + ORDER BY ALL + SETTINGS + object_storage_cluster='cluster_simple', + object_storage_cluster_join_mode='local' + """ + ) + + assert res == "Jack\tSparrow\nJohn\tDow\n" + + res = node.query( + f""" + SELECT name + FROM {CATALOG_NAME}.`{root_namespace}.{table_name}` + WHERE tag in ( + SELECT id + FROM {CATALOG_NAME}.`{root_namespace}.{table_name_2}` + ) + ORDER BY ALL + SETTINGS + object_storage_cluster='cluster_simple', + object_storage_cluster_join_mode='local' + """ + ) + + assert res == "Jack\nJohn\n" + + res = node.query( + f""" + SELECT t1.name,t2.second_name + FROM {CATALOG_NAME}.`{root_namespace}.{table_name}` AS t1 + JOIN `{table_name_local}` AS t2 + ON t1.tag=t2.id + WHERE t1.tag < 10 AND t2.id < 20 + ORDER BY ALL + SETTINGS + object_storage_cluster='cluster_simple', + object_storage_cluster_join_mode='local' + """ + ) + + assert res == "Jack\tBlack\nJohn\tSilver\n" + + res = node.query( + f""" + SELECT name + FROM {CATALOG_NAME}.`{root_namespace}.{table_name}` + WHERE tag in ( + SELECT id + FROM `{table_name_local}` + ) + ORDER BY ALL + SETTINGS + object_storage_cluster='cluster_simple', + object_storage_cluster_join_mode='local' + """ + ) + + assert res == "Jack\nJohn\n" + + res = node.query( + f""" + SELECT t1.name,t2.second_name + FROM {CATALOG_NAME}.`{root_namespace}.{table_name}` AS t1 + CROSS JOIN `{table_name_local}` AS t2 + WHERE t1.tag < 10 AND t2.id < 20 + ORDER BY ALL + SETTINGS + object_storage_cluster='cluster_simple', + object_storage_cluster_join_mode='local' + """ + ) + + assert res == "Jack\tBlack\nJack\tSilver\nJohn\tBlack\nJohn\tSilver\n" +>>>>>>> d9d3710bd9b (Merge pull request #1646 from Altinity/frontport/antalya-26.3/fix_remote_calls) diff --git a/tests/integration/test_s3_cluster/test.py b/tests/integration/test_s3_cluster/test.py index 30d5a6190217..3371f7013325 100644 --- a/tests/integration/test_s3_cluster/test.py +++ b/tests/integration/test_s3_cluster/test.py @@ -3,6 +3,12 @@ import os import shutil import uuid +<<<<<<< HEAD +======= +import threading +import time +from email.errors import HeaderParseError +>>>>>>> d9d3710bd9b (Merge pull request #1646 from Altinity/frontport/antalya-26.3/fix_remote_calls) import pytest @@ -551,9 +557,366 @@ def test_cluster_default_expression(started_cluster): assert result == expected_result +<<<<<<< HEAD @pytest.mark.parametrize("allow_experimental_analyzer", [0, 1]) @pytest.mark.parametrize("use_partition_strategy", [False, True]) def test_hive_partitioning(started_cluster, allow_experimental_analyzer, use_partition_strategy): +======= +def test_remote_hedged(started_cluster): + node = started_cluster.instances["s0_0_0"] + pure_s3 = node.query( + f""" + SELECT * from s3( + 'http://minio1:9001/root/data/{{clickhouse,database}}/*', + 'minio', '{minio_secret_key}', 'CSV', + 'name String, value UInt32, polygon Array(Array(Tuple(Float64, Float64)))') + ORDER BY (name, value, polygon) + LIMIT 1 + """ + ) + s3_distributed = node.query( + f""" + SELECT * from remote('s0_0_1', s3Cluster( + 'cluster_simple', + 'http://minio1:9001/root/data/{{clickhouse,database}}/*', 'minio', '{minio_secret_key}', 'CSV', + 'name String, value UInt32, polygon Array(Array(Tuple(Float64, Float64)))')) + ORDER BY (name, value, polygon) + LIMIT 1 + SETTINGS use_hedged_requests=True + """ + ) + + assert TSV(pure_s3) == TSV(s3_distributed) + + +def test_remote_no_hedged(started_cluster): + node = started_cluster.instances["s0_0_0"] + pure_s3 = node.query( + f""" + SELECT * from s3( + 'http://minio1:9001/root/data/{{clickhouse,database}}/*', + 'minio', '{minio_secret_key}', 'CSV', + 'name String, value UInt32, polygon Array(Array(Tuple(Float64, Float64)))') + ORDER BY (name, value, polygon) + LIMIT 1 + """ + ) + s3_distributed = node.query( + f""" + SELECT * from remote('s0_0_1', s3Cluster( + 'cluster_simple', + 'http://minio1:9001/root/data/{{clickhouse,database}}/*', 'minio', '{minio_secret_key}', 'CSV', + 'name String, value UInt32, polygon Array(Array(Tuple(Float64, Float64)))')) + ORDER BY (name, value, polygon) + LIMIT 1 + SETTINGS use_hedged_requests=False + """ + ) + + assert TSV(pure_s3) == TSV(s3_distributed) + + +@pytest.mark.parametrize("allow_experimental_analyzer", [0, 1]) +def test_hive_partitioning(started_cluster, allow_experimental_analyzer): + node = started_cluster.instances["s0_0_0"] + + node.query(f"SET allow_experimental_analyzer = {allow_experimental_analyzer}") + + for i in range(1, 5): + exists = node.query( + f""" + SELECT + count() + FROM s3('http://minio1:9001/root/data/hive/key={i}/*', 'minio', '{minio_secret_key}', 'Parquet', 'key Int32, value Int32') + GROUP BY ALL + FORMAT TSV + """ + ) + if int(exists) == 0: + node.query( + f""" + INSERT + INTO FUNCTION s3('http://minio1:9001/root/data/hive/key={i}/data.parquet', 'minio', '{minio_secret_key}', 'Parquet', 'key Int32, value Int32') + SELECT {i}, {i} + SETTINGS use_hive_partitioning = 0 + """ + ) + + settings = "enable_filesystem_cache = 0, use_query_cache = 0, use_cache_for_count_from_files = 0, use_iceberg_metadata_files_cache = 0, use_parquet_metadata_cache = 0, use_page_cache_for_object_storage = 0" + + query_id_full = str(uuid.uuid4()) + result = node.query( + f""" + SELECT count() + FROM s3('http://minio1:9001/root/data/hive/key=**.parquet', 'minio', '{minio_secret_key}', 'Parquet', 'key Int32, value Int32') + WHERE key <= 2 + FORMAT TSV + SETTINGS {settings}, use_hive_partitioning = 0 + """, + query_id=query_id_full, + ) + result = int(result) + assert result == 2 + + query_id_optimized = str(uuid.uuid4()) + result = node.query( + f""" + SELECT count() + FROM s3('http://minio1:9001/root/data/hive/key=**.parquet', 'minio', '{minio_secret_key}', 'Parquet', 'key Int32, value Int32') + WHERE key <= 2 + FORMAT TSV + SETTINGS {settings}, use_hive_partitioning = 1 + """, + query_id=query_id_optimized, + ) + result = int(result) + assert result == 2 + + query_id_cluster_full = str(uuid.uuid4()) + result = node.query( + f""" + SELECT count() + FROM s3Cluster(cluster_simple, 'http://minio1:9001/root/data/hive/key=**.parquet', 'minio', '{minio_secret_key}', 'Parquet', 'key Int32, value Int32') + WHERE key <= 2 + FORMAT TSV + SETTINGS {settings}, use_hive_partitioning = 0 + """, + query_id=query_id_cluster_full, + ) + result = int(result) + assert result == 2 + + query_id_cluster_optimized = str(uuid.uuid4()) + result = node.query( + f""" + SELECT count() + FROM s3Cluster(cluster_simple, 'http://minio1:9001/root/data/hive/key=**.parquet', 'minio', '{minio_secret_key}', 'Parquet', 'key Int32, value Int32') + WHERE key <= 2 + FORMAT TSV + SETTINGS {settings}, use_hive_partitioning = 1 + """, + query_id=query_id_cluster_optimized, + ) + result = int(result) + assert result == 2 + + node.query("SYSTEM FLUSH LOGS ON CLUSTER 'cluster_simple'") + + full_traffic = node.query( + f""" + SELECT sum(ProfileEvents['ReadBufferFromS3Bytes']) + FROM clusterAllReplicas(cluster_simple, system.query_log) + WHERE type='QueryFinish' AND initial_query_id='{query_id_full}' + FORMAT TSV + """ + ) + full_traffic = int(full_traffic) + assert full_traffic > 0 # 612*4 + + optimized_traffic = node.query( + f""" + SELECT sum(ProfileEvents['ReadBufferFromS3Bytes']) + FROM clusterAllReplicas(cluster_simple, system.query_log) + WHERE type='QueryFinish' AND initial_query_id='{query_id_optimized}' + FORMAT TSV + """ + ) + optimized_traffic = int(optimized_traffic) + assert optimized_traffic > 0 # 612*2 + assert full_traffic > optimized_traffic + + cluster_full_traffic = node.query( + f""" + SELECT sum(ProfileEvents['ReadBufferFromS3Bytes']) + FROM clusterAllReplicas(cluster_simple, system.query_log) + WHERE type='QueryFinish' AND initial_query_id='{query_id_cluster_full}' + FORMAT TSV + """ + ) + cluster_full_traffic = int(cluster_full_traffic) + assert cluster_full_traffic == full_traffic + + cluster_optimized_traffic = node.query( + f""" + SELECT sum(ProfileEvents['ReadBufferFromS3Bytes']) + FROM clusterAllReplicas(cluster_simple, system.query_log) + WHERE type='QueryFinish' AND initial_query_id='{query_id_cluster_optimized}' + FORMAT TSV + """ + ) + cluster_optimized_traffic = int(cluster_optimized_traffic) + assert cluster_optimized_traffic == optimized_traffic + + node.query("SET allow_experimental_analyzer = DEFAULT") + + +def test_joins(started_cluster): + node = started_cluster.instances["s0_0_0"] + + # Table join_table only exists on the node 's0_0_0'. + node.query("DROP TABLE IF EXISTS join_table SYNC") + node.query( + """ + CREATE TABLE IF NOT EXISTS join_table ( + id UInt32, + name String + ) ENGINE=MergeTree() + ORDER BY id; + """ + ) + + node.query( + f""" + INSERT INTO join_table + SELECT value, concat(name, '_jt') FROM s3Cluster('cluster_simple', + 'http://minio1:9001/root/data/{{clickhouse,database}}/*', 'minio', '{minio_secret_key}', 'CSV', + 'name String, value UInt32, polygon Array(Array(Tuple(Float64, Float64)))'); + """ + ) + + result1 = node.query( + f""" + SELECT t1.name, t2.name FROM + s3Cluster('cluster_simple', + 'http://minio1:9001/root/data/{{clickhouse,database}}/*', 'minio', '{minio_secret_key}', 'CSV', + 'name String, value UInt32, polygon Array(Array(Tuple(Float64, Float64)))') AS t1 + JOIN + join_table AS t2 + ON t1.value = t2.id + ORDER BY t1.name + SETTINGS object_storage_cluster_join_mode='local'; + """ + ) + + res = list(map(str.split, result1.splitlines())) + assert len(res) == 25 + + for line in res: + if len(line) == 2: + assert line[1] == f"{line[0]}_jt" + else: + assert line == ["_jt"] # for empty name + + result2 = node.query( + f""" + SELECT t1.name, t2.name FROM + join_table AS t2 + JOIN + s3Cluster('cluster_simple', + 'http://minio1:9001/root/data/{{clickhouse,database}}/*', 'minio', '{minio_secret_key}', 'CSV', + 'name String, value UInt32, polygon Array(Array(Tuple(Float64, Float64)))') AS t1 + ON t1.value = t2.id + ORDER BY t1.name + SETTINGS object_storage_cluster_join_mode='local'; + """ + ) + + assert result1 == result2 + + # With WHERE clause with remote column only + result3 = node.query( + f""" + SELECT t1.name, t2.name FROM + s3Cluster('cluster_simple', + 'http://minio1:9001/root/data/{{clickhouse,database}}/*', 'minio', '{minio_secret_key}', 'CSV', + 'name String, value UInt32, polygon Array(Array(Tuple(Float64, Float64)))') AS t1 + JOIN + join_table AS t2 + ON t1.value = t2.id + WHERE (t1.value % 2) + ORDER BY t1.name + SETTINGS object_storage_cluster_join_mode='local'; + """ + ) + + res = list(map(str.split, result3.splitlines())) + assert len(res) == 8 + + # With WHERE clause with local column only + result4 = node.query( + f""" + SELECT t1.name, t2.name FROM + s3Cluster('cluster_simple', + 'http://minio1:9001/root/data/{{clickhouse,database}}/*', 'minio', '{minio_secret_key}', 'CSV', + 'name String, value UInt32, polygon Array(Array(Tuple(Float64, Float64)))') AS t1 + JOIN + join_table AS t2 + ON t1.value = t2.id + WHERE (t2.id % 2) + ORDER BY t1.name + SETTINGS object_storage_cluster_join_mode='local'; + """ + ) + + assert result3 == result4 + + # With WHERE clause with local and remote columns + result5 = node.query( + f""" + SELECT t1.name, t2.name FROM + s3Cluster('cluster_simple', + 'http://minio1:9001/root/data/{{clickhouse,database}}/*', 'minio', '{minio_secret_key}', 'CSV', + 'name String, value UInt32, polygon Array(Array(Tuple(Float64, Float64)))') AS t1 + JOIN + join_table AS t2 + ON t1.value = t2.id + WHERE (t1.value % 2) AND ((t2.id % 3) == 2) + ORDER BY t1.name + SETTINGS object_storage_cluster_join_mode='local'; + """ + ) + + res = list(map(str.split, result5.splitlines())) + assert len(res) == 6 + + result6 = node.query( + f""" + SELECT name FROM + s3Cluster('cluster_simple', + 'http://minio1:9001/root/data/{{clickhouse,database}}/*', 'minio', '{minio_secret_key}', 'CSV', + 'name String, value UInt32, polygon Array(Array(Tuple(Float64, Float64)))') + WHERE value IN (SELECT id FROM join_table) + ORDER BY name + SETTINGS object_storage_cluster_join_mode='local'; + """ + ) + res = list(map(str.split, result6.splitlines())) + assert len(res) == 25 + + result7 = node.query( + f""" + SELECT count() FROM + s3Cluster('cluster_simple', + 'http://minio1:9001/root/data/{{clickhouse,database}}/*', 'minio', '{minio_secret_key}', 'CSV', + 'name String, value UInt32, polygon Array(Array(Tuple(Float64, Float64)))') AS t1 + JOIN + join_table AS t2 + ON 1 + GROUP BY ALL + SETTINGS object_storage_cluster_join_mode='local'; + """ + ) + assert result7.strip() == "625" + + result8 = node.query( + f""" + SELECT count(), t2.id FROM + s3Cluster('cluster_simple', + 'http://minio1:9001/root/data/{{clickhouse,database}}/*', 'minio', '{minio_secret_key}', 'CSV', + 'name String, value UInt32, polygon Array(Array(Tuple(Float64, Float64)))') AS t1 + JOIN + join_table AS t2 + ON 1 + GROUP BY ALL + SETTINGS object_storage_cluster_join_mode='local'; + """ + ) + res = list(map(str.split, result8.splitlines())) + assert len(res) == 25 + + +def test_graceful_shutdown(started_cluster): +>>>>>>> d9d3710bd9b (Merge pull request #1646 from Altinity/frontport/antalya-26.3/fix_remote_calls) node = started_cluster.instances["s0_0_0"] data_path = f"root/data/hive_{allow_experimental_analyzer}/{random_string(6)}" diff --git a/tests/integration/test_storage_iceberg_with_spark/test_cluster_joins.py b/tests/integration/test_storage_iceberg_with_spark/test_cluster_joins.py new file mode 100644 index 000000000000..bb637f8e8cc2 --- /dev/null +++ b/tests/integration/test_storage_iceberg_with_spark/test_cluster_joins.py @@ -0,0 +1,154 @@ +import pytest + +from helpers.iceberg_utils import ( + get_uuid_str, + get_creation_expression, + execute_spark_query_general, +) + +# TODO - turn on after merge alternative syntax +@pytest.mark.parametrize("storage_type", ["s3", "azure"]) +def _test_cluster_joins(started_cluster_iceberg_with_spark, storage_type): + instance = started_cluster_iceberg_with_spark.instances["node1"] + spark = started_cluster_iceberg_with_spark.spark_session + TABLE_NAME = "test_cluster_joins_" + storage_type + "_" + get_uuid_str() + TABLE_NAME_2 = "test_cluster_joins_2_" + storage_type + "_" + get_uuid_str() + TABLE_NAME_LOCAL = "test_cluster_joins_local_" + storage_type + "_" + get_uuid_str() + + def execute_spark_query(query: str, table_name): + return execute_spark_query_general( + spark, + started_cluster_iceberg_with_spark, + storage_type, + table_name, + query, + ) + + execute_spark_query( + f""" + CREATE TABLE {TABLE_NAME} ( + tag INT, + name VARCHAR(50) + ) + USING iceberg + OPTIONS('format-version'='2') + """, TABLE_NAME + ) + + execute_spark_query( + f""" + INSERT INTO {TABLE_NAME} VALUES + (1, 'john'), + (2, 'jack') + """, TABLE_NAME + ) + + execute_spark_query( + f""" + CREATE TABLE {TABLE_NAME_2} ( + id INT, + second_name VARCHAR(50) + ) + USING iceberg + OPTIONS('format-version'='2') + """, TABLE_NAME_2 + ) + + execute_spark_query( + f""" + INSERT INTO {TABLE_NAME_2} VALUES + (1, 'dow'), + (2, 'sparrow') + """, TABLE_NAME_2 + ) + + creation_expression = get_creation_expression( + storage_type, TABLE_NAME, started_cluster_iceberg_with_spark, table_function=True, run_on_cluster=True + ) + + creation_expression_2 = get_creation_expression( + storage_type, TABLE_NAME_2, started_cluster_iceberg_with_spark, table_function=True, run_on_cluster=True + ) + + instance.query(f"CREATE TABLE `{TABLE_NAME_LOCAL}` (id Int64, second_name String) ENGINE = Memory()") + instance.query(f"INSERT INTO `{TABLE_NAME_LOCAL}` VALUES (1, 'silver'), (2, 'black')") + + res = instance.query( + f""" + SELECT t1.name,t2.second_name + FROM {creation_expression} AS t1 + JOIN {creation_expression_2} AS t2 + ON t1.tag=t2.id + ORDER BY ALL + SETTINGS + object_storage_cluster='cluster_simple', + object_storage_cluster_join_mode='local' + """ + ) + + assert res == "jack\tsparrow\njohn\tdow\n" + + res = instance.query( + f""" + SELECT name + FROM {creation_expression} + WHERE tag in ( + SELECT id + FROM {creation_expression_2} + ) + ORDER BY ALL + SETTINGS + object_storage_cluster='cluster_simple', + object_storage_cluster_join_mode='local' + """ + ) + + assert res == "jack\njohn\n" + + res = instance.query( + f""" + SELECT t1.name,t2.second_name + FROM {creation_expression} AS t1 + JOIN `{TABLE_NAME_LOCAL}` AS t2 + ON t1.tag=t2.id + WHERE t1.tag < 10 AND t2.id < 20 + ORDER BY ALL + SETTINGS + object_storage_cluster='cluster_simple', + object_storage_cluster_join_mode='local' + """ + ) + + assert res == "jack\tblack\njohn\tsilver\n" + + res = instance.query( + f""" + SELECT name + FROM {creation_expression} + WHERE tag in ( + SELECT id + FROM `{TABLE_NAME_LOCAL}` + ) + ORDER BY ALL + SETTINGS + object_storage_cluster='cluster_simple', + object_storage_cluster_join_mode='local' + """ + ) + + assert res == "jack\njohn\n" + + res = instance.query( + f""" + SELECT t1.name,t2.second_name + FROM {creation_expression} AS t1 + CROSS JOIN `{TABLE_NAME_LOCAL}` AS t2 + WHERE t1.tag < 10 AND t2.id < 20 + ORDER BY ALL + SETTINGS + object_storage_cluster='cluster_simple', + object_storage_cluster_join_mode='local' + """ + ) + + assert res == "jack\tblack\njack\tsilver\njohn\tblack\njohn\tsilver\n" diff --git a/tests/integration/test_storage_iceberg_with_spark/test_cluster_table_function.py b/tests/integration/test_storage_iceberg_with_spark/test_cluster_table_function.py index 7f1701158bf8..a145b087d4b2 100644 --- a/tests/integration/test_storage_iceberg_with_spark/test_cluster_table_function.py +++ b/tests/integration/test_storage_iceberg_with_spark/test_cluster_table_function.py @@ -130,6 +130,24 @@ def add_df(mode): # write 3 times assert int(instance.query(f"SELECT count() FROM {table_function_expr_cluster}")) == 100 * 3 + + # Cluster Query with node1 as coordinator + table_function_expr_cluster = get_creation_expression( + storage_type, + TABLE_NAME, + started_cluster_iceberg_with_spark, + table_function=True, + run_on_cluster=True, + ) + select_remote_cluster = ( + instance.query(f"SELECT * FROM remote('node2',{table_function_expr_cluster})") + .strip() + .split() + ) + assert len(select_remote_cluster) == 600 + assert select_remote_cluster == select_regular + + @pytest.mark.parametrize("format_version", ["1", "2"]) @pytest.mark.parametrize("storage_type", ["s3", "azure"]) def test_writes_cluster_table_function(started_cluster_iceberg_with_spark, format_version, storage_type): diff --git a/tests/queries/0_stateless/02126_dist_desc.sql.j2 b/tests/queries/0_stateless/02126_dist_desc.sql.j2 index c0e1b5f8abd2..93f23dc9eb14 100644 --- a/tests/queries/0_stateless/02126_dist_desc.sql.j2 +++ b/tests/queries/0_stateless/02126_dist_desc.sql.j2 @@ -10,7 +10,7 @@ select * from remote('{{host}}', {{args}}) format Null; {% endfor -%} system flush logs query_log; -select anyIf(query, is_initial_query), groupArrayIf(query, query_kind = 'Describe' and not is_initial_query) from system.query_log +select anyIf(query, initial_query_id == query_id), groupArrayIf(query, query_kind = 'Describe' and initial_query_id != query_id) from system.query_log where event_date >= yesterday() AND event_time >= now() - 600 AND type = 'QueryFinish' and diff --git a/tests/queries/0_stateless/03550_analyzer_remote_view_columns.sql b/tests/queries/0_stateless/03550_analyzer_remote_view_columns.sql index 833106925266..820a7cd95f2b 100644 --- a/tests/queries/0_stateless/03550_analyzer_remote_view_columns.sql +++ b/tests/queries/0_stateless/03550_analyzer_remote_view_columns.sql @@ -39,4 +39,4 @@ WHERE event_date >= yesterday() AND event_time >= now() - 600 AND AND log_comment = 'THIS IS A COMMENT TO MARK THE INITIAL QUERY' LIMIT 1) AND type = 'QueryFinish' - AND NOT is_initial_query; + AND query_id != initial_query_id; diff --git a/tests/queries/0_stateless/03620_analyzer_distributed_global_in.reference b/tests/queries/0_stateless/03620_analyzer_distributed_global_in.reference index 689e784cc037..dc1a034d49e0 100644 --- a/tests/queries/0_stateless/03620_analyzer_distributed_global_in.reference +++ b/tests/queries/0_stateless/03620_analyzer_distributed_global_in.reference @@ -60,7 +60,7 @@ CreatingSets (Create sets before main query execution) ReadFromSystemNumbers system flush logs query_log; -- SKIP: current_database = currentDatabase() -select normalizeQuery(replace(query, currentDatabase(), 'default')) from system.query_log where event_date >= yesterday() AND event_time >= now() - 600 and log_comment like '%' || currentDatabase() || '%' and not is_initial_query and type != 'QueryStart' and query_kind = 'Select' order by event_time_microseconds; +select normalizeQuery(replace(query, currentDatabase(), 'default')) from system.query_log where event_date >= yesterday() AND event_time >= now() - 600 and log_comment like '%' || currentDatabase() || '%' and initial_query_id != query_id and type != 'QueryStart' and query_kind = 'Select' order by event_time_microseconds; SELECT `__table1`.`x` AS `x`, `__table1`.`y` AS `y` FROM `default`.`tab0` AS `__table1` HAVING in(`x`, (SELECT `__table1`.`number` + ? AS `?` FROM numbers(?) AS `__table1`)) SELECT `__table1`.`x` AS `x`, `__table1`.`y` AS `y` FROM `default`.`tab0` AS `__table1` HAVING globalIn(`x`, `?`) SELECT `__table1`.`x` AS `x`, `__table1`.`y` AS `y` FROM `default`.`tab0` AS `__table1` HAVING in(`x`, (SELECT `__table1`.`number` + ? AS `?` FROM numbers(?) AS `__table1`)) diff --git a/tests/queries/0_stateless/03620_analyzer_distributed_global_in.sql b/tests/queries/0_stateless/03620_analyzer_distributed_global_in.sql index 031dccb75f89..87193c7e06a8 100644 --- a/tests/queries/0_stateless/03620_analyzer_distributed_global_in.sql +++ b/tests/queries/0_stateless/03620_analyzer_distributed_global_in.sql @@ -30,4 +30,4 @@ select * from (explain indexes=1, distributed=1 ); system flush logs query_log; -- SKIP: current_database = currentDatabase() -select normalizeQuery(replace(query, currentDatabase(), 'default')) from system.query_log where event_date >= yesterday() AND event_time >= now() - 600 and log_comment like '%' || currentDatabase() || '%' and not is_initial_query and type != 'QueryStart' and query_kind = 'Select' order by event_time_microseconds; +select normalizeQuery(replace(query, currentDatabase(), 'default')) from system.query_log where event_date >= yesterday() AND event_time >= now() - 600 and log_comment like '%' || currentDatabase() || '%' and initial_query_id != query_id and type != 'QueryStart' and query_kind = 'Select' order by event_time_microseconds; From f13ad1d3aa1790024f73b9e94b13d7dfe06264d5 Mon Sep 17 00:00:00 2001 From: Andrey Zvonov <32552679+zvonand@users.noreply.github.com> Date: Mon, 3 Aug 2026 13:56:55 +0200 Subject: [PATCH 04/43] Resolve conflicts in cherry-pick of #1646 Kept antalya-26.6's already-present, evolved copies of the pieces this frontport re-adds (ObjectFilterStep, its registry entry, the snapshot-based virtuals list in StorageObjectStorageCluster and test_hive_partitioning in test_s3_cluster), and applied the source PR's genuinely new additions on top: the analyzer-side addObjectFilterStep in Planner.cpp, RemoteQueryExecutor::setRemoteFunction/setShardCount plus their ReadFromRemote call sites, and the new integration tests (test_remote_hedged, test_remote_no_hedged, test_joins, _test_cluster_joins). Context-only lines from the 26.3 source that antalya-26.6 no longer has (duplicated_part_uuids, threading/time/HeaderParseError and FloatType imports) were not re-introduced. SettingsChangesHistory: uncommented the existing object_storage_cluster_join_mode row in place and dropped the cherry-pick's duplicate (which had landed inside an unrelated version block). Source-PR: #1646 (https://github.com/Altinity/ClickHouse/pull/1646) --- src/Core/SettingsChangesHistory.cpp | 5 +- src/Planner/Planner.cpp | 3 - src/Processors/QueryPlan/ObjectFilterStep.cpp | 7 - src/Processors/QueryPlan/ObjectFilterStep.h | 14 -- .../QueryPlan/QueryPlanStepRegistry.cpp | 7 - src/Processors/QueryPlan/ReadFromRemote.cpp | 3 - src/QueryPipeline/RemoteQueryExecutor.h | 10 +- .../StorageObjectStorageCluster.cpp | 4 - .../integration/test_database_iceberg/test.py | 35 ++-- tests/integration/test_s3_cluster/test.py | 150 +----------------- 10 files changed, 21 insertions(+), 217 deletions(-) diff --git a/src/Core/SettingsChangesHistory.cpp b/src/Core/SettingsChangesHistory.cpp index 786b2ca773bc..4309c206d199 100644 --- a/src/Core/SettingsChangesHistory.cpp +++ b/src/Core/SettingsChangesHistory.cpp @@ -197,9 +197,6 @@ const VersionToSettingsChangesMap & getSettingsChangesHistory() {"ai_function_throw_on_quota_exceeded", true, true, "New setting"}, {"variant_throw_on_type_mismatch", true, true, "New setting to control type mismatch behavior in default Variant implementation"}, {"dynamic_throw_on_type_mismatch", true, true, "New setting to control type mismatch behavior in default Dynamic implementation"}, - addSettingsChanges(settings_changes_history, "26.3.1.20001.altinityantalya", - { - {"object_storage_cluster_join_mode", "allow", "allow", "New setting"}, }); addSettingsChanges(settings_changes_history, "26.3", { @@ -472,7 +469,7 @@ const VersionToSettingsChangesMap & getSettingsChangesHistory() // {"iceberg_timezone_for_timestamptz", "UTC", "UTC", "New setting."}, // {"object_storage_remote_initiator", false, false, "New setting."}, // {"allow_experimental_iceberg_read_optimization", true, true, "New setting."}, - // {"object_storage_cluster_join_mode", "allow", "allow", "New setting"}, + {"object_storage_cluster_join_mode", "allow", "allow", "New setting"}, // {"lock_object_storage_task_distribution_ms", 500, 500, "New setting."}, // {"allow_retries_in_cluster_requests", false, false, "New setting"}, // {"allow_experimental_export_merge_tree_part", false, true, "Turned ON by default for Antalya."}, diff --git a/src/Planner/Planner.cpp b/src/Planner/Planner.cpp index e0304721e792..b36bc3b00cd7 100644 --- a/src/Planner/Planner.cpp +++ b/src/Planner/Planner.cpp @@ -41,11 +41,8 @@ #include #include #include -<<<<<<< HEAD #include -======= #include ->>>>>>> d9d3710bd9b (Merge pull request #1646 from Altinity/frontport/antalya-26.3/fix_remote_calls) #include #include diff --git a/src/Processors/QueryPlan/ObjectFilterStep.cpp b/src/Processors/QueryPlan/ObjectFilterStep.cpp index f5d2d3ee667b..cfa44162feaf 100644 --- a/src/Processors/QueryPlan/ObjectFilterStep.cpp +++ b/src/Processors/QueryPlan/ObjectFilterStep.cpp @@ -15,11 +15,7 @@ namespace ErrorCodes } ObjectFilterStep::ObjectFilterStep( -<<<<<<< HEAD const SharedHeader & input_header_, -======= - SharedHeader input_header_, ->>>>>>> d9d3710bd9b (Merge pull request #1646 from Altinity/frontport/antalya-26.3/fix_remote_calls) ActionsDAG actions_dag_, String filter_column_name_) : actions_dag(std::move(actions_dag_)) @@ -59,10 +55,7 @@ std::unique_ptr ObjectFilterStep::deserialize(Deserialization & return std::make_unique(ctx.input_headers.front(), std::move(actions_dag), std::move(filter_column_name)); } -<<<<<<< HEAD void registerObjectFilterStep(QueryPlanStepRegistry & registry); -======= ->>>>>>> d9d3710bd9b (Merge pull request #1646 from Altinity/frontport/antalya-26.3/fix_remote_calls) void registerObjectFilterStep(QueryPlanStepRegistry & registry) { registry.registerStep("ObjectFilter", ObjectFilterStep::deserialize); diff --git a/src/Processors/QueryPlan/ObjectFilterStep.h b/src/Processors/QueryPlan/ObjectFilterStep.h index f2cb3db195cb..f0030a6624e6 100644 --- a/src/Processors/QueryPlan/ObjectFilterStep.h +++ b/src/Processors/QueryPlan/ObjectFilterStep.h @@ -5,45 +5,31 @@ namespace DB { -<<<<<<< HEAD /// Implements WHERE condition only to filter objects in object storage /// Difference with FilterStep is that ObjectFilterStep is added only for distributed calls /// (table functions like `s3Cluster`) and is used only to filter objects, /// not to filter data after reading, because initiator can have not this column /// In query like `SELECT count() FROM s3Cluster('cluster', ...) WHERE key=42` /// column `key` does not exist in blocks getting from cluster replicas. -======= -/// Implements WHERE operation. ->>>>>>> d9d3710bd9b (Merge pull request #1646 from Altinity/frontport/antalya-26.3/fix_remote_calls) class ObjectFilterStep : public IQueryPlanStep { public: ObjectFilterStep( -<<<<<<< HEAD const SharedHeader & input_header_, -======= - SharedHeader input_header_, ->>>>>>> d9d3710bd9b (Merge pull request #1646 from Altinity/frontport/antalya-26.3/fix_remote_calls) ActionsDAG actions_dag_, String filter_column_name_); String getName() const override { return "ObjectFilter"; } QueryPipelineBuilderPtr updatePipeline(QueryPipelineBuilders pipelines, const BuildQueryPipelineSettings & settings) override; -<<<<<<< HEAD bool hasCorrelatedExpressions() const override { return actions_dag.hasCorrelatedColumns(); } -======= ->>>>>>> d9d3710bd9b (Merge pull request #1646 from Altinity/frontport/antalya-26.3/fix_remote_calls) const ActionsDAG & getExpression() const { return actions_dag; } ActionsDAG & getExpression() { return actions_dag; } const String & getFilterColumnName() const { return filter_column_name; } void serialize(Serialization & ctx) const override; -<<<<<<< HEAD bool isSerializable() const override { return true; } -======= ->>>>>>> d9d3710bd9b (Merge pull request #1646 from Altinity/frontport/antalya-26.3/fix_remote_calls) static std::unique_ptr deserialize(Deserialization & ctx); diff --git a/src/Processors/QueryPlan/QueryPlanStepRegistry.cpp b/src/Processors/QueryPlan/QueryPlanStepRegistry.cpp index d434c7cda4ef..08fcf3126283 100644 --- a/src/Processors/QueryPlan/QueryPlanStepRegistry.cpp +++ b/src/Processors/QueryPlan/QueryPlanStepRegistry.cpp @@ -56,7 +56,6 @@ void registerFilterStep(QueryPlanStepRegistry & registry); void registerTotalsHavingStep(QueryPlanStepRegistry & registry); void registerExtremesStep(QueryPlanStepRegistry & registry); void registerJoinStep(QueryPlanStepRegistry & registry); -<<<<<<< HEAD void registerShuffleSendStep(QueryPlanStepRegistry & registry); void registerShuffleReceiveStep(QueryPlanStepRegistry & registry); void registerGatherSendStep(QueryPlanStepRegistry & registry); @@ -64,9 +63,6 @@ void registerGatherReceiveStep(QueryPlanStepRegistry & registry); void registerBroadcastSendStep(QueryPlanStepRegistry & registry); void registerBroadcastReceiveStep(QueryPlanStepRegistry & registry); void registerReadFromMergeTreeStep(QueryPlanStepRegistry & registry); -======= -void registerObjectFilterStep(QueryPlanStepRegistry & registry); ->>>>>>> d9d3710bd9b (Merge pull request #1646 from Altinity/frontport/antalya-26.3/fix_remote_calls) void registerReadFromTableStep(QueryPlanStepRegistry & registry); void registerReadFromTableFunctionStep(QueryPlanStepRegistry & registry); @@ -113,12 +109,9 @@ void QueryPlanStepRegistry::registerPlanSteps() registerReadFromTableFunctionStep(registry); registerBuildRuntimeFilterStep(registry); registerObjectFilterStep(registry); -<<<<<<< HEAD registerReadFromStorageStep(registry); -======= ->>>>>>> d9d3710bd9b (Merge pull request #1646 from Altinity/frontport/antalya-26.3/fix_remote_calls) } } diff --git a/src/Processors/QueryPlan/ReadFromRemote.cpp b/src/Processors/QueryPlan/ReadFromRemote.cpp index eba29d63b116..b41f36a4343e 100644 --- a/src/Processors/QueryPlan/ReadFromRemote.cpp +++ b/src/Processors/QueryPlan/ReadFromRemote.cpp @@ -710,12 +710,9 @@ void ReadFromRemote::addLazyPipe( auto remote_query_executor = std::make_shared( std::move(connections), query_string, header, my_context, my_throttler, my_scalars, my_external_tables, stage_to_use, my_shard.query_plan, /*extension=*/std::nullopt, my_shard.shard_info.pool); -<<<<<<< HEAD remote_query_executor->setDistributedFanout(my_distributed_fanout); -======= remote_query_executor->setRemoteFunction(my_is_remote_function); remote_query_executor->setShardCount(my_shard_count); ->>>>>>> d9d3710bd9b (Merge pull request #1646 from Altinity/frontport/antalya-26.3/fix_remote_calls) auto pipe = createRemoteSourcePipe( remote_query_executor, add_agg_info, add_totals, add_extremes, async_read, async_query_sending, parallel_marshalling_threads); diff --git a/src/QueryPipeline/RemoteQueryExecutor.h b/src/QueryPipeline/RemoteQueryExecutor.h index 855b9974547f..ffcb32e6a2df 100644 --- a/src/QueryPipeline/RemoteQueryExecutor.h +++ b/src/QueryPipeline/RemoteQueryExecutor.h @@ -215,13 +215,11 @@ class RemoteQueryExecutor void setUnavailableShardTracker(UnavailableShardTrackerPtr tracker) { unavailable_shard_tracker = std::move(tracker); } -<<<<<<< HEAD void setDistributedFanout(size_t total_connections) { distributed_fanout = total_connections; } -======= + void setRemoteFunction(bool is_remote_function_ = true) { is_remote_function = is_remote_function_; } void setShardCount(UInt32 shard_count_) { shard_count = shard_count_; } ->>>>>>> d9d3710bd9b (Merge pull request #1646 from Altinity/frontport/antalya-26.3/fix_remote_calls) const Block & getHeader() const { return *header; } const SharedHeader & getSharedHeader() const { return header; } @@ -315,15 +313,9 @@ class RemoteQueryExecutor bool packet_in_progress = false; #endif -<<<<<<< HEAD -======= bool is_remote_function = false; UInt32 shard_count = 0; - /// Parts uuids, collected from remote replicas - std::vector duplicated_part_uuids; - ->>>>>>> d9d3710bd9b (Merge pull request #1646 from Altinity/frontport/antalya-26.3/fix_remote_calls) PoolMode pool_mode = PoolMode::GET_MANY; StorageID main_table = StorageID::createEmpty(); diff --git a/src/Storages/ObjectStorage/StorageObjectStorageCluster.cpp b/src/Storages/ObjectStorage/StorageObjectStorageCluster.cpp index 40b00dff7341..061fcf839d0d 100644 --- a/src/Storages/ObjectStorage/StorageObjectStorageCluster.cpp +++ b/src/Storages/ObjectStorage/StorageObjectStorageCluster.cpp @@ -284,11 +284,7 @@ RemoteQueryExecutor::Extension StorageObjectStorageCluster::getTaskIteratorExten local_context, predicate, filter, -<<<<<<< HEAD storage_metadata_snapshot->virtuals.getSampleBlock(VirtualsKind::All, VirtualsMaterializationPlace::Reader).getNamesAndTypesList(), -======= - getVirtualsList(), ->>>>>>> d9d3710bd9b (Merge pull request #1646 from Altinity/frontport/antalya-26.3/fix_remote_calls) hive_partition_columns_to_read_from_file_path, nullptr, local_context->getFileProgressCallback(), diff --git a/tests/integration/test_database_iceberg/test.py b/tests/integration/test_database_iceberg/test.py index 3f4e60d43afd..bb24e9329fcc 100644 --- a/tests/integration/test_database_iceberg/test.py +++ b/tests/integration/test_database_iceberg/test.py @@ -17,11 +17,7 @@ from pyiceberg.transforms import DayTransform, IdentityTransform from pyiceberg.types import ( DoubleType, -<<<<<<< HEAD -======= LongType, - FloatType, ->>>>>>> d9d3710bd9b (Merge pull request #1646 from Altinity/frontport/antalya-26.3/fix_remote_calls) NestedField, StringType, StructType, @@ -1150,7 +1146,6 @@ def test_gcs(started_cluster): assert "Google cloud storage converts to S3" in str(err.value) -<<<<<<< HEAD def test_invalid_auth_header_format(started_cluster): node = started_cluster.instances["node1"] @@ -1187,23 +1182,11 @@ def test_iceberg_file_progress_callback(started_cluster): test_ref = f"test_progress_callback_{uuid.uuid4().hex[:8]}" table_name = f"{test_ref}_table" -======= -# TODO - turn on after merge alternative syntax -def _test_cluster_joins(started_cluster): - node = started_cluster.instances["node1"] - - test_ref = f"test_join_tables_{uuid.uuid4()}" - table_name = f"{test_ref}_table" - table_name_2 = f"{test_ref}_table_2" - table_name_local = f"{test_ref}_table_local" - ->>>>>>> d9d3710bd9b (Merge pull request #1646 from Altinity/frontport/antalya-26.3/fix_remote_calls) root_namespace = f"{test_ref}_namespace" catalog = load_catalog_impl(started_cluster) catalog.create_namespace(root_namespace) -<<<<<<< HEAD table = create_table( catalog, root_namespace, @@ -1258,7 +1241,22 @@ def _test_cluster_joins(started_cluster): f"`IcebergIterator::next` did not invoke the file-progress callback " f"(regression of PR #105413 wiring)." ) -======= + + +# TODO - turn on after merge alternative syntax +def _test_cluster_joins(started_cluster): + node = started_cluster.instances["node1"] + + test_ref = f"test_join_tables_{uuid.uuid4()}" + table_name = f"{test_ref}_table" + table_name_2 = f"{test_ref}_table_2" + table_name_local = f"{test_ref}_table_local" + + root_namespace = f"{test_ref}_namespace" + + catalog = load_catalog_impl(started_cluster) + catalog.create_namespace(root_namespace) + schema = Schema( NestedField( field_id=1, @@ -1384,4 +1382,3 @@ def _test_cluster_joins(started_cluster): ) assert res == "Jack\tBlack\nJack\tSilver\nJohn\tBlack\nJohn\tSilver\n" ->>>>>>> d9d3710bd9b (Merge pull request #1646 from Altinity/frontport/antalya-26.3/fix_remote_calls) diff --git a/tests/integration/test_s3_cluster/test.py b/tests/integration/test_s3_cluster/test.py index 3371f7013325..82c68c405598 100644 --- a/tests/integration/test_s3_cluster/test.py +++ b/tests/integration/test_s3_cluster/test.py @@ -3,12 +3,6 @@ import os import shutil import uuid -<<<<<<< HEAD -======= -import threading -import time -from email.errors import HeaderParseError ->>>>>>> d9d3710bd9b (Merge pull request #1646 from Altinity/frontport/antalya-26.3/fix_remote_calls) import pytest @@ -557,11 +551,6 @@ def test_cluster_default_expression(started_cluster): assert result == expected_result -<<<<<<< HEAD -@pytest.mark.parametrize("allow_experimental_analyzer", [0, 1]) -@pytest.mark.parametrize("use_partition_strategy", [False, True]) -def test_hive_partitioning(started_cluster, allow_experimental_analyzer, use_partition_strategy): -======= def test_remote_hedged(started_cluster): node = started_cluster.instances["s0_0_0"] pure_s3 = node.query( @@ -616,140 +605,6 @@ def test_remote_no_hedged(started_cluster): assert TSV(pure_s3) == TSV(s3_distributed) -@pytest.mark.parametrize("allow_experimental_analyzer", [0, 1]) -def test_hive_partitioning(started_cluster, allow_experimental_analyzer): - node = started_cluster.instances["s0_0_0"] - - node.query(f"SET allow_experimental_analyzer = {allow_experimental_analyzer}") - - for i in range(1, 5): - exists = node.query( - f""" - SELECT - count() - FROM s3('http://minio1:9001/root/data/hive/key={i}/*', 'minio', '{minio_secret_key}', 'Parquet', 'key Int32, value Int32') - GROUP BY ALL - FORMAT TSV - """ - ) - if int(exists) == 0: - node.query( - f""" - INSERT - INTO FUNCTION s3('http://minio1:9001/root/data/hive/key={i}/data.parquet', 'minio', '{minio_secret_key}', 'Parquet', 'key Int32, value Int32') - SELECT {i}, {i} - SETTINGS use_hive_partitioning = 0 - """ - ) - - settings = "enable_filesystem_cache = 0, use_query_cache = 0, use_cache_for_count_from_files = 0, use_iceberg_metadata_files_cache = 0, use_parquet_metadata_cache = 0, use_page_cache_for_object_storage = 0" - - query_id_full = str(uuid.uuid4()) - result = node.query( - f""" - SELECT count() - FROM s3('http://minio1:9001/root/data/hive/key=**.parquet', 'minio', '{minio_secret_key}', 'Parquet', 'key Int32, value Int32') - WHERE key <= 2 - FORMAT TSV - SETTINGS {settings}, use_hive_partitioning = 0 - """, - query_id=query_id_full, - ) - result = int(result) - assert result == 2 - - query_id_optimized = str(uuid.uuid4()) - result = node.query( - f""" - SELECT count() - FROM s3('http://minio1:9001/root/data/hive/key=**.parquet', 'minio', '{minio_secret_key}', 'Parquet', 'key Int32, value Int32') - WHERE key <= 2 - FORMAT TSV - SETTINGS {settings}, use_hive_partitioning = 1 - """, - query_id=query_id_optimized, - ) - result = int(result) - assert result == 2 - - query_id_cluster_full = str(uuid.uuid4()) - result = node.query( - f""" - SELECT count() - FROM s3Cluster(cluster_simple, 'http://minio1:9001/root/data/hive/key=**.parquet', 'minio', '{minio_secret_key}', 'Parquet', 'key Int32, value Int32') - WHERE key <= 2 - FORMAT TSV - SETTINGS {settings}, use_hive_partitioning = 0 - """, - query_id=query_id_cluster_full, - ) - result = int(result) - assert result == 2 - - query_id_cluster_optimized = str(uuid.uuid4()) - result = node.query( - f""" - SELECT count() - FROM s3Cluster(cluster_simple, 'http://minio1:9001/root/data/hive/key=**.parquet', 'minio', '{minio_secret_key}', 'Parquet', 'key Int32, value Int32') - WHERE key <= 2 - FORMAT TSV - SETTINGS {settings}, use_hive_partitioning = 1 - """, - query_id=query_id_cluster_optimized, - ) - result = int(result) - assert result == 2 - - node.query("SYSTEM FLUSH LOGS ON CLUSTER 'cluster_simple'") - - full_traffic = node.query( - f""" - SELECT sum(ProfileEvents['ReadBufferFromS3Bytes']) - FROM clusterAllReplicas(cluster_simple, system.query_log) - WHERE type='QueryFinish' AND initial_query_id='{query_id_full}' - FORMAT TSV - """ - ) - full_traffic = int(full_traffic) - assert full_traffic > 0 # 612*4 - - optimized_traffic = node.query( - f""" - SELECT sum(ProfileEvents['ReadBufferFromS3Bytes']) - FROM clusterAllReplicas(cluster_simple, system.query_log) - WHERE type='QueryFinish' AND initial_query_id='{query_id_optimized}' - FORMAT TSV - """ - ) - optimized_traffic = int(optimized_traffic) - assert optimized_traffic > 0 # 612*2 - assert full_traffic > optimized_traffic - - cluster_full_traffic = node.query( - f""" - SELECT sum(ProfileEvents['ReadBufferFromS3Bytes']) - FROM clusterAllReplicas(cluster_simple, system.query_log) - WHERE type='QueryFinish' AND initial_query_id='{query_id_cluster_full}' - FORMAT TSV - """ - ) - cluster_full_traffic = int(cluster_full_traffic) - assert cluster_full_traffic == full_traffic - - cluster_optimized_traffic = node.query( - f""" - SELECT sum(ProfileEvents['ReadBufferFromS3Bytes']) - FROM clusterAllReplicas(cluster_simple, system.query_log) - WHERE type='QueryFinish' AND initial_query_id='{query_id_cluster_optimized}' - FORMAT TSV - """ - ) - cluster_optimized_traffic = int(cluster_optimized_traffic) - assert cluster_optimized_traffic == optimized_traffic - - node.query("SET allow_experimental_analyzer = DEFAULT") - - def test_joins(started_cluster): node = started_cluster.instances["s0_0_0"] @@ -915,8 +770,9 @@ def test_joins(started_cluster): assert len(res) == 25 -def test_graceful_shutdown(started_cluster): ->>>>>>> d9d3710bd9b (Merge pull request #1646 from Altinity/frontport/antalya-26.3/fix_remote_calls) +@pytest.mark.parametrize("allow_experimental_analyzer", [0, 1]) +@pytest.mark.parametrize("use_partition_strategy", [False, True]) +def test_hive_partitioning(started_cluster, allow_experimental_analyzer, use_partition_strategy): node = started_cluster.instances["s0_0_0"] data_path = f"root/data/hive_{allow_experimental_analyzer}/{random_string(6)}" From 8c26761f67c0779dc0f7b24dfe36b2226ffe3086 Mon Sep 17 00:00:00 2001 From: Andrey Zvonov <32552679+zvonand@users.noreply.github.com> Date: Fri, 8 May 2026 00:17:13 +0200 Subject: [PATCH 05/43] Cherry-pick of https://github.com/Altinity/ClickHouse/pull/1640 with unresolved conflict markers (resolution in next commit) --- Original cherry-pick message follows: Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax 26.3 Antalya port - Alternative syntax for cluster functions # Conflicts: # docs/en/sql-reference/table-functions/iceberg.md # src/Analyzer/FunctionNode.h # src/Common/ErrorCodes.cpp # src/Core/Settings.cpp # src/Databases/DataLake/DatabaseDataLake.cpp # src/Databases/DataLake/DatabaseDataLakeSettings.cpp # src/Databases/DataLake/GlueCatalog.cpp # src/Databases/DataLake/ICatalog.cpp # src/Databases/DataLake/RestCatalog.cpp # src/Databases/DataLake/UnityCatalog.cpp # src/Disks/DiskObjectStorage/ObjectStorages/S3/S3ObjectStorage.cpp # src/IO/S3/URI.cpp # src/IO/S3/URI.h # src/IO/S3/getObjectInfo.cpp # src/Interpreters/IcebergMetadataLog.cpp # src/Parsers/FunctionSecretArgumentsFinder.h # src/Server/TCPHandler.cpp # src/Storages/IStorageCluster.h # src/Storages/ObjectStorage/DataLakes/DataLakeConfiguration.h # src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergMetadata.cpp # src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergMetadata.h # src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergWrites.cpp # src/Storages/ObjectStorage/DataLakes/Iceberg/PersistentTableComponents.h # src/Storages/ObjectStorage/DataLakes/Iceberg/SchemaProcessor.cpp # src/Storages/ObjectStorage/DataLakes/Iceberg/SchemaProcessor.h # src/Storages/ObjectStorage/DataLakes/Iceberg/Utils.cpp # src/Storages/ObjectStorage/S3/Configuration.cpp # src/Storages/ObjectStorage/StorageObjectStorage.cpp # src/Storages/ObjectStorage/StorageObjectStorageCluster.cpp # src/Storages/ObjectStorage/StorageObjectStorageConfiguration.h # src/Storages/ObjectStorage/StorageObjectStorageSource.cpp # src/Storages/ObjectStorage/registerStorageObjectStorage.cpp # src/Storages/System/StorageSystemTables.cpp # src/TableFunctions/TableFunctionObjectStorage.cpp # tests/integration/compose/docker_compose_iceberg_rest_catalog.yml # tests/integration/helpers/iceberg_utils.py # tests/integration/test_database_delta/test.py # tests/integration/test_database_glue/test.py # tests/integration/test_database_iceberg/test.py # tests/integration/test_mask_sensitive_info/test.py # tests/integration/test_s3_cluster/test.py --- docs/en/antalya/swarm.md | 73 ++ docs/en/engines/database-engines/datalake.md | 57 +- .../table-engines/integrations/iceberg.md | 56 ++ .../sql-reference/distribution-on-cluster.md | 23 + .../azureBlobStorageCluster.md | 14 + .../table-functions/deltalakeCluster.md | 11 + .../table-functions/hdfsCluster.md | 12 + .../table-functions/hudiCluster.md | 12 + .../sql-reference/table-functions/iceberg.md | 43 + .../table-functions/icebergCluster.md | 75 ++ .../table-functions/s3Cluster.md | 17 + src/Analyzer/FunctionNode.cpp | 30 +- src/Analyzer/FunctionNode.h | 19 + .../FunctionSecretArgumentsFinderTreeNode.h | 10 +- src/Analyzer/QueryTreeBuilder.cpp | 7 +- src/Analyzer/Resolve/QueryAnalyzer.cpp | 1 + src/Common/ErrorCodes.cpp | 4 + src/Common/ProfileEvents.cpp | 8 + src/Core/Settings.cpp | 42 + src/Core/SettingsChangesHistory.cpp | 27 +- src/Databases/DataLake/Common.cpp | 12 +- src/Databases/DataLake/Common.h | 3 +- src/Databases/DataLake/DataLakeConstants.h | 1 + src/Databases/DataLake/DatabaseDataLake.cpp | 77 +- src/Databases/DataLake/DatabaseDataLake.h | 2 +- .../DataLake/DatabaseDataLakeSettings.cpp | 5 + src/Databases/DataLake/GlueCatalog.cpp | 40 +- src/Databases/DataLake/GlueCatalog.h | 5 + src/Databases/DataLake/HiveCatalog.cpp | 16 +- src/Databases/DataLake/HiveCatalog.h | 14 +- src/Databases/DataLake/ICatalog.cpp | 64 +- src/Databases/DataLake/ICatalog.h | 11 + src/Databases/DataLake/PaimonRestCatalog.cpp | 6 +- src/Databases/DataLake/PaimonRestCatalog.h | 4 +- src/Databases/DataLake/RestCatalog.cpp | 135 ++- src/Databases/DataLake/RestCatalog.h | 27 + src/Databases/DataLake/UnityCatalog.cpp | 23 +- src/Databases/DataLake/UnityCatalog.h | 7 + .../gtest_rest_catalog_allowed_namespaces.cpp | 69 ++ .../AzureBlobStorage/AzureObjectStorage.cpp | 13 +- .../ObjectStorages/S3/S3ObjectStorage.cpp | 18 +- src/Disks/DiskType.cpp | 63 +- src/Disks/DiskType.h | 5 +- src/IO/ReadBufferFromS3.cpp | 6 + src/IO/S3/Client.cpp | 14 +- src/IO/S3/Client.h | 2 +- src/IO/S3/URI.cpp | 63 ++ src/IO/S3/URI.h | 6 + src/IO/S3/getObjectInfo.cpp | 7 + src/Interpreters/Cluster.cpp | 37 +- src/Interpreters/Cluster.h | 7 +- src/Interpreters/ClusterDiscovery.cpp | 37 +- src/Interpreters/IcebergMetadataLog.cpp | 8 +- src/Interpreters/IcebergMetadataLog.h | 4 +- src/Interpreters/InterpreterCreateQuery.cpp | 3 +- src/Interpreters/InterpreterInsertQuery.cpp | 3 + src/Parsers/ASTSetQuery.cpp | 3 +- src/Parsers/FunctionSecretArgumentsFinder.h | 122 ++- .../FunctionSecretArgumentsFinderAST.h | 9 +- .../QueryPlan/ReadFromObjectStorageStep.cpp | 2 +- src/Server/TCPHandler.cpp | 5 + src/Storages/HivePartitioningUtils.cpp | 6 +- src/Storages/IStorage.h | 5 +- src/Storages/IStorageCluster.cpp | 189 +++- src/Storages/IStorageCluster.h | 62 +- .../MergeTree/ExportPartitionUtils.cpp | 4 +- src/Storages/MergeTree/MergeTreeData.cpp | 9 +- .../ObjectStorage/Azure/Configuration.cpp | 41 +- .../ObjectStorage/Azure/Configuration.h | 15 +- .../Common/AvroForIcebergDeserializer.cpp | 10 + .../DataLakes/DataLakeConfiguration.h | 573 +++++++++++- .../DataLakes/DataLakeStorageSettings.h | 3 + .../DeltaLakeMetadataDeltaKernel.cpp | 10 +- .../ObjectStorage/DataLakes/HudiMetadata.cpp | 8 +- .../ObjectStorage/DataLakes/HudiMetadata.h | 2 +- .../DataLakes/Iceberg/ChunkPartitioner.cpp | 12 +- .../DataLakes/Iceberg/ChunkPartitioner.h | 1 + .../DataLakes/Iceberg/IcebergMetadata.cpp | 51 +- .../DataLakes/Iceberg/IcebergMetadata.h | 4 + .../DataLakes/Iceberg/IcebergWrites.cpp | 6 + .../Iceberg/ManifestFileIterator.cpp | 20 +- .../Iceberg/ManifestFilesPruning.cpp | 11 +- .../DataLakes/Iceberg/ManifestFilesPruning.h | 2 +- .../Iceberg/PersistentTableComponents.h | 4 + .../DataLakes/Iceberg/SchemaProcessor.cpp | 73 +- .../DataLakes/Iceberg/SchemaProcessor.h | 27 +- .../Iceberg/StatelessMetadataFileGetter.cpp | 6 +- .../ObjectStorage/DataLakes/Iceberg/Utils.cpp | 48 +- .../ObjectStorage/DataLakes/Iceberg/Utils.h | 7 +- .../ObjectStorage/HDFS/Configuration.cpp | 8 + .../ObjectStorage/HDFS/Configuration.h | 2 + .../ObjectStorage/Local/Configuration.cpp | 17 + .../ObjectStorage/Local/Configuration.h | 2 + .../MultiFileStorageObjectStorageSink.cpp | 6 +- .../ObjectStorage/ReadBufferIterator.cpp | 10 +- .../ObjectStorage/S3/Configuration.cpp | 30 + src/Storages/ObjectStorage/S3/Configuration.h | 2 + .../ObjectStorage/StorageObjectStorage.cpp | 87 +- .../ObjectStorage/StorageObjectStorage.h | 6 +- .../StorageObjectStorageCluster.cpp | 843 +++++++++++++++++- .../StorageObjectStorageCluster.h | 177 +++- .../StorageObjectStorageConfiguration.cpp | 51 +- .../StorageObjectStorageConfiguration.h | 76 +- .../StorageObjectStorageSettings.h | 10 + .../StorageObjectStorageSink.cpp | 6 +- .../StorageObjectStorageSource.cpp | 29 +- src/Storages/ObjectStorage/Utils.cpp | 15 +- src/Storages/ObjectStorage/Utils.h | 3 +- .../registerStorageObjectStorage.cpp | 38 +- .../StorageObjectStorageQueue.cpp | 8 +- .../StorageObjectStorageQueue.h | 2 +- .../registerQueueStorage.cpp | 2 +- src/Storages/StorageDistributed.cpp | 4 +- src/Storages/StorageFileCluster.cpp | 6 +- src/Storages/StorageFileCluster.h | 6 +- src/Storages/StorageReplicatedMergeTree.cpp | 10 +- src/Storages/StorageURLCluster.cpp | 6 +- src/Storages/StorageURLCluster.h | 6 +- .../System/StorageSystemIcebergHistory.cpp | 6 +- src/Storages/System/StorageSystemTables.cpp | 95 ++ .../extractTableFunctionFromSelectQuery.h | 3 +- src/TableFunctions/ITableFunction.h | 2 +- src/TableFunctions/ITableFunctionCluster.h | 15 +- .../TableFunctionObjectStorage.cpp | 124 +-- .../TableFunctionObjectStorage.h | 34 +- .../TableFunctionObjectStorageCluster.cpp | 53 +- .../TableFunctionObjectStorageCluster.h | 19 +- ...leFunctionObjectStorageClusterFallback.cpp | 458 ++++++++++ ...ableFunctionObjectStorageClusterFallback.h | 50 ++ src/TableFunctions/TableFunctionRemote.h | 2 + src/TableFunctions/registerTableFunctions.cpp | 1 + src/TableFunctions/registerTableFunctions.h | 1 + .../docker_compose_iceberg_rest_catalog.yml | 25 + tests/integration/helpers/cluster.py | 9 + tests/integration/helpers/iceberg_utils.py | 84 +- tests/integration/test_database_delta/test.py | 43 + tests/integration/test_database_glue/test.py | 42 + .../configs/iceberg_partition_timezone.xml | 7 + .../configs/timezone.xml | 3 + .../integration/test_database_iceberg/test.py | 169 +++- .../test_partition_timezone.py | 186 ++++ .../test_mask_sensitive_info/test.py | 106 ++- .../test_s3_cluster/configs/cluster.xml | 102 +++ .../configs/hidden_clusters.xml | 20 + .../test_s3_cluster/configs/users.xml | 13 + tests/integration/test_s3_cluster/test.py | 550 ++++++++++++ .../configs/config.d/named_collections.xml | 14 + .../configs/config.d/named_collections.xml | 14 + .../test_cluster_joins.py | 3 +- .../test_cluster_table_function.py | 185 +++- .../test_minmax_pruning_with_null.py | 8 +- .../test_partition_pruning.py | 2 +- .../test_types.py | 46 + .../01625_constraints_index_append.reference | 4 +- 154 files changed, 5995 insertions(+), 679 deletions(-) create mode 100644 docs/en/antalya/swarm.md create mode 100644 docs/en/sql-reference/distribution-on-cluster.md create mode 100644 src/Databases/DataLake/tests/gtest_rest_catalog_allowed_namespaces.cpp create mode 100644 src/TableFunctions/TableFunctionObjectStorageClusterFallback.cpp create mode 100644 src/TableFunctions/TableFunctionObjectStorageClusterFallback.h create mode 100644 tests/integration/test_database_iceberg/configs/iceberg_partition_timezone.xml create mode 100644 tests/integration/test_database_iceberg/configs/timezone.xml create mode 100644 tests/integration/test_database_iceberg/test_partition_timezone.py create mode 100644 tests/integration/test_s3_cluster/configs/hidden_clusters.xml diff --git a/docs/en/antalya/swarm.md b/docs/en/antalya/swarm.md new file mode 100644 index 000000000000..a26f9de26e0a --- /dev/null +++ b/docs/en/antalya/swarm.md @@ -0,0 +1,73 @@ +# Antalya branch + +## Swarm + +### Difference with upstream version + +#### `storage_type` argument in object storage functions + +In upstream ClickHouse, there are several table functions to read Iceberg tables from different storage backends such as `icebergLocal`, `icebergS3`, `icebergAzure`, `icebergHDFS`, cluster variants, the `iceberg` function as a synonym for `icebergS3`, and table engines like `IcebergLocal`, `IcebergS3`, `IcebergAzure`, `IcebergHDFS`. + +In the Antalya branch, the `iceberg` table function and the `Iceberg` table engine unify all variants into one by using a new named argument, `storage_type`, which can be one of `local`, `s3`, `azure`, or `hdfs`. + +Old syntax examples: + +```sql +SELECT * FROM icebergS3('http://minio1:9000/root/table_data', 'minio', 'minio123', 'Parquet'); +SELECT * FROM icebergAzureCluster('mycluster', 'http://azurite1:30000/devstoreaccount1', 'cont', '/table_data', 'devstoreaccount1', 'Eby8vdM02xNOcqFlqUwJPLlmEtlCDXJ1OUzFT50uSRZ6IFsuFq2UVErCz4I6tq/K1SZFPTOtr/KBHBeksoGMGw==', 'Parquet'); +CREATE TABLE mytable ENGINE=IcebergHDFS('/table_data', 'Parquet'); +``` + +New syntax examples: + +```sql +SELECT * FROM iceberg(storage_type='s3', 'http://minio1:9000/root/table_data', 'minio', 'minio123', 'Parquet'); +SELECT * FROM icebergCluster('mycluster', storage_type='azure', 'http://azurite1:30000/devstoreaccount1', 'cont', '/table_data', 'devstoreaccount1', 'Eby8vdM02xNOcqFlqUwJPLlmEtlCDXJ1OUzFT50uSRZ6IFsuFq2UVErCz4I6tq/K1SZFPTOtr/KBHBeksoGMGw==', 'Parquet'); +CREATE TABLE mytable ENGINE=Iceberg('/table_data', 'Parquet', storage_type='hdfs'); +``` + +Also, if a named collection is used to store access parameters, the field `storage_type` can be included in the same named collection: + +```xml + + + http://minio1:9001/root/ + minio + minio123 + s3 + + +``` + +```sql +SELECT * FROM iceberg(s3, filename='table_data'); +``` + +By default `storage_type` is `'s3'` to maintain backward compatibility. + + +#### `object_storage_cluster` setting + +The new setting `object_storage_cluster` controls whether a single-node or cluster variant of table functions reading from object storage (e.g., `s3`, `azure`, `iceberg`, and their cluster variants like `s3Cluster`, `azureCluster`, `icebergCluster`) is used. + +Old syntax examples: + +```sql +SELECT * from s3Cluster('myCluster', 'http://minio1:9001/root/data/{clickhouse,database}/*', 'minio', 'minio123', 'CSV', + 'name String, value UInt32, polygon Array(Array(Tuple(Float64, Float64)))'); +SELECT * FROM icebergAzureCluster('mycluster', 'http://azurite1:30000/devstoreaccount1', 'cont', '/table_data', 'devstoreaccount1', 'Eby8vdM02xNOcqFlqUwJPLlmEtlCDXJ1OUzFT50uSRZ6IFsuFq2UVErCz4I6tq/K1SZFPTOtr/KBHBeksoGMGw==', 'Parquet'); +``` + +New syntax examples: + +```sql +SELECT * from s3('http://minio1:9001/root/data/{clickhouse,database}/*', 'minio', 'minio123', 'CSV', + 'name String, value UInt32, polygon Array(Array(Tuple(Float64, Float64)))') + SETTINGS object_storage_cluster='myCluster'; +SELECT * FROM icebergAzure('http://azurite1:30000/devstoreaccount1', 'cont', '/table_data', 'devstoreaccount1', 'Eby8vdM02xNOcqFlqUwJPLlmEtlCDXJ1OUzFT50uSRZ6IFsuFq2UVErCz4I6tq/K1SZFPTOtr/KBHBeksoGMGw==', 'Parquet') + SETTINGS object_storage_cluster='myCluster'; +``` + +This setting also applies to table engines and can be used with tables managed by Iceberg Catalog. + +Note: The upstream ClickHouse has introduced analogous settings, such as `parallel_replicas_for_cluster_engines` and `cluster_for_parallel_replicas`. Since version 25.10, these settings work with table engines. It is possible that in the future, the `object_storage_cluster` setting will be deprecated. diff --git a/docs/en/engines/database-engines/datalake.md b/docs/en/engines/database-engines/datalake.md index b37fc38f790d..10fb660035f7 100644 --- a/docs/en/engines/database-engines/datalake.md +++ b/docs/en/engines/database-engines/datalake.md @@ -44,21 +44,22 @@ catalog_type, The following settings are supported: -| Setting | Description | -|-------------------------|-----------------------------------------------------------------------------------------| -| `catalog_type` | Type of catalog: `glue`, `unity` (Delta), `rest` (Iceberg), `hive`, `onelake` (Iceberg) | -| `warehouse` | The warehouse/database name to use in the catalog. | -| `catalog_credential` | Authentication credential for the catalog (e.g., API key or token) | -| `auth_header` | Custom HTTP header for authentication with the catalog service | -| `auth_scope` | OAuth2 scope for authentication (if using OAuth) | -| `storage_endpoint` | Endpoint URL for the underlying storage | -| `oauth_server_uri` | URI of the OAuth2 authorization server for authentication | +| Setting | Description | +|-------------------------|-----------------------------------------------------------------------------------------------| +| `catalog_type` | Type of catalog: `glue`, `unity` (Delta), `rest` (Iceberg), `hive`, `onelake` (Iceberg) | +| `warehouse` | The warehouse/database name to use in the catalog. | +| `catalog_credential` | Authentication credential for the catalog (e.g., API key or token) | +| `auth_header` | Custom HTTP header for authentication with the catalog service | +| `auth_scope` | OAuth2 scope for authentication (if using OAuth) | +| `storage_endpoint` | Endpoint URL for the underlying storage | +| `oauth_server_uri` | URI of the OAuth2 authorization server for authentication | | `vended_credentials` | Boolean indicating whether to use vended credentials from the catalog (supports AWS S3 and Azure ADLS Gen2) | -| `aws_access_key_id` | AWS access key ID for S3/Glue access (if not using vended credentials) | -| `aws_secret_access_key` | AWS secret access key for S3/Glue access (if not using vended credentials) | -| `region` | AWS region for the service (e.g., `us-east-1`) | -| `dlf_access_key_id` | Access key ID for DLF access | -| `dlf_access_key_secret` | Access key Secret for DLF access | +| `aws_access_key_id` | AWS access key ID for S3/Glue access (if not using vended credentials) | +| `aws_secret_access_key` | AWS secret access key for S3/Glue access (if not using vended credentials) | +| `region` | AWS region for the service (e.g., `us-east-1`) | +| `dlf_access_key_id` | Access key ID for DLF access | +| `dlf_access_key_secret` | Access key Secret for DLF access | +| `namespaces` | Comma-separated list of namespaces, implemented for catalog types: `rest`, `glue` and `unity` | ## Examples {#examples} @@ -81,4 +82,30 @@ SETTINGS onelake_client_secret = client_secret; SHOW TABLES IN database_name; SELECT count() from database_name.table_name; -``` \ No newline at end of file +``` + +## Namespace filter {#namespace} + +By default, ClickHouse reads tables from all namespaces available in the catalog. You can limit this behavior using the `namespaces` database setting. The value should be a comma‑separated list of namespaces that are allowed to be read. + +Supported catalog types are `rest`, `glue` and `unity`. + +For example, if the catalog contains three namespaces - `dev`, `stage`, and `prod` - and you want to read data only from dev and stage, set: +``` +namespaces='dev,stage' +``` + +### Nested namespaces {#namespace-nested} + +The Iceberg (`rest`) catalog supports nested namespaces. The `namespaces` filter accepts the following patterns: + +- `namespace` - includes tables from the specified namespace, but not from its nested namespaces. +- `namespace.nested` - includes tables from the nested namespace, but not from the parent. +- `namespace.*` - includes tables from all nested namespaces, but not from the parent. + +If you need to include both a namespace and its nested namespaces, specify both explicitly. For example: +``` +namespaces='namespace,namespace.*' +``` + +The default value is '*', which means all namespaces are included. diff --git a/docs/en/engines/table-engines/integrations/iceberg.md b/docs/en/engines/table-engines/integrations/iceberg.md index 7a93d3f18c35..e36e90ffb3fd 100644 --- a/docs/en/engines/table-engines/integrations/iceberg.md +++ b/docs/en/engines/table-engines/integrations/iceberg.md @@ -356,6 +356,62 @@ SETTINGS iceberg_metadata_staleness_ms=120000 **Note**: Current expectation is that metadata cache size is sufficient to hold the latest metadata snapshot in full for all active tables, if asynchronous prefetching is enabled. +## Altinity Antalya branch + +### Specify storage type in arguments + +Only in the Altinity Antalya branch does `Iceberg` table engine support all storage types. The storage type can be specified using the named argument `storage_type`. Supported values are `s3`, `azure`, `hdfs`, and `local`. + +```sql +CREATE TABLE iceberg_table_s3 + ENGINE = Iceberg(storage_type='s3', url, [, NOSIGN | access_key_id, secret_access_key, [session_token]], format, [,compression]) + +CREATE TABLE iceberg_table_azure + ENGINE = Iceberg(storage_type='azure', connection_string|storage_account_url, container_name, blobpath, [account_name, account_key, format, compression]) + +CREATE TABLE iceberg_table_hdfs + ENGINE = Iceberg(storage_type='hdfs', path_to_table, [,format] [,compression_method]) + +CREATE TABLE iceberg_table_local + ENGINE = Iceberg(storage_type='local', path_to_table, [,format] [,compression_method]) +``` + +### Specify storage type in named collection + +Only in Altinity Antalya branch `storage_type` can be included as part of a named collection. This allows for centralized configuration of storage settings. + +```xml + + + + http://test.s3.amazonaws.com/clickhouse-bucket/ + test + test + auto + auto + s3 + + + +``` + +```sql +CREATE TABLE iceberg_table ENGINE=Iceberg(iceberg_conf, filename = 'test_table') +``` + +The default value for `storage_type` is `s3`. + +### The `object_storage_cluster` setting. + +Only in the Altinity Antalya branch is an alternative syntax for the `Iceberg` table engine available. This syntax allows execution on a cluster when the `object_storage_cluster` setting is non-empty and contains the cluster name. + +```sql +CREATE TABLE iceberg_table_s3 + ENGINE = Iceberg(storage_type='s3', url, [, NOSIGN | access_key_id, secret_access_key, [session_token]], format, [,compression]); + +SELECT * FROM iceberg_table_s3 SETTINGS object_storage_cluster='cluster_simple'; +``` + ## See also {#see-also} - [iceberg table function](/sql-reference/table-functions/iceberg.md) diff --git a/docs/en/sql-reference/distribution-on-cluster.md b/docs/en/sql-reference/distribution-on-cluster.md new file mode 100644 index 000000000000..3a9835e23856 --- /dev/null +++ b/docs/en/sql-reference/distribution-on-cluster.md @@ -0,0 +1,23 @@ +# Task distribution in *Cluster family functions + +## Task distribution algorithm + +Table functions such as `s3Cluster`, `azureBlobStorageCluster`, `hdsfCluster`, `icebergCluster`, and table engines like `S3`, `Azure`, `HDFS`, `Iceberg` with the setting `object_storage_cluster` distribute tasks across all cluster nodes or a subset limited by the `object_storage_max_nodes` setting. This setting limits the number of nodes involved in processing a distributed query, randomly selecting nodes for each query. + +A single task corresponds to processing one source file. + +For each file, one cluster node is selected as the primary node using a consistent Rendezvous Hashing algorithm. This algorithm guarantees that: + * The same node is consistently selected as primary for each file, as long as the cluster remains unchanged. + * When the cluster changes (nodes added or removed), only files assigned to those affected nodes change their primary node assignment. + +This improves cache efficiency by minimizing data movement among nodes. + +## `lock_object_storage_task_distribution_ms` setting + +Each node begins processing files for which it is the primary node. After completing its assigned files, a node may take tasks from other nodes, either immediately or after waiting for `lock_object_storage_task_distribution_ms` milliseconds if the primary node does not request new files during that interval. The default value of `lock_object_storage_task_distribution_ms` is 500 milliseconds. This setting balances between caching efficiency and workload redistribution when nodes are imbalanced. + +## `SYSTEM STOP SWARM MODE` command + +If a node needs to shut down gracefully, the command `SYSTEM STOP SWARM MODE` prevents the node from receiving new tasks for *Cluster-family queries. The node finishes processing already assigned files before it can safely shut down without errors. + +Receiving new tasks can be resumed with the command `SYSTEM START SWARM MODE`. diff --git a/docs/en/sql-reference/table-functions/azureBlobStorageCluster.md b/docs/en/sql-reference/table-functions/azureBlobStorageCluster.md index b7c98cbfede2..e9c154576473 100644 --- a/docs/en/sql-reference/table-functions/azureBlobStorageCluster.md +++ b/docs/en/sql-reference/table-functions/azureBlobStorageCluster.md @@ -52,6 +52,20 @@ SELECT count(*) FROM azureBlobStorageCluster( See [azureBlobStorage](/sql-reference/table-functions/azureBlobStorage#using-shared-access-signatures-sas-sas-tokens) for examples. +## Altinity Antalya branch + +### `object_storage_cluster` setting. + +Only in the Altinity Antalya branch, the alternative syntax for the `azureBlobStorageCluster` table function is avilable. This allows the `azureBlobStorage` function to be used with the non-empty `object_storage_cluster` setting, specifying a cluster name. This enables distributed queries over Azure Blob Storage across a ClickHouse cluster. + +```sql +SELECT count(*) FROM azureBlobStorage( + 'http://azurite1:10000/devstoreaccount1', 'testcontainer', 'test_cluster_count.csv', 'devstoreaccount1', + 'Eby8vdM02xNOcqFlqUwJPLlmEtlCDXJ1OUzFT50uSRZ6IFsuFq2UVErCz4I6tq/K1SZFPTOtr/KBHBeksoGMGw==', 'CSV', + 'auto', 'key UInt64') +SETTINGS object_storage_cluster='cluster_simple' +``` + ## Related {#related} - [AzureBlobStorage engine](../../engines/table-engines/integrations/azureBlobStorage.md) diff --git a/docs/en/sql-reference/table-functions/deltalakeCluster.md b/docs/en/sql-reference/table-functions/deltalakeCluster.md index 3ac9a4b3b6a2..35451433992b 100644 --- a/docs/en/sql-reference/table-functions/deltalakeCluster.md +++ b/docs/en/sql-reference/table-functions/deltalakeCluster.md @@ -43,6 +43,17 @@ A table with the specified structure for reading data from cluster in the specif - `_time` — Last modified time of the file. Type: `Nullable(DateTime)`. If the time is unknown, the value is `NULL`. - `_etag` — The etag of the file. Type: `LowCardinality(String)`. If the etag is unknown, the value is `NULL`. +## Altinity Antalya branch + +### `object_storage_cluster` setting. + +Only in the Altinity Antalya branch alternative syntax for `deltaLakeCluster` table function is available. This allows the `deltaLake` function to be used with the non-empty `object_storage_cluster` setting, specifying a cluster name. This enables distributed queries over Delta Lake Storage across a ClickHouse cluster. + +```sql +SELECT count(*) FROM deltaLake(url [,aws_access_key_id, aws_secret_access_key] [,format] [,structure] [,compression]) +SETTINGS object_storage_cluster='cluster_simple' +``` + ## Related {#related} - [deltaLake engine](/engines/table-engines/integrations/deltalake.md) diff --git a/docs/en/sql-reference/table-functions/hdfsCluster.md b/docs/en/sql-reference/table-functions/hdfsCluster.md index 8c25a713fade..b2559e805960 100644 --- a/docs/en/sql-reference/table-functions/hdfsCluster.md +++ b/docs/en/sql-reference/table-functions/hdfsCluster.md @@ -58,6 +58,18 @@ FROM hdfsCluster('cluster_simple', 'hdfs://hdfs1:9000/{some,another}_dir/*', 'TS If your listing of files contains number ranges with leading zeros, use the construction with braces for each digit separately or use `?`. ::: +## Altinity Antalya branch + +### `object_storage_cluster` setting. + +Only in the Altinity Antalya branch alternative syntax for `hdfsCluster` table function is available. This allows the `hdfs` function to be used with the non-empty `object_storage_cluster` setting, specifying a cluster name. This enables distributed queries over HDFS Storage across a ClickHouse cluster. + +```sql +SELECT count(*) +FROM hdfs('hdfs://hdfs1:9000/{some,another}_dir/*', 'TSV', 'name String, value UInt32') +SETTINGS object_storage_cluster='cluster_simple' +``` + ## Related {#related} - [HDFS engine](../../engines/table-engines/integrations/hdfs.md) diff --git a/docs/en/sql-reference/table-functions/hudiCluster.md b/docs/en/sql-reference/table-functions/hudiCluster.md index 064f6e956ec9..8acd3963512b 100644 --- a/docs/en/sql-reference/table-functions/hudiCluster.md +++ b/docs/en/sql-reference/table-functions/hudiCluster.md @@ -42,6 +42,18 @@ A table with the specified structure for reading data from cluster in the specif - `_time` — Last modified time of the file. Type: `Nullable(DateTime)`. If the time is unknown, the value is `NULL`. - `_etag` — The etag of the file. Type: `LowCardinality(String)`. If the etag is unknown, the value is `NULL`. +## Altinity Antalya branch + +### `object_storage_cluster` setting. + +Only in the Altinity Antalya branch alternative syntax for `hudiCluster` table function is available. This allows the `hudi` function to be used with the non-empty `object_storage_cluster` setting, specifying a cluster name. This enables distributed queries over Hudi Storage across a ClickHouse cluster. + +```sql +SELECT * +FROM hudi(url [,aws_access_key_id, aws_secret_access_key] [,format] [,structure] [,compression]) +SETTINGS object_storage_cluster='cluster_simple' +``` + ## Related {#related} - [Hudi engine](/engines/table-engines/integrations/hudi.md) diff --git a/docs/en/sql-reference/table-functions/iceberg.md b/docs/en/sql-reference/table-functions/iceberg.md index 22f567d590cd..ba59c52ace13 100644 --- a/docs/en/sql-reference/table-functions/iceberg.md +++ b/docs/en/sql-reference/table-functions/iceberg.md @@ -633,6 +633,7 @@ GRANT ALTER TABLE ON my_iceberg_table TO my_user; - The catalog's own authorization (REST catalog auth, AWS Glue IAM, etc.) is enforced independently when ClickHouse updates the metadata ::: +<<<<<<< HEAD ### Remove Orphan Files {#iceberg-remove-orphan-files} Orphan files are files on storage that are not referenced by any snapshot in the Iceberg table metadata. They accumulate from failed writes, partial cleanup after compaction, and interrupted operations, causing unbounded storage growth. The `remove_orphan_files` command identifies and removes these orphan files. @@ -714,6 +715,48 @@ The command returns a table with `metric_name` and `metric_value` columns showin - Use `dry_run = 1` to preview orphan files before deletion - The `older_than` threshold protects against deleting files from in-progress writes — the default 3-day threshold provides a generous safety margin ::: +======= +## Altinity Antalya branch + +### Specify storage type in arguments + +Only in the Altinity Antalya branch does the `iceberg` table function support all storage types. The storage type can be specified using the named argument `storage_type`. Supported values are `s3`, `azure`, `hdfs`, and `local`. + +```sql +iceberg(storage_type='s3', url [, NOSIGN | access_key_id, secret_access_key, [session_token]] [,format] [,compression_method]) + +iceberg(storage_type='azure', connection_string|storage_account_url, container_name, blobpath, [,account_name], [,account_key] [,format] [,compression_method]) + +iceberg(storage_type='hdfs', path_to_table, [,format] [,compression_method]) + +iceberg(storage_type='local', path_to_table, [,format] [,compression_method]) +``` + +### Specify storage type in named collection + +Only in the Altinity Antalya branch can storage_type be included as part of a named collection. This allows for centralized configuration of storage settings. + +```xml + + + + http://test.s3.amazonaws.com/clickhouse-bucket/ + test + test + auto + auto + s3 + + + +``` + +```sql +iceberg(named_collection[, option=value [,..]]) +``` + +The default value for `storage_type` is `s3`. +>>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) ## See Also {#see-also} diff --git a/docs/en/sql-reference/table-functions/icebergCluster.md b/docs/en/sql-reference/table-functions/icebergCluster.md index 7e8ed1ebf688..91d6c9dea2ac 100644 --- a/docs/en/sql-reference/table-functions/icebergCluster.md +++ b/docs/en/sql-reference/table-functions/icebergCluster.md @@ -49,6 +49,81 @@ SELECT * FROM icebergS3Cluster('cluster_simple', 'http://test.s3.amazonaws.com/c - `_time` — Last modified time of the file. Type: `Nullable(DateTime)`. If the time is unknown, the value is `NULL`. - `_etag` — The etag of the file. Type: `LowCardinality(String)`. If the etag is unknown, the value is `NULL`. +## Altinity Antalya branch + +### `icebergLocalCluster` table function + +Only in the Altinity Antalya branch, `icebergLocalCluster` designed to make distributed cluster queries when Iceberg data is stored on shared network storage mounted with a local path. The path must be identical on all replicas. + +```sql +icebergLocalCluster(cluster_name, path_to_table, [,format] [,compression_method]) +``` + +### Specify storage type in function arguments + +Only in the Altinity Antalya branch, the `icebergCluster` table function supports all storage backends. The storage backend can be specified using the named argument `storage_type`. Valid values include `s3`, `azure`, `hdfs`, and `local`. + +```sql +icebergCluster(storage_type='s3', cluster_name, url [, NOSIGN | access_key_id, secret_access_key, [session_token]] [,format] [,compression_method]) + +icebergCluster(storage_type='azure', cluster_name, connection_string|storage_account_url, container_name, blobpath, [,account_name], [,account_key] [,format] [,compression_method]) + +icebergCluster(storage_type='hdfs', cluster_name, path_to_table, [,format] [,compression_method]) + +icebergCluster(storage_type='local', cluster_name, path_to_table, [,format] [,compression_method]) +``` + +### Specify storage type in a named collection + +Only in the Altinity Antalya branch, `storage_type` can be part of a named collection. + +```xml + + + + http://test.s3.amazonaws.com/clickhouse-bucket/ + test + test + auto + auto + s3 + + + +``` + +```sql +icebergCluster(iceberg_conf[, option=value [,..]]) +``` + +The default value for `storage_type` is `s3`. + +### `object_storage_cluster` setting. + +Only in the Altinity Antalya branch, an alternative syntax for `icebergCluster` table function is available. This allows the `iceberg` function to be used with the non-empty `object_storage_cluster` setting, specifying a cluster name. This enables distributed queries over Iceberg table across a ClickHouse cluster. + +```sql +icebergS3(url [, NOSIGN | access_key_id, secret_access_key, [session_token]] [,format] [,compression_method]) SETTINGS object_storage_cluster='cluster_name' + +icebergAzure(connection_string|storage_account_url, container_name, blobpath, [,account_name], [,account_key] [,format] [,compression_method]) SETTINGS object_storage_cluster='cluster_name' + +icebergHDFS(path_to_table, [,format] [,compression_method]) SETTINGS object_storage_cluster='cluster_name' + +icebergLocal(path_to_table, [,format] [,compression_method]) SETTINGS object_storage_cluster='cluster_name' + +icebergS3(option=value [,..]) SETTINGS object_storage_cluster='cluster_name' + +iceberg(storage_type='s3', url [, NOSIGN | access_key_id, secret_access_key, [session_token]] [,format] [,compression_method]) SETTINGS object_storage_cluster='cluster_name' + +iceberg(storage_type='azure', connection_string|storage_account_url, container_name, blobpath, [,account_name], [,account_key] [,format] [,compression_method]) SETTINGS object_storage_cluster='cluster_name' + +iceberg(storage_type='hdfs', path_to_table, [,format] [,compression_method]) SETTINGS object_storage_cluster='cluster_name' + +iceberg(storage_type='local', path_to_table, [,format] [,compression_method]) SETTINGS object_storage_cluster='cluster_name' + +iceberg(iceberg_conf[, option=value [,..]]) SETTINGS object_storage_cluster='cluster_name' +``` + **See Also** - [Iceberg engine](/engines/table-engines/integrations/iceberg.md) diff --git a/docs/en/sql-reference/table-functions/s3Cluster.md b/docs/en/sql-reference/table-functions/s3Cluster.md index a5a86601a3af..32e4f0df4471 100644 --- a/docs/en/sql-reference/table-functions/s3Cluster.md +++ b/docs/en/sql-reference/table-functions/s3Cluster.md @@ -89,6 +89,23 @@ Users can use the same approaches as document for the s3 function [here](/sql-re For details on optimizing the performance of the s3 function see [our detailed guide](/integrations/s3/performance). +## Altinity Antalya branch + +### `object_storage_cluster` setting. + +Only in the Altinity Antalya branch alternative syntax for `s3Cluster` table function is available. This allows the `s3` function to be used with the non-empty `object_storage_cluster` setting, specifying a cluster name. This enables distributed queries over S3 Storage across a ClickHouse cluster. + +```sql +SELECT * FROM s3( + 'http://minio1:9001/root/data/{clickhouse,database}/*', + 'minio', + 'ClickHouse_Minio_P@ssw0rd', + 'CSV', + 'name String, value UInt32, polygon Array(Array(Tuple(Float64, Float64)))' +) ORDER BY (name, value, polygon) +SETTINGS object_storage_cluster='cluster_simple' +``` + ## Related {#related} - [S3 engine](../../engines/table-engines/integrations/s3.md) diff --git a/src/Analyzer/FunctionNode.cpp b/src/Analyzer/FunctionNode.cpp index a154393cb1af..f4b6afc8b977 100644 --- a/src/Analyzer/FunctionNode.cpp +++ b/src/Analyzer/FunctionNode.cpp @@ -12,6 +12,7 @@ #include #include +#include #include @@ -164,6 +165,13 @@ void FunctionNode::dumpTreeImpl(WriteBuffer & buffer, FormatState & format_state buffer << '\n' << std::string(indent + 2, ' ') << "WINDOW\n"; getWindowNode()->dumpTreeImpl(buffer, format_state, indent + 4); } + + if (!settings_changes.empty()) + { + buffer << '\n' << std::string(indent + 2, ' ') << "SETTINGS"; + for (const auto & change : settings_changes) + buffer << fmt::format(" {}={}", change.name, fieldToString(change.value)); + } } bool FunctionNode::isEqualImpl(const IQueryTreeNode & rhs, CompareOptions /*compare_options*/) const @@ -171,7 +179,7 @@ bool FunctionNode::isEqualImpl(const IQueryTreeNode & rhs, CompareOptions /*comp const auto & rhs_typed = assert_cast(rhs); if (function_name != rhs_typed.function_name || isAggregateFunction() != rhs_typed.isAggregateFunction() || isOrdinaryFunction() != rhs_typed.isOrdinaryFunction() || isWindowFunction() != rhs_typed.isWindowFunction() - || nulls_action != rhs_typed.nulls_action) + || nulls_action != rhs_typed.nulls_action || settings_changes != rhs_typed.settings_changes) return false; /// is_operator is ignored here because it affects only AST formatting @@ -203,6 +211,17 @@ void FunctionNode::updateTreeHashImpl(HashState & hash_state, CompareOptions /*c hash_state.update(isWindowFunction()); hash_state.update(nulls_action); + hash_state.update(settings_changes.size()); + for (const auto & change : settings_changes) + { + hash_state.update(change.name.size()); + hash_state.update(change.name); + + const auto & value_dump = change.value.dump(); + hash_state.update(value_dump.size()); + hash_state.update(value_dump); + } + /// is_operator is ignored here because it affects only AST formatting if (!isResolved()) @@ -224,6 +243,7 @@ QueryTreeNodePtr FunctionNode::cloneImpl() const result_function->nulls_action = nulls_action; result_function->wrap_with_nullable = wrap_with_nullable; result_function->is_operator = is_operator; + result_function->settings_changes = settings_changes; return result_function; } @@ -287,6 +307,14 @@ ASTPtr FunctionNode::toASTImpl(const ConvertToASTOptions & options) const function_ast->window_definition = window_node->toAST(new_options); } + if (!settings_changes.empty()) + { + auto settings_ast = make_intrusive(); + settings_ast->changes = settings_changes; + settings_ast->is_standalone = false; + function_ast->arguments->children.push_back(settings_ast); + } + return function_ast; } diff --git a/src/Analyzer/FunctionNode.h b/src/Analyzer/FunctionNode.h index 396991a9d860..d6ae27404e97 100644 --- a/src/Analyzer/FunctionNode.h +++ b/src/Analyzer/FunctionNode.h @@ -9,6 +9,11 @@ #include #include #include +<<<<<<< HEAD +======= +#include +#include +>>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) namespace DB { @@ -203,6 +208,18 @@ class FunctionNode final : public IQueryTreeNode wrap_with_nullable = true; } + /// Get settings changes passed to table function + const SettingsChanges & getSettingsChanges() const + { + return settings_changes; + } + + /// Set settings changes passed as last argument to table function + void setSettingsChanges(SettingsChanges settings_changes_) + { + settings_changes = std::move(settings_changes_); + } + void dumpTreeImpl(WriteBuffer & buffer, FormatState & format_state, size_t indent) const override; protected: @@ -227,6 +244,8 @@ class FunctionNode final : public IQueryTreeNode static constexpr size_t arguments_child_index = 1; static constexpr size_t window_child_index = 2; static constexpr size_t children_size = window_child_index + 1; + + SettingsChanges settings_changes; }; } diff --git a/src/Analyzer/FunctionSecretArgumentsFinderTreeNode.h b/src/Analyzer/FunctionSecretArgumentsFinderTreeNode.h index 8bcb6e147420..e4f63192c95b 100644 --- a/src/Analyzer/FunctionSecretArgumentsFinderTreeNode.h +++ b/src/Analyzer/FunctionSecretArgumentsFinderTreeNode.h @@ -71,8 +71,14 @@ class FunctionTreeNodeImpl : public AbstractFunction { public: explicit ArgumentsTreeNode(const QueryTreeNodes * arguments_) : arguments(arguments_) {} - size_t size() const override { return arguments ? arguments->size() : 0; } - std::unique_ptr at(size_t n) const override { return std::make_unique(arguments->at(n).get()); } + size_t size() const override + { /// size withous skipped indexes + return arguments ? arguments->size() - skippedSize() : 0; + } + std::unique_ptr at(size_t n) const override + { /// n is relative index, some can be skipped + return std::make_unique(arguments->at(getRealIndex(n)).get()); + } private: const QueryTreeNodes * arguments = nullptr; }; diff --git a/src/Analyzer/QueryTreeBuilder.cpp b/src/Analyzer/QueryTreeBuilder.cpp index 1f4101d083bf..cfb28c56e041 100644 --- a/src/Analyzer/QueryTreeBuilder.cpp +++ b/src/Analyzer/QueryTreeBuilder.cpp @@ -765,7 +765,12 @@ QueryTreeNodePtr QueryTreeBuilder::buildExpression(const ASTPtr & expression, co { const auto & function_arguments_list = function->arguments->as()->children; for (const auto & argument : function_arguments_list) - function_node->getArguments().getNodes().push_back(buildExpression(argument, context)); + { + if (const auto * ast_set = argument->as()) + function_node->setSettingsChanges(ast_set->changes); + else + function_node->getArguments().getNodes().push_back(buildExpression(argument, context)); + } } if (function->isWindowFunction()) diff --git a/src/Analyzer/Resolve/QueryAnalyzer.cpp b/src/Analyzer/Resolve/QueryAnalyzer.cpp index 353e6d381388..ec08c7bbfae0 100644 --- a/src/Analyzer/Resolve/QueryAnalyzer.cpp +++ b/src/Analyzer/Resolve/QueryAnalyzer.cpp @@ -4511,6 +4511,7 @@ void QueryAnalyzer::resolveTableFunction(QueryTreeNodePtr & table_function_node, { auto table_function_node_to_resolve_typed = std::make_shared(table_function_argument_function_name); table_function_node_to_resolve_typed->getArgumentsNode() = table_function_argument_function->getArgumentsNode(); + table_function_node_to_resolve_typed->setSettingsChanges(table_function_argument_function->getSettingsChanges()); QueryTreeNodePtr table_function_node_to_resolve = std::move(table_function_node_to_resolve_typed); if (table_function_argument_function_name == "view" diff --git a/src/Common/ErrorCodes.cpp b/src/Common/ErrorCodes.cpp index d6e809c6ed33..216571ab8f26 100644 --- a/src/Common/ErrorCodes.cpp +++ b/src/Common/ErrorCodes.cpp @@ -650,6 +650,7 @@ M(768, CANNOT_EXECUTE_PROMQL_QUERY) \ M(769, NAMED_COLLECTION_IS_USED) \ M(770, WASM_ERROR) \ +<<<<<<< HEAD M(771, CACHE_CANNOT_WRITE_TO_CACHE_DISK) \ M(772, INCOMPATIBLE_SCHEMA) \ M(773, MALFORMED_AI_PROVIDER_RESPONSE) \ @@ -658,6 +659,9 @@ M(776, RESOURCE_LIMIT_EXCEEDED) \ M(777, MEMORY_RESERVATION_KILLED) \ M(778, MEMORY_RESERVATION_FAILED) \ +======= + M(771, CATALOG_NAMESPACE_DISABLED) \ +>>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) \ M(900, DISTRIBUTED_CACHE_ERROR) \ M(901, CANNOT_USE_DISTRIBUTED_CACHE) \ diff --git a/src/Common/ProfileEvents.cpp b/src/Common/ProfileEvents.cpp index 8f3745c38762..bcc334491632 100644 --- a/src/Common/ProfileEvents.cpp +++ b/src/Common/ProfileEvents.cpp @@ -422,6 +422,11 @@ M(IcebergTrivialCountOptimizationApplied, "Trivial count optimization applied while reading from Iceberg", ValueType::Number) \ M(IcebergVersionHintUsed, "Number of times version-hint.text has been used.", ValueType::Number) \ M(IcebergMinMaxIndexPrunedFiles, "Number of skipped files by using MinMax index in Iceberg", ValueType::Number) \ + M(IcebergAvroFileParsing, "Number of times avro metadata files have been parsed.", ValueType::Number) \ + M(IcebergAvroFileParsingMicroseconds, "Time spent for parsing avro metadata files for Iceberg tables.", ValueType::Microseconds) \ + M(IcebergJsonFileParsing, "Number of times json metadata files have been parsed.", ValueType::Number) \ + M(IcebergJsonFileParsingMicroseconds, "Time spent for parsing json metadata files for Iceberg tables.", ValueType::Microseconds) \ + \ M(JoinBuildTableRowCount, "Total number of rows in the build table for a JOIN operation.", ValueType::Number) \ M(JoinProbeTableRowCount, "Total number of rows in the probe table for a JOIN operation.", ValueType::Number) \ M(JoinResultRowCount, "Total number of rows in the result of a JOIN operation.", ValueType::Number) \ @@ -771,8 +776,10 @@ The server successfully detected this situation and will download merged part fr M(S3DeleteObjects, "Number of S3 API DeleteObject(s) calls.", ValueType::Number) \ M(S3CopyObject, "Number of S3 API CopyObject calls.", ValueType::Number) \ M(S3ListObjects, "Number of S3 API ListObjects calls.", ValueType::Number) \ + M(S3ListObjectsMicroseconds, "Time of S3 API ListObjects execution.", ValueType::Microseconds) \ M(S3HeadObject, "Number of S3 API HeadObject calls.", ValueType::Number) \ M(S3GetObjectTagging, "Number of S3 API GetObjectTagging calls.", ValueType::Number) \ + M(S3HeadObjectMicroseconds, "Time of S3 API HeadObject execution.", ValueType::Microseconds) \ M(S3CreateMultipartUpload, "Number of S3 API CreateMultipartUpload calls.", ValueType::Number) \ M(S3UploadPartCopy, "Number of S3 API UploadPartCopy calls.", ValueType::Number) \ M(S3UploadPart, "Number of S3 API UploadPart calls.", ValueType::Number) \ @@ -828,6 +835,7 @@ The server successfully detected this situation and will download merged part fr M(AzureCopyObject, "Number of Azure blob storage API CopyObject calls", ValueType::Number) \ M(AzureDeleteObjects, "Number of Azure blob storage API DeleteObject(s) calls.", ValueType::Number) \ M(AzureListObjects, "Number of Azure blob storage API ListObjects calls.", ValueType::Number) \ + M(AzureListObjectsMicroseconds, "Time of Azure blob storage API ListObjects execution.", ValueType::Microseconds) \ M(AzureGetProperties, "Number of Azure blob storage API GetProperties calls.", ValueType::Number) \ M(AzureCreateContainer, "Number of Azure blob storage API CreateContainer calls.", ValueType::Number) \ \ diff --git a/src/Core/Settings.cpp b/src/Core/Settings.cpp index 6ce6575d8d5c..3266e1881d90 100644 --- a/src/Core/Settings.cpp +++ b/src/Core/Settings.cpp @@ -7986,6 +7986,7 @@ Use `allow_nullable_tuple_in_extracted_subcolumns` to control whether extracted )", BETA, enable_nullable_tuple_type) \ DECLARE(UInt64, archive_adaptive_buffer_max_size_bytes, 8 * DBMS_DEFAULT_BUFFER_SIZE, R"( Limits the maximum size of the adaptive buffer used when writing to archive files (for example, tar archives)", 0) \ +<<<<<<< HEAD DECLARE(UInt64, shared_merge_tree_sequential_consistency_initial_parts_update_backoff_ms, 50, R"( Initial backoff in milliseconds for parts update when using `select_sequential_consistency` with `SharedMergeTree`. Only available in ClickHouse Cloud. )", 0) \ @@ -8014,6 +8015,26 @@ Enable converting the hash table to a flat array for joins when the key is a sin )", 0) \ DECLARE(UInt64, query_plan_min_columns_for_join_lazy_indexing, 3, R"( Control the minimum number of payload columns from the left side required for enabling lazy indexing optimization in JOIN. 0 means the optimization is disabled. +======= + DECLARE(Timezone, iceberg_timezone_for_timestamptz, "UTC", R"( +Timezone for Iceberg timestamptz field. + +Possible values: + +- Any valid timezone, e.g. `Europe/Berlin`, `UTC` or `Zulu` +- `` (empty value) - use session timezone + +Default value is `UTC`. +)", 0) \ + DECLARE(Timezone, iceberg_partition_timezone, "", R"( +Time zone by which partitioning of Iceberg tables was performed. +Possible values: + +- Any valid timezone, e.g. `Europe/Berlin`, `UTC` or `Zulu` +- `` (empty value) - use server or session timezone + +Default value is empty. +>>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) )", 0) \ DECLARE(Bool, export_merge_tree_part_overwrite_file_if_exists, false, R"( Overwrite file if it already exists when exporting a merge tree part @@ -8247,6 +8268,15 @@ Source SQL dialect for the polyglot transpiler (e.g. 'sqlite', 'mysql', 'postgre )", EXPERIMENTAL) \ DECLARE(Bool, enable_adaptive_memory_spill_scheduler, false, R"( Trigger processor to spill data into external storage adpatively. grace join is supported at present. +)", EXPERIMENTAL) \ + DECLARE(String, object_storage_cluster, "", R"( +Cluster to make distributed requests to object storages with alternative syntax. +)", EXPERIMENTAL) \ + DECLARE(UInt64, object_storage_max_nodes, 0, R"( +Limit for hosts used for request in object storage cluster table functions - azureBlobStorageCluster, s3Cluster, hdfsCluster, etc. +Possible values: +- Positive integer. +- 0 — All hosts in cluster. )", EXPERIMENTAL) \ DECLARE(Bool, allow_experimental_delta_kernel_rs, true, R"( Allow experimental delta-kernel-rs implementation. @@ -8350,6 +8380,18 @@ When the hash join build side was converted to a FixedHashMap (see `enable_join_ DECLARE(Bool, rewrite_in_to_join, false, R"( Rewrite expressions like 'x IN subquery' to JOIN. This might be useful for optimizing the whole query with join reordering. )", EXPERIMENTAL) \ +<<<<<<< HEAD +======= + DECLARE(Bool, object_storage_remote_initiator, false, R"( +Execute request to object storage as remote on one of object_storage_cluster nodes. +)", EXPERIMENTAL) \ + DECLARE(String, object_storage_remote_initiator_cluster, "", R"( +Cluster to choose remote initiator, when `object_storage_remote_initiator` is true. When empty, `object_storage_cluster` is used. +)", EXPERIMENTAL) \ + DECLARE(Bool, allow_experimental_iceberg_read_optimization, true, R"( +Allow Iceberg read optimization based on Iceberg metadata. +)", EXPERIMENTAL) \ +>>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) \ /** Experimental timeSeries* aggregate functions. */ \ DECLARE_WITH_ALIAS(Bool, allow_experimental_time_series_aggregate_functions, false, R"( diff --git a/src/Core/SettingsChangesHistory.cpp b/src/Core/SettingsChangesHistory.cpp index 4309c206d199..e0fc89469657 100644 --- a/src/Core/SettingsChangesHistory.cpp +++ b/src/Core/SettingsChangesHistory.cpp @@ -267,12 +267,12 @@ const VersionToSettingsChangesMap & getSettingsChangesHistory() }); addSettingsChanges(settings_changes_history, "26.1.3.20001.altinityantalya", { - // {"iceberg_partition_timezone", "", "", "New setting."}, + {"iceberg_partition_timezone", "", "", "New setting."}, // {"s3_propagate_credentials_to_other_storages", false, false, "New setting"}, {"export_merge_tree_part_filename_pattern", "", "{part_name}_{checksum}", "New setting"}, // {"use_parquet_metadata_cache", false, true, "Enables cache of parquet file metadata."}, // {"input_format_parquet_use_metadata_cache", true, false, "Obsolete. No-op"}, // https://github.com/Altinity/ClickHouse/pull/586 - // {"object_storage_remote_initiator_cluster", "", "", "New setting."}, + {"object_storage_remote_initiator_cluster", "", "", "New setting."}, // {"iceberg_metadata_staleness_ms", 0, 0, "New setting allowing using cached metadata version at READ operations to prevent fetching from remote catalog"}, {"export_merge_tree_partition_task_timeout_seconds", 0, 3600, "New setting to control the timeout for export partition tasks."}, {"export_merge_tree_partition_manifest_ttl", 180, 86400, "Reasonable default for real usage"}, @@ -360,7 +360,6 @@ const VersionToSettingsChangesMap & getSettingsChangesHistory() {"insert_select_deduplicate", Field{"auto"}, Field{"auto"}, "New setting"}, {"output_format_pretty_named_tuples_as_json", false, true, "New setting to control whether named tuples in Pretty format are output as JSON objects"}, {"deduplicate_insert_select", "enable_even_for_bad_queries", "enable_even_for_bad_queries", "New setting, replace insert_select_deduplicate"}, - }); addSettingsChanges(settings_changes_history, "25.11", { @@ -459,12 +458,12 @@ const VersionToSettingsChangesMap & getSettingsChangesHistory() }); addSettingsChanges(settings_changes_history, "25.8.16.20001.altinityantalya", { - // {"allow_experimental_database_iceberg", false, true, "Turned ON by default for Antalya."}, - // {"allow_experimental_database_unity_catalog", false, true, "Turned ON by default for Antalya."}, - // {"allow_experimental_database_glue_catalog", false, true, "Turned ON by default for Antalya."}, - // {"allow_database_iceberg", false, true, "Turned ON by default for Antalya (alias)."}, - // {"allow_database_unity_catalog", false, true, "Turned ON by default for Antalya (alias)."}, - // {"allow_database_glue_catalog", false, true, "Turned ON by default for Antalya (alias)."}, + {"allow_experimental_database_iceberg", false, true, "Turned ON by default for Antalya."}, + {"allow_experimental_database_unity_catalog", false, true, "Turned ON by default for Antalya."}, + {"allow_experimental_database_glue_catalog", false, true, "Turned ON by default for Antalya."}, + {"allow_database_iceberg", false, true, "Turned ON by default for Antalya (alias)."}, + {"allow_database_unity_catalog", false, true, "Turned ON by default for Antalya (alias)."}, + {"allow_database_glue_catalog", false, true, "Turned ON by default for Antalya (alias)."}, // {"input_format_parquet_use_metadata_cache", true, true, "New setting, turned ON by default"}, // https://github.com/Altinity/ClickHouse/pull/586 // {"iceberg_timezone_for_timestamptz", "UTC", "UTC", "New setting."}, // {"object_storage_remote_initiator", false, false, "New setting."}, @@ -487,6 +486,12 @@ const VersionToSettingsChangesMap & getSettingsChangesHistory() // {"export_merge_tree_partition_system_table_prefer_remote_information", true, true, "New setting."}, // {"export_merge_tree_part_throw_on_pending_mutations", true, true, "New setting."}, // {"export_merge_tree_part_throw_on_pending_patch_parts", true, true, "New setting."}, + {"iceberg_timezone_for_timestamptz", "UTC", "UTC", "New setting."}, + {"object_storage_remote_initiator", false, false, "New setting."}, + {"allow_experimental_iceberg_read_optimization", true, true, "New setting."}, + // {"object_storage_cluster_join_mode", "allow", "allow", "New setting"}, + {"lock_object_storage_task_distribution_ms", 500, 500, "New setting."}, + {"allow_retries_in_cluster_requests", false, false, "New setting"}, {"allow_experimental_export_merge_tree_part", false, true, "Turned ON by default for Antalya."}, {"export_merge_tree_part_overwrite_file_if_exists", false, false, "New setting."}, {"export_merge_tree_partition_force_export", false, false, "New setting."}, @@ -499,8 +504,8 @@ const VersionToSettingsChangesMap & getSettingsChangesHistory() {"export_merge_tree_partition_system_table_prefer_remote_information", true, true, "New setting."}, {"export_merge_tree_part_throw_on_pending_mutations", true, true, "New setting."}, {"export_merge_tree_part_throw_on_pending_patch_parts", true, true, "New setting."}, - // {"object_storage_cluster", "", "", "Antalya: New setting"}, - // {"object_storage_max_nodes", 0, 0, "Antalya: New setting"}, + {"object_storage_cluster", "", "", "Antalya: New setting"}, + {"object_storage_max_nodes", 0, 0, "Antalya: New setting"}, {"use_object_storage_list_objects_cache", false, false, "New setting."}, }); addSettingsChanges(settings_changes_history, "25.8", diff --git a/src/Databases/DataLake/Common.cpp b/src/Databases/DataLake/Common.cpp index 681dd957b43f..8946d3412d70 100644 --- a/src/Databases/DataLake/Common.cpp +++ b/src/Databases/DataLake/Common.cpp @@ -61,14 +61,14 @@ std::vector splitTypeArguments(const String & type_str) return args; } -DB::DataTypePtr getType(const String & type_name, bool nullable, const String & prefix) +DB::DataTypePtr getType(const String & type_name, bool nullable, DB::ContextPtr context, const String & prefix) { String name = trim(type_name); if (name.starts_with("array<") && name.ends_with(">")) { String inner = name.substr(6, name.size() - 7); - return std::make_shared(getType(inner, nullable)); + return std::make_shared(getType(inner, nullable, context)); } if (name.starts_with("map<") && name.ends_with(">")) @@ -79,7 +79,7 @@ DB::DataTypePtr getType(const String & type_name, bool nullable, const String & if (args.size() != 2) throw DB::Exception(DB::ErrorCodes::DATALAKE_DATABASE_ERROR, "Invalid data type {}", type_name); - return std::make_shared(getType(args[0], false), getType(args[1], nullable)); + return std::make_shared(getType(args[0], false, context), getType(args[1], nullable, context)); } if (name.starts_with("struct<") && name.ends_with(">")) @@ -101,13 +101,13 @@ DB::DataTypePtr getType(const String & type_name, bool nullable, const String & String full_field_name = prefix.empty() ? field_name : prefix + "." + field_name; field_names.push_back(full_field_name); - field_types.push_back(getType(field_type, nullable, full_field_name)); + field_types.push_back(getType(field_type, nullable, context, full_field_name)); } return std::make_shared(field_types, field_names); } - return nullable ? DB::makeNullable(DB::Iceberg::IcebergSchemaProcessor::getSimpleType(name)) - : DB::Iceberg::IcebergSchemaProcessor::getSimpleType(name); + return nullable ? DB::makeNullable(DB::Iceberg::IcebergSchemaProcessor::getSimpleType(name, context)) + : DB::Iceberg::IcebergSchemaProcessor::getSimpleType(name, context); } std::pair parseTableName(const std::string & name) diff --git a/src/Databases/DataLake/Common.h b/src/Databases/DataLake/Common.h index cd4b6214e343..9b0dd7c626a6 100644 --- a/src/Databases/DataLake/Common.h +++ b/src/Databases/DataLake/Common.h @@ -2,6 +2,7 @@ #include #include +#include namespace DataLake { @@ -10,7 +11,7 @@ String trim(const String & str); std::vector splitTypeArguments(const String & type_str); -DB::DataTypePtr getType(const String & type_name, bool nullable, const String & prefix = ""); +DB::DataTypePtr getType(const String & type_name, bool nullable, DB::ContextPtr context, const String & prefix = ""); /// Parse a string, containing at least one dot, into a two substrings: /// A.B.C.D.E -> A.B.C.D and E, where diff --git a/src/Databases/DataLake/DataLakeConstants.h b/src/Databases/DataLake/DataLakeConstants.h index 0b228bf310ec..372cc92a6631 100644 --- a/src/Databases/DataLake/DataLakeConstants.h +++ b/src/Databases/DataLake/DataLakeConstants.h @@ -8,6 +8,7 @@ namespace DataLake { static constexpr auto DATABASE_ENGINE_NAME = "DataLakeCatalog"; +static constexpr auto DATABASE_ALIAS_NAME = "Iceberg"; static constexpr std::string_view FILE_PATH_PREFIX = "file:/"; /// Some catalogs (Unity or Glue) may store not only Iceberg/DeltaLake tables but other kinds of "tables" diff --git a/src/Databases/DataLake/DatabaseDataLake.cpp b/src/Databases/DataLake/DatabaseDataLake.cpp index 12fbb051ba4a..859bb0d9d021 100644 --- a/src/Databases/DataLake/DatabaseDataLake.cpp +++ b/src/Databases/DataLake/DatabaseDataLake.cpp @@ -63,6 +63,7 @@ namespace DatabaseDataLakeSetting extern const DatabaseDataLakeSettingsString oauth_server_uri; extern const DatabaseDataLakeSettingsBool oauth_server_use_request_body; extern const DatabaseDataLakeSettingsBool vended_credentials; + extern const DatabaseDataLakeSettingsString object_storage_cluster; extern const DatabaseDataLakeSettingsString aws_access_key_id; extern const DatabaseDataLakeSettingsString aws_secret_access_key; extern const DatabaseDataLakeSettingsString region; @@ -75,6 +76,7 @@ namespace DatabaseDataLakeSetting extern const DatabaseDataLakeSettingsBool onelake_use_blob_endpoint; extern const DatabaseDataLakeSettingsString dlf_access_key_id; extern const DatabaseDataLakeSettingsString dlf_access_key_secret; + extern const DatabaseDataLakeSettingsString namespaces; extern const DatabaseDataLakeSettingsString google_project_id; extern const DatabaseDataLakeSettingsString google_service_account; extern const DatabaseDataLakeSettingsString google_metadata_service; @@ -178,9 +180,14 @@ void DatabaseDataLake::initialize() const .aws_access_key_id = settings[DatabaseDataLakeSetting::aws_access_key_id].value, .aws_secret_access_key = settings[DatabaseDataLakeSetting::aws_secret_access_key].value, .region = settings[DatabaseDataLakeSetting::region].value, + .namespaces = settings[DatabaseDataLakeSetting::namespaces].value, .aws_role_arn = settings[DatabaseDataLakeSetting::aws_role_arn].value, +<<<<<<< HEAD .aws_role_session_name = settings[DatabaseDataLakeSetting::aws_role_session_name].value, .aws_external_id = settings[DatabaseDataLakeSetting::aws_external_id].value, +======= + .aws_role_session_name = settings[DatabaseDataLakeSetting::aws_role_session_name].value +>>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) }; switch (settings[DatabaseDataLakeSetting::catalog_type].value) @@ -195,6 +202,7 @@ void DatabaseDataLake::initialize() const settings[DatabaseDataLakeSetting::auth_header], settings[DatabaseDataLakeSetting::oauth_server_uri].value, settings[DatabaseDataLakeSetting::oauth_server_use_request_body].value, + settings[DatabaseDataLakeSetting::namespaces].value, Context::getGlobalContextInstance()); break; } @@ -209,6 +217,7 @@ void DatabaseDataLake::initialize() const settings[DatabaseDataLakeSetting::auth_scope].value, settings[DatabaseDataLakeSetting::oauth_server_uri].value, settings[DatabaseDataLakeSetting::oauth_server_use_request_body].value, + settings[DatabaseDataLakeSetting::namespaces].value, Context::getGlobalContextInstance()); break; } @@ -239,6 +248,7 @@ void DatabaseDataLake::initialize() const google_adc_client_secret, google_adc_refresh_token, google_adc_quota_project_id, + settings[DatabaseDataLakeSetting::namespaces].value, Context::getGlobalContextInstance()); break; } @@ -248,6 +258,7 @@ void DatabaseDataLake::initialize() const settings[DatabaseDataLakeSetting::warehouse].value, url, settings[DatabaseDataLakeSetting::catalog_credential].value, + settings[DatabaseDataLakeSetting::namespaces].value, Context::getGlobalContextInstance()); break; } @@ -322,7 +333,7 @@ std::shared_ptr DatabaseDataLake::getCatalog() const return catalog_impl; } -std::shared_ptr DatabaseDataLake::getConfiguration( +StorageObjectStorageConfigurationPtr DatabaseDataLake::getConfiguration( DatabaseDataLakeStorageType type, DataLakeStorageSettingsPtr storage_settings) const { @@ -356,24 +367,24 @@ std::shared_ptr DatabaseDataLake::getConfigur #if USE_AWS_S3 case DB::DatabaseDataLakeStorageType::S3: { - return std::make_shared(storage_settings); + return std::make_shared(storage_settings, settings[DatabaseDataLakeSetting::namespaces].value); } #endif #if USE_AZURE_BLOB_STORAGE case DB::DatabaseDataLakeStorageType::Azure: { - return std::make_shared(storage_settings); + return std::make_shared(storage_settings, settings[DatabaseDataLakeSetting::namespaces].value); } #endif #if USE_HDFS case DB::DatabaseDataLakeStorageType::HDFS: { - return std::make_shared(storage_settings); + return std::make_shared(storage_settings, settings[DatabaseDataLakeSetting::namespaces].value); } #endif case DB::DatabaseDataLakeStorageType::Local: { - return std::make_shared(storage_settings); + return std::make_shared(storage_settings, settings[DatabaseDataLakeSetting::namespaces].value); } /// Fake storage in case when catalog store not only /// primary-type tables (DeltaLake or Iceberg), but for @@ -385,7 +396,7 @@ std::shared_ptr DatabaseDataLake::getConfigur /// dependencies and the most lightweight case DB::DatabaseDataLakeStorageType::Other: { - return std::make_shared(storage_settings); + return std::make_shared(storage_settings, settings[DatabaseDataLakeSetting::namespaces].value); } #if !USE_AWS_S3 || !USE_AZURE_BLOB_STORAGE || !USE_HDFS default: @@ -402,7 +413,7 @@ std::shared_ptr DatabaseDataLake::getConfigur #if USE_AWS_S3 case DB::DatabaseDataLakeStorageType::S3: { - return std::make_shared(storage_settings); + return std::make_shared(storage_settings, settings[DatabaseDataLakeSetting::namespaces].value); } #endif #if USE_AZURE_BLOB_STORAGE @@ -413,7 +424,7 @@ std::shared_ptr DatabaseDataLake::getConfigur #endif case DB::DatabaseDataLakeStorageType::Local: { - return std::make_shared(storage_settings); + return std::make_shared(storage_settings, settings[DatabaseDataLakeSetting::namespaces].value); } /// Fake storage in case when catalog store not only /// primary-type tables (DeltaLake or Iceberg), but for @@ -425,7 +436,7 @@ std::shared_ptr DatabaseDataLake::getConfigur /// dependencies and the most lightweight case DB::DatabaseDataLakeStorageType::Other: { - return std::make_shared(storage_settings); + return std::make_shared(storage_settings, settings[DatabaseDataLakeSetting::namespaces].value); } default: throw Exception(ErrorCodes::BAD_ARGUMENTS, @@ -440,12 +451,12 @@ std::shared_ptr DatabaseDataLake::getConfigur #if USE_AWS_S3 case DB::DatabaseDataLakeStorageType::S3: { - return std::make_shared(storage_settings); + return std::make_shared(storage_settings, settings[DatabaseDataLakeSetting::namespaces].value); } #endif case DB::DatabaseDataLakeStorageType::Other: { - return std::make_shared(storage_settings); + return std::make_shared(storage_settings, settings[DatabaseDataLakeSetting::namespaces].value); } default: throw Exception(ErrorCodes::BAD_ARGUMENTS, @@ -544,7 +555,7 @@ StoragePtr DatabaseDataLake::tryGetTableImpl(const String & name, ContextPtr con auto [namespace_name, table_name] = DataLake::parseTableName(name); - if (!catalog->tryGetTableMetadata(namespace_name, table_name, table_metadata)) + if (!catalog->tryGetTableMetadata(namespace_name, table_name, context_, table_metadata)) return nullptr; if (ignore_if_not_iceberg && !table_metadata.isDefaultReadableTable()) return nullptr; @@ -680,7 +691,7 @@ StoragePtr DatabaseDataLake::tryGetTableImpl(const String & name, ContextPtr con /// with_table_structure = false: because there will be /// no table structure in table definition AST. - StorageObjectStorageConfiguration::initialize(*configuration, args, context_copy, /* with_table_structure */false); + configuration->initialize(args, context_copy, /* with_table_structure */false); const auto & query_settings = context_->getSettingsRef(); @@ -692,6 +703,7 @@ StoragePtr DatabaseDataLake::tryGetTableImpl(const String & name, ContextPtr con const auto is_secondary_query = context_->getClientInfo().query_kind == ClientInfo::QueryKind::SECONDARY_QUERY; +<<<<<<< HEAD const auto catalog_uuid = table_metadata.getTableUUID(); const UUID table_uuid = catalog_uuid ? parseFromString(*catalog_uuid) : UUIDHelpers::Nil; @@ -732,26 +744,49 @@ StoragePtr DatabaseDataLake::tryGetTableImpl(const String & name, ContextPtr con configuration->createObjectStorage(context_copy, /* is_readonly */ false, catalog->getCredentialsConfigurationCallback(StorageID(getDatabaseName(), name, table_uuid))), context_copy, StorageID(getDatabaseName(), name, table_uuid), +======= + std::string cluster_name = configuration->isClusterSupported() ? settings[DatabaseDataLakeSetting::object_storage_cluster].value : ""; + + if (cluster_name.empty() && can_use_parallel_replicas && !is_secondary_query) + cluster_name = parallel_replicas_cluster_name; + + auto storage_cluster = std::make_shared( + cluster_name, + configuration, + configuration->createObjectStorage(context_copy, /* is_readonly */ false, catalog->getCredentialsConfigurationCallback(StorageID(getDatabaseName(), name))), + StorageID(getDatabaseName(), name), +>>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) /* columns */columns, /* constraints */ConstraintsDescription{}, - /* comment */"", + /* partition_by */nullptr, + /* order_by */nullptr, + context_copy, + /* comment */ "", getFormatSettings(context_copy), LoadingStrictnessLevel::CREATE, getCatalog(), /* if_not_exists*/true, /* is_datalake_query*/true, +<<<<<<< HEAD distributed_processing, /* partition_by */nullptr, /* order_by */nullptr, +======= +>>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) /// Use is_table_function = true, /// because this table is actually stateless like a table function. /* is_table_function */true, /* lazy_init */true); +<<<<<<< HEAD if (context_->hasQueryContext() && context_->getSettingsRef()[Setting::log_queries]) context_->getQueryContext()->addQueryFactoriesInfo(Context::QueryLogFactories::Storage, result_storage->getName()); return result_storage; +======= + storage_cluster->startup(); + return storage_cluster; +>>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) } void DatabaseDataLake::dropTable( /// NOLINT @@ -930,7 +965,7 @@ void DatabaseDataLake::checkDatabase() const ASTPtr DatabaseDataLake::getCreateTableQueryImpl( const String & name, - ContextPtr /* context_ */, + ContextPtr context_, bool throw_on_error) const { auto catalog = getCatalog(); @@ -940,7 +975,7 @@ ASTPtr DatabaseDataLake::getCreateTableQueryImpl( const auto [namespace_name, table_name] = DataLake::parseTableName(name); - if (!catalog->tryGetTableMetadata(namespace_name, table_name, table_metadata)) + if (!catalog->tryGetTableMetadata(namespace_name, table_name, context_, table_metadata)) { if (throw_on_error) throw Exception(ErrorCodes::CANNOT_GET_CREATE_TABLE_QUERY, "Table `{}` doesn't exist", name); @@ -1055,6 +1090,11 @@ void registerDatabaseDataLake(DatabaseFactory & factory) throw Exception(ErrorCodes::BAD_ARGUMENTS, "Engine `{}` must have arguments", database_engine_name); } + if (database_engine_name == "Iceberg" && catalog_type != DatabaseDataLakeCatalogType::ICEBERG_REST) + { + throw Exception(ErrorCodes::BAD_ARGUMENTS, "Engine `Iceberg` must have `rest` catalog type only"); + } + for (auto & engine_arg : engine_args) engine_arg = evaluateConstantExpressionOrIdentifierAsLiteral(engine_arg, args.context); @@ -1151,6 +1191,7 @@ void registerDatabaseDataLake(DatabaseFactory & factory) args.uuid, /*lazy_init=*/args.create_query.attach); }; +<<<<<<< HEAD /// TODO: DataLakeCatalog is polymorphic — underlying source (S3, Azure, HDFS, etc.) depends /// on the catalog type chosen at runtime. Consider adding source_access_type once a mechanism /// for runtime-dependent or composite source checks exist. @@ -1239,6 +1280,10 @@ SELECT count() from database_name.table_name; )DOCS_MD", .syntax = "ENGINE = DataLakeCatalog('catalog_url'[, 'user', 'password']) SETTINGS catalog_type = '...'", .related = {}}); +======= + factory.registerDatabase("DataLakeCatalog", create_fn, { .supports_arguments = true, .supports_settings = true }); + factory.registerDatabase("Iceberg", create_fn, { .supports_arguments = true, .supports_settings = true }); +>>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) } } diff --git a/src/Databases/DataLake/DatabaseDataLake.h b/src/Databases/DataLake/DatabaseDataLake.h index 1b08632dc4ab..bd5fc6ac44f0 100644 --- a/src/Databases/DataLake/DatabaseDataLake.h +++ b/src/Databases/DataLake/DatabaseDataLake.h @@ -97,7 +97,7 @@ class DatabaseDataLake final : public IDatabase, WithContext /// front. Guarded by `catalog_mutex` because lazy initialization can race concurrent readers. void initialize() const TSA_REQUIRES(catalog_mutex); - std::shared_ptr getConfiguration( + StorageObjectStorageConfigurationPtr getConfiguration( DatabaseDataLakeStorageType type, DataLakeStorageSettingsPtr storage_settings) const; diff --git a/src/Databases/DataLake/DatabaseDataLakeSettings.cpp b/src/Databases/DataLake/DatabaseDataLakeSettings.cpp index 969b0769d13a..294178061b90 100644 --- a/src/Databases/DataLake/DatabaseDataLakeSettings.cpp +++ b/src/Databases/DataLake/DatabaseDataLakeSettings.cpp @@ -47,7 +47,12 @@ namespace ErrorCodes DECLARE(String, google_adc_credentials_file, "", "Deprecated setting, will throw an exception if used", 0) \ DECLARE(String, dlf_access_key_id, "", "Access id of DLF token for Paimon REST Catalog", 0) \ DECLARE(String, dlf_access_key_secret, "", "Access secret of DLF token for Paimon REST Catalog", 0) \ +<<<<<<< HEAD DECLARE(Bool, force_add_bucket, false, "Add bucket name to the metadata path", 0) \ +======= + DECLARE(String, namespaces, "*", "Comma-separated list of allowed namespaces", 0) \ + DECLARE(Bool, polaris_style_paths, true, "Enable Polaris/ADLS Gen2 path convention: the container name is prepended to the path in ABFSS locations (e.g. abfss://c@account/c/actual/path). When enabled, the redundant container prefix is stripped when building Azure HTTPS URLs. Disable if a real directory inside the container has the same name as the container itself.", 0) \ +>>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) #define LIST_OF_DATABASE_ICEBERG_SETTINGS(M, ALIAS) \ DATABASE_ICEBERG_RELATED_SETTINGS(M, ALIAS) \ diff --git a/src/Databases/DataLake/GlueCatalog.cpp b/src/Databases/DataLake/GlueCatalog.cpp index 53ac171c79ff..7cb4cb51bebf 100644 --- a/src/Databases/DataLake/GlueCatalog.cpp +++ b/src/Databases/DataLake/GlueCatalog.cpp @@ -58,12 +58,16 @@ namespace DB::ErrorCodes { extern const int BAD_ARGUMENTS; extern const int DATALAKE_DATABASE_ERROR; +<<<<<<< HEAD extern const int FAULT_INJECTED; } namespace DB::FailPoints { extern const char check_database_datalake_negative[]; +======= + extern const int CATALOG_NAMESPACE_DISABLED; +>>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) } namespace DB::Setting @@ -172,9 +176,9 @@ GlueCatalog::GlueCatalog( LOG_TRACE(log, "Creating AWS glue client with credentials empty {}, region '{}', endpoint '{}'", credentials.IsEmpty(), region, endpoint); } + boost::split(allowed_namespaces, settings.namespaces, boost::is_any_of(", "), boost::token_compress_on); credentials_provider = DB::S3::getCredentialsProvider(poco_config, credentials, creds_config); glue_client = std::make_unique(credentials_provider, endpoint_provider, client_configuration); - } GlueCatalog::~GlueCatalog() = default; @@ -200,8 +204,9 @@ DataLake::ICatalog::Namespaces GlueCatalog::getDatabases(const std::string & pre for (const auto & db : dbs) { const auto & db_name = db.GetName(); - if (!db_name.starts_with(prefix)) + if (!isNamespaceAllowed(db_name) || !db_name.starts_with(prefix)) continue; + result.push_back(db_name); if (limit != 0 && result.size() >= limit) break; @@ -286,6 +291,9 @@ DB::Names GlueCatalog::getTables() const bool GlueCatalog::existsTable(const std::string & database_name, const std::string & table_name) const { + if (!isNamespaceAllowed(database_name)) + throw DB::Exception(DB::ErrorCodes::CATALOG_NAMESPACE_DISABLED, "Namespace {} is filtered by `namespaces` database parameter", database_name); + Aws::Glue::Model::GetTableRequest request; request.SetDatabaseName(database_name); request.SetName(table_name); @@ -297,8 +305,12 @@ bool GlueCatalog::existsTable(const std::string & database_name, const std::stri bool GlueCatalog::tryGetTableMetadata( const std::string & database_name, const std::string & table_name, + DB::ContextPtr /* context_ */, TableMetadata & result) const { + if (!isNamespaceAllowed(database_name)) + throw DB::Exception(DB::ErrorCodes::CATALOG_NAMESPACE_DISABLED, "Namespace {} is filtered by `namespaces` database parameter", database_name); + Aws::Glue::Model::GetTableRequest request; request.SetDatabaseName(database_name); request.SetName(table_name); @@ -389,7 +401,7 @@ bool GlueCatalog::tryGetTableMetadata( column_type = getActualTimestampType(column.GetName(), result, column_type); } - schema.push_back({column.GetName(), getType(column_type, can_be_nullable)}); + schema.push_back({column.GetName(), getType(column_type, can_be_nullable, getContext())}); } result.setSchema(schema); } @@ -411,9 +423,10 @@ bool GlueCatalog::tryGetTableMetadata( void GlueCatalog::getTableMetadata( const std::string & database_name, const std::string & table_name, + DB::ContextPtr context_, TableMetadata & result) const { - if (!tryGetTableMetadata(database_name, table_name, result)) + if (!tryGetTableMetadata(database_name, table_name, context_, result)) { throw DB::Exception( DB::ErrorCodes::DATALAKE_DATABASE_ERROR, @@ -543,8 +556,8 @@ GlueCatalog::ObjectStorageWithPath GlueCatalog::createObjectStorageForEarlyTable auto storage_settings = std::make_shared(); storage_settings->loadFromSettingsChanges(settings.allChanged()); - auto configuration = std::make_shared(storage_settings); - DB::StorageObjectStorageConfiguration::initialize(*configuration, args, getContext(), false); + auto configuration = std::make_shared(storage_settings, settings.namespaces); + configuration->initialize(args, getContext(), false); auto object_storage = configuration->createObjectStorage(getContext(), true, {}); @@ -604,6 +617,11 @@ void GlueCatalog::createNamespaceIfNotExists(const String & namespace_name) cons void GlueCatalog::createTable(const String & namespace_name, const String & table_name, const String & new_metadata_path, Poco::JSON::Object::Ptr /*metadata_content*/) const { + if (!isNamespaceAllowed(namespace_name)) + throw DB::Exception(DB::ErrorCodes::CATALOG_NAMESPACE_DISABLED, + "Failed to create table {}, namespace {} is filtered by `namespaces` database parameter", + table_name, namespace_name); + createNamespaceIfNotExists(namespace_name); Aws::Glue::Model::CreateTableRequest request; @@ -686,6 +704,11 @@ bool GlueCatalog::updateSchema( void GlueCatalog::dropTable(const String & namespace_name, const String & table_name) const { + if (!isNamespaceAllowed(namespace_name)) + throw DB::Exception(DB::ErrorCodes::CATALOG_NAMESPACE_DISABLED, + "Failed to drop table {}, namespace {} is filtered by `namespaces` database parameter", + table_name, namespace_name); + Aws::Glue::Model::DeleteTableRequest request; request.SetDatabaseName(namespace_name); request.SetName(table_name); @@ -699,6 +722,11 @@ void GlueCatalog::dropTable(const String & namespace_name, const String & table_ response.GetError().GetMessage()); } +bool GlueCatalog::isNamespaceAllowed(const std::string & namespace_) const +{ + return allowed_namespaces.contains("*") || allowed_namespaces.contains(namespace_); +} + } #endif diff --git a/src/Databases/DataLake/GlueCatalog.h b/src/Databases/DataLake/GlueCatalog.h index 919b13a5669f..4b2a6f0d570c 100644 --- a/src/Databases/DataLake/GlueCatalog.h +++ b/src/Databases/DataLake/GlueCatalog.h @@ -46,11 +46,13 @@ class GlueCatalog final : public ICatalog, private DB::WithContext void getTableMetadata( const std::string & database_name, const std::string & table_name, + DB::ContextPtr context_, TableMetadata & result) const override; bool tryGetTableMetadata( const std::string & database_name, const std::string & table_name, + DB::ContextPtr context_, TableMetadata & result) const override; std::optional getStorageType() const override @@ -100,6 +102,9 @@ class GlueCatalog final : public ICatalog, private DB::WithContext std::string region; CatalogSettings settings; DB::ASTPtr table_engine_definition; + std::unordered_set allowed_namespaces; + + bool isNamespaceAllowed(const std::string & namespace_) const; DataLake::ICatalog::Namespaces getDatabases(const std::string & prefix, size_t limit = 0) const; DB::Names getTablesForDatabase(const std::string & db_name, size_t limit = 0) const; diff --git a/src/Databases/DataLake/HiveCatalog.cpp b/src/Databases/DataLake/HiveCatalog.cpp index 92d935aa6825..5170e45171df 100644 --- a/src/Databases/DataLake/HiveCatalog.cpp +++ b/src/Databases/DataLake/HiveCatalog.cpp @@ -183,13 +183,21 @@ bool HiveCatalog::existsTable(const std::string & namespace_name, const std::str return true; } -void HiveCatalog::getTableMetadata(const std::string & namespace_name, const std::string & table_name, TableMetadata & result) const +void HiveCatalog::getTableMetadata( + const std::string & namespace_name, + const std::string & table_name, + DB::ContextPtr context_, + TableMetadata & result) const { - if (!tryGetTableMetadata(namespace_name, table_name, result)) + if (!tryGetTableMetadata(namespace_name, table_name, context_, result)) throw DB::Exception(DB::ErrorCodes::DATALAKE_DATABASE_ERROR, "No response from iceberg catalog"); } -bool HiveCatalog::tryGetTableMetadata(const std::string & namespace_name, const std::string & table_name, TableMetadata & result) const +bool HiveCatalog::tryGetTableMetadata( + const std::string & namespace_name, + const std::string & table_name, + DB::ContextPtr context_, + TableMetadata & result) const { Apache::Hadoop::Hive::Table table; @@ -214,7 +222,7 @@ bool HiveCatalog::tryGetTableMetadata(const std::string & namespace_name, const auto columns = table.sd.cols; for (const auto & column : columns) { - schema.push_back({column.name, getType(column.type, true)}); + schema.push_back({column.name, getType(column.type, true, context_)}); } result.setSchema(schema); } diff --git a/src/Databases/DataLake/HiveCatalog.h b/src/Databases/DataLake/HiveCatalog.h index 4c6314aba505..d626b73c3871 100644 --- a/src/Databases/DataLake/HiveCatalog.h +++ b/src/Databases/DataLake/HiveCatalog.h @@ -38,9 +38,17 @@ class HiveCatalog final : public ICatalog, private DB::WithContext bool existsTable(const std::string & namespace_name, const std::string & table_name) const override; - void getTableMetadata(const std::string & namespace_name, const std::string & table_name, TableMetadata & result) const override; - - bool tryGetTableMetadata(const std::string & namespace_name, const std::string & table_name, TableMetadata & result) const override; + void getTableMetadata( + const std::string & namespace_name, + const std::string & table_name, + DB::ContextPtr context_, + TableMetadata & result) const override; + + bool tryGetTableMetadata( + const std::string & namespace_name, + const std::string & table_name, + DB::ContextPtr context_, + TableMetadata & result) const override; std::optional getStorageType() const override; diff --git a/src/Databases/DataLake/ICatalog.cpp b/src/Databases/DataLake/ICatalog.cpp index 432b4d8b61c5..d51660fb8793 100644 --- a/src/Databases/DataLake/ICatalog.cpp +++ b/src/Databases/DataLake/ICatalog.cpp @@ -103,19 +103,8 @@ void TableMetadata::setLocation(const std::string & location_) auto pos_to_path = location_.substr(pos_to_bucket).find('/'); if (pos_to_path == std::string::npos) - throw DB::Exception(DB::ErrorCodes::NOT_IMPLEMENTED, "Unexpected location format: {}", location_); - - pos_to_path = pos_to_bucket + pos_to_path; - - location_without_path = location_.substr(0, pos_to_path); - path = location_.substr(pos_to_path + 1); - - /// For Azure ABFSS format: abfss://container@account.dfs.core.windows.net/path - /// The bucket (container) is the part before '@', not the whole string before '/' - String bucket_part = location_.substr(pos_to_bucket, pos_to_path - pos_to_bucket); - auto at_pos = bucket_part.find('@'); - if (at_pos != std::string::npos) { +<<<<<<< HEAD /// Azure ABFSS format: extract container (before @) and account (after @) bucket = bucket_part.substr(0, at_pos); azure_account_with_suffix = bucket_part.substr(at_pos + 1); @@ -123,14 +112,55 @@ void TableMetadata::setLocation(const std::string & location_) LOG_TEST(getLogger("TableMetadata"), "Parsed Azure location - container: {}, account: {}, path: {}", bucket, azure_account_with_suffix, path); +======= + if (storage_type_str == "s3://") + { // empty path is allowed for AWS S3Table + location_without_path = location_; + path.clear(); + bucket = location_.substr(pos_to_bucket); + } + else + throw DB::Exception(DB::ErrorCodes::NOT_IMPLEMENTED, "Unexpected location format: {}", location_); +>>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) } else { - /// Standard format (S3, GCS, etc.) - bucket = bucket_part; - LOG_TEST(getLogger("TableMetadata"), - "Parsed location without path: {}, path: {}", - location_without_path, path); + pos_to_path = pos_to_bucket + pos_to_path; + + location_without_path = location_.substr(0, pos_to_path); + path = location_.substr(pos_to_path + 1); + + /// For Azure ABFSS format: abfss://container@account.dfs.core.windows.net/path + /// The bucket (container) is the part before '@', not the whole string before '/' + String bucket_part = location_.substr(pos_to_bucket, pos_to_path - pos_to_bucket); + auto at_pos = bucket_part.find('@'); + if (at_pos != std::string::npos) + { + /// Azure ABFSS format: extract container (before @) and account (after @) + bucket = bucket_part.substr(0, at_pos); + azure_account_with_suffix = bucket_part.substr(at_pos + 1); + + /// Some catalogs (e.g. Apache Polaris) follow the ADLS Gen2 filesystem convention + /// of including the container name as the first segment of the path in abfss:// locations, + /// e.g. abfss://container@account.dfs.core.windows.net/container/actual/path. + /// We record this as a flag so that `constructLocation` and `getMetadataLocation` can + /// strip the redundant prefix when needed, while `path` itself is left intact so that + /// `getLocation` remains a round-trip of `setLocation`. + if (polaris_style_abfss_paths && path.starts_with(bucket + "/")) + abfss_has_container_path_prefix = true; + + LOG_TEST(getLogger("TableMetadata"), + "Parsed Azure location - container: {}, account: {}, path: {}", + bucket, azure_account_with_suffix, path); + } + else + { + /// Standard format (S3, GCS, etc.) + bucket = bucket_part; + LOG_TEST(getLogger("TableMetadata"), + "Parsed location without path: {}, path: {}", + location_without_path, path); + } } } diff --git a/src/Databases/DataLake/ICatalog.h b/src/Databases/DataLake/ICatalog.h index e14b00ac3732..b60c114eca32 100644 --- a/src/Databases/DataLake/ICatalog.h +++ b/src/Databases/DataLake/ICatalog.h @@ -10,6 +10,14 @@ #include #include +namespace DB +{ + +class Context; +using ContextPtr = std::shared_ptr; + +} + namespace DataLake { @@ -128,6 +136,7 @@ struct CatalogSettings String aws_access_key_id; String aws_secret_access_key; String region; + String namespaces; String aws_role_arn; String aws_role_session_name; String aws_external_id; @@ -165,6 +174,7 @@ class ICatalog virtual void getTableMetadata( const std::string & namespace_name, const std::string & table_name, + DB::ContextPtr context, TableMetadata & result) const = 0; /// Get table metadata in the given namespace. @@ -172,6 +182,7 @@ class ICatalog virtual bool tryGetTableMetadata( const std::string & namespace_name, const std::string & table_name, + DB::ContextPtr context, TableMetadata & result) const = 0; /// Get storage type, where Iceberg tables' data is stored. diff --git a/src/Databases/DataLake/PaimonRestCatalog.cpp b/src/Databases/DataLake/PaimonRestCatalog.cpp index 7b7fa8d96056..a2d2dd04c080 100644 --- a/src/Databases/DataLake/PaimonRestCatalog.cpp +++ b/src/Databases/DataLake/PaimonRestCatalog.cpp @@ -467,7 +467,7 @@ bool PaimonRestCatalog::existsTable(const String & database_name, const String & return true; } -bool PaimonRestCatalog::tryGetTableMetadata(const String & database_name, const String & table_name, TableMetadata & result) const +bool PaimonRestCatalog::tryGetTableMetadata(const String & database_name, const String & table_name, DB::ContextPtr /*context_*/, TableMetadata & result) const { try { @@ -593,9 +593,9 @@ Poco::JSON::Object::Ptr PaimonRestCatalog::requestRest( return json.extract(); } -void PaimonRestCatalog::getTableMetadata(const String & database_name, const String & table_name, TableMetadata & result) const +void PaimonRestCatalog::getTableMetadata(const String & database_name, const String & table_name, DB::ContextPtr context_, TableMetadata & result) const { - if (!tryGetTableMetadata(database_name, table_name, result)) + if (!tryGetTableMetadata(database_name, table_name, context_, result)) { throw DB::Exception(DB::ErrorCodes::DATALAKE_DATABASE_ERROR, "No response from paimon rest catalog"); } diff --git a/src/Databases/DataLake/PaimonRestCatalog.h b/src/Databases/DataLake/PaimonRestCatalog.h index 78713832e288..c81722c63964 100644 --- a/src/Databases/DataLake/PaimonRestCatalog.h +++ b/src/Databases/DataLake/PaimonRestCatalog.h @@ -89,9 +89,9 @@ class PaimonRestCatalog final : public ICatalog, private DB::WithContext bool existsTable(const String & database_name, const String & table_name) const override; - void getTableMetadata(const String & database_name, const String & table_name, TableMetadata & result) const override; + void getTableMetadata(const String & database_name, const String & table_name, DB::ContextPtr context_, TableMetadata & result) const override; - bool tryGetTableMetadata(const String & database_name, const String & table_name, TableMetadata & result) const override; + bool tryGetTableMetadata(const String & database_name, const String & table_name, DB::ContextPtr /*context_*/, TableMetadata & result) const override; std::optional getStorageType() const override { return storage_type; } diff --git a/src/Databases/DataLake/RestCatalog.cpp b/src/Databases/DataLake/RestCatalog.cpp index 28c1195082e4..fc693e8c0bd6 100644 --- a/src/Databases/DataLake/RestCatalog.cpp +++ b/src/Databases/DataLake/RestCatalog.cpp @@ -55,6 +55,7 @@ namespace DB::ErrorCodes extern const int DATALAKE_DATABASE_ERROR; extern const int LOGICAL_ERROR; extern const int BAD_ARGUMENTS; +<<<<<<< HEAD extern const int FAULT_INJECTED; } @@ -66,6 +67,9 @@ namespace DB::Setting namespace DB::FailPoints { extern const char check_database_datalake_negative[]; +======= + extern const int CATALOG_NAMESPACE_DISABLED; +>>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) } namespace DataLake @@ -170,6 +174,7 @@ RestCatalog::RestCatalog( const std::string & auth_header_, const std::string & oauth_server_uri_, bool oauth_server_use_request_body_, + const std::string & namespaces_, DB::ContextPtr context_) : ICatalog(warehouse_) , DB::WithContext(context_) @@ -178,6 +183,7 @@ RestCatalog::RestCatalog( , auth_scope(auth_scope_) , oauth_server_uri(oauth_server_uri_) , oauth_server_use_request_body(oauth_server_use_request_body_) + , allowed_namespaces(namespaces_) { if (!catalog_credential_.empty()) { @@ -205,6 +211,7 @@ RestCatalog::RestCatalog( const std::string & auth_scope_, const std::string & oauth_server_uri_, bool oauth_server_use_request_body_, + const std::string & namespaces_, DB::ContextPtr context_) : ICatalog(warehouse_) , DB::WithContext(context_) @@ -213,6 +220,7 @@ RestCatalog::RestCatalog( , auth_scope(auth_scope_) , oauth_server_uri(oauth_server_uri_) , oauth_server_use_request_body(oauth_server_use_request_body_) + , allowed_namespaces(namespaces_) { } @@ -297,8 +305,9 @@ OneLakeCatalog::OneLakeCatalog( const std::string & auth_scope_, const std::string & oauth_server_uri_, bool oauth_server_use_request_body_, + const std::string & namespaces_, DB::ContextPtr context_) - : RestCatalog(warehouse_, base_url_, auth_scope_, oauth_server_uri_, oauth_server_use_request_body_, context_) + : RestCatalog(warehouse_, base_url_, auth_scope_, oauth_server_uri_, oauth_server_use_request_body_, namespaces_, context_) , tenant_id(onelake_tenant_id) { client_id = onelake_client_id; @@ -409,8 +418,9 @@ BigLakeCatalog::BigLakeCatalog( const std::string & google_adc_client_secret_, const std::string & google_adc_refresh_token_, const std::string & google_adc_quota_project_id_, + const std::string & namespaces_, DB::ContextPtr context_) - : RestCatalog(warehouse_, base_url_, "", "", false, context_) + : RestCatalog(warehouse_, base_url_, "", "", false, namespaces_, context_) , google_project_id(google_project_id_) , google_service_account(google_service_account_) , google_metadata_service(google_metadata_service_) @@ -640,9 +650,14 @@ bool RestCatalog::empty() const bool found_table = false; auto stop_condition = [&](const std::string & namespace_name) -> bool { +<<<<<<< HEAD if (found_table) return true; +======= + if (!allowed_namespaces.isNamespaceAllowed(namespace_name, /*nested*/ false)) + return false; +>>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) const auto tables = getTables(namespace_name, /* limit */1); if (!tables.empty()) found_table = true; @@ -668,6 +683,8 @@ DB::Names RestCatalog::getTables() const auto execute_for_each_namespace = [&](const std::string & current_namespace) { + if (!allowed_namespaces.isNamespaceAllowed(current_namespace, /*nested*/ false)) + return; runner.enqueueAndKeepTrack( [=, &tables, &mutex, this] { @@ -710,9 +727,21 @@ void RestCatalog::getNamespacesRecursive( break; if (func) - func(current_namespace); + { + if (allowed_namespaces.isNamespaceAllowed(current_namespace, /*nested*/ false)) + func(current_namespace); + else + { + LOG_DEBUG(log, "Tables in namespace {} are filtered", current_namespace); + } + } - getNamespacesRecursive(current_namespace, result, stop_condition, func); + if (allowed_namespaces.isNamespaceAllowed(current_namespace, /*nested*/ true)) + getNamespacesRecursive(current_namespace, result, stop_condition, func); + else + { + LOG_DEBUG(log, "Nested namespaces in namespace {} are filtered", current_namespace); + } } } @@ -887,6 +916,10 @@ RestCatalog::Namespaces RestCatalog::parseNamespaces(DB::ReadBuffer & buf, const DB::Names RestCatalog::getTables(const std::string & base_namespace, size_t limit) const { + if (!allowed_namespaces.isNamespaceAllowed(base_namespace, /*nested*/ false)) + throw DB::Exception(DB::ErrorCodes::CATALOG_NAMESPACE_DISABLED, + "Namespace {} is filtered by `namespaces` database parameter", base_namespace); + auto encoded_namespace = encodeNamespaceForURI(base_namespace); const std::string endpoint = std::filesystem::path(NAMESPACES_ENDPOINT) / encoded_namespace / "tables"; @@ -996,20 +1029,23 @@ DB::Names RestCatalog::parseTables(DB::ReadBuffer & buf, const std::string & bas bool RestCatalog::existsTable(const std::string & namespace_name, const std::string & table_name) const { TableMetadata table_metadata; - return tryGetTableMetadata(namespace_name, table_name, table_metadata); + return tryGetTableMetadata(namespace_name, table_name, getContext(), table_metadata); } bool RestCatalog::tryGetTableMetadata( const std::string & namespace_name, const std::string & table_name, + DB::ContextPtr context_, TableMetadata & result) const { try { - return getTableMetadataImpl(namespace_name, table_name, result); + return getTableMetadataImpl(namespace_name, table_name, context_, result); } catch (const DB::Exception & ex) { + if (ex.code() == DB::ErrorCodes::CATALOG_NAMESPACE_DISABLED) + throw; LOG_DEBUG(log, "tryGetTableMetadata response: {}", ex.what()); return false; } @@ -1018,19 +1054,25 @@ bool RestCatalog::tryGetTableMetadata( void RestCatalog::getTableMetadata( const std::string & namespace_name, const std::string & table_name, + DB::ContextPtr context_, TableMetadata & result) const { - if (!getTableMetadataImpl(namespace_name, table_name, result)) + if (!getTableMetadataImpl(namespace_name, table_name, context_, result)) throw DB::Exception(DB::ErrorCodes::DATALAKE_DATABASE_ERROR, "No response from iceberg catalog"); } bool RestCatalog::getTableMetadataImpl( const std::string & namespace_name, const std::string & table_name, + DB::ContextPtr context_, TableMetadata & result) const { LOG_DEBUG(log, "Checking table {} in namespace {}", table_name, namespace_name); + if (!allowed_namespaces.isNamespaceAllowed(namespace_name, /*nested*/ false)) + throw DB::Exception(DB::ErrorCodes::CATALOG_NAMESPACE_DISABLED, + "Namespace {} is filtered by `namespaces` database parameter", namespace_name); + DB::HTTPHeaderEntries headers; if (result.requiresCredentials()) { @@ -1086,10 +1128,16 @@ bool RestCatalog::getTableMetadataImpl( if (result.requiresSchema()) { +<<<<<<< HEAD const bool allow_geo_parser = getContext()->getSettingsRef()[DB::Setting::allow_experimental_geo_types_in_iceberg].value; auto schema_processor = DB::Iceberg::IcebergSchemaProcessor(allow_geo_parser); auto id = DB::IcebergMetadata::parseTableSchema(metadata_object, schema_processor, log); +======= + // int format_version = metadata_object->getValue("format-version"); + auto schema_processor = DB::Iceberg::IcebergSchemaProcessor(context_); + auto id = DB::IcebergMetadata::parseTableSchema(metadata_object, schema_processor, context_, log); +>>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) auto schema = schema_processor.getClickhouseTableSchemaById(id); result.setSchema(*schema); } @@ -1190,6 +1238,10 @@ void RestCatalog::createNamespaceIfNotExists(const String & namespace_name, cons void RestCatalog::createTable(const String & namespace_name, const String & table_name, const String & /*new_metadata_path*/, Poco::JSON::Object::Ptr metadata_content) const { + if (!allowed_namespaces.isNamespaceAllowed(namespace_name, /*nested*/ false)) + throw DB::Exception(DB::ErrorCodes::CATALOG_NAMESPACE_DISABLED, + "Failed to create table {}, namespace {} is filtered by `namespaces` database parameter", table_name, namespace_name); + createNamespaceIfNotExists(namespace_name, metadata_content->getValue("location")); const std::string endpoint = (base_url / config.prefix / NAMESPACES_ENDPOINT / encodeNamespaceForURI(namespace_name) / "tables").generic_string(); @@ -1361,6 +1413,11 @@ bool RestCatalog::updateSchema( void RestCatalog::dropTable(const String & namespace_name, const String & table_name) const { + if (!allowed_namespaces.isNamespaceAllowed(namespace_name, /*nested*/ false)) + throw DB::Exception(DB::ErrorCodes::CATALOG_NAMESPACE_DISABLED, + "Failed to drop table {}, namespace {} is filtered by `namespaces` database parameter", + table_name, namespace_name); + const std::string endpoint = fmt::format("{}/namespaces/{}/tables/{}?purgeRequested=False", base_url, namespace_name, table_name); Poco::JSON::Object::Ptr request_body = nullptr; @@ -1497,6 +1554,70 @@ ICatalog::CredentialsRefreshCallback RestCatalog::getCredentialsConfigurationCal }; } +/// "alpha,alpha.a1,bravo,bravo.*,charlie,delta.d1,echo.*" +/// allows tables from +/// - "alpha" namespace +/// - "alpha.a1" namespace +/// - "bravo" namespace +/// - any nested namespaces of "bravo" +/// - "charlie" namespace, but not from nested of "charlie" +/// - "delta.d1" namespace, but not from "delta" +/// - any nested namespaces of "echo", but not "echo" itself +/// "bravo.*.b2" makes no sense for now, asterisk allows all nested +RestCatalog::AllowedNamespaces::AllowedNamespaces(const std::string & namespaces_) +{ + std::vector list_of_namespaces; + boost::split(list_of_namespaces, namespaces_, boost::is_any_of(", "), boost::token_compress_on); + for (const auto & ns : list_of_namespaces) + { + std::vector list_of_nested_namespaces; + boost::split(list_of_nested_namespaces, ns, boost::is_any_of(".")); + + size_t len = list_of_nested_namespaces.size(); + if (!len) + continue; + + AllowedNamespaces * current = &(nested_namespaces[list_of_nested_namespaces[0]]); + for (size_t i = 1; i <= len; ++i) + { + if (i == len) + current->allow_tables = true; + else + { + current = &(current->nested_namespaces[list_of_nested_namespaces[i]]); + if (list_of_nested_namespaces[i] == "*") + { + current->allow_tables = true; + break; + } + } + } + } +} + +bool RestCatalog::AllowedNamespaces::isNamespaceAllowed(const std::string & namespace_, bool nested) const +{ + // Trivial case, check here to avoid split namespace on nested + if (nested_namespaces.contains("*")) + return true; + + std::vector list_of_nested_namespaces; + boost::split(list_of_nested_namespaces, namespace_, boost::is_any_of(".")); + + const AllowedNamespaces * current = this; + for (const auto & nns : list_of_nested_namespaces) + { + if (current->nested_namespaces.contains("*")) + return true; + auto it = current->nested_namespaces.find(nns); + if (it == current->nested_namespaces.end()) + return false; + current = &(it->second); + } + + return nested ? !current->nested_namespaces.empty() : current->allow_tables; +} + } #endif diff --git a/src/Databases/DataLake/RestCatalog.h b/src/Databases/DataLake/RestCatalog.h index 982475ee2c96..d31cb2d5de41 100644 --- a/src/Databases/DataLake/RestCatalog.h +++ b/src/Databases/DataLake/RestCatalog.h @@ -43,6 +43,7 @@ class RestCatalog : public ICatalog, public DB::WithContext const std::string & auth_header_, const std::string & oauth_server_uri_, bool oauth_server_use_request_body_, + const std::string & namespaces_, DB::ContextPtr context_); ~RestCatalog() override = default; @@ -56,11 +57,13 @@ class RestCatalog : public ICatalog, public DB::WithContext void getTableMetadata( const std::string & namespace_name, const std::string & table_name, + DB::ContextPtr context_, TableMetadata & result) const override; bool tryGetTableMetadata( const std::string & namespace_name, const std::string & table_name, + DB::ContextPtr context_, TableMetadata & result) const override; std::optional getStorageType() const override; @@ -97,6 +100,7 @@ class RestCatalog : public ICatalog, public DB::WithContext const std::string & auth_scope_, const std::string & oauth_server_uri_, bool oauth_server_use_request_body_, + const std::string & namespaces_, DB::ContextPtr context_); void createNamespaceIfNotExists(const String & namespace_name, const String & location) const; @@ -131,6 +135,26 @@ class RestCatalog : public ICatalog, public DB::WithContext bool oauth_server_use_request_body; mutable MultiVersion access_token; +public: + class AllowedNamespaces + { + public: + AllowedNamespaces() {} + explicit AllowedNamespaces(const std::string & namespaces_); + + /// Check if nested namespaces (nested=true) or tables (nested=false) are allowed in namespace + bool isNamespaceAllowed(const std::string & namespace_, bool nested) const; + + private: + /// List of allowed nested namespaces + std::unordered_map nested_namespaces; + /// Tables from current level are allowed + bool allow_tables = false; + }; + +protected: + AllowedNamespaces allowed_namespaces; + Poco::Net::HTTPBasicCredentials credentials{}; DB::ReadWriteBufferFromHTTPPtr createReadBuffer( @@ -160,6 +184,7 @@ class RestCatalog : public ICatalog, public DB::WithContext bool getTableMetadataImpl( const std::string & namespace_name, const std::string & table_name, + DB::ContextPtr context_, TableMetadata & result) const; Config loadConfig(); @@ -189,6 +214,7 @@ class OneLakeCatalog : public RestCatalog const std::string & auth_scope_, const std::string & oauth_server_uri_, bool oauth_server_use_request_body_, + const std::string & namespaces_, DB::ContextPtr context_); DB::DatabaseDataLakeCatalogType getCatalogType() const override @@ -216,6 +242,7 @@ class BigLakeCatalog : public RestCatalog const std::string & google_adc_client_secret_, const std::string & google_adc_refresh_token_, const std::string & google_adc_quota_project_id_, + const std::string & namespaces_, DB::ContextPtr context_); DB::DatabaseDataLakeCatalogType getCatalogType() const override diff --git a/src/Databases/DataLake/UnityCatalog.cpp b/src/Databases/DataLake/UnityCatalog.cpp index 414b7e439ecf..4fc035b85ad8 100644 --- a/src/Databases/DataLake/UnityCatalog.cpp +++ b/src/Databases/DataLake/UnityCatalog.cpp @@ -19,7 +19,11 @@ namespace DB::ErrorCodes { extern const int DATALAKE_DATABASE_ERROR; extern const int LOGICAL_ERROR; +<<<<<<< HEAD extern const int BAD_ARGUMENTS; +======= + extern const int CATALOG_NAMESPACE_DISABLED; +>>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) } namespace @@ -93,9 +97,10 @@ DB::Names UnityCatalog::getTables() const void UnityCatalog::getTableMetadata( const std::string & namespace_name, const std::string & table_name, + DB::ContextPtr context_, TableMetadata & result) const { - if (!tryGetTableMetadata(namespace_name, table_name, result)) + if (!tryGetTableMetadata(namespace_name, table_name, context_, result)) throw DB::Exception(DB::ErrorCodes::DATALAKE_DATABASE_ERROR, "No response from unity catalog"); } @@ -160,8 +165,12 @@ void UnityCatalog::getCredentials(const String & table_id, TableMetadata & metad bool UnityCatalog::tryGetTableMetadata( const std::string & schema_name, const std::string & table_name, + DB::ContextPtr /* context_ */, TableMetadata & result) const { + if (!isNamespaceAllowed(schema_name)) + throw DB::Exception(DB::ErrorCodes::CATALOG_NAMESPACE_DISABLED, "Namespace {} is filtered by `namespaces` database parameter", schema_name); + auto full_table_name = warehouse + "." + schema_name + "." + table_name; Poco::Dynamic::Var json; std::string json_str; @@ -284,6 +293,9 @@ bool UnityCatalog::tryGetTableMetadata( bool UnityCatalog::existsTable(const std::string & schema_name, const std::string & table_name) const { + if (!isNamespaceAllowed(schema_name)) + throw DB::Exception(DB::ErrorCodes::CATALOG_NAMESPACE_DISABLED, "Namespace {} is filtered by `namespaces` database parameter", schema_name); + String json_str; Poco::Dynamic::Var json; try @@ -393,7 +405,7 @@ DataLake::ICatalog::Namespaces UnityCatalog::getSchemas(const std::string & base chassert(schema_info->get("catalog_name").extract() == warehouse); UnityCatalogFullSchemaName schema_name = parseFullSchemaName(schema_info->get("full_name").extract()); - if (schema_name.schema_name.starts_with(base_prefix)) + if (isNamespaceAllowed(schema_name.schema_name) && schema_name.schema_name.starts_with(base_prefix)) schemas.push_back(schema_name.schema_name); if (limit && schemas.size() > limit) @@ -435,6 +447,7 @@ UnityCatalog::UnityCatalog( const std::string & catalog_, const std::string & base_url_, const std::string & catalog_credential_, + const std::string & namespaces_, DB::ContextPtr context_) : ICatalog(catalog_) , DB::WithContext(context_) @@ -442,6 +455,12 @@ UnityCatalog::UnityCatalog( , log(getLogger("UnityCatalog(" + catalog_ + ")")) , auth_header("Authorization", "Bearer " + catalog_credential_) { + boost::split(allowed_namespaces, namespaces_, boost::is_any_of(", "), boost::token_compress_on); +} + +bool UnityCatalog::isNamespaceAllowed(const std::string & namespace_) const +{ + return allowed_namespaces.contains("*") || allowed_namespaces.contains(namespace_); } /// getCredentialsConfigurationCallback method is supported only for S3 storage diff --git a/src/Databases/DataLake/UnityCatalog.h b/src/Databases/DataLake/UnityCatalog.h index d835ee221f78..caa66cf90044 100644 --- a/src/Databases/DataLake/UnityCatalog.h +++ b/src/Databases/DataLake/UnityCatalog.h @@ -22,6 +22,7 @@ class UnityCatalog final : public ICatalog, private DB::WithContext const std::string & catalog_, const std::string & base_url_, const std::string & catalog_credential_, + const std::string & namespaces_, DB::ContextPtr context_); ~UnityCatalog() override = default; @@ -35,11 +36,13 @@ class UnityCatalog final : public ICatalog, private DB::WithContext void getTableMetadata( const std::string & namespace_name, const std::string & table_name, + DB::ContextPtr context_, TableMetadata & result) const override; bool tryGetTableMetadata( const std::string & schema_name, const std::string & table_name, + DB::ContextPtr context_, TableMetadata & result) const override; std::optional getStorageType() const override { return std::nullopt; } @@ -60,6 +63,10 @@ class UnityCatalog final : public ICatalog, private DB::WithContext Poco::Net::HTTPBasicCredentials credentials{}; + std::unordered_set allowed_namespaces; + + bool isNamespaceAllowed(const std::string & namespace_) const; + DataLake::ICatalog::Namespaces getSchemas(const std::string & base_prefix, size_t limit = 0) const; DB::Names getTablesForSchema(const std::string & schema, size_t limit = 0) const; diff --git a/src/Databases/DataLake/tests/gtest_rest_catalog_allowed_namespaces.cpp b/src/Databases/DataLake/tests/gtest_rest_catalog_allowed_namespaces.cpp new file mode 100644 index 000000000000..7a1981511c3f --- /dev/null +++ b/src/Databases/DataLake/tests/gtest_rest_catalog_allowed_namespaces.cpp @@ -0,0 +1,69 @@ +#include +#include + + +TEST(TestRestCatalogAllowedNamespaces, TestAllAllowed) +{ + DataLake::RestCatalog::AllowedNamespaces namespaces("*"); + EXPECT_TRUE(namespaces.isNamespaceAllowed("foo", /*nested*/ true)); + EXPECT_TRUE(namespaces.isNamespaceAllowed("foo", /*nested*/ false)); + EXPECT_TRUE(namespaces.isNamespaceAllowed("foo.bar", /*nested*/ true)); + EXPECT_TRUE(namespaces.isNamespaceAllowed("foo.bar", /*nested*/ false)); +} + +TEST(TestRestCatalogAllowedNamespaces, TestAllBlocked) +{ + DataLake::RestCatalog::AllowedNamespaces namespaces(""); + EXPECT_FALSE(namespaces.isNamespaceAllowed("foo", /*nested*/ true)); + EXPECT_FALSE(namespaces.isNamespaceAllowed("foo", /*nested*/ false)); + EXPECT_FALSE(namespaces.isNamespaceAllowed("foo.bar", /*nested*/ true)); + EXPECT_FALSE(namespaces.isNamespaceAllowed("foo.bar", /*nested*/ false)); +} + +TEST(TestRestCatalogAllowedNamespaces, TestTableInNamespaceAllowed) +{ + DataLake::RestCatalog::AllowedNamespaces namespaces("foo"); + EXPECT_FALSE(namespaces.isNamespaceAllowed("foo", /*nested*/ true)); + EXPECT_TRUE(namespaces.isNamespaceAllowed("foo", /*nested*/ false)); + EXPECT_FALSE(namespaces.isNamespaceAllowed("foo.bar", /*nested*/ true)); + EXPECT_FALSE(namespaces.isNamespaceAllowed("foo.bar", /*nested*/ false)); + EXPECT_FALSE(namespaces.isNamespaceAllowed("biz", /*nested*/ true)); + EXPECT_FALSE(namespaces.isNamespaceAllowed("biz", /*nested*/ false)); +} + +TEST(TestRestCatalogAllowedNamespaces, TestSpecificNestedNamespaceAllowed) +{ + DataLake::RestCatalog::AllowedNamespaces namespaces("foo.bar"); + EXPECT_TRUE(namespaces.isNamespaceAllowed("foo", /*nested*/ true)); + EXPECT_FALSE(namespaces.isNamespaceAllowed("foo", /*nested*/ false)); + EXPECT_FALSE(namespaces.isNamespaceAllowed("foo.bar", /*nested*/ true)); + EXPECT_TRUE(namespaces.isNamespaceAllowed("foo.bar", /*nested*/ false)); + EXPECT_FALSE(namespaces.isNamespaceAllowed("bar", /*nested*/ true)); + EXPECT_FALSE(namespaces.isNamespaceAllowed("bar", /*nested*/ false)); + EXPECT_FALSE(namespaces.isNamespaceAllowed("biz", /*nested*/ true)); + EXPECT_FALSE(namespaces.isNamespaceAllowed("biz", /*nested*/ false)); + EXPECT_FALSE(namespaces.isNamespaceAllowed("foo.biz", /*nested*/ true)); + EXPECT_FALSE(namespaces.isNamespaceAllowed("foo.biz", /*nested*/ false)); +} + +TEST(TestRestCatalogAllowedNamespaces, TestNestedNamespacesAllowed) +{ + DataLake::RestCatalog::AllowedNamespaces namespaces("foo.*"); + EXPECT_TRUE(namespaces.isNamespaceAllowed("foo", /*nested*/ true)); + EXPECT_FALSE(namespaces.isNamespaceAllowed("foo", /*nested*/ false)); + EXPECT_TRUE(namespaces.isNamespaceAllowed("foo.bar", /*nested*/ true)); + EXPECT_TRUE(namespaces.isNamespaceAllowed("foo.bar", /*nested*/ false)); + EXPECT_FALSE(namespaces.isNamespaceAllowed("biz", /*nested*/ true)); + EXPECT_FALSE(namespaces.isNamespaceAllowed("biz", /*nested*/ false)); +} + +TEST(TestRestCatalogAllowedNamespaces, TestTablesAndNestedNamespacesAllowed) +{ + DataLake::RestCatalog::AllowedNamespaces namespaces("foo,foo.*"); + EXPECT_TRUE(namespaces.isNamespaceAllowed("foo", /*nested*/ true)); + EXPECT_TRUE(namespaces.isNamespaceAllowed("foo", /*nested*/ false)); + EXPECT_TRUE(namespaces.isNamespaceAllowed("foo.bar", /*nested*/ true)); + EXPECT_TRUE(namespaces.isNamespaceAllowed("foo.bar", /*nested*/ false)); + EXPECT_FALSE(namespaces.isNamespaceAllowed("biz", /*nested*/ true)); + EXPECT_FALSE(namespaces.isNamespaceAllowed("biz", /*nested*/ false)); +} diff --git a/src/Disks/DiskObjectStorage/ObjectStorages/AzureBlobStorage/AzureObjectStorage.cpp b/src/Disks/DiskObjectStorage/ObjectStorages/AzureBlobStorage/AzureObjectStorage.cpp index aac0ef6a094a..0526708271ec 100644 --- a/src/Disks/DiskObjectStorage/ObjectStorages/AzureBlobStorage/AzureObjectStorage.cpp +++ b/src/Disks/DiskObjectStorage/ObjectStorages/AzureBlobStorage/AzureObjectStorage.cpp @@ -26,6 +26,7 @@ #include #include #include +#include namespace CurrentMetrics @@ -38,6 +39,7 @@ namespace CurrentMetrics namespace ProfileEvents { extern const Event AzureListObjects; + extern const Event AzureListObjectsMicroseconds; extern const Event DiskAzureListObjects; extern const Event AzureDeleteObjects; extern const Event DiskAzureDeleteObjects; @@ -89,6 +91,7 @@ class AzureIteratorAsync final : public IObjectStorageIteratorAsync ProfileEvents::increment(ProfileEvents::AzureListObjects); if (client->IsClientForDisk()) ProfileEvents::increment(ProfileEvents::DiskAzureListObjects); + ProfileEventTimeIncrement watch(ProfileEvents::AzureListObjectsMicroseconds); chassert(batch.empty()); auto blob_list_response = client->ListBlobs(options); @@ -196,7 +199,15 @@ void AzureObjectStorage::listObjects(const std::string & path, RelativePathsWith else options.PageSizeHint = settings.get()->list_object_keys_size; - for (auto blob_list_response = client_ptr->ListBlobs(options); blob_list_response.HasPage(); blob_list_response.MoveToNextPage()) + AzureBlobStorage::ListBlobsPagedResponse blob_list_response; + + auto list_blobs = [&]()->void + { + ProfileEventTimeIncrement watch(ProfileEvents::AzureListObjectsMicroseconds); + blob_list_response = client_ptr->ListBlobs(options); + }; + + for (list_blobs(); blob_list_response.HasPage(); blob_list_response.MoveToNextPage()) { ProfileEvents::increment(ProfileEvents::AzureListObjects); if (client_ptr->IsClientForDisk()) diff --git a/src/Disks/DiskObjectStorage/ObjectStorages/S3/S3ObjectStorage.cpp b/src/Disks/DiskObjectStorage/ObjectStorages/S3/S3ObjectStorage.cpp index 34eb1eaebb9d..16dd3e27cda2 100644 --- a/src/Disks/DiskObjectStorage/ObjectStorages/S3/S3ObjectStorage.cpp +++ b/src/Disks/DiskObjectStorage/ObjectStorages/S3/S3ObjectStorage.cpp @@ -35,6 +35,7 @@ #include #include #include +#include #include #include @@ -42,6 +43,7 @@ namespace ProfileEvents { extern const Event S3ListObjects; + extern const Event S3ListObjectsMicroseconds; extern const Event DiskS3DeleteObjects; extern const Event DiskS3ListObjects; } @@ -168,7 +170,12 @@ class S3IteratorAsync final : public IObjectStorageIteratorAsync ProfileEvents::increment(ProfileEvents::S3ListObjects); ProfileEvents::increment(ProfileEvents::DiskS3ListObjects); - auto outcome = client->ListObjectsV2(*request); + Aws::S3::Model::ListObjectsV2Outcome outcome; + + { + ProfileEventTimeIncrement watch(ProfileEvents::S3ListObjectsMicroseconds); + outcome = client->ListObjectsV2(*request); + } /// Outcome failure will be handled on the caller side. if (outcome.IsSuccess()) @@ -363,8 +370,17 @@ void S3ObjectStorage::listObjects(const std::string & path, RelativePathsWithMet ProfileEvents::increment(ProfileEvents::S3ListObjects); ProfileEvents::increment(ProfileEvents::DiskS3ListObjects); +<<<<<<< HEAD outcome = client.get()->ListObjectsV2(request); throwIfError(outcome, "while listing objects in bucket '{}' with prefix '{}' on disk '{}'", uri.bucket, path, disk_name); +======= + { + ProfileEventTimeIncrement watch(ProfileEvents::S3ListObjectsMicroseconds); + outcome = client.get()->ListObjectsV2(request); + } + + throwIfError(outcome); +>>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) auto result = outcome.GetResult(); auto objects = result.GetContents(); diff --git a/src/Disks/DiskType.cpp b/src/Disks/DiskType.cpp index bf4506b4cbf6..ddc42cd07dc3 100644 --- a/src/Disks/DiskType.cpp +++ b/src/Disks/DiskType.cpp @@ -10,7 +10,7 @@ namespace ErrorCodes extern const int LOGICAL_ERROR; } -MetadataStorageType metadataTypeFromString(const String & type) +MetadataStorageType metadataTypeFromString(const std::string & type) { auto check_type = Poco::toLower(type); if (check_type == "local") @@ -58,25 +58,7 @@ String DataSourceDescription::name() const case DataSourceType::RAM: return "memory"; case DataSourceType::ObjectStorage: - { - switch (object_storage_type) - { - case ObjectStorageType::S3: - return "s3"; - case ObjectStorageType::HDFS: - return "hdfs"; - case ObjectStorageType::Azure: - return "azure_blob_storage"; - case ObjectStorageType::Local: - return "local_blob_storage"; - case ObjectStorageType::Web: - return "web"; - case ObjectStorageType::None: - return "none"; - case ObjectStorageType::Max: - throw Exception(ErrorCodes::LOGICAL_ERROR, "Unexpected object storage type: Max"); - } - } + return DB::toString(object_storage_type); } } @@ -86,4 +68,45 @@ String DataSourceDescription::toString() const name(), description, is_encrypted, is_cached, zookeeper_name); } +ObjectStorageType objectStorageTypeFromString(const std::string & type) +{ + auto check_type = Poco::toLower(type); + if (check_type == "s3") + return ObjectStorageType::S3; + if (check_type == "hdfs") + return ObjectStorageType::HDFS; + if (check_type == "azure_blob_storage" || check_type == "azure") + return ObjectStorageType::Azure; + if (check_type == "local_blob_storage" || check_type == "local") + return ObjectStorageType::Local; + if (check_type == "web") + return ObjectStorageType::Web; + if (check_type == "none") + return ObjectStorageType::None; + + throw Exception(ErrorCodes::UNKNOWN_ELEMENT_IN_CONFIG, + "Unknown object storage type: {}", type); +} + +std::string toString(ObjectStorageType type) +{ + switch (type) + { + case ObjectStorageType::S3: + return "s3"; + case ObjectStorageType::HDFS: + return "hdfs"; + case ObjectStorageType::Azure: + return "azure_blob_storage"; + case ObjectStorageType::Local: + return "local_blob_storage"; + case ObjectStorageType::Web: + return "web"; + case ObjectStorageType::None: + return "none"; + case ObjectStorageType::Max: + throw Exception(ErrorCodes::LOGICAL_ERROR, "Unexpected object storage type: Max"); + } +} + } diff --git a/src/Disks/DiskType.h b/src/Disks/DiskType.h index 726557d5575d..53b2b28213ba 100644 --- a/src/Disks/DiskType.h +++ b/src/Disks/DiskType.h @@ -36,7 +36,10 @@ enum class MetadataStorageType : uint8_t Memory, }; -MetadataStorageType metadataTypeFromString(const String & type); +MetadataStorageType metadataTypeFromString(const std::string & type); + +ObjectStorageType objectStorageTypeFromString(const std::string & type); +std::string toString(ObjectStorageType type); struct DataSourceDescription { diff --git a/src/IO/ReadBufferFromS3.cpp b/src/IO/ReadBufferFromS3.cpp index f2d184fd8e71..67c92a0b2e94 100644 --- a/src/IO/ReadBufferFromS3.cpp +++ b/src/IO/ReadBufferFromS3.cpp @@ -566,6 +566,12 @@ Aws::S3::Model::GetObjectResult ReadBufferFromS3::sendRequest(size_t attempt, si log, "Read S3 object. Bucket: {}, Key: {}, Version: {}, Offset: {}", bucket, key, version_id.empty() ? "Latest" : version_id, range_begin); } + else + { + LOG_TEST( + log, "Read S3 object. Bucket: {}, Key: {}, Version: {}", + bucket, key, version_id.empty() ? "Latest" : version_id); + } ProfileEvents::increment(ProfileEvents::S3GetObject); if (client_ptr->isClientForDisk()) diff --git a/src/IO/S3/Client.cpp b/src/IO/S3/Client.cpp index af6f0a2e6894..01a3df1c99c0 100644 --- a/src/IO/S3/Client.cpp +++ b/src/IO/S3/Client.cpp @@ -463,7 +463,7 @@ Model::HeadObjectOutcome Client::HeadObject(HeadObjectRequest & request) const auto bucket_uri = getURIForBucket(bucket); if (!bucket_uri) { - if (auto maybe_error = updateURIForBucketForHead(bucket); maybe_error.has_value()) + if (auto maybe_error = updateURIForBucketForHead(bucket, request.GetKey()); maybe_error.has_value()) return *maybe_error; if (auto region = getRegionForBucket(bucket); !region.empty()) @@ -680,7 +680,6 @@ Client::doRequest(RequestType & request, RequestFn request_fn) const if (auto uri = getURIForBucket(bucket); uri.has_value()) request.overrideURI(std::move(*uri)); - bool found_new_endpoint = false; // if we found correct endpoint after 301 responses, update the cache for future requests SCOPE_EXIT( @@ -1049,12 +1048,15 @@ std::optional Client::getURIFromError(const Aws::S3::S3Error & error) c } // Do a list request because head requests don't have body in response -std::optional Client::updateURIForBucketForHead(const std::string & bucket) const +// S3 Tables don't support ListObjects, so made dirty workaroung - changed on GetObject +std::optional Client::updateURIForBucketForHead(const std::string & bucket, const std::string & key) const { - ListObjectsV2Request req; + GetObjectRequest req; req.SetBucket(bucket); - req.SetMaxKeys(1); - auto result = ListObjectsV2(req); + req.SetKey(key); + req.SetRange("bytes=0-1"); + auto result = GetObject(req); + if (result.IsSuccess()) return std::nullopt; return result.GetError(); diff --git a/src/IO/S3/Client.h b/src/IO/S3/Client.h index ad4d685d88a2..0dd58e76780e 100644 --- a/src/IO/S3/Client.h +++ b/src/IO/S3/Client.h @@ -301,7 +301,7 @@ class Client : private Aws::S3::S3Client void updateURIForBucket(const std::string & bucket, S3::URI new_uri) const; std::optional getURIFromError(const Aws::S3::S3Error & error) const; - std::optional updateURIForBucketForHead(const std::string & bucket) const; + std::optional updateURIForBucketForHead(const std::string & bucket, const std::string & key) const; std::optional getURIForBucket(const std::string & bucket) const; diff --git a/src/IO/S3/URI.cpp b/src/IO/S3/URI.cpp index dd5429dcf55f..0028db01b004 100644 --- a/src/IO/S3/URI.cpp +++ b/src/IO/S3/URI.cpp @@ -156,6 +156,7 @@ URI::URI(const std::string & uri_, bool allow_archive_path_syntax, bool keep_pre validateKey(key, uri); } +<<<<<<< HEAD bool URI::tryInitPathStyle() { /// Case when bucket name and key represented in the path of S3 URL. @@ -211,12 +212,74 @@ bool URI::tryInitVirtualHostedStyle(bool is_using_aws_private_link_interface, bo else storage_name = name; return true; +======= +bool URI::isAWSRegion(std::string_view region) +{ + /// List from https://docs.aws.amazon.com/general/latest/gr/s3.html + static const std::unordered_set regions = { + "us-east-2", + "us-east-1", + "us-west-1", + "us-west-2", + "af-south-1", + "ap-east-1", + "ap-south-2", + "ap-southeast-3", + "ap-southeast-5", + "ap-southeast-4", + "ap-south-1", + "ap-northeast-3", + "ap-northeast-2", + "ap-southeast-1", + "ap-southeast-2", + "ap-east-2", + "ap-southeast-7", + "ap-northeast-1", + "ca-central-1", + "ca-west-1", + "eu-central-1", + "eu-west-1", + "eu-west-2", + "eu-south-1", + "eu-west-3", + "eu-south-2", + "eu-north-1", + "eu-central-2", + "il-central-1", + "mx-central-1", + "me-south-1", + "me-central-1", + "sa-east-1", + "us-gov-east-1", + "us-gov-west-1" + }; + + /// 's3-us-west-2' is a legacy region format for S3 storage, equals to 'us-west-2' + /// See https://docs.aws.amazon.com/AmazonS3/latest/userguide/VirtualHosting.html#VirtualHostingBackwardsCompatibility + if (region.substr(0, 3) == "s3-") + region = region.substr(3); + + return regions.contains(region); +>>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) } void URI::addRegionToURI(const std::string ®ion) { if (auto pos = endpoint.find(".amazonaws.com"); pos != std::string::npos) + { + if (pos > 0) + { /// Check if region is already in endpoint to avoid add it second time + auto prev_pos = endpoint.find_last_of("/.", pos - 1); + if (prev_pos == std::string::npos) + prev_pos = 0; + else + ++prev_pos; + std::string_view endpoint_region = std::string_view(endpoint).substr(prev_pos, pos - prev_pos); + if (isAWSRegion(endpoint_region)) + return; + } endpoint = endpoint.substr(0, pos) + "." + region + endpoint.substr(pos); + } } void URI::validateBucket(const String & bucket, const Poco::URI & uri) diff --git a/src/IO/S3/URI.h b/src/IO/S3/URI.h index 64b4def76744..ada42a34f672 100644 --- a/src/IO/S3/URI.h +++ b/src/IO/S3/URI.h @@ -47,9 +47,15 @@ struct URI static void validateBucket(const std::string & bucket, const Poco::URI & uri); static void validateKey(const std::string & key, const Poco::URI & uri); +<<<<<<< HEAD private: bool tryInitPathStyle(); bool tryInitVirtualHostedStyle(bool is_using_aws_private_link_interface, bool use_strict_pattern); +======= + /// Returns true if 'region' string is an AWS S3 region + /// https://docs.aws.amazon.com/general/latest/gr/s3.html + static bool isAWSRegion(std::string_view region); +>>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) }; } diff --git a/src/IO/S3/getObjectInfo.cpp b/src/IO/S3/getObjectInfo.cpp index 1cdcd9f94e37..e9aff5925e38 100644 --- a/src/IO/S3/getObjectInfo.cpp +++ b/src/IO/S3/getObjectInfo.cpp @@ -1,6 +1,7 @@ #include #include #include +#include #if USE_AWS_S3 @@ -8,6 +9,11 @@ namespace ProfileEvents { extern const Event S3GetObjectTagging; extern const Event S3HeadObject; +<<<<<<< HEAD +======= + extern const Event S3HeadObjectMicroseconds; + extern const Event DiskS3GetObject; +>>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) extern const Event DiskS3GetObjectTagging; extern const Event DiskS3HeadObject; } @@ -27,6 +33,7 @@ namespace ProfileEvents::increment(ProfileEvents::S3HeadObject); if (client.isClientForDisk()) ProfileEvents::increment(ProfileEvents::DiskS3HeadObject); + ProfileEventTimeIncrement watch(ProfileEvents::S3HeadObjectMicroseconds); S3::HeadObjectRequest req; req.SetBucket(bucket); diff --git a/src/Interpreters/Cluster.cpp b/src/Interpreters/Cluster.cpp index 2967d2e36a23..60c5bbed04c5 100644 --- a/src/Interpreters/Cluster.cpp +++ b/src/Interpreters/Cluster.cpp @@ -754,9 +754,9 @@ void Cluster::initMisc() } } -std::unique_ptr Cluster::getClusterWithReplicasAsShards(const Settings & settings, size_t max_replicas_from_shard) const +std::unique_ptr Cluster::getClusterWithReplicasAsShards(const Settings & settings, size_t max_replicas_from_shard, size_t max_hosts) const { - return std::unique_ptr{ new Cluster(ReplicasAsShardsTag{}, *this, settings, max_replicas_from_shard)}; + return std::unique_ptr{ new Cluster(ReplicasAsShardsTag{}, *this, settings, max_replicas_from_shard, max_hosts)}; } std::unique_ptr Cluster::getClusterWithSingleShard(size_t index) const @@ -805,7 +805,7 @@ void shuffleReplicas(std::vector & replicas, const Settings & } -Cluster::Cluster(Cluster::ReplicasAsShardsTag, const Cluster & from, const Settings & settings, size_t max_replicas_from_shard) +Cluster::Cluster(Cluster::ReplicasAsShardsTag, const Cluster & from, const Settings & settings, size_t max_replicas_from_shard, size_t max_hosts) { if (from.addresses_with_failover.empty()) throw Exception(ErrorCodes::LOGICAL_ERROR, "Cluster is empty"); @@ -827,6 +827,7 @@ Cluster::Cluster(Cluster::ReplicasAsShardsTag, const Cluster & from, const Setti if (address.is_local) info.local_addresses.push_back(address); + addresses_with_failover.emplace_back(Addresses({address})); auto pool = ConnectionPoolFactory::instance().get( static_cast(settings[Setting::distributed_connections_pool_size]), @@ -850,9 +851,6 @@ Cluster::Cluster(Cluster::ReplicasAsShardsTag, const Cluster & from, const Setti info.per_replica_pools = {std::move(pool)}; info.default_database = address.default_database; - addresses_with_failover.emplace_back(Addresses{address}); - - slot_to_shard.insert(std::end(slot_to_shard), info.weight, shards_info.size()); shards_info.emplace_back(std::move(info)); } }; @@ -874,10 +872,37 @@ Cluster::Cluster(Cluster::ReplicasAsShardsTag, const Cluster & from, const Setti secret = from.secret; name = from.name; + constrainShardInfoAndAddressesToMaxHosts(max_hosts); + + for (size_t i = 0; i < shards_info.size(); ++i) + slot_to_shard.insert(std::end(slot_to_shard), shards_info[i].weight, i); + initMisc(); } +void Cluster::constrainShardInfoAndAddressesToMaxHosts(size_t max_hosts) +{ + if (max_hosts == 0 || shards_info.size() <= max_hosts) + return; + + pcg64_fast gen{randomSeed()}; + std::shuffle(shards_info.begin(), shards_info.end(), gen); + shards_info.resize(max_hosts); + + AddressesWithFailover addresses_with_failover_; + + UInt32 shard_num = 0; + for (auto & shard_info : shards_info) + { + addresses_with_failover_.push_back(addresses_with_failover[shard_info.shard_num - 1]); + shard_info.shard_num = ++shard_num; + } + + addresses_with_failover.swap(addresses_with_failover_); +} + + Cluster::Cluster(Cluster::SubclusterTag, const Cluster & from, const std::vector & indices) { for (size_t index : indices) diff --git a/src/Interpreters/Cluster.h b/src/Interpreters/Cluster.h index 2b74c2b9e6c8..2d707b51265f 100644 --- a/src/Interpreters/Cluster.h +++ b/src/Interpreters/Cluster.h @@ -285,7 +285,7 @@ class Cluster std::unique_ptr getClusterWithMultipleShards(const std::vector & indices) const; /// Get a new Cluster that contains all servers (all shards with all replicas) from existing cluster as independent shards. - std::unique_ptr getClusterWithReplicasAsShards(const Settings & settings, size_t max_replicas_from_shard = 0) const; + std::unique_ptr getClusterWithReplicasAsShards(const Settings & settings, size_t max_replicas_from_shard = 0, size_t max_hosts = 0) const; /// Returns false if cluster configuration doesn't allow to use it for cross-replication. /// NOTE: true does not mean, that it's actually a cross-replication cluster. @@ -311,7 +311,7 @@ class Cluster /// For getClusterWithReplicasAsShards implementation struct ReplicasAsShardsTag {}; - Cluster(ReplicasAsShardsTag, const Cluster & from, const Settings & settings, size_t max_replicas_from_shard); + Cluster(ReplicasAsShardsTag, const Cluster & from, const Settings & settings, size_t max_replicas_from_shard, size_t max_hosts); void addShard( const Settings & settings, @@ -322,6 +322,9 @@ class Cluster UInt32 weight = 1, bool internal_replication = false); + /// Reduce size of cluster to max_hosts + void constrainShardInfoAndAddressesToMaxHosts(size_t max_hosts); + /// Inter-server secret String secret; diff --git a/src/Interpreters/ClusterDiscovery.cpp b/src/Interpreters/ClusterDiscovery.cpp index 20efaebf2f4b..e92b6ac8683f 100644 --- a/src/Interpreters/ClusterDiscovery.cpp +++ b/src/Interpreters/ClusterDiscovery.cpp @@ -295,17 +295,32 @@ Strings ClusterDiscovery::getNodeNames(zkutil::ZooKeeperPtr & zk, auto callback = get_nodes_callbacks.find(cluster_name); if (callback == get_nodes_callbacks.end()) { - auto watch_dynamic_callback = std::make_shared([ - cluster_name, - my_clusters_to_update = clusters_to_update, - my_discovery_paths_need_update = multicluster_discovery_paths[zk_root_index - 1].need_update - ](auto) - { - my_discovery_paths_need_update->store(true); - my_clusters_to_update->set(cluster_name); - }); - auto res = get_nodes_callbacks.insert(std::make_pair(cluster_name, watch_dynamic_callback)); - callback = res.first; + if (zk_root_index > 0) + { + auto watch_dynamic_callback = std::make_shared([ + cluster_name, + my_clusters_to_update = clusters_to_update, + my_discovery_paths_need_update = multicluster_discovery_paths[zk_root_index - 1].need_update + ](auto) + { + my_discovery_paths_need_update->store(true); + my_clusters_to_update->set(cluster_name); + }); + auto res = get_nodes_callbacks.insert(std::make_pair(cluster_name, watch_dynamic_callback)); + callback = res.first; + } + else + { // zk_root_index == 0 for static clusters + auto watch_dynamic_callback = std::make_shared([ + cluster_name, + my_clusters_to_update = clusters_to_update + ](auto) + { + my_clusters_to_update->set(cluster_name); + }); + auto res = get_nodes_callbacks.insert(std::make_pair(cluster_name, watch_dynamic_callback)); + callback = res.first; + } } nodes = zk->getChildrenWatch( getShardsListPath(zk_root), diff --git a/src/Interpreters/IcebergMetadataLog.cpp b/src/Interpreters/IcebergMetadataLog.cpp index 9536bc4ae96b..fe8c14dd3ec4 100644 --- a/src/Interpreters/IcebergMetadataLog.cpp +++ b/src/Interpreters/IcebergMetadataLog.cpp @@ -12,6 +12,7 @@ #include #include #include +#include #include #include #include @@ -84,7 +85,7 @@ void IcebergMetadataLogElement::appendToBlock(MutableColumns & columns) const void insertRowToLogTable( const ContextPtr & local_context, - String row, + std::function get_row, IcebergMetadataLogLevel row_log_level, const String & table_path, const Iceberg::IcebergPathFromMetadata & file_path, @@ -111,8 +112,13 @@ void insertRowToLogTable( .query_id = local_context->getCurrentQueryId(), .content_type = row_log_level, .table_path = table_path, +<<<<<<< HEAD .file_path = file_path.serialize(), .metadata_content = row, +======= + .file_path = file_path, + .metadata_content = get_row(), +>>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) .row_in_file = row_in_file, .pruning_status = pruning_status}); } diff --git a/src/Interpreters/IcebergMetadataLog.h b/src/Interpreters/IcebergMetadataLog.h index b850fb0a8500..cdad76443d52 100644 --- a/src/Interpreters/IcebergMetadataLog.h +++ b/src/Interpreters/IcebergMetadataLog.h @@ -27,9 +27,11 @@ struct IcebergMetadataLogElement void appendToBlock(MutableColumns & columns) const; }; +/// Here `get_row` function is used instead `row` string to calculate string only when required. +/// Inside `insertRowToLogTable` code can exit immediately after `iceberg_metadata_log_level` setting check. void insertRowToLogTable( const ContextPtr & local_context, - String row, + std::function get_row, IcebergMetadataLogLevel row_log_level, const String & table_path, const Iceberg::IcebergPathFromMetadata & file_path, diff --git a/src/Interpreters/InterpreterCreateQuery.cpp b/src/Interpreters/InterpreterCreateQuery.cpp index 7fba368de6a6..b6ed676f3e99 100644 --- a/src/Interpreters/InterpreterCreateQuery.cpp +++ b/src/Interpreters/InterpreterCreateQuery.cpp @@ -2125,8 +2125,7 @@ bool InterpreterCreateQuery::doCreateTable(ASTCreateQuery & create, auto table_function_ast = create.as_table_function->ptr(); auto table_function = TableFunctionFactory::instance().get(table_function_ast, getContext()); - if (!table_function->canBeUsedToCreateTable()) - throw Exception(ErrorCodes::BAD_ARGUMENTS, "Table function '{}' cannot be used to create a table", table_function->getName()); + table_function->validateUseToCreateTable(); /// In case of CREATE AS table_function() query we should use global context /// in storage creation because there will be no query context on server startup diff --git a/src/Interpreters/InterpreterInsertQuery.cpp b/src/Interpreters/InterpreterInsertQuery.cpp index 21bdbd90108c..667b0817070a 100644 --- a/src/Interpreters/InterpreterInsertQuery.cpp +++ b/src/Interpreters/InterpreterInsertQuery.cpp @@ -885,6 +885,9 @@ std::optional InterpreterInsertQuery::distributedWriteIntoReplica if (!src_storage_cluster) return {}; + if (src_storage_cluster->getClusterName(local_context).empty()) + return {}; + if (!isInsertSelectTrivialEnoughForDistributedExecution(query)) return {}; diff --git a/src/Parsers/ASTSetQuery.cpp b/src/Parsers/ASTSetQuery.cpp index f4fe077280ac..968e4f4ee569 100644 --- a/src/Parsers/ASTSetQuery.cpp +++ b/src/Parsers/ASTSetQuery.cpp @@ -131,7 +131,8 @@ void ASTSetQuery::formatImpl(WriteBuffer & ostr, const FormatSettings & format, return true; } - if (DataLake::DATABASE_ENGINE_NAME == state.create_engine_name) + if (DataLake::DATABASE_ENGINE_NAME == state.create_engine_name + || DataLake::DATABASE_ALIAS_NAME == state.create_engine_name) { if (DataLake::SETTINGS_TO_HIDE.contains(change.name)) { diff --git a/src/Parsers/FunctionSecretArgumentsFinder.h b/src/Parsers/FunctionSecretArgumentsFinder.h index d5c490cd426f..525597f6db1a 100644 --- a/src/Parsers/FunctionSecretArgumentsFinder.h +++ b/src/Parsers/FunctionSecretArgumentsFinder.h @@ -3,9 +3,12 @@ #include #include #include +#include +#include #include #include #include +#include #include @@ -30,6 +33,21 @@ class AbstractFunction virtual ~Arguments() = default; virtual size_t size() const = 0; virtual std::unique_ptr at(size_t n) const = 0; + void skipArgument(size_t n) { skipped_indexes.insert(n); } + void unskipArguments() { skipped_indexes.clear(); } + size_t getRealIndex(size_t n) const + { + for (auto idx : skipped_indexes) + { + if (n < idx) + break; + ++n; + } + return n; + } + size_t skippedSize() const { return skipped_indexes.size(); } + private: + std::set skipped_indexes; }; virtual ~AbstractFunction() = default; @@ -78,11 +96,13 @@ class FunctionSecretArgumentsFinder { if (index >= function->arguments->size()) return; + auto real_index = function->arguments->getRealIndex(index); if (!result.count) { - result.start = index; + result.start = real_index; result.are_named = argument_is_named; } +<<<<<<< HEAD chassert(result.replacement.empty()); /// We shouldn't use replacement with masking other arguments /// Widen the masked range to cover `index`. Arguments are normally marked consecutively in /// increasing order, but a malformed query can mix the named secret form (`key = ...`) with the @@ -91,6 +111,11 @@ class FunctionSecretArgumentsFinder size_t end = std::max(result.start + result.count, index + 1); result.start = std::min(result.start, index); result.count = end - result.start; +======= + chassert(real_index >= result.start); /// We always check arguments consecutively + chassert(result.replacement.empty()); /// We shouldn't use replacement with masking other arguments + result.count = real_index + 1 - result.start; +>>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) if (!argument_is_named) result.are_named = false; } @@ -108,18 +133,36 @@ class FunctionSecretArgumentsFinder { findMongoDBSecretArguments(); } + else if (function->name() == "iceberg") + { + findIcebergFunctionSecretArguments(/* is_cluster_function= */ false); + } + else if (function ->name() == "icebergCluster") + { + findIcebergFunctionSecretArguments(/* is_cluster_function= */ true); + } else if ((function->name() == "s3") || (function->name() == "cosn") || (function->name() == "oss") || +<<<<<<< HEAD (function->name() == "deltaLake") || (function->name() == "deltaLakeS3") || (function->name() == "hudi") || (function->name() == "iceberg") || (function->name() == "gcs") || (function->name() == "icebergS3") || (function->name() == "paimon") || (function->name() == "paimonS3")) +======= + (function->name() == "deltaLake") || (function->name() == "hudi") || + (function->name() == "gcs") || (function->name() == "icebergS3") || (function->name() == "paimon") || + (function->name() == "paimonS3")) +>>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) { /// s3('url', 'aws_access_key_id', 'aws_secret_access_key', ...) findS3FunctionSecretArguments(/* is_cluster_function= */ false); } else if ((function->name() == "s3Cluster") || (function ->name() == "hudiCluster") || (function ->name() == "deltaLakeCluster") || (function ->name() == "deltaLakeS3Cluster") || +<<<<<<< HEAD (function ->name() == "icebergS3Cluster") || (function ->name() == "icebergCluster") || (function ->name() == "paimonCluster") || (function ->name() == "paimonS3Cluster")) +======= + (function ->name() == "icebergS3Cluster")) +>>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) { /// s3Cluster('cluster_name', 'url', 'aws_access_key_id', 'aws_secret_access_key', ...) findS3FunctionSecretArguments(/* is_cluster_function= */ true); @@ -360,6 +403,12 @@ class FunctionSecretArgumentsFinder findSecretNamedArgument("secret_access_key", 1); return; } + if (is_cluster_function && isNamedCollectionName(1)) + { + /// s3Cluster(cluster, named_collection, ..., secret_access_key = 'secret_access_key', ...) + findSecretNamedArgument("secret_access_key", 2); + return; + } findSecretNamedArgument("secret_access_key", url_arg_idx); @@ -367,6 +416,7 @@ class FunctionSecretArgumentsFinder /// s3('url', NOSIGN, 'format' [, 'compression'] [, extra_credentials(..)] [, headers(..)]) /// s3('url', 'format', 'structure' [, 'compression'] [, extra_credentials(..)] [, headers(..)]) size_t count = excludeS3OrURLNestedMaps(); + if ((url_arg_idx + 3 <= count) && (count <= url_arg_idx + 4)) { String second_arg; @@ -431,6 +481,48 @@ class FunctionSecretArgumentsFinder markSecretArgument(url_arg_idx + 4); } + std::string findIcebergStorageType(bool is_cluster_function) + { + std::string storage_type = "s3"; + + size_t count = function->arguments->size(); + if (!count) + return storage_type; + + auto storage_type_idx = findNamedArgument(&storage_type, "storage_type"); + if (storage_type_idx != -1) + { + storage_type = Poco::toLower(storage_type); + function->arguments->skipArgument(storage_type_idx); + } + else if (isNamedCollectionName(is_cluster_function ? 1 : 0)) + { + std::string collection_name; + if (function->arguments->at(is_cluster_function ? 1 : 0)->tryGetString(&collection_name, true)) + { + NamedCollectionPtr collection = NamedCollectionFactory::instance().tryGet(collection_name); + if (collection && collection->has("storage_type")) + { + storage_type = Poco::toLower(collection->get("storage_type")); + } + } + } + + return storage_type; + } + + void findIcebergFunctionSecretArguments(bool is_cluster_function) + { + auto storage_type = findIcebergStorageType(is_cluster_function); + + if (storage_type == "s3") + findS3FunctionSecretArguments(is_cluster_function); + else if (storage_type == "azure") + findAzureBlobStorageFunctionSecretArguments(is_cluster_function); + + function->arguments->unskipArguments(); + } + bool maskAzureConnectionString(ssize_t url_arg_idx, bool argument_is_named = false, size_t start = 0) { String url_arg; @@ -454,7 +546,7 @@ class FunctionSecretArgumentsFinder if (RE2::Replace(&url_arg, account_key_pattern, "AccountKey=[HIDDEN]\\1")) { chassert(result.count == 0); /// We shouldn't use replacement with masking other arguments - result.start = url_arg_idx; + result.start = function->arguments->getRealIndex(url_arg_idx); result.are_named = argument_is_named; result.count = 1; result.replacement = url_arg; @@ -465,7 +557,7 @@ class FunctionSecretArgumentsFinder if (RE2::Replace(&url_arg, sas_signature_pattern, "SharedAccessSignature=[HIDDEN]\\1")) { chassert(result.count == 0); /// We shouldn't use replacement with masking other arguments - result.start = url_arg_idx; + result.start = function->arguments->getRealIndex(url_arg_idx); result.are_named = argument_is_named; result.count = 1; result.replacement = url_arg; @@ -636,6 +728,7 @@ class FunctionSecretArgumentsFinder void findTableEngineSecretArguments() { const String & engine_name = function->name(); + if (engine_name == "ExternalDistributed") { /// ExternalDistributed('engine', 'host:port', 'database', 'table', 'user', 'password') @@ -653,10 +746,13 @@ class FunctionSecretArgumentsFinder { findMongoDBSecretArguments(); } + else if (engine_name == "Iceberg") + { + findIcebergTableEngineSecretArguments(); + } else if ((engine_name == "S3") || (engine_name == "COSN") || (engine_name == "OSS") || (engine_name == "DeltaLake") || (engine_name == "Hudi") - || (engine_name == "Iceberg") || (engine_name == "IcebergS3") - || (engine_name == "S3Queue")) + || (engine_name == "IcebergS3") || (engine_name == "S3Queue")) { /// S3('url', ['aws_access_key_id', 'aws_secret_access_key',] ...) findS3TableEngineSecretArguments(); @@ -665,7 +761,7 @@ class FunctionSecretArgumentsFinder { findURLSecretArguments(); } - else if (engine_name == "AzureBlobStorage" || engine_name == "AzureQueue") + else if (engine_name == "AzureBlobStorage" || engine_name == "AzureQueue" || engine_name == "IcebergAzure") { findAzureBlobStorageTableEngineSecretArguments(); } @@ -790,6 +886,18 @@ class FunctionSecretArgumentsFinder markSecretArgument(2); } + void findIcebergTableEngineSecretArguments() + { + auto storage_type = findIcebergStorageType(0); + + if (storage_type == "s3") + findS3TableEngineSecretArguments(); + else if (storage_type == "azure") + findAzureBlobStorageTableEngineSecretArguments(); + + function->arguments->unskipArguments(); + } + void findDatabaseEngineSecretArguments() { const String & engine_name = function->name(); @@ -806,7 +914,7 @@ class FunctionSecretArgumentsFinder /// S3('url', 'access_key_id', 'secret_access_key') findS3DatabaseSecretArguments(); } - else if (engine_name == "DataLakeCatalog") + else if (engine_name == "DataLakeCatalog" || engine_name == "Iceberg") { findDataLakeCatalogSecretArguments(); } diff --git a/src/Parsers/FunctionSecretArgumentsFinderAST.h b/src/Parsers/FunctionSecretArgumentsFinderAST.h index 86211b3a299c..42d6ffc806c0 100644 --- a/src/Parsers/FunctionSecretArgumentsFinderAST.h +++ b/src/Parsers/FunctionSecretArgumentsFinderAST.h @@ -54,10 +54,13 @@ class FunctionAST : public AbstractFunction { public: explicit ArgumentsAST(const ASTs * arguments_) : arguments(arguments_) {} - size_t size() const override { return arguments ? arguments->size() : 0; } + size_t size() const override + { /// size withous skipped indexes + return arguments ? arguments->size() - skippedSize() : 0; + } std::unique_ptr at(size_t n) const override - { - return std::make_unique(arguments->at(n).get()); + { /// n is relative index, some can be skipped + return std::make_unique(arguments->at(getRealIndex(n)).get()); } private: const ASTs * arguments = nullptr; diff --git a/src/Processors/QueryPlan/ReadFromObjectStorageStep.cpp b/src/Processors/QueryPlan/ReadFromObjectStorageStep.cpp index b3a1bcab66cd..6daa8ead6201 100644 --- a/src/Processors/QueryPlan/ReadFromObjectStorageStep.cpp +++ b/src/Processors/QueryPlan/ReadFromObjectStorageStep.cpp @@ -146,7 +146,7 @@ void ReadFromObjectStorageStep::initializePipeline(QueryPipelineBuilder & pipeli size_t output_ports = pipe.numOutputPorts(); const bool parallelize_output = context->getSettingsRef()[Setting::parallelize_output_from_storages]; if (parallelize_output - && FormatFactory::instance().checkParallelizeOutputAfterReading(configuration->format, context) + && FormatFactory::instance().checkParallelizeOutputAfterReading(configuration->getFormat(), context) && output_ports > 0 && output_ports < max_num_streams) pipe.resize(max_num_streams); diff --git a/src/Server/TCPHandler.cpp b/src/Server/TCPHandler.cpp index 13012d38525f..d0708fa5c64d 100644 --- a/src/Server/TCPHandler.cpp +++ b/src/Server/TCPHandler.cpp @@ -25,6 +25,7 @@ #include #include #include +#include #include #include #include @@ -35,7 +36,11 @@ #include #include #include +<<<<<<< HEAD #include +======= +#include +>>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) #include #include #include diff --git a/src/Storages/HivePartitioningUtils.cpp b/src/Storages/HivePartitioningUtils.cpp index be72281a9ed6..f635ce13b14a 100644 --- a/src/Storages/HivePartitioningUtils.cpp +++ b/src/Storages/HivePartitioningUtils.cpp @@ -226,9 +226,9 @@ HivePartitionColumnsWithFileColumnsPair setupHivePartitioningForObjectStorage( * Otherwise, in case `use_hive_partitioning=1`, we can keep the old behavior of extracting it from the sample path. * And if the schema was inferred (not specified in the table definition), we need to enrich it with the path partition columns */ - if (configuration->partition_strategy && configuration->partition_strategy_type == PartitionStrategyFactory::StrategyType::HIVE) + if (configuration->getPartitionStrategy() && configuration->getPartitionStrategyType() == PartitionStrategyFactory::StrategyType::HIVE) { - hive_partition_columns_to_read_from_file_path = configuration->partition_strategy->getPartitionColumns(); + hive_partition_columns_to_read_from_file_path = configuration->getPartitionStrategy()->getPartitionColumns(); sanityCheckSchemaAndHivePartitionColumns(hive_partition_columns_to_read_from_file_path, columns, /* check_contained_in_schema */true); } else if (context->getSettingsRef()[Setting::use_hive_partitioning]) @@ -242,7 +242,7 @@ HivePartitionColumnsWithFileColumnsPair setupHivePartitioningForObjectStorage( sanityCheckSchemaAndHivePartitionColumns(hive_partition_columns_to_read_from_file_path, columns, /* check_contained_in_schema */false); } - if (configuration->partition_columns_in_data_file) + if (configuration->getPartitionColumnsInDataFile()) { file_columns = columns.getAllPhysical(); } diff --git a/src/Storages/IStorage.h b/src/Storages/IStorage.h index 9c02ade588df..890cae49c236 100644 --- a/src/Storages/IStorage.h +++ b/src/Storages/IStorage.h @@ -73,6 +73,9 @@ using ConditionSelectivityEstimatorPtr = std::shared_ptr; + class ActionsDAG; /** Storage. Describes the table. Responsible for @@ -397,6 +400,7 @@ class IStorage : public std::enable_shared_from_this, public TypePromo size_t /*max_block_size*/, size_t /*num_streams*/); +public: /// Should we process blocks of data returned by the storage in parallel /// even when the storage returned only one stream of data for reading? /// It is beneficial, for example, when you read from a file quickly, @@ -407,7 +411,6 @@ class IStorage : public std::enable_shared_from_this, public TypePromo /// useless). virtual bool parallelizeOutputAfterReading(ContextPtr) const { return !isSystemStorage(); } -public: /// Other version of read which adds reading step to query plan. /// Default implementation creates ReadFromStorageStep and uses usual read. /// Can be called after `shutdown`, but not after `drop`. diff --git a/src/Storages/IStorageCluster.cpp b/src/Storages/IStorageCluster.cpp index 90c887dd3eb4..8749a4dc571a 100644 --- a/src/Storages/IStorageCluster.cpp +++ b/src/Storages/IStorageCluster.cpp @@ -1,5 +1,8 @@ #include +#include +#include + #include #include #include @@ -12,6 +15,7 @@ #include #include #include +#include #include #include #include @@ -19,6 +23,9 @@ #include #include #include +#include +#include +#include #include #include #include @@ -49,19 +56,18 @@ namespace Setting extern const SettingsBool async_query_sending_for_remote; extern const SettingsBool async_socket_for_remote; extern const SettingsBool skip_unavailable_shards; - extern const SettingsBool parallel_replicas_local_plan; - extern const SettingsString cluster_for_parallel_replicas; extern const SettingsNonZeroUInt64 max_parallel_replicas; + extern const SettingsUInt64 object_storage_max_nodes; + extern const SettingsBool object_storage_remote_initiator; + extern const SettingsString object_storage_remote_initiator_cluster; extern const SettingsObjectStorageClusterJoinMode object_storage_cluster_join_mode; } namespace ErrorCodes { extern const int LOGICAL_ERROR; -} - -namespace ErrorCodes -{ + extern const int NOT_IMPLEMENTED; + extern const int BAD_ARGUMENTS; extern const int ALL_CONNECTION_TRIES_FAILED; } @@ -309,24 +315,58 @@ void IStorageCluster::read( SelectQueryInfo & query_info, ContextPtr context, QueryProcessingStage::Enum processed_stage, - size_t /*max_block_size*/, - size_t /*num_streams*/) + size_t max_block_size, + size_t num_streams) { + if (!isClusterSupported()) + { + readFallBackToPure(query_plan, column_names, storage_snapshot, query_info, context, processed_stage, max_block_size, num_streams); + return; + } + + auto cluster_name_from_settings = getClusterName(context); + const auto & settings = context->getSettingsRef(); + ASTPtr query_to_send = query_info.query; + + if (cluster_name_from_settings.empty()) + { + if (settings[Setting::object_storage_remote_initiator]) + { + /// rewrite query to execute `remote('remote_host', s3(...))` + /// remote_host can execute query itself or make on-cluster query depends on own `object_storage_cluster` setting + updateConfigurationIfNeeded(context); + updateQueryWithJoinToSendIfNeeded(query_to_send, query_info.query_tree, context); + updateQueryToSendIfNeeded(query_to_send, storage_snapshot, context, /*make_cluster_function*/ false); + + auto remote_initiator_cluster_name = settings[Setting::object_storage_remote_initiator_cluster].value; + if (remote_initiator_cluster_name.empty()) + throw Exception(ErrorCodes::BAD_ARGUMENTS, "Setting 'object_storage_remote_initiator' can be used only with 'object_storage_remote_initiator_cluster' or 'object_storage_cluster'"); + + auto remote_initiator_cluster = getClusterImpl(context, remote_initiator_cluster_name); + auto storage_and_context = convertToRemote(remote_initiator_cluster, context, remote_initiator_cluster_name, query_to_send); + auto src_distributed = std::dynamic_pointer_cast(storage_and_context.storage); + auto modified_query_info = query_info; + modified_query_info.cluster = src_distributed->getCluster(); + auto new_storage_snapshot = storage_and_context.storage->getStorageSnapshot(storage_snapshot->metadata, storage_and_context.context); + storage_and_context.storage->read(query_plan, column_names, new_storage_snapshot, modified_query_info, storage_and_context.context, processed_stage, max_block_size, num_streams); + return; + } + + readFallBackToPure(query_plan, column_names, storage_snapshot, query_info, context, processed_stage, max_block_size, num_streams); + return; + } + updateConfigurationIfNeeded(context); storage_snapshot->check(column_names); - updateBeforeRead(context); - auto cluster = getCluster(context); - /// Calculate the header. This is significant, because some columns could be thrown away in some cases like query with count(*) SharedHeader sample_block; - ASTPtr query_to_send = query_info.query; updateQueryWithJoinToSendIfNeeded(query_to_send, query_info.query_tree, context); - if (context->getSettingsRef()[Setting::allow_experimental_analyzer]) + if (settings[Setting::allow_experimental_analyzer]) { sample_block = InterpreterSelectQueryAnalyzer::getSampleBlock(query_to_send, context, SelectQueryOptions(processed_stage)); } @@ -337,7 +377,32 @@ void IStorageCluster::read( query_to_send = interpreter.getQueryInfo().query->clone(); } - updateQueryToSendIfNeeded(query_to_send, storage_snapshot, context); + updateQueryToSendIfNeeded(query_to_send, storage_snapshot, context, /*make_cluster_function*/ true); + + /// In case the current node is not supposed to initiate the clustered query + /// Sends this query to a remote initiator using the `remote` table function + if (settings[Setting::object_storage_remote_initiator]) + { + /// Re-writes queries in the form of: + /// Input: SELECT * FROM iceberg(...) SETTINGS object_storage_cluster='swarm', object_storage_remote_initiator=1 + /// Output: SELECT * FROM remote('remote_host', icebergCluster('swarm', ...) + /// Where `remote_host` is a random host from the cluster which will execute the query + /// This means the initiator node belongs to the same cluster that will execute the query + /// In case remote_initiator_cluster_name is set, the initiator might be set to a different cluster + auto remote_initiator_cluster_name = settings[Setting::object_storage_remote_initiator_cluster].value; + if (remote_initiator_cluster_name.empty()) + remote_initiator_cluster_name = cluster_name_from_settings; + auto remote_initiator_cluster = getClusterImpl(context, remote_initiator_cluster_name); + auto storage_and_context = convertToRemote(remote_initiator_cluster, context, remote_initiator_cluster_name, query_to_send); + auto src_distributed = std::dynamic_pointer_cast(storage_and_context.storage); + auto modified_query_info = query_info; + modified_query_info.cluster = src_distributed->getCluster(); + auto new_storage_snapshot = storage_and_context.storage->getStorageSnapshot(storage_snapshot->metadata, storage_and_context.context); + storage_and_context.storage->read(query_plan, column_names, new_storage_snapshot, modified_query_info, storage_and_context.context, processed_stage, max_block_size, num_streams); + return; + } + + auto cluster = getClusterImpl(context, cluster_name_from_settings, isObjectStorage() ? settings[Setting::object_storage_max_nodes] : 0); RestoreQualifiedNamesVisitor::Data data; data.distributed_table = DatabaseAndTableWithAlias(*getTableExpression(query_to_send->as(), 0)); @@ -366,6 +431,98 @@ void IStorageCluster::read( query_plan.addStep(std::move(reading)); } +IStorageCluster::RemoteCallVariables IStorageCluster::convertToRemote( + ClusterPtr cluster, + ContextPtr context, + const std::string & cluster_name_from_settings, + ASTPtr query_to_send) +{ + /// TODO: Allow to use secret for remote queries + if (!cluster->getSecret().empty()) + throw Exception(ErrorCodes::NOT_IMPLEMENTED, "Can't convert query to remote when cluster uses secret"); + + auto host_addresses = cluster->getShardsAddresses(); + if (host_addresses.empty()) + throw Exception(ErrorCodes::BAD_ARGUMENTS, "Empty cluster {}", cluster_name_from_settings); + + pcg64 rng(randomSeed()); + size_t shard_num = rng() % host_addresses.size(); + auto shard_addresses = host_addresses[shard_num]; + /// After getClusterImpl each shard must have exactly 1 replica + if (shard_addresses.size() != 1) + throw Exception(ErrorCodes::BAD_ARGUMENTS, "Size of shard {} in cluster {} is not equal 1", shard_num, cluster_name_from_settings); + std::string host_name; + Poco::URI::decode(shard_addresses[0].toString(), host_name); + + LOG_INFO(log, "Choose remote initiator '{}'", host_name); + + bool secure = shard_addresses[0].secure == Protocol::Secure::Enable; + std::string remote_function_name = secure ? "remoteSecure" : "remote"; + + /// Clean object_storage_remote_initiator setting to avoid infinite remote call + auto new_context = Context::createCopy(context); + std::vector settings_to_remove = {"object_storage_remote_initiator", "object_storage_remote_initiator_cluster"}; + new_context->resetSettingsToDefaultValue(settings_to_remove); + + auto * select_query = query_to_send->as(); + if (!select_query) + throw Exception(ErrorCodes::LOGICAL_ERROR, "Expected SELECT query"); + + auto query_settings = select_query->settings(); + if (query_settings) + { + auto & settings_ast = query_settings->as(); + bool settings_changed = false; + for (const auto & setting_to_remove : settings_to_remove) + settings_changed |= settings_ast.changes.removeSetting(setting_to_remove); + if (settings_changed && settings_ast.changes.empty()) + select_query->setExpression(ASTSelectQuery::Expression::SETTINGS, {}); + } + + ASTTableExpression * table_expression = extractTableExpressionASTPtrFromSelectQuery(query_to_send); + if (!table_expression) + throw Exception(ErrorCodes::LOGICAL_ERROR, "Can't find table expression"); + if (!table_expression->table_function) + throw Exception(ErrorCodes::LOGICAL_ERROR, "Can't find table function in table expression"); + + boost::intrusive_ptr remote_query; + + if (shard_addresses[0].user_specified) + { // with user/password for clsuter access remote query is executed from this user, add it in query parameters + remote_query = makeASTFunction(remote_function_name, + make_intrusive(host_name), + table_expression->table_function, + make_intrusive(shard_addresses[0].user), + make_intrusive(shard_addresses[0].password)); + } + else + { // without specified user/password remote query is executed from default user + remote_query = makeASTFunction(remote_function_name, make_intrusive(host_name), table_expression->table_function); + } + + table_expression->table_function = remote_query; + + auto remote_function = TableFunctionFactory::instance().get(remote_query, new_context); + + auto storage = remote_function->execute(query_to_send, new_context, remote_function_name); + + return RemoteCallVariables{storage, new_context}; +} + +SinkToStoragePtr IStorageCluster::write( + const ASTPtr & query, + const StorageMetadataPtr & metadata_snapshot, + ContextPtr context, + bool async_insert) +{ + auto cluster_name_from_settings = getClusterName(context); + + if (cluster_name_from_settings.empty()) + return writeFallBackToPure(query, metadata_snapshot, context, async_insert); + + throw Exception(ErrorCodes::NOT_IMPLEMENTED, "Method write is not supported by storage {}", getName()); +} + void ReadFromCluster::initializePipeline(QueryPipelineBuilder & pipeline, const BuildQueryPipelineSettings &) { const Scalars & scalars = context->hasQueryContext() ? context->getQueryContext()->getScalars() : Scalars{}; @@ -511,9 +668,9 @@ ContextPtr ReadFromCluster::updateSettings(const Settings & settings) return new_context; } -ClusterPtr IStorageCluster::getCluster(ContextPtr context) const +ClusterPtr IStorageCluster::getClusterImpl(ContextPtr context, const String & cluster_name_, size_t max_hosts) { - return context->getCluster(cluster_name)->getClusterWithReplicasAsShards(context->getSettingsRef()); + return context->getCluster(cluster_name_)->getClusterWithReplicasAsShards(context->getSettingsRef(), /* max_replicas_from_shard */ 0, max_hosts); } } diff --git a/src/Storages/IStorageCluster.h b/src/Storages/IStorageCluster.h index ac76323a22c1..351bde8618fb 100644 --- a/src/Storages/IStorageCluster.h +++ b/src/Storages/IStorageCluster.h @@ -31,10 +31,16 @@ class IStorageCluster : public IStorage SelectQueryInfo & query_info, ContextPtr context, QueryProcessingStage::Enum processed_stage, - size_t /*max_block_size*/, - size_t /*num_streams*/) override; + size_t max_block_size, + size_t num_streams) override; - ClusterPtr getCluster(ContextPtr context) const; + SinkToStoragePtr write( + const ASTPtr & query, + const StorageMetadataPtr & metadata_snapshot, + ContextPtr context, + bool async_insert) override; + + ClusterPtr getCluster(ContextPtr context) const { return getClusterImpl(context, cluster_name); } /// Query is needed for pruning by virtual columns (_file, _path) virtual RemoteQueryExecutor::Extension getTaskIteratorExtension( @@ -52,16 +58,62 @@ class IStorageCluster : public IStorage bool supportsOptimizationToSubcolumns() const override { return false; } bool supportsTrivialCountOptimization(const StorageSnapshotPtr &, ContextPtr) const override { return true; } +<<<<<<< HEAD const String & getClusterName() const { return cluster_name; } +======= + const String & getOriginalClusterName() const { return cluster_name; } + virtual String getClusterName(ContextPtr /* context */) const { return getOriginalClusterName(); } +>>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) protected: - virtual void updateBeforeRead(const ContextPtr &) {} - virtual void updateQueryToSendIfNeeded(ASTPtr & /*query*/, const StorageSnapshotPtr & /*storage_snapshot*/, const ContextPtr & /*context*/) {} + virtual void updateQueryToSendIfNeeded( + ASTPtr & /*query*/, + const StorageSnapshotPtr & /*storage_snapshot*/, + const ContextPtr & /*context*/, + bool /*make_cluster_function*/) {} void updateQueryWithJoinToSendIfNeeded(ASTPtr & query_to_send, QueryTreeNodePtr query_tree, const ContextPtr & context); virtual void updateConfigurationIfNeeded(ContextPtr /* context */) {} + struct RemoteCallVariables + { + StoragePtr storage; + ContextPtr context; + }; + + RemoteCallVariables convertToRemote( + ClusterPtr cluster, + ContextPtr context, + const std::string & cluster_name_from_settings, + ASTPtr query_to_send); + + virtual void readFallBackToPure( + QueryPlan & /* query_plan */, + const Names & /* column_names */, + const StorageSnapshotPtr & /* storage_snapshot */, + SelectQueryInfo & /* query_info */, + ContextPtr /* context */, + QueryProcessingStage::Enum /* processed_stage */, + size_t /* max_block_size */, + size_t /* num_streams */) + { + throw Exception(ErrorCodes::NOT_IMPLEMENTED, "Method readFallBackToPure is not supported by storage {}", getName()); + } + + virtual SinkToStoragePtr writeFallBackToPure( + const ASTPtr & /*query*/, + const StorageMetadataPtr & /*metadata_snapshot*/, + ContextPtr /*context*/, + bool /*async_insert*/) + { + throw Exception(ErrorCodes::NOT_IMPLEMENTED, "Method writeFallBackToPure is not supported by storage {}", getName()); + } + private: + static ClusterPtr getClusterImpl(ContextPtr context, const String & cluster_name_, size_t max_hosts = 0); + + virtual bool isClusterSupported() const { return true; } + LoggerPtr log; String cluster_name; diff --git a/src/Storages/MergeTree/ExportPartitionUtils.cpp b/src/Storages/MergeTree/ExportPartitionUtils.cpp index 02069754bd3d..da11b5a11bbd 100644 --- a/src/Storages/MergeTree/ExportPartitionUtils.cpp +++ b/src/Storages/MergeTree/ExportPartitionUtils.cpp @@ -474,8 +474,8 @@ namespace ExportPartitionUtils /// produced by different writers (ClickHouse vs Spark/Trino) compare equal. /// Comparison is on {function_name, argument}; time_zone is writer-specific /// and not part of the partition spec identity. - const auto expected_canonical = Iceberg::parseTransformAndArgument(expected_transform); - const auto actual_canonical = Iceberg::parseTransformAndArgument(actual_transform); + const auto expected_canonical = Iceberg::parseTransformAndArgument(expected_transform, ""); + const auto actual_canonical = Iceberg::parseTransformAndArgument(actual_transform, ""); const bool transforms_match = (expected_canonical && actual_canonical) ? (expected_canonical->transform_name == actual_canonical->transform_name diff --git a/src/Storages/MergeTree/MergeTreeData.cpp b/src/Storages/MergeTree/MergeTreeData.cpp index 0ff8975a4cbb..a20a916c68a8 100644 --- a/src/Storages/MergeTree/MergeTreeData.cpp +++ b/src/Storages/MergeTree/MergeTreeData.cpp @@ -7133,14 +7133,19 @@ void MergeTreeData::exportPartToTable( else { auto * object_storage = dynamic_cast(dest_storage.get()); + auto * object_storage_cluster = dynamic_cast(dest_storage.get()); /// in theory this should never happen, but just in case - if (!object_storage) + if (!object_storage && !object_storage_cluster) { throw Exception(ErrorCodes::BAD_ARGUMENTS, "Destination storage {} is not a StorageObjectStorage", dest_storage->getName()); } - auto * iceberg_metadata = dynamic_cast(object_storage->getExternalMetadata(query_context)); + IcebergMetadata * iceberg_metadata = nullptr; + if (object_storage) + iceberg_metadata = dynamic_cast(object_storage->getExternalMetadata(query_context)); + else if (object_storage_cluster) + iceberg_metadata = dynamic_cast(object_storage_cluster->getExternalMetadata(query_context)); if (!iceberg_metadata) { throw Exception(ErrorCodes::BAD_ARGUMENTS, "Destination storage {} is a data lake but not an iceberg table", dest_storage->getName()); diff --git a/src/Storages/ObjectStorage/Azure/Configuration.cpp b/src/Storages/ObjectStorage/Azure/Configuration.cpp index a5930cc25015..23ebdc7cb4c5 100644 --- a/src/Storages/ObjectStorage/Azure/Configuration.cpp +++ b/src/Storages/ObjectStorage/Azure/Configuration.cpp @@ -65,6 +65,7 @@ const std::unordered_set optional_configuration_keys = { "partition_columns_in_data_file", "client_id", "tenant_id", + "storage_type", }; void StorageAzureConfiguration::check(ContextPtr context) @@ -211,10 +212,6 @@ void AzureStorageParsedArguments::fromNamedCollection(const NamedCollection & co String connection_url; String container_name; - std::optional account_name; - std::optional account_key; - std::optional client_id; - std::optional tenant_id; if (collection.has("connection_string")) connection_url = collection.get("connection_string"); @@ -406,16 +403,10 @@ void AzureStorageParsedArguments::fromAST(ASTs & engine_args, ContextPtr context std::unordered_map engine_args_to_idx; - String connection_url = checkAndGetLiteralArgument(engine_args[0], "connection_string/storage_account_url"); String container_name = checkAndGetLiteralArgument(engine_args[1], "container"); blob_path = checkAndGetLiteralArgument(engine_args[2], "blobpath"); - std::optional account_name; - std::optional account_key; - std::optional client_id; - std::optional tenant_id; - collectCredentials(extra_credentials, client_id, tenant_id, context); auto is_format_arg = [] (const std::string & s) -> bool @@ -465,8 +456,7 @@ void AzureStorageParsedArguments::fromAST(ASTs & engine_args, ContextPtr context auto sixth_arg = checkAndGetLiteralArgument(engine_args[5], "partition_strategy/structure"); if (magic_enum::enum_contains(sixth_arg, magic_enum::case_insensitive)) { - partition_strategy_type - = magic_enum::enum_cast(sixth_arg, magic_enum::case_insensitive).value(); + partition_strategy_type = magic_enum::enum_cast(sixth_arg, magic_enum::case_insensitive).value(); } else { @@ -588,8 +578,7 @@ void AzureStorageParsedArguments::fromAST(ASTs & engine_args, ContextPtr context auto eighth_arg = checkAndGetLiteralArgument(engine_args[7], "partition_strategy/structure"); if (magic_enum::enum_contains(eighth_arg, magic_enum::case_insensitive)) { - partition_strategy_type - = magic_enum::enum_cast(eighth_arg, magic_enum::case_insensitive).value(); + partition_strategy_type = magic_enum::enum_cast(eighth_arg, magic_enum::case_insensitive).value(); } else { @@ -843,6 +832,26 @@ void StorageAzureConfiguration::initializeFromParsedArguments(const AzureStorage StorageObjectStorageConfiguration::initializeFromParsedArguments(parsed_arguments); blob_path = parsed_arguments.blob_path; connection_params = parsed_arguments.connection_params; + account_name = parsed_arguments.account_name; + account_key = parsed_arguments.account_key; + client_id = parsed_arguments.client_id; + tenant_id = parsed_arguments.tenant_id; +} + +ASTPtr StorageAzureConfiguration::createArgsWithAccessData() const +{ + auto arguments = make_intrusive(); + + arguments->children.push_back(make_intrusive(connection_params.endpoint.storage_account_url)); + arguments->children.push_back(make_intrusive(connection_params.endpoint.container_name)); + arguments->children.push_back(make_intrusive(blob_path.path)); + if (account_name && account_key) + { + arguments->children.push_back(make_intrusive(*account_name)); + arguments->children.push_back(make_intrusive(*account_key)); + } + + return arguments; } void StorageAzureConfiguration::addStructureAndFormatToArgsIfNeeded( @@ -850,13 +859,13 @@ void StorageAzureConfiguration::addStructureAndFormatToArgsIfNeeded( { if (disk) { - if (format == "auto") + if (getFormat() == "auto") { ASTs format_equal_func_args = {make_intrusive("format"), make_intrusive(format_)}; auto format_equal_func = makeASTFunction("equals", std::move(format_equal_func_args)); args.push_back(format_equal_func); } - if (structure == "auto") + if (getStructure() == "auto") { ASTs structure_equal_func_args = {make_intrusive("structure"), make_intrusive(structure_)}; auto structure_equal_func = makeASTFunction("equals", std::move(structure_equal_func_args)); diff --git a/src/Storages/ObjectStorage/Azure/Configuration.h b/src/Storages/ObjectStorage/Azure/Configuration.h index e74279ae1682..44bce3b903ed 100644 --- a/src/Storages/ObjectStorage/Azure/Configuration.h +++ b/src/Storages/ObjectStorage/Azure/Configuration.h @@ -77,6 +77,11 @@ struct AzureStorageParsedArguments : private StorageParsedArguments Path blob_path; AzureBlobStorage::ConnectionParams connection_params; + + std::optional account_name; + std::optional account_key; + std::optional client_id; + std::optional tenant_id; }; class StorageAzureConfiguration : public StorageObjectStorageConfiguration @@ -132,6 +137,7 @@ class StorageAzureConfiguration : public StorageObjectStorageConfiguration onelake_tenant_id = tenant_id_; onelake_use_blob_endpoint = use_blob_endpoint_; } + ASTPtr createArgsWithAccessData() const override; protected: void fromDisk(const String & disk_name, ASTs & args, ContextPtr context, bool with_structure) override; @@ -143,15 +149,22 @@ class StorageAzureConfiguration : public StorageObjectStorageConfiguration Path blob_path; Paths blobs_paths; AzureBlobStorage::ConnectionParams connection_params; - DiskPtr disk; + + std::optional account_name; + std::optional account_key; + std::optional client_id; + std::optional tenant_id; String onelake_client_id; String onelake_client_secret; String onelake_tenant_id; bool onelake_use_blob_endpoint = true; + DiskPtr disk; + void initializeFromParsedArguments(const AzureStorageParsedArguments & parsed_arguments); }; + } #endif diff --git a/src/Storages/ObjectStorage/DataLakes/Common/AvroForIcebergDeserializer.cpp b/src/Storages/ObjectStorage/DataLakes/Common/AvroForIcebergDeserializer.cpp index 4517834d0dec..a2328976f12f 100644 --- a/src/Storages/ObjectStorage/DataLakes/Common/AvroForIcebergDeserializer.cpp +++ b/src/Storages/ObjectStorage/DataLakes/Common/AvroForIcebergDeserializer.cpp @@ -16,6 +16,7 @@ #include #include #include +#include namespace DB::ErrorCodes { @@ -23,6 +24,12 @@ namespace DB::ErrorCodes extern const int INCORRECT_DATA; } +namespace ProfileEvents +{ + extern const Event IcebergAvroFileParsing; + extern const Event IcebergAvroFileParsingMicroseconds; +} + namespace DB::Iceberg { @@ -36,6 +43,9 @@ try : buffer(std::move(buffer_)) , manifest_file_path(manifest_file_path_) { + ProfileEvents::increment(ProfileEvents::IcebergAvroFileParsing); + ProfileEventTimeIncrement watch(ProfileEvents::IcebergAvroFileParsingMicroseconds); + auto manifest_file_reader = std::make_unique(std::make_unique(*buffer), MAX_AVRO_SCHEMA_DEPTH); diff --git a/src/Storages/ObjectStorage/DataLakes/DataLakeConfiguration.h b/src/Storages/ObjectStorage/DataLakes/DataLakeConfiguration.h index f22078d2799c..4faf3bae4cae 100644 --- a/src/Storages/ObjectStorage/DataLakes/DataLakeConfiguration.h +++ b/src/Storages/ObjectStorage/DataLakes/DataLakeConfiguration.h @@ -7,6 +7,7 @@ #include #include +#include #include #include #include @@ -17,11 +18,21 @@ #include #include #include -#include +#include #include #include #include +<<<<<<< HEAD #include +======= +#include +#include +#include +#include +#include +#include + +>>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) #include #include #include @@ -53,21 +64,21 @@ namespace ErrorCodes namespace DataLakeStorageSetting { - extern DataLakeStorageSettingsDatabaseDataLakeCatalogType storage_catalog_type; - extern DataLakeStorageSettingsString object_storage_endpoint; - extern DataLakeStorageSettingsString storage_aws_access_key_id; - extern DataLakeStorageSettingsString storage_aws_secret_access_key; - extern DataLakeStorageSettingsString storage_region; - extern DataLakeStorageSettingsString storage_aws_role_arn; - extern DataLakeStorageSettingsString storage_aws_role_session_name; - extern DataLakeStorageSettingsString storage_catalog_url; - extern DataLakeStorageSettingsString storage_warehouse; - extern DataLakeStorageSettingsString storage_catalog_credential; - - extern DataLakeStorageSettingsString storage_auth_scope; - extern DataLakeStorageSettingsString storage_auth_header; - extern DataLakeStorageSettingsString storage_oauth_server_uri; - extern DataLakeStorageSettingsBool storage_oauth_server_use_request_body; + extern const DataLakeStorageSettingsDatabaseDataLakeCatalogType storage_catalog_type; + extern const DataLakeStorageSettingsString object_storage_endpoint; + extern const DataLakeStorageSettingsString storage_aws_access_key_id; + extern const DataLakeStorageSettingsString storage_aws_secret_access_key; + extern const DataLakeStorageSettingsString storage_region; + extern const DataLakeStorageSettingsString storage_aws_role_arn; + extern const DataLakeStorageSettingsString storage_aws_role_session_name; + extern const DataLakeStorageSettingsString storage_catalog_url; + extern const DataLakeStorageSettingsString storage_warehouse; + extern const DataLakeStorageSettingsString storage_catalog_credential; + extern const DataLakeStorageSettingsString storage_auth_scope; + extern const DataLakeStorageSettingsString storage_auth_header; + extern const DataLakeStorageSettingsString storage_oauth_server_uri; + extern const DataLakeStorageSettingsBool storage_oauth_server_use_request_body; + extern const DataLakeStorageSettingsString iceberg_metadata_file_path; } struct FormatParserSharedResources; @@ -76,11 +87,17 @@ using FormatParserSharedResourcesPtr = std::shared_ptr concept StorageConfiguration = std::derived_from; -template +template class DataLakeConfiguration : public BaseStorageConfiguration, public std::enable_shared_from_this { public: - explicit DataLakeConfiguration(DataLakeStorageSettingsPtr settings_) : settings(settings_) {} + DataLakeConfiguration() {} + + explicit DataLakeConfiguration( + DataLakeStorageSettingsPtr settings_, + std::optional catalog_namespaces_ = std::nullopt) + : settings(settings_) + , catalog_namespaces(catalog_namespaces_.value_or("*")) {} bool isDataLakeConfiguration() const override { return true; } @@ -105,6 +122,7 @@ class DataLakeConfiguration : public BaseStorageConfiguration, public std::enabl return StorageObjectStorageConfiguration::Path(result.ends_with('/') ? result : result + "/"); } + void setRawPath(const StorageObjectStorageConfiguration::Path & path) override { BaseStorageConfiguration::setRawPath(path); } void update(ObjectStoragePtr object_storage, ContextPtr local_context) override { @@ -146,13 +164,13 @@ class DataLakeConfiguration : public BaseStorageConfiguration, public std::enabl bool supportsDelete() const override { - assertInitialized(); + assertInitializedDL(); return current_metadata->supportsDelete(); } bool supportsParallelInsert() const override { - assertInitialized(); + assertInitializedDL(); return current_metadata->supportsParallelInsert(); } @@ -164,19 +182,32 @@ class DataLakeConfiguration : public BaseStorageConfiguration, public std::enabl std::shared_ptr catalog, const std::optional & format_settings) override { +<<<<<<< HEAD assertInitialized(); current_metadata->mutate(commands, storage_ptr, context, storage_id, metadata_snapshot, catalog, format_settings); +======= + assertInitializedDL(); + current_metadata->mutate(commands, shared_from_this(), context, storage_id, metadata_snapshot, catalog, format_settings); +>>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) } void checkMutationIsPossible(ObjectStoragePtr object_storage, ContextPtr context, const MutationCommands & commands) override { +<<<<<<< HEAD lazyInitializeIfNeeded(object_storage, context); +======= + assertInitializedDL(); +>>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) current_metadata->checkMutationIsPossible(commands); } void checkAlterIsPossible(ObjectStoragePtr object_storage, ContextPtr context, const AlterCommands & commands) override { +<<<<<<< HEAD lazyInitializeIfNeeded(object_storage, context); +======= + assertInitializedDL(); +>>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) current_metadata->checkAlterIsPossible(commands); } @@ -187,8 +218,14 @@ class DataLakeConfiguration : public BaseStorageConfiguration, public std::enabl const StorageID & storage_id, std::shared_ptr catalog) override { +<<<<<<< HEAD lazyInitializeIfNeeded(object_storage, context); current_metadata->alter(params, context, storage_id, catalog); +======= + assertInitializedDL(); + current_metadata->alter(params, context); + +>>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) } ObjectStoragePtr createObjectStorage(ContextPtr context, bool is_readonly, StorageObjectStorageConfiguration::CredentialsConfigurationCallback refresh_credentials_callback) override @@ -200,7 +237,7 @@ class DataLakeConfiguration : public BaseStorageConfiguration, public std::enabl std::optional tryGetTableStructureFromMetadata(ContextPtr local_context) const override { - assertInitialized(); + assertInitializedDL(); if (auto schema = current_metadata->getTableSchema(local_context); !schema.empty()) return ColumnsDescription(std::move(schema)); return std::nullopt; @@ -213,7 +250,7 @@ class DataLakeConfiguration : public BaseStorageConfiguration, public std::enabl std::optional totalRows(ContextPtr local_context) override { - assertInitialized(); + assertInitializedDL(); return current_metadata->totalRows(local_context); } @@ -224,38 +261,38 @@ class DataLakeConfiguration : public BaseStorageConfiguration, public std::enabl std::optional totalBytes(ContextPtr local_context) override { - assertInitialized(); + assertInitializedDL(); return current_metadata->totalBytes(local_context); } bool isDataSortedBySortingKey(StorageMetadataPtr metadata_snapshot, ContextPtr local_context) const override { - assertInitialized(); + assertInitializedDL(); return current_metadata->isDataSortedBySortingKey(metadata_snapshot, local_context); } std::shared_ptr getInitialSchemaByPath(ContextPtr local_context, ObjectInfoPtr object_info) const override { - assertInitialized(); + assertInitializedDL(); return current_metadata->getInitialSchemaByPath(local_context, object_info); } std::shared_ptr getSchemaTransformer(ContextPtr local_context, ObjectInfoPtr object_info) const override { - assertInitialized(); + assertInitializedDL(); return current_metadata->getSchemaTransformer(local_context, object_info); } std::optional getTableStateSnapshot(ContextPtr context) const override { - assertInitialized(); + assertInitializedDL(); return current_metadata->getTableStateSnapshot(context); } std::unique_ptr buildStorageMetadataFromState( const DataLakeTableStateSnapshot & state, ContextPtr context) const override { - assertInitialized(); + assertInitializedDL(); auto metadata = current_metadata->buildStorageMetadataFromState(state, context); if (metadata) LOG_TEST(log, "Built storage metadata from state with columns: {}", @@ -265,13 +302,13 @@ class DataLakeConfiguration : public BaseStorageConfiguration, public std::enabl bool shouldReloadSchemaForConsistency(ContextPtr context) const override { - assertInitialized(); + assertInitializedDL(); return current_metadata->shouldReloadSchemaForConsistency(context); } IDataLakeMetadata * getExternalMetadata() override { - assertInitialized(); + assertInitializedDL(); return current_metadata.get(); } @@ -279,7 +316,7 @@ class DataLakeConfiguration : public BaseStorageConfiguration, public std::enabl bool supportsWrites() const override { - assertInitialized(); + assertInitializedDL(); return current_metadata->supportsWrites(); } @@ -290,7 +327,7 @@ class DataLakeConfiguration : public BaseStorageConfiguration, public std::enabl StorageMetadataPtr storage_metadata, ContextPtr context) override { - assertInitialized(); + assertInitializedDL(); return current_metadata->iterate(filter_dag, callback, list_batch_size, storage_metadata, context); } @@ -302,7 +339,7 @@ class DataLakeConfiguration : public BaseStorageConfiguration, public std::enabl /// because the code will be removed ASAP anyway) DeltaLakePartitionColumns getDeltaLakePartitionColumns() const { - assertInitialized(); + assertInitializedDL(); const auto * delta_lake_metadata = dynamic_cast(current_metadata.get()); if (delta_lake_metadata) return delta_lake_metadata->getPartitionColumns(); @@ -312,18 +349,18 @@ class DataLakeConfiguration : public BaseStorageConfiguration, public std::enabl void modifyFormatSettings(FormatSettings & settings_, const Context & local_context) const override { - assertInitialized(); + assertInitializedDL(); current_metadata->modifyFormatSettings(settings_, local_context); } ColumnMapperPtr getColumnMapperForObject(ObjectInfoPtr object_info) const override { - assertInitialized(); + assertInitializedDL(); return current_metadata->getColumnMapperForObject(object_info); } ColumnMapperPtr getColumnMapperForCurrentSchema(StorageMetadataPtr storage_metadata_snapshot, ContextPtr context) const override { - assertInitialized(); + assertInitializedDL(); return current_metadata->getColumnMapperForCurrentSchema(storage_metadata_snapshot, context); } @@ -354,6 +391,7 @@ class DataLakeConfiguration : public BaseStorageConfiguration, public std::enabl std::shared_ptr getCatalog([[maybe_unused]] ContextPtr context, [[maybe_unused]] const StorageID & table_id) const override { +<<<<<<< HEAD #if USE_AVRO && USE_PARQUET if ((*settings)[DataLakeStorageSetting::storage_catalog_type].changed || (*settings)[DataLakeStorageSetting::storage_catalog_url].changed @@ -370,13 +408,56 @@ class DataLakeConfiguration : public BaseStorageConfiguration, public std::enabl return nullptr; return datalake_database->getCatalog(); #else +======= +#if USE_AWS_S3 && USE_AVRO + if ((*settings)[DataLakeStorageSetting::storage_catalog_type].value == DatabaseDataLakeCatalogType::GLUE) + { + auto catalog_parameters = DataLake::CatalogSettings{ + .storage_endpoint = (*settings)[DataLakeStorageSetting::object_storage_endpoint].value, + .aws_access_key_id = (*settings)[DataLakeStorageSetting::storage_aws_access_key_id].value, + .aws_secret_access_key = (*settings)[DataLakeStorageSetting::storage_aws_secret_access_key].value, + .region = (*settings)[DataLakeStorageSetting::storage_region].value, + .namespaces = catalog_namespaces, + .aws_role_arn = (*settings)[DataLakeStorageSetting::storage_aws_role_arn].value, + .aws_role_session_name = (*settings)[DataLakeStorageSetting::storage_aws_role_session_name].value + }; + + return std::make_shared( + (*settings)[DataLakeStorageSetting::storage_catalog_url].value, + context, + catalog_parameters, + /* table_engine_definition */nullptr + ); + } + /// Attach condition is provided for compatibility. + if ((*settings)[DataLakeStorageSetting::storage_catalog_type].value == DatabaseDataLakeCatalogType::ICEBERG_REST || + (is_attach && (*settings)[DataLakeStorageSetting::storage_catalog_type].value == DatabaseDataLakeCatalogType::NONE && !(*settings)[DataLakeStorageSetting::storage_catalog_url].value.empty())) + { + return std::make_shared( + (*settings)[DataLakeStorageSetting::storage_warehouse].value, + (*settings)[DataLakeStorageSetting::storage_catalog_url].value, + (*settings)[DataLakeStorageSetting::storage_catalog_credential].value, + (*settings)[DataLakeStorageSetting::storage_auth_scope].value, + (*settings)[DataLakeStorageSetting::storage_auth_header], + (*settings)[DataLakeStorageSetting::storage_oauth_server_uri].value, + (*settings)[DataLakeStorageSetting::storage_oauth_server_use_request_body].value, + catalog_namespaces, + context); + } + +#endif +>>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) return nullptr; #endif } bool optimize(ObjectStoragePtr object_storage, const StorageMetadataPtr & metadata_snapshot, ContextPtr context, const std::optional & format_settings) override { +<<<<<<< HEAD lazyInitializeIfNeeded(object_storage, context); +======= + assertInitializedDL(); +>>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) return current_metadata->optimize(metadata_snapshot, context, format_settings); } @@ -404,9 +485,48 @@ class DataLakeConfiguration : public BaseStorageConfiguration, public std::enabl #endif } + bool isClusterSupported() const override { return is_cluster_supported; } + + ASTPtr createArgsWithAccessData() const override + { + auto res = BaseStorageConfiguration::createArgsWithAccessData(); + + auto iceberg_metadata_file_path = (*settings)[DataLakeStorageSetting::iceberg_metadata_file_path]; + + if (iceberg_metadata_file_path.changed) + { + auto * arguments = res->template as(); + if (!arguments) + throw Exception(ErrorCodes::LOGICAL_ERROR, "Arguments are not an expression list"); + + bool has_settings = false; + + for (auto & arg : arguments->children) + { + if (auto * settings_ast = arg->template as()) + { + has_settings = true; + settings_ast->changes.setSetting("iceberg_metadata_file_path", iceberg_metadata_file_path.value); + break; + } + } + + if (!has_settings) + { + boost::intrusive_ptr settings_ast = make_intrusive(); + settings_ast->is_standalone = false; + settings_ast->changes.setSetting("iceberg_metadata_file_path", iceberg_metadata_file_path.value); + arguments->children.push_back(settings_ast); + } + } + + return res; + } + private: const DataLakeStorageSettingsPtr settings; ObjectStoragePtr ready_object_storage; + std::string catalog_namespaces; DataLakeMetadataPtr current_metadata; LoggerPtr log = getLogger("DataLakeConfiguration"); @@ -421,8 +541,9 @@ class DataLakeConfiguration : public BaseStorageConfiguration, public std::enabl } } - void assertInitialized() const + void assertInitializedDL() const { + BaseStorageConfiguration::assertInitialized(); if (!current_metadata) throw Exception(ErrorCodes::LOGICAL_ERROR, "Metadata is not initialized"); } @@ -456,18 +577,388 @@ using StorageS3IcebergConfiguration = DataLakeConfiguration; #endif -#if USE_AZURE_BLOB_STORAGE +# if USE_AZURE_BLOB_STORAGE using StorageAzureIcebergConfiguration = DataLakeConfiguration; using StorageAzurePaimonConfiguration = DataLakeConfiguration; #endif -#if USE_HDFS +# if USE_HDFS using StorageHDFSIcebergConfiguration = DataLakeConfiguration; using StorageHDFSPaimonConfiguration = DataLakeConfiguration; #endif using StorageLocalIcebergConfiguration = DataLakeConfiguration; -using StorageLocalPaimonConfiguration = DataLakeConfiguration; +using StorageLocalPaimonConfiguration = DataLakeConfiguration; + +/// Class detects storage type by `storage_type` parameter if exists +/// and uses appropriate implementation - S3, Azure, HDFS or Local +class StorageIcebergConfiguration : public StorageObjectStorageConfiguration, public std::enable_shared_from_this +{ + friend class StorageObjectStorageConfiguration; + +public: + StorageIcebergConfiguration() {} + + explicit StorageIcebergConfiguration(DataLakeStorageSettingsPtr settings_) : settings(settings_) {} + + void initialize( + ASTs & engine_args, + ContextPtr local_context, + bool with_table_structure, + const StorageID * table_id = nullptr) override + { + createDynamicConfiguration(engine_args, local_context); + getImpl().initialize(engine_args, local_context, with_table_structure, table_id); + } + + ObjectStorageType getType() const override { return getImpl().getType(); } + + std::string getTypeName() const override { return getImpl().getTypeName(); } + std::string getEngineName() const override { return getImpl().getEngineName(); } + std::string getNamespaceType() const override { return getImpl().getNamespaceType(); } + + Path getRawPath() const override { return getImpl().getRawPath(); } + void setRawPath(const Path & path) override { getImpl().setRawPath(path); } + const String & getRawURI() const override { return getImpl().getRawURI(); } + const Path & getPathForRead() const override { return getImpl().getPathForRead(); } + Path getPathForWrite(const std::string & partition_id) const override { return getImpl().getPathForWrite(partition_id); } + + void setPathForRead(const Path & path) override { getImpl().setPathForRead(path); } + + const Paths & getPaths() const override { return getImpl().getPaths(); } + void setPaths(const Paths & paths) override { getImpl().setPaths(paths); } + + String getDataSourceDescription() const override { return getImpl().getDataSourceDescription(); } + String getNamespace() const override { return getImpl().getNamespace(); } + + StorageObjectStorageQuerySettings getQuerySettings(const ContextPtr & context) const override + { return getImpl().getQuerySettings(context); } + + void addStructureAndFormatToArgsIfNeeded( + ASTs & args, const String & structure_, const String & format_, ContextPtr context, bool with_structure) override + { getImpl().addStructureAndFormatToArgsIfNeeded(args, structure_, format_, context, with_structure); } + + bool isNamespaceWithGlobs() const override { return getImpl().isNamespaceWithGlobs(); } + + bool isArchive() const override { return getImpl().isArchive(); } + bool isPathInArchiveWithGlobs() const override { return getImpl().isPathInArchiveWithGlobs(); } + std::string getPathInArchive() const override { return getImpl().getPathInArchive(); } + + void check(ContextPtr context) override { getImpl().check(context); } + void validateNamespace(const String & name) const override { getImpl().validateNamespace(name); } + + ObjectStoragePtr createObjectStorage(ContextPtr context, bool is_readonly, CredentialsConfigurationCallback refresh_credentials_callback) override + { return getImpl().createObjectStorage(context, is_readonly, refresh_credentials_callback); } + bool isStaticConfiguration() const override { return getImpl().isStaticConfiguration(); } + + bool isDataLakeConfiguration() const override { return getImpl().isDataLakeConfiguration(); } + + bool supportsTotalRows(ContextPtr context, ObjectStorageType storage_type) const override { return getImpl().supportsTotalRows(context, storage_type); } + std::optional totalRows(ContextPtr context) override { return getImpl().totalRows(context); } + bool supportsTotalBytes(ContextPtr context, ObjectStorageType storage_type) const override { return getImpl().supportsTotalBytes(context, storage_type); } + std::optional totalBytes(ContextPtr context) override { return getImpl().totalBytes(context); } + bool isDataSortedBySortingKey(StorageMetadataPtr storage_metadata, ContextPtr context) const override + { return getImpl().isDataSortedBySortingKey(storage_metadata, context); } + + IDataLakeMetadata * getExternalMetadata() override { return getImpl().getExternalMetadata(); } + + std::shared_ptr getInitialSchemaByPath(ContextPtr context, ObjectInfoPtr object_info) const override + { return getImpl().getInitialSchemaByPath(context, object_info); } + + std::shared_ptr getSchemaTransformer(ContextPtr context, ObjectInfoPtr object_info) const override + { return getImpl().getSchemaTransformer(context, object_info); } + + void modifyFormatSettings(FormatSettings & settings_, const Context & context) const override + { getImpl().modifyFormatSettings(settings_, context); } + + void addDeleteTransformers( + ObjectInfoPtr object_info, + QueryPipelineBuilder & builder, + const std::optional & format_settings, + FormatParserSharedResourcesPtr parser_shared_resources, + ContextPtr local_context) const override + { getImpl().addDeleteTransformers(object_info, builder, format_settings, parser_shared_resources, local_context); } + + ReadFromFormatInfo prepareReadingFromFormat( + ObjectStoragePtr object_storage, + const Strings & requested_columns, + const StorageSnapshotPtr & storage_snapshot, + bool supports_subset_of_columns, + bool supports_tuple_elements, + ContextPtr local_context, + const PrepareReadingFromFormatHiveParams & hive_parameters) override + { + return getImpl().prepareReadingFromFormat( + object_storage, + requested_columns, + storage_snapshot, + supports_subset_of_columns, + supports_tuple_elements, + local_context, + hive_parameters); + } + + void setSchemaHash(const String & hash) override { getImpl().setSchemaHash(hash); } + + void initPartitionStrategy(ASTPtr partition_by, const ColumnsDescription & columns, ContextPtr context) override + { getImpl().initPartitionStrategy(partition_by, columns, context); } + + std::optional getTableStateSnapshot(ContextPtr local_context) const override { return getImpl().getTableStateSnapshot(local_context); } + std::unique_ptr buildStorageMetadataFromState(const DataLakeTableStateSnapshot & state, ContextPtr local_context) const override + { return getImpl().buildStorageMetadataFromState(state, local_context); } + bool shouldReloadSchemaForConsistency(ContextPtr local_context) const override { return getImpl().shouldReloadSchemaForConsistency(local_context); } + std::optional tryGetTableStructureFromMetadata(ContextPtr local_context) const override + { return getImpl().tryGetTableStructureFromMetadata(local_context); } + + bool supportsFileIterator() const override { return getImpl().supportsFileIterator(); } + bool supportsParallelInsert() const override { return getImpl().supportsParallelInsert(); } + bool supportsWrites() const override { return getImpl().supportsWrites(); } + + bool supportsPartialPathPrefix() const override { return getImpl().supportsPartialPathPrefix(); } + + ObjectIterator iterate( + const ActionsDAG * filter_dag, + IDataLakeMetadata::FileProgressCallback callback, + size_t list_batch_size, + StorageMetadataPtr storage_metadata, + ContextPtr context) override + { + return getImpl().iterate(filter_dag, callback, list_batch_size, storage_metadata, context); + } + + void update( + ObjectStoragePtr object_storage_ptr, + ContextPtr context) override + { + getImpl().update(object_storage_ptr, context); + } + void lazyInitializeIfNeeded(ObjectStoragePtr object_storage, ContextPtr local_context) override + { return getImpl().lazyInitializeIfNeeded(object_storage, local_context); } + + void create( + ObjectStoragePtr object_storage, + ContextPtr local_context, + const std::optional & columns, + ASTPtr partition_by, + ASTPtr order_by, + bool if_not_exists, + std::shared_ptr catalog, + const StorageID & table_id_) override + { + getImpl().create(object_storage, local_context, columns, partition_by, order_by, if_not_exists, catalog, table_id_); + } + + SinkToStoragePtr write( + SharedHeader sample_block, + const StorageID & table_id, + ObjectStoragePtr object_storage, + const std::optional & format_settings, + ContextPtr context, + std::shared_ptr catalog) override + { + return getImpl().write(sample_block, table_id, object_storage, format_settings, context, catalog); + } + + bool supportsDelete() const override { return getImpl().supportsDelete(); } + void mutate(const MutationCommands & commands, + ContextPtr context, + const StorageID & storage_id, + StorageMetadataPtr metadata_snapshot, + std::shared_ptr catalog, + const std::optional & format_settings) override + { + getImpl().mutate(commands, context, storage_id, metadata_snapshot, catalog, format_settings); + } + void checkMutationIsPossible(const MutationCommands & commands) override { getImpl().checkMutationIsPossible(commands); } + + void checkAlterIsPossible(const AlterCommands & commands) override { getImpl().checkAlterIsPossible(commands); } + + void alter(const AlterCommands & params, ContextPtr context) override { getImpl().alter(params, context); } + + const DataLakeStorageSettings & getDataLakeSettings() const override { return getImpl().getDataLakeSettings(); } + + ASTPtr createArgsWithAccessData() const override + { + return getImpl().createArgsWithAccessData(); + } + + void fromNamedCollection(const NamedCollection & collection, ContextPtr context) override + { getImpl().fromNamedCollection(collection, context); } + void fromAST(ASTs & args, ContextPtr context, bool with_structure) override + { getImpl().fromAST(args, context, with_structure); } + void fromDisk(const String & disk_name, ASTs & args, ContextPtr context, bool with_structure) override + { getImpl().fromDisk(disk_name, args, context, with_structure); } + + /// Find storage_type argument and remove it from args if exists. + /// Return storage type. + ObjectStorageType extractDynamicStorageType(ASTs & args, ContextPtr context, ASTPtr * type_arg, bool cluster_name_first) const override + { + static const auto * const storage_type_name = "storage_type"; + + { + auto args_copy = args; + if (cluster_name_first) + { + // Remove cluster name from args to avoid confusing cluster name and named collection name + args_copy.erase(args_copy.begin()); + } + + if (auto named_collection = tryGetNamedCollectionWithOverrides(args_copy, context)) + { + if (named_collection->has(storage_type_name)) + { + return objectStorageTypeFromString(named_collection->get(storage_type_name)); + } + } + } + + auto type_it = args.end(); + + /// S3 by default for backward compatibility + /// Iceberg without storage_type == IcebergS3 + ObjectStorageType type = ObjectStorageType::S3; + + for (auto arg_it = args.begin(); arg_it != args.end(); ++arg_it) + { + const auto * type_ast_function = (*arg_it)->as(); + + if (type_ast_function && type_ast_function->name == "equals" + && type_ast_function->arguments && type_ast_function->arguments->children.size() == 2) + { + auto * name = type_ast_function->arguments->children[0]->as(); + + if (name && name->name() == storage_type_name) + { + if (type_it != args.end()) + { + throw Exception( + ErrorCodes::BAD_ARGUMENTS, + "DataLake can have only one key-value argument: storage_type='type'."); + } + + auto * value = type_ast_function->arguments->children[1]->as(); + + if (!value) + { + throw Exception( + ErrorCodes::BAD_ARGUMENTS, + "DataLake parameter 'storage_type' has wrong type, string literal expected."); + } + + if (value->value.getType() != Field::Types::String) + { + throw Exception( + ErrorCodes::BAD_ARGUMENTS, + "DataLake parameter 'storage_type' has wrong value type, string expected."); + } + + type = objectStorageTypeFromString(value->value.safeGet()); + + type_it = arg_it; + } + } + } + + if (type_it != args.end()) + { + if (type_arg) + *type_arg = *type_it; + args.erase(type_it); + } + + return type; + } + + const String & getFormat() const override { return getImpl().getFormat(); } + const String & getCompressionMethod() const override { return getImpl().getCompressionMethod(); } + const String & getStructure() const override { return getImpl().getStructure(); } + + PartitionStrategyFactory::StrategyType getPartitionStrategyType() const override { return getImpl().getPartitionStrategyType(); } + bool getPartitionColumnsInDataFile() const override { return getImpl().getPartitionColumnsInDataFile(); } + std::shared_ptr getPartitionStrategy() const override { return getImpl().getPartitionStrategy(); } + + void setFormat(const String & format_) override { getImpl().setFormat(format_); } + void setCompressionMethod(const String & compression_method_) override { getImpl().setCompressionMethod(compression_method_); } + void setStructure(const String & structure_) override { getImpl().setStructure(structure_); } + + void setPartitionStrategyType(PartitionStrategyFactory::StrategyType partition_strategy_type_) override + { getImpl().setPartitionStrategyType(partition_strategy_type_); } + void setPartitionColumnsInDataFile(bool partition_columns_in_data_file_) override + { getImpl().setPartitionColumnsInDataFile(partition_columns_in_data_file_); } + void setPartitionStrategy(const std::shared_ptr & partition_strategy_) override + { getImpl().setPartitionStrategy(partition_strategy_); } + + void assertInitialized() const override { getImpl().assertInitialized(); } + + ColumnMapperPtr getColumnMapperForObject(ObjectInfoPtr obj) const override { return getImpl().getColumnMapperForObject(obj); } + + ColumnMapperPtr getColumnMapperForCurrentSchema(StorageMetadataPtr storage_metadata_snapshot, ContextPtr context) const override + { return getImpl().getColumnMapperForCurrentSchema(storage_metadata_snapshot, context); } + + std::shared_ptr getCatalog(ContextPtr context, bool is_attach) const override + { return getImpl().getCatalog(context, is_attach); } + + bool optimize(const StorageMetadataPtr & metadata_snapshot, ContextPtr context, const std::optional & format_settings) override + { return getImpl().optimize(metadata_snapshot, context, format_settings); } + + bool supportsPrewhere() const override { return getImpl().supportsPrewhere(); } + + void drop(ContextPtr context) override { getImpl().drop(context); } + +protected: + void createDynamicConfiguration(ASTs & args, ContextPtr context) + { + ObjectStorageType type = extractDynamicStorageType(args, context, nullptr, false); + createDynamicStorage(type); + } + +private: + inline StorageObjectStorageConfiguration & getImpl() const + { + if (!impl) + throw Exception(ErrorCodes::LOGICAL_ERROR, "Dynamic DataLake storage not initialized"); + + return *impl; + } + + void createDynamicStorage(ObjectStorageType type) + { + if (impl) + { + if (impl->getType() == type) + return; + + throw Exception(ErrorCodes::LOGICAL_ERROR, "Can't change datalake engine storage"); + } + + switch (type) + { +# if USE_AWS_S3 + case ObjectStorageType::S3: + impl = std::make_unique(settings); + break; +# endif +# if USE_AZURE_BLOB_STORAGE + case ObjectStorageType::Azure: + impl = std::make_unique(settings); + break; +# endif +# if USE_HDFS + case ObjectStorageType::HDFS: + impl = std::make_unique(settings); + break; +# endif + case ObjectStorageType::Local: + impl = std::make_unique(settings); + break; + default: + throw Exception(ErrorCodes::LOGICAL_ERROR, "Unsuported DataLake storage {}", type); + } + } + + StorageObjectStorageConfigurationPtr impl; + DataLakeStorageSettingsPtr settings; +}; #endif #if USE_PARQUET @@ -479,7 +970,7 @@ using StorageS3DeltaLakeConfiguration = DataLakeConfiguration; #endif -using StorageLocalDeltaLakeConfiguration = DataLakeConfiguration; +using StorageLocalDeltaLakeConfiguration = DataLakeConfiguration; #endif diff --git a/src/Storages/ObjectStorage/DataLakes/DataLakeStorageSettings.h b/src/Storages/ObjectStorage/DataLakes/DataLakeStorageSettings.h index c5a4db758418..cd52fd104cbf 100644 --- a/src/Storages/ObjectStorage/DataLakes/DataLakeStorageSettings.h +++ b/src/Storages/ObjectStorage/DataLakes/DataLakeStorageSettings.h @@ -62,6 +62,9 @@ The period in milliseconds to asynchronously prefetch the latest metadata snapsh )", 0) \ DECLARE(Bool, iceberg_use_version_hint, false, R"( Get latest metadata path from version-hint.text file. +)", 0) \ + DECLARE(String, object_storage_cluster, "", R"( +Cluster for distributed requests )", 0) \ DECLARE(NonZeroUInt64, iceberg_format_version, 2, R"( Metadata format version. diff --git a/src/Storages/ObjectStorage/DataLakes/DeltaLakeMetadataDeltaKernel.cpp b/src/Storages/ObjectStorage/DataLakes/DeltaLakeMetadataDeltaKernel.cpp index 9ef10e9a9198..5278cca5b3f3 100644 --- a/src/Storages/ObjectStorage/DataLakes/DeltaLakeMetadataDeltaKernel.cpp +++ b/src/Storages/ObjectStorage/DataLakes/DeltaLakeMetadataDeltaKernel.cpp @@ -117,7 +117,7 @@ DeltaLakeMetadataDeltaKernel::DeltaLakeMetadataDeltaKernel( : log(getLogger("DeltaLakeMetadata")) , kernel_helper(DB::getKernelHelper(configuration_.lock(), object_storage_)) , object_storage(object_storage_) - , format_name(configuration_.lock()->format) + , format_name(configuration_.lock()->getFormat()) /// TODO: Supports size limit, not just elements limit. /// TODO: Support weight function (by default weight = 1 for all elements). /// TODO: Add a setting for cache size. @@ -643,8 +643,8 @@ SinkToStoragePtr DeltaLakeMetadataDeltaKernel::write( context, sample_block, format_settings, - configuration->format, - configuration->compression_method); + configuration->getFormat(), + configuration->getCompressionMethod()); } return std::make_shared( @@ -654,8 +654,8 @@ SinkToStoragePtr DeltaLakeMetadataDeltaKernel::write( context, sample_block, format_settings, - configuration->format, - configuration->compression_method); + configuration->getFormat(), + configuration->getCompressionMethod()); } void DeltaLakeMetadataDeltaKernel::logMetadataFiles(ContextPtr context) const diff --git a/src/Storages/ObjectStorage/DataLakes/HudiMetadata.cpp b/src/Storages/ObjectStorage/DataLakes/HudiMetadata.cpp index aeb4f9989dd2..c0f527d63d87 100644 --- a/src/Storages/ObjectStorage/DataLakes/HudiMetadata.cpp +++ b/src/Storages/ObjectStorage/DataLakes/HudiMetadata.cpp @@ -92,11 +92,11 @@ HudiMetadata::HudiMetadata(ObjectStoragePtr object_storage_, StorageObjectStorag : WithContext(context_) , object_storage(object_storage_) , table_path(configuration_->getPathForRead().path) - , format(configuration_->format) + , format(configuration_->getFormat()) { } -Strings HudiMetadata::getDataFiles(const ActionsDAG *) const +Strings HudiMetadata::getDataFiles() const { if (data_files.empty()) data_files = getDataFilesImpl(); @@ -104,13 +104,13 @@ Strings HudiMetadata::getDataFiles(const ActionsDAG *) const } ObjectIterator HudiMetadata::iterate( - const ActionsDAG * filter_dag, + const ActionsDAG * /* filter_dag */, FileProgressCallback callback, size_t /* list_batch_size */, StorageMetadataPtr /* storage_metadata_snapshot*/, ContextPtr /* context */) const { - return createKeysIterator(getDataFiles(filter_dag), object_storage, callback); + return createKeysIterator(getDataFiles(), object_storage, callback); } } diff --git a/src/Storages/ObjectStorage/DataLakes/HudiMetadata.h b/src/Storages/ObjectStorage/DataLakes/HudiMetadata.h index d2700f405fc8..b941a84a3747 100644 --- a/src/Storages/ObjectStorage/DataLakes/HudiMetadata.h +++ b/src/Storages/ObjectStorage/DataLakes/HudiMetadata.h @@ -65,7 +65,7 @@ class HudiMetadata final : public IDataLakeMetadata, private WithContext mutable Strings data_files; Strings getDataFilesImpl() const; - Strings getDataFiles(const ActionsDAG * filter_dag) const; + Strings getDataFiles() const; }; } diff --git a/src/Storages/ObjectStorage/DataLakes/Iceberg/ChunkPartitioner.cpp b/src/Storages/ObjectStorage/DataLakes/Iceberg/ChunkPartitioner.cpp index 8d22f09eb4ff..502e31463702 100644 --- a/src/Storages/ObjectStorage/DataLakes/Iceberg/ChunkPartitioner.cpp +++ b/src/Storages/ObjectStorage/DataLakes/Iceberg/ChunkPartitioner.cpp @@ -18,6 +18,7 @@ namespace DB namespace Setting { extern const SettingsUInt64 iceberg_insert_max_partitions; + extern const SettingsTimezone iceberg_partition_timezone; } namespace ErrorCodes @@ -52,7 +53,7 @@ ChunkPartitioner::ChunkPartitioner( auto & factory = FunctionFactory::instance(); - auto transform_and_argument = Iceberg::parseTransformAndArgument(transform_name); + auto transform_and_argument = Iceberg::parseTransformAndArgument(transform_name, context->getSettingsRef()[Setting::iceberg_partition_timezone]); if (!transform_and_argument) throw Exception(ErrorCodes::BAD_ARGUMENTS, "Unknown transform {}", transform_name); @@ -66,6 +67,7 @@ ChunkPartitioner::ChunkPartitioner( result_data_types.push_back(function->getReturnType(columns_for_function)); functions.push_back(function); function_params.push_back(transform_and_argument->argument); + function_time_zones.push_back(transform_and_argument->time_zone); columns_to_apply.push_back(column_name); } } @@ -107,6 +109,14 @@ ChunkPartitioner::partitionChunk(const Chunk & chunk) arguments.push_back(ColumnWithTypeAndName(const_column->clone(), type, "#")); } arguments.push_back(name_to_column[columns_to_apply[transform_ind]]); + if (function_time_zones[transform_ind].has_value()) + { + auto type = std::make_shared(); + auto column_value = ColumnString::create(); + column_value->insert(*function_time_zones[transform_ind]); + auto const_column = ColumnConst::create(std::move(column_value), chunk.getNumRows()); + arguments.push_back(ColumnWithTypeAndName(const_column->clone(), type, "PartitioningTimezone")); + } auto result = functions[transform_ind]->build(arguments)->execute(arguments, result_data_types[transform_ind], chunk.getNumRows(), false); functions_columns.push_back(result); diff --git a/src/Storages/ObjectStorage/DataLakes/Iceberg/ChunkPartitioner.h b/src/Storages/ObjectStorage/DataLakes/Iceberg/ChunkPartitioner.h index fa386b2de0cb..7395ed0c8b9c 100644 --- a/src/Storages/ObjectStorage/DataLakes/Iceberg/ChunkPartitioner.h +++ b/src/Storages/ObjectStorage/DataLakes/Iceberg/ChunkPartitioner.h @@ -43,6 +43,7 @@ class ChunkPartitioner std::vector functions; std::vector> function_params; + std::vector> function_time_zones; std::vector columns_to_apply; std::vector result_data_types; diff --git a/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergMetadata.cpp b/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergMetadata.cpp index 6292aa10bd49..277bc3b4c41e 100644 --- a/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergMetadata.cpp +++ b/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergMetadata.cpp @@ -242,14 +242,22 @@ Iceberg::PersistentTableComponents IcebergMetadata::initializePersistentTableCom } auto table_path = configuration->getPathForRead().path; return PersistentTableComponents{ +<<<<<<< HEAD .schema_processor = std::make_shared(context_->getSettingsRef()[Setting::allow_experimental_geo_types_in_iceberg]), +======= + .schema_processor = std::make_shared(context_), +>>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) .metadata_cache = cache_ptr, .format_version = format_version, .table_location = table_location, .metadata_compression_method = compression_method, .table_path = table_path, .table_uuid = table_uuid, +<<<<<<< HEAD .path_resolver = IcebergPathResolver(table_location, table_path, configuration->getTypeName(), configuration->getNamespace()), +======= + .common_namespace = configuration->getNamespace(), +>>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) }; } @@ -277,7 +285,7 @@ IcebergMetadata::IcebergMetadata( , object_storage(std::move(object_storage_)) , persistent_components(std::move(persistent_components_)) , data_lake_settings(configuration_->getDataLakeSettings()) - , write_format(configuration_->format) + , write_format(configuration_->getFormat()) { /// TODO: for now it's okay to start/stop the task via constructor/destructor. Once refactored, we'd need to plumb startup/shutdown and schedule the task from there if (persistent_components.metadata_cache && data_lake_settings[DataLakeStorageSetting::iceberg_metadata_async_prefetch_period_ms] != 0) @@ -354,6 +362,10 @@ void IcebergMetadata::backgroundMetadataPrefetcherThread() Int32 IcebergMetadata::parseTableSchema( const Poco::JSON::Object::Ptr & metadata_object, IcebergSchemaProcessor & schema_processor, +<<<<<<< HEAD +======= + ContextPtr context_, +>>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) LoggerPtr metadata_logger) { const auto format_version = metadata_object->getValue(f_format_version); @@ -361,7 +373,7 @@ Int32 IcebergMetadata::parseTableSchema( if (format_version == 2) { auto [schema, current_schema_id] = parseTableSchemaV2Method(metadata_object); - schema_processor.addIcebergTableSchema(schema); + schema_processor.addIcebergTableSchema(schema, context_); return current_schema_id; } else @@ -369,7 +381,7 @@ Int32 IcebergMetadata::parseTableSchema( try { auto [schema, current_schema_id] = parseTableSchemaV1Method(metadata_object); - schema_processor.addIcebergTableSchema(schema); + schema_processor.addIcebergTableSchema(schema, context_); return current_schema_id; } catch (const Exception & first_error) @@ -379,7 +391,7 @@ Int32 IcebergMetadata::parseTableSchema( try { auto [schema, current_schema_id] = parseTableSchemaV2Method(metadata_object); - schema_processor.addIcebergTableSchema(schema); + schema_processor.addIcebergTableSchema(schema, context_); LOG_WARNING( metadata_logger, "Iceberg table schema was parsed using v2 specification, but it was impossible to parse it using v1 " @@ -402,8 +414,16 @@ Int32 IcebergMetadata::parseTableSchema( } } +<<<<<<< HEAD static Poco::JSON::Object::Ptr traverseMetadataAndFindNecessarySnapshotObject( Poco::JSON::Object::Ptr metadata_object, Int64 snapshot_id, IcebergSchemaProcessorPtr schema_processor) +======= +Poco::JSON::Object::Ptr traverseMetadataAndFindNecessarySnapshotObject( + Poco::JSON::Object::Ptr metadata_object, + Int64 snapshot_id, + IcebergSchemaProcessorPtr schema_processor, + ContextPtr local_context) +>>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) { if (!metadata_object->has(f_snapshots)) throw Exception(ErrorCodes::ICEBERG_SPECIFICATION_VIOLATION, "No snapshot set found in metadata for iceberg file"); @@ -411,7 +431,7 @@ static Poco::JSON::Object::Ptr traverseMetadataAndFindNecessarySnapshotObject( for (UInt32 j = 0; j < schemas->size(); ++j) { auto schema = schemas->getObject(j); - schema_processor->addIcebergTableSchema(schema); + schema_processor->addIcebergTableSchema(schema, local_context); } Poco::JSON::Object::Ptr current_snapshot = nullptr; auto snapshots = metadata_object->get(f_snapshots).extract(); @@ -474,7 +494,11 @@ IcebergDataSnapshotPtr IcebergMetadata::createIcebergDataSnapshotFromSnapshotJSO IcebergDataSnapshotPtr IcebergMetadata::getIcebergDataSnapshot(Poco::JSON::Object::Ptr metadata_object, Int64 snapshot_id, ContextPtr local_context) const { - auto object = traverseMetadataAndFindNecessarySnapshotObject(metadata_object, snapshot_id, persistent_components.schema_processor); + auto object = traverseMetadataAndFindNecessarySnapshotObject( + metadata_object, + snapshot_id, + persistent_components.schema_processor, + local_context); if (!object) throw Exception(ErrorCodes::ICEBERG_SPECIFICATION_VIOLATION, "No snapshot found for id `{}`", snapshot_id); @@ -559,7 +583,7 @@ IcebergMetadata::getStateImpl(const ContextPtr & local_context, Poco::JSON::Obje } else { - auto schema_id = parseTableSchema(metadata_object, *persistent_components.schema_processor, log); + auto schema_id = parseTableSchema(metadata_object, *persistent_components.schema_processor, local_context, log); if (!metadata_object->has(f_current_snapshot_id)) { return {nullptr, schema_id}; @@ -584,9 +608,10 @@ IcebergMetadata::getState(const ContextPtr & local_context, const String & metad auto metadata_object = getMetadataJSONObject( metadata_path, object_storage, persistent_components.metadata_cache, local_context, log, persistent_components.metadata_compression_method, persistent_components.table_uuid); + auto dump_metadata = [&]()->String { return dumpMetadataObjectToString(metadata_object); }; insertRowToLogTable( local_context, - dumpMetadataObjectToString(metadata_object), + dump_metadata, DB::IcebergMetadataLogLevel::Metadata, persistent_components.path_resolver.getTableRoot(), Iceberg::IcebergPathFromMetadata::deserialize(metadata_path), @@ -617,14 +642,16 @@ std::shared_ptr IcebergMetadata::getInitialSchemaByPath(Conte : nullptr; } -std::shared_ptr IcebergMetadata::getSchemaTransformer(ContextPtr, ObjectInfoPtr object_info) const +std::shared_ptr IcebergMetadata::getSchemaTransformer(ContextPtr context_, ObjectInfoPtr object_info) const { IcebergDataObjectInfo * iceberg_object_info = dynamic_cast(object_info.get()); if (!iceberg_object_info) return nullptr; return (iceberg_object_info->info.underlying_format_read_schema_id != iceberg_object_info->info.schema_id_relevant_to_iterator) ? persistent_components.schema_processor->getSchemaTransformationDagByIds( - iceberg_object_info->info.underlying_format_read_schema_id, iceberg_object_info->info.schema_id_relevant_to_iterator) + context_, + iceberg_object_info->info.underlying_format_read_schema_id, + iceberg_object_info->info.schema_id_relevant_to_iterator) : nullptr; } @@ -848,7 +875,7 @@ Iceberg::IcebergDataSnapshotPtr IcebergMetadata::getRelevantDataSnapshotFromTabl if (!table_state_snapshot.snapshot_id.has_value()) return nullptr; Poco::JSON::Object::Ptr snapshot_object = traverseMetadataAndFindNecessarySnapshotObject( - metadata_object, *table_state_snapshot.snapshot_id, persistent_components.schema_processor); + metadata_object, *table_state_snapshot.snapshot_id, persistent_components.schema_processor, local_context); return createIcebergDataSnapshotFromSnapshotJSON(snapshot_object, *table_state_snapshot.snapshot_id, local_context); } @@ -1548,7 +1575,7 @@ bool IcebergMetadata::commitImportPartitionTransactionImpl( { const auto & [namespace_name, table_name] = DataLake::parseTableName(table_id.getTableName()); DataLake::TableMetadata table_metadata = DataLake::TableMetadata().withLocation().withDataLakeSpecificProperties(); - catalog->getTableMetadata(namespace_name, table_name, table_metadata); + catalog->getTableMetadata(namespace_name, table_name, context, table_metadata); auto table_specific_properties = table_metadata.getDataLakeSpecificProperties(); if (!table_specific_properties.has_value() || table_specific_properties->iceberg_metadata_file_location.empty()) diff --git a/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergMetadata.h b/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergMetadata.h index 07e002ae42c9..f8b7f503799c 100644 --- a/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergMetadata.h +++ b/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergMetadata.h @@ -102,6 +102,10 @@ class IcebergMetadata : public IDataLakeMetadata static Int32 parseTableSchema( const Poco::JSON::Object::Ptr & metadata_object, Iceberg::IcebergSchemaProcessor & schema_processor, +<<<<<<< HEAD +======= + ContextPtr context_, +>>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) LoggerPtr metadata_logger); bool supportsUpdate() const override { return true; } diff --git a/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergWrites.cpp b/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergWrites.cpp index 86420c029fbc..c1f9269b3106 100644 --- a/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergWrites.cpp +++ b/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergWrites.cpp @@ -877,7 +877,13 @@ IcebergStorageSink::IcebergStorageSink( , table_id(table_id_) , persistent_table_components(persistent_table_components_) , data_lake_settings(configuration_->getDataLakeSettings()) +<<<<<<< HEAD , write_format(configuration_->format) +======= + , write_format(configuration_->getFormat()) + , blob_storage_type_name(configuration_->getTypeName()) + , blob_storage_namespace_name(configuration_->getNamespace()) +>>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) { auto [last_version, metadata_path, compression_method] = getLatestOrExplicitMetadataFileAndVersion( object_storage, diff --git a/src/Storages/ObjectStorage/DataLakes/Iceberg/ManifestFileIterator.cpp b/src/Storages/ObjectStorage/DataLakes/Iceberg/ManifestFileIterator.cpp index d21cf0de238c..c6b0e106197f 100644 --- a/src/Storages/ObjectStorage/DataLakes/Iceberg/ManifestFileIterator.cpp +++ b/src/Storages/ObjectStorage/DataLakes/Iceberg/ManifestFileIterator.cpp @@ -6,6 +6,7 @@ #include #include +#include #include #include @@ -14,6 +15,7 @@ #include #include +#include #include #include #include @@ -34,6 +36,11 @@ namespace DB::ErrorCodes extern const int BAD_ARGUMENTS; } +namespace DB::Setting +{ + extern const SettingsTimezone iceberg_partition_timezone; +} + namespace ProfileEvents { extern const Event IcebergPartitionPrunedFiles; @@ -232,9 +239,10 @@ std::shared_ptr ManifestFileIterator::create( std::shared_ptr filter_dag_, Int32 table_snapshot_schema_id_) { + auto dump_metadata = [&]()->String { return manifest_file_deserializer_->getMetadataContent(); }; insertRowToLogTable( context_, - manifest_file_deserializer_->getMetadataContent(), + dump_metadata, DB::IcebergMetadataLogLevel::ManifestFileMetadata, path_resolver_.getTableRoot(), path_to_manifest_file_, @@ -284,7 +292,7 @@ std::shared_ptr ManifestFileIterator::create( const Poco::JSON::Object::Ptr & schema_object = json.extract(); Int32 manifest_schema_id = schema_object->getValue(f_schema_id); - schema_processor.addIcebergTableSchema(schema_object); + schema_processor.addIcebergTableSchema(schema_object, context_); PartitionSpecification partition_spec_vec; for (size_t i = 0; i != partition_specification->size(); ++i) @@ -302,7 +310,7 @@ std::shared_ptr ManifestFileIterator::create( auto transform_name = partition_specification_field->getValue(f_partition_transform); auto partition_name = partition_specification_field->getValue(f_partition_name); partition_spec_vec.emplace_back(source_id, transform_name, partition_name); - auto partition_ast = getASTFromTransform(transform_name, numeric_column_name); + auto partition_ast = getASTFromTransform(transform_name, numeric_column_name, context_->getSettingsRef()[Setting::iceberg_partition_timezone]); /// Unsupported partition key expression if (partition_ast == nullptr) continue; @@ -383,9 +391,10 @@ ProcessedManifestFileEntryPtr ManifestFileIterator::processRow(size_t row_index) if (parsed_entry->status == ManifestEntryStatus::DELETED) { + auto dump_metadata = [&]()->String { return manifest_file_deserializer->getContent(row_index); }; insertRowToLogTable( context, - manifest_file_deserializer->getContent(row_index), + dump_metadata, DB::IcebergMetadataLogLevel::ManifestFileEntry, path_resolver.getTableRoot(), path_to_manifest_file, @@ -496,9 +505,10 @@ ProcessedManifestFileEntryPtr ManifestFileIterator::processRow(size_t row_index) const ManifestFilesPruner * current_pruner = getOrCreatePruner(entry->resolved_schema_id); pruning_status = current_pruner->canBePruned(entry, hyperrectangles); } + auto dump_metadata = [&]()->String { return manifest_file_deserializer->getContent(row_index); }; insertRowToLogTable( context, - manifest_file_deserializer->getContent(row_index), + dump_metadata, DB::IcebergMetadataLogLevel::ManifestFileEntry, path_resolver.getTableRoot(), path_to_manifest_file, diff --git a/src/Storages/ObjectStorage/DataLakes/Iceberg/ManifestFilesPruning.cpp b/src/Storages/ObjectStorage/DataLakes/Iceberg/ManifestFilesPruning.cpp index dc4804de1ab5..0a02fd75c04f 100644 --- a/src/Storages/ObjectStorage/DataLakes/Iceberg/ManifestFilesPruning.cpp +++ b/src/Storages/ObjectStorage/DataLakes/Iceberg/ManifestFilesPruning.cpp @@ -27,9 +27,9 @@ using namespace DB; namespace DB::Iceberg { -DB::ASTPtr getASTFromTransform(const String & transform_name_src, const String & column_name) +DB::ASTPtr getASTFromTransform(const String & transform_name_src, const String & column_name, const String & time_zone) { - auto transform_and_argument = parseTransformAndArgument(transform_name_src); + auto transform_and_argument = parseTransformAndArgument(transform_name_src, time_zone); if (!transform_and_argument) { LOG_WARNING(&Poco::Logger::get("Iceberg Partition Pruning"), "Cannot parse iceberg transform name: {}.", transform_name_src); @@ -48,6 +48,13 @@ DB::ASTPtr getASTFromTransform(const String & transform_name_src, const String & return makeASTFunction( transform_and_argument->transform_name, make_intrusive(*transform_and_argument->argument), make_intrusive(column_name)); } + if (transform_and_argument->time_zone) + { + return makeASTFunction( + transform_and_argument->transform_name, + make_intrusive(column_name), + make_intrusive(*transform_and_argument->time_zone)); + } return makeASTFunction(transform_and_argument->transform_name, make_intrusive(column_name)); } diff --git a/src/Storages/ObjectStorage/DataLakes/Iceberg/ManifestFilesPruning.h b/src/Storages/ObjectStorage/DataLakes/Iceberg/ManifestFilesPruning.h index 78c136167a88..d04ced3796ee 100644 --- a/src/Storages/ObjectStorage/DataLakes/Iceberg/ManifestFilesPruning.h +++ b/src/Storages/ObjectStorage/DataLakes/Iceberg/ManifestFilesPruning.h @@ -30,7 +30,7 @@ namespace DB::Iceberg struct ProcessedManifestFileEntry; class ManifestFileIterator; -DB::ASTPtr getASTFromTransform(const String & transform_name_src, const String & column_name); +DB::ASTPtr getASTFromTransform(const String & transform_name_src, const String & column_name, const String & time_zone); /// Prune specific data files based on manifest content class ManifestFilesPruner diff --git a/src/Storages/ObjectStorage/DataLakes/Iceberg/PersistentTableComponents.h b/src/Storages/ObjectStorage/DataLakes/Iceberg/PersistentTableComponents.h index 11d1b74eaab1..91f7dc19c7d0 100644 --- a/src/Storages/ObjectStorage/DataLakes/Iceberg/PersistentTableComponents.h +++ b/src/Storages/ObjectStorage/DataLakes/Iceberg/PersistentTableComponents.h @@ -24,6 +24,7 @@ struct PersistentTableComponents const CompressionMethod metadata_compression_method; const String table_path; const std::optional table_uuid; +<<<<<<< HEAD const IcebergPathResolver path_resolver; /// Invalidate cached metadata for this table under both keys we may have used to cache it @@ -36,6 +37,9 @@ struct PersistentTableComponents if (table_uuid.has_value()) metadata_cache->remove(*table_uuid); } +======= + const String common_namespace; +>>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) }; } diff --git a/src/Storages/ObjectStorage/DataLakes/Iceberg/SchemaProcessor.cpp b/src/Storages/ObjectStorage/DataLakes/Iceberg/SchemaProcessor.cpp index 8c6fd9ee22a2..6975efa85cbd 100644 --- a/src/Storages/ObjectStorage/DataLakes/Iceberg/SchemaProcessor.cpp +++ b/src/Storages/ObjectStorage/DataLakes/Iceberg/SchemaProcessor.cpp @@ -18,6 +18,7 @@ #include #include #include +#include #include #include @@ -33,6 +34,8 @@ #include #include #include +#include +#include #include @@ -48,6 +51,10 @@ extern const int BAD_ARGUMENTS; extern const int ICEBERG_SPECIFICATION_VIOLATION; } +namespace Setting +{ +extern const SettingsTimezone iceberg_timezone_for_timestamptz; +} namespace { @@ -205,7 +212,7 @@ namespace Iceberg std::string IcebergSchemaProcessor::default_link{}; -void IcebergSchemaProcessor::addIcebergTableSchema(Poco::JSON::Object::Ptr schema_ptr) +void IcebergSchemaProcessor::addIcebergTableSchema(Poco::JSON::Object::Ptr schema_ptr, ContextPtr context_) { std::lock_guard lock(mutex); @@ -239,7 +246,7 @@ void IcebergSchemaProcessor::addIcebergTableSchema(Poco::JSON::Object::Ptr schem auto name = field->getValue(f_name); bool required = field->getValue(f_required); current_full_name = name; - auto type = getFieldType(field, f_type, required, current_full_name, true); + auto type = getFieldType(field, f_type, context_, required, current_full_name, true); clickhouse_schema->push_back(NameAndTypePair{name, type}); clickhouse_types_by_source_ids[{schema_id, field->getValue(f_id)}] = NameAndTypePair{current_full_name, type}; clickhouse_ids_by_source_names[{schema_id, current_full_name}] = field->getValue(f_id); @@ -293,7 +300,11 @@ NamesAndTypesList IcebergSchemaProcessor::tryGetFieldsCharacteristics(Int32 sche return fields; } +<<<<<<< HEAD DataTypePtr IcebergSchemaProcessor::getSimpleType(const String & type_name, bool allow_geo_parser) +======= +DataTypePtr IcebergSchemaProcessor::getSimpleType(const String & type_name, ContextPtr context_) +>>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) { if (type_name == f_boolean) return DataTypeFactory::instance().get("Bool"); @@ -312,11 +323,18 @@ DataTypePtr IcebergSchemaProcessor::getSimpleType(const String & type_name, bool if (type_name == f_timestamp) return std::make_shared(6); if (type_name == f_timestamptz) +<<<<<<< HEAD return std::make_shared(6, "UTC"); if (type_name == f_timestamp_ns) return std::make_shared(9); if (type_name == f_timestamptz_ns) return std::make_shared(9, "UTC"); +======= + { + std::string timezone = context_->getSettingsRef()[Setting::iceberg_timezone_for_timestamptz]; + return std::make_shared(6, timezone); + } +>>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) if (type_name == f_string || type_name == f_binary) return std::make_shared(); @@ -350,21 +368,25 @@ DataTypePtr IcebergSchemaProcessor::getSimpleType(const String & type_name, bool } DataTypePtr -IcebergSchemaProcessor::getComplexTypeFromObject(const Poco::JSON::Object::Ptr & type, String & current_full_name, bool is_subfield_of_root) +IcebergSchemaProcessor::getComplexTypeFromObject( + const Poco::JSON::Object::Ptr & type, + String & current_full_name, + ContextPtr context_, + bool is_subfield_of_root) { String type_name = type->getValue(f_type); if (type_name == f_list) { bool element_required = type->getValue("element-required"); - auto element_type = getFieldType(type, f_element, element_required); + auto element_type = getFieldType(type, f_element, context_, element_required); return std::make_shared(element_type); } if (type_name == f_map) { - auto key_type = getFieldType(type, f_key, true); + auto key_type = getFieldType(type, f_key, context_, true); auto value_required = type->getValue("value-required"); - auto value_type = getFieldType(type, f_value, value_required); + auto value_type = getFieldType(type, f_value, context_, value_required); return std::make_shared(key_type, value_type); } @@ -388,7 +410,7 @@ IcebergSchemaProcessor::getComplexTypeFromObject(const Poco::JSON::Object::Ptr & (current_full_name += ".").append(element_names.back()); scope_guard guard([&] { current_full_name.resize(current_full_name.size() - element_names.back().size() - 1); }); - element_types.push_back(getFieldType(field, f_type, required, current_full_name, true)); + element_types.push_back(getFieldType(field, f_type, context_, required, current_full_name, true)); TSA_SUPPRESS_WARNING_FOR_WRITE(clickhouse_types_by_source_ids) [{schema_id, field->getValue(f_id)}] = NameAndTypePair{current_full_name, element_types.back()}; @@ -397,7 +419,7 @@ IcebergSchemaProcessor::getComplexTypeFromObject(const Poco::JSON::Object::Ptr & } else { - element_types.push_back(getFieldType(field, f_type, required)); + element_types.push_back(getFieldType(field, f_type, context_, required)); } } @@ -408,17 +430,27 @@ IcebergSchemaProcessor::getComplexTypeFromObject(const Poco::JSON::Object::Ptr & } DataTypePtr IcebergSchemaProcessor::getFieldType( - const Poco::JSON::Object::Ptr & field, const String & type_key, bool required, String & current_full_name, bool is_subfield_of_root) + const Poco::JSON::Object::Ptr & field, + const String & type_key, + ContextPtr context_, + bool required, + String & current_full_name, + bool is_subfield_of_root) { if (field->isObject(type_key)) - return getComplexTypeFromObject(field->getObject(type_key), current_full_name, is_subfield_of_root); + return getComplexTypeFromObject(field->getObject(type_key), current_full_name, context_, is_subfield_of_root); auto type = field->get(type_key); if (type.isString()) { const String & type_name = type.extract(); +<<<<<<< HEAD auto data_type = getSimpleType(type_name, allow_geo_parser); return required || !data_type->canBeInsideNullable() ? data_type : makeNullable(data_type); +======= + auto data_type = getSimpleType(type_name, context_); + return required ? data_type : makeNullable(data_type); +>>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) } throw Exception(ErrorCodes::BAD_ARGUMENTS, "Unexpected 'type' field: {}", type.toString()); @@ -447,7 +479,11 @@ bool IcebergSchemaProcessor::allowPrimitiveTypeConversion(const String & old_typ // Ids are passed only for error logging purposes std::shared_ptr IcebergSchemaProcessor::getSchemaTransformationDag( - const Poco::JSON::Object::Ptr & old_schema, const Poco::JSON::Object::Ptr & new_schema, Int32 old_id, Int32 new_id) + const Poco::JSON::Object::Ptr & old_schema, + const Poco::JSON::Object::Ptr & new_schema, + ContextPtr context_, + Int32 old_id, + Int32 new_id) { std::unordered_map> old_schema_entries; auto old_schema_fields = old_schema->get(f_fields).extract(); @@ -459,7 +495,7 @@ std::shared_ptr IcebergSchemaProcessor::getSchemaTransformationDag( size_t id = field->getValue(f_id); auto name = field->getValue(f_name); bool required = field->getValue(f_required); - old_schema_entries[id] = {field, &dag->addInput(name, getFieldType(field, f_type, required))}; + old_schema_entries[id] = {field, &dag->addInput(name, getFieldType(field, f_type, context_, required))}; } auto new_schema_fields = new_schema->get(f_fields).extract(); for (size_t i = 0; i != new_schema_fields->size(); ++i) @@ -468,7 +504,7 @@ std::shared_ptr IcebergSchemaProcessor::getSchemaTransformationDag( size_t id = field->getValue(f_id); auto name = field->getValue(f_name); bool required = field->getValue(f_required); - auto type = getFieldType(field, f_type, required); + auto type = getFieldType(field, f_type, context_, required); auto old_node_it = old_schema_entries.find(id); if (old_node_it != old_schema_entries.end()) { @@ -478,7 +514,7 @@ std::shared_ptr IcebergSchemaProcessor::getSchemaTransformationDag( || field->getObject(f_type)->getValue(f_type) == "list" || field->getObject(f_type)->getValue(f_type) == "map")) { - auto old_type = getFieldType(old_json, "type", required); + auto old_type = getFieldType(old_json, "type", context_, required); auto transform = std::make_shared(std::vector{type}, std::vector{old_type}, old_json, field); old_node = &dag->addFunction(transform, std::vector{old_node}, name); @@ -508,7 +544,7 @@ std::shared_ptr IcebergSchemaProcessor::getSchemaTransformationDag( } else if (allowPrimitiveTypeConversion(old_type, new_type)) { - node = &dag->addCast(*old_node, getFieldType(field, f_type, required), name, nullptr); + node = &dag->addCast(*old_node, getFieldType(field, f_type, context_, required), name, nullptr); } outputs.push_back(node); } @@ -534,7 +570,10 @@ std::shared_ptr IcebergSchemaProcessor::getSchemaTransformationDag( return dag; } -std::shared_ptr IcebergSchemaProcessor::getSchemaTransformationDagByIds(Int32 old_id, Int32 new_id) +std::shared_ptr IcebergSchemaProcessor::getSchemaTransformationDagByIds( + ContextPtr context_, + Int32 old_id, + Int32 new_id) { if (old_id == new_id) return nullptr; @@ -553,7 +592,7 @@ std::shared_ptr IcebergSchemaProcessor::getSchemaTransformatio throw Exception(ErrorCodes::BAD_ARGUMENTS, "Schema with schema-id {} is unknown", new_id); return transform_dags_by_ids[{old_id, new_id}] - = getSchemaTransformationDag(old_schema_it->second, new_schema_it->second, old_id, new_id); + = getSchemaTransformationDag(old_schema_it->second, new_schema_it->second, context_, old_id, new_id); } Poco::JSON::Object::Ptr IcebergSchemaProcessor::getIcebergTableSchemaById(Int32 id) const diff --git a/src/Storages/ObjectStorage/DataLakes/Iceberg/SchemaProcessor.h b/src/Storages/ObjectStorage/DataLakes/Iceberg/SchemaProcessor.h index 2c58f25c4fb9..bff81414d642 100644 --- a/src/Storages/ObjectStorage/DataLakes/Iceberg/SchemaProcessor.h +++ b/src/Storages/ObjectStorage/DataLakes/Iceberg/SchemaProcessor.h @@ -75,18 +75,24 @@ ColumnMapperPtr createColumnMapper(Poco::JSON::Object::Ptr schema_object); * } * } */ -class IcebergSchemaProcessor +class IcebergSchemaProcessor : private WithContext { static std::string default_link; using Node = ActionsDAG::Node; public: +<<<<<<< HEAD explicit IcebergSchemaProcessor(bool allow_geo_parser_ = false) : allow_geo_parser(allow_geo_parser_) {} void addIcebergTableSchema(Poco::JSON::Object::Ptr schema_ptr); +======= + explicit IcebergSchemaProcessor(ContextPtr context_) : WithContext(context_) {} + + void addIcebergTableSchema(Poco::JSON::Object::Ptr schema_ptr, ContextPtr context_); +>>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) std::shared_ptr getClickhouseTableSchemaById(Int32 id); - std::shared_ptr getSchemaTransformationDagByIds(Int32 old_id, Int32 new_id); + std::shared_ptr getSchemaTransformationDagByIds(ContextPtr context_, Int32 old_id, Int32 new_id); NameAndTypePair getFieldCharacteristics(Int32 schema_version, Int32 source_id) const; std::optional tryGetFieldCharacteristics(Int32 schema_version, Int32 source_id) const; NamesAndTypesList tryGetFieldsCharacteristics(Int32 schema_id, const std::vector & source_ids) const; @@ -94,7 +100,11 @@ class IcebergSchemaProcessor Poco::JSON::Object::Ptr getIcebergTableSchemaById(Int32 id) const; bool hasClickhouseTableSchemaById(Int32 id) const; +<<<<<<< HEAD static DataTypePtr getSimpleType(const String & type_name, bool allow_geo_parser = true); +======= + static DataTypePtr getSimpleType(const String & type_name, ContextPtr context_); +>>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) static std::unordered_map traverseSchema(Poco::JSON::Array::Ptr schema); @@ -114,10 +124,15 @@ class IcebergSchemaProcessor std::unordered_map schema_id_by_snapshot TSA_GUARDED_BY(mutex); NamesAndTypesList getSchemaType(const Poco::JSON::Object::Ptr & schema); - DataTypePtr getComplexTypeFromObject(const Poco::JSON::Object::Ptr & type, String & current_full_name, bool is_subfield_of_root); + DataTypePtr getComplexTypeFromObject( + const Poco::JSON::Object::Ptr & type, + String & current_full_name, + ContextPtr context_, + bool is_subfield_of_root); DataTypePtr getFieldType( const Poco::JSON::Object::Ptr & field, const String & type_key, + ContextPtr context_, bool required, String & current_full_name = default_link, bool is_subfield_of_root = false); @@ -126,7 +141,11 @@ class IcebergSchemaProcessor const Node * getDefaultNodeForField(const Poco::JSON::Object::Ptr & field); std::shared_ptr getSchemaTransformationDag( - const Poco::JSON::Object::Ptr & old_schema, const Poco::JSON::Object::Ptr & new_schema, Int32 old_id, Int32 new_id); + const Poco::JSON::Object::Ptr & old_schema, + const Poco::JSON::Object::Ptr & new_schema, + ContextPtr context_, + Int32 old_id, + Int32 new_id); mutable SharedMutex mutex; bool allow_geo_parser = true; diff --git a/src/Storages/ObjectStorage/DataLakes/Iceberg/StatelessMetadataFileGetter.cpp b/src/Storages/ObjectStorage/DataLakes/Iceberg/StatelessMetadataFileGetter.cpp index 045470229dc9..feeee16e53b8 100644 --- a/src/Storages/ObjectStorage/DataLakes/Iceberg/StatelessMetadataFileGetter.cpp +++ b/src/Storages/ObjectStorage/DataLakes/Iceberg/StatelessMetadataFileGetter.cpp @@ -168,9 +168,10 @@ ManifestFileCacheKeys getManifestList( ManifestFileCacheKeys manifest_file_cache_keys; + auto dump_metadata = [&]()->String { return manifest_list_deserializer.getMetadataContent(); }; insertRowToLogTable( local_context, - manifest_list_deserializer.getMetadataContent(), + dump_metadata, DB::IcebergMetadataLogLevel::ManifestListMetadata, persistent_table_components.path_resolver.getTableRoot(), filename, @@ -211,9 +212,10 @@ ManifestFileCacheKeys getManifestList( manifest_file_cache_keys.emplace_back( manifest_file_name, manifest_length, added_sequence_number, added_snapshot_id.safeGet(), content_type); + auto dump_row_metadata = [&]()->String { return manifest_list_deserializer.getContent(i); }; insertRowToLogTable( local_context, - manifest_list_deserializer.getContent(i), + dump_row_metadata, DB::IcebergMetadataLogLevel::ManifestListEntry, persistent_table_components.path_resolver.getTableRoot(), filename, diff --git a/src/Storages/ObjectStorage/DataLakes/Iceberg/Utils.cpp b/src/Storages/ObjectStorage/DataLakes/Iceberg/Utils.cpp index 72a73c026b70..7d83abf630f8 100644 --- a/src/Storages/ObjectStorage/DataLakes/Iceberg/Utils.cpp +++ b/src/Storages/ObjectStorage/DataLakes/Iceberg/Utils.cpp @@ -57,6 +57,7 @@ #include #include +#include using namespace DB; @@ -88,12 +89,15 @@ extern const SettingsString iceberg_metadata_compression_method; namespace ProfileEvents { extern const Event IcebergVersionHintUsed; + extern const Event IcebergJsonFileParsing; + extern const Event IcebergJsonFileParsingMicroseconds; } namespace DB::Setting { extern const SettingsUInt64 iceberg_metadata_staleness_ms; extern const SettingsUInt64 output_format_compression_level; + extern const SettingsTimezone iceberg_partition_timezone; } /// Hard to imagine a hint file larger than 10 MB @@ -173,6 +177,7 @@ Iceberg::MetadataFileWithInfo getMetadataFileAndVersion(const std::string & path path); } String version_str; +<<<<<<< HEAD /// `vN.metadata.json` or `vN-.metadata.json` (the latter is what /// `apache/iceberg-rest-fixture` and other iceberg-java REST catalogs /// write when committing a new metadata file). @@ -186,6 +191,12 @@ Iceberg::MetadataFileWithInfo getMetadataFileAndVersion(const std::string & path file_name); version_str = String(file_name.begin() + 1, file_name.begin() + end_pos); } +======= + /// v.metadata.json + /// v-.metadata.json - generated by FileNamesGenerator::generateMetadataName with use_uuid_in_metadata flag + if (file_name.starts_with('v')) + version_str = String(file_name.begin() + 1, file_name.begin() + file_name.find_first_of(".-")); +>>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) /// -.metadata.json else { @@ -380,27 +391,31 @@ bool writeMetadataFileAndVersionHint( } -std::optional parseTransformAndArgument(const String & transform_name_src) +std::optional parseTransformAndArgument(const String & transform_name_src, const String & time_zone) { std::string transform_name = Poco::toLower(transform_name_src); + std::optional time_zone_opt; + if (!time_zone.empty()) + time_zone_opt = time_zone; + if (transform_name == "year" || transform_name == "years") - return TransformAndArgument{"toYearNumSinceEpoch", std::nullopt}; + return TransformAndArgument{"toYearNumSinceEpoch", std::nullopt, time_zone_opt}; if (transform_name == "month" || transform_name == "months") - return TransformAndArgument{"toMonthNumSinceEpoch", std::nullopt}; + return TransformAndArgument{"toMonthNumSinceEpoch", std::nullopt, time_zone_opt}; if (transform_name == "day" || transform_name == "date" || transform_name == "days" || transform_name == "dates") - return TransformAndArgument{"toRelativeDayNum", std::nullopt}; + return TransformAndArgument{"toRelativeDayNum", std::nullopt, time_zone_opt}; if (transform_name == "hour" || transform_name == "hours") - return TransformAndArgument{"toRelativeHourNum", std::nullopt}; + return TransformAndArgument{"toRelativeHourNum", std::nullopt, time_zone_opt}; if (transform_name == "identity") - return TransformAndArgument{"identity", std::nullopt}; + return TransformAndArgument{"identity", std::nullopt, std::nullopt}; if (transform_name == "void") - return TransformAndArgument{"tuple", std::nullopt}; + return TransformAndArgument{"tuple", std::nullopt, std::nullopt}; if (transform_name.starts_with("truncate") || transform_name.starts_with("bucket")) { @@ -424,11 +439,11 @@ std::optional parseTransformAndArgument(const String & tra if (transform_name.starts_with("truncate")) { - return TransformAndArgument{"icebergTruncate", argument}; + return TransformAndArgument{"icebergTruncate", argument, std::nullopt}; } else if (transform_name.starts_with("bucket")) { - return TransformAndArgument{"icebergBucket", argument}; + return TransformAndArgument{"icebergBucket", argument, std::nullopt}; } } return std::nullopt; @@ -492,6 +507,9 @@ Poco::JSON::Object::Ptr getMetadataJSONObject( return json_str; }; + ProfileEvents::increment(ProfileEvents::IcebergJsonFileParsing); + ProfileEventTimeIncrement watch(ProfileEvents::IcebergJsonFileParsingMicroseconds); + String metadata_json_str; if (metadata_cache && table_uuid.has_value()) metadata_json_str = metadata_cache->getOrSetTableMetadata( @@ -1358,10 +1376,15 @@ KeyDescription getSortingKeyDescriptionFromMetadata(Poco::JSON::Object::Ptr meta auto column_name = source_id_to_column_name[source_id]; int direction = field->getValue(f_direction) == "asc" ? 1 : -1; auto iceberg_transform_name = field->getValue(f_transform); +<<<<<<< HEAD auto clickhouse_transform_name = parseTransformAndArgument(iceberg_transform_name); /// Quote the column name so identifiers with special characters (e.g. `@timestamp`) /// produce a parseable ORDER BY clause. auto quoted_column_name = backQuoteIfNeed(column_name); +======= + auto clickhouse_transform_name = parseTransformAndArgument(iceberg_transform_name, + local_context->getSettingsRef()[Setting::iceberg_partition_timezone]); +>>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) String full_argument; if (clickhouse_transform_name->transform_name != "identity") { @@ -1370,7 +1393,14 @@ KeyDescription getSortingKeyDescriptionFromMetadata(Poco::JSON::Object::Ptr meta { full_argument += std::to_string(*clickhouse_transform_name->argument) + ", "; } +<<<<<<< HEAD full_argument += quoted_column_name + ")"; +======= + full_argument += column_name; + if (clickhouse_transform_name->time_zone) + full_argument += ", '" + *clickhouse_transform_name->time_zone + "'"; + full_argument += ")"; +>>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) } else { diff --git a/src/Storages/ObjectStorage/DataLakes/Iceberg/Utils.h b/src/Storages/ObjectStorage/DataLakes/Iceberg/Utils.h index cf3ebbff2eea..debfc69548d4 100644 --- a/src/Storages/ObjectStorage/DataLakes/Iceberg/Utils.h +++ b/src/Storages/ObjectStorage/DataLakes/Iceberg/Utils.h @@ -59,9 +59,14 @@ struct TransformAndArgument { String transform_name; std::optional argument; + /// When Iceberg table is partitioned by time, splitting by partitions can be made using different timezone + /// (UTC in most cases). This timezone can be set with setting `iceberg_partition_timezone`, value is in this member. + /// When Iceberg partition condition converted to ClickHouse function in `parseTransformAndArgument` method + /// `time_zone` added as second argument to functions like `toRelativeDayNum`, `toYearNumSinceEpoch`, etc. + std::optional time_zone; }; -std::optional parseTransformAndArgument(const String & transform_name_src); +std::optional parseTransformAndArgument(const String & transform_name_src, const String & time_zone); CompressionMethod getCompressionMethodFromMetadataFile(const String & path); diff --git a/src/Storages/ObjectStorage/HDFS/Configuration.cpp b/src/Storages/ObjectStorage/HDFS/Configuration.cpp index 8527b4a1be5e..81e89095eb30 100644 --- a/src/Storages/ObjectStorage/HDFS/Configuration.cpp +++ b/src/Storages/ObjectStorage/HDFS/Configuration.cpp @@ -235,6 +235,14 @@ void StorageHDFSConfiguration::addStructureAndFormatToArgsIfNeeded( { addStructureAndFormatToArgsIfNeededHDFS(args, structure_, format_, context, with_structure); } + +ASTPtr StorageHDFSConfiguration::createArgsWithAccessData() const +{ + auto arguments = make_intrusive(); + arguments->children.push_back(make_intrusive(url + path.path)); + return arguments; +} + } #endif diff --git a/src/Storages/ObjectStorage/HDFS/Configuration.h b/src/Storages/ObjectStorage/HDFS/Configuration.h index c52567f4cc72..79c23bbb81b9 100644 --- a/src/Storages/ObjectStorage/HDFS/Configuration.h +++ b/src/Storages/ObjectStorage/HDFS/Configuration.h @@ -81,6 +81,8 @@ class StorageHDFSConfiguration : public StorageObjectStorageConfiguration void addStructureAndFormatToArgsIfNeeded( ASTs & args, const String & structure_, const String & format_, ContextPtr context, bool with_structure) override; + ASTPtr createArgsWithAccessData() const override; + private: void initializeFromParsedArguments(const HDFSStorageParsedArguments & parsed_arguments); void setURL(const std::string & url_); diff --git a/src/Storages/ObjectStorage/Local/Configuration.cpp b/src/Storages/ObjectStorage/Local/Configuration.cpp index 2d88b97dfebf..1a8dc9c75551 100644 --- a/src/Storages/ObjectStorage/Local/Configuration.cpp +++ b/src/Storages/ObjectStorage/Local/Configuration.cpp @@ -126,4 +126,21 @@ void StorageLocalConfiguration::fromNamedCollection(const NamedCollection & coll initializeFromParsedArguments(parsed_arguments); paths = {path}; } + +ASTPtr StorageLocalConfiguration::createArgsWithAccessData() const +{ + auto arguments = make_intrusive(); + + arguments->children.push_back(make_intrusive(path.path)); + if (getFormat() != "auto") + arguments->children.push_back(make_intrusive(getFormat())); + if (getStructure() != "auto") + arguments->children.push_back(make_intrusive(getStructure())); + if (getCompressionMethod() != "auto") + arguments->children.push_back(make_intrusive(getCompressionMethod())); + + return arguments; +} + + } diff --git a/src/Storages/ObjectStorage/Local/Configuration.h b/src/Storages/ObjectStorage/Local/Configuration.h index 615f38e6757b..4d16e4bcb32b 100644 --- a/src/Storages/ObjectStorage/Local/Configuration.h +++ b/src/Storages/ObjectStorage/Local/Configuration.h @@ -87,6 +87,8 @@ class StorageLocalConfiguration : public StorageObjectStorageConfiguration void addStructureAndFormatToArgsIfNeeded(ASTs &, const String &, const String &, ContextPtr, bool) override { } + ASTPtr createArgsWithAccessData() const override; + protected: void fromAST(ASTs & args, ContextPtr context, bool with_structure) override; void fromDisk(const String & disk_name_, ASTs & args, ContextPtr context, bool with_structure) override; diff --git a/src/Storages/ObjectStorage/MultiFileStorageObjectStorageSink.cpp b/src/Storages/ObjectStorage/MultiFileStorageObjectStorageSink.cpp index f46985b9a52f..c14223d94044 100644 --- a/src/Storages/ObjectStorage/MultiFileStorageObjectStorageSink.cpp +++ b/src/Storages/ObjectStorage/MultiFileStorageObjectStorageSink.cpp @@ -51,7 +51,7 @@ MultiFileStorageObjectStorageSink::~MultiFileStorageObjectStorageSink() /// Output is `table_root/year=2025/month=12/day=12/file.1.parquet` std::string MultiFileStorageObjectStorageSink::generateNewFilePath() { - const auto file_format = Poco::toLower(configuration->format); + const auto file_format = Poco::toLower(configuration->getFormat()); const auto index_string = std::to_string(file_paths.size() + 1); std::size_t pos = base_path.rfind(file_format); @@ -91,8 +91,8 @@ std::shared_ptr MultiFileStorageObjectStorageSink::cre format_settings, sample_block, context, - configuration->format, - configuration->compression_method); + configuration->getFormat(), + configuration->getCompressionMethod()); } void MultiFileStorageObjectStorageSink::consume(Chunk & chunk) diff --git a/src/Storages/ObjectStorage/ReadBufferIterator.cpp b/src/Storages/ObjectStorage/ReadBufferIterator.cpp index 3531a28e9a78..fe77b33ea7c7 100644 --- a/src/Storages/ObjectStorage/ReadBufferIterator.cpp +++ b/src/Storages/ObjectStorage/ReadBufferIterator.cpp @@ -39,8 +39,8 @@ ReadBufferIterator::ReadBufferIterator( , read_keys(read_keys_) , prev_read_keys_size(read_keys_.size()) { - if (configuration->format != "auto") - format = configuration->format; + if (configuration->getFormat() != "auto") + format = configuration->getFormat(); } SchemaCache::Key ReadBufferIterator::getKeyForSchemaCache(const ObjectInfo & object_info, const String & format_name) const @@ -143,7 +143,7 @@ std::unique_ptr ReadBufferIterator::recreateLastReadBuffer() auto impl = createReadBuffer(current_object_info->relative_path_with_metadata, object_storage, context, getLogger("ReadBufferIterator")); - const auto compression_method = chooseCompressionMethod(current_object_info->getFileName(), configuration->compression_method); + const auto compression_method = chooseCompressionMethod(current_object_info->getFileName(), configuration->getCompressionMethod()); const auto zstd_window = static_cast(context->getSettingsRef()[Setting::zstd_window_log_max]); return wrapReadBufferWithCompressionMethod(std::move(impl), compression_method, zstd_window); @@ -260,13 +260,13 @@ ReadBufferIterator::Data ReadBufferIterator::next() using ObjectInfoInArchive = StorageObjectStorageSource::ArchiveIterator::ObjectInfoInArchive; if (const auto * object_info_in_archive = dynamic_cast(current_object_info.get())) { - compression_method = chooseCompressionMethod(filename, configuration->compression_method); + compression_method = chooseCompressionMethod(filename, configuration->getCompressionMethod()); const auto & archive_reader = object_info_in_archive->archive_reader; read_buf = archive_reader->readFile(object_info_in_archive->path_in_archive, /*throw_on_not_found=*/true); } else { - compression_method = chooseCompressionMethod(filename, configuration->compression_method); + compression_method = chooseCompressionMethod(filename, configuration->getCompressionMethod()); read_buf = createReadBuffer( current_object_info->relative_path_with_metadata, object_storage, getContext(), getLogger("ReadBufferIterator")); } diff --git a/src/Storages/ObjectStorage/S3/Configuration.cpp b/src/Storages/ObjectStorage/S3/Configuration.cpp index f60d69ed1f5c..d2f95cf379ce 100644 --- a/src/Storages/ObjectStorage/S3/Configuration.cpp +++ b/src/Storages/ObjectStorage/S3/Configuration.cpp @@ -108,7 +108,11 @@ static const std::unordered_set optional_configuration_keys = "partition_strategy", "partition_columns_in_data_file", "storage_class_name", +<<<<<<< HEAD "storage_class", /// Interchangeable alias for `storage_class_name`, see issue #68551 +======= + "storage_type", +>>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) /// Private configuration options "role_arn", /// for extra_credentials "role_session_name", /// for extra_credentials @@ -688,6 +692,7 @@ void S3StorageParsedArguments::fromAST(ASTs & args, ContextPtr context, bool wit compression_method = compression_method_value.value(); } + if (auto partition_strategy_value = getFromPositionOrKeyValue("partition_strategy", args, engine_args_to_idx, key_value_args); partition_strategy_value.has_value()) { @@ -1090,6 +1095,31 @@ void StorageS3Configuration::addStructureAndFormatToArgsIfNeeded( addStructureAndFormatToArgsIfNeededS3( args, structure_, format_, context, with_structure, S3StorageParsedArguments::getMaxNumberOfArguments(with_structure)); } + +ASTPtr StorageS3Configuration::createArgsWithAccessData() const +{ + auto arguments = make_intrusive(); + + arguments->children.push_back(make_intrusive(url.uri_str)); + if (s3_settings->auth_settings[S3AuthSetting::no_sign_request]) + { + arguments->children.push_back(make_intrusive("NOSIGN")); + } + else + { + arguments->children.push_back(make_intrusive(s3_settings->auth_settings[S3AuthSetting::access_key_id].value)); + arguments->children.push_back(make_intrusive(s3_settings->auth_settings[S3AuthSetting::secret_access_key].value)); + if (!s3_settings->auth_settings[S3AuthSetting::session_token].value.empty()) + arguments->children.push_back(make_intrusive(s3_settings->auth_settings[S3AuthSetting::session_token].value)); + if (getFormat() != "auto") + arguments->children.push_back(make_intrusive(getFormat())); + if (!getCompressionMethod().empty()) + arguments->children.push_back(make_intrusive(getCompressionMethod())); + } + + return arguments; +} + } #endif diff --git a/src/Storages/ObjectStorage/S3/Configuration.h b/src/Storages/ObjectStorage/S3/Configuration.h index 13b581c07585..5beda2db5f50 100644 --- a/src/Storages/ObjectStorage/S3/Configuration.h +++ b/src/Storages/ObjectStorage/S3/Configuration.h @@ -145,6 +145,8 @@ class StorageS3Configuration : public StorageObjectStorageConfiguration ContextPtr context, bool with_structure) override; + ASTPtr createArgsWithAccessData() const override; + static bool collectCredentials(ASTPtr maybe_credentials, S3::S3AuthSettings & auth_settings_, ContextPtr local_context); S3::URI url; diff --git a/src/Storages/ObjectStorage/StorageObjectStorage.cpp b/src/Storages/ObjectStorage/StorageObjectStorage.cpp index 9ffa15fa654d..94e3321ec3dd 100644 --- a/src/Storages/ObjectStorage/StorageObjectStorage.cpp +++ b/src/Storages/ObjectStorage/StorageObjectStorage.cpp @@ -97,6 +97,18 @@ String StorageObjectStorage::getPathSample(ContextPtr context) query_settings.throw_on_zero_files_match = false; query_settings.ignore_non_existent_file = true; +<<<<<<< HEAD +======= + bool local_distributed_processing = distributed_processing; + if (context->getSettingsRef()[Setting::use_hive_partitioning]) + local_distributed_processing = false; + + const auto path = configuration->getRawPath(); + + if (!configuration->isArchive() && !path.hasGlobs() && !local_distributed_processing) + return path.path; + +>>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) auto file_iterator = StorageObjectStorageSource::createFileIterator( configuration, query_settings, @@ -131,13 +143,15 @@ StorageObjectStorage::StorageObjectStorage( std::optional format_settings_, LoadingStrictnessLevel mode, std::shared_ptr catalog_, - bool if_not_exists_, + bool /*if_not_exists_*/, bool is_datalake_query, bool distributed_processing_, ASTPtr partition_by_, - ASTPtr order_by_, + ASTPtr /*order_by_*/, bool is_table_function_, - bool lazy_init) + bool lazy_init, + bool updated_configuration, + std::optional sample_path_) : IStorage(table_id_) , configuration(configuration_) , object_storage(object_storage_) @@ -150,9 +164,9 @@ StorageObjectStorage::StorageObjectStorage( , background_operations_assignee(*this, table_id_, BackgroundJobsAssignee::Type::DataProcessing, Context::getGlobalContextInstance()) { configuration->initPartitionStrategy(partition_by_, columns_in_table_or_function_definition, context); - const bool need_resolve_columns_or_format = columns_in_table_or_function_definition.empty() || (configuration->format == "auto"); + const bool need_resolve_columns_or_format = columns_in_table_or_function_definition.empty() || (configuration->getFormat() == "auto"); const bool need_resolve_sample_path = context->getSettingsRef()[Setting::use_hive_partitioning] - && !configuration->partition_strategy + && !configuration->getPartitionStrategy() && !configuration->isDataLakeConfiguration(); const bool do_lazy_init = lazy_init && !need_resolve_columns_or_format && !need_resolve_sample_path; @@ -170,17 +184,9 @@ StorageObjectStorage::StorageObjectStorage( throw Exception(ErrorCodes::BAD_ARGUMENTS, "Delta lake CDF is allowed only for deltaLake table function"); } - if (!is_table_function && !columns_in_table_or_function_definition.empty() && !is_datalake_query && mode == LoadingStrictnessLevel::CREATE) - { - LOG_DEBUG(log, "Creating new storage with specified columns"); - configuration->create( - object_storage, context, columns_in_table_or_function_definition, partition_by_, order_by_, if_not_exists_, catalog, storage_id); - } - - bool updated_configuration = false; try { - if (!do_lazy_init) + if (!do_lazy_init && !updated_configuration) { if (is_table_function) configuration->lazyInitializeIfNeeded(object_storage, context); @@ -200,7 +206,7 @@ StorageObjectStorage::StorageObjectStorage( tryLogCurrentException(log, /*start of message = */ "", LogsLevel::warning); } - std::string sample_path; + std::string sample_path = sample_path_.value_or(""); ColumnsDescription columns{columns_in_table_or_function_definition}; @@ -209,7 +215,7 @@ StorageObjectStorage::StorageObjectStorage( if (configuration->isDataLakeConfiguration()) throw Exception(ErrorCodes::BAD_ARGUMENTS, "The _schema_hash placeholder is not supported for DataLake engines"); - if (configuration->partition_strategy_type == PartitionStrategyFactory::StrategyType::HIVE) + if (configuration->getPartitionStrategyType() == PartitionStrategyFactory::StrategyType::HIVE) throw Exception(ErrorCodes::BAD_ARGUMENTS, "The _schema_hash placeholder is not supported with hive partition strategy"); if (columns.empty()) @@ -219,7 +225,7 @@ StorageObjectStorage::StorageObjectStorage( } if (need_resolve_columns_or_format) - resolveSchemaAndFormat(columns, configuration->format, object_storage, configuration, format_settings, sample_path, context); + resolveSchemaAndFormat(columns, object_storage, configuration, format_settings, sample_path, context); else validateSupportedColumns(columns, *configuration); @@ -227,7 +233,7 @@ StorageObjectStorage::StorageObjectStorage( /// FIXME: We need to call getPathSample() lazily on select /// in case it failed to be initialized in constructor. - if (updated_configuration && sample_path.empty() && need_resolve_sample_path && !configuration->partition_strategy) + if (updated_configuration && sample_path.empty() && need_resolve_sample_path && !configuration->getPartitionStrategy()) { try { @@ -259,7 +265,7 @@ StorageObjectStorage::StorageObjectStorage( sample_path); } - bool format_supports_prewhere = FormatFactory::instance().checkIfFormatSupportsPrewhere(configuration->format, context, format_settings); + bool format_supports_prewhere = FormatFactory::instance().checkIfFormatSupportsPrewhere(configuration->getFormat(), context, format_settings); /// TODO: Known problems with datalake prewhere: /// * If the iceberg table went through schema evolution, columns read from file may need to @@ -315,14 +321,16 @@ StorageObjectStorage::StorageObjectStorage( metadata.setConstraints(constraints_); metadata.setComment(comment); - if (configuration->partition_strategy) - metadata.partition_key = configuration->partition_strategy->getPartitionKeyDescription(); + if (configuration->getPartitionStrategy()) + { + metadata.partition_key = configuration->getPartitionStrategy()->getPartitionKeyDescription(); + } metadata.setVirtuals(VirtualColumnUtils::getVirtualsForFileLikeStorage( metadata.columns, context, format_settings, - configuration->partition_strategy_type, + configuration->getPartitionStrategyType(), sample_path)); setInMemoryMetadata(metadata); @@ -335,17 +343,17 @@ String StorageObjectStorage::getName() const bool StorageObjectStorage::prefersLargeBlocks() const { - return FormatFactory::instance().checkIfOutputFormatPrefersLargeBlocks(configuration->format); + return FormatFactory::instance().checkIfOutputFormatPrefersLargeBlocks(configuration->getFormat()); } bool StorageObjectStorage::parallelizeOutputAfterReading(ContextPtr context) const { - return FormatFactory::instance().checkParallelizeOutputAfterReading(configuration->format, context); + return FormatFactory::instance().checkParallelizeOutputAfterReading(configuration->getFormat(), context); } bool StorageObjectStorage::supportsSubsetOfColumns(const ContextPtr & context) const { - return FormatFactory::instance().checkIfFormatSupportsSubsetOfColumns(configuration->format, context, format_settings); + return FormatFactory::instance().checkIfFormatSupportsSubsetOfColumns(configuration->getFormat(), context, format_settings); } bool StorageObjectStorage::supportsPrewhere() const @@ -487,8 +495,7 @@ void StorageObjectStorage::read( configuration->update(object_storage, local_context); } - - if (configuration->partition_strategy && configuration->partition_strategy_type != PartitionStrategyFactory::StrategyType::HIVE) + if (configuration->getPartitionStrategy() && configuration->getPartitionStrategyType() != PartitionStrategyFactory::StrategyType::HIVE) { throw Exception(ErrorCodes::NOT_IMPLEMENTED, "Reading from a partitioned {} storage is not implemented yet", @@ -620,10 +627,10 @@ SinkToStoragePtr StorageObjectStorage::write( /// Not a data lake, just raw object storage - if (configuration->partition_strategy) + if (configuration->getPartitionStrategy()) { auto sink_creator = std::make_shared(object_storage, configuration, format_settings, sample_block, local_context); - return std::make_shared(configuration->partition_strategy, sink_creator, local_context, sample_block); + return std::make_shared(configuration->getPartitionStrategy(), sink_creator, local_context, sample_block); } auto paths = configuration->getPaths(); @@ -639,8 +646,8 @@ SinkToStoragePtr StorageObjectStorage::write( format_settings, sample_block, local_context, - configuration->format, - configuration->compression_method); + configuration->getFormat(), + configuration->getCompressionMethod()); } bool StorageObjectStorage::optimize( @@ -664,13 +671,13 @@ bool StorageObjectStorage::supportsImport(ContextPtr local_context) const return configuration->getExternalMetadata()->supportsImport(local_context); } - if (!configuration->partition_strategy) + if (!configuration->getPartitionStrategy()) return false; - if (configuration->partition_strategy_type == PartitionStrategyFactory::StrategyType::WILDCARD) + if (configuration->getPartitionStrategyType() == PartitionStrategyFactory::StrategyType::WILDCARD) return configuration->getRawPath().hasExportFilenameWildcard(); - return configuration->partition_strategy_type == PartitionStrategyFactory::StrategyType::HIVE; + return configuration->getPartitionStrategyType() == PartitionStrategyFactory::StrategyType::HIVE; } SinkToStoragePtr StorageObjectStorage::import( @@ -698,9 +705,9 @@ SinkToStoragePtr StorageObjectStorage::import( std::string partition_key; - if (configuration->partition_strategy) + if (configuration->getPartitionStrategy()) { - const auto column_with_partition_key = configuration->partition_strategy->computePartitionKey(block_with_partition_values); + const auto column_with_partition_key = configuration->getPartitionStrategy()->computePartitionKey(block_with_partition_values); if (!column_with_partition_key->empty()) { @@ -867,7 +874,7 @@ ColumnsDescription StorageObjectStorage::resolveSchemaFromData( { ObjectInfos read_keys; auto iterator = createReadBufferIterator(object_storage, configuration, format_settings, read_keys, context); - auto schema = readSchemaFromFormat(configuration->format, format_settings, *iterator, context); + auto schema = readSchemaFromFormat(configuration->getFormat(), format_settings, *iterator, context); sample_path = iterator->getLastFilePath(); return schema; } @@ -888,7 +895,7 @@ std::string StorageObjectStorage::resolveFormatFromData( std::pair StorageObjectStorage::resolveSchemaAndFormatFromData( const ObjectStoragePtr & object_storage, - const StorageObjectStorageConfigurationPtr & configuration, + StorageObjectStorageConfigurationPtr & configuration, const std::optional & format_settings, std::string & sample_path, const ContextPtr & context) @@ -897,13 +904,13 @@ std::pair StorageObjectStorage::resolveSchemaAn auto iterator = createReadBufferIterator(object_storage, configuration, format_settings, read_keys, context); auto [columns, format] = detectFormatAndReadSchema(format_settings, *iterator, context); sample_path = iterator->getLastFilePath(); - configuration->format = format; + configuration->setFormat(format); return std::pair(columns, format); } void StorageObjectStorage::addInferredEngineArgsToCreateQuery(ASTs & args, const ContextPtr & context) const { - configuration->addStructureAndFormatToArgsIfNeeded(args, "", configuration->format, context, /*with_structure=*/false); + configuration->addStructureAndFormatToArgsIfNeeded(args, "", configuration->getFormat(), context, /*with_structure=*/false); } SchemaCache & StorageObjectStorage::getSchemaCache(const ContextPtr & context, const std::string & storage_engine_name) diff --git a/src/Storages/ObjectStorage/StorageObjectStorage.h b/src/Storages/ObjectStorage/StorageObjectStorage.h index 233dcf71cc66..45807c7c89c4 100644 --- a/src/Storages/ObjectStorage/StorageObjectStorage.h +++ b/src/Storages/ObjectStorage/StorageObjectStorage.h @@ -60,7 +60,9 @@ class StorageObjectStorage : public IStorage, public IBackgroundOperation ASTPtr partition_by_ = nullptr, ASTPtr order_by_ = nullptr, bool is_table_function_ = false, - bool lazy_init = false); + bool lazy_init = false, + bool updated_configuration = false, // avoid double update configuration from cluster and local versions + std::optional sample_path_ = std::nullopt); String getName() const override; @@ -160,7 +162,7 @@ class StorageObjectStorage : public IStorage, public IBackgroundOperation static std::pair resolveSchemaAndFormatFromData( const ObjectStoragePtr & object_storage, - const StorageObjectStorageConfigurationPtr & configuration, + StorageObjectStorageConfigurationPtr & configuration, const std::optional & format_settings, std::string & sample_path, const ContextPtr & context); diff --git a/src/Storages/ObjectStorage/StorageObjectStorageCluster.cpp b/src/Storages/ObjectStorage/StorageObjectStorageCluster.cpp index 061fcf839d0d..b48b5b3b22c8 100644 --- a/src/Storages/ObjectStorage/StorageObjectStorageCluster.cpp +++ b/src/Storages/ObjectStorage/StorageObjectStorageCluster.cpp @@ -11,9 +11,16 @@ #include #include +#include +#include +#include +#include +#include #include #include #include +#include +#include #include #include @@ -30,6 +37,15 @@ namespace Setting extern const SettingsBool use_hive_partitioning; extern const SettingsBool cluster_function_process_archive_on_multiple_nodes; extern const SettingsObjectStorageGranularityLevel cluster_table_function_split_granularity; +<<<<<<< HEAD +======= + extern const SettingsBool parallel_replicas_for_cluster_engines; + extern const SettingsString object_storage_cluster; + extern const SettingsInt64 delta_lake_snapshot_start_version; + extern const SettingsInt64 delta_lake_snapshot_end_version; + extern const SettingsUInt64 lock_object_storage_task_distribution_ms; + extern const SettingsBool allow_experimental_iceberg_read_optimization; +>>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) } namespace ErrorCodes @@ -47,6 +63,14 @@ String StorageObjectStorageCluster::getPathSample(ContextPtr context) auto query_settings = configuration->getQuerySettings(context); /// We don't want to throw an exception if there are no files with specified path. query_settings.throw_on_zero_files_match = false; + + if (!configuration->isArchive()) + { + const auto & path = configuration->getPathForRead(); + if (!path.hasGlobs()) + return path.path; + } + auto file_iterator = StorageObjectStorageSource::createFileIterator( configuration, query_settings, @@ -59,11 +83,14 @@ String StorageObjectStorageCluster::getPathSample(ContextPtr context) {}, // virtual_columns {}, // hive_columns nullptr, // read_keys - {} // file_progress_callback + {}, // file_progress_callback + false, // ignore_archive_globs + true // skip_object_metadata ); if (auto file = file_iterator->next(0)) return file->getPath(); + return ""; } @@ -75,29 +102,97 @@ StorageObjectStorageCluster::StorageObjectStorageCluster( const ColumnsDescription & columns_in_table_or_function_definition, const ConstraintsDescription & constraints_, const ASTPtr & partition_by, + const ASTPtr & order_by, ContextPtr context_, - bool is_table_function) + const String & comment_, + std::optional format_settings_, + LoadingStrictnessLevel mode_, + std::shared_ptr catalog, + bool if_not_exists, + bool is_datalake_query, + bool is_table_function, + bool lazy_init) : IStorageCluster( cluster_name_, table_id_, getLogger(fmt::format("{}({})", configuration_->getEngineName(), table_id_.table_name))) , configuration{configuration_} , object_storage(object_storage_) + , cluster_name_in_settings(false) { configuration->initPartitionStrategy(partition_by, columns_in_table_or_function_definition, context_); - /// We allow exceptions to be thrown on update(), - /// because Cluster engine can only be used as table function, - /// so no lazy initialization is allowed. - configuration->update(object_storage, context_); + + const bool need_resolve_columns_or_format = columns_in_table_or_function_definition.empty() || (configuration->getFormat() == "auto"); + const bool do_lazy_init = lazy_init && !need_resolve_columns_or_format && catalog; + + auto log = getLogger("StorageObjectStorageCluster"); + + bool is_delta_lake_cdf = context_->getSettingsRef()[Setting::delta_lake_snapshot_start_version] != -1 + || context_->getSettingsRef()[Setting::delta_lake_snapshot_end_version] != -1; + + if (!is_table_function && is_delta_lake_cdf) + { + throw Exception(ErrorCodes::BAD_ARGUMENTS, "Delta lake CDF is allowed only for deltaLake table function"); + } + + if (!is_table_function && !columns_in_table_or_function_definition.empty() && !is_datalake_query && mode_ == LoadingStrictnessLevel::CREATE) + { + LOG_DEBUG(log, "Creating new storage with specified columns"); + configuration->create( + object_storage, context_, columns_in_table_or_function_definition, partition_by, order_by, if_not_exists, catalog, table_id_); + } + + bool updated_configuration = false; + try + { + if (!do_lazy_init) + { + if (is_table_function) + configuration->lazyInitializeIfNeeded(object_storage, context_); + else + configuration->update(object_storage, context_); + updated_configuration = true; + } + } + catch (...) + { + // If we don't have format or schema yet, we can't ignore failed configuration update, + // because relevant configuration is crucial for format and schema inference + if (mode_ <= LoadingStrictnessLevel::CREATE || need_resolve_columns_or_format) + { + throw; + } + tryLogCurrentException(log); + } ColumnsDescription columns{columns_in_table_or_function_definition}; + + if (configuration->getRawPath().hasSchemaHashWildcard()) + { + if (configuration->isDataLakeConfiguration()) + throw Exception(ErrorCodes::BAD_ARGUMENTS, "The _schema_hash placeholder is not supported for DataLake engines"); + + if (configuration->getPartitionStrategyType() == PartitionStrategyFactory::StrategyType::HIVE) + throw Exception(ErrorCodes::BAD_ARGUMENTS, "The _schema_hash placeholder is not supported with hive partition strategy"); + + if (columns.empty()) + throw Exception(ErrorCodes::BAD_ARGUMENTS, "Cannot use _schema_hash placeholder without explicitly specifying columns"); + + configuration->setSchemaHash(StorageObjectStorageConfiguration::computeSchemaHash(columns)); + } + std::string sample_path; - resolveSchemaAndFormat(columns, configuration->format, object_storage, configuration, {}, sample_path, context_); + if (need_resolve_columns_or_format) + resolveSchemaAndFormat(columns, object_storage, configuration, {}, sample_path, context_); + else + validateSupportedColumns(columns, *configuration); configuration->check(context_); - if (sample_path.empty() - && context_->getSettingsRef()[Setting::use_hive_partitioning] - && !configuration->isDataLakeConfiguration() - && !configuration->partition_strategy) + if (updated_configuration && sample_path.empty() + && context_->getSettingsRef()[Setting::use_hive_partitioning] + && !configuration->isDataLakeConfiguration() + && !configuration->getPartitionStrategy()) + { sample_path = getPathSample(context_); + } /// Not grabbing the file_columns because it is not necessary to do it here. std::tie(hive_partition_columns_to_read_from_file_path, std::ignore) = HivePartitioningUtils::setupHivePartitioningForObjectStorage( @@ -110,7 +205,8 @@ StorageObjectStorageCluster::StorageObjectStorageCluster( StorageInMemoryMetadata metadata; metadata.setColumns(columns); - if (is_table_function && configuration->isDataLakeConfiguration()) + + if (!do_lazy_init && is_table_function && configuration->isDataLakeConfiguration()) { /// For datalake table functions, always pin the current snapshot version so that /// query execution uses the same snapshot as query analysis (logical-race fix). @@ -128,19 +224,54 @@ StorageObjectStorageCluster::StorageObjectStorageCluster( metadata.setConstraints(constraints_); - if (configuration->partition_strategy) + if (configuration->getPartitionStrategy()) { - metadata.partition_key = configuration->partition_strategy->getPartitionKeyDescription(); + metadata.partition_key = configuration->getPartitionStrategy()->getPartitionKeyDescription(); } metadata.setVirtuals(VirtualColumnUtils::getVirtualsForFileLikeStorage( metadata.columns, context_, /* format_settings */std::nullopt, - configuration->partition_strategy_type, + configuration->getPartitionStrategyType(), sample_path)); setInMemoryMetadata(metadata); + + const auto can_use_parallel_replicas = !cluster_name_.empty() + && context_->getSettingsRef()[Setting::parallel_replicas_for_cluster_engines] + && context_->canUseTaskBasedParallelReplicas() + && !context_->isDistributed(); + + bool can_use_distributed_iterator = + context_->getClientInfo().collaborate_with_initiator && + can_use_parallel_replicas; + + pure_storage = std::make_shared( + configuration, + object_storage, + context_, + getStorageID(), + IStorageCluster::getInMemoryMetadata().getColumns(), + IStorageCluster::getInMemoryMetadata().getConstraints(), + comment_, + format_settings_, + mode_, + catalog, + if_not_exists, + is_datalake_query, + /* distributed_processing */can_use_distributed_iterator, + partition_by, + order_by, + /* is_table_function */is_table_function, + /* lazy_init */lazy_init, + updated_configuration, + sample_path); + + auto virtuals_ = getVirtualsPtr(); + if (virtuals_) + pure_storage->setVirtuals(*virtuals_); + pure_storage->setInMemoryMetadata(IStorageCluster::getInMemoryMetadata()); } std::string StorageObjectStorageCluster::getName() const @@ -150,6 +281,8 @@ std::string StorageObjectStorageCluster::getName() const std::optional StorageObjectStorageCluster::totalRows(ContextPtr query_context) const { + if (pure_storage) + return pure_storage->totalRows(query_context); configuration->lazyInitializeIfNeeded( object_storage, query_context); @@ -158,17 +291,152 @@ std::optional StorageObjectStorageCluster::totalRows(ContextPtr query_co std::optional StorageObjectStorageCluster::totalBytes(ContextPtr query_context) const { + if (pure_storage) + return pure_storage->totalBytes(query_context); configuration->lazyInitializeIfNeeded( object_storage, query_context); return configuration->totalBytes(query_context); } +void StorageObjectStorageCluster::updateQueryForDistributedEngineIfNeeded(ASTPtr & query, ContextPtr context, bool make_cluster_function) +{ + // Change table engine on table function for distributed request + // CREATE TABLE t (...) ENGINE=IcebergS3(...) + // SELECT * FROM t + // change on + // SELECT * FROM icebergS3(...) + // to execute on cluster nodes + + auto * select_query = query->as(); + if (!select_query || !select_query->tables()) + return; + + auto * tables = select_query->tables()->as(); + + if (tables->children.empty()) + throw Exception( + ErrorCodes::LOGICAL_ERROR, + "Expected SELECT query from table with engine {}, got '{}'", + configuration->getEngineName(), query->formatForLogging()); + + auto * table_expression = tables->children[0]->as()->table_expression->as(); + + if (!table_expression) + return; + + if (!table_expression->database_and_table_name) + return; + + auto & table_identifier_typed = table_expression->database_and_table_name->as(); + + auto table_alias = table_identifier_typed.tryGetAlias(); + + auto storage_engine_name = configuration->getEngineName(); + if (storage_engine_name == "Iceberg") + { + switch (configuration->getType()) + { + case ObjectStorageType::S3: + storage_engine_name = "IcebergS3"; + break; + case ObjectStorageType::Azure: + storage_engine_name = "IcebergAzure"; + break; + case ObjectStorageType::HDFS: + storage_engine_name = "IcebergHDFS"; + break; + default: + throw Exception( + ErrorCodes::LOGICAL_ERROR, + "Can't find table function for engine {}", + storage_engine_name + ); + } + } + + static std::unordered_map engine_to_function = { + {"S3", "s3"}, + {"Azure", "azureBlobStorage"}, + {"HDFS", "hdfs"}, + {"Iceberg", "iceberg"}, + {"IcebergS3", "icebergS3"}, + {"IcebergAzure", "icebergAzure"}, + {"IcebergHDFS", "icebergHDFS"}, + {"IcebergLocal", "icebergLocal"}, + {"DeltaLake", "deltaLake"}, + {"DeltaLakeS3", "deltaLakeS3"}, + {"DeltaLakeAzure", "deltaLakeAzure"}, + {"DeltaLakeLocal", "deltaLakeLocal"}, + {"Hudi", "hudi"}, + {"COSN", "cosn"}, + {"GCS", "gcs"}, + {"OSS", "oss"}, + }; + + auto p = engine_to_function.find(storage_engine_name); + if (p == engine_to_function.end()) + { + throw Exception( + ErrorCodes::LOGICAL_ERROR, + "Can't find table function for engine {}", + storage_engine_name + ); + } + + std::string table_function_name = p->second; + + auto function_ast = make_intrusive(); + function_ast->name = table_function_name; + + function_ast->arguments = configuration->createArgsWithAccessData(); + function_ast->children.push_back(function_ast->arguments); + function_ast->setAlias(table_alias); + + ASTPtr function_ast_ptr(function_ast); + + table_expression->database_and_table_name = nullptr; + table_expression->table_function = function_ast_ptr; + table_expression->children[0] = function_ast_ptr; + + if (make_cluster_function) + { + auto cluster_name = getClusterName(context); + + if (cluster_name.empty()) + { + throw Exception( + ErrorCodes::LOGICAL_ERROR, + "Can't be here without cluster name, no cluster name in query {}", + query->formatForLogging()); + } + + auto settings = select_query->settings(); + if (settings) + { + auto & settings_ast = settings->as(); + settings_ast.changes.insertSetting("object_storage_cluster", cluster_name); + } + else + { + auto settings_ast_ptr = make_intrusive(); + settings_ast_ptr->is_standalone = false; + settings_ast_ptr->changes.setSetting("object_storage_cluster", cluster_name); + select_query->setExpression(ASTSelectQuery::Expression::SETTINGS, std::move(settings_ast_ptr)); + } + + cluster_name_in_settings = true; + } +} + void StorageObjectStorageCluster::updateQueryToSendIfNeeded( ASTPtr & query, const DB::StorageSnapshotPtr & storage_snapshot, - const ContextPtr & context) + const ContextPtr & context, + bool make_cluster_function) { + updateQueryForDistributedEngineIfNeeded(query, context, make_cluster_function); + auto * table_function = extractTableFunctionFromSelectQuery(query); if (!table_function) return; @@ -191,6 +459,9 @@ void StorageObjectStorageCluster::updateQueryToSendIfNeeded( configuration->getEngineName()); } + ASTPtr object_storage_type_arg; + configuration->extractDynamicStorageType(args, context, &object_storage_type_arg, !cluster_name_in_settings); + ASTPtr settings_temporary_storage = nullptr; for (auto it = args.begin(); it != args.end(); ++it) { @@ -203,6 +474,7 @@ void StorageObjectStorageCluster::updateQueryToSendIfNeeded( } } +<<<<<<< HEAD if (!endsWith(table_function->name, "Cluster")) { configuration->addStructureAndFormatToArgsIfNeeded(args, structure, configuration->format, context, /*with_structure=*/true); @@ -223,16 +495,80 @@ void StorageObjectStorageCluster::updateQueryToSendIfNeeded( } } else +======= + if (cluster_name_in_settings || !endsWith(table_function->name, "Cluster")) +>>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) { + configuration->addStructureAndFormatToArgsIfNeeded(args, structure, configuration->getFormat(), context, /*with_structure=*/true); + + if (make_cluster_function) + { + /// Convert to old-stype *Cluster table function. + /// This allows to use old clickhouse versions in cluster. + static std::unordered_map function_to_cluster_function = { + {"s3", "s3Cluster"}, + {"azureBlobStorage", "azureBlobStorageCluster"}, + {"hdfs", "hdfsCluster"}, + {"iceberg", "icebergCluster"}, + {"icebergS3", "icebergS3Cluster"}, + {"icebergAzure", "icebergAzureCluster"}, + {"icebergHDFS", "icebergHDFSCluster"}, + {"icebergLocal", "icebergLocalCluster"}, + {"deltaLake", "deltaLakeCluster"}, + {"deltaLakeS3", "deltaLakeS3Cluster"}, + {"deltaLakeAzure", "deltaLakeAzureCluster"}, + {"hudi", "hudiCluster"}, + {"paimonS3", "paimonS3Cluster"}, + {"paimonAzure", "paimonAzureCluster"}, + }; + + auto p = function_to_cluster_function.find(table_function->name); + if (p == function_to_cluster_function.end()) + { + throw Exception( + ErrorCodes::LOGICAL_ERROR, + "Can't find cluster variant for table function {}", + table_function->name); + } + + table_function->name = p->second; + + auto cluster_name = getClusterName(context); + auto cluster_name_arg = make_intrusive(cluster_name); + args.insert(args.begin(), cluster_name_arg); + + auto * select_query = query->as(); + if (!select_query) + throw Exception( + ErrorCodes::LOGICAL_ERROR, + "Expected SELECT query from table function {}", + configuration->getEngineName()); + + auto settings = select_query->settings(); + if (settings) + { + auto & settings_ast = settings->as(); + if (settings_ast.changes.removeSetting("object_storage_cluster") && settings_ast.changes.empty()) + { + select_query->setExpression(ASTSelectQuery::Expression::SETTINGS, {}); + } + /// No throw if not found - `object_storage_cluster` can be global setting. + } + } + } + else + { /// *Cluster function has cluster name as first argument. Temporary remove it before add structure and format ASTPtr cluster_name_arg = args.front(); args.erase(args.begin()); - configuration->addStructureAndFormatToArgsIfNeeded(args, structure, configuration->format, context, /*with_structure=*/true); + configuration->addStructureAndFormatToArgsIfNeeded(args, structure, configuration->getFormat(), context, /*with_structure=*/true); args.insert(args.begin(), cluster_name_arg); } if (settings_temporary_storage) { args.insert(args.end(), std::move(settings_temporary_storage)); } + if (object_storage_type_arg) + args.insert(args.end(), object_storage_type_arg); } void StorageObjectStorageCluster::updateExternalDynamicMetadataIfExists(ContextPtr query_context) @@ -261,11 +597,18 @@ void StorageObjectStorageCluster::updateExternalDynamicMetadataIfExists(ContextP new_metadata = *metadata_snapshot; } +<<<<<<< HEAD setInMemoryMetadata(new_metadata.withVirtuals(VirtualColumnUtils::getVirtualsForFileLikeStorage( new_metadata.columns, query_context, /* format_settings */ std::nullopt, configuration->partition_strategy_type))); +======= + setInMemoryMetadata(new_metadata); + + if (pure_storage) + pure_storage->setInMemoryMetadata(IStorageCluster::getInMemoryMetadata()); +>>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) } RemoteQueryExecutor::Extension StorageObjectStorageCluster::getTaskIteratorExtension( @@ -295,7 +638,7 @@ RemoteQueryExecutor::Extension StorageObjectStorageCluster::getTaskIteratorExten { iterator = std::make_shared( std::move(iterator), - configuration->format, + configuration->getFormat(), object_storage, local_context ); @@ -336,5 +679,469 @@ RemoteQueryExecutor::Extension StorageObjectStorageCluster::getTaskIteratorExten return RemoteQueryExecutor::Extension{ .task_iterator = std::move(callback) }; } +void StorageObjectStorageCluster::readFallBackToPure( + QueryPlan & query_plan, + const Names & column_names, + const StorageSnapshotPtr & storage_snapshot, + SelectQueryInfo & query_info, + ContextPtr context, + QueryProcessingStage::Enum processed_stage, + size_t max_block_size, + size_t num_streams) +{ + pure_storage->read(query_plan, column_names, storage_snapshot, query_info, context, processed_stage, max_block_size, num_streams); +} + +bool StorageObjectStorageCluster::isClusterSupported() const +{ + return configuration->isClusterSupported(); +} + +SinkToStoragePtr StorageObjectStorageCluster::writeFallBackToPure( + const ASTPtr & query, + const StorageMetadataPtr & metadata_snapshot, + ContextPtr context, + bool async_insert) +{ + return pure_storage->write(query, metadata_snapshot, context, async_insert); +} + +String StorageObjectStorageCluster::getClusterName(ContextPtr context) const +{ + /// StorageObjectStorageCluster is always created for cluster or non-cluster variants. + /// User can specify cluster name in table definition or in setting `object_storage_cluster` + /// only for several queries. When it specified in both places, priority is given to the query setting. + /// When it is empty, non-cluster realization is used. + + if (!isClusterSupported()) + return ""; + + auto cluster_name_from_settings = context->getSettingsRef()[Setting::object_storage_cluster].value; + if (cluster_name_from_settings.empty()) + cluster_name_from_settings = getOriginalClusterName(); + return cluster_name_from_settings; +} + +QueryProcessingStage::Enum StorageObjectStorageCluster::getQueryProcessingStage( + ContextPtr context, QueryProcessingStage::Enum to_stage, const StorageSnapshotPtr & storage_snapshot, SelectQueryInfo & query_info) const +{ + /// Full query if fall back to pure storage. + if (getClusterName(context).empty()) + return QueryProcessingStage::Enum::FetchColumns; + + /// Distributed storage. + return IStorageCluster::getQueryProcessingStage(context, to_stage, storage_snapshot, query_info); +} + +std::optional StorageObjectStorageCluster::distributedWrite( + const ASTInsertQuery & query, + ContextPtr context) +{ + if (getClusterName(context).empty()) + return pure_storage->distributedWrite(query, context); + return IStorageCluster::distributedWrite(query, context); +} + +void StorageObjectStorageCluster::drop() +{ + if (pure_storage) + { + pure_storage->drop(); + return; + } + IStorageCluster::drop(); +} + +void StorageObjectStorageCluster::dropInnerTableIfAny(bool sync, ContextPtr context) +{ + if (getClusterName(context).empty()) + { + pure_storage->dropInnerTableIfAny(sync, context); + return; + } + IStorageCluster::dropInnerTableIfAny(sync, context); } +void StorageObjectStorageCluster::truncate( + const ASTPtr & query, + const StorageMetadataPtr & metadata_snapshot, + ContextPtr local_context, + TableExclusiveLockHolder & lock_holder) +{ + /// Full query if fall back to pure storage. + if (getClusterName(local_context).empty()) + { + pure_storage->truncate(query, metadata_snapshot, local_context, lock_holder); + return; + } + + throw Exception(ErrorCodes::NOT_IMPLEMENTED, "Truncate is not supported by storage {}", getName()); +} + +void StorageObjectStorageCluster::checkTableCanBeRenamed(const StorageID & new_name) const +{ + if (pure_storage) + pure_storage->checkTableCanBeRenamed(new_name); + IStorageCluster::checkTableCanBeRenamed(new_name); +} + +void StorageObjectStorageCluster::rename(const String & new_path_to_table_data, const StorageID & new_table_id) +{ + if (pure_storage) + pure_storage->rename(new_path_to_table_data, new_table_id); + IStorageCluster::rename(new_path_to_table_data, new_table_id); +} + +void StorageObjectStorageCluster::renameInMemory(const StorageID & new_table_id) +{ + if (pure_storage) + pure_storage->renameInMemory(new_table_id); + IStorageCluster::renameInMemory(new_table_id); +} + +void StorageObjectStorageCluster::alter(const AlterCommands & params, ContextPtr context, AlterLockHolder & alter_lock_holder) +{ + if (getClusterName(context).empty()) + { + pure_storage->alter(params, context, alter_lock_holder); + setInMemoryMetadata(pure_storage->getInMemoryMetadata()); + return; + } + IStorageCluster::alter(params, context, alter_lock_holder); + pure_storage->setInMemoryMetadata(IStorageCluster::getInMemoryMetadata()); +} + +void StorageObjectStorageCluster::addInferredEngineArgsToCreateQuery(ASTs & args, const ContextPtr & context) const +{ + configuration->addStructureAndFormatToArgsIfNeeded(args, "", configuration->getFormat(), context, /*with_structure=*/false); +} + +StorageMetadataPtr StorageObjectStorageCluster::getInMemoryMetadataPtr(bool bypass_metadata_cache) const +{ + if (pure_storage) + return pure_storage->getInMemoryMetadataPtr(bypass_metadata_cache); + return IStorageCluster::getInMemoryMetadataPtr(bypass_metadata_cache); +} + +IDataLakeMetadata * StorageObjectStorageCluster::getExternalMetadata(ContextPtr query_context) +{ + if (getClusterName(query_context).empty()) + return pure_storage->getExternalMetadata(query_context); + + configuration->update( + object_storage, + query_context); + + return configuration->getExternalMetadata(); +} + +void StorageObjectStorageCluster::checkAlterIsPossible(const AlterCommands & commands, ContextPtr context) const +{ + if (getClusterName(context).empty()) + { + pure_storage->checkAlterIsPossible(commands, context); + return; + } + IStorageCluster::checkAlterIsPossible(commands, context); +} + +void StorageObjectStorageCluster::checkMutationIsPossible(const MutationCommands & commands, const Settings & settings) const +{ + if (pure_storage) + { + pure_storage->checkMutationIsPossible(commands, settings); + return; + } + IStorageCluster::checkMutationIsPossible(commands, settings); +} + +Pipe StorageObjectStorageCluster::alterPartition( + const StorageMetadataPtr & metadata_snapshot, + const PartitionCommands & commands, + ContextPtr context) +{ + if (getClusterName(context).empty()) + return pure_storage->alterPartition(metadata_snapshot, commands, context); + return IStorageCluster::alterPartition(metadata_snapshot, commands, context); +} + +void StorageObjectStorageCluster::checkAlterPartitionIsPossible( + const PartitionCommands & commands, + const StorageMetadataPtr & metadata_snapshot, + const Settings & settings, + ContextPtr context) const +{ + if (getClusterName(context).empty()) + { + pure_storage->checkAlterPartitionIsPossible(commands, metadata_snapshot, settings, context); + return; + } + IStorageCluster::checkAlterPartitionIsPossible(commands, metadata_snapshot, settings, context); +} + +bool StorageObjectStorageCluster::optimize( + const ASTPtr & query, + const StorageMetadataPtr & metadata_snapshot, + const ASTPtr & partition, + bool final, + bool deduplicate, + const Names & deduplicate_by_columns, + bool cleanup, + ContextPtr context) +{ + if (getClusterName(context).empty()) + return pure_storage->optimize(query, metadata_snapshot, partition, final, deduplicate, deduplicate_by_columns, cleanup, context); + return IStorageCluster::optimize(query, metadata_snapshot, partition, final, deduplicate, deduplicate_by_columns, cleanup, context); +} + +QueryPipeline StorageObjectStorageCluster::updateLightweight(const MutationCommands & commands, ContextPtr context) +{ + if (getClusterName(context).empty()) + return pure_storage->updateLightweight(commands, context); + return IStorageCluster::updateLightweight(commands, context); +} + +void StorageObjectStorageCluster::mutate(const MutationCommands & commands, ContextPtr context) +{ + if (getClusterName(context).empty()) + { + pure_storage->mutate(commands, context); + return; + } + IStorageCluster::mutate(commands, context); +} + +CancellationCode StorageObjectStorageCluster::killMutation(const String & mutation_id) +{ + if (pure_storage) + return pure_storage->killMutation(mutation_id); + return IStorageCluster::killMutation(mutation_id); +} + +void StorageObjectStorageCluster::waitForMutation(const String & mutation_id, bool wait_for_another_mutation) +{ + if (pure_storage) + { + pure_storage->waitForMutation(mutation_id, wait_for_another_mutation); + return; + } + IStorageCluster::waitForMutation(mutation_id, wait_for_another_mutation); +} + +void StorageObjectStorageCluster::setMutationCSN(const String & mutation_id, UInt64 csn) +{ + if (pure_storage) + { + pure_storage->setMutationCSN(mutation_id, csn); + return; + } + IStorageCluster::setMutationCSN(mutation_id, csn); +} + +CancellationCode StorageObjectStorageCluster::killPartMoveToShard(const UUID & task_uuid) +{ + if (pure_storage) + return pure_storage->killPartMoveToShard(task_uuid); + return IStorageCluster::killPartMoveToShard(task_uuid); +} + +void StorageObjectStorageCluster::startup() +{ + if (pure_storage) + { + pure_storage->startup(); + return; + } + IStorageCluster::startup(); +} + +void StorageObjectStorageCluster::shutdown(bool is_drop) +{ + if (pure_storage) + { + pure_storage->shutdown(is_drop); + return; + } + IStorageCluster::shutdown(is_drop); +} + +void StorageObjectStorageCluster::flushAndPrepareForShutdown() +{ + if (pure_storage) + { + pure_storage->flushAndPrepareForShutdown(); + return; + } + IStorageCluster::flushAndPrepareForShutdown(); +} + +ActionLock StorageObjectStorageCluster::getActionLock(StorageActionBlockType action_type) +{ + if (pure_storage) + return pure_storage->getActionLock(action_type); + return IStorageCluster::getActionLock(action_type); +} + +void StorageObjectStorageCluster::onActionLockRemove(StorageActionBlockType action_type) +{ + if (pure_storage) + { + pure_storage->onActionLockRemove(action_type); + return; + } + IStorageCluster::onActionLockRemove(action_type); +} + +bool StorageObjectStorageCluster::supportsDelete() const +{ + if (pure_storage) + return pure_storage->supportsDelete(); + return IStorageCluster::supportsDelete(); +} + +bool StorageObjectStorageCluster::supportsParallelInsert() const +{ + if (pure_storage) + return pure_storage->supportsParallelInsert(); + return IStorageCluster::supportsParallelInsert(); +} + +bool StorageObjectStorageCluster::prefersLargeBlocks() const +{ + if (pure_storage) + return pure_storage->prefersLargeBlocks(); + return IStorageCluster::prefersLargeBlocks(); +} + +bool StorageObjectStorageCluster::supportsPartitionBy() const +{ + if (pure_storage) + return pure_storage->supportsPartitionBy(); + return IStorageCluster::supportsPartitionBy(); +} + +bool StorageObjectStorageCluster::supportsSubcolumns() const +{ + if (pure_storage) + return pure_storage->supportsSubcolumns(); + return IStorageCluster::supportsSubcolumns(); +} + +bool StorageObjectStorageCluster::supportsTrivialCountOptimization(const StorageSnapshotPtr & snapshot, ContextPtr context) const +{ + if (pure_storage) + return pure_storage->supportsTrivialCountOptimization(snapshot, context); + return IStorageCluster::supportsTrivialCountOptimization(snapshot, context); +} + +bool StorageObjectStorageCluster::supportsPrewhere() const +{ + if (pure_storage) + return pure_storage->supportsPrewhere(); + return IStorageCluster::supportsPrewhere(); +} + +bool StorageObjectStorageCluster::canMoveConditionsToPrewhere() const +{ + if (pure_storage) + return pure_storage->canMoveConditionsToPrewhere(); + return IStorageCluster::canMoveConditionsToPrewhere(); +} + +std::optional StorageObjectStorageCluster::supportedPrewhereColumns() const +{ + if (pure_storage) + return pure_storage->supportedPrewhereColumns(); + return IStorageCluster::supportedPrewhereColumns(); +} + +IStorageCluster::ColumnSizeByName StorageObjectStorageCluster::getColumnSizes() const +{ + if (pure_storage) + return pure_storage->getColumnSizes(); + return IStorageCluster::getColumnSizes(); +} + +bool StorageObjectStorageCluster::parallelizeOutputAfterReading(ContextPtr context) const +{ + if (pure_storage) + return pure_storage->parallelizeOutputAfterReading(context); + return IStorageCluster::parallelizeOutputAfterReading(context); +} + +Pipe StorageObjectStorageCluster::executeCommand(const String & command_name, const ASTPtr & args, ContextPtr context) +{ + if (pure_storage) + return pure_storage->executeCommand(command_name, args, context); + return IStorageCluster::executeCommand(command_name, args, context); +} + +bool StorageObjectStorageCluster::supportsImport(ContextPtr context) const +{ + if (pure_storage) + return pure_storage->supportsImport(context); + return IStorageCluster::supportsImport(context); +} + +SinkToStoragePtr StorageObjectStorageCluster::import( + const std::string & file_name, + Block & block_with_partition_values, + const std::function & new_file_path_callback, + bool overwrite_if_exists, + std::size_t max_bytes_per_file, + std::size_t max_rows_per_file, + const std::optional & iceberg_metadata_json_string, + const std::optional & format_settings_, + ContextPtr context) +{ + if (pure_storage) + return pure_storage->import( + file_name, + block_with_partition_values, + new_file_path_callback, + overwrite_if_exists, + max_bytes_per_file, + max_rows_per_file, + iceberg_metadata_json_string, + format_settings_, + context); + return IStorageCluster::import( + file_name, + block_with_partition_values, + new_file_path_callback, + overwrite_if_exists, + max_bytes_per_file, + max_rows_per_file, + iceberg_metadata_json_string, + format_settings_, + context); +} + +void StorageObjectStorageCluster::commitExportPartitionTransaction( + const String & transaction_id, + const String & partition_id, + const Strings & exported_paths, + const IcebergCommitExportPartitionArguments & iceberg_commit_export_partition_arguments, + ContextPtr local_context) +{ + if (pure_storage) + { + pure_storage->commitExportPartitionTransaction( + transaction_id, + partition_id, + exported_paths, + iceberg_commit_export_partition_arguments, + local_context + ); + return; + } + IStorageCluster::commitExportPartitionTransaction( + transaction_id, + partition_id, + exported_paths, + iceberg_commit_export_partition_arguments, + local_context + ); +} + +} diff --git a/src/Storages/ObjectStorage/StorageObjectStorageCluster.h b/src/Storages/ObjectStorage/StorageObjectStorageCluster.h index 2d7c4a660077..c168af53c28b 100644 --- a/src/Storages/ObjectStorage/StorageObjectStorageCluster.h +++ b/src/Storages/ObjectStorage/StorageObjectStorageCluster.h @@ -18,11 +18,39 @@ class StorageObjectStorageCluster : public IStorageCluster const ColumnsDescription & columns_in_table_or_function_definition, const ConstraintsDescription & constraints_, const ASTPtr & partition_by, + const ASTPtr & order_by, ContextPtr context_, - bool is_table_function_ = false); + const String & comment_, + std::optional format_settings_, + LoadingStrictnessLevel mode_, + std::shared_ptr catalog, + bool if_not_exists, + bool is_datalake_query, + bool is_table_function_ = false, + bool lazy_init = false); std::string getName() const override; + bool supportsImport(ContextPtr context) const override; + + SinkToStoragePtr import( + const std::string & file_name, + Block & block_with_partition_values, + const std::function & new_file_path_callback, + bool overwrite_if_exists, + std::size_t max_bytes_per_file, + std::size_t max_rows_per_file, + const std::optional & iceberg_metadata_json_string, + const std::optional & format_settings_, + ContextPtr context) override; + + void commitExportPartitionTransaction( + const String & transaction_id, + const String & partition_id, + const Strings & exported_paths, + const IcebergCommitExportPartitionArguments & iceberg_commit_export_partition_arguments, + ContextPtr local_context) override; + RemoteQueryExecutor::Extension getTaskIteratorExtension( const ActionsDAG::Node * predicate, const ActionsDAG * filter, @@ -34,19 +62,162 @@ class StorageObjectStorageCluster : public IStorageCluster std::optional totalRows(ContextPtr query_context) const override; std::optional totalBytes(ContextPtr query_context) const override; + void setClusterNameInSettings(bool cluster_name_in_settings_) { cluster_name_in_settings = cluster_name_in_settings_; } + + String getClusterName(ContextPtr context) const override; + + QueryProcessingStage::Enum getQueryProcessingStage(ContextPtr, QueryProcessingStage::Enum, const StorageSnapshotPtr &, SelectQueryInfo &) const override; + + std::optional distributedWrite( + const ASTInsertQuery & query, + ContextPtr context) override; + + void drop() override; + + void dropInnerTableIfAny(bool sync, ContextPtr context) override; + + void truncate( + const ASTPtr & query, + const StorageMetadataPtr & metadata_snapshot, + ContextPtr local_context, + TableExclusiveLockHolder &) override; + + void checkTableCanBeRenamed(const StorageID & new_name) const override; + + void rename(const String & new_path_to_table_data, const StorageID & new_table_id) override; + + void renameInMemory(const StorageID & new_table_id) override; + + void alter(const AlterCommands & params, ContextPtr context, AlterLockHolder & alter_lock_holder) override; + + void addInferredEngineArgsToCreateQuery(ASTs & args, const ContextPtr & context) const override; + + IDataLakeMetadata * getExternalMetadata(ContextPtr query_context); + + StorageMetadataPtr getInMemoryMetadataPtr(bool bypass_metadata_cache = false) const override; + + void checkAlterIsPossible(const AlterCommands & commands, ContextPtr context) const override; + + void checkMutationIsPossible(const MutationCommands & commands, const Settings & settings) const override; + + Pipe alterPartition( + const StorageMetadataPtr & metadata_snapshot, + const PartitionCommands & commands, + ContextPtr context) override; + + void checkAlterPartitionIsPossible( + const PartitionCommands & commands, + const StorageMetadataPtr & metadata_snapshot, + const Settings & settings, + ContextPtr context) const override; + + bool optimize( + const ASTPtr & query, + const StorageMetadataPtr & metadata_snapshot, + const ASTPtr & partition, + bool final, + bool deduplicate, + const Names & deduplicate_by_columns, + bool cleanup, + ContextPtr context) override; + + QueryPipeline updateLightweight(const MutationCommands & commands, ContextPtr context) override; + + void mutate(const MutationCommands & commands, ContextPtr context) override; + + Pipe executeCommand(const String & command_name, const ASTPtr & args, ContextPtr context) override; + + CancellationCode killMutation(const String & mutation_id) override; + + void waitForMutation(const String & mutation_id, bool wait_for_another_mutation) override; + + void setMutationCSN(const String & mutation_id, UInt64 csn) override; + + CancellationCode killPartMoveToShard(const UUID & task_uuid) override; + + void startup() override; + + void shutdown(bool is_drop = false) override; + + void flushAndPrepareForShutdown() override; + + ActionLock getActionLock(StorageActionBlockType action_type) override; + + void onActionLockRemove(StorageActionBlockType action_type) override; void updateExternalDynamicMetadataIfExists(ContextPtr query_context) override; + bool supportsDelete() const override; + + bool supportsParallelInsert() const override; + + bool prefersLargeBlocks() const override; + + bool supportsPartitionBy() const override; + + bool supportsSubcolumns() const override; + + bool supportsTrivialCountOptimization(const StorageSnapshotPtr &, ContextPtr) const override; + + /// Things required for PREWHERE. + bool supportsPrewhere() const override; + bool canMoveConditionsToPrewhere() const override; + std::optional supportedPrewhereColumns() const override; + ColumnSizeByName getColumnSizes() const override; + + bool parallelizeOutputAfterReading(ContextPtr context) const override; + + bool isObjectStorage() const override { return true; } + + bool isDataLake() const override { return configuration->isDataLakeConfiguration(); } + private: void updateQueryToSendIfNeeded( ASTPtr & query, const StorageSnapshotPtr & storage_snapshot, - const ContextPtr & context) override; + const ContextPtr & context, + bool make_cluster_function) override; + + bool isClusterSupported() const override; + + void readFallBackToPure( + QueryPlan & query_plan, + const Names & column_names, + const StorageSnapshotPtr & storage_snapshot, + SelectQueryInfo & query_info, + ContextPtr context, + QueryProcessingStage::Enum processed_stage, + size_t max_block_size, + size_t num_streams) override; + + SinkToStoragePtr writeFallBackToPure( + const ASTPtr & query, + const StorageMetadataPtr & metadata_snapshot, + ContextPtr context, + bool async_insert) override; + + /* + In case the table was created with `object_storage_cluster` setting, + modify the AST query object so that it uses the table function implementation + by mapping the engine name to table function name and setting `object_storage_cluster`. + For table like + CREATE TABLE table ENGINE=S3(...) SETTINGS object_storage_cluster='cluster' + coverts request + SELECT * FROM table + to + SELECT * FROM s3(...) SETTINGS object_storage_cluster='cluster' + to make distributed request over cluster 'cluster'. + */ + void updateQueryForDistributedEngineIfNeeded(ASTPtr & query, ContextPtr context, bool make_cluster_function); const String engine_name; - const StorageObjectStorageConfigurationPtr configuration; + StorageObjectStorageConfigurationPtr configuration; const ObjectStoragePtr object_storage; NamesAndTypesList hive_partition_columns_to_read_from_file_path; + bool cluster_name_in_settings; + + /// non-clustered storage to fall back on pure realisation if needed + std::shared_ptr pure_storage; }; } diff --git a/src/Storages/ObjectStorage/StorageObjectStorageConfiguration.cpp b/src/Storages/ObjectStorage/StorageObjectStorageConfiguration.cpp index 9e33987cae0e..214088660607 100644 --- a/src/Storages/ObjectStorage/StorageObjectStorageConfiguration.cpp +++ b/src/Storages/ObjectStorage/StorageObjectStorageConfiguration.cpp @@ -95,80 +95,79 @@ bool StorageObjectStorageConfiguration::shouldReloadSchemaForConsistency(Context void StorageObjectStorageConfiguration::initialize( - StorageObjectStorageConfiguration & configuration_to_initialize, ASTs & engine_args, ContextPtr local_context, bool with_table_structure, const StorageID * table_id) { std::string disk_name; - if (configuration_to_initialize.isDataLakeConfiguration()) + if (isDataLakeConfiguration()) { - const auto & storage_settings = configuration_to_initialize.getDataLakeSettings(); + const auto & storage_settings = getDataLakeSettings(); disk_name = storage_settings[DataLakeStorageSetting::disk].changed ? storage_settings[DataLakeStorageSetting::disk].value : ""; } if (!disk_name.empty()) - configuration_to_initialize.fromDisk(disk_name, engine_args, local_context, with_table_structure); + fromDisk(disk_name, engine_args, local_context, with_table_structure); else if (auto named_collection = tryGetNamedCollectionWithOverrides(engine_args, local_context, true, nullptr, table_id)) - configuration_to_initialize.fromNamedCollection(*named_collection, local_context); + fromNamedCollection(*named_collection, local_context); else - configuration_to_initialize.fromAST(engine_args, local_context, with_table_structure); + fromAST(engine_args, local_context, with_table_structure); - if (configuration_to_initialize.isNamespaceWithGlobs()) + if (isNamespaceWithGlobs()) throw Exception(ErrorCodes::BAD_ARGUMENTS, - "Expression can not have wildcards inside {} name", configuration_to_initialize.getNamespaceType()); + "Expression can not have wildcards inside {} name", getNamespaceType()); - if (configuration_to_initialize.isDataLakeConfiguration()) + if (isDataLakeConfiguration()) { - if (configuration_to_initialize.partition_strategy_type != PartitionStrategyFactory::StrategyType::NONE) + if (getPartitionStrategyType() != PartitionStrategyFactory::StrategyType::NONE) { throw Exception(ErrorCodes::BAD_ARGUMENTS, "The `partition_strategy` argument is incompatible with data lakes"); } } - else if (configuration_to_initialize.partition_strategy_type == PartitionStrategyFactory::StrategyType::NONE) + else if (getPartitionStrategyType() == PartitionStrategyFactory::StrategyType::NONE) { - if (configuration_to_initialize.getRawPath().hasPartitionWildcard()) + if (getRawPath().hasPartitionWildcard()) { // Promote to wildcard in case it is not data lake to make it backwards compatible - configuration_to_initialize.partition_strategy_type = PartitionStrategyFactory::StrategyType::WILDCARD; + setPartitionStrategyType(PartitionStrategyFactory::StrategyType::WILDCARD); } } - if (configuration_to_initialize.format == "auto") + if (format == "auto") { - if (configuration_to_initialize.isDataLakeConfiguration()) + if (isDataLakeConfiguration()) { - configuration_to_initialize.format = "Parquet"; + format = "Parquet"; } else { - configuration_to_initialize.format + format = FormatFactory::instance() - .tryGetFormatFromFileName(configuration_to_initialize.isArchive() ? configuration_to_initialize.getPathInArchive() : configuration_to_initialize.getRawPath().path) + .tryGetFormatFromFileName(isArchive() ? getPathInArchive() : getRawPath().path) .value_or("auto"); } } else - FormatFactory::instance().checkFormatName(configuration_to_initialize.format); + FormatFactory::instance().checkFormatName(format); - if (configuration_to_initialize.partition_strategy_type == PartitionStrategyFactory::StrategyType::HIVE) + if (partition_strategy_type == PartitionStrategyFactory::StrategyType::HIVE) { - configuration_to_initialize.file_path_generator = std::make_shared( - configuration_to_initialize.getRawPath().path, - configuration_to_initialize.format); + file_path_generator = std::make_shared( + getRawPath().path, + format); } else { - configuration_to_initialize.file_path_generator = std::make_shared(configuration_to_initialize.getRawPath().path); + file_path_generator = std::make_shared(getRawPath().path); } /// We shouldn't set path for disk setup because path prefix is already set in used object_storage. if (disk_name.empty()) - configuration_to_initialize.read_path = configuration_to_initialize.file_path_generator->getPathForRead(); + read_path = file_path_generator->getPathForRead(); - configuration_to_initialize.initialized = true; + initialized = true; } String StorageObjectStorageConfiguration::computeSchemaHash(const ColumnsDescription & columns) diff --git a/src/Storages/ObjectStorage/StorageObjectStorageConfiguration.h b/src/Storages/ObjectStorage/StorageObjectStorageConfiguration.h index 8b9ad1c7ebeb..f62628e19844 100644 --- a/src/Storages/ObjectStorage/StorageObjectStorageConfiguration.h +++ b/src/Storages/ObjectStorage/StorageObjectStorageConfiguration.h @@ -88,8 +88,7 @@ class StorageObjectStorageConfiguration using Paths = std::vector; /// Initialize configuration from either AST or NamedCollection. - static void initialize( - StorageObjectStorageConfiguration & configuration_to_initialize, + virtual void initialize( ASTs & engine_args, ContextPtr local_context, bool with_table_structure, @@ -112,13 +111,13 @@ class StorageObjectStorageConfiguration /// Raw URI, specified by a user. Used in permission check. virtual const String & getRawURI() const = 0; - const Path & getPathForRead() const; + virtual const Path & getPathForRead() const; // Path used for writing, it should not be globbed and might contain a partition key - Path getPathForWrite(const std::string & partition_id = "") const; - Path getPathForWrite(const std::string & partition_id, const std::string & filename_override) const; + virtual Path getPathForWrite(const std::string & partition_id = "") const; + virtual Path getPathForWrite(const std::string & partition_id, const std::string & filename_override) const; - void setPathForRead(const Path & path) + virtual void setPathForRead(const Path & path) { read_path = path; } @@ -140,10 +139,10 @@ class StorageObjectStorageConfiguration virtual void addStructureAndFormatToArgsIfNeeded( ASTs & args, const String & structure_, const String & format_, ContextPtr context, bool with_structure) = 0; - bool isNamespaceWithGlobs() const; + virtual bool isNamespaceWithGlobs() const; virtual bool isArchive() const { return false; } - bool isPathInArchiveWithGlobs() const; + virtual bool isPathInArchiveWithGlobs() const; virtual std::string getPathInArchive() const; virtual void check(ContextPtr context); @@ -189,9 +188,9 @@ class StorageObjectStorageConfiguration const PrepareReadingFromFormatHiveParams & hive_parameters); static String computeSchemaHash(const ColumnsDescription & columns); - void setSchemaHash(const String & hash); + virtual void setSchemaHash(const String & hash); - void initPartitionStrategy(ASTPtr partition_by, const ColumnsDescription & columns, ContextPtr context); + virtual void initPartitionStrategy(ASTPtr partition_by, const ColumnsDescription & columns, ContextPtr context); virtual std::optional getTableStateSnapshot(ContextPtr local_context) const; virtual std::unique_ptr buildStorageMetadataFromState(const DataLakeTableStateSnapshot & state, ContextPtr local_context) const; @@ -276,6 +275,49 @@ class StorageObjectStorageConfiguration throw Exception(ErrorCodes::NOT_IMPLEMENTED, "Method getDataLakeSettings() is not implemented for configuration type {}", getTypeName()); } + /// Create arguments for table function with path and access parameters + virtual ASTPtr createArgsWithAccessData() const + { + throw Exception(ErrorCodes::NOT_IMPLEMENTED, "Method createArgsWithAccessData is not supported by storage {}", getEngineName()); + } + + virtual void fromNamedCollection(const NamedCollection & collection, ContextPtr context) = 0; + virtual void fromAST(ASTs & args, ContextPtr context, bool with_structure) = 0; + virtual void fromDisk(const String & /*disk_name*/, ASTs & /*args*/, ContextPtr /*context*/, bool /*with_structure*/) + { + throw Exception(ErrorCodes::NOT_IMPLEMENTED, "method fromDisk is not implemented"); + } + + virtual ObjectStorageType extractDynamicStorageType(ASTs & /* args */, ContextPtr /* context */, ASTPtr * /* type_arg */, bool /* cluster_name_first */) const + { return ObjectStorageType::None; } + + virtual const String & getFormat() const { return format; } + virtual const String & getCompressionMethod() const { return compression_method; } + virtual const String & getStructure() const { return structure; } + + virtual PartitionStrategyFactory::StrategyType getPartitionStrategyType() const { return partition_strategy_type; } + virtual bool getPartitionColumnsInDataFile() const { return partition_columns_in_data_file; } + virtual std::shared_ptr getPartitionStrategy() const { return partition_strategy; } + + virtual void setFormat(const String & format_) { format = format_; } + virtual void setCompressionMethod(const String & compression_method_) { compression_method = compression_method_; } + virtual void setStructure(const String & structure_) { structure = structure_; } + + virtual void setPartitionStrategyType(PartitionStrategyFactory::StrategyType partition_strategy_type_) + { + partition_strategy_type = partition_strategy_type_; + } + virtual void setPartitionColumnsInDataFile(bool partition_columns_in_data_file_) + { + partition_columns_in_data_file = partition_columns_in_data_file_; + } + virtual void setPartitionStrategy(const std::shared_ptr & partition_strategy_) + { + partition_strategy = partition_strategy_; + } + + virtual void assertInitialized() const; + virtual ColumnMapperPtr getColumnMapperForObject(ObjectInfoPtr /**/) const { return nullptr; } virtual ColumnMapperPtr getColumnMapperForCurrentSchema(StorageMetadataPtr /**/, ContextPtr /**/) const { return nullptr; } @@ -298,6 +340,7 @@ class StorageObjectStorageConfiguration virtual void drop(ContextPtr) {} +<<<<<<< HEAD virtual bool isBackgroundExecutable() const { return false; @@ -315,6 +358,11 @@ class StorageObjectStorageConfiguration return 0; } +======= + virtual bool isClusterSupported() const { return true; } + +private: +>>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) String format = "auto"; String compression_method = "auto"; String structure = "auto"; @@ -331,14 +379,6 @@ class StorageObjectStorageConfiguration protected: void initializeFromParsedArguments(const StorageParsedArguments & parsed_arguments); - virtual void fromNamedCollection(const NamedCollection & collection, ContextPtr context) = 0; - virtual void fromAST(ASTs & args, ContextPtr context, bool with_structure) = 0; - virtual void fromDisk(const String & /*disk_name*/, ASTs & /*args*/, ContextPtr /*context*/, bool /*with_structure*/) - { - throw Exception(ErrorCodes::NOT_IMPLEMENTED, "method fromDisk is not implemented"); - } - - void assertInitialized() const; bool initialized = false; String schema_hash; diff --git a/src/Storages/ObjectStorage/StorageObjectStorageSettings.h b/src/Storages/ObjectStorage/StorageObjectStorageSettings.h index 82b443b38fd6..543ccc2eb615 100644 --- a/src/Storages/ObjectStorage/StorageObjectStorageSettings.h +++ b/src/Storages/ObjectStorage/StorageObjectStorageSettings.h @@ -70,7 +70,17 @@ struct StorageObjectStorageSettings using StorageObjectStorageSettingsPtr = std::shared_ptr; +// clang-format off + +#define STORAGE_OBJECT_STORAGE_RELATED_SETTINGS(DECLARE, ALIAS) \ + DECLARE(String, object_storage_cluster, "", R"( +Cluster for distributed requests +)", 0) \ + +// clang-format on + #define LIST_OF_STORAGE_OBJECT_STORAGE_SETTINGS(M, ALIAS) \ + STORAGE_OBJECT_STORAGE_RELATED_SETTINGS(M, ALIAS) \ LIST_OF_ALL_FORMAT_SETTINGS(M, ALIAS) } diff --git a/src/Storages/ObjectStorage/StorageObjectStorageSink.cpp b/src/Storages/ObjectStorage/StorageObjectStorageSink.cpp index 0669c4e4524d..ed443accbe79 100644 --- a/src/Storages/ObjectStorage/StorageObjectStorageSink.cpp +++ b/src/Storages/ObjectStorage/StorageObjectStorageSink.cpp @@ -183,10 +183,10 @@ SinkPtr PartitionedStorageObjectStorageSink::createSinkForPartition(const String file_path, object_storage, format_settings, - std::make_shared(configuration->partition_strategy->getFormatHeader()), + std::make_shared(configuration->getPartitionStrategy()->getFormatHeader()), context, - configuration->format, - configuration->compression_method + configuration->getFormat(), + configuration->getCompressionMethod() ); } diff --git a/src/Storages/ObjectStorage/StorageObjectStorageSource.cpp b/src/Storages/ObjectStorage/StorageObjectStorageSource.cpp index 088695b2ee46..627c30e1ff4c 100644 --- a/src/Storages/ObjectStorage/StorageObjectStorageSource.cpp +++ b/src/Storages/ObjectStorage/StorageObjectStorageSource.cpp @@ -646,7 +646,7 @@ void StorageObjectStorageSource::addNumRowsToCache(const ObjectInfo & object_inf { const auto cache_key = getKeyForSchemaCache( getUniqueStoragePathIdentifier(*configuration, object_info), - object_info.getFileFormat().value_or(configuration->format), + object_info.getFileFormat().value_or(configuration->getFormat()), format_settings, read_context); schema_cache.addNumRows(cache_key, num_rows); @@ -770,7 +770,7 @@ StorageObjectStorageSource::ReaderHolder StorageObjectStorageSource::createReade const auto cache_key = getKeyForSchemaCache( getUniqueStoragePathIdentifier(*configuration, *object_info), - object_info->getFileFormat().value_or(configuration->format), + object_info->getFileFormat().value_or(configuration->getFormat()), format_settings, context_); @@ -808,6 +808,7 @@ StorageObjectStorageSource::ReaderHolder StorageObjectStorageSource::createReade CompressionMethod compression_method = {}; if (input_format_does_not_read_file) { +<<<<<<< HEAD /// `One` produces a single row per object without consuming the underlying `ReadBuffer`. read_buf = std::make_unique(); compression_method = CompressionMethod::None; @@ -816,13 +817,20 @@ StorageObjectStorageSource::ReaderHolder StorageObjectStorageSource::createReade { ProfileEvents::increment(ProfileEvents::ObjectStorageReadObjects); compression_method = chooseCompressionMethod(configuration->getPathInArchive(), configuration->compression_method); +======= + compression_method = chooseCompressionMethod(configuration->getPathInArchive(), configuration->getCompressionMethod()); +>>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) const auto & archive_reader = object_info_in_archive->archive_reader; read_buf = archive_reader->readFile(object_info_in_archive->path_in_archive, /*throw_on_not_found=*/true); } else { +<<<<<<< HEAD ProfileEvents::increment(ProfileEvents::ObjectStorageReadObjects); compression_method = chooseCompressionMethod(object_info->getFileName(), configuration->compression_method); +======= + compression_method = chooseCompressionMethod(object_info->getFileName(), configuration->getCompressionMethod()); +>>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) read_buf = createReadBuffer(object_info->relative_path_with_metadata, object_storage, context_, log); } @@ -930,20 +938,33 @@ StorageObjectStorageSource::ReaderHolder StorageObjectStorageSource::createReade "Reading object '{}', size: {} bytes, with format: {}", object_info->getPath(), object_info->getObjectMetadata()->size_bytes, +<<<<<<< HEAD format_name); +======= + object_info->getFileFormat().value_or(configuration->getFormat())); +>>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) logIcebergFileStats(object_info, log); InputFormatPtr input_format; +<<<<<<< HEAD if (context_->getSettingsRef()[Setting::use_parquet_metadata_cache] && (Poco::toLower(format_name) == "parquet") +======= + if (context_->getSettingsRef()[Setting::use_parquet_metadata_cache] && use_native_reader_v3 + && (object_info->getFileFormat().value_or(configuration->getFormat()) == "Parquet") +>>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) && !object_info->getObjectMetadata()->etag.empty()) { std::optional object_with_metadata = object_info->relative_path_with_metadata; if (object_info->isArchive()) object_with_metadata->relative_path = object_info->getPath(); input_format = FormatFactory::instance().getInputWithMetadata( +<<<<<<< HEAD format_name, +======= + object_info->getFileFormat().value_or(configuration->getFormat()), +>>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) *read_buf, initial_header, context_, @@ -962,7 +983,11 @@ StorageObjectStorageSource::ReaderHolder StorageObjectStorageSource::createReade else { input_format = FormatFactory::instance().getInput( +<<<<<<< HEAD format_name, +======= + object_info->getFileFormat().value_or(configuration->getFormat()), +>>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) *read_buf, initial_header, context_, diff --git a/src/Storages/ObjectStorage/Utils.cpp b/src/Storages/ObjectStorage/Utils.cpp index 8fa02319c9ac..8443bfeb0370 100644 --- a/src/Storages/ObjectStorage/Utils.cpp +++ b/src/Storages/ObjectStorage/Utils.cpp @@ -71,14 +71,13 @@ std::optional checkAndGetNewFileOnInsertIfNeeded( void resolveSchemaAndFormat( ColumnsDescription & columns, - std::string & format, ObjectStoragePtr object_storage, - const StorageObjectStorageConfigurationPtr & configuration, + StorageObjectStorageConfigurationPtr & configuration, std::optional format_settings, std::string & sample_path, const ContextPtr & context) { - if (format == "auto") + if (configuration->getFormat() == "auto") { if (configuration->isDataLakeConfiguration()) { @@ -100,21 +99,23 @@ void resolveSchemaAndFormat( if (columns.empty()) { - if (format == "auto") + if (configuration->getFormat() == "auto") { + std::string format; std::tie(columns, format) = StorageObjectStorage::resolveSchemaAndFormatFromData( object_storage, configuration, format_settings, sample_path, context); + configuration->setFormat(format); } else { - chassert(!format.empty()); + chassert(!configuration->getFormat().empty()); columns = StorageObjectStorage::resolveSchemaFromData(object_storage, configuration, format_settings, sample_path, context); } } } - else if (format == "auto") + else if (configuration->getFormat() == "auto") { - format = StorageObjectStorage::resolveFormatFromData(object_storage, configuration, format_settings, sample_path, context); + configuration->setFormat(StorageObjectStorage::resolveFormatFromData(object_storage, configuration, format_settings, sample_path, context)); } validateSupportedColumns(columns, *configuration); diff --git a/src/Storages/ObjectStorage/Utils.h b/src/Storages/ObjectStorage/Utils.h index 931ebcbed9ac..4dd7495cebab 100644 --- a/src/Storages/ObjectStorage/Utils.h +++ b/src/Storages/ObjectStorage/Utils.h @@ -18,9 +18,8 @@ std::optional checkAndGetNewFileOnInsertIfNeeded( void resolveSchemaAndFormat( ColumnsDescription & columns, - std::string & format, ObjectStoragePtr object_storage, - const StorageObjectStorageConfigurationPtr & configuration, + StorageObjectStorageConfigurationPtr & configuration, std::optional format_settings, std::string & sample_path, const ContextPtr & context); diff --git a/src/Storages/ObjectStorage/registerStorageObjectStorage.cpp b/src/Storages/ObjectStorage/registerStorageObjectStorage.cpp index b1ca240b9689..2636e201fcfa 100644 --- a/src/Storages/ObjectStorage/registerStorageObjectStorage.cpp +++ b/src/Storages/ObjectStorage/registerStorageObjectStorage.cpp @@ -14,7 +14,11 @@ #include #include #include +<<<<<<< HEAD #include +======= +#include +>>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) #include #include #include @@ -45,11 +49,20 @@ namespace // LocalObjectStorage is only supported for Iceberg Datalake operations where Avro format is required. For regular file access, use FileStorage instead. #if USE_AWS_S3 || USE_AZURE_BLOB_STORAGE || USE_HDFS || USE_AVRO -std::shared_ptr +StoragePtr createStorageObjectStorage(const StorageFactory::Arguments & args, StorageObjectStorageConfigurationPtr configuration) { const auto context = args.getLocalContext(); - StorageObjectStorageConfiguration::initialize(*configuration, args.engine_args, context, false, &args.table_id); + + std::string cluster_name; + + if (args.storage_def->settings) + { + if (const auto * value = args.storage_def->settings->changes.tryGet("object_storage_cluster")) + cluster_name = value->safeGet(); + } + + configuration->initialize(args.engine_args, context, false, &args.table_id); // Use format settings from global server context + settings from // the SETTINGS clause of the create query. Settings from current @@ -80,24 +93,26 @@ createStorageObjectStorage(const StorageFactory::Arguments & args, StorageObject ContextMutablePtr context_copy = Context::createCopy(args.getContext()); Settings settings_copy = args.getLocalContext()->getSettingsCopy(); context_copy->setSettings(settings_copy); - return std::make_shared( + return std::make_shared( + cluster_name, configuration, // We only want to perform write actions (e.g. create a container in Azure) when the table is being created, // and we want to avoid it when we load the table after a server restart. configuration->createObjectStorage(context, /* is_readonly */ args.mode != LoadingStrictnessLevel::CREATE, std::nullopt), - context_copy, /// Use global context. args.table_id, args.columns, args.constraints, + partition_by, + order_by, + context_copy, /// Use global context. args.comment, format_settings, args.mode, configuration->getCatalog(context, args.table_id), args.query.if_not_exists, - /* is_datalake_query*/ false, - /* distributed_processing */ false, - partition_by, - order_by); + /* is_datalake_query */ false, + /* is_table_function */ false, + /* lazy_init */ false); } #endif @@ -1074,9 +1089,8 @@ void registerStorageIceberg(StorageFactory & factory) } } else -#if USE_AWS_S3 - configuration = std::make_shared(storage_settings); -#endif + configuration = std::make_shared(storage_settings); + if (configuration == nullptr) { throw Exception(ErrorCodes::BAD_ARGUMENTS, "This storage configuration is not available at this build"); @@ -2078,7 +2092,7 @@ Data types supported in Paimon partition keys: void registerStorageDeltaLake(StorageFactory & factory); void registerStorageDeltaLake(StorageFactory & factory) { -#if USE_AWS_S3 +# if USE_AWS_S3 factory.registerStorage( DeltaLakeDefinition::storage_engine_name, [&](const StorageFactory::Arguments & args) diff --git a/src/Storages/ObjectStorageQueue/StorageObjectStorageQueue.cpp b/src/Storages/ObjectStorageQueue/StorageObjectStorageQueue.cpp index 343fb3b8408b..8a0afc74bf28 100644 --- a/src/Storages/ObjectStorageQueue/StorageObjectStorageQueue.cpp +++ b/src/Storages/ObjectStorageQueue/StorageObjectStorageQueue.cpp @@ -327,12 +327,12 @@ StorageObjectStorageQueue::StorageObjectStorageQueue( validateSettings(*queue_settings_, is_attach); object_storage = configuration->createObjectStorage(context_, /* is_readonly */true, std::nullopt); - FormatFactory::instance().checkFormatName(configuration->format); + FormatFactory::instance().checkFormatName(configuration->getFormat()); configuration->check(context_); ColumnsDescription columns{columns_}; std::string sample_path; - resolveSchemaAndFormat(columns, configuration->format, object_storage, configuration, format_settings, sample_path, context_); + resolveSchemaAndFormat(columns, object_storage, configuration, format_settings, sample_path, context_); configuration->check(context_); bool is_path_with_hive_partitioning = false; @@ -389,7 +389,7 @@ StorageObjectStorageQueue::StorageObjectStorageQueue( zk_path, *queue_settings_, storage_metadata.getColumns(), - configuration_->format, + configuration_->getFormat(), context_, is_attach, log); @@ -540,7 +540,7 @@ void StorageObjectStorageQueue::renameInMemory(const StorageID & new_table_id) bool StorageObjectStorageQueue::supportsSubsetOfColumns(const ContextPtr & context_) const { - return FormatFactory::instance().checkIfFormatSupportsSubsetOfColumns(configuration->format, context_, format_settings); + return FormatFactory::instance().checkIfFormatSupportsSubsetOfColumns(configuration->getFormat(), context_, format_settings); } class ReadFromObjectStorageQueue : public SourceStepWithFilter diff --git a/src/Storages/ObjectStorageQueue/StorageObjectStorageQueue.h b/src/Storages/ObjectStorageQueue/StorageObjectStorageQueue.h index ed7d46e842ef..dd55c2763db3 100644 --- a/src/Storages/ObjectStorageQueue/StorageObjectStorageQueue.h +++ b/src/Storages/ObjectStorageQueue/StorageObjectStorageQueue.h @@ -60,7 +60,7 @@ class StorageObjectStorageQueue : public IStorage, WithContext void renameInMemory(const StorageID & new_table_id) override; - const auto & getFormatName() const { return configuration->format; } + const auto & getFormatName() const { return configuration->getFormat(); } const fs::path & getZooKeeperPath() const { return zk_path; } diff --git a/src/Storages/ObjectStorageQueue/registerQueueStorage.cpp b/src/Storages/ObjectStorageQueue/registerQueueStorage.cpp index 09f0898d0264..ec0851ec47dc 100644 --- a/src/Storages/ObjectStorageQueue/registerQueueStorage.cpp +++ b/src/Storages/ObjectStorageQueue/registerQueueStorage.cpp @@ -49,7 +49,7 @@ StoragePtr createQueueStorage(const StorageFactory::Arguments & args) throw Exception(ErrorCodes::BAD_ARGUMENTS, "External data source must have arguments"); auto configuration = std::make_shared(); - StorageObjectStorageConfiguration::initialize(*configuration, args.engine_args, args.getContext(), false, &args.table_id); + configuration->initialize(args.engine_args, args.getContext(), false, &args.table_id); // Use format settings from global server context + settings from // the SETTINGS clause of the create query. Settings from current diff --git a/src/Storages/StorageDistributed.cpp b/src/Storages/StorageDistributed.cpp index f8ea0d59c3c3..6dab3173be2d 100644 --- a/src/Storages/StorageDistributed.cpp +++ b/src/Storages/StorageDistributed.cpp @@ -886,6 +886,7 @@ QueryTreeNodePtr buildQueryTreeDistributed(SelectQueryInfo & query_info, auto table_function_node = std::make_shared(remote_table_function_node.getFunctionName()); table_function_node->getArgumentsNode() = remote_table_function_node.getArgumentsNode(); + table_function_node->setSettingsChanges(remote_table_function_node.getSettingsChanges()); if (table_expression_modifiers) table_function_node->setTableExpressionModifiers(*table_expression_modifiers); @@ -1403,7 +1404,8 @@ std::optional StorageDistributed::distributedWrite(const ASTInser } if (auto src_storage_cluster = std::dynamic_pointer_cast(src_storage)) { - return distributedWriteFromClusterStorage(*src_storage_cluster, query, local_context); + if (!src_storage_cluster->getClusterName(local_context).empty()) + return distributedWriteFromClusterStorage(*src_storage_cluster, query, local_context); } return {}; diff --git a/src/Storages/StorageFileCluster.cpp b/src/Storages/StorageFileCluster.cpp index 3f32b4387272..6cd6ec0ec2cd 100644 --- a/src/Storages/StorageFileCluster.cpp +++ b/src/Storages/StorageFileCluster.cpp @@ -81,7 +81,11 @@ StorageFileCluster::StorageFileCluster( setInMemoryMetadata(storage_metadata); } -void StorageFileCluster::updateQueryToSendIfNeeded(DB::ASTPtr & query, const StorageSnapshotPtr & storage_snapshot, const DB::ContextPtr & context) +void StorageFileCluster::updateQueryToSendIfNeeded( + DB::ASTPtr & query, + const StorageSnapshotPtr & storage_snapshot, + const DB::ContextPtr & context, + bool /*make_cluster_function*/) { auto * table_function = extractTableFunctionFromSelectQuery(query); if (!table_function) diff --git a/src/Storages/StorageFileCluster.h b/src/Storages/StorageFileCluster.h index 869dc9f8e58b..eb2a70f60b89 100644 --- a/src/Storages/StorageFileCluster.h +++ b/src/Storages/StorageFileCluster.h @@ -36,7 +36,11 @@ class StorageFileCluster : public IStorageCluster StorageMetadataPtr) const override; private: - void updateQueryToSendIfNeeded(ASTPtr & query, const StorageSnapshotPtr & storage_snapshot, const ContextPtr & context) override; + void updateQueryToSendIfNeeded( + ASTPtr & query, + const StorageSnapshotPtr & storage_snapshot, + const ContextPtr & context, + bool /*make_cluster_function*/) override; Strings paths; String filename; diff --git a/src/Storages/StorageReplicatedMergeTree.cpp b/src/Storages/StorageReplicatedMergeTree.cpp index adf50f731fd9..86119b70df72 100644 --- a/src/Storages/StorageReplicatedMergeTree.cpp +++ b/src/Storages/StorageReplicatedMergeTree.cpp @@ -141,6 +141,7 @@ #include #include #include +#include #include #include @@ -8682,14 +8683,19 @@ void StorageReplicatedMergeTree::exportPartitionToTable(const PartitionCommand & { #if USE_AVRO auto * object_storage = dynamic_cast(dest_storage.get()); + auto * object_storage_cluster = dynamic_cast(dest_storage.get()); /// in theory this should never happen, but just in case - if (!object_storage) + if (!object_storage && !object_storage_cluster) { throw Exception(ErrorCodes::BAD_ARGUMENTS, "Destination storage {} is not a StorageObjectStorage", dest_storage->getName()); } - auto * iceberg_metadata = dynamic_cast(object_storage->getExternalMetadata(query_context)); + IcebergMetadata * iceberg_metadata = nullptr; + if (object_storage) + iceberg_metadata = dynamic_cast(object_storage->getExternalMetadata(query_context)); + else if (object_storage_cluster) + iceberg_metadata = dynamic_cast(object_storage_cluster->getExternalMetadata(query_context)); if (!iceberg_metadata) { throw Exception(ErrorCodes::BAD_ARGUMENTS, "Destination storage {} is a data lake but not an iceberg table", dest_storage->getName()); diff --git a/src/Storages/StorageURLCluster.cpp b/src/Storages/StorageURLCluster.cpp index e6378b222182..2232645e5a7f 100644 --- a/src/Storages/StorageURLCluster.cpp +++ b/src/Storages/StorageURLCluster.cpp @@ -109,7 +109,11 @@ StorageURLCluster::StorageURLCluster( setInMemoryMetadata(storage_metadata); } -void StorageURLCluster::updateQueryToSendIfNeeded(ASTPtr & query, const StorageSnapshotPtr & storage_snapshot, const ContextPtr & context) +void StorageURLCluster::updateQueryToSendIfNeeded( + ASTPtr & query, + const StorageSnapshotPtr & storage_snapshot, + const ContextPtr & context, + bool /*make_cluster_function*/) { auto * table_function = extractTableFunctionFromSelectQuery(query); if (!table_function) diff --git a/src/Storages/StorageURLCluster.h b/src/Storages/StorageURLCluster.h index bee90c4ba7b0..2ed1e28922f1 100644 --- a/src/Storages/StorageURLCluster.h +++ b/src/Storages/StorageURLCluster.h @@ -38,7 +38,11 @@ class StorageURLCluster : public IStorageCluster StorageMetadataPtr) const override; private: - void updateQueryToSendIfNeeded(ASTPtr & query, const StorageSnapshotPtr & storage_snapshot, const ContextPtr & context) override; + void updateQueryToSendIfNeeded( + ASTPtr & query, + const StorageSnapshotPtr & storage_snapshot, + const ContextPtr & context, + bool /*make_cluster_function*/) override; String uri; String format_name; diff --git a/src/Storages/System/StorageSystemIcebergHistory.cpp b/src/Storages/System/StorageSystemIcebergHistory.cpp index c9566dd0cada..239e01816527 100644 --- a/src/Storages/System/StorageSystemIcebergHistory.cpp +++ b/src/Storages/System/StorageSystemIcebergHistory.cpp @@ -17,7 +17,7 @@ #include #include #include -#include +#include #include #include #include @@ -85,7 +85,7 @@ void StorageSystemIcebergHistory::fillData([[maybe_unused]] MutableColumns & res const auto access = context_copy->getAccess(); - auto add_history_record = [&](const DatabaseTablesIteratorPtr & it, StorageObjectStorage * object_storage) + auto add_history_record = [&](const DatabaseTablesIteratorPtr & it, StorageObjectStorageCluster * object_storage) { if (!access->isGranted(AccessType::SHOW_TABLES, it->databaseName(), it->name())) return; @@ -149,7 +149,7 @@ void StorageSystemIcebergHistory::fillData([[maybe_unused]] MutableColumns & res // Table was dropped while acquiring the lock, skipping table continue; - if (auto * object_storage_table = dynamic_cast(storage.get())) + if (auto * object_storage_table = dynamic_cast(storage.get())) { add_history_record(iterator, object_storage_table); } diff --git a/src/Storages/System/StorageSystemTables.cpp b/src/Storages/System/StorageSystemTables.cpp index b87f1ce9f1d0..6657fec7da52 100644 --- a/src/Storages/System/StorageSystemTables.cpp +++ b/src/Storages/System/StorageSystemTables.cpp @@ -27,6 +27,12 @@ #include #include #include +<<<<<<< HEAD +======= +#include +#include +#include +>>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) #include #include #include @@ -698,18 +704,107 @@ class TablesBlockSource final : public ISource ASTPtr expression_ptr; if (columns_mask[src_index++]) { +<<<<<<< HEAD if (metadata_snapshot && (expression_ptr = metadata_snapshot->getPartitionKeyAST())) res_columns[res_index++]->insert(format({context, *expression_ptr})); else res_columns[res_index++]->insertDefault(); +======= + bool inserted = false; + + try + { + // Extract from specific DataLake metadata if suitable + if (auto * obj = dynamic_cast(table.get())) + { + if (auto * dl_meta = obj->getExternalMetadata(context)) + { + if (auto p = dl_meta->partitionKey(context); p.has_value()) + { + res_columns[res_index++]->insert(*p); + inserted = true; + } + } + } + else if (auto * clobj = dynamic_cast(table.get())) + { + if (auto * dl_meta = clobj->getExternalMetadata(context)) + { + if (auto p = dl_meta->partitionKey(context); p.has_value()) + { + res_columns[res_index++]->insert(*p); + inserted = true; + } + } + + } + } + catch (const Exception &) + { + /// Failed to get info. It's not critical, just log it. + tryLogCurrentException("StorageSystemTables"); + } + + if (!inserted) + { + if (metadata_snapshot && (expression_ptr = metadata_snapshot->getPartitionKeyAST())) + res_columns[res_index++]->insert(format({context, *expression_ptr})); + else + res_columns[res_index++]->insertDefault(); + } +>>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) } if (columns_mask[src_index++]) { +<<<<<<< HEAD if (metadata_snapshot && (expression_ptr = metadata_snapshot->getSortingKey().expression_list_ast)) res_columns[res_index++]->insert(format({context, *expression_ptr})); else res_columns[res_index++]->insertDefault(); +======= + bool inserted = false; + + try + { + // Extract from specific DataLake metadata if suitable + if (auto * obj = dynamic_cast(table.get())) + { + if (auto * dl_meta = obj->getExternalMetadata(context)) + { + if (auto p = dl_meta->sortingKey(context); p.has_value()) + { + res_columns[res_index++]->insert(*p); + inserted = true; + } + } + } + else if (auto * clobj = dynamic_cast(table.get())) + { + if (auto * dl_meta = clobj->getExternalMetadata(context)) + { + if (auto p = dl_meta->sortingKey(context); p.has_value()) + { + res_columns[res_index++]->insert(*p); + inserted = true; + } + } + } + } + catch (const Exception &) + { + /// Failed to get info. It's not critical, just log it. + tryLogCurrentException("StorageSystemTables"); + } + + if (!inserted) + { + if (metadata_snapshot && (expression_ptr = metadata_snapshot->getSortingKey().expression_list_ast)) + res_columns[res_index++]->insert(format({context, *expression_ptr})); + else + res_columns[res_index++]->insertDefault(); + } +>>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) } if (columns_mask[src_index++]) diff --git a/src/Storages/extractTableFunctionFromSelectQuery.h b/src/Storages/extractTableFunctionFromSelectQuery.h index 20cc1ae93896..2a845477df82 100644 --- a/src/Storages/extractTableFunctionFromSelectQuery.h +++ b/src/Storages/extractTableFunctionFromSelectQuery.h @@ -1,8 +1,8 @@ #pragma once #include -#include #include +#include namespace DB { @@ -12,5 +12,6 @@ ASTTableExpression * extractTableExpressionASTPtrFromSelectQuery(ASTPtr & query) ASTPtr extractTableFunctionASTPtrFromSelectQuery(ASTPtr & query); ASTPtr extractTableASTPtrFromSelectQuery(ASTPtr & query); ASTFunction * extractTableFunctionFromSelectQuery(ASTPtr & query); +ASTExpressionList * extractTableFunctionArgumentsFromSelectQuery(ASTPtr & query); } diff --git a/src/TableFunctions/ITableFunction.h b/src/TableFunctions/ITableFunction.h index 5b583329d342..186d1d25a855 100644 --- a/src/TableFunctions/ITableFunction.h +++ b/src/TableFunctions/ITableFunction.h @@ -81,7 +81,7 @@ class ITableFunction : public std::enable_shared_from_this virtual bool supportsReadingSubsetOfColumns(const ContextPtr &) { return true; } - virtual bool canBeUsedToCreateTable() const { return true; } + virtual void validateUseToCreateTable() const {} // INSERT INTO TABLE FUNCTION ... PARTITION BY // Set partition by expression so `ITableFunctionObjectStorage` can construct a proper representation diff --git a/src/TableFunctions/ITableFunctionCluster.h b/src/TableFunctions/ITableFunctionCluster.h index 5345e1a0f0db..920f271f0535 100644 --- a/src/TableFunctions/ITableFunctionCluster.h +++ b/src/TableFunctions/ITableFunctionCluster.h @@ -16,6 +16,7 @@ namespace ErrorCodes extern const int NUMBER_OF_ARGUMENTS_DOESNT_MATCH; extern const int CLUSTER_DOESNT_EXIST; extern const int LOGICAL_ERROR; + extern const int BAD_ARGUMENTS; } /// Base class for *Cluster table functions that require cluster_name for the first argument. @@ -46,9 +47,13 @@ class ITableFunctionCluster : public Base throw Exception(ErrorCodes::LOGICAL_ERROR, "Unexpected table function name: {}", table_function->name); } - bool canBeUsedToCreateTable() const override { return false; } bool isClusterFunction() const override { return true; } + void validateUseToCreateTable() const override + { + throw Exception(ErrorCodes::BAD_ARGUMENTS, "Table function '{}' cannot be used to create a table", getName()); + } + protected: void parseArguments(const ASTPtr & ast, ContextPtr context) override { @@ -70,9 +75,11 @@ class ITableFunctionCluster : public Base /// Cluster name is always the first cluster_name = checkAndGetLiteralArgument(args[0], "cluster_name"); - - if (!context->tryGetCluster(cluster_name)) - throw Exception(ErrorCodes::CLUSTER_DOESNT_EXIST, "Requested cluster '{}' not found", cluster_name); + /// Remove check cluster existing here + /// In query like + /// remote('remote_host', xxxCluster('remote_cluster', ...)) + /// 'remote_cluster' can be defined only on 'remote_host' + /// If cluster not exists, query falls later /// Just cut the first arg (cluster_name) and try to parse other table function arguments as is args.erase(args.begin()); diff --git a/src/TableFunctions/TableFunctionObjectStorage.cpp b/src/TableFunctions/TableFunctionObjectStorage.cpp index 99038321e27d..4b3c985ed03c 100644 --- a/src/TableFunctions/TableFunctionObjectStorage.cpp +++ b/src/TableFunctions/TableFunctionObjectStorage.cpp @@ -194,7 +194,7 @@ template ColumnsDescription TableFunctionObjectStorage< Definition, Configuration, is_data_lake>::getActualTableStructure(ContextPtr context, bool is_insert_query) const { - if (configuration->structure == "auto") + if (configuration->getStructure() == "auto") { auto storage = getObjectStorage(context, !is_insert_query); configuration->lazyInitializeIfNeeded(object_storage, context); @@ -203,7 +203,6 @@ ColumnsDescription TableFunctionObjectStorage< ColumnsDescription columns; resolveSchemaAndFormat( columns, - configuration->format, std::move(storage), configuration, /* format_settings */std::nullopt, @@ -220,7 +219,7 @@ ColumnsDescription TableFunctionObjectStorage< return columns; } - return parseColumnsListFromString(configuration->structure, context); + return parseColumnsListFromString(configuration->getStructure(), context); } template @@ -234,8 +233,8 @@ StoragePtr TableFunctionObjectStorage:: chassert(configuration); ColumnsDescription columns; - if (configuration->structure != "auto") - columns = parseColumnsListFromString(configuration->structure, context); + if (configuration->getStructure() != "auto") + columns = parseColumnsListFromString(configuration->getStructure(), context); else if (!structure_hint.empty()) columns = structure_hint; else if (!cached_columns.empty()) @@ -266,8 +265,15 @@ StoragePtr TableFunctionObjectStorage:: columns, ConstraintsDescription{}, partition_by, + /* order_by */ nullptr, context, - /* is_table_function */true); + /* comment */ String{}, + /* format_settings */ std::nullopt, /// No format_settings + /* mode */ LoadingStrictnessLevel::CREATE, + configuration->getCatalog(context, /* attach */ false), + /* if_not_exists */ false, + /* is_datalake_query*/ false, + /* is_table_function */ true); storage->startup(); return storage; @@ -315,16 +321,7 @@ void registerTableFunctionObjectStorage(TableFunctionFactory & factory) { UNUSED(factory); #if USE_AWS_S3 - factory.registerFunction>( - { - .description=R"(The table function can be used to read the data stored on AWS S3.)", - .examples{{S3Definition::name, "SELECT * FROM s3(url, access_key_id, secret_access_key)", ""}}, - .category = FunctionDocumentation::Category::TableFunction - }, - {.allow_readonly = false} - ); - - factory.registerFunction>( + factory.registerFunction>( { .description=R"(The table function can be used to read the data stored on GCS.)", .examples{{GCSDefinition::name, "SELECT * FROM gcs(url, access_key_id, secret_access_key)", ""}}, @@ -333,7 +330,7 @@ void registerTableFunctionObjectStorage(TableFunctionFactory & factory) {.allow_readonly = false} ); - factory.registerFunction>( + factory.registerFunction>( { .description=R"(The table function can be used to read the data stored on COSN.)", .examples{{COSNDefinition::name, "SELECT * FROM cosn(url, access_key_id, secret_access_key)", ""}}, @@ -342,7 +339,7 @@ void registerTableFunctionObjectStorage(TableFunctionFactory & factory) {.allow_readonly = false} ); - factory.registerFunction>( + factory.registerFunction>( { .description=R"(The table function can be used to read the data stored on OSS.)", .examples{{OSSDefinition::name, "SELECT * FROM oss(url, access_key_id, secret_access_key)", ""}}, @@ -351,54 +348,28 @@ void registerTableFunctionObjectStorage(TableFunctionFactory & factory) {.allow_readonly = false} ); #endif - -#if USE_AZURE_BLOB_STORAGE - factory.registerFunction>( - { - .description=R"(The table function can be used to read the data stored on Azure Blob Storage.)", - .examples{ - { - AzureDefinition::name, - "SELECT * FROM azureBlobStorage(connection_string|storage_account_url, container_name, blobpath, " - "[account_name, account_key, format, compression, structure])", "" - }}, - .category = FunctionDocumentation::Category::TableFunction - }, - {.allow_readonly = false} - ); -#endif -#if USE_HDFS - factory.registerFunction>( - { - .description=R"(The table function can be used to read the data stored on HDFS virtual filesystem.)", - .examples{ - { - HDFSDefinition::name, - "SELECT * FROM hdfs(url, format, compression, structure])", "" - }}, - .category = FunctionDocumentation::Category::TableFunction - }, - {.allow_readonly = false} - ); -#endif } #if USE_AZURE_BLOB_STORAGE -template class TableFunctionObjectStorage; -template class TableFunctionObjectStorage; +template class TableFunctionObjectStorage; +template class TableFunctionObjectStorage; #endif #if USE_AWS_S3 -template class TableFunctionObjectStorage; -template class TableFunctionObjectStorage; -template class TableFunctionObjectStorage; -template class TableFunctionObjectStorage; -template class TableFunctionObjectStorage; +template class TableFunctionObjectStorage; +template class TableFunctionObjectStorage; +template class TableFunctionObjectStorage; +template class TableFunctionObjectStorage; +template class TableFunctionObjectStorage; #endif #if USE_HDFS -template class TableFunctionObjectStorage; -template class TableFunctionObjectStorage; +template class TableFunctionObjectStorage; +template class TableFunctionObjectStorage; +#endif + +#if USE_AVRO +template class TableFunctionObjectStorage; #endif #if USE_AVRO @@ -444,6 +415,7 @@ template class TableFunctionObjectStorage; #endif +<<<<<<< HEAD #if USE_AVRO void registerTableFunctionIceberg(TableFunctionFactory & factory); void registerTableFunctionIceberg(TableFunctionFactory & factory) @@ -483,6 +455,8 @@ void registerTableFunctionIceberg(TableFunctionFactory & factory) } #endif +======= +>>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) #if USE_AVRO void registerTableFunctionPaimon(TableFunctionFactory & factory); @@ -527,28 +501,6 @@ void registerTableFunctionPaimon(TableFunctionFactory & factory) void registerTableFunctionDeltaLake(TableFunctionFactory & factory); void registerTableFunctionDeltaLake(TableFunctionFactory & factory) { -#if USE_AWS_S3 - factory.registerFunction( - {.description = R"(The table function can be used to read the DeltaLake table stored on S3, alias of deltaLakeS3.)", - .examples{{DeltaLakeDefinition::name, "SELECT * FROM deltaLake(url, access_key_id, secret_access_key)", ""}}, - .category = FunctionDocumentation::Category::TableFunction}, - {.allow_readonly = false}); - - factory.registerFunction( - {.description = R"(The table function can be used to read the DeltaLake table stored on S3.)", - .examples{{DeltaLakeS3Definition::name, "SELECT * FROM deltaLakeS3(url, access_key_id, secret_access_key)", ""}}, - .category = FunctionDocumentation::Category::TableFunction}, - {.allow_readonly = false}); -#endif - -#if USE_AZURE_BLOB_STORAGE - factory.registerFunction( - {.description = R"(The table function can be used to read the DeltaLake table stored on Azure object store.)", - .examples{{DeltaLakeAzureDefinition::name, "SELECT * FROM deltaLakeAzure(connection_string|storage_account_url, container_name, blobpath, \"\n" - " \"[account_name, account_key, format, compression, structure])", ""}}, - .category = FunctionDocumentation::Category::TableFunction}, - {.allow_readonly = false}); -#endif // Register the new local Delta Lake table function factory.registerFunction( {.description = R"(The table function can be used to read the DeltaLake table stored locally.)", @@ -558,6 +510,7 @@ void registerTableFunctionDeltaLake(TableFunctionFactory & factory) } #endif +<<<<<<< HEAD #if USE_AWS_S3 void registerTableFunctionHudi(TableFunctionFactory & factory); void registerTableFunctionHudi(TableFunctionFactory & factory) @@ -570,22 +523,17 @@ void registerTableFunctionHudi(TableFunctionFactory & factory) } #endif +======= +>>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) void registerDataLakeTableFunctions(TableFunctionFactory & factory) { UNUSED(factory); -#if USE_AVRO - registerTableFunctionIceberg(factory); -#endif - -#if USE_AVRO - registerTableFunctionPaimon(factory); -#endif #if USE_PARQUET && USE_DELTA_KERNEL_RS registerTableFunctionDeltaLake(factory); #endif -#if USE_AWS_S3 - registerTableFunctionHudi(factory); +#if USE_AVRO + registerTableFunctionPaimon(factory); #endif } } diff --git a/src/TableFunctions/TableFunctionObjectStorage.h b/src/TableFunctions/TableFunctionObjectStorage.h index 24e6597779a6..a15395d11e68 100644 --- a/src/TableFunctions/TableFunctionObjectStorage.h +++ b/src/TableFunctions/TableFunctionObjectStorage.h @@ -25,10 +25,12 @@ struct S3StorageSettings; struct AzureStorageSettings; struct HDFSStorageSettings; -template +template class TableFunctionObjectStorage : public ITableFunction { public: + using Configuration = StorageConfiguration; + static constexpr auto name = Definition::name; using Settings = typename std::conditional_t< is_data_lake, @@ -37,15 +39,16 @@ class TableFunctionObjectStorage : public ITableFunction String getName() const override { return name; } - bool hasStaticStructure() const override { return configuration->structure != "auto"; } + bool hasStaticStructure() const override { return configuration->getStructure() != "auto"; } - bool needStructureHint() const override { return configuration->structure == "auto"; } + bool needStructureHint() const override { return configuration->getStructure() == "auto"; } void setStructureHint(const ColumnsDescription & structure_hint_) override { structure_hint = structure_hint_; } bool supportsReadingSubsetOfColumns(const ContextPtr & context) override { - return configuration->format != "auto" && FormatFactory::instance().checkIfFormatSupportsSubsetOfColumns(configuration->format, context); + return configuration->getFormat() != "auto" + && FormatFactory::instance().checkIfFormatSupportsSubsetOfColumns(configuration->getFormat(), context); } NameSet getVirtualsToCheckBeforeUsingStructureHint() const override @@ -55,7 +58,7 @@ class TableFunctionObjectStorage : public ITableFunction virtual void parseArgumentsImpl(ASTs & args, const ContextPtr & context) { - StorageObjectStorageConfiguration::initialize(*getConfiguration(context), args, context, true); + getConfiguration(context)->initialize(args, context, true); } static void updateStructureAndFormatArgumentsIfNeeded( @@ -67,8 +70,8 @@ class TableFunctionObjectStorage : public ITableFunction if constexpr (is_data_lake) { Configuration configuration(createEmptySettings()); - if (configuration.format == "auto") - configuration.format = "Parquet"; /// Default format of data lakes. + if (configuration.getFormat() == "auto") + configuration.setFormat("Parquet"); /// Default format of data lakes. configuration.addStructureAndFormatToArgsIfNeeded(args, structure, format, context, /*with_structure=*/true); } @@ -110,21 +113,22 @@ class TableFunctionObjectStorage : public ITableFunction }; #if USE_AWS_S3 -using TableFunctionS3 = TableFunctionObjectStorage; +using TableFunctionS3 = TableFunctionObjectStorage; #endif #if USE_AZURE_BLOB_STORAGE -using TableFunctionAzureBlob = TableFunctionObjectStorage; +using TableFunctionAzureBlob = TableFunctionObjectStorage; #endif #if USE_HDFS -using TableFunctionHDFS = TableFunctionObjectStorage; +using TableFunctionHDFS = TableFunctionObjectStorage; #endif #if USE_AVRO +using TableFunctionIceberg = TableFunctionObjectStorage; + # if USE_AWS_S3 -using TableFunctionIceberg = TableFunctionObjectStorage; using TableFunctionIcebergS3 = TableFunctionObjectStorage; # endif # if USE_AZURE_BLOB_STORAGE @@ -149,13 +153,13 @@ using TableFunctionPaimonHDFS = TableFunctionObjectStorage; #endif #if USE_PARQUET && USE_DELTA_KERNEL_RS -#if USE_AWS_S3 +# if USE_AWS_S3 using TableFunctionDeltaLake = TableFunctionObjectStorage; using TableFunctionDeltaLakeS3 = TableFunctionObjectStorage; -#endif -#if USE_AZURE_BLOB_STORAGE +# endif +# if USE_AZURE_BLOB_STORAGE using TableFunctionDeltaLakeAzure = TableFunctionObjectStorage; -#endif +# endif // New alias for local Delta Lake table function using TableFunctionDeltaLakeLocal = TableFunctionObjectStorage; #endif diff --git a/src/TableFunctions/TableFunctionObjectStorageCluster.cpp b/src/TableFunctions/TableFunctionObjectStorageCluster.cpp index fa65ca56d6fa..552a262a5ce2 100644 --- a/src/TableFunctions/TableFunctionObjectStorageCluster.cpp +++ b/src/TableFunctions/TableFunctionObjectStorageCluster.cpp @@ -31,8 +31,9 @@ StoragePtr TableFunctionObjectStorageClusterstructure != "auto") - columns = parseColumnsListFromString(configuration->structure, context); + + if (configuration->getStructure() != "auto") + columns = parseColumnsListFromString(configuration->getStructure(), context); else if (!Base::structure_hint.empty()) columns = Base::structure_hint; else if (!cached_columns.empty()) @@ -79,8 +80,16 @@ StoragePtr TableFunctionObjectStorageClusterstartup(); @@ -145,47 +154,53 @@ void registerTableFunctionIcebergCluster(TableFunctionFactory & factory) {.allow_readonly = false} ); -#if USE_AWS_S3 factory.registerFunction( { - .description = R"(The table function can be used to read the Iceberg table stored on store from disk in parallel for many nodes in a specified cluster.)", - .examples{{IcebergClusterDefinition::name, "SELECT * FROM icebergCluster(cluster) SETTINGS disk = 'disk'", ""},{IcebergClusterDefinition::name, "SELECT * FROM icebergCluster(cluster, url, [, NOSIGN | access_key_id, secret_access_key, [session_token]], format, [,compression])", ""}}, + .description = R"(The table function can be used to read the Iceberg table stored on any object store in parallel for many nodes in a specified cluster.)", + .examples{ +# if USE_AWS_S3 + {"icebergCluster", "SELECT * FROM icebergCluster(cluster, url, [, NOSIGN | access_key_id, secret_access_key, [session_token]], format, [,compression], storage_type='s3')", ""}, +# endif +# if USE_AZURE_BLOB_STORAGE + {"icebergCluster", "SELECT * FROM icebergCluster(cluster, connection_string|storage_account_url, container_name, blobpath, [account_name, account_key, format, compression], storage_type='azure')", ""}, +# endif +# if USE_HDFS + {"icebergCluster", "SELECT * FROM icebergCluster(cluster, uri, [format], [structure], [compression_method], storage_type='hdfs')", ""}, +# endif + }, .category = FunctionDocumentation::Category::TableFunction }, - {.allow_readonly = false} - ); + {.allow_readonly = false}); +# if USE_AWS_S3 factory.registerFunction( { .description = R"(The table function can be used to read the Iceberg table stored on S3 object store in parallel for many nodes in a specified cluster.)", .examples{{IcebergS3ClusterDefinition::name, "SELECT * FROM icebergS3Cluster(cluster, url, [, NOSIGN | access_key_id, secret_access_key, [session_token]], format, [,compression])", ""}}, .category = FunctionDocumentation::Category::TableFunction }, - {.allow_readonly = false} - ); -#endif + {.allow_readonly = false}); +# endif -#if USE_AZURE_BLOB_STORAGE +# if USE_AZURE_BLOB_STORAGE factory.registerFunction( { .description = R"(The table function can be used to read the Iceberg table stored on Azure object store in parallel for many nodes in a specified cluster.)", .examples{{IcebergAzureClusterDefinition::name, "SELECT * FROM icebergAzureCluster(cluster, connection_string|storage_account_url, container_name, blobpath, [account_name, account_key, format, compression])", ""}}, .category = FunctionDocumentation::Category::TableFunction }, - {.allow_readonly = false} - ); -#endif + {.allow_readonly = false}); +# endif -#if USE_HDFS +# if USE_HDFS factory.registerFunction( { .description = R"(The table function can be used to read the Iceberg table stored on HDFS virtual filesystem in parallel for many nodes in a specified cluster.)", .examples{{IcebergHDFSClusterDefinition::name, "SELECT * FROM icebergHDFSCluster(cluster, uri, [format], [structure], [compression_method])", ""}}, .category = FunctionDocumentation::Category::TableFunction }, - {.allow_readonly = false} - ); -#endif + {.allow_readonly = false}); +# endif } void registerTableFunctionPaimonCluster(TableFunctionFactory & factory); diff --git a/src/TableFunctions/TableFunctionObjectStorageCluster.h b/src/TableFunctions/TableFunctionObjectStorageCluster.h index 58acf48d4f2a..26faefc2c5c2 100644 --- a/src/TableFunctions/TableFunctionObjectStorageCluster.h +++ b/src/TableFunctions/TableFunctionObjectStorageCluster.h @@ -12,8 +12,6 @@ namespace DB class Context; -class StorageS3Settings; -class StorageAzureBlobSettings; class StorageS3Configuration; class StorageAzureConfiguration; @@ -47,21 +45,25 @@ class TableFunctionObjectStorageCluster : public ITableFunctionClusterstructure != "auto"; } - bool needStructureHint() const override { return Base::getConfiguration(getQueryOrGlobalContext())->structure == "auto"; } + bool hasStaticStructure() const override { return Base::getConfiguration(getQueryOrGlobalContext())->getStructure() != "auto"; } + bool needStructureHint() const override { return Base::getConfiguration(getQueryOrGlobalContext())->getStructure() == "auto"; } void setStructureHint(const ColumnsDescription & structure_hint_) override { Base::structure_hint = structure_hint_; } }; #if USE_AWS_S3 -using TableFunctionS3Cluster = TableFunctionObjectStorageCluster; +using TableFunctionS3Cluster = TableFunctionObjectStorageCluster; #endif #if USE_AZURE_BLOB_STORAGE -using TableFunctionAzureBlobCluster = TableFunctionObjectStorageCluster; +using TableFunctionAzureBlobCluster = TableFunctionObjectStorageCluster; #endif #if USE_HDFS -using TableFunctionHDFSCluster = TableFunctionObjectStorageCluster; +using TableFunctionHDFSCluster = TableFunctionObjectStorageCluster; +#endif + +#if USE_AVRO +using TableFunctionIcebergCluster = TableFunctionObjectStorageCluster; #endif #if USE_AVRO @@ -70,7 +72,6 @@ using TableFunctionIcebergLocalCluster = TableFunctionObjectStorageCluster; -using TableFunctionIcebergCluster = TableFunctionObjectStorageCluster; #endif #if USE_AVRO && USE_AZURE_BLOB_STORAGE @@ -95,7 +96,7 @@ using TableFunctionPaimonHDFSCluster = TableFunctionObjectStorageCluster; using TableFunctionDeltaLakeS3Cluster = TableFunctionObjectStorageCluster; #endif diff --git a/src/TableFunctions/TableFunctionObjectStorageClusterFallback.cpp b/src/TableFunctions/TableFunctionObjectStorageClusterFallback.cpp new file mode 100644 index 000000000000..8b2819e42d46 --- /dev/null +++ b/src/TableFunctions/TableFunctionObjectStorageClusterFallback.cpp @@ -0,0 +1,458 @@ +#include +#include +#include +#include +#include + +namespace DB +{ + +namespace Setting +{ + extern const SettingsString object_storage_cluster; + extern const SettingsBool object_storage_remote_initiator; + extern const SettingsString object_storage_remote_initiator_cluster; +} + +namespace ErrorCodes +{ + extern const int NUMBER_OF_ARGUMENTS_DOESNT_MATCH; + extern const int BAD_ARGUMENTS; +} + +struct S3ClusterFallbackDefinition +{ + static constexpr auto name = "s3"; + static constexpr auto storage_engine_name = "S3"; + static constexpr auto storage_engine_cluster_name = "S3Cluster"; +}; + +struct AzureClusterFallbackDefinition +{ + static constexpr auto name = "azureBlobStorage"; + static constexpr auto storage_engine_name = "Azure"; + static constexpr auto storage_engine_cluster_name = "AzureBlobStorageCluster"; +}; + +struct HDFSClusterFallbackDefinition +{ + static constexpr auto name = "hdfs"; + static constexpr auto storage_engine_name = "HDFS"; + static constexpr auto storage_engine_cluster_name = "HDFSCluster"; +}; + +struct IcebergClusterFallbackDefinition +{ + static constexpr auto name = "iceberg"; + static constexpr auto storage_engine_name = "UNDEFINED"; + static constexpr auto storage_engine_cluster_name = "IcebergCluster"; +}; + +struct IcebergS3ClusterFallbackDefinition +{ + static constexpr auto name = "icebergS3"; + static constexpr auto storage_engine_name = "S3"; + static constexpr auto storage_engine_cluster_name = "IcebergS3Cluster"; +}; + +struct IcebergAzureClusterFallbackDefinition +{ + static constexpr auto name = "icebergAzure"; + static constexpr auto storage_engine_name = "Azure"; + static constexpr auto storage_engine_cluster_name = "IcebergAzureCluster"; +}; + +struct IcebergHDFSClusterFallbackDefinition +{ + static constexpr auto name = "icebergHDFS"; + static constexpr auto storage_engine_name = "HDFS"; + static constexpr auto storage_engine_cluster_name = "IcebergHDFSCluster"; +}; + +struct IcebergLocalClusterFallbackDefinition +{ + static constexpr auto name = "icebergLocal"; + static constexpr auto storage_engine_name = "Local"; + static constexpr auto storage_engine_cluster_name = "IcebergLocalCluster"; +}; + +struct DeltaLakeClusterFallbackDefinition +{ + static constexpr auto name = "deltaLake"; + static constexpr auto storage_engine_name = "S3"; + static constexpr auto storage_engine_cluster_name = "DeltaLakeS3Cluster"; +}; + +struct DeltaLakeS3ClusterFallbackDefinition +{ + static constexpr auto name = "deltaLakeS3"; + static constexpr auto storage_engine_name = "S3"; + static constexpr auto storage_engine_cluster_name = "DeltaLakeS3Cluster"; +}; + +struct DeltaLakeAzureClusterFallbackDefinition +{ + static constexpr auto name = "deltaLakeAzure"; + static constexpr auto storage_engine_name = "Azure"; + static constexpr auto storage_engine_cluster_name = "DeltaLakeAzureCluster"; +}; + +struct HudiClusterFallbackDefinition +{ + static constexpr auto name = "hudi"; + static constexpr auto storage_engine_name = "S3"; + static constexpr auto storage_engine_cluster_name = "HudiS3Cluster"; +}; + +template +void TableFunctionObjectStorageClusterFallback::parseArgumentsImpl(ASTs & args, const ContextPtr & context) +{ + if (args.empty()) + throw Exception( + ErrorCodes::NUMBER_OF_ARGUMENTS_DOESNT_MATCH, + "The function {} should have arguments. The first argument must be the cluster name and the rest are the arguments of " + "corresponding table function", + getName()); + + const auto & settings = context->getSettingsRef(); + + is_cluster_function = !settings[Setting::object_storage_cluster].value.empty() && typename Base::Configuration().isClusterSupported(); + is_remote = settings[Setting::object_storage_remote_initiator]; + + if (is_cluster_function) + { + /// Name may be empty, but cluster workaround may be used in remote initiator case + ASTPtr cluster_name_arg = make_intrusive(settings[Setting::object_storage_cluster].value); + args.insert(args.begin(), cluster_name_arg); + BaseCluster::parseArgumentsImpl(args, context); + args.erase(args.begin()); + } + else + BaseSimple::parseArgumentsImpl(args, context); // NOLINT(bugprone-parent-virtual-call) +} + +template +StoragePtr TableFunctionObjectStorageClusterFallback::executeImpl( + const ASTPtr & ast_function, + ContextPtr context, + const std::string & table_name, + ColumnsDescription cached_columns, + bool is_insert_query) const +{ + if (is_cluster_function || is_remote) + { + auto result = BaseCluster::executeImpl(ast_function, context, table_name, cached_columns, is_insert_query); + if (auto storage = typeid_cast>(result)) + storage->setClusterNameInSettings(true); + return result; + } + else + return BaseSimple::executeImpl(ast_function, context, table_name, cached_columns, is_insert_query); // NOLINT(bugprone-parent-virtual-call) +} + +template +void TableFunctionObjectStorageClusterFallback::validateUseToCreateTable() const +{ + if (is_cluster_function || is_remote) + throw Exception( + ErrorCodes::BAD_ARGUMENTS, + "Table function '{}' cannot be used to create a table in cluster mode or with remote initiator", + getName()); +} + +#if USE_AWS_S3 +using TableFunctionS3ClusterFallback = TableFunctionObjectStorageClusterFallback; +#endif + +#if USE_AZURE_BLOB_STORAGE +using TableFunctionAzureClusterFallback = TableFunctionObjectStorageClusterFallback; +#endif + +#if USE_HDFS +using TableFunctionHDFSClusterFallback = TableFunctionObjectStorageClusterFallback; +#endif + +#if USE_AVRO +using TableFunctionIcebergClusterFallback = TableFunctionObjectStorageClusterFallback; +using TableFunctionIcebergLocalClusterFallback = TableFunctionObjectStorageClusterFallback; +#endif + +#if USE_AVRO && USE_AWS_S3 +using TableFunctionIcebergS3ClusterFallback = TableFunctionObjectStorageClusterFallback; +#endif + +#if USE_AVRO && USE_AZURE_BLOB_STORAGE +using TableFunctionIcebergAzureClusterFallback = TableFunctionObjectStorageClusterFallback; +#endif + +#if USE_AVRO && USE_HDFS +using TableFunctionIcebergHDFSClusterFallback = TableFunctionObjectStorageClusterFallback; +#endif + +#if USE_AWS_S3 && USE_PARQUET && USE_DELTA_KERNEL_RS +using TableFunctionDeltaLakeClusterFallback = TableFunctionObjectStorageClusterFallback; +using TableFunctionDeltaLakeS3ClusterFallback = TableFunctionObjectStorageClusterFallback; +#endif + +#if USE_AZURE_BLOB_STORAGE && USE_PARQUET && USE_DELTA_KERNEL_RS +using TableFunctionDeltaLakeAzureClusterFallback = TableFunctionObjectStorageClusterFallback; +#endif + +#if USE_AWS_S3 +using TableFunctionHudiClusterFallback = TableFunctionObjectStorageClusterFallback; +#endif + +void registerTableFunctionObjectStorageClusterFallback(TableFunctionFactory & factory) +{ + UNUSED(factory); +#if USE_AWS_S3 + factory.registerFunction( + { + .description=R"(The table function can be used to read the data stored on S3 in parallel for many nodes in a specified cluster or from single node.)", + .examples{ + {"s3", "SELECT * FROM s3(url, format, structure)", ""}, + {"s3", "SELECT * FROM s3(url, format, structure) SETTINGS object_storage_cluster='cluster'", ""} + }, + .category = FunctionDocumentation::Category::TableFunction + }, + {.allow_readonly = false} + ); +#endif + +#if USE_AZURE_BLOB_STORAGE + factory.registerFunction( + { + .description=R"(The table function can be used to read the data stored on Azure Blob Storage in parallel for many nodes in a specified cluster or from single node.)", + .examples{ + { + "azureBlobStorage", + "SELECT * FROM azureBlobStorage(connection_string|storage_account_url, container_name, blobpath, " + "[account_name, account_key, format, compression, structure])", "" + }, + { + "azureBlobStorage", + "SELECT * FROM azureBlobStorage(connection_string|storage_account_url, container_name, blobpath, " + "[account_name, account_key, format, compression, structure]) " + "SETTINGS object_storage_cluster='cluster'", "" + }, + }, + .category = FunctionDocumentation::Category::TableFunction + }, + {.allow_readonly = false} + ); +#endif + +#if USE_HDFS + factory.registerFunction( + { + .description=R"(The table function can be used to read the data stored on HDFS virtual filesystem in parallel for many nodes in a specified cluster or from single node.)", + .examples{ + { + "hdfs", + "SELECT * FROM hdfs(url, format, compression, structure])", "" + }, + { + "hdfs", + "SELECT * FROM hdfs(url, format, compression, structure]) " + "SETTINGS object_storage_cluster='cluster'", "" + }, + }, + .category = FunctionDocumentation::Category::TableFunction + }, + {.allow_readonly = false} + ); +#endif + +#if USE_AVRO + factory.registerFunction( + { + .description=R"(The table function can be used to read the Iceberg table stored on different object store in parallel for many nodes in a specified cluster or from single node.)", + .examples{ + { + "iceberg", + "SELECT * FROM iceberg(url, access_key_id, secret_access_key, storage_type='s3')", "" + }, + { + "iceberg", + "SELECT * FROM iceberg(url, access_key_id, secret_access_key, storage_type='s3') " + "SETTINGS object_storage_cluster='cluster'", "" + }, + { + "iceberg", + "SELECT * FROM iceberg(url, access_key_id, secret_access_key, storage_type='azure')", "" + }, + { + "iceberg", + "SELECT * FROM iceberg(url, storage_type='hdfs') SETTINGS object_storage_cluster='cluster'", "" + }, + }, + .category = FunctionDocumentation::Category::TableFunction + }, + {.allow_readonly = false} + ); + + factory.registerFunction( + { + .description=R"(The table function can be used to read the Iceberg table stored on shared disk in parallel for many nodes in a specified cluster or from single node.)", + .examples{ + { + "icebergLocal", + "SELECT * FROM icebergLocal(filename)", "" + }, + { + "icebergLocal", + "SELECT * FROM icebergLocal(filename) " + "SETTINGS object_storage_cluster='cluster'", "" + }, + }, + .category = FunctionDocumentation::Category::TableFunction + }, + {.allow_readonly = false} + ); +#endif + +#if USE_AVRO && USE_AWS_S3 + factory.registerFunction( + { + .description=R"(The table function can be used to read the Iceberg table stored on S3 object store in parallel for many nodes in a specified cluster or from single node.)", + .examples{ + { + "icebergS3", + "SELECT * FROM icebergS3(url, access_key_id, secret_access_key)", "" + }, + { + "icebergS3", + "SELECT * FROM icebergS3(url, access_key_id, secret_access_key) " + "SETTINGS object_storage_cluster='cluster'", "" + }, + }, + .category = FunctionDocumentation::Category::TableFunction + }, + {.allow_readonly = false} + ); +#endif + +#if USE_AVRO && USE_AZURE_BLOB_STORAGE + factory.registerFunction( + { + .description=R"(The table function can be used to read the Iceberg table stored on Azure object store in parallel for many nodes in a specified cluster or from single node.)", + .examples{ + { + "icebergAzure", + "SELECT * FROM icebergAzure(url, access_key_id, secret_access_key)", "" + }, + { + "icebergAzure", + "SELECT * FROM icebergAzure(url, access_key_id, secret_access_key) " + "SETTINGS object_storage_cluster='cluster'", "" + }, + }, + .category = FunctionDocumentation::Category::TableFunction + }, + {.allow_readonly = false} + ); +#endif + +#if USE_AVRO && USE_HDFS + factory.registerFunction( + { + .description=R"(The table function can be used to read the Iceberg table stored on HDFS virtual filesystem in parallel for many nodes in a specified cluster or from single node.)", + .examples{ + { + "icebergHDFS", + "SELECT * FROM icebergHDFS(url)", "" + }, + { + "icebergHDFS", + "SELECT * FROM icebergHDFS(url) SETTINGS object_storage_cluster='cluster'", "" + }, + }, + .category = FunctionDocumentation::Category::TableFunction + }, + {.allow_readonly = false} + ); +#endif + +#if USE_PARQUET && USE_DELTA_KERNEL_RS +# if USE_AWS_S3 + factory.registerFunction( + { + .description=R"(The table function can be used to read the DeltaLake table stored on object store in parallel for many nodes in a specified cluster or from single node.)", + .examples{ + { + "deltaLake", + "SELECT * FROM deltaLake(url, access_key_id, secret_access_key)", "" + }, + { + "deltaLake", + "SELECT * FROM deltaLake(url, access_key_id, secret_access_key) " + "SETTINGS object_storage_cluster='cluster'", "" + }, + }, + .category = FunctionDocumentation::Category::TableFunction + }, + {.allow_readonly = false} + ); + factory.registerFunction( + { + .description=R"(The table function can be used to read the DeltaLake table stored on object store in parallel for many nodes in a specified cluster or from single node.)", + .examples{ + { + "deltaLakeS3", + "SELECT * FROM deltaLakeS3(url, access_key_id, secret_access_key)", "" + }, + { + "deltaLakeS3", + "SELECT * FROM deltaLakeS3(url, access_key_id, secret_access_key) " + "SETTINGS object_storage_cluster='cluster'", "" + }, + }, + .category = FunctionDocumentation::Category::TableFunction + }, + {.allow_readonly = false} + ); +# endif +# if USE_AZURE_BLOB_STORAGE + factory.registerFunction( + { + .description=R"(The table function can be used to read the DeltaLake table stored on object store in parallel for many nodes in a specified cluster or from single node.)", + .examples{ + { + "deltaLakeAzure", + "SELECT * FROM deltaLakeAzure(url, access_key_id, secret_access_key)", "" + }, + { + "deltaLakeAzure", + "SELECT * FROM deltaLakeAzure(url, access_key_id, secret_access_key) " + "SETTINGS object_storage_cluster='cluster'", "" + }, + }, + .category = FunctionDocumentation::Category::TableFunction + }, + {.allow_readonly = false} + ); +# endif +#endif + +#if USE_AWS_S3 + factory.registerFunction( + { + .description=R"(The table function can be used to read the Hudi table stored on object store in parallel for many nodes in a specified cluster or from single node.)", + .examples{ + { + "hudi", + "SELECT * FROM hudi(url, access_key_id, secret_access_key)", "" + }, + { + "hudi", + "SELECT * FROM hudi(url, access_key_id, secret_access_key) SETTINGS object_storage_cluster='cluster'", "" + }, + }, + .category = FunctionDocumentation::Category::TableFunction + }, + {.allow_readonly = false} + ); +#endif +} + +} diff --git a/src/TableFunctions/TableFunctionObjectStorageClusterFallback.h b/src/TableFunctions/TableFunctionObjectStorageClusterFallback.h new file mode 100644 index 000000000000..a21cc963d4c0 --- /dev/null +++ b/src/TableFunctions/TableFunctionObjectStorageClusterFallback.h @@ -0,0 +1,50 @@ +#pragma once +#include "config.h" +#include + +namespace DB +{ + +/** +* Class implementing s3/hdfs/azureBlobStorage(...) table functions, +* which allow to use simple or distributed function variant based on settings. +* If setting `object_storage_cluster` is empty, +* simple single-host variant is used, if setting not empty, cluster variant is used. +* `SELECT * FROM s3('s3://...', ...) SETTINGS object_storage_cluster='cluster'` +* is equal to +* `SELECT * FROM s3Cluster('cluster', 's3://...', ...)` +*/ + +template +class TableFunctionObjectStorageClusterFallback : public Base +{ +public: + using BaseCluster = Base; + using BaseSimple = BaseCluster::Base; + + static constexpr auto name = Definition::name; + + String getName() const override { return name; } + + void validateUseToCreateTable() const override; + +private: + const char * getStorageEngineName() const override + { + return is_cluster_function ? Definition::storage_engine_cluster_name : Definition::storage_engine_name; + } + + StoragePtr executeImpl( + const ASTPtr & ast_function, + ContextPtr context, + const std::string & table_name, + ColumnsDescription cached_columns, + bool is_insert_query) const override; + + void parseArgumentsImpl(ASTs & args, const ContextPtr & context) override; + + bool is_cluster_function = false; + bool is_remote = false; +}; + +} diff --git a/src/TableFunctions/TableFunctionRemote.h b/src/TableFunctions/TableFunctionRemote.h index e58d30cf48df..498339231153 100644 --- a/src/TableFunctions/TableFunctionRemote.h +++ b/src/TableFunctions/TableFunctionRemote.h @@ -26,6 +26,8 @@ class TableFunctionRemote : public ITableFunction bool needStructureConversion() const override { return false; } + void setRemoteTableFunction(ASTPtr remote_table_function_ptr_) { remote_table_function_ptr = remote_table_function_ptr_; } + private: StoragePtr executeImpl(const ASTPtr & ast_function, ContextPtr context, const std::string & table_name, ColumnsDescription cached_columns, bool is_insert_query) const override; diff --git a/src/TableFunctions/registerTableFunctions.cpp b/src/TableFunctions/registerTableFunctions.cpp index e023be33e6c5..d70464d419a9 100644 --- a/src/TableFunctions/registerTableFunctions.cpp +++ b/src/TableFunctions/registerTableFunctions.cpp @@ -73,6 +73,7 @@ void registerTableFunctions() registerTableFunctionObjectStorage(factory); registerTableFunctionObjectStorageCluster(factory); registerDataLakeTableFunctions(factory); + registerTableFunctionObjectStorageClusterFallback(factory); registerDataLakeClusterTableFunctions(factory); #if USE_YTSAURUS diff --git a/src/TableFunctions/registerTableFunctions.h b/src/TableFunctions/registerTableFunctions.h index 92d87bea980c..a113fc1d5d46 100644 --- a/src/TableFunctions/registerTableFunctions.h +++ b/src/TableFunctions/registerTableFunctions.h @@ -74,6 +74,7 @@ void registerTableFunctionExplain(TableFunctionFactory & factory); void registerTableFunctionObjectStorage(TableFunctionFactory & factory); void registerTableFunctionObjectStorageCluster(TableFunctionFactory & factory); void registerDataLakeTableFunctions(TableFunctionFactory & factory); +void registerTableFunctionObjectStorageClusterFallback(TableFunctionFactory & factory); void registerDataLakeClusterTableFunctions(TableFunctionFactory & factory); void registerTableFunctionTimeSeries(TableFunctionFactory & factory); diff --git a/tests/integration/compose/docker_compose_iceberg_rest_catalog.yml b/tests/integration/compose/docker_compose_iceberg_rest_catalog.yml index f584a954da8c..c993bf122a5d 100644 --- a/tests/integration/compose/docker_compose_iceberg_rest_catalog.yml +++ b/tests/integration/compose/docker_compose_iceberg_rest_catalog.yml @@ -1,8 +1,33 @@ services: +<<<<<<< HEAD rest: image: tabulario/iceberg-rest:1.6.0 ports: - "${ICEBERG_REST_CATALOG_PORT}:8181" +======= + spark-iceberg: + image: tabulario/spark-iceberg:3.5.5_1.8.1 + build: spark/ + depends_on: + rest: + condition: service_healthy + minio: + condition: service_started + environment: + - AWS_ACCESS_KEY_ID=admin + - AWS_SECRET_ACCESS_KEY=password + - AWS_REGION=us-east-1 + ports: + - ${SPARK_ICEBERG_EXTERNAL_PORT:-8080}:8080 + - ${SPARK_ICEBERG_EXTERNAL_PORT_2:-10002}:10000 + - ${SPARK_ICEBERG_EXTERNAL_PORT_3:-10003}:10001 + stop_grace_period: 5s + cpus: 3 + rest: + image: tabulario/iceberg-rest:1.6.0 + ports: + - ${ICEBERG_REST_EXTERNAL_PORT:-8182}:8181 +>>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) environment: - AWS_ACCESS_KEY_ID=minio - AWS_SECRET_ACCESS_KEY=ClickHouse_Minio_P@ssw0rd diff --git a/tests/integration/helpers/cluster.py b/tests/integration/helpers/cluster.py index 946ed4bf9ddf..d2b80d35ebb0 100644 --- a/tests/integration/helpers/cluster.py +++ b/tests/integration/helpers/cluster.py @@ -724,6 +724,9 @@ def __init__( self.minio_secret_key = minio_secret_key self.spark_session = None + self.spark_iceberg_external_port = 8080 + self.spark_iceberg_external_port_2 = 10002 + self.spark_iceberg_external_port_3 = 10003 self.with_iceberg_catalog = False self._iceberg_rest_catalog_port = None self._iceberg_minio_port = None @@ -938,6 +941,8 @@ def __init__( self._letsencrypt_pebble_api_port = 14000 self._letsencrypt_pebble_management_port = 15000 + self.iceberg_rest_external_port = 8182 + self.docker_client: docker.DockerClient = None self.is_up = False self.env = os.environ.copy() @@ -1870,6 +1875,10 @@ def setup_hms_catalog_cmd(self, instance, env_variables, docker_compose_yml_dir) def setup_iceberg_catalog_cmd( self, instance, env_variables, docker_compose_yml_dir, extra_parameters=None ): + env_variables["ICEBERG_REST_EXTERNAL_PORT"] = str(self.iceberg_rest_external_port) + env_variables["SPARK_ICEBERG_EXTERNAL_PORT"] = str(self.spark_iceberg_external_port) + env_variables["SPARK_ICEBERG_EXTERNAL_PORT_2"] = str(self.spark_iceberg_external_port_2) + env_variables["SPARK_ICEBERG_EXTERNAL_PORT_3"] = str(self.spark_iceberg_external_port_3) self.with_iceberg_catalog = True file_name = "docker_compose_iceberg_rest_catalog.yml" if extra_parameters is not None and extra_parameters["docker_compose_file_name"] != "": diff --git a/tests/integration/helpers/iceberg_utils.py b/tests/integration/helpers/iceberg_utils.py index c3a993b9d557..136689e3bcec 100644 --- a/tests/integration/helpers/iceberg_utils.py +++ b/tests/integration/helpers/iceberg_utils.py @@ -236,8 +236,12 @@ def get_creation_expression( table_function=False, use_version_hint=False, run_on_cluster=False, + object_storage_cluster=False, explicit_metadata_path="", additional_settings = [], + storage_type_as_arg=False, + storage_type_in_named_collection=False, + cluster_name_as_literal=True, **kwargs, ): settings_array = list(additional_settings) @@ -248,6 +252,9 @@ def get_creation_expression( if use_version_hint: settings_array.append("iceberg_use_version_hint = true") + if object_storage_cluster: + settings_array.append(f"object_storage_cluster = '{object_storage_cluster}'") + if partition_by: partition_by = "PARTITION BY " + partition_by @@ -264,6 +271,24 @@ def get_creation_expression( else: settings_expression = "" + cluster_name = "'cluster_simple'" if cluster_name_as_literal else "cluster_simple" + + storage_arg = storage_type + engine_part = "" + if (storage_type_in_named_collection): + storage_arg += "_with_type" + elif (storage_type_as_arg): + storage_arg += f", storage_type='{storage_type}'" + else: + if (storage_type == "s3"): + engine_part = "S3" + elif (storage_type == "azure"): + engine_part = "Azure" + elif (storage_type == "hdfs"): + engine_part = "HDFS" + elif (storage_type == "local"): + engine_part = "Local" + if_not_exists_prefix = "" if if_not_exists: if_not_exists_prefix = "IF NOT EXISTS" @@ -276,16 +301,16 @@ def get_creation_expression( if run_on_cluster: assert table_function - return f"icebergS3Cluster('cluster_simple', s3, filename = 'var/lib/clickhouse/user_files/iceberg_data/default/{table_name}/', format={format}, url = 'http://minio1:9001/{bucket}/')" + return f"iceberg{engine_part}Cluster({cluster_name}, {storage_arg}, filename = 'var/lib/clickhouse/user_files/iceberg_data/default/{table_name}/', format={format}, url = 'http://minio1:9001/{bucket}/')" else: if table_function: - return f"icebergS3(s3, filename = 'var/lib/clickhouse/user_files/iceberg_data/default/{table_name}/', format={format}, url = 'http://minio1:9001/{bucket}/')" + return f"iceberg{engine_part}({storage_arg}, filename = 'var/lib/clickhouse/user_files/iceberg_data/default/{table_name}/', format={format}, url = 'http://minio1:9001/{bucket}/')" else: return ( f""" DROP TABLE IF EXISTS {table_name}; CREATE TABLE {if_not_exists_prefix} {table_name} {schema} - ENGINE=IcebergS3(s3, filename = 'var/lib/clickhouse/user_files/iceberg_data/default/{table_name}/', format={format}, url = 'http://minio1:9001/{bucket}/') + ENGINE=Iceberg{engine_part}({storage_arg}, filename = 'var/lib/clickhouse/user_files/iceberg_data/default/{table_name}/', format={format}, url = 'http://minio1:9001/{bucket}/') {order_by} {partition_by} {settings_expression}; @@ -296,19 +321,19 @@ def get_creation_expression( if run_on_cluster: assert table_function return f""" - icebergAzureCluster('cluster_simple', azure, container = '{cluster.azure_container_name}', storage_account_url = '{cluster.env_variables["AZURITE_STORAGE_ACCOUNT_URL"]}', blob_path = '/var/lib/clickhouse/user_files/iceberg_data/default/{table_name}/', format={format}) + iceberg{engine_part}Cluster({cluster_name}, {storage_arg}, container = '{cluster.azure_container_name}', storage_account_url = '{cluster.env_variables["AZURITE_STORAGE_ACCOUNT_URL"]}', blob_path = '/var/lib/clickhouse/user_files/iceberg_data/default/{table_name}/', format={format}) """ else: if table_function: return f""" - icebergAzure(azure, container = '{cluster.azure_container_name}', storage_account_url = '{cluster.env_variables["AZURITE_STORAGE_ACCOUNT_URL"]}', blob_path = '/var/lib/clickhouse/user_files/iceberg_data/default/{table_name}/', format={format}) + iceberg{engine_part}({storage_arg}, container = '{cluster.azure_container_name}', storage_account_url = '{cluster.env_variables["AZURITE_STORAGE_ACCOUNT_URL"]}', blob_path = '/var/lib/clickhouse/user_files/iceberg_data/default/{table_name}/', format={format}) """ else: return ( f""" DROP TABLE IF EXISTS {table_name}; CREATE TABLE {if_not_exists_prefix} {table_name} {schema} - ENGINE=IcebergAzure(azure, container = {cluster.azure_container_name}, storage_account_url = '{cluster.env_variables["AZURITE_STORAGE_ACCOUNT_URL"]}', blob_path = '/var/lib/clickhouse/user_files/iceberg_data/default/{table_name}/', format={format}) + ENGINE=Iceberg{engine_part}({storage_arg}, container = {cluster.azure_container_name}, storage_account_url = '{cluster.env_variables["AZURITE_STORAGE_ACCOUNT_URL"]}', blob_path = '/var/lib/clickhouse/user_files/iceberg_data/default/{table_name}/', format={format}) {order_by} {partition_by} {settings_expression} @@ -319,19 +344,19 @@ def get_creation_expression( if run_on_cluster: assert table_function return f""" - icebergLocalCluster('cluster_simple', local, path = '/var/lib/clickhouse/user_files/iceberg_data/default/{table_name}', format={format}) + iceberg{engine_part}Cluster({cluster_name}, {storage_arg}, path = '/var/lib/clickhouse/user_files/iceberg_data/default/{table_name}/', format={format}) """ else: if table_function: return f""" - icebergLocal(local, path = '/var/lib/clickhouse/user_files/iceberg_data/default/{table_name}', format={format}) + iceberg{engine_part}({storage_arg}, path = '/var/lib/clickhouse/user_files/iceberg_data/default/{table_name}', format={format}) """ else: return ( f""" DROP TABLE IF EXISTS {table_name}; CREATE TABLE {if_not_exists_prefix} {table_name} {schema} - ENGINE=IcebergLocal(local, path = '/var/lib/clickhouse/user_files/iceberg_data/default/{table_name}', format={format}) + ENGINE=Iceberg{engine_part}({storage_arg}, path = '/var/lib/clickhouse/user_files/iceberg_data/default/{table_name}/', format={format}) {order_by} {partition_by} {settings_expression} @@ -422,6 +447,7 @@ def create_iceberg_table( run_on_cluster=False, format="Parquet", order_by="", +<<<<<<< HEAD settings=None, **kwargs, ): @@ -429,6 +455,46 @@ def create_iceberg_table( get_creation_expression(storage_type, table_name, cluster, schema, format_version, partition_by, if_not_exists, compression_method, format, order_by, run_on_cluster=run_on_cluster, **kwargs), settings=settings, ) +======= + object_storage_cluster=False, + **kwargs, +): + if 'output_format_parquet_use_custom_encoder' in kwargs: + node.query( + get_creation_expression( + storage_type, + table_name, + cluster, + schema, + format_version, + partition_by, + if_not_exists, + compression_method, + format, + order_by, + run_on_cluster=run_on_cluster, + object_storage_cluster=object_storage_cluster, + **kwargs), + settings={"output_format_parquet_use_custom_encoder" : 0, "output_format_parquet_parallel_encoding" : 0} + ) + else: + node.query( + get_creation_expression( + storage_type, + table_name, + cluster, + schema, + format_version, + partition_by, + if_not_exists, + compression_method, + format, + order_by, + run_on_cluster=run_on_cluster, + object_storage_cluster=object_storage_cluster, + **kwargs), + ) +>>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) def drop_iceberg_table( diff --git a/tests/integration/test_database_delta/test.py b/tests/integration/test_database_delta/test.py index 76ec56858eec..0385f458bb26 100644 --- a/tests/integration/test_database_delta/test.py +++ b/tests/integration/test_database_delta/test.py @@ -13,6 +13,9 @@ UC_LOG = "/var/lib/clickhouse/user_files/unitycatalog/uc.log" +CATALOG_NAME = "unity_catalog_test_db" + + def start_unity_catalog(node): node.exec_in_container( [ @@ -953,6 +956,7 @@ def get_table_versions(): ) +<<<<<<< HEAD @pytest.mark.parametrize("use_delta_kernel", ["1", "0"]) def test_varchar_char_types_via_unity_catalog(started_cluster, use_delta_kernel): """ @@ -1026,3 +1030,42 @@ def test_varchar_char_types_via_unity_catalog(started_cluster, use_delta_kernel) .strip() ) assert row == "1\thello varchar\thello char" +======= +def test_namespace_filter(started_cluster): + node = started_cluster.instances["node1"] + + # Use the same table name in all namespaces + table_name = f"table_{uuid.uuid4()}".replace("-", "_") + namespace_prefix = f"namespace_{uuid.uuid4()}_".replace("-", "_") + + + def create_namespace(suffix): + namespace = f"{namespace_prefix}{suffix}" + execute_spark_query( + node, f"CREATE SCHEMA {namespace}" + ) + execute_spark_query( + node, f"CREATE TABLE {namespace}.{table_name} (col1 int, col2 double) using Delta location '/var/lib/clickhouse/user_files/tmp/{namespace}/{table_name}'" + ) + + create_namespace("alpha"); + create_namespace("bravo"); + + node.query( + f""" + drop database if exists {CATALOG_NAME}; + create database {CATALOG_NAME} + engine DataLakeCatalog('http://localhost:8080/api/2.1/unity-catalog') + settings warehouse = 'unity', catalog_type='unity', vended_credentials=false, namespaces = '{namespace_prefix}alpha' + """, + settings={"allow_database_unity_catalog": "1"}, + ) + + assert node.query(f"SELECT name FROM system.tables WHERE database='{CATALOG_NAME}' ORDER BY name", settings={"show_data_lake_catalogs_in_system_tables": 1}) == TSV( + [ + [f"{namespace_prefix}alpha.{table_name}"], + ]) + + assert node.query(f"SELECT count() FROM {CATALOG_NAME}.`{namespace_prefix}alpha.{table_name}`") == "0\n" + assert "is filtered by `namespaces` database parameter." in node.query_and_get_error(f"SELECT count() FROM {CATALOG_NAME}.`{namespace_prefix}bravo.{table_name}`") +>>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) diff --git a/tests/integration/test_database_glue/test.py b/tests/integration/test_database_glue/test.py index 96a5eddcb7a2..b0a40879f302 100644 --- a/tests/integration/test_database_glue/test.py +++ b/tests/integration/test_database_glue/test.py @@ -16,6 +16,7 @@ from pyiceberg.table.sorting import SortField, SortOrder from pyiceberg.transforms import DayTransform, IdentityTransform from helpers.config_cluster import minio_access_key, minio_secret_key +from helpers.test_tools import TSV import decimal from pyiceberg.types import ( DoubleType, @@ -1068,6 +1069,7 @@ def test_table_without_metadata_location(started_cluster): node.query(f"DROP DATABASE IF EXISTS {db_name} SYNC") +<<<<<<< HEAD def test_check_database(started_cluster): """Test that CHECK DATABASE works with Glue catalog database.""" node = started_cluster.instances["node1"] @@ -1129,6 +1131,46 @@ def test_check_database(started_cluster): "SYSTEM DISABLE FAILPOINT check_database_datalake_negative" ) +======= +def test_namespace_filter(started_cluster): + node = started_cluster.instances["node1"] + + # Use the same table name in all namespaces + table_name = f"table_{uuid.uuid4()}" + table2_name = f"table2_{uuid.uuid4()}" + namespace_prefix = f"namespace_{uuid.uuid4()}_" + + catalog = load_catalog_impl(started_cluster) + + def create_namespace(suffix): + namespace = f"{namespace_prefix}{suffix}" + catalog.create_namespace(namespace) + create_table(catalog, namespace, table_name, DEFAULT_SCHEMA, PartitionSpec(), DEFAULT_SORT_ORDER) + + create_namespace("alpha"); + create_namespace("bravo"); + + create_clickhouse_glue_database(started_cluster, node, CATALOG_NAME, + additional_settings={ + "namespaces": f"{namespace_prefix}alpha" + }) + + assert node.query(f"SELECT name FROM system.tables WHERE database='{CATALOG_NAME}' ORDER BY name", settings={"show_data_lake_catalogs_in_system_tables": 1}) == TSV( + [ + [f"{namespace_prefix}alpha.{table_name}"], + ]) + + assert node.query(f"SELECT count() FROM {CATALOG_NAME}.`{namespace_prefix}alpha.{table_name}`") == "0\n" + assert "is filtered by `namespaces` database parameter." in node.query_and_get_error(f"SELECT count() FROM {CATALOG_NAME}.`{namespace_prefix}bravo.{table_name}`") + + node.query(f"CREATE TABLE {CATALOG_NAME}.`{namespace_prefix}alpha.{table2_name}` (x String) ENGINE = IcebergS3('http://minio:9000/warehouse-glue/{namespace_prefix}alpha/a1/{table2_name}/', '{minio_access_key}', '{minio_secret_key}')") + assert "is filtered by `namespaces` database parameter." in node.query_and_get_error(f"CREATE TABLE {CATALOG_NAME}.`{namespace_prefix}bravo.{table2_name}` (x String) ENGINE = IcebergS3('http://minio:9000/warehouse-glue/{namespace_prefix}bravo/{table2_name}/', '{minio_access_key}', '{minio_secret_key}')") + + node.query(f"DROP TABLE {CATALOG_NAME}.`{namespace_prefix}alpha.{table_name}`") + assert "is filtered by `namespaces` database parameter." in node.query_and_get_error(f"DROP TABLE {CATALOG_NAME}.`{namespace_prefix}bravo.{table_name}`") + + +>>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) def test_sts_smoke(started_cluster): """Test that STS authentication works with Glue catalog using role_arn and role_session_name""" node = started_cluster.instances["node1"] diff --git a/tests/integration/test_database_iceberg/configs/iceberg_partition_timezone.xml b/tests/integration/test_database_iceberg/configs/iceberg_partition_timezone.xml new file mode 100644 index 000000000000..40aebd33c515 --- /dev/null +++ b/tests/integration/test_database_iceberg/configs/iceberg_partition_timezone.xml @@ -0,0 +1,7 @@ + + + + UTC + + + diff --git a/tests/integration/test_database_iceberg/configs/timezone.xml b/tests/integration/test_database_iceberg/configs/timezone.xml new file mode 100644 index 000000000000..269e52ef2247 --- /dev/null +++ b/tests/integration/test_database_iceberg/configs/timezone.xml @@ -0,0 +1,3 @@ + + Asia/Istanbul + \ No newline at end of file diff --git a/tests/integration/test_database_iceberg/test.py b/tests/integration/test_database_iceberg/test.py index bb24e9329fcc..88ea5ae0b204 100644 --- a/tests/integration/test_database_iceberg/test.py +++ b/tests/integration/test_database_iceberg/test.py @@ -66,6 +66,8 @@ DEFAULT_SORT_ORDER = SortOrder(SortField(source_id=2, transform=IdentityTransform())) +AVAILABLE_ENGINES = ["DataLakeCatalog", "Iceberg"] + def list_namespaces(started_cluster): base_url_local = f"http://localhost:{started_cluster.iceberg_rest_catalog_port}/v1" @@ -118,7 +120,7 @@ def generate_record(): def create_clickhouse_iceberg_database( - started_cluster, node, name, additional_settings={} + started_cluster, node, name, additional_settings={}, engine='DataLakeCatalog' ): settings = { "catalog_type": "rest", @@ -131,7 +133,13 @@ def create_clickhouse_iceberg_database( node.query( f""" DROP DATABASE IF EXISTS {name}; +<<<<<<< HEAD CREATE DATABASE {name} ENGINE = DataLakeCatalog('{BASE_URL}', 'minio', '{minio_secret_key}') +======= +SET allow_database_iceberg=true; +SET write_full_path_in_iceberg_metadata=1; +CREATE DATABASE {name} ENGINE = {engine}('{BASE_URL}', 'minio', '{minio_secret_key}') +>>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) SETTINGS {",".join((k+"="+repr(v) for k, v in settings.items()))} """, settings={ @@ -216,7 +224,8 @@ def started_cluster(): cluster.shutdown() -def test_list_tables(started_cluster): +@pytest.mark.parametrize("engine", AVAILABLE_ENGINES) +def test_list_tables(started_cluster, engine): node = started_cluster.instances["node1"] root_namespace = f"clickhouse_{uuid.uuid4()}" @@ -247,7 +256,7 @@ def test_list_tables(started_cluster): for namespace in [namespace_1, namespace_2]: assert len(catalog.list_tables(namespace)) == 0 - create_clickhouse_iceberg_database(started_cluster, node, CATALOG_NAME) + create_clickhouse_iceberg_database(started_cluster, node, CATALOG_NAME, engine=engine) tables_list = "" for table in namespace_1_tables: @@ -282,6 +291,7 @@ def test_list_tables(started_cluster): ) +<<<<<<< HEAD def test_check_database(started_cluster): node = started_cluster.instances["node1"] @@ -347,6 +357,10 @@ def test_check_database(started_cluster): def test_many_namespaces(started_cluster): +======= +@pytest.mark.parametrize("engine", AVAILABLE_ENGINES) +def test_many_namespaces(started_cluster, engine): +>>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) node = started_cluster.instances["node1"] root_namespace_1 = f"A_{uuid.uuid4()}" root_namespace_2 = f"B_{uuid.uuid4()}" @@ -367,7 +381,7 @@ def test_many_namespaces(started_cluster): for table in tables: create_table(catalog, namespace, table) - create_clickhouse_iceberg_database(started_cluster, node, CATALOG_NAME) + create_clickhouse_iceberg_database(started_cluster, node, CATALOG_NAME, engine=engine) for namespace in namespaces: for table in tables: @@ -379,7 +393,8 @@ def test_many_namespaces(started_cluster): ) -def test_select(started_cluster): +@pytest.mark.parametrize("engine", AVAILABLE_ENGINES) +def test_select(started_cluster, engine): node = started_cluster.instances["node1"] test_ref = f"test_list_tables_{uuid.uuid4()}" @@ -407,7 +422,7 @@ def test_select(started_cluster): df = pa.Table.from_pylist(data) table.append(df) - create_clickhouse_iceberg_database(started_cluster, node, CATALOG_NAME) + create_clickhouse_iceberg_database(started_cluster, node, CATALOG_NAME, engine=engine) expected = DEFAULT_CREATE_TABLE.format(CATALOG_NAME, namespace, table_name) assert expected == node.query( @@ -421,7 +436,8 @@ def test_select(started_cluster): assert int(node.query(f"SELECT count() FROM system.iceberg_history WHERE table = '{namespace}.{table_name}' and database = '{CATALOG_NAME}'").strip()) == 1 -def test_hide_sensitive_info(started_cluster): +@pytest.mark.parametrize("engine", AVAILABLE_ENGINES) +def test_hide_sensitive_info(started_cluster, engine): node = started_cluster.instances["node1"] test_ref = f"test_hide_sensitive_info_{uuid.uuid4()}" @@ -434,6 +450,7 @@ def test_hide_sensitive_info(started_cluster): create_table(catalog, namespace, table_name) +<<<<<<< HEAD def check_secret_hidden(secret, additional_settings): settings = { "catalog_type": "rest", @@ -489,6 +506,23 @@ def test_no_secrets_in_logs(started_cluster): "allow_database_iceberg": 1, "write_full_path_in_iceberg_metadata": 1, }, +======= + create_clickhouse_iceberg_database( + started_cluster, + node, + CATALOG_NAME, + additional_settings={"catalog_credential": "SECRET_1"}, + engine=engine, + ) + assert "SECRET_1" not in node.query(f"SHOW CREATE DATABASE {CATALOG_NAME}") + + create_clickhouse_iceberg_database( + started_cluster, + node, + CATALOG_NAME, + additional_settings={"auth_header": "SECRET_2"}, + engine=engine, +>>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) ) qid_table = uuid.uuid4().hex @@ -549,7 +583,8 @@ def test_no_secrets_in_logs(started_cluster): assert minio_secret_key not in val -def test_tables_with_same_location(started_cluster): +@pytest.mark.parametrize("engine", AVAILABLE_ENGINES) +def test_tables_with_same_location(started_cluster, engine): node = started_cluster.instances["node1"] test_ref = f"test_tables_with_same_location_{uuid.uuid4()}" @@ -580,7 +615,7 @@ def record(key): df = pa.Table.from_pylist(data) table_2.append(df) - create_clickhouse_iceberg_database(started_cluster, node, CATALOG_NAME) + create_clickhouse_iceberg_database(started_cluster, node, CATALOG_NAME, engine=engine) assert 'aaa\naaa\naaa' == node.query(f"SELECT symbol FROM {CATALOG_NAME}.`{namespace}.{table_name}`").strip() assert 'bbb\nbbb\nbbb' == node.query(f"SELECT symbol FROM {CATALOG_NAME}.`{namespace}.{table_name_2}`").strip() @@ -725,6 +760,52 @@ def test_timestamps(started_cluster): assert node.query(f"SHOW CREATE TABLE {CATALOG_NAME}.`{root_namespace}.{table_name}`") == f"CREATE TABLE {CATALOG_NAME}.`{root_namespace}.{table_name}`\\n(\\n `timestamp` Nullable(DateTime64(6)),\\n `timestamptz` Nullable(DateTime64(6, \\'UTC\\'))\\n)\\nENGINE = Iceberg(\\'http://minio1:9001/warehouse-rest/data/\\', \\'minio\\', \\'[HIDDEN]\\')\n" assert node.query(f"SELECT * FROM {CATALOG_NAME}.`{root_namespace}.{table_name}`") == "2024-01-01 12:00:00.000000\t2024-01-01 12:00:00.000000\n" + # Berlin - UTC+1 at winter + # Istanbul - UTC+3 at winter + + # 'UTC' is default value, responce is equal to query above + assert node.query(f""" + SELECT * FROM {CATALOG_NAME}.`{root_namespace}.{table_name}` + SETTINGS iceberg_timezone_for_timestamptz='UTC' + """) == "2024-01-01 12:00:00.000000\t2024-01-01 12:00:00.000000\n" + # Timezone from setting + assert node.query(f""" + SELECT * FROM {CATALOG_NAME}.`{root_namespace}.{table_name}` + SETTINGS iceberg_timezone_for_timestamptz='Europe/Berlin' + """) == "2024-01-01 12:00:00.000000\t2024-01-01 13:00:00.000000\n" + # Empty value means session timezone, by default it is 'UTC' too + assert node.query(f""" + SELECT * FROM {CATALOG_NAME}.`{root_namespace}.{table_name}` + SETTINGS iceberg_timezone_for_timestamptz='' + """) == "2024-01-01 12:00:00.000000\t2024-01-01 12:00:00.000000\n" + # If session timezone is used, `timestamptz` does not changed, 'UTC' by default + assert node.query(f""" + SELECT * FROM {CATALOG_NAME}.`{root_namespace}.{table_name}` + SETTINGS session_timezone='Asia/Istanbul' + """) == "2024-01-01 15:00:00.000000\t2024-01-01 12:00:00.000000\n" + # Setiing `iceberg_timezone_for_timestamptz` does not affect `timestamp` column + assert node.query(f""" + SELECT * FROM {CATALOG_NAME}.`{root_namespace}.{table_name}` + SETTINGS session_timezone='Asia/Istanbul', iceberg_timezone_for_timestamptz='Europe/Berlin' + """) == "2024-01-01 15:00:00.000000\t2024-01-01 13:00:00.000000\n" + # Empty value, used non-default session timezone + assert node.query(f""" + SELECT * FROM {CATALOG_NAME}.`{root_namespace}.{table_name}` + SETTINGS session_timezone='Asia/Istanbul', iceberg_timezone_for_timestamptz='' + """) == "2024-01-01 15:00:00.000000\t2024-01-01 15:00:00.000000\n" + # Invalid timezone + assert "Invalid time zone: Foo/Bar" in node.query_and_get_error(f""" + SELECT * FROM {CATALOG_NAME}.`{root_namespace}.{table_name}` + SETTINGS iceberg_timezone_for_timestamptz='Foo/Bar' + """) + + assert node.query(f"SHOW CREATE TABLE {CATALOG_NAME}.`{root_namespace}.{table_name}` SETTINGS iceberg_timezone_for_timestamptz='UTC'") == f"CREATE TABLE {CATALOG_NAME}.`{root_namespace}.{table_name}`\\n(\\n `timestamp` Nullable(DateTime64(6)),\\n `timestamptz` Nullable(DateTime64(6, \\'UTC\\'))\\n)\\nENGINE = Iceberg(\\'http://minio:9000/warehouse-rest/data/\\', \\'minio\\', \\'[HIDDEN]\\')\n" + assert node.query(f"SHOW CREATE TABLE {CATALOG_NAME}.`{root_namespace}.{table_name}` SETTINGS iceberg_timezone_for_timestamptz='Europe/Berlin'") == f"CREATE TABLE {CATALOG_NAME}.`{root_namespace}.{table_name}`\\n(\\n `timestamp` Nullable(DateTime64(6)),\\n `timestamptz` Nullable(DateTime64(6, \\'Europe/Berlin\\'))\\n)\\nENGINE = Iceberg(\\'http://minio:9000/warehouse-rest/data/\\', \\'minio\\', \\'[HIDDEN]\\')\n" + + assert node.query(f"SELECT timezoneOf(timestamptz) FROM {CATALOG_NAME}.`{root_namespace}.{table_name}` LIMIT 1") == "UTC\n" + assert node.query(f"SELECT timezoneOf(timestamptz) FROM {CATALOG_NAME}.`{root_namespace}.{table_name}` LIMIT 1 SETTINGS iceberg_timezone_for_timestamptz='UTC'") == "UTC\n" + assert node.query(f"SELECT timezoneOf(timestamptz) FROM {CATALOG_NAME}.`{root_namespace}.{table_name}` LIMIT 1 SETTINGS iceberg_timezone_for_timestamptz='Europe/Berlin'") == "Europe/Berlin\n" + def test_insert(started_cluster): node = started_cluster.instances["node1"] @@ -1146,6 +1227,7 @@ def test_gcs(started_cluster): assert "Google cloud storage converts to S3" in str(err.value) +<<<<<<< HEAD def test_invalid_auth_header_format(started_cluster): node = started_cluster.instances["node1"] @@ -1245,6 +1327,75 @@ def test_iceberg_file_progress_callback(started_cluster): # TODO - turn on after merge alternative syntax def _test_cluster_joins(started_cluster): +======= +def test_namespace_filter(started_cluster): + node = started_cluster.instances["node1"] + + # Use the same table name in all namespaces + table_name = f"table_{uuid.uuid4()}" + table2_name = f"table2_{uuid.uuid4()}" + namespace_prefix = f"namespace_{uuid.uuid4()}_" + + catalog = load_catalog_impl(started_cluster) + + def create_namespace(suffix): + namespace = f"{namespace_prefix}{suffix}" + catalog.create_namespace(namespace) + create_table(catalog, namespace, table_name, DEFAULT_SCHEMA, PartitionSpec(), DEFAULT_SORT_ORDER) + + create_namespace("alpha"); + create_namespace("alpha.a1"); + create_namespace("alpha.a2"); + create_namespace("bravo"); + create_namespace("bravo.b1"); + create_namespace("charlie"); + create_namespace("charlie.c1"); + create_namespace("delta"); + create_namespace("delta.d1"); + create_namespace("delta.d2"); + create_namespace("echo"); + create_namespace("echo.e1"); + + create_clickhouse_iceberg_database(started_cluster, node, CATALOG_NAME, + additional_settings={ + "namespaces": f"{namespace_prefix}alpha,{namespace_prefix}alpha.a1,{namespace_prefix}bravo,{namespace_prefix}bravo.*,{namespace_prefix}charlie,{namespace_prefix}delta.d1,{namespace_prefix}echo.*" + }) + + assert node.query(f"SELECT name FROM system.tables WHERE database='{CATALOG_NAME}' ORDER BY name", settings={"show_data_lake_catalogs_in_system_tables": 1}) == TSV( + [ + [f"{namespace_prefix}alpha.a1.{table_name}"], + [f"{namespace_prefix}alpha.{table_name}"], + [f"{namespace_prefix}bravo.b1.{table_name}"], + [f"{namespace_prefix}bravo.{table_name}"], + [f"{namespace_prefix}charlie.{table_name}"], + [f"{namespace_prefix}delta.d1.{table_name}"], + [f"{namespace_prefix}echo.e1.{table_name}"], + ]) + + assert node.query(f"SELECT count() FROM {CATALOG_NAME}.`{namespace_prefix}alpha.{table_name}`") == "0\n" + assert node.query(f"SELECT count() FROM {CATALOG_NAME}.`{namespace_prefix}alpha.a1.{table_name}`") == "0\n" + assert "is filtered by `namespaces` database parameter." in node.query_and_get_error(f"SELECT count() FROM {CATALOG_NAME}.`{namespace_prefix}alpha.a2.{table_name}`") + assert node.query(f"SELECT count() FROM {CATALOG_NAME}.`{namespace_prefix}bravo.{table_name}`") == "0\n" + assert node.query(f"SELECT count() FROM {CATALOG_NAME}.`{namespace_prefix}bravo.b1.{table_name}`") == "0\n" + assert node.query(f"SELECT count() FROM {CATALOG_NAME}.`{namespace_prefix}charlie.{table_name}`") == "0\n" + assert "is filtered by `namespaces` database parameter." in node.query_and_get_error(f"SELECT count() FROM {CATALOG_NAME}.`{namespace_prefix}charlie.c1.{table_name}`") + assert "is filtered by `namespaces` database parameter." in node.query_and_get_error(f"SELECT count() FROM {CATALOG_NAME}.`{namespace_prefix}delta.{table_name}`") + assert node.query(f"SELECT count() FROM {CATALOG_NAME}.`{namespace_prefix}delta.d1.{table_name}`") == "0\n" + assert "is filtered by `namespaces` database parameter." in node.query_and_get_error(f"SELECT count() FROM {CATALOG_NAME}.`{namespace_prefix}delta.d2.{table_name}`") + assert "is filtered by `namespaces` database parameter." in node.query_and_get_error(f"SELECT count() FROM {CATALOG_NAME}.`{namespace_prefix}echo.{table_name}`") + assert node.query(f"SELECT count() FROM {CATALOG_NAME}.`{namespace_prefix}echo.e1.{table_name}`") == "0\n" + + node.query(f"CREATE TABLE {CATALOG_NAME}.`{namespace_prefix}alpha.{table2_name}` (x String) ENGINE = IcebergS3('http://minio:9000/warehouse-rest/{namespace_prefix}alpha/{table2_name}/', '{minio_access_key}', '{minio_secret_key}')") + node.query(f"CREATE TABLE {CATALOG_NAME}.`{namespace_prefix}alpha.a1.{table2_name}` (x String) ENGINE = IcebergS3('http://minio:9000/warehouse-rest/{namespace_prefix}alpha/a1/{table2_name}/', '{minio_access_key}', '{minio_secret_key}')") + assert "is filtered by `namespaces` database parameter." in node.query_and_get_error(f"CREATE TABLE {CATALOG_NAME}.`{namespace_prefix}alpha.a2.{table2_name}` (x String) ENGINE = IcebergS3('http://minio:9000/warehouse-rest/{namespace_prefix}alpha/a2/{table2_name}/', '{minio_access_key}', '{minio_secret_key}')") + + node.query(f"DROP TABLE {CATALOG_NAME}.`{namespace_prefix}alpha.{table_name}`") + node.query(f"DROP TABLE {CATALOG_NAME}.`{namespace_prefix}alpha.a1.{table_name}`") + assert "is filtered by `namespaces` database parameter." in node.query_and_get_error(f"DROP TABLE {CATALOG_NAME}.`{namespace_prefix}alpha.a2.{table_name}`") + + +def test_cluster_joins(started_cluster): +>>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) node = started_cluster.instances["node1"] test_ref = f"test_join_tables_{uuid.uuid4()}" diff --git a/tests/integration/test_database_iceberg/test_partition_timezone.py b/tests/integration/test_database_iceberg/test_partition_timezone.py new file mode 100644 index 000000000000..1a43f3481ad7 --- /dev/null +++ b/tests/integration/test_database_iceberg/test_partition_timezone.py @@ -0,0 +1,186 @@ +import glob +import json +import logging +import os +import random +import time +import uuid +from datetime import datetime, timedelta + +import pyarrow as pa +import pytest +import requests +import urllib3 +import pytz +from minio import Minio +from pyiceberg.catalog import load_catalog +from pyiceberg.partitioning import PartitionField, PartitionSpec, UNPARTITIONED_PARTITION_SPEC +from pyiceberg.schema import Schema +from pyiceberg.table.sorting import SortField, SortOrder +from pyiceberg.transforms import DayTransform, IdentityTransform +from pyiceberg.types import ( + DoubleType, + LongType, + FloatType, + NestedField, + StringType, + StructType, + TimestampType, + TimestamptzType +) +from pyiceberg.table.sorting import UNSORTED_SORT_ORDER + +from helpers.cluster import ClickHouseCluster, ClickHouseInstance, is_arm +from helpers.config_cluster import minio_secret_key, minio_access_key +from helpers.s3_tools import get_file_contents, list_s3_objects, prepare_s3_bucket +from helpers.test_tools import TSV, csv_compare +from helpers.config_cluster import minio_secret_key + +ICEBERG_PORT = 8183 + +BASE_URL = "http://rest:8181/v1" +BASE_URL_LOCAL = f"http://localhost:{ICEBERG_PORT}/v1" +BASE_URL_LOCAL_RAW = f"http://localhost:{ICEBERG_PORT}" + +CATALOG_NAME = "demo" + +DEFAULT_PARTITION_SPEC = PartitionSpec( + PartitionField( + source_id=1, field_id=1000, transform=DayTransform(), name="datetime_day" + ) +) +DEFAULT_SORT_ORDER = SortOrder(SortField(source_id=1, transform=DayTransform())) +DEFAULT_SCHEMA = Schema( + NestedField(field_id=1, name="datetime", field_type=TimestampType(), required=False), + NestedField(field_id=2, name="value", field_type=LongType(), required=False), +) + + +@pytest.fixture(scope="module") +def started_cluster(): + try: + cluster = ClickHouseCluster(__file__) + cluster.iceberg_rest_external_port = ICEBERG_PORT + cluster.spark_iceberg_external_port = 10004 + cluster.spark_iceberg_external_port_2 = 10005 + cluster.spark_iceberg_external_port_3 = 10006 + cluster.add_instance( + "node1", + main_configs=["configs/timezone.xml", "configs/cluster.xml"], + user_configs=["configs/iceberg_partition_timezone.xml"], + stay_alive=True, + with_iceberg_catalog=True, + with_zookeeper=True, + ) + + logging.info("Starting cluster...") + cluster.start() + + # TODO: properly wait for container + time.sleep(10) + + yield cluster + + finally: + cluster.shutdown() + + +def load_catalog_impl(started_cluster): + return load_catalog( + CATALOG_NAME, + **{ + "uri": BASE_URL_LOCAL_RAW, + "type": "rest", + "s3.endpoint": f"http://{started_cluster.get_instance_ip('minio')}:9000", + "s3.access-key-id": minio_access_key, + "s3.secret-access-key": minio_secret_key, + }, + ) + + +def create_table( + catalog, + namespace, + table, + schema=DEFAULT_SCHEMA, + partition_spec=DEFAULT_PARTITION_SPEC, + sort_order=DEFAULT_SORT_ORDER, +): + return catalog.create_table( + identifier=f"{namespace}.{table}", + schema=schema, + location=f"s3://warehouse-rest/data", + partition_spec=partition_spec, + sort_order=sort_order, + ) + + +def create_clickhouse_iceberg_database( + node, name, additional_settings={}, engine='DataLakeCatalog' +): + settings = { + "catalog_type": "rest", + "warehouse": "demo", + "storage_endpoint": "http://minio:9000/warehouse-rest", + } + + settings.update(additional_settings) + + node.query( + f""" +DROP DATABASE IF EXISTS {name}; +SET allow_database_iceberg=true; +SET write_full_path_in_iceberg_metadata=1; +CREATE DATABASE {name} ENGINE = {engine}('{BASE_URL}', 'minio', '{minio_secret_key}') +SETTINGS {",".join((k+"="+repr(v) for k, v in settings.items()))} + """ + ) + show_result = node.query(f"SHOW DATABASE {name}") + assert minio_secret_key not in show_result + assert "HIDDEN" in show_result + + +def test_partition_timezone(started_cluster): + catalog = load_catalog_impl(started_cluster) + namespace = f"timezone_ns_{uuid.uuid4()}" + table_name = f"tz_table__{uuid.uuid4()}" + catalog.create_namespace(namespace) + table = create_table( + catalog, + namespace, + table_name, + ) + + # catalog accept data in UTC + data = [{"datetime": datetime(2024, 1, 1, 20, 0), "value": 1}, # partition 20240101 + {"datetime": datetime(2024, 1, 1, 23, 0), "value": 2}, # partition 20240101 + {"datetime": datetime(2024, 1, 2, 2, 0), "value": 3}] # partition 20240102 + df = pa.Table.from_pylist(data) + table.append(df) + + node = started_cluster.instances["node1"] + create_clickhouse_iceberg_database(node, CATALOG_NAME) + + # server timezone is Asia/Istanbul (UTC+3) + assert node.query(f""" + SELECT datetime, value + FROM {CATALOG_NAME}.`{namespace}.{table_name}` + ORDER BY datetime + """, timeout=10) == TSV( + [ + ["2024-01-01 23:00:00.000000", 1], + ["2024-01-02 02:00:00.000000", 2], + ["2024-01-02 05:00:00.000000", 3], + ]) + + # partitioning works correctly + assert node.query(f""" + SELECT datetime, value + FROM {CATALOG_NAME}.`{namespace}.{table_name}` + WHERE datetime >= '2024-01-02 00:00:00' + ORDER BY datetime + """, timeout=10) == TSV( + [ + ["2024-01-02 02:00:00.000000", 2], + ["2024-01-02 05:00:00.000000", 3], + ]) diff --git a/tests/integration/test_mask_sensitive_info/test.py b/tests/integration/test_mask_sensitive_info/test.py index a30d0ec40d63..b160f9fed188 100644 --- a/tests/integration/test_mask_sensitive_info/test.py +++ b/tests/integration/test_mask_sensitive_info/test.py @@ -3,6 +3,7 @@ import string import pytest +import uuid from helpers.cluster import ClickHouseCluster from helpers.test_tools import TSV @@ -247,6 +248,8 @@ def test_create_table(): azure_account_name = "devstoreaccount1" azure_account_key = "Eby8vdM02xNOcqFlqUwJPLlmEtlCDXJ1OUzFT50uSRZ6IFsuFq2UVErCz4I6tq/K1SZFPTOtr/KBHBeksoGMGw==" + table_suffix = uuid.uuid4().hex + table_engines = [ f"MySQL('mysql80:3306', 'mysql_db', 'mysql_table', 'mysql_user', '{password}')", f"PostgreSQL('postgres1:5432', 'postgres_db', 'postgres_table', 'postgres_user', '{password}')", @@ -278,19 +281,43 @@ def test_create_table(): f"IcebergS3('http://minio1:9001/root/data/test11.csv.gz', 'minio', '{password}')", "DNS_ERROR", ), + ( + f"Iceberg(storage_type='s3', 'http://minio1:9001/root/data/test11.csv.gz', 'minio', '{password}')", + "DNS_ERROR", + ), f"AzureBlobStorage('{azure_conn_string}', 'cont', 'test_simple.csv', 'CSV')", f"AzureBlobStorage('{azure_conn_string}', 'cont', 'test_simple_1.csv', 'CSV', 'none')", - f"AzureBlobStorage('{azure_storage_account_url}', 'cont', 'test_simple_2.csv', '{azure_account_name}', '{azure_account_key}')", - f"AzureBlobStorage('{azure_storage_account_url}', 'cont', 'test_simple_3.csv', '{azure_account_name}', '{azure_account_key}', 'CSV')", - f"AzureBlobStorage('{azure_storage_account_url}', 'cont', 'test_simple_4.csv', '{azure_account_name}', '{azure_account_key}', 'CSV', 'none')", + f"AzureQueue('{azure_conn_string}', 'cont', '*', 'CSV') SETTINGS mode = 'unordered'", f"AzureQueue('{azure_conn_string}', 'cont', '*', 'CSV', 'none') SETTINGS mode = 'unordered'", f"AzureQueue('{azure_conn_string}', 'cont', '*', 'CSV') SETTINGS mode = 'unordered', after_processing = 'move', after_processing_move_connection_string = '{azure_conn_string}', after_processing_move_container = 'chprocessed'", f"AzureQueue('{azure_conn_string}', 'cont', '*', 'CSV') SETTINGS mode = 'unordered', after_processing = 'move', after_processing_move_connection_string = '{azure_sas_conn_string}', after_processing_move_container = 'chprocessed'", f"AzureQueue('{azure_storage_account_url}', 'cont', '*', '{azure_account_name}', '{azure_account_key}', 'CSV') SETTINGS mode = 'unordered'", f"AzureQueue('{azure_storage_account_url}', 'cont', '*', '{azure_account_name}', '{azure_account_key}', 'CSV', 'none') SETTINGS mode = 'unordered'", +<<<<<<< HEAD "AzureBlobStorage('BlobEndpoint=https://my-endpoint/;SharedAccessSignature=sp=r&st=2025-09-29T14:58:11Z&se=2025-09-29T00:00:00Z&spr=https&sv=2022-11-02&sr=c&sig=SECRET%SECRET%SECRET%SECRET', 'exampledatasets', 'example.csv')", f"S3('https://my-s3-endpoint/bucket/data.csv', 'myaccess', '{password}', 'CSV')", +======= + ( + f"AzureBlobStorage('BlobEndpoint=https://my-endpoint/;SharedAccessSignature=sp=r&st=2025-09-29T14:58:11Z&se=2025-09-29T00:00:00Z&spr=https&sv=2022-11-02&sr=c&sig=SECRET%SECRET%SECRET%SECRET', 'exampledatasets', 'example.csv')", + "STD_EXCEPTION", + ), + + f"AzureBlobStorage(named_collection_2, connection_string = '{azure_conn_string}', container = 'cont', blob_path = 'test_simple_7.csv', format = 'CSV')", + f"AzureBlobStorage(named_collection_2, storage_account_url = '{azure_storage_account_url}', container = 'cont', blob_path = 'test_simple_8.csv', account_name = '{azure_account_name}', account_key = '{azure_account_key}')", + f"AzureBlobStorage('{azure_storage_account_url}', 'cont', 'test_simple_3.csv', '{azure_account_name}', '{azure_account_key}')", + f"AzureBlobStorage('{azure_storage_account_url}', 'cont', 'test_simple_4.csv', '{azure_account_name}', '{azure_account_key}', 'CSV')", + f"AzureBlobStorage('{azure_storage_account_url}', 'cont', 'test_simple_5.csv', '{azure_account_name}', '{azure_account_key}', 'CSV', 'none')", + f"IcebergAzure('{azure_conn_string}', 'cont', 'test_simple_0_{table_suffix}.csv')", + f"IcebergAzure('{azure_storage_account_url}', 'cont', 'test_simple_1_{table_suffix}.csv', '{azure_account_name}', '{azure_account_key}')", + f"IcebergAzure(named_collection_2, connection_string = '{azure_conn_string}', container = 'cont', blob_path = 'test_simple_2_{table_suffix}.csv', format = 'CSV')", + f"IcebergAzure(named_collection_2, storage_account_url = '{azure_storage_account_url}', container = 'cont', blob_path = 'test_simple_3_{table_suffix}.csv', account_name = '{azure_account_name}', account_key = '{azure_account_key}')", + f"Iceberg(storage_type='azure', '{azure_conn_string}', 'cont', 'test_simple_4_{table_suffix}.csv')", + f"Iceberg(storage_type='azure', '{azure_storage_account_url}', 'cont', 'test_simple_5_{table_suffix}.csv', '{azure_account_name}', '{azure_account_key}')", + f"Iceberg(storage_type='azure', named_collection_2, connection_string = '{azure_conn_string}', container = 'cont', blob_path = 'test_simple_6_{table_suffix}.csv', format = 'CSV')", + f"Iceberg(storage_type='azure', named_collection_2, storage_account_url = '{azure_storage_account_url}', container = 'cont', blob_path = 'test_simple_7_{table_suffix}.csv', account_name = '{azure_account_name}', account_key = '{azure_account_key}')", + +>>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) f"Kafka() SETTINGS kafka_broker_list = '127.0.0.1', kafka_topic_list = 'topic', kafka_group_name = 'group', kafka_format = 'JSONEachRow', kafka_security_protocol = 'sasl_ssl', kafka_sasl_mechanism = 'PLAIN', kafka_sasl_username = 'user', kafka_sasl_password = '{password}', format_avro_schema_registry_url = 'http://schema_user:{password}@'", f"Kafka() SETTINGS kafka_broker_list = '127.0.0.1', kafka_topic_list = 'topic', kafka_group_name = 'group', kafka_format = 'JSONEachRow', kafka_security_protocol = 'sasl_ssl', kafka_sasl_mechanism = 'PLAIN', kafka_sasl_username = 'user', kafka_sasl_password = '{password}', format_avro_schema_registry_url = 'http://schema_user:{password}@domain.com'", f"S3('http://minio1:9001/root/data/test5.csv.gz', 'CSV', access_key_id = 'minio', secret_access_key = '{password}', compression_method = 'gzip')", @@ -312,7 +339,7 @@ def test_create_table(): ] def make_test_case(i): - table_name = f"table{i}" + table_name = f"table{i}_{table_suffix}" table_engine = table_engines[i] error = None if isinstance(table_engine, tuple): @@ -331,18 +358,18 @@ def make_test_case(i): for toggle, secret in enumerate(["[HIDDEN]", password]): assert ( - node.query(f"SHOW CREATE TABLE table0 {show_secrets}={toggle}") - == "CREATE TABLE default.table0\\n(\\n `x` Int32\\n)\\n" + node.query(f"SHOW CREATE TABLE table0_{table_suffix} {show_secrets}={toggle}") + == f"CREATE TABLE default.table0_{table_suffix}\\n(\\n `x` Int32\\n)\\n" "ENGINE = MySQL(\\'mysql80:3306\\', \\'mysql_db\\', " f"\\'mysql_table\\', \\'mysql_user\\', \\'{secret}\\')\n" ) assert node.query( - f"SELECT create_table_query, engine_full FROM system.tables WHERE name = 'table0' {show_secrets}={toggle}" + f"SELECT create_table_query, engine_full FROM system.tables WHERE name = 'table0_{table_suffix}' {show_secrets}={toggle}" ) == TSV( [ [ - "CREATE TABLE default.table0 (`x` Int32) ENGINE = MySQL(\\'mysql80:3306\\', \\'mysql_db\\', " + f"CREATE TABLE default.table0_{table_suffix} (`x` Int32) ENGINE = MySQL(\\'mysql80:3306\\', \\'mysql_db\\', " f"\\'mysql_table\\', \\'mysql_user\\', \\'{secret}\\')", f"MySQL(\\'mysql80:3306\\', \\'mysql_db\\', \\'mysql_table\\', \\'mysql_user\\', \\'{secret}\\')", ], @@ -352,7 +379,7 @@ def make_test_case(i): create_table_statement_counter = 0 def generate_create_table_numbered(tail): nonlocal create_table_statement_counter - result = f"CREATE TABLE table{create_table_statement_counter} {tail}" + result = f"CREATE TABLE table{create_table_statement_counter}_{table_suffix} {tail}" create_table_statement_counter += 1 return result @@ -383,11 +410,9 @@ def generate_create_table_numbered(tail): generate_create_table_numbered("(`x` int) ENGINE = S3Queue('http://minio1:9001/root/data/', 'CSV') SETTINGS mode = 'ordered', after_processing = 'move', after_processing_move_uri = 'http://minio1:9001/chprocessed', after_processing_move_access_key_id = 'minio', after_processing_move_secret_access_key = '[HIDDEN]'"), generate_create_table_numbered("(`x` int) ENGINE = Iceberg('http://minio1:9001/root/data/test11.csv.gz', 'minio', '[HIDDEN]')"), generate_create_table_numbered("(`x` int) ENGINE = IcebergS3('http://minio1:9001/root/data/test11.csv.gz', 'minio', '[HIDDEN]')"), + generate_create_table_numbered("(`x` int) ENGINE = Iceberg(storage_type = 's3', 'http://minio1:9001/root/data/test11.csv.gz', 'minio', '[HIDDEN]')"), generate_create_table_numbered(f"(`x` int) ENGINE = AzureBlobStorage('{masked_azure_conn_string}', 'cont', 'test_simple.csv', 'CSV')"), generate_create_table_numbered(f"(`x` int) ENGINE = AzureBlobStorage('{masked_azure_conn_string}', 'cont', 'test_simple_1.csv', 'CSV', 'none')"), - generate_create_table_numbered(f"(`x` int) ENGINE = AzureBlobStorage('{azure_storage_account_url}', 'cont', 'test_simple_2.csv', '{azure_account_name}', '[HIDDEN]')"), - generate_create_table_numbered(f"(`x` int) ENGINE = AzureBlobStorage('{azure_storage_account_url}', 'cont', 'test_simple_3.csv', '{azure_account_name}', '[HIDDEN]', 'CSV')"), - generate_create_table_numbered(f"(`x` int) ENGINE = AzureBlobStorage('{azure_storage_account_url}', 'cont', 'test_simple_4.csv', '{azure_account_name}', '[HIDDEN]', 'CSV', 'none')"), generate_create_table_numbered(f"(`x` int) ENGINE = AzureQueue('{masked_azure_conn_string}', 'cont', '*', 'CSV') SETTINGS mode = 'unordered'"), generate_create_table_numbered(f"(`x` int) ENGINE = AzureQueue('{masked_azure_conn_string}', 'cont', '*', 'CSV', 'none') SETTINGS mode = 'unordered'"), generate_create_table_numbered(f"(`x` int) ENGINE = AzureQueue('{masked_azure_conn_string}', 'cont', '*', 'CSV') SETTINGS mode = 'unordered', after_processing = 'move', after_processing_move_connection_string = '{masked_azure_conn_string}', after_processing_move_container = 'chprocessed'",), @@ -395,7 +420,23 @@ def generate_create_table_numbered(tail): generate_create_table_numbered(f"(`x` int) ENGINE = AzureQueue('{azure_storage_account_url}', 'cont', '*', '{azure_account_name}', '[HIDDEN]', 'CSV') SETTINGS mode = 'unordered'"), generate_create_table_numbered(f"(`x` int) ENGINE = AzureQueue('{azure_storage_account_url}', 'cont', '*', '{azure_account_name}', '[HIDDEN]', 'CSV', 'none') SETTINGS mode = 'unordered'"), generate_create_table_numbered(f"(`x` int) ENGINE = AzureBlobStorage('{masked_sas_conn_string}', 'exampledatasets', 'example.csv')"), +<<<<<<< HEAD generate_create_table_numbered("(`x` int) ENGINE = S3('https://my-s3-endpoint/bucket/data.csv', 'myaccess', '[HIDDEN]', 'CSV')"), +======= + generate_create_table_numbered(f"(`x` int) ENGINE = AzureBlobStorage(named_collection_2, connection_string = '{masked_azure_conn_string}', container = 'cont', blob_path = 'test_simple_7.csv', format = 'CSV')"), + generate_create_table_numbered(f"(`x` int) ENGINE = AzureBlobStorage(named_collection_2, storage_account_url = '{azure_storage_account_url}', container = 'cont', blob_path = 'test_simple_8.csv', account_name = '{azure_account_name}', account_key = '[HIDDEN]')"), + generate_create_table_numbered(f"(`x` int) ENGINE = AzureBlobStorage('{azure_storage_account_url}', 'cont', 'test_simple_3.csv', '{azure_account_name}', '[HIDDEN]')"), + generate_create_table_numbered(f"(`x` int) ENGINE = AzureBlobStorage('{azure_storage_account_url}', 'cont', 'test_simple_4.csv', '{azure_account_name}', '[HIDDEN]', 'CSV')"), + generate_create_table_numbered(f"(`x` int) ENGINE = AzureBlobStorage('{azure_storage_account_url}', 'cont', 'test_simple_5.csv', '{azure_account_name}', '[HIDDEN]', 'CSV', 'none')"), + generate_create_table_numbered(f"(`x` int) ENGINE = IcebergAzure('{masked_azure_conn_string}', 'cont', 'test_simple_0_{table_suffix}.csv')"), + generate_create_table_numbered(f"(`x` int) ENGINE = IcebergAzure('{azure_storage_account_url}', 'cont', 'test_simple_1_{table_suffix}.csv', '{azure_account_name}', '[HIDDEN]')"), + generate_create_table_numbered(f"(`x` int) ENGINE = IcebergAzure(named_collection_2, connection_string = '{masked_azure_conn_string}', container = 'cont', blob_path = 'test_simple_2_{table_suffix}.csv', format = 'CSV')"), + generate_create_table_numbered(f"(`x` int) ENGINE = IcebergAzure(named_collection_2, storage_account_url = '{azure_storage_account_url}', container = 'cont', blob_path = 'test_simple_3_{table_suffix}.csv', account_name = '{azure_account_name}', account_key = '[HIDDEN]')"), + generate_create_table_numbered(f"(`x` int) ENGINE = Iceberg(storage_type = 'azure', '{masked_azure_conn_string}', 'cont', 'test_simple_4_{table_suffix}.csv')"), + generate_create_table_numbered(f"(`x` int) ENGINE = Iceberg(storage_type = 'azure', '{azure_storage_account_url}', 'cont', 'test_simple_5_{table_suffix}.csv', '{azure_account_name}', '[HIDDEN]')"), + generate_create_table_numbered(f"(`x` int) ENGINE = Iceberg(storage_type = 'azure', named_collection_2, connection_string = '{masked_azure_conn_string}', container = 'cont', blob_path = 'test_simple_6_{table_suffix}.csv', format = 'CSV')"), + generate_create_table_numbered(f"(`x` int) ENGINE = Iceberg(storage_type = 'azure', named_collection_2, storage_account_url = '{azure_storage_account_url}', container = 'cont', blob_path = 'test_simple_7_{table_suffix}.csv', account_name = '{azure_account_name}', account_key = '[HIDDEN]')"), +>>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) generate_create_table_numbered("(`x` int) ENGINE = Kafka SETTINGS kafka_broker_list = '127.0.0.1', kafka_topic_list = 'topic', kafka_group_name = 'group', kafka_format = 'JSONEachRow', kafka_security_protocol = 'sasl_ssl', kafka_sasl_mechanism = 'PLAIN', kafka_sasl_username = 'user', kafka_sasl_password = '[HIDDEN]', format_avro_schema_registry_url = 'http://schema_user:[HIDDEN]@'"), generate_create_table_numbered("(`x` int) ENGINE = Kafka SETTINGS kafka_broker_list = '127.0.0.1', kafka_topic_list = 'topic', kafka_group_name = 'group', kafka_format = 'JSONEachRow', kafka_security_protocol = 'sasl_ssl', kafka_sasl_mechanism = 'PLAIN', kafka_sasl_username = 'user', kafka_sasl_password = '[HIDDEN]', format_avro_schema_registry_url = 'http://schema_user:[HIDDEN]@domain.com'"), generate_create_table_numbered("(`x` int) ENGINE = S3('http://minio1:9001/root/data/test5.csv.gz', 'CSV', access_key_id = 'minio', secret_access_key = '[HIDDEN]', compression_method = 'gzip')"), @@ -535,9 +576,22 @@ def test_table_functions(): f"azureBlobStorage(named_collection_2, connection_string = '{azure_conn_string}', container = 'cont', blob_path = 'test_simple_7.csv', format = 'CSV')", f"azureBlobStorage(named_collection_2, storage_account_url = '{azure_storage_account_url}', container = 'cont', blob_path = 'test_simple_8.csv', account_name = '{azure_account_name}', account_key = '{azure_account_key}')", f"iceberg('http://minio1:9001/root/data/test11.csv.gz', 'minio', '{password}')", - f"gcs('http://minio1:9001/root/data/test11.csv.gz', 'minio', '{password}')", + f"iceberg(named_collection_2, url = 'http://minio1:9001/root/data/test4.csv', access_key_id = 'minio', secret_access_key = '{password}')", f"icebergS3('http://minio1:9001/root/data/test11.csv.gz', 'minio', '{password}')", + f"icebergS3(named_collection_2, url = 'http://minio1:9001/root/data/test4.csv', access_key_id = 'minio', secret_access_key = '{password}')", + f"icebergAzure('{azure_conn_string}', 'cont', 'test_simple.csv')", + f"icebergAzure('{azure_storage_account_url}', 'cont', 'test_simple.csv', '{azure_account_name}', '{azure_account_key}')", f"icebergAzure('{azure_storage_account_url}', 'cont', 'test_simple_6.csv', '{azure_account_name}', '{azure_account_key}', 'CSV', 'none', 'auto')", + f"icebergAzure(named_collection_2, connection_string = '{azure_conn_string}', container = 'cont', blob_path = 'test_simple_7.csv', format = 'CSV')", + f"icebergAzure(named_collection_2, storage_account_url = '{azure_storage_account_url}', container = 'cont', blob_path = 'test_simple_8.csv', account_name = '{azure_account_name}', account_key = '{azure_account_key}')", + f"iceberg(storage_type='s3', 'http://minio1:9001/root/data/test11.csv.gz', 'minio', '{password}')", + f"iceberg(storage_type='s3', named_collection_2, url = 'http://minio1:9001/root/data/test4.csv', access_key_id = 'minio', secret_access_key = '{password}')", + f"iceberg(storage_type='azure', '{azure_conn_string}', 'cont', 'test_simple.csv')", + f"iceberg(storage_type='azure', '{azure_storage_account_url}', 'cont', 'test_simple.csv', '{azure_account_name}', '{azure_account_key}')", + f"iceberg(storage_type='azure', '{azure_storage_account_url}', 'cont', 'test_simple_6.csv', '{azure_account_name}', '{azure_account_key}', 'CSV', 'none', 'auto')", + f"iceberg(storage_type='azure', named_collection_2, connection_string = '{azure_conn_string}', container = 'cont', blob_path = 'test_simple_7.csv', format = 'CSV')", + f"iceberg(storage_type='azure', named_collection_2, storage_account_url = '{azure_storage_account_url}', container = 'cont', blob_path = 'test_simple_8.csv', account_name = '{azure_account_name}', account_key = '{azure_account_key}')", + f"gcs('http://minio1:9001/root/data/test11.csv.gz', 'minio', '{password}')", f"deltaLakeAzure('{azure_storage_account_url}', 'cont', 'test_simple_6.csv', '{azure_account_name}', '{azure_account_key}', 'CSV', 'none', 'auto')" if has_delta_lake else (f"deltaLakeAzure('{azure_storage_account_url}', 'cont', 'test_simple_6.csv', '{azure_account_name}', '{azure_account_key}', 'CSV', 'none', 'auto')", "UNKNOWN_FUNCTION"), f"hudi('http://minio1:9001/root/data/test7.csv', 'minio', '{password}')", f"arrowFlight('arrowflight1:5006', 'dataset', 'arrowflight_user', '{password}')", @@ -642,8 +696,9 @@ def make_test_case(i): f"CREATE TABLE tablefunc37 (`x` int) AS azureBlobStorage(named_collection_2, connection_string = '{masked_azure_conn_string}', container = 'cont', blob_path = 'test_simple_7.csv', format = 'CSV')", f"CREATE TABLE tablefunc38 (`x` int) AS azureBlobStorage(named_collection_2, storage_account_url = '{azure_storage_account_url}', container = 'cont', blob_path = 'test_simple_8.csv', account_name = '{azure_account_name}', account_key = '[HIDDEN]')", "CREATE TABLE tablefunc39 (`x` int) AS iceberg('http://minio1:9001/root/data/test11.csv.gz', 'minio', '[HIDDEN]')", - "CREATE TABLE tablefunc40 (`x` int) AS gcs('http://minio1:9001/root/data/test11.csv.gz', 'minio', '[HIDDEN]')", + "CREATE TABLE tablefunc40 (`x` int) AS iceberg(named_collection_2, url = 'http://minio1:9001/root/data/test4.csv', access_key_id = 'minio', secret_access_key = '[HIDDEN]')", "CREATE TABLE tablefunc41 (`x` int) AS icebergS3('http://minio1:9001/root/data/test11.csv.gz', 'minio', '[HIDDEN]')", +<<<<<<< HEAD f"CREATE TABLE tablefunc42 (`x` int) AS icebergAzure('{azure_storage_account_url}', 'cont', 'test_simple_6.csv', '{azure_account_name}', '[HIDDEN]', 'CSV', 'none', 'auto')", f"CREATE TABLE tablefunc43 (`x` int) AS deltaLakeAzure('{azure_storage_account_url}', 'cont', 'test_simple_6.csv', '{azure_account_name}', '[HIDDEN]', 'CSV', 'none', 'auto')", "CREATE TABLE tablefunc44 (`x` int) AS hudi('http://minio1:9001/root/data/test7.csv', 'minio', '[HIDDEN]')", @@ -666,6 +721,29 @@ def make_test_case(i): "CREATE TABLE tablefunc61 (`x` int) AS paimon('http://minio1:9001/root/data/test11.csv.gz', 'minio', '[HIDDEN]')", "CREATE TABLE tablefunc62 (`x` int) AS paimonS3('http://minio1:9001/root/data/test11.csv.gz', 'minio', '[HIDDEN]')", f"CREATE TABLE tablefunc63 (`x` int) AS paimonAzure('{azure_storage_account_url}', 'cont', 'test_simple_6.csv', '{azure_account_name}', '[HIDDEN]', 'CSV', 'none', 'auto')", +======= + "CREATE TABLE tablefunc42 (`x` int) AS icebergS3(named_collection_2, url = 'http://minio1:9001/root/data/test4.csv', access_key_id = 'minio', secret_access_key = '[HIDDEN]')", + f"CREATE TABLE tablefunc43 (`x` int) AS icebergAzure('{masked_azure_conn_string}', 'cont', 'test_simple.csv')", + f"CREATE TABLE tablefunc44 (`x` int) AS icebergAzure('{azure_storage_account_url}', 'cont', 'test_simple.csv', '{azure_account_name}', '[HIDDEN]')", + f"CREATE TABLE tablefunc45 (`x` int) AS icebergAzure('{azure_storage_account_url}', 'cont', 'test_simple_6.csv', '{azure_account_name}', '[HIDDEN]', 'CSV', 'none', 'auto')", + f"CREATE TABLE tablefunc46 (`x` int) AS icebergAzure(named_collection_2, connection_string = '{masked_azure_conn_string}', container = 'cont', blob_path = 'test_simple_7.csv', format = 'CSV')", + f"CREATE TABLE tablefunc47 (`x` int) AS icebergAzure(named_collection_2, storage_account_url = '{azure_storage_account_url}', container = 'cont', blob_path = 'test_simple_8.csv', account_name = '{azure_account_name}', account_key = '[HIDDEN]')", + "CREATE TABLE tablefunc48 (`x` int) AS iceberg(storage_type = 's3', 'http://minio1:9001/root/data/test11.csv.gz', 'minio', '[HIDDEN]')", + "CREATE TABLE tablefunc49 (`x` int) AS iceberg(storage_type = 's3', named_collection_2, url = 'http://minio1:9001/root/data/test4.csv', access_key_id = 'minio', secret_access_key = '[HIDDEN]')", + f"CREATE TABLE tablefunc50 (`x` int) AS iceberg(storage_type = 'azure', '{masked_azure_conn_string}', 'cont', 'test_simple.csv')", + f"CREATE TABLE tablefunc51 (`x` int) AS iceberg(storage_type = 'azure', '{azure_storage_account_url}', 'cont', 'test_simple.csv', '{azure_account_name}', '[HIDDEN]')", + f"CREATE TABLE tablefunc52 (`x` int) AS iceberg(storage_type = 'azure', '{azure_storage_account_url}', 'cont', 'test_simple_6.csv', '{azure_account_name}', '[HIDDEN]', 'CSV', 'none', 'auto')", + f"CREATE TABLE tablefunc53 (`x` int) AS iceberg(storage_type = 'azure', named_collection_2, connection_string = '{masked_azure_conn_string}', container = 'cont', blob_path = 'test_simple_7.csv', format = 'CSV')", + f"CREATE TABLE tablefunc54 (`x` int) AS iceberg(storage_type = 'azure', named_collection_2, storage_account_url = '{azure_storage_account_url}', container = 'cont', blob_path = 'test_simple_8.csv', account_name = '{azure_account_name}', account_key = '[HIDDEN]')", + "CREATE TABLE tablefunc55 (`x` int) AS gcs('http://minio1:9001/root/data/test11.csv.gz', 'minio', '[HIDDEN]')", + f"CREATE TABLE tablefunc56 (`x` int) AS deltaLakeAzure('{azure_storage_account_url}', 'cont', 'test_simple_6.csv', '{azure_account_name}', '[HIDDEN]', 'CSV', 'none', 'auto')", + "CREATE TABLE tablefunc57 (`x` int) AS hudi('http://minio1:9001/root/data/test7.csv', 'minio', '[HIDDEN]')", + "CREATE TABLE tablefunc58 (`x` int) AS arrowFlight('arrowflight1:5006', 'dataset', 'arrowflight_user', '[HIDDEN]')", + "CREATE TABLE tablefunc59 (`x` int) AS arrowFlight(named_collection_1, host = 'arrowflight1', port = 5006, dataset = 'dataset', username = 'arrowflight_user', password = '[HIDDEN]')", + "CREATE TABLE tablefunc60 (`x` int) AS arrowflight(named_collection_1, host = 'arrowflight1', port = 5006, dataset = 'dataset', username = 'arrowflight_user', password = '[HIDDEN]')", + "CREATE TABLE tablefunc61 (`x` int) AS url('https://username:[HIDDEN]@domain.com/path', 'CSV')", + "CREATE TABLE tablefunc62 (`x` int) AS redis('localhost', 'key', 'key Int64', 0, '[HIDDEN]')", +>>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) ], must_not_contain=[password], ) diff --git a/tests/integration/test_s3_cluster/configs/cluster.xml b/tests/integration/test_s3_cluster/configs/cluster.xml index 84e6afd12f71..9d98df479576 100644 --- a/tests/integration/test_s3_cluster/configs/cluster.xml +++ b/tests/integration/test_s3_cluster/configs/cluster.xml @@ -20,6 +20,20 @@ + + + + + s0_0_1 + 9000 + + + s0_1_0 + 9000 + + + + @@ -49,6 +63,94 @@ + + + + c2.s0_0_0 + 9000 + + + c2.s0_0_1 + 9000 + + + + + + + + s0_0_1 + 9000 + foo + bar + + + s0_1_0 + 9000 + foo + bar + + + + + + + + c2.s0_0_0 + 9000 + biz + bar + + + c2.s0_0_1 + 9000 + biz + bar + + + + + + baz + + + s0_0_1 + 9000 + foo + + + s0_1_0 + 9000 + foo + + + + + + + + s0_0_0 + 9000 + + + s0_0_1 + 9000 + + + s0_1_0 + 9000 + + + c2.s0_0_0 + 9000 + + + c2.s0_0_1 + 9000 + + + + cluster_simple diff --git a/tests/integration/test_s3_cluster/configs/hidden_clusters.xml b/tests/integration/test_s3_cluster/configs/hidden_clusters.xml new file mode 100644 index 000000000000..8816cca1c79b --- /dev/null +++ b/tests/integration/test_s3_cluster/configs/hidden_clusters.xml @@ -0,0 +1,20 @@ + + + + + + s0_0_1 + 9000 + foo + bar + + + s0_1_0 + 9000 + foo + bar + + + + + diff --git a/tests/integration/test_s3_cluster/configs/users.xml b/tests/integration/test_s3_cluster/configs/users.xml index 4b6ba057ecb1..e0d2160a8a22 100644 --- a/tests/integration/test_s3_cluster/configs/users.xml +++ b/tests/integration/test_s3_cluster/configs/users.xml @@ -5,5 +5,18 @@ default 1 + + bar + default + + + bar + osc + + + + hidden_cluster_with_username_and_password + + diff --git a/tests/integration/test_s3_cluster/test.py b/tests/integration/test_s3_cluster/test.py index 82c68c405598..7c021a5384a7 100644 --- a/tests/integration/test_s3_cluster/test.py +++ b/tests/integration/test_s3_cluster/test.py @@ -129,6 +129,22 @@ def started_cluster(): macros={"replica": "replica1", "shard": "shard2"}, with_zookeeper=True, ) + cluster.add_instance( + "c2.s0_0_0", + main_configs=["configs/cluster.xml", "configs/named_collections.xml", "configs/hidden_clusters.xml"], + user_configs=["configs/users.xml"], + macros={"replica": "replica1", "shard": "shard1"}, + with_zookeeper=True, + stay_alive=True, + ) + cluster.add_instance( + "c2.s0_0_1", + main_configs=["configs/cluster.xml", "configs/named_collections.xml", "configs/hidden_clusters.xml"], + user_configs=["configs/users.xml"], + macros={"replica": "replica2", "shard": "shard1"}, + with_zookeeper=True, + stay_alive=True, + ) logging.info("Starting cluster...") cluster.start() @@ -274,6 +290,21 @@ def test_wrong_cluster(started_cluster): assert "not found" in error + error = node.query_and_get_error( + f""" + SELECT count(*) from s3( + 'http://minio1:9001/root/data/{{clickhouse,database}}/*', + 'minio', '{minio_secret_key}', 'CSV', 'name String, value UInt32, polygon Array(Array(Tuple(Float64, Float64)))') + UNION ALL + SELECT count(*) from s3( + 'http://minio1:9001/root/data/{{clickhouse,database}}/*', + 'minio', '{minio_secret_key}', 'CSV', 'name String, value UInt32, polygon Array(Array(Tuple(Float64, Float64)))') + SETTINGS object_storage_cluster = 'non_existing_cluster' + """ + ) + + assert "not found" in error + def test_ambiguous_join(started_cluster): node = started_cluster.instances["s0_0_0"] @@ -292,6 +323,20 @@ def test_ambiguous_join(started_cluster): ) assert "AMBIGUOUS_COLUMN_NAME" not in result + result = node.query( + f""" + SELECT l.name, r.value from s3( + 'http://minio1:9001/root/data/{{clickhouse,database}}/*', 'minio', '{minio_secret_key}', 'CSV', + 'name String, value UInt32, polygon Array(Array(Tuple(Float64, Float64)))') as l + JOIN s3( + 'http://minio1:9001/root/data/{{clickhouse,database}}/*', 'minio', '{minio_secret_key}', 'CSV', + 'name String, value UInt32, polygon Array(Array(Tuple(Float64, Float64)))') as r + ON l.name = r.name + SETTINGS object_storage_cluster = 'cluster_simple' + """ + ) + assert "AMBIGUOUS_COLUMN_NAME" not in result + def test_skip_unavailable_shards(started_cluster): node = started_cluster.instances["s0_0_0"] @@ -307,6 +352,17 @@ def test_skip_unavailable_shards(started_cluster): assert result == "10\n" + result = node.query( + f""" + SELECT count(*) from s3( + 'http://minio1:9001/root/data/clickhouse/part1.csv', + 'minio', '{minio_secret_key}', 'CSV', 'name String, value UInt32, polygon Array(Array(Tuple(Float64, Float64)))') + SETTINGS skip_unavailable_shards = 1, object_storage_cluster = 'cluster_non_existent_port' + """ + ) + + assert result == "10\n" + def test_unset_skip_unavailable_shards(started_cluster): # Although skip_unavailable_shards is not set, cluster table functions should always skip unavailable shards. @@ -322,6 +378,17 @@ def test_unset_skip_unavailable_shards(started_cluster): assert result == "10\n" + result = node.query( + f""" + SELECT count(*) from s3( + 'http://minio1:9001/root/data/clickhouse/part1.csv', + 'minio', '{minio_secret_key}', 'CSV', 'name String, value UInt32, polygon Array(Array(Tuple(Float64, Float64)))') + SETTINGS object_storage_cluster = 'cluster_non_existent_port' + """ + ) + + assert result == "10\n" + def test_distributed_insert_select_with_replicated(started_cluster): first_replica_first_shard = started_cluster.instances["s0_0_0"] @@ -502,6 +569,18 @@ def test_cluster_format_detection(started_cluster): assert result == expected_result + result = node.query( + f"SELECT * FROM s3('http://minio1:9001/root/data/generated/*', 'minio', '{minio_secret_key}') order by c1, c2 SETTINGS object_storage_cluster = 'cluster_simple'" + ) + + assert result == expected_result + + result = node.query( + f"SELECT * FROM s3('http://minio1:9001/root/data/generated/*', 'minio', '{minio_secret_key}', auto, 'a String, b UInt64') order by a, b SETTINGS object_storage_cluster = 'cluster_simple'" + ) + + assert result == expected_result + def test_cluster_default_expression(started_cluster): node = started_cluster.instances["s0_0_0"] @@ -550,6 +629,377 @@ def test_cluster_default_expression(started_cluster): assert result == expected_result + result = node.query( + f"SELECT * FROM s3('http://minio1:9001/root/data/data{{1,2,3}}', 'minio', '{minio_secret_key}', 'JSONEachRow', 'id UInt32, date Date DEFAULT 18262') order by id SETTINGS object_storage_cluster = 'cluster_simple'" + ) + + assert result == expected_result + + result = node.query( + f"SELECT * FROM s3('http://minio1:9001/root/data/data{{1,2,3}}', 'minio', '{minio_secret_key}', 'auto', 'id UInt32, date Date DEFAULT 18262') order by id SETTINGS object_storage_cluster = 'cluster_simple'" + ) + + assert result == expected_result + + result = node.query( + f"SELECT * FROM s3('http://minio1:9001/root/data/data{{1,2,3}}', 'minio', '{minio_secret_key}', 'JSONEachRow', 'id UInt32, date Date DEFAULT 18262', 'auto') order by id SETTINGS object_storage_cluster = 'cluster_simple'" + ) + + assert result == expected_result + + result = node.query( + f"SELECT * FROM s3('http://minio1:9001/root/data/data{{1,2,3}}', 'minio', '{minio_secret_key}', 'auto', 'id UInt32, date Date DEFAULT 18262', 'auto') order by id SETTINGS object_storage_cluster = 'cluster_simple'" + ) + + assert result == expected_result + + result = node.query( + "SELECT * FROM s3(test_s3_with_default) order by id SETTINGS object_storage_cluster = 'cluster_simple'" + ) + + assert result == expected_result + + +def test_distributed_s3_table_engine(started_cluster): + node = started_cluster.instances["s0_0_0"] + + resp_def = node.query( + f""" + SELECT * from s3Cluster( + 'cluster_simple', + 'http://minio1:9001/root/data/{{clickhouse,database}}/*', 'minio', '{minio_secret_key}', 'CSV', + 'name String, value UInt32, polygon Array(Array(Tuple(Float64, Float64)))') ORDER BY (name, value, polygon) + """ + ) + + node.query("DROP TABLE IF EXISTS single_node"); + node.query( + f""" + CREATE TABLE single_node + (name String, value UInt32, polygon Array(Array(Tuple(Float64, Float64)))) + ENGINE=S3('http://minio1:9001/root/data/{{clickhouse,database}}/*', 'minio', '{minio_secret_key}', 'CSV') + """ + ) + query_id_engine_single_node = str(uuid.uuid4()) + resp_engine_single_node = node.query( + """ + SELECT * FROM single_node ORDER BY (name, value, polygon) + """, + query_id = query_id_engine_single_node + ) + assert resp_def == resp_engine_single_node + + node.query("DROP TABLE IF EXISTS distributed"); + node.query( + f""" + CREATE TABLE distributed + (name String, value UInt32, polygon Array(Array(Tuple(Float64, Float64)))) + ENGINE=S3('http://minio1:9001/root/data/{{clickhouse,database}}/*', 'minio', '{minio_secret_key}', 'CSV') + SETTINGS object_storage_cluster='cluster_simple' + """ + ) + query_id_engine_distributed = str(uuid.uuid4()) + resp_engine_distributed = node.query( + """ + SELECT * FROM distributed ORDER BY (name, value, polygon) + """, + query_id = query_id_engine_distributed + ) + assert resp_def == resp_engine_distributed + + node.query("SYSTEM FLUSH LOGS ON CLUSTER 'cluster_simple'") + + hosts_engine_single_node = node.query( + f""" + SELECT uniq(hostname) + FROM clusterAllReplicas('cluster_simple', system.query_log) + WHERE type='QueryFinish' AND initial_query_id='{query_id_engine_single_node}' + """ + ) + assert int(hosts_engine_single_node) == 1 + hosts_engine_distributed = node.query( + f""" + SELECT uniq(hostname) + FROM clusterAllReplicas('cluster_simple', system.query_log) + WHERE type='QueryFinish' AND initial_query_id='{query_id_engine_distributed}' + """ + ) + assert int(hosts_engine_distributed) == 3 + + +def test_cluster_hosts_limit(started_cluster): + node = started_cluster.instances["s0_0_0"] + + query_id_def = str(uuid.uuid4()) + resp_def = node.query( + f""" + SELECT * from s3Cluster( + 'cluster_simple', + 'http://minio1:9001/root/data/{{clickhouse,database}}/*', 'minio', '{minio_secret_key}', 'CSV', + 'name String, value UInt32, polygon Array(Array(Tuple(Float64, Float64)))') ORDER BY (name, value, polygon) + """, + query_id = query_id_def + ) + + # object_storage_max_nodes is greater than number of hosts in cluster + query_id_4_hosts = str(uuid.uuid4()) + resp_4_hosts = node.query( + f""" + SELECT * from s3Cluster( + 'cluster_simple', + 'http://minio1:9001/root/data/{{clickhouse,database}}/*', 'minio', '{minio_secret_key}', 'CSV', + 'name String, value UInt32, polygon Array(Array(Tuple(Float64, Float64)))') ORDER BY (name, value, polygon) + SETTINGS object_storage_max_nodes=4 + """, + query_id = query_id_4_hosts + ) + assert resp_def == resp_4_hosts + + # object_storage_max_nodes is equal number of hosts in cluster + query_id_3_hosts = str(uuid.uuid4()) + resp_3_hosts = node.query( + f""" + SELECT * from s3Cluster( + 'cluster_simple', + 'http://minio1:9001/root/data/{{clickhouse,database}}/*', 'minio', '{minio_secret_key}', 'CSV', + 'name String, value UInt32, polygon Array(Array(Tuple(Float64, Float64)))') ORDER BY (name, value, polygon) + SETTINGS object_storage_max_nodes=3 + """, + query_id = query_id_3_hosts + ) + assert resp_def == resp_3_hosts + + # object_storage_max_nodes is less than number of hosts in cluster + query_id_2_hosts = str(uuid.uuid4()) + resp_2_hosts = node.query( + f""" + SELECT * from s3Cluster( + 'cluster_simple', + 'http://minio1:9001/root/data/{{clickhouse,database}}/*', 'minio', '{minio_secret_key}', 'CSV', + 'name String, value UInt32, polygon Array(Array(Tuple(Float64, Float64)))') ORDER BY (name, value, polygon) + SETTINGS object_storage_max_nodes=2 + """, + query_id = query_id_2_hosts + ) + assert resp_def == resp_2_hosts + + node.query("SYSTEM FLUSH LOGS ON CLUSTER 'cluster_simple'") + + hosts_def = node.query( + f""" + SELECT uniq(hostname) + FROM clusterAllReplicas('cluster_simple', system.query_log) + WHERE type='QueryFinish' AND initial_query_id='{query_id_def}' AND query_id!='{query_id_def}' + """ + ) + assert int(hosts_def) == 3 + + hosts_4 = node.query( + f""" + SELECT uniq(hostname) + FROM clusterAllReplicas('cluster_simple', system.query_log) + WHERE type='QueryFinish' AND initial_query_id='{query_id_4_hosts}' AND query_id!='{query_id_4_hosts}' + """ + ) + assert int(hosts_4) == 3 + + hosts_3 = node.query( + f""" + SELECT uniq(hostname) + FROM clusterAllReplicas('cluster_simple', system.query_log) + WHERE type='QueryFinish' AND initial_query_id='{query_id_3_hosts}' AND query_id!='{query_id_3_hosts}' + """ + ) + assert int(hosts_3) == 3 + + hosts_2 = node.query( + f""" + SELECT uniq(hostname) + FROM clusterAllReplicas('cluster_simple', system.query_log) + WHERE type='QueryFinish' AND initial_query_id='{query_id_2_hosts}' AND query_id!='{query_id_2_hosts}' + """ + ) + assert int(hosts_2) == 2 + + +def test_object_storage_remote_initiator(started_cluster): + node = started_cluster.instances["s0_0_0"] + + # Simple cluster + query_id = uuid.uuid4().hex + result = node.query( + f""" + SELECT * from s3Cluster( + 'cluster_remote', + 'http://minio1:9001/root/data/{{clickhouse,database}}/*', 'minio', '{minio_secret_key}', 'CSV', + 'name String, value UInt32, polygon Array(Array(Tuple(Float64, Float64)))') ORDER BY (name, value, polygon) + SETTINGS object_storage_remote_initiator=1 + """, + query_id = query_id, + ) + + assert result is not None + + node.query("SYSTEM FLUSH LOGS ON CLUSTER 'cluster_all'") + queries = node.query( + f""" + SELECT count() + FROM clusterAllReplicas('cluster_all', system.query_log) + WHERE type='QueryFinish' AND initial_query_id='{query_id}' + FORMAT TSV + """ + ).splitlines() + + # initial node + describe table + remote initiator + 2 subqueries on replicas + assert queries == ["5"] + + # Cluster with dots in the host names + query_id = uuid.uuid4().hex + result = node.query( + f""" + SELECT * from s3Cluster( + 'cluster_with_dots', + 'http://minio1:9001/root/data/{{clickhouse,database}}/*', 'minio', '{minio_secret_key}', 'CSV', + 'name String, value UInt32, polygon Array(Array(Tuple(Float64, Float64)))') ORDER BY (name, value, polygon) + SETTINGS object_storage_remote_initiator=1 + """, + query_id = query_id, + ) + + assert result is not None + + node.query("SYSTEM FLUSH LOGS ON CLUSTER 'cluster_all'") + queries = node.query( + f""" + SELECT count() + FROM clusterAllReplicas('cluster_all', system.query_log) + WHERE type='QueryFinish' AND initial_query_id='{query_id}' + FORMAT TSV + """ + ).splitlines() + + # initial node + describe table + remote initiator + 2 subqueries on replicas + assert queries == ["5"] + + users = node.query( + f""" + SELECT DISTINCT hostname, user + FROM clusterAllReplicas('cluster_all', system.query_log) + WHERE type='QueryFinish' AND initial_query_id='{query_id}' + ORDER BY ALL + FORMAT TSV + """ + ).splitlines() + + assert users == ["c2.s0_0_0\tdefault", + "c2.s0_0_1\tdefault", + "s0_0_0\tdefault"] + + # Cluster with user and password + query_id = uuid.uuid4().hex + result = node.query( + f""" + SELECT * from s3Cluster( + 'cluster_with_username_and_password', + 'http://minio1:9001/root/data/{{clickhouse,database}}/*', 'minio', '{minio_secret_key}', 'CSV', + 'name String, value UInt32, polygon Array(Array(Tuple(Float64, Float64)))') ORDER BY (name, value, polygon) + SETTINGS object_storage_remote_initiator=1 + """, + query_id = query_id, + ) + + assert result is not None + + node.query("SYSTEM FLUSH LOGS ON CLUSTER 'cluster_all'") + queries = node.query( + f""" + SELECT count() + FROM clusterAllReplicas('cluster_all', system.query_log) + WHERE type='QueryFinish' AND initial_query_id='{query_id}' + FORMAT TSV + """ + ).splitlines() + + # initial node + describe table + remote initiator + 2 subqueries on replicas + assert queries == ["5"] + + users = node.query( + f""" + SELECT DISTINCT hostname, user + FROM clusterAllReplicas('cluster_all', system.query_log) + WHERE type='QueryFinish' AND initial_query_id='{query_id}' + ORDER BY ALL + FORMAT TSV + """ + ).splitlines() + + assert users == ["s0_0_0\tdefault", + "s0_0_1\tfoo", + "s0_1_0\tfoo"] + + # Cluster with secret + query_id = uuid.uuid4().hex + result = node.query_and_get_error( + f""" + SELECT * from s3Cluster( + 'cluster_with_secret', + 'http://minio1:9001/root/data/{{clickhouse,database}}/*', 'minio', '{minio_secret_key}', 'CSV', + 'name String, value UInt32, polygon Array(Array(Tuple(Float64, Float64)))') ORDER BY (name, value, polygon) + SETTINGS object_storage_remote_initiator=1 + """, + query_id = query_id, + ) + + assert "Can't convert query to remote when cluster uses secret" in result + + # Different cluster for remote initiator and query execution + # with `hidden_cluster_with_username_and_password` existed only in `cluster_with_dots` nodes + query_id = uuid.uuid4().hex + + result = node.query( + f""" + SELECT * from s3( + 'http://minio1:9001/root/data/{{clickhouse,database}}/*', 'minio', '{minio_secret_key}', 'CSV', + 'name String, value UInt32, polygon Array(Array(Tuple(Float64, Float64)))') ORDER BY (name, value, polygon) + SETTINGS + object_storage_remote_initiator=1, + object_storage_cluster='hidden_cluster_with_username_and_password', + object_storage_remote_initiator_cluster='cluster_with_dots' + """, + query_id = query_id, + ) + + assert result is not None + + node.query("SYSTEM FLUSH LOGS ON CLUSTER 'cluster_all'") + queries = node.query( + f""" + SELECT count() + FROM clusterAllReplicas('cluster_all', system.query_log) + WHERE type='QueryFinish' AND initial_query_id='{query_id}' + FORMAT TSV + """ + ).splitlines() + + # initial node + describe table + remote initiator + 2 subqueries on replicas + assert queries == ["5"] + + users = node.query( + f""" + SELECT DISTINCT hostname, user + FROM clusterAllReplicas('cluster_all', system.query_log) + WHERE type='QueryFinish' AND initial_query_id='{query_id}' + ORDER BY ALL + FORMAT TSV + """ + ).splitlines() + + # Random host from 'cluster_with_dots' for remote query + assert users[0] in ["c2.s0_0_0\tdefault", "c2.s0_0_1\tdefault"] + assert users[1:] == ["s0_0_0\tdefault", + "s0_0_1\tfoo", + "s0_1_0\tfoo"] + def test_remote_hedged(started_cluster): node = started_cluster.instances["s0_0_0"] @@ -902,6 +1352,7 @@ def test_hive_partitioning(started_cluster, allow_experimental_analyzer, use_par cluster_full_traffic = int(cluster_full_traffic) assert cluster_full_traffic == full_traffic +<<<<<<< HEAD cluster_optimized_traffic = node.query( f""" SELECT sum(ProfileEvents['{event}']) @@ -983,3 +1434,102 @@ def test_iceberg_s3_cluster_read_task_failpoint(started_cluster): ) node.query(f"DROP TABLE IF EXISTS {dst_table}") node.query(f"DROP TABLE IF EXISTS {iceberg_table}") +======= + assert errors == 0 + + +def test_object_storage_remote_initiator_without_cluster_function(started_cluster): + node = started_cluster.instances["s0_0_0"] + + # Remove initiator without cluster request + # Query executed on random node of object_storage_remote_initiator_cluster + query_id = uuid.uuid4().hex + + result = node.query( + f""" + SELECT * from s3( + 'http://minio1:9001/root/data/{{clickhouse,database}}/*', 'minio', '{minio_secret_key}', 'CSV', + 'name String, value UInt32, polygon Array(Array(Tuple(Float64, Float64)))') ORDER BY (name, value, polygon) + SETTINGS + object_storage_remote_initiator=1, + object_storage_remote_initiator_cluster='cluster_with_dots' + """, + query_id = query_id, + ) + + assert result is not None + + node.query("SYSTEM FLUSH LOGS ON CLUSTER 'cluster_all'") + queries = node.query( + f""" + SELECT count() + FROM clusterAllReplicas('cluster_all', system.query_log) + WHERE type='QueryFinish' AND initial_query_id='{query_id}' + FORMAT TSV + """ + ).splitlines() + + # initial node + describe table + remote initiator + assert queries == ["3"] + + users = node.query( + f""" + SELECT DISTINCT hostname, user + FROM clusterAllReplicas('cluster_all', system.query_log) + WHERE type='QueryFinish' AND initial_query_id='{query_id}' + ORDER BY ALL + FORMAT TSV + """ + ).splitlines() + + # Random host from 'cluster_with_dots' for remote query + assert users[0] in ["c2.s0_0_0\tdefault", "c2.s0_0_1\tdefault"] + assert users[1:] == ["s0_0_0\tdefault"] + + # Remove initiator without cluster request + # but with `object_storage_cluster` specified for user on remote cluster + query_id = uuid.uuid4().hex + + result = node.query( + f""" + SELECT * from s3( + 'http://minio1:9001/root/data/{{clickhouse,database}}/*', 'minio', '{minio_secret_key}', 'CSV', + 'name String, value UInt32, polygon Array(Array(Tuple(Float64, Float64)))') ORDER BY (name, value, polygon) + SETTINGS + object_storage_remote_initiator=1, + object_storage_remote_initiator_cluster='cluster_with_dots_and_user' + """, + query_id = query_id, + ) + + assert result is not None + + node.query("SYSTEM FLUSH LOGS ON CLUSTER 'cluster_all'") + queries = node.query( + f""" + SELECT count() + FROM clusterAllReplicas('cluster_all', system.query_log) + WHERE type='QueryFinish' AND initial_query_id='{query_id}' + FORMAT TSV + """ + ).splitlines() + + # initial node + describe table + remote initiator + 2 subqueries on replicas + assert queries == ["5"] + + users = node.query( + f""" + SELECT DISTINCT hostname, user + FROM clusterAllReplicas('cluster_all', system.query_log) + WHERE type='QueryFinish' AND initial_query_id='{query_id}' + ORDER BY ALL + FORMAT TSV + """ + ).splitlines() + + # Random host from 'cluster_with_dots' for remote query + assert users[0] in ["c2.s0_0_0\tbiz", "c2.s0_0_1\tbiz"] + assert users[1:] == ["s0_0_0\tdefault", + "s0_0_1\tfoo", + "s0_1_0\tfoo"] +>>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) diff --git a/tests/integration/test_storage_iceberg_no_spark/configs/config.d/named_collections.xml b/tests/integration/test_storage_iceberg_no_spark/configs/config.d/named_collections.xml index 516e4ba63a3a..7dfec41b2df8 100644 --- a/tests/integration/test_storage_iceberg_no_spark/configs/config.d/named_collections.xml +++ b/tests/integration/test_storage_iceberg_no_spark/configs/config.d/named_collections.xml @@ -11,5 +11,19 @@ + + http://minio1:9001/root/ + minio + ClickHouse_Minio_P@ssw0rd + s3 + + + devstoreaccount1 + Eby8vdM02xNOcqFlqUwJPLlmEtlCDXJ1OUzFT50uSRZ6IFsuFq2UVErCz4I6tq/K1SZFPTOtr/KBHBeksoGMGw== + azure + + + local + diff --git a/tests/integration/test_storage_iceberg_with_spark/configs/config.d/named_collections.xml b/tests/integration/test_storage_iceberg_with_spark/configs/config.d/named_collections.xml index 516e4ba63a3a..7dfec41b2df8 100644 --- a/tests/integration/test_storage_iceberg_with_spark/configs/config.d/named_collections.xml +++ b/tests/integration/test_storage_iceberg_with_spark/configs/config.d/named_collections.xml @@ -11,5 +11,19 @@ + + http://minio1:9001/root/ + minio + ClickHouse_Minio_P@ssw0rd + s3 + + + devstoreaccount1 + Eby8vdM02xNOcqFlqUwJPLlmEtlCDXJ1OUzFT50uSRZ6IFsuFq2UVErCz4I6tq/K1SZFPTOtr/KBHBeksoGMGw== + azure + + + local + diff --git a/tests/integration/test_storage_iceberg_with_spark/test_cluster_joins.py b/tests/integration/test_storage_iceberg_with_spark/test_cluster_joins.py index bb637f8e8cc2..c04940c1eeb8 100644 --- a/tests/integration/test_storage_iceberg_with_spark/test_cluster_joins.py +++ b/tests/integration/test_storage_iceberg_with_spark/test_cluster_joins.py @@ -6,9 +6,8 @@ execute_spark_query_general, ) -# TODO - turn on after merge alternative syntax @pytest.mark.parametrize("storage_type", ["s3", "azure"]) -def _test_cluster_joins(started_cluster_iceberg_with_spark, storage_type): +def test_cluster_joins(started_cluster_iceberg_with_spark, storage_type): instance = started_cluster_iceberg_with_spark.instances["node1"] spark = started_cluster_iceberg_with_spark.spark_session TABLE_NAME = "test_cluster_joins_" + storage_type + "_" + get_uuid_str() diff --git a/tests/integration/test_storage_iceberg_with_spark/test_cluster_table_function.py b/tests/integration/test_storage_iceberg_with_spark/test_cluster_table_function.py index a145b087d4b2..078e34004ec9 100644 --- a/tests/integration/test_storage_iceberg_with_spark/test_cluster_table_function.py +++ b/tests/integration/test_storage_iceberg_with_spark/test_cluster_table_function.py @@ -16,9 +16,30 @@ from helpers.config_cluster import minio_secret_key +def count_secondary_subqueries(started_cluster, query_id, expected, comment): + for node_name, replica in started_cluster.instances.items(): + cluster_secondary_queries = ( + replica.query( + f""" + SELECT count(*) FROM system.query_log + WHERE + type = 'QueryFinish' + AND NOT is_initial_query + AND initial_query_id='{query_id}' + """ + ) + .strip() + ) + + logging.info( + f"[{node_name}] cluster_secondary_queries {comment}: {cluster_secondary_queries}" + ) + assert int(cluster_secondary_queries) == expected + @pytest.mark.parametrize("format_version", ["1", "2"]) @pytest.mark.parametrize("storage_type", ["s3", "azure", "local"]) -def test_cluster_table_function(started_cluster_iceberg_with_spark, format_version, storage_type): +@pytest.mark.parametrize("cluster_name_as_literal", [True, False]) +def test_cluster_table_function(started_cluster_iceberg_with_spark, format_version, storage_type, cluster_name_as_literal): instance = started_cluster_iceberg_with_spark.instances["node1"] spark = started_cluster_iceberg_with_spark.spark_session @@ -76,59 +97,159 @@ def add_df(mode): # Regular Query only node1 table_function_expr = get_creation_expression( - storage_type, TABLE_NAME, started_cluster_iceberg_with_spark, table_function=True + storage_type, TABLE_NAME, started_cluster_iceberg_with_spark, table_function=True, cluster_name_as_literal=cluster_name_as_literal ) select_regular = ( instance.query(f"SELECT * FROM {table_function_expr}").strip().split() ) + def make_query_from_function( + run_on_cluster=False, + alt_syntax=False, + remote=False, + storage_type_as_arg=False, + storage_type_in_named_collection=False, + ): + expr = get_creation_expression( + storage_type, + TABLE_NAME, + started_cluster_iceberg_with_spark, + table_function=True, + run_on_cluster=run_on_cluster, + storage_type_as_arg=storage_type_as_arg, + storage_type_in_named_collection=storage_type_in_named_collection, + cluster_name_as_literal=cluster_name_as_literal, + ) + query_id = str(uuid.uuid4()) + settings = f"SETTINGS object_storage_cluster='cluster_simple'" if (alt_syntax and not run_on_cluster) else "" + if remote: + query = f"SELECT * FROM remote('node2', {expr}) {settings}" + else: + query = f"SELECT * FROM {expr} {settings}" + responce = instance.query(query, query_id=query_id).strip().split() + return responce, query_id + # Cluster Query with node1 as coordinator - table_function_expr_cluster = get_creation_expression( - storage_type, - TABLE_NAME, - started_cluster_iceberg_with_spark, - table_function=True, + select_cluster, query_id_cluster = make_query_from_function(run_on_cluster=True) + + # Cluster Query with node1 as coordinator with alternative syntax + select_cluster_alt_syntax, query_id_cluster_alt_syntax = make_query_from_function( + run_on_cluster=True, + alt_syntax=True) + + # Cluster Query with node1 as coordinator and storage type as arg + select_cluster_with_type_arg, query_id_cluster_with_type_arg = make_query_from_function( run_on_cluster=True, + storage_type_as_arg=True, ) - select_cluster = ( - instance.query(f"SELECT * FROM {table_function_expr_cluster}").strip().split() + + # Cluster Query with node1 as coordinator and storage type in named collection + select_cluster_with_type_in_nc, query_id_cluster_with_type_in_nc = make_query_from_function( + run_on_cluster=True, + storage_type_in_named_collection=True, + ) + + # Cluster Query with node1 as coordinator and storage type as arg, alternative syntax + select_cluster_with_type_arg_alt_syntax, query_id_cluster_with_type_arg_alt_syntax = make_query_from_function( + storage_type_as_arg=True, + alt_syntax=True, ) + # Cluster Query with node1 as coordinator and storage type in named collection, alternative syntax + select_cluster_with_type_in_nc_alt_syntax, query_id_cluster_with_type_in_nc_alt_syntax = make_query_from_function( + storage_type_in_named_collection=True, + alt_syntax=True, + ) + + #select_remote_cluster, _ = make_query_from_function(run_on_cluster=True, remote=True) + + def make_query_from_table(alt_syntax=False): + query_id = str(uuid.uuid4()) + settings = "SETTINGS object_storage_cluster='cluster_simple'" if alt_syntax else "" + responce = ( + instance.query( + f"SELECT * FROM {TABLE_NAME} {settings}", + query_id=query_id, + ) + .strip() + .split() + ) + return responce, query_id + + create_iceberg_table(storage_type, instance, TABLE_NAME, started_cluster_iceberg_with_spark, object_storage_cluster='cluster_simple') + select_cluster_table_engine, query_id_cluster_table_engine = make_query_from_table() + + #select_remote_cluster = ( + # instance.query(f"SELECT * FROM remote('node2',{table_function_expr_cluster})") + # .strip() + # .split() + #) + + instance.query(f"DROP TABLE IF EXISTS `{TABLE_NAME}` SYNC") + + create_iceberg_table(storage_type, instance, TABLE_NAME, started_cluster_iceberg_with_spark) + select_pure_table_engine, query_id_pure_table_engine = make_query_from_table() + select_pure_table_engine_cluster, query_id_pure_table_engine_cluster = make_query_from_table(alt_syntax=True) + + create_iceberg_table(storage_type, instance, TABLE_NAME, started_cluster_iceberg_with_spark, storage_type_as_arg=True) + select_pure_table_engine_with_type_arg, query_id_pure_table_engine_with_type_arg = make_query_from_table() + select_pure_table_engine_cluster_with_type_arg, query_id_pure_table_engine_cluster_with_type_arg = make_query_from_table(alt_syntax=True) + + create_iceberg_table(storage_type, instance, TABLE_NAME, started_cluster_iceberg_with_spark, storage_type_in_named_collection=True) + select_pure_table_engine_with_type_in_nc, query_id_pure_table_engine_with_type_in_nc = make_query_from_table() + select_pure_table_engine_cluster_with_type_in_nc, query_id_pure_table_engine_cluster_with_type_in_nc = make_query_from_table(alt_syntax=True) + # Simple size check assert len(select_regular) == 600 assert len(select_cluster) == 600 + assert len(select_cluster_alt_syntax) == 600 + assert len(select_cluster_table_engine) == 600 + #assert len(select_remote_cluster) == 600 + assert len(select_cluster_with_type_arg) == 600 + assert len(select_cluster_with_type_in_nc) == 600 + assert len(select_cluster_with_type_arg_alt_syntax) == 600 + assert len(select_cluster_with_type_in_nc_alt_syntax) == 600 + assert len(select_pure_table_engine) == 600 + assert len(select_pure_table_engine_cluster) == 600 + assert len(select_pure_table_engine_with_type_arg) == 600 + assert len(select_pure_table_engine_cluster_with_type_arg) == 600 + assert len(select_pure_table_engine_with_type_in_nc) == 600 + assert len(select_pure_table_engine_cluster_with_type_in_nc) == 600 # Actual check assert select_cluster == select_regular + assert select_cluster_alt_syntax == select_regular + assert select_cluster_table_engine == select_regular + #assert select_remote_cluster == select_regular + assert select_cluster_with_type_arg == select_regular + assert select_cluster_with_type_in_nc == select_regular + assert select_cluster_with_type_arg_alt_syntax == select_regular + assert select_cluster_with_type_in_nc_alt_syntax == select_regular + assert select_pure_table_engine == select_regular + assert select_pure_table_engine_cluster == select_regular + assert select_pure_table_engine_with_type_arg == select_regular + assert select_pure_table_engine_cluster_with_type_arg == select_regular + assert select_pure_table_engine_with_type_in_nc == select_regular + assert select_pure_table_engine_cluster_with_type_in_nc == select_regular # Check query_log for replica in started_cluster_iceberg_with_spark.instances.values(): replica.query("SYSTEM FLUSH LOGS") - for node_name, replica in started_cluster_iceberg_with_spark.instances.items(): - cluster_secondary_queries = ( - replica.query( - f""" - SELECT query, type, is_initial_query, read_rows, read_bytes FROM system.query_log - WHERE - type = 'QueryStart' AND - positionCaseInsensitive(query, '{storage_type}Cluster') != 0 AND - position(query, '{TABLE_NAME}') != 0 AND - position(query, 'system.query_log') = 0 AND - NOT is_initial_query - """ - ) - .strip() - .split("\n") - ) - - logging.info( - f"[{node_name}] cluster_secondary_queries: {cluster_secondary_queries}" - ) - assert len(cluster_secondary_queries) == 1 + count_secondary_subqueries(started_cluster_iceberg_with_spark, query_id_cluster, 1, "table function") + count_secondary_subqueries(started_cluster_iceberg_with_spark, query_id_cluster_alt_syntax, 1, "table function alt syntax") + count_secondary_subqueries(started_cluster_iceberg_with_spark, query_id_cluster_table_engine, 1, "cluster table engine") + count_secondary_subqueries(started_cluster_iceberg_with_spark, query_id_cluster_with_type_arg, 1, "table function with storage type in args") + count_secondary_subqueries(started_cluster_iceberg_with_spark, query_id_cluster_with_type_in_nc, 1, "table function with storage type in named collection") + count_secondary_subqueries(started_cluster_iceberg_with_spark, query_id_cluster_with_type_arg_alt_syntax, 1, "table function with storage type in args alt syntax") + count_secondary_subqueries(started_cluster_iceberg_with_spark, query_id_cluster_with_type_in_nc_alt_syntax, 1, "table function with storage type in named collection alt syntax") + count_secondary_subqueries(started_cluster_iceberg_with_spark, query_id_pure_table_engine, 0, "table engine") + count_secondary_subqueries(started_cluster_iceberg_with_spark, query_id_pure_table_engine_cluster, 1, "table engine with cluster setting") + count_secondary_subqueries(started_cluster_iceberg_with_spark, query_id_pure_table_engine_with_type_arg, 0, "table engine with storage type in args") + count_secondary_subqueries(started_cluster_iceberg_with_spark, query_id_pure_table_engine_cluster_with_type_arg, 1, "table engine with cluster setting with storage type in args") + count_secondary_subqueries(started_cluster_iceberg_with_spark, query_id_pure_table_engine_with_type_in_nc, 0, "table engine with storage type in named collection") + count_secondary_subqueries(started_cluster_iceberg_with_spark, query_id_pure_table_engine_cluster_with_type_in_nc, 1, "table engine with cluster setting with storage type in named collection") - # write 3 times - assert int(instance.query(f"SELECT count() FROM {table_function_expr_cluster}")) == 100 * 3 # Cluster Query with node1 as coordinator diff --git a/tests/integration/test_storage_iceberg_with_spark/test_minmax_pruning_with_null.py b/tests/integration/test_storage_iceberg_with_spark/test_minmax_pruning_with_null.py index ceb630acbd73..93ba2f765914 100644 --- a/tests/integration/test_storage_iceberg_with_spark/test_minmax_pruning_with_null.py +++ b/tests/integration/test_storage_iceberg_with_spark/test_minmax_pruning_with_null.py @@ -9,7 +9,10 @@ ) @pytest.mark.parametrize("storage_type", ["s3", "azure", "local"]) -def test_minmax_pruning_with_null(started_cluster_iceberg_with_spark, storage_type): +@pytest.mark.parametrize("run_on_cluster", [False, True]) +def test_minmax_pruning_with_null(started_cluster_iceberg_with_spark, storage_type, run_on_cluster): + if run_on_cluster and storage_type == "local": + pytest.skip("Local storage is not supported on cluster") instance = started_cluster_iceberg_with_spark.instances["node1"] spark = started_cluster_iceberg_with_spark.spark_session TABLE_NAME = "test_minmax_pruning_with_null" + storage_type + "_" + get_uuid_str() @@ -21,6 +24,7 @@ def execute_spark_query(query: str): storage_type, TABLE_NAME, query, + additional_nodes=["node2", "node3"] if storage_type=="local" else [], ) execute_spark_query( @@ -79,7 +83,7 @@ def execute_spark_query(query: str): ) creation_expression = get_creation_expression( - storage_type, TABLE_NAME, started_cluster_iceberg_with_spark, table_function=True + storage_type, TABLE_NAME, started_cluster_iceberg_with_spark, table_function=True, run_on_cluster=run_on_cluster ) def check_validity_and_get_prunned_files(select_expression): diff --git a/tests/integration/test_storage_iceberg_with_spark/test_partition_pruning.py b/tests/integration/test_storage_iceberg_with_spark/test_partition_pruning.py index 6ade42e72537..4c6a6b4c7bd7 100644 --- a/tests/integration/test_storage_iceberg_with_spark/test_partition_pruning.py +++ b/tests/integration/test_storage_iceberg_with_spark/test_partition_pruning.py @@ -9,7 +9,7 @@ @pytest.mark.parametrize( "storage_type, run_on_cluster", - [("s3", False), ("s3", True), ("azure", False), ("local", False), ("local", True)], + [("s3", False), ("s3", True), ("azure", False), ("azure", True), ("local", False), ("local", True)], ) def test_partition_pruning(started_cluster_iceberg_with_spark, storage_type, run_on_cluster): instance = started_cluster_iceberg_with_spark.instances["node1"] diff --git a/tests/integration/test_storage_iceberg_with_spark/test_types.py b/tests/integration/test_storage_iceberg_with_spark/test_types.py index 7f63df522db1..1dd605098279 100644 --- a/tests/integration/test_storage_iceberg_with_spark/test_types.py +++ b/tests/integration/test_storage_iceberg_with_spark/test_types.py @@ -86,3 +86,49 @@ def test_types(started_cluster_iceberg_with_spark, format_version, storage_type) ["e", "Nullable(Bool)"], ] ) + + # Test storage type as function argument + table_function_expr = get_creation_expression( + storage_type, + TABLE_NAME, + started_cluster_iceberg_with_spark, + table_function=True, + storage_type_as_arg=True, + ) + assert ( + instance.query(f"SELECT a, b, c, d, e FROM {table_function_expr}").strip() + == "123\tstring\t2000-01-01\t['str1','str2']\ttrue" + ) + + assert instance.query(f"DESCRIBE {table_function_expr} FORMAT TSV") == TSV( + [ + ["a", "Nullable(Int32)"], + ["b", "Nullable(String)"], + ["c", "Nullable(Date32)"], + ["d", "Array(Nullable(String))"], + ["e", "Nullable(Bool)"], + ] + ) + + # Test storage type as field in named collection + table_function_expr = get_creation_expression( + storage_type, + TABLE_NAME, + started_cluster_iceberg_with_spark, + table_function=True, + storage_type_in_named_collection=True, + ) + assert ( + instance.query(f"SELECT a, b, c, d, e FROM {table_function_expr}").strip() + == "123\tstring\t2000-01-01\t['str1','str2']\ttrue" + ) + + assert instance.query(f"DESCRIBE {table_function_expr} FORMAT TSV") == TSV( + [ + ["a", "Nullable(Int32)"], + ["b", "Nullable(String)"], + ["c", "Nullable(Date32)"], + ["d", "Array(Nullable(String))"], + ["e", "Nullable(Bool)"], + ] + ) diff --git a/tests/queries/0_stateless/01625_constraints_index_append.reference b/tests/queries/0_stateless/01625_constraints_index_append.reference index b68b514ca8bd..bf6f37328286 100644 --- a/tests/queries/0_stateless/01625_constraints_index_append.reference +++ b/tests/queries/0_stateless/01625_constraints_index_append.reference @@ -13,14 +13,14 @@ Prewhere info Prewhere filter Prewhere filter column: less(multiply(2, b), 100) - Filter column: and(equals(a, 0), indexHint(greater(plus(i, 40), 0))) (removed) + Filter column: and(indexHint(greater(plus(i, 40), 0)), equals(a, 0)) (removed) Prewhere info Prewhere filter Prewhere filter column: equals(a, 0) Prewhere info Prewhere filter Prewhere filter column: less(a, 0) (removed) - Filter column: and(greaterOrEquals(a, 0), indexHint(greater(plus(i, 40), 0))) (removed) + Filter column: and(indexHint(greater(plus(i, 40), 0)), greaterOrEquals(a, 0)) (removed) Prewhere info Prewhere filter Prewhere filter column: greaterOrEquals(a, 0) From 8d29b7ef133604d85d08757e919acc48a717b33a Mon Sep 17 00:00:00 2001 From: Andrey Zvonov <32552679+zvonand@users.noreply.github.com> Date: Wed, 5 Aug 2026 17:41:16 +0200 Subject: [PATCH 06/43] Resolve conflicts in cherry-pick of #1640 Removed the conflict markers left by the cherry-pick and kept the source PR's changes, translated to the shapes antalya-26.6 already has (private configuration members behind getters, renamed assertInitializedDL, ContextPtr threaded through IcebergSchemaProcessor, IcebergPathResolver, table UUIDs in DataLake StorageIDs). Source-PR: #1640 (https://github.com/Altinity/ClickHouse/pull/1640) --- .../sql-reference/table-functions/iceberg.md | 4 +- src/Analyzer/FunctionNode.h | 4 - src/Common/ErrorCodes.cpp | 5 +- src/Core/Settings.cpp | 10 +- src/Core/SettingsChangesHistory.cpp | 10 +- src/Databases/DataLake/DatabaseDataLake.cpp | 64 +------------ .../DataLake/DatabaseDataLakeSettings.cpp | 4 - src/Databases/DataLake/GlueCatalog.cpp | 5 +- src/Databases/DataLake/ICatalog.cpp | 19 ---- src/Databases/DataLake/RestCatalog.cpp | 17 +--- src/Databases/DataLake/UnityCatalog.cpp | 3 - .../ObjectStorages/S3/S3ObjectStorage.cpp | 7 +- src/IO/S3/URI.cpp | 5 +- src/IO/S3/URI.h | 10 +- src/IO/S3/getObjectInfo.cpp | 4 - src/Interpreters/IcebergMetadataLog.cpp | 5 - src/Parsers/FunctionSecretArgumentsFinder.h | 24 +---- .../QueryPlan/ReadFromObjectStorageStep.cpp | 2 +- src/Server/TCPHandler.cpp | 5 - src/Storages/IStorageCluster.h | 4 +- .../DataLakes/DataLakeConfiguration.h | 76 +-------------- .../DataLakes/Iceberg/IcebergMetadata.cpp | 18 +--- .../DataLakes/Iceberg/IcebergMetadata.h | 3 - .../DataLakes/Iceberg/IcebergWrites.cpp | 6 -- .../Iceberg/PersistentTableComponents.h | 5 +- .../DataLakes/Iceberg/SchemaProcessor.cpp | 25 ++--- .../DataLakes/Iceberg/SchemaProcessor.h | 15 +-- .../ObjectStorage/DataLakes/Iceberg/Utils.cpp | 21 +--- .../tests/gtest_iceberg_schema_processor.cpp | 37 ++++---- .../ObjectStorage/S3/Configuration.cpp | 3 - .../ObjectStorage/StorageObjectStorage.cpp | 14 +-- .../StorageObjectStorageCluster.cpp | 34 +------ .../StorageObjectStorageConfiguration.h | 3 - .../StorageObjectStorageSource.cpp | 31 +----- .../registerStorageObjectStorage.cpp | 3 - src/Storages/System/StorageSystemTables.cpp | 95 ------------------- .../TableFunctionObjectStorage.cpp | 57 ----------- .../docker_compose_iceberg_rest_catalog.yml | 25 ----- tests/integration/helpers/iceberg_utils.py | 44 +-------- tests/integration/test_database_delta/test.py | 5 +- tests/integration/test_database_glue/test.py | 6 +- .../integration/test_database_iceberg/test.py | 38 +------- .../test_mask_sensitive_info/test.py | 55 ++++------- tests/integration/test_s3_cluster/test.py | 4 - 44 files changed, 93 insertions(+), 741 deletions(-) diff --git a/docs/en/sql-reference/table-functions/iceberg.md b/docs/en/sql-reference/table-functions/iceberg.md index ba59c52ace13..64b9d9c62ba6 100644 --- a/docs/en/sql-reference/table-functions/iceberg.md +++ b/docs/en/sql-reference/table-functions/iceberg.md @@ -633,7 +633,6 @@ GRANT ALTER TABLE ON my_iceberg_table TO my_user; - The catalog's own authorization (REST catalog auth, AWS Glue IAM, etc.) is enforced independently when ClickHouse updates the metadata ::: -<<<<<<< HEAD ### Remove Orphan Files {#iceberg-remove-orphan-files} Orphan files are files on storage that are not referenced by any snapshot in the Iceberg table metadata. They accumulate from failed writes, partial cleanup after compaction, and interrupted operations, causing unbounded storage growth. The `remove_orphan_files` command identifies and removes these orphan files. @@ -715,7 +714,7 @@ The command returns a table with `metric_name` and `metric_value` columns showin - Use `dry_run = 1` to preview orphan files before deletion - The `older_than` threshold protects against deleting files from in-progress writes — the default 3-day threshold provides a generous safety margin ::: -======= + ## Altinity Antalya branch ### Specify storage type in arguments @@ -756,7 +755,6 @@ iceberg(named_collection[, option=value [,..]]) ``` The default value for `storage_type` is `s3`. ->>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) ## See Also {#see-also} diff --git a/src/Analyzer/FunctionNode.h b/src/Analyzer/FunctionNode.h index d6ae27404e97..f72f285480cc 100644 --- a/src/Analyzer/FunctionNode.h +++ b/src/Analyzer/FunctionNode.h @@ -9,11 +9,7 @@ #include #include #include -<<<<<<< HEAD -======= -#include #include ->>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) namespace DB { diff --git a/src/Common/ErrorCodes.cpp b/src/Common/ErrorCodes.cpp index 216571ab8f26..4664dff73f1c 100644 --- a/src/Common/ErrorCodes.cpp +++ b/src/Common/ErrorCodes.cpp @@ -650,7 +650,6 @@ M(768, CANNOT_EXECUTE_PROMQL_QUERY) \ M(769, NAMED_COLLECTION_IS_USED) \ M(770, WASM_ERROR) \ -<<<<<<< HEAD M(771, CACHE_CANNOT_WRITE_TO_CACHE_DISK) \ M(772, INCOMPATIBLE_SCHEMA) \ M(773, MALFORMED_AI_PROVIDER_RESPONSE) \ @@ -659,9 +658,7 @@ M(776, RESOURCE_LIMIT_EXCEEDED) \ M(777, MEMORY_RESERVATION_KILLED) \ M(778, MEMORY_RESERVATION_FAILED) \ -======= - M(771, CATALOG_NAMESPACE_DISABLED) \ ->>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) + M(779, CATALOG_NAMESPACE_DISABLED) \ \ M(900, DISTRIBUTED_CACHE_ERROR) \ M(901, CANNOT_USE_DISTRIBUTED_CACHE) \ diff --git a/src/Core/Settings.cpp b/src/Core/Settings.cpp index 3266e1881d90..67241001c1b0 100644 --- a/src/Core/Settings.cpp +++ b/src/Core/Settings.cpp @@ -7986,7 +7986,6 @@ Use `allow_nullable_tuple_in_extracted_subcolumns` to control whether extracted )", BETA, enable_nullable_tuple_type) \ DECLARE(UInt64, archive_adaptive_buffer_max_size_bytes, 8 * DBMS_DEFAULT_BUFFER_SIZE, R"( Limits the maximum size of the adaptive buffer used when writing to archive files (for example, tar archives)", 0) \ -<<<<<<< HEAD DECLARE(UInt64, shared_merge_tree_sequential_consistency_initial_parts_update_backoff_ms, 50, R"( Initial backoff in milliseconds for parts update when using `select_sequential_consistency` with `SharedMergeTree`. Only available in ClickHouse Cloud. )", 0) \ @@ -8015,7 +8014,7 @@ Enable converting the hash table to a flat array for joins when the key is a sin )", 0) \ DECLARE(UInt64, query_plan_min_columns_for_join_lazy_indexing, 3, R"( Control the minimum number of payload columns from the left side required for enabling lazy indexing optimization in JOIN. 0 means the optimization is disabled. -======= +)", 0) \ DECLARE(Timezone, iceberg_timezone_for_timestamptz, "UTC", R"( Timezone for Iceberg timestamptz field. @@ -8034,7 +8033,6 @@ Possible values: - `` (empty value) - use server or session timezone Default value is empty. ->>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) )", 0) \ DECLARE(Bool, export_merge_tree_part_overwrite_file_if_exists, false, R"( Overwrite file if it already exists when exporting a merge tree part @@ -8380,18 +8378,12 @@ When the hash join build side was converted to a FixedHashMap (see `enable_join_ DECLARE(Bool, rewrite_in_to_join, false, R"( Rewrite expressions like 'x IN subquery' to JOIN. This might be useful for optimizing the whole query with join reordering. )", EXPERIMENTAL) \ -<<<<<<< HEAD -======= DECLARE(Bool, object_storage_remote_initiator, false, R"( Execute request to object storage as remote on one of object_storage_cluster nodes. )", EXPERIMENTAL) \ DECLARE(String, object_storage_remote_initiator_cluster, "", R"( Cluster to choose remote initiator, when `object_storage_remote_initiator` is true. When empty, `object_storage_cluster` is used. )", EXPERIMENTAL) \ - DECLARE(Bool, allow_experimental_iceberg_read_optimization, true, R"( -Allow Iceberg read optimization based on Iceberg metadata. -)", EXPERIMENTAL) \ ->>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) \ /** Experimental timeSeries* aggregate functions. */ \ DECLARE_WITH_ALIAS(Bool, allow_experimental_time_series_aggregate_functions, false, R"( diff --git a/src/Core/SettingsChangesHistory.cpp b/src/Core/SettingsChangesHistory.cpp index e0fc89469657..080d15b05918 100644 --- a/src/Core/SettingsChangesHistory.cpp +++ b/src/Core/SettingsChangesHistory.cpp @@ -465,8 +465,8 @@ const VersionToSettingsChangesMap & getSettingsChangesHistory() {"allow_database_unity_catalog", false, true, "Turned ON by default for Antalya (alias)."}, {"allow_database_glue_catalog", false, true, "Turned ON by default for Antalya (alias)."}, // {"input_format_parquet_use_metadata_cache", true, true, "New setting, turned ON by default"}, // https://github.com/Altinity/ClickHouse/pull/586 - // {"iceberg_timezone_for_timestamptz", "UTC", "UTC", "New setting."}, - // {"object_storage_remote_initiator", false, false, "New setting."}, + {"iceberg_timezone_for_timestamptz", "UTC", "UTC", "New setting."}, + {"object_storage_remote_initiator", false, false, "New setting."}, // {"allow_experimental_iceberg_read_optimization", true, true, "New setting."}, {"object_storage_cluster_join_mode", "allow", "allow", "New setting"}, // {"lock_object_storage_task_distribution_ms", 500, 500, "New setting."}, @@ -486,12 +486,6 @@ const VersionToSettingsChangesMap & getSettingsChangesHistory() // {"export_merge_tree_partition_system_table_prefer_remote_information", true, true, "New setting."}, // {"export_merge_tree_part_throw_on_pending_mutations", true, true, "New setting."}, // {"export_merge_tree_part_throw_on_pending_patch_parts", true, true, "New setting."}, - {"iceberg_timezone_for_timestamptz", "UTC", "UTC", "New setting."}, - {"object_storage_remote_initiator", false, false, "New setting."}, - {"allow_experimental_iceberg_read_optimization", true, true, "New setting."}, - // {"object_storage_cluster_join_mode", "allow", "allow", "New setting"}, - {"lock_object_storage_task_distribution_ms", 500, 500, "New setting."}, - {"allow_retries_in_cluster_requests", false, false, "New setting"}, {"allow_experimental_export_merge_tree_part", false, true, "Turned ON by default for Antalya."}, {"export_merge_tree_part_overwrite_file_if_exists", false, false, "New setting."}, {"export_merge_tree_partition_force_export", false, false, "New setting."}, diff --git a/src/Databases/DataLake/DatabaseDataLake.cpp b/src/Databases/DataLake/DatabaseDataLake.cpp index 859bb0d9d021..1306ee870e5c 100644 --- a/src/Databases/DataLake/DatabaseDataLake.cpp +++ b/src/Databases/DataLake/DatabaseDataLake.cpp @@ -182,12 +182,8 @@ void DatabaseDataLake::initialize() const .region = settings[DatabaseDataLakeSetting::region].value, .namespaces = settings[DatabaseDataLakeSetting::namespaces].value, .aws_role_arn = settings[DatabaseDataLakeSetting::aws_role_arn].value, -<<<<<<< HEAD .aws_role_session_name = settings[DatabaseDataLakeSetting::aws_role_session_name].value, .aws_external_id = settings[DatabaseDataLakeSetting::aws_external_id].value, -======= - .aws_role_session_name = settings[DatabaseDataLakeSetting::aws_role_session_name].value ->>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) }; switch (settings[DatabaseDataLakeSetting::catalog_type].value) @@ -703,48 +699,9 @@ StoragePtr DatabaseDataLake::tryGetTableImpl(const String & name, ContextPtr con const auto is_secondary_query = context_->getClientInfo().query_kind == ClientInfo::QueryKind::SECONDARY_QUERY; -<<<<<<< HEAD const auto catalog_uuid = table_metadata.getTableUUID(); const UUID table_uuid = catalog_uuid ? parseFromString(*catalog_uuid) : UUIDHelpers::Nil; - if (can_use_parallel_replicas && !is_secondary_query) - { - auto storage_id = StorageID(getDatabaseName(), name, table_uuid); - auto storage_cluster = std::make_shared( - parallel_replicas_cluster_name, - configuration, - configuration->createObjectStorage(context_copy, /* is_readonly */ false, catalog->getCredentialsConfigurationCallback(storage_id)), - storage_id, - columns, - ConstraintsDescription{}, - nullptr, - context_, - /// Use is_table_function = true, - /// because this table is actually stateless like a table function. - /* is_table_function */true); - - if (context_->hasQueryContext() && context_->getSettingsRef()[Setting::log_queries]) - context_->getQueryContext()->addQueryFactoriesInfo(Context::QueryLogFactories::Storage, storage_cluster->getName()); - - storage_cluster->startup(); - return storage_cluster; - } - - /// Unlike table functions (s3, url, etc.), DataLake tables are queried as - /// `SELECT * FROM catalog.table` — the query sent to shards cannot be rewritten - /// into a Cluster table function variant. So when the initiator created a - /// StorageObjectStorageCluster (the branch above) and the shard is collaborating - /// with it, we need distributed_processing=true to use the task iterator. - const bool distributed_processing = - context_->getClientInfo().collaborate_with_initiator - && can_use_parallel_replicas; - - auto result_storage = std::make_shared( - configuration, - configuration->createObjectStorage(context_copy, /* is_readonly */ false, catalog->getCredentialsConfigurationCallback(StorageID(getDatabaseName(), name, table_uuid))), - context_copy, - StorageID(getDatabaseName(), name, table_uuid), -======= std::string cluster_name = configuration->isClusterSupported() ? settings[DatabaseDataLakeSetting::object_storage_cluster].value : ""; if (cluster_name.empty() && can_use_parallel_replicas && !is_secondary_query) @@ -753,9 +710,8 @@ StoragePtr DatabaseDataLake::tryGetTableImpl(const String & name, ContextPtr con auto storage_cluster = std::make_shared( cluster_name, configuration, - configuration->createObjectStorage(context_copy, /* is_readonly */ false, catalog->getCredentialsConfigurationCallback(StorageID(getDatabaseName(), name))), - StorageID(getDatabaseName(), name), ->>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) + configuration->createObjectStorage(context_copy, /* is_readonly */ false, catalog->getCredentialsConfigurationCallback(StorageID(getDatabaseName(), name, table_uuid))), + StorageID(getDatabaseName(), name, table_uuid), /* columns */columns, /* constraints */ConstraintsDescription{}, /* partition_by */nullptr, @@ -767,26 +723,16 @@ StoragePtr DatabaseDataLake::tryGetTableImpl(const String & name, ContextPtr con getCatalog(), /* if_not_exists*/true, /* is_datalake_query*/true, -<<<<<<< HEAD - distributed_processing, - /* partition_by */nullptr, - /* order_by */nullptr, -======= ->>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) /// Use is_table_function = true, /// because this table is actually stateless like a table function. /* is_table_function */true, /* lazy_init */true); -<<<<<<< HEAD if (context_->hasQueryContext() && context_->getSettingsRef()[Setting::log_queries]) - context_->getQueryContext()->addQueryFactoriesInfo(Context::QueryLogFactories::Storage, result_storage->getName()); + context_->getQueryContext()->addQueryFactoriesInfo(Context::QueryLogFactories::Storage, storage_cluster->getName()); - return result_storage; -======= storage_cluster->startup(); return storage_cluster; ->>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) } void DatabaseDataLake::dropTable( /// NOLINT @@ -1191,7 +1137,6 @@ void registerDatabaseDataLake(DatabaseFactory & factory) args.uuid, /*lazy_init=*/args.create_query.attach); }; -<<<<<<< HEAD /// TODO: DataLakeCatalog is polymorphic — underlying source (S3, Azure, HDFS, etc.) depends /// on the catalog type chosen at runtime. Consider adding source_access_type once a mechanism /// for runtime-dependent or composite source checks exist. @@ -1280,10 +1225,7 @@ SELECT count() from database_name.table_name; )DOCS_MD", .syntax = "ENGINE = DataLakeCatalog('catalog_url'[, 'user', 'password']) SETTINGS catalog_type = '...'", .related = {}}); -======= - factory.registerDatabase("DataLakeCatalog", create_fn, { .supports_arguments = true, .supports_settings = true }); factory.registerDatabase("Iceberg", create_fn, { .supports_arguments = true, .supports_settings = true }); ->>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) } } diff --git a/src/Databases/DataLake/DatabaseDataLakeSettings.cpp b/src/Databases/DataLake/DatabaseDataLakeSettings.cpp index 294178061b90..a982ea0a7663 100644 --- a/src/Databases/DataLake/DatabaseDataLakeSettings.cpp +++ b/src/Databases/DataLake/DatabaseDataLakeSettings.cpp @@ -47,12 +47,8 @@ namespace ErrorCodes DECLARE(String, google_adc_credentials_file, "", "Deprecated setting, will throw an exception if used", 0) \ DECLARE(String, dlf_access_key_id, "", "Access id of DLF token for Paimon REST Catalog", 0) \ DECLARE(String, dlf_access_key_secret, "", "Access secret of DLF token for Paimon REST Catalog", 0) \ -<<<<<<< HEAD DECLARE(Bool, force_add_bucket, false, "Add bucket name to the metadata path", 0) \ -======= DECLARE(String, namespaces, "*", "Comma-separated list of allowed namespaces", 0) \ - DECLARE(Bool, polaris_style_paths, true, "Enable Polaris/ADLS Gen2 path convention: the container name is prepended to the path in ABFSS locations (e.g. abfss://c@account/c/actual/path). When enabled, the redundant container prefix is stripped when building Azure HTTPS URLs. Disable if a real directory inside the container has the same name as the container itself.", 0) \ ->>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) #define LIST_OF_DATABASE_ICEBERG_SETTINGS(M, ALIAS) \ DATABASE_ICEBERG_RELATED_SETTINGS(M, ALIAS) \ diff --git a/src/Databases/DataLake/GlueCatalog.cpp b/src/Databases/DataLake/GlueCatalog.cpp index 7cb4cb51bebf..cba05247177d 100644 --- a/src/Databases/DataLake/GlueCatalog.cpp +++ b/src/Databases/DataLake/GlueCatalog.cpp @@ -58,16 +58,13 @@ namespace DB::ErrorCodes { extern const int BAD_ARGUMENTS; extern const int DATALAKE_DATABASE_ERROR; -<<<<<<< HEAD extern const int FAULT_INJECTED; + extern const int CATALOG_NAMESPACE_DISABLED; } namespace DB::FailPoints { extern const char check_database_datalake_negative[]; -======= - extern const int CATALOG_NAMESPACE_DISABLED; ->>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) } namespace DB::Setting diff --git a/src/Databases/DataLake/ICatalog.cpp b/src/Databases/DataLake/ICatalog.cpp index d51660fb8793..5b487faa5e3a 100644 --- a/src/Databases/DataLake/ICatalog.cpp +++ b/src/Databases/DataLake/ICatalog.cpp @@ -104,15 +104,6 @@ void TableMetadata::setLocation(const std::string & location_) if (pos_to_path == std::string::npos) { -<<<<<<< HEAD - /// Azure ABFSS format: extract container (before @) and account (after @) - bucket = bucket_part.substr(0, at_pos); - azure_account_with_suffix = bucket_part.substr(at_pos + 1); - - LOG_TEST(getLogger("TableMetadata"), - "Parsed Azure location - container: {}, account: {}, path: {}", - bucket, azure_account_with_suffix, path); -======= if (storage_type_str == "s3://") { // empty path is allowed for AWS S3Table location_without_path = location_; @@ -121,7 +112,6 @@ void TableMetadata::setLocation(const std::string & location_) } else throw DB::Exception(DB::ErrorCodes::NOT_IMPLEMENTED, "Unexpected location format: {}", location_); ->>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) } else { @@ -140,15 +130,6 @@ void TableMetadata::setLocation(const std::string & location_) bucket = bucket_part.substr(0, at_pos); azure_account_with_suffix = bucket_part.substr(at_pos + 1); - /// Some catalogs (e.g. Apache Polaris) follow the ADLS Gen2 filesystem convention - /// of including the container name as the first segment of the path in abfss:// locations, - /// e.g. abfss://container@account.dfs.core.windows.net/container/actual/path. - /// We record this as a flag so that `constructLocation` and `getMetadataLocation` can - /// strip the redundant prefix when needed, while `path` itself is left intact so that - /// `getLocation` remains a round-trip of `setLocation`. - if (polaris_style_abfss_paths && path.starts_with(bucket + "/")) - abfss_has_container_path_prefix = true; - LOG_TEST(getLogger("TableMetadata"), "Parsed Azure location - container: {}, account: {}, path: {}", bucket, azure_account_with_suffix, path); diff --git a/src/Databases/DataLake/RestCatalog.cpp b/src/Databases/DataLake/RestCatalog.cpp index fc693e8c0bd6..a52b9f65fe3a 100644 --- a/src/Databases/DataLake/RestCatalog.cpp +++ b/src/Databases/DataLake/RestCatalog.cpp @@ -55,8 +55,8 @@ namespace DB::ErrorCodes extern const int DATALAKE_DATABASE_ERROR; extern const int LOGICAL_ERROR; extern const int BAD_ARGUMENTS; -<<<<<<< HEAD extern const int FAULT_INJECTED; + extern const int CATALOG_NAMESPACE_DISABLED; } namespace DB::Setting @@ -67,9 +67,6 @@ namespace DB::Setting namespace DB::FailPoints { extern const char check_database_datalake_negative[]; -======= - extern const int CATALOG_NAMESPACE_DISABLED; ->>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) } namespace DataLake @@ -650,14 +647,12 @@ bool RestCatalog::empty() const bool found_table = false; auto stop_condition = [&](const std::string & namespace_name) -> bool { -<<<<<<< HEAD if (found_table) return true; -======= if (!allowed_namespaces.isNamespaceAllowed(namespace_name, /*nested*/ false)) return false; ->>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) + const auto tables = getTables(namespace_name, /* limit */1); if (!tables.empty()) found_table = true; @@ -1128,16 +1123,10 @@ bool RestCatalog::getTableMetadataImpl( if (result.requiresSchema()) { -<<<<<<< HEAD const bool allow_geo_parser = getContext()->getSettingsRef()[DB::Setting::allow_experimental_geo_types_in_iceberg].value; - auto schema_processor = DB::Iceberg::IcebergSchemaProcessor(allow_geo_parser); - auto id = DB::IcebergMetadata::parseTableSchema(metadata_object, schema_processor, log); -======= - // int format_version = metadata_object->getValue("format-version"); - auto schema_processor = DB::Iceberg::IcebergSchemaProcessor(context_); + auto schema_processor = DB::Iceberg::IcebergSchemaProcessor(context_, allow_geo_parser); auto id = DB::IcebergMetadata::parseTableSchema(metadata_object, schema_processor, context_, log); ->>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) auto schema = schema_processor.getClickhouseTableSchemaById(id); result.setSchema(*schema); } diff --git a/src/Databases/DataLake/UnityCatalog.cpp b/src/Databases/DataLake/UnityCatalog.cpp index 4fc035b85ad8..3478dec403d5 100644 --- a/src/Databases/DataLake/UnityCatalog.cpp +++ b/src/Databases/DataLake/UnityCatalog.cpp @@ -19,11 +19,8 @@ namespace DB::ErrorCodes { extern const int DATALAKE_DATABASE_ERROR; extern const int LOGICAL_ERROR; -<<<<<<< HEAD extern const int BAD_ARGUMENTS; -======= extern const int CATALOG_NAMESPACE_DISABLED; ->>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) } namespace diff --git a/src/Disks/DiskObjectStorage/ObjectStorages/S3/S3ObjectStorage.cpp b/src/Disks/DiskObjectStorage/ObjectStorages/S3/S3ObjectStorage.cpp index 16dd3e27cda2..4ea325577d48 100644 --- a/src/Disks/DiskObjectStorage/ObjectStorages/S3/S3ObjectStorage.cpp +++ b/src/Disks/DiskObjectStorage/ObjectStorages/S3/S3ObjectStorage.cpp @@ -370,17 +370,12 @@ void S3ObjectStorage::listObjects(const std::string & path, RelativePathsWithMet ProfileEvents::increment(ProfileEvents::S3ListObjects); ProfileEvents::increment(ProfileEvents::DiskS3ListObjects); -<<<<<<< HEAD - outcome = client.get()->ListObjectsV2(request); - throwIfError(outcome, "while listing objects in bucket '{}' with prefix '{}' on disk '{}'", uri.bucket, path, disk_name); -======= { ProfileEventTimeIncrement watch(ProfileEvents::S3ListObjectsMicroseconds); outcome = client.get()->ListObjectsV2(request); } - throwIfError(outcome); ->>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) + throwIfError(outcome, "while listing objects in bucket '{}' with prefix '{}' on disk '{}'", uri.bucket, path, disk_name); auto result = outcome.GetResult(); auto objects = result.GetContents(); diff --git a/src/IO/S3/URI.cpp b/src/IO/S3/URI.cpp index 0028db01b004..26b16cf2bed5 100644 --- a/src/IO/S3/URI.cpp +++ b/src/IO/S3/URI.cpp @@ -156,7 +156,6 @@ URI::URI(const std::string & uri_, bool allow_archive_path_syntax, bool keep_pre validateKey(key, uri); } -<<<<<<< HEAD bool URI::tryInitPathStyle() { /// Case when bucket name and key represented in the path of S3 URL. @@ -212,7 +211,8 @@ bool URI::tryInitVirtualHostedStyle(bool is_using_aws_private_link_interface, bo else storage_name = name; return true; -======= +} + bool URI::isAWSRegion(std::string_view region) { /// List from https://docs.aws.amazon.com/general/latest/gr/s3.html @@ -260,7 +260,6 @@ bool URI::isAWSRegion(std::string_view region) region = region.substr(3); return regions.contains(region); ->>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) } void URI::addRegionToURI(const std::string ®ion) diff --git a/src/IO/S3/URI.h b/src/IO/S3/URI.h index ada42a34f672..f1fce02452ea 100644 --- a/src/IO/S3/URI.h +++ b/src/IO/S3/URI.h @@ -47,15 +47,13 @@ struct URI static void validateBucket(const std::string & bucket, const Poco::URI & uri); static void validateKey(const std::string & key, const Poco::URI & uri); -<<<<<<< HEAD -private: - bool tryInitPathStyle(); - bool tryInitVirtualHostedStyle(bool is_using_aws_private_link_interface, bool use_strict_pattern); -======= /// Returns true if 'region' string is an AWS S3 region /// https://docs.aws.amazon.com/general/latest/gr/s3.html static bool isAWSRegion(std::string_view region); ->>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) + +private: + bool tryInitPathStyle(); + bool tryInitVirtualHostedStyle(bool is_using_aws_private_link_interface, bool use_strict_pattern); }; } diff --git a/src/IO/S3/getObjectInfo.cpp b/src/IO/S3/getObjectInfo.cpp index e9aff5925e38..deec76d3fdc6 100644 --- a/src/IO/S3/getObjectInfo.cpp +++ b/src/IO/S3/getObjectInfo.cpp @@ -9,11 +9,7 @@ namespace ProfileEvents { extern const Event S3GetObjectTagging; extern const Event S3HeadObject; -<<<<<<< HEAD -======= extern const Event S3HeadObjectMicroseconds; - extern const Event DiskS3GetObject; ->>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) extern const Event DiskS3GetObjectTagging; extern const Event DiskS3HeadObject; } diff --git a/src/Interpreters/IcebergMetadataLog.cpp b/src/Interpreters/IcebergMetadataLog.cpp index fe8c14dd3ec4..02bc2cbbd98e 100644 --- a/src/Interpreters/IcebergMetadataLog.cpp +++ b/src/Interpreters/IcebergMetadataLog.cpp @@ -112,13 +112,8 @@ void insertRowToLogTable( .query_id = local_context->getCurrentQueryId(), .content_type = row_log_level, .table_path = table_path, -<<<<<<< HEAD .file_path = file_path.serialize(), - .metadata_content = row, -======= - .file_path = file_path, .metadata_content = get_row(), ->>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) .row_in_file = row_in_file, .pruning_status = pruning_status}); } diff --git a/src/Parsers/FunctionSecretArgumentsFinder.h b/src/Parsers/FunctionSecretArgumentsFinder.h index 525597f6db1a..49134c1eb7b4 100644 --- a/src/Parsers/FunctionSecretArgumentsFinder.h +++ b/src/Parsers/FunctionSecretArgumentsFinder.h @@ -102,20 +102,14 @@ class FunctionSecretArgumentsFinder result.start = real_index; result.are_named = argument_is_named; } -<<<<<<< HEAD chassert(result.replacement.empty()); /// We shouldn't use replacement with masking other arguments /// Widen the masked range to cover `index`. Arguments are normally marked consecutively in /// increasing order, but a malformed query can mix the named secret form (`key = ...`) with the /// positional form and ask to mask an earlier index after a later one. Masking is best-effort /// over arbitrary user input, so it must widen the range rather than assert on the order. - size_t end = std::max(result.start + result.count, index + 1); - result.start = std::min(result.start, index); + size_t end = std::max(result.start + result.count, real_index + 1); + result.start = std::min(result.start, real_index); result.count = end - result.start; -======= - chassert(real_index >= result.start); /// We always check arguments consecutively - chassert(result.replacement.empty()); /// We shouldn't use replacement with masking other arguments - result.count = real_index + 1 - result.start; ->>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) if (!argument_is_named) result.are_named = false; } @@ -142,27 +136,17 @@ class FunctionSecretArgumentsFinder findIcebergFunctionSecretArguments(/* is_cluster_function= */ true); } else if ((function->name() == "s3") || (function->name() == "cosn") || (function->name() == "oss") || -<<<<<<< HEAD (function->name() == "deltaLake") || (function->name() == "deltaLakeS3") || (function->name() == "hudi") || - (function->name() == "iceberg") || (function->name() == "gcs") || (function->name() == "icebergS3") || + (function->name() == "gcs") || (function->name() == "icebergS3") || (function->name() == "paimon") || (function->name() == "paimonS3")) -======= - (function->name() == "deltaLake") || (function->name() == "hudi") || - (function->name() == "gcs") || (function->name() == "icebergS3") || (function->name() == "paimon") || - (function->name() == "paimonS3")) ->>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) { /// s3('url', 'aws_access_key_id', 'aws_secret_access_key', ...) findS3FunctionSecretArguments(/* is_cluster_function= */ false); } else if ((function->name() == "s3Cluster") || (function ->name() == "hudiCluster") || (function ->name() == "deltaLakeCluster") || (function ->name() == "deltaLakeS3Cluster") || -<<<<<<< HEAD - (function ->name() == "icebergS3Cluster") || (function ->name() == "icebergCluster") || + (function ->name() == "icebergS3Cluster") || (function ->name() == "paimonCluster") || (function ->name() == "paimonS3Cluster")) -======= - (function ->name() == "icebergS3Cluster")) ->>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) { /// s3Cluster('cluster_name', 'url', 'aws_access_key_id', 'aws_secret_access_key', ...) findS3FunctionSecretArguments(/* is_cluster_function= */ true); diff --git a/src/Processors/QueryPlan/ReadFromObjectStorageStep.cpp b/src/Processors/QueryPlan/ReadFromObjectStorageStep.cpp index 6daa8ead6201..5e7ef89e439d 100644 --- a/src/Processors/QueryPlan/ReadFromObjectStorageStep.cpp +++ b/src/Processors/QueryPlan/ReadFromObjectStorageStep.cpp @@ -71,7 +71,7 @@ void ReadFromObjectStorageStep::applyFilters(ActionDAGNodes added_filter_nodes) if (!filter_actions_dag) return; - if (boost::iequals(configuration->format, "Parquet") || boost::iequals(configuration->format, "ORC")) + if (boost::iequals(configuration->getFormat(), "Parquet") || boost::iequals(configuration->getFormat(), "ORC")) prepareEagerKeyConditionSets( filter_actions_dag, storage_snapshot, info.source_header, diff --git a/src/Server/TCPHandler.cpp b/src/Server/TCPHandler.cpp index d0708fa5c64d..219303fe6b93 100644 --- a/src/Server/TCPHandler.cpp +++ b/src/Server/TCPHandler.cpp @@ -36,11 +36,6 @@ #include #include #include -<<<<<<< HEAD -#include -======= -#include ->>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) #include #include #include diff --git a/src/Storages/IStorageCluster.h b/src/Storages/IStorageCluster.h index 351bde8618fb..e42c841baf14 100644 --- a/src/Storages/IStorageCluster.h +++ b/src/Storages/IStorageCluster.h @@ -58,12 +58,10 @@ class IStorageCluster : public IStorage bool supportsOptimizationToSubcolumns() const override { return false; } bool supportsTrivialCountOptimization(const StorageSnapshotPtr &, ContextPtr) const override { return true; } -<<<<<<< HEAD const String & getClusterName() const { return cluster_name; } -======= + const String & getOriginalClusterName() const { return cluster_name; } virtual String getClusterName(ContextPtr /* context */) const { return getOriginalClusterName(); } ->>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) protected: virtual void updateQueryToSendIfNeeded( diff --git a/src/Storages/ObjectStorage/DataLakes/DataLakeConfiguration.h b/src/Storages/ObjectStorage/DataLakes/DataLakeConfiguration.h index 4faf3bae4cae..3ad8d3c716a5 100644 --- a/src/Storages/ObjectStorage/DataLakes/DataLakeConfiguration.h +++ b/src/Storages/ObjectStorage/DataLakes/DataLakeConfiguration.h @@ -22,17 +22,12 @@ #include #include #include -<<<<<<< HEAD -#include -======= -#include #include #include #include #include #include - ->>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) +#include #include #include #include @@ -182,32 +177,19 @@ class DataLakeConfiguration : public BaseStorageConfiguration, public std::enabl std::shared_ptr catalog, const std::optional & format_settings) override { -<<<<<<< HEAD - assertInitialized(); - current_metadata->mutate(commands, storage_ptr, context, storage_id, metadata_snapshot, catalog, format_settings); -======= assertInitializedDL(); - current_metadata->mutate(commands, shared_from_this(), context, storage_id, metadata_snapshot, catalog, format_settings); ->>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) + current_metadata->mutate(commands, storage_ptr, context, storage_id, metadata_snapshot, catalog, format_settings); } void checkMutationIsPossible(ObjectStoragePtr object_storage, ContextPtr context, const MutationCommands & commands) override { -<<<<<<< HEAD lazyInitializeIfNeeded(object_storage, context); -======= - assertInitializedDL(); ->>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) current_metadata->checkMutationIsPossible(commands); } void checkAlterIsPossible(ObjectStoragePtr object_storage, ContextPtr context, const AlterCommands & commands) override { -<<<<<<< HEAD lazyInitializeIfNeeded(object_storage, context); -======= - assertInitializedDL(); ->>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) current_metadata->checkAlterIsPossible(commands); } @@ -218,14 +200,8 @@ class DataLakeConfiguration : public BaseStorageConfiguration, public std::enabl const StorageID & storage_id, std::shared_ptr catalog) override { -<<<<<<< HEAD lazyInitializeIfNeeded(object_storage, context); current_metadata->alter(params, context, storage_id, catalog); -======= - assertInitializedDL(); - current_metadata->alter(params, context); - ->>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) } ObjectStoragePtr createObjectStorage(ContextPtr context, bool is_readonly, StorageObjectStorageConfiguration::CredentialsConfigurationCallback refresh_credentials_callback) override @@ -391,7 +367,6 @@ class DataLakeConfiguration : public BaseStorageConfiguration, public std::enabl std::shared_ptr getCatalog([[maybe_unused]] ContextPtr context, [[maybe_unused]] const StorageID & table_id) const override { -<<<<<<< HEAD #if USE_AVRO && USE_PARQUET if ((*settings)[DataLakeStorageSetting::storage_catalog_type].changed || (*settings)[DataLakeStorageSetting::storage_catalog_url].changed @@ -408,56 +383,13 @@ class DataLakeConfiguration : public BaseStorageConfiguration, public std::enabl return nullptr; return datalake_database->getCatalog(); #else -======= -#if USE_AWS_S3 && USE_AVRO - if ((*settings)[DataLakeStorageSetting::storage_catalog_type].value == DatabaseDataLakeCatalogType::GLUE) - { - auto catalog_parameters = DataLake::CatalogSettings{ - .storage_endpoint = (*settings)[DataLakeStorageSetting::object_storage_endpoint].value, - .aws_access_key_id = (*settings)[DataLakeStorageSetting::storage_aws_access_key_id].value, - .aws_secret_access_key = (*settings)[DataLakeStorageSetting::storage_aws_secret_access_key].value, - .region = (*settings)[DataLakeStorageSetting::storage_region].value, - .namespaces = catalog_namespaces, - .aws_role_arn = (*settings)[DataLakeStorageSetting::storage_aws_role_arn].value, - .aws_role_session_name = (*settings)[DataLakeStorageSetting::storage_aws_role_session_name].value - }; - - return std::make_shared( - (*settings)[DataLakeStorageSetting::storage_catalog_url].value, - context, - catalog_parameters, - /* table_engine_definition */nullptr - ); - } - /// Attach condition is provided for compatibility. - if ((*settings)[DataLakeStorageSetting::storage_catalog_type].value == DatabaseDataLakeCatalogType::ICEBERG_REST || - (is_attach && (*settings)[DataLakeStorageSetting::storage_catalog_type].value == DatabaseDataLakeCatalogType::NONE && !(*settings)[DataLakeStorageSetting::storage_catalog_url].value.empty())) - { - return std::make_shared( - (*settings)[DataLakeStorageSetting::storage_warehouse].value, - (*settings)[DataLakeStorageSetting::storage_catalog_url].value, - (*settings)[DataLakeStorageSetting::storage_catalog_credential].value, - (*settings)[DataLakeStorageSetting::storage_auth_scope].value, - (*settings)[DataLakeStorageSetting::storage_auth_header], - (*settings)[DataLakeStorageSetting::storage_oauth_server_uri].value, - (*settings)[DataLakeStorageSetting::storage_oauth_server_use_request_body].value, - catalog_namespaces, - context); - } - -#endif ->>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) return nullptr; #endif } bool optimize(ObjectStoragePtr object_storage, const StorageMetadataPtr & metadata_snapshot, ContextPtr context, const std::optional & format_settings) override { -<<<<<<< HEAD lazyInitializeIfNeeded(object_storage, context); -======= - assertInitializedDL(); ->>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) return current_metadata->optimize(metadata_snapshot, context, format_settings); } @@ -895,8 +827,8 @@ class StorageIcebergConfiguration : public StorageObjectStorageConfiguration, pu ColumnMapperPtr getColumnMapperForCurrentSchema(StorageMetadataPtr storage_metadata_snapshot, ContextPtr context) const override { return getImpl().getColumnMapperForCurrentSchema(storage_metadata_snapshot, context); } - std::shared_ptr getCatalog(ContextPtr context, bool is_attach) const override - { return getImpl().getCatalog(context, is_attach); } + std::shared_ptr getCatalog(ContextPtr context, const StorageID & table_id) const override + { return getImpl().getCatalog(context, table_id); } bool optimize(const StorageMetadataPtr & metadata_snapshot, ContextPtr context, const std::optional & format_settings) override { return getImpl().optimize(metadata_snapshot, context, format_settings); } diff --git a/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergMetadata.cpp b/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergMetadata.cpp index 277bc3b4c41e..5b554cc0e03a 100644 --- a/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergMetadata.cpp +++ b/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergMetadata.cpp @@ -242,22 +242,16 @@ Iceberg::PersistentTableComponents IcebergMetadata::initializePersistentTableCom } auto table_path = configuration->getPathForRead().path; return PersistentTableComponents{ -<<<<<<< HEAD - .schema_processor = std::make_shared(context_->getSettingsRef()[Setting::allow_experimental_geo_types_in_iceberg]), -======= - .schema_processor = std::make_shared(context_), ->>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) + .schema_processor = std::make_shared( + context_, context_->getSettingsRef()[Setting::allow_experimental_geo_types_in_iceberg]), .metadata_cache = cache_ptr, .format_version = format_version, .table_location = table_location, .metadata_compression_method = compression_method, .table_path = table_path, .table_uuid = table_uuid, -<<<<<<< HEAD .path_resolver = IcebergPathResolver(table_location, table_path, configuration->getTypeName(), configuration->getNamespace()), -======= .common_namespace = configuration->getNamespace(), ->>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) }; } @@ -362,10 +356,7 @@ void IcebergMetadata::backgroundMetadataPrefetcherThread() Int32 IcebergMetadata::parseTableSchema( const Poco::JSON::Object::Ptr & metadata_object, IcebergSchemaProcessor & schema_processor, -<<<<<<< HEAD -======= ContextPtr context_, ->>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) LoggerPtr metadata_logger) { const auto format_version = metadata_object->getValue(f_format_version); @@ -414,16 +405,11 @@ Int32 IcebergMetadata::parseTableSchema( } } -<<<<<<< HEAD static Poco::JSON::Object::Ptr traverseMetadataAndFindNecessarySnapshotObject( - Poco::JSON::Object::Ptr metadata_object, Int64 snapshot_id, IcebergSchemaProcessorPtr schema_processor) -======= -Poco::JSON::Object::Ptr traverseMetadataAndFindNecessarySnapshotObject( Poco::JSON::Object::Ptr metadata_object, Int64 snapshot_id, IcebergSchemaProcessorPtr schema_processor, ContextPtr local_context) ->>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) { if (!metadata_object->has(f_snapshots)) throw Exception(ErrorCodes::ICEBERG_SPECIFICATION_VIOLATION, "No snapshot set found in metadata for iceberg file"); diff --git a/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergMetadata.h b/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergMetadata.h index f8b7f503799c..a60b6784f519 100644 --- a/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergMetadata.h +++ b/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergMetadata.h @@ -102,10 +102,7 @@ class IcebergMetadata : public IDataLakeMetadata static Int32 parseTableSchema( const Poco::JSON::Object::Ptr & metadata_object, Iceberg::IcebergSchemaProcessor & schema_processor, -<<<<<<< HEAD -======= ContextPtr context_, ->>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) LoggerPtr metadata_logger); bool supportsUpdate() const override { return true; } diff --git a/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergWrites.cpp b/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergWrites.cpp index c1f9269b3106..6ec1e4c35a7b 100644 --- a/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergWrites.cpp +++ b/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergWrites.cpp @@ -877,13 +877,7 @@ IcebergStorageSink::IcebergStorageSink( , table_id(table_id_) , persistent_table_components(persistent_table_components_) , data_lake_settings(configuration_->getDataLakeSettings()) -<<<<<<< HEAD - , write_format(configuration_->format) -======= , write_format(configuration_->getFormat()) - , blob_storage_type_name(configuration_->getTypeName()) - , blob_storage_namespace_name(configuration_->getNamespace()) ->>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) { auto [last_version, metadata_path, compression_method] = getLatestOrExplicitMetadataFileAndVersion( object_storage, diff --git a/src/Storages/ObjectStorage/DataLakes/Iceberg/PersistentTableComponents.h b/src/Storages/ObjectStorage/DataLakes/Iceberg/PersistentTableComponents.h index 91f7dc19c7d0..0a7539f9a12d 100644 --- a/src/Storages/ObjectStorage/DataLakes/Iceberg/PersistentTableComponents.h +++ b/src/Storages/ObjectStorage/DataLakes/Iceberg/PersistentTableComponents.h @@ -24,8 +24,8 @@ struct PersistentTableComponents const CompressionMethod metadata_compression_method; const String table_path; const std::optional table_uuid; -<<<<<<< HEAD const IcebergPathResolver path_resolver; + const String common_namespace; /// Invalidate cached metadata for this table under both keys we may have used to cache it /// (`table_path` and `table_uuid`). @@ -37,9 +37,6 @@ struct PersistentTableComponents if (table_uuid.has_value()) metadata_cache->remove(*table_uuid); } -======= - const String common_namespace; ->>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) }; } diff --git a/src/Storages/ObjectStorage/DataLakes/Iceberg/SchemaProcessor.cpp b/src/Storages/ObjectStorage/DataLakes/Iceberg/SchemaProcessor.cpp index 6975efa85cbd..4324008ea54c 100644 --- a/src/Storages/ObjectStorage/DataLakes/Iceberg/SchemaProcessor.cpp +++ b/src/Storages/ObjectStorage/DataLakes/Iceberg/SchemaProcessor.cpp @@ -300,11 +300,7 @@ NamesAndTypesList IcebergSchemaProcessor::tryGetFieldsCharacteristics(Int32 sche return fields; } -<<<<<<< HEAD -DataTypePtr IcebergSchemaProcessor::getSimpleType(const String & type_name, bool allow_geo_parser) -======= -DataTypePtr IcebergSchemaProcessor::getSimpleType(const String & type_name, ContextPtr context_) ->>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) +DataTypePtr IcebergSchemaProcessor::getSimpleType(const String & type_name, ContextPtr context_, bool allow_geo_parser) { if (type_name == f_boolean) return DataTypeFactory::instance().get("Bool"); @@ -323,18 +319,14 @@ DataTypePtr IcebergSchemaProcessor::getSimpleType(const String & type_name, Cont if (type_name == f_timestamp) return std::make_shared(6); if (type_name == f_timestamptz) -<<<<<<< HEAD - return std::make_shared(6, "UTC"); - if (type_name == f_timestamp_ns) - return std::make_shared(9); - if (type_name == f_timestamptz_ns) - return std::make_shared(9, "UTC"); -======= { std::string timezone = context_->getSettingsRef()[Setting::iceberg_timezone_for_timestamptz]; return std::make_shared(6, timezone); } ->>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) + if (type_name == f_timestamp_ns) + return std::make_shared(9); + if (type_name == f_timestamptz_ns) + return std::make_shared(9, "UTC"); if (type_name == f_string || type_name == f_binary) return std::make_shared(); @@ -444,13 +436,8 @@ DataTypePtr IcebergSchemaProcessor::getFieldType( if (type.isString()) { const String & type_name = type.extract(); -<<<<<<< HEAD - auto data_type = getSimpleType(type_name, allow_geo_parser); + auto data_type = getSimpleType(type_name, context_, allow_geo_parser); return required || !data_type->canBeInsideNullable() ? data_type : makeNullable(data_type); -======= - auto data_type = getSimpleType(type_name, context_); - return required ? data_type : makeNullable(data_type); ->>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) } throw Exception(ErrorCodes::BAD_ARGUMENTS, "Unexpected 'type' field: {}", type.toString()); diff --git a/src/Storages/ObjectStorage/DataLakes/Iceberg/SchemaProcessor.h b/src/Storages/ObjectStorage/DataLakes/Iceberg/SchemaProcessor.h index bff81414d642..8533614fe799 100644 --- a/src/Storages/ObjectStorage/DataLakes/Iceberg/SchemaProcessor.h +++ b/src/Storages/ObjectStorage/DataLakes/Iceberg/SchemaProcessor.h @@ -82,15 +82,10 @@ class IcebergSchemaProcessor : private WithContext using Node = ActionsDAG::Node; public: -<<<<<<< HEAD - explicit IcebergSchemaProcessor(bool allow_geo_parser_ = false) : allow_geo_parser(allow_geo_parser_) {} - - void addIcebergTableSchema(Poco::JSON::Object::Ptr schema_ptr); -======= - explicit IcebergSchemaProcessor(ContextPtr context_) : WithContext(context_) {} + explicit IcebergSchemaProcessor(ContextPtr context_, bool allow_geo_parser_ = false) + : WithContext(context_), allow_geo_parser(allow_geo_parser_) {} void addIcebergTableSchema(Poco::JSON::Object::Ptr schema_ptr, ContextPtr context_); ->>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) std::shared_ptr getClickhouseTableSchemaById(Int32 id); std::shared_ptr getSchemaTransformationDagByIds(ContextPtr context_, Int32 old_id, Int32 new_id); NameAndTypePair getFieldCharacteristics(Int32 schema_version, Int32 source_id) const; @@ -100,11 +95,7 @@ class IcebergSchemaProcessor : private WithContext Poco::JSON::Object::Ptr getIcebergTableSchemaById(Int32 id) const; bool hasClickhouseTableSchemaById(Int32 id) const; -<<<<<<< HEAD - static DataTypePtr getSimpleType(const String & type_name, bool allow_geo_parser = true); -======= - static DataTypePtr getSimpleType(const String & type_name, ContextPtr context_); ->>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) + static DataTypePtr getSimpleType(const String & type_name, ContextPtr context_, bool allow_geo_parser = true); static std::unordered_map traverseSchema(Poco::JSON::Array::Ptr schema); diff --git a/src/Storages/ObjectStorage/DataLakes/Iceberg/Utils.cpp b/src/Storages/ObjectStorage/DataLakes/Iceberg/Utils.cpp index 7d83abf630f8..f6165514addd 100644 --- a/src/Storages/ObjectStorage/DataLakes/Iceberg/Utils.cpp +++ b/src/Storages/ObjectStorage/DataLakes/Iceberg/Utils.cpp @@ -177,7 +177,6 @@ Iceberg::MetadataFileWithInfo getMetadataFileAndVersion(const std::string & path path); } String version_str; -<<<<<<< HEAD /// `vN.metadata.json` or `vN-.metadata.json` (the latter is what /// `apache/iceberg-rest-fixture` and other iceberg-java REST catalogs /// write when committing a new metadata file). @@ -191,12 +190,6 @@ Iceberg::MetadataFileWithInfo getMetadataFileAndVersion(const std::string & path file_name); version_str = String(file_name.begin() + 1, file_name.begin() + end_pos); } -======= - /// v.metadata.json - /// v-.metadata.json - generated by FileNamesGenerator::generateMetadataName with use_uuid_in_metadata flag - if (file_name.starts_with('v')) - version_str = String(file_name.begin() + 1, file_name.begin() + file_name.find_first_of(".-")); ->>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) /// -.metadata.json else { @@ -1376,15 +1369,11 @@ KeyDescription getSortingKeyDescriptionFromMetadata(Poco::JSON::Object::Ptr meta auto column_name = source_id_to_column_name[source_id]; int direction = field->getValue(f_direction) == "asc" ? 1 : -1; auto iceberg_transform_name = field->getValue(f_transform); -<<<<<<< HEAD - auto clickhouse_transform_name = parseTransformAndArgument(iceberg_transform_name); + auto clickhouse_transform_name = parseTransformAndArgument(iceberg_transform_name, + local_context->getSettingsRef()[Setting::iceberg_partition_timezone]); /// Quote the column name so identifiers with special characters (e.g. `@timestamp`) /// produce a parseable ORDER BY clause. auto quoted_column_name = backQuoteIfNeed(column_name); -======= - auto clickhouse_transform_name = parseTransformAndArgument(iceberg_transform_name, - local_context->getSettingsRef()[Setting::iceberg_partition_timezone]); ->>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) String full_argument; if (clickhouse_transform_name->transform_name != "identity") { @@ -1393,14 +1382,10 @@ KeyDescription getSortingKeyDescriptionFromMetadata(Poco::JSON::Object::Ptr meta { full_argument += std::to_string(*clickhouse_transform_name->argument) + ", "; } -<<<<<<< HEAD - full_argument += quoted_column_name + ")"; -======= - full_argument += column_name; + full_argument += quoted_column_name; if (clickhouse_transform_name->time_zone) full_argument += ", '" + *clickhouse_transform_name->time_zone + "'"; full_argument += ")"; ->>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) } else { diff --git a/src/Storages/ObjectStorage/DataLakes/Iceberg/tests/gtest_iceberg_schema_processor.cpp b/src/Storages/ObjectStorage/DataLakes/Iceberg/tests/gtest_iceberg_schema_processor.cpp index 8719762b0ba2..9a3834aac4d8 100644 --- a/src/Storages/ObjectStorage/DataLakes/Iceberg/tests/gtest_iceberg_schema_processor.cpp +++ b/src/Storages/ObjectStorage/DataLakes/Iceberg/tests/gtest_iceberg_schema_processor.cpp @@ -1,5 +1,6 @@ #include +#include #include #include #include @@ -8,107 +9,107 @@ using namespace DB::Iceberg; TEST(IcebergSchemaProcessor, GetSimpleTypeBoolean) { - auto type = IcebergSchemaProcessor::getSimpleType("boolean"); + auto type = IcebergSchemaProcessor::getSimpleType("boolean", getContext().context); EXPECT_EQ(type->getName(), "Bool"); } TEST(IcebergSchemaProcessor, GetSimpleTypeInt) { - auto type = IcebergSchemaProcessor::getSimpleType("int"); + auto type = IcebergSchemaProcessor::getSimpleType("int", getContext().context); EXPECT_EQ(type->getName(), "Int32"); } TEST(IcebergSchemaProcessor, GetSimpleTypeLong) { - auto type = IcebergSchemaProcessor::getSimpleType("long"); + auto type = IcebergSchemaProcessor::getSimpleType("long", getContext().context); EXPECT_EQ(type->getName(), "Int64"); } TEST(IcebergSchemaProcessor, GetSimpleTypeBigint) { - auto type = IcebergSchemaProcessor::getSimpleType("bigint"); + auto type = IcebergSchemaProcessor::getSimpleType("bigint", getContext().context); EXPECT_EQ(type->getName(), "Int64"); } TEST(IcebergSchemaProcessor, GetSimpleTypeFloat) { - auto type = IcebergSchemaProcessor::getSimpleType("float"); + auto type = IcebergSchemaProcessor::getSimpleType("float", getContext().context); EXPECT_EQ(type->getName(), "Float32"); } TEST(IcebergSchemaProcessor, GetSimpleTypeDouble) { - auto type = IcebergSchemaProcessor::getSimpleType("double"); + auto type = IcebergSchemaProcessor::getSimpleType("double", getContext().context); EXPECT_EQ(type->getName(), "Float64"); } TEST(IcebergSchemaProcessor, GetSimpleTypeDate) { - auto type = IcebergSchemaProcessor::getSimpleType("date"); + auto type = IcebergSchemaProcessor::getSimpleType("date", getContext().context); EXPECT_EQ(type->getName(), "Date32"); } TEST(IcebergSchemaProcessor, GetSimpleTypeTime) { - auto type = IcebergSchemaProcessor::getSimpleType("time"); + auto type = IcebergSchemaProcessor::getSimpleType("time", getContext().context); EXPECT_EQ(type->getName(), "Int64"); } TEST(IcebergSchemaProcessor, GetSimpleTypeTimestamp) { - auto type = IcebergSchemaProcessor::getSimpleType("timestamp"); + auto type = IcebergSchemaProcessor::getSimpleType("timestamp", getContext().context); EXPECT_EQ(type->getName(), "DateTime64(6)"); } TEST(IcebergSchemaProcessor, GetSimpleTypeTimestamptz) { - auto type = IcebergSchemaProcessor::getSimpleType("timestamptz"); + auto type = IcebergSchemaProcessor::getSimpleType("timestamptz", getContext().context); EXPECT_EQ(type->getName(), "DateTime64(6, 'UTC')"); } TEST(IcebergSchemaProcessor, GetSimpleTypeTimestampNs) { - auto type = IcebergSchemaProcessor::getSimpleType("timestamp_ns"); + auto type = IcebergSchemaProcessor::getSimpleType("timestamp_ns", getContext().context); EXPECT_EQ(type->getName(), "DateTime64(9)"); } TEST(IcebergSchemaProcessor, GetSimpleTypeTimestamptzNs) { - auto type = IcebergSchemaProcessor::getSimpleType("timestamptz_ns"); + auto type = IcebergSchemaProcessor::getSimpleType("timestamptz_ns", getContext().context); EXPECT_EQ(type->getName(), "DateTime64(9, 'UTC')"); } TEST(IcebergSchemaProcessor, GetSimpleTypeString) { - auto type = IcebergSchemaProcessor::getSimpleType("string"); + auto type = IcebergSchemaProcessor::getSimpleType("string", getContext().context); EXPECT_EQ(type->getName(), "String"); } TEST(IcebergSchemaProcessor, GetSimpleTypeBinary) { - auto type = IcebergSchemaProcessor::getSimpleType("binary"); + auto type = IcebergSchemaProcessor::getSimpleType("binary", getContext().context); EXPECT_EQ(type->getName(), "String"); } TEST(IcebergSchemaProcessor, GetSimpleTypeUuid) { - auto type = IcebergSchemaProcessor::getSimpleType("uuid"); + auto type = IcebergSchemaProcessor::getSimpleType("uuid", getContext().context); EXPECT_EQ(type->getName(), "UUID"); } TEST(IcebergSchemaProcessor, GetSimpleTypeFixed) { - auto type = IcebergSchemaProcessor::getSimpleType("fixed[16]"); + auto type = IcebergSchemaProcessor::getSimpleType("fixed[16]", getContext().context); EXPECT_EQ(type->getName(), "FixedString(16)"); } TEST(IcebergSchemaProcessor, GetSimpleTypeDecimal) { - auto type = IcebergSchemaProcessor::getSimpleType("decimal(10, 2)"); + auto type = IcebergSchemaProcessor::getSimpleType("decimal(10, 2)", getContext().context); EXPECT_EQ(type->getName(), "Decimal(10, 2)"); } TEST(IcebergSchemaProcessor, GetSimpleTypeUnknownThrows) { - EXPECT_THROW(IcebergSchemaProcessor::getSimpleType("unknown_type"), DB::Exception); + EXPECT_THROW(IcebergSchemaProcessor::getSimpleType("unknown_type", getContext().context), DB::Exception); } diff --git a/src/Storages/ObjectStorage/S3/Configuration.cpp b/src/Storages/ObjectStorage/S3/Configuration.cpp index d2f95cf379ce..4433b373efcd 100644 --- a/src/Storages/ObjectStorage/S3/Configuration.cpp +++ b/src/Storages/ObjectStorage/S3/Configuration.cpp @@ -108,11 +108,8 @@ static const std::unordered_set optional_configuration_keys = "partition_strategy", "partition_columns_in_data_file", "storage_class_name", -<<<<<<< HEAD "storage_class", /// Interchangeable alias for `storage_class_name`, see issue #68551 -======= "storage_type", ->>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) /// Private configuration options "role_arn", /// for extra_credentials "role_session_name", /// for extra_credentials diff --git a/src/Storages/ObjectStorage/StorageObjectStorage.cpp b/src/Storages/ObjectStorage/StorageObjectStorage.cpp index 94e3321ec3dd..6cb98ae5dfca 100644 --- a/src/Storages/ObjectStorage/StorageObjectStorage.cpp +++ b/src/Storages/ObjectStorage/StorageObjectStorage.cpp @@ -97,18 +97,6 @@ String StorageObjectStorage::getPathSample(ContextPtr context) query_settings.throw_on_zero_files_match = false; query_settings.ignore_non_existent_file = true; -<<<<<<< HEAD -======= - bool local_distributed_processing = distributed_processing; - if (context->getSettingsRef()[Setting::use_hive_partitioning]) - local_distributed_processing = false; - - const auto path = configuration->getRawPath(); - - if (!configuration->isArchive() && !path.hasGlobs() && !local_distributed_processing) - return path.path; - ->>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) auto file_iterator = StorageObjectStorageSource::createFileIterator( configuration, query_settings, @@ -443,7 +431,7 @@ void StorageObjectStorage::updateExternalDynamicMetadataIfExists(ContextPtr quer new_metadata.columns, query_context, format_settings, - configuration->partition_strategy_type))); + configuration->getPartitionStrategyType()))); } diff --git a/src/Storages/ObjectStorage/StorageObjectStorageCluster.cpp b/src/Storages/ObjectStorage/StorageObjectStorageCluster.cpp index b48b5b3b22c8..26acb15c7f68 100644 --- a/src/Storages/ObjectStorage/StorageObjectStorageCluster.cpp +++ b/src/Storages/ObjectStorage/StorageObjectStorageCluster.cpp @@ -37,15 +37,10 @@ namespace Setting extern const SettingsBool use_hive_partitioning; extern const SettingsBool cluster_function_process_archive_on_multiple_nodes; extern const SettingsObjectStorageGranularityLevel cluster_table_function_split_granularity; -<<<<<<< HEAD -======= extern const SettingsBool parallel_replicas_for_cluster_engines; extern const SettingsString object_storage_cluster; extern const SettingsInt64 delta_lake_snapshot_start_version; extern const SettingsInt64 delta_lake_snapshot_end_version; - extern const SettingsUInt64 lock_object_storage_task_distribution_ms; - extern const SettingsBool allow_experimental_iceberg_read_optimization; ->>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) } namespace ErrorCodes @@ -474,30 +469,7 @@ void StorageObjectStorageCluster::updateQueryToSendIfNeeded( } } -<<<<<<< HEAD - if (!endsWith(table_function->name, "Cluster")) - { - configuration->addStructureAndFormatToArgsIfNeeded(args, structure, configuration->format, context, /*with_structure=*/true); - - /// When a non-cluster table function (e.g. `s3`) was auto-converted to cluster mode - /// by the `parallel_replicas_for_cluster_engines` setting, rename it to the Cluster variant - /// (e.g. `s3Cluster`) and prepend the cluster name argument. This ensures that on the shard, - /// `TableFunctionObjectStorageCluster::executeImpl` is called, which correctly handles - /// `distributed_processing` for task-based file distribution from the initiator. - /// - /// Some table functions (e.g. `paimonLocal`, `deltaLakeLocal`) do not have a Cluster variant, - /// so we only rename when the target function actually exists. - const String cluster_function_name = table_function->name + "Cluster"; - if (TableFunctionFactory::instance().isTableFunctionName(cluster_function_name)) - { - args.insert(args.begin(), make_intrusive(getClusterName())); - table_function->name = cluster_function_name; - } - } - else -======= if (cluster_name_in_settings || !endsWith(table_function->name, "Cluster")) ->>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) { configuration->addStructureAndFormatToArgsIfNeeded(args, structure, configuration->getFormat(), context, /*with_structure=*/true); @@ -597,18 +569,14 @@ void StorageObjectStorageCluster::updateExternalDynamicMetadataIfExists(ContextP new_metadata = *metadata_snapshot; } -<<<<<<< HEAD setInMemoryMetadata(new_metadata.withVirtuals(VirtualColumnUtils::getVirtualsForFileLikeStorage( new_metadata.columns, query_context, /* format_settings */ std::nullopt, - configuration->partition_strategy_type))); -======= - setInMemoryMetadata(new_metadata); + configuration->getPartitionStrategyType()))); if (pure_storage) pure_storage->setInMemoryMetadata(IStorageCluster::getInMemoryMetadata()); ->>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) } RemoteQueryExecutor::Extension StorageObjectStorageCluster::getTaskIteratorExtension( diff --git a/src/Storages/ObjectStorage/StorageObjectStorageConfiguration.h b/src/Storages/ObjectStorage/StorageObjectStorageConfiguration.h index f62628e19844..fbcc1fde2111 100644 --- a/src/Storages/ObjectStorage/StorageObjectStorageConfiguration.h +++ b/src/Storages/ObjectStorage/StorageObjectStorageConfiguration.h @@ -340,7 +340,6 @@ class StorageObjectStorageConfiguration virtual void drop(ContextPtr) {} -<<<<<<< HEAD virtual bool isBackgroundExecutable() const { return false; @@ -358,11 +357,9 @@ class StorageObjectStorageConfiguration return 0; } -======= virtual bool isClusterSupported() const { return true; } private: ->>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) String format = "auto"; String compression_method = "auto"; String structure = "auto"; diff --git a/src/Storages/ObjectStorage/StorageObjectStorageSource.cpp b/src/Storages/ObjectStorage/StorageObjectStorageSource.cpp index 627c30e1ff4c..fa40e1fffe7b 100644 --- a/src/Storages/ObjectStorage/StorageObjectStorageSource.cpp +++ b/src/Storages/ObjectStorage/StorageObjectStorageSource.cpp @@ -746,7 +746,7 @@ StorageObjectStorageSource::ReaderHolder StorageObjectStorageSource::createReade continue; auto file_bucket_info = FormatFactory::instance().getFileBucketInfo( - object_info->getFileFormat().value_or(configuration->format)); + object_info->getFileFormat().value_or(configuration->getFormat())); if (file_bucket_info) { auto filtered = file_bucket_info->filterByMatchingRowGroups(matching_row_groups); @@ -802,13 +802,12 @@ StorageObjectStorageSource::ReaderHolder StorageObjectStorageSource::createReade } else { - const auto format_name = object_info->getFileFormat().value_or(configuration->format); + const auto format_name = object_info->getFileFormat().value_or(configuration->getFormat()); const bool input_format_does_not_read_file = Poco::toLower(format_name) == "one"; CompressionMethod compression_method = {}; if (input_format_does_not_read_file) { -<<<<<<< HEAD /// `One` produces a single row per object without consuming the underlying `ReadBuffer`. read_buf = std::make_unique(); compression_method = CompressionMethod::None; @@ -816,21 +815,14 @@ StorageObjectStorageSource::ReaderHolder StorageObjectStorageSource::createReade else if (const auto * object_info_in_archive = dynamic_cast(object_info.get())) { ProfileEvents::increment(ProfileEvents::ObjectStorageReadObjects); - compression_method = chooseCompressionMethod(configuration->getPathInArchive(), configuration->compression_method); -======= compression_method = chooseCompressionMethod(configuration->getPathInArchive(), configuration->getCompressionMethod()); ->>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) const auto & archive_reader = object_info_in_archive->archive_reader; read_buf = archive_reader->readFile(object_info_in_archive->path_in_archive, /*throw_on_not_found=*/true); } else { -<<<<<<< HEAD ProfileEvents::increment(ProfileEvents::ObjectStorageReadObjects); - compression_method = chooseCompressionMethod(object_info->getFileName(), configuration->compression_method); -======= compression_method = chooseCompressionMethod(object_info->getFileName(), configuration->getCompressionMethod()); ->>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) read_buf = createReadBuffer(object_info->relative_path_with_metadata, object_storage, context_, log); } @@ -861,7 +853,7 @@ StorageObjectStorageSource::ReaderHolder StorageObjectStorageSource::createReade /// tables (e.g. Iceberg with Parquet + ORC files), table-level PREWHERE support /// may not match the individual file's format capabilities. /// See https://github.com/ClickHouse/ClickHouse/issues/96829 - const auto actual_format = object_info->getFileFormat().value_or(configuration->format); + const auto actual_format = object_info->getFileFormat().value_or(configuration->getFormat()); const bool format_supports_prewhere = FormatFactory::instance().checkIfFormatSupportsPrewhere(actual_format, context_, format_settings); @@ -938,33 +930,20 @@ StorageObjectStorageSource::ReaderHolder StorageObjectStorageSource::createReade "Reading object '{}', size: {} bytes, with format: {}", object_info->getPath(), object_info->getObjectMetadata()->size_bytes, -<<<<<<< HEAD format_name); -======= - object_info->getFileFormat().value_or(configuration->getFormat())); ->>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) logIcebergFileStats(object_info, log); InputFormatPtr input_format; -<<<<<<< HEAD if (context_->getSettingsRef()[Setting::use_parquet_metadata_cache] && (Poco::toLower(format_name) == "parquet") -======= - if (context_->getSettingsRef()[Setting::use_parquet_metadata_cache] && use_native_reader_v3 - && (object_info->getFileFormat().value_or(configuration->getFormat()) == "Parquet") ->>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) && !object_info->getObjectMetadata()->etag.empty()) { std::optional object_with_metadata = object_info->relative_path_with_metadata; if (object_info->isArchive()) object_with_metadata->relative_path = object_info->getPath(); input_format = FormatFactory::instance().getInputWithMetadata( -<<<<<<< HEAD format_name, -======= - object_info->getFileFormat().value_or(configuration->getFormat()), ->>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) *read_buf, initial_header, context_, @@ -983,11 +962,7 @@ StorageObjectStorageSource::ReaderHolder StorageObjectStorageSource::createReade else { input_format = FormatFactory::instance().getInput( -<<<<<<< HEAD format_name, -======= - object_info->getFileFormat().value_or(configuration->getFormat()), ->>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) *read_buf, initial_header, context_, diff --git a/src/Storages/ObjectStorage/registerStorageObjectStorage.cpp b/src/Storages/ObjectStorage/registerStorageObjectStorage.cpp index 2636e201fcfa..2688f1bae1a6 100644 --- a/src/Storages/ObjectStorage/registerStorageObjectStorage.cpp +++ b/src/Storages/ObjectStorage/registerStorageObjectStorage.cpp @@ -14,11 +14,8 @@ #include #include #include -<<<<<<< HEAD #include -======= #include ->>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) #include #include #include diff --git a/src/Storages/System/StorageSystemTables.cpp b/src/Storages/System/StorageSystemTables.cpp index 6657fec7da52..b87f1ce9f1d0 100644 --- a/src/Storages/System/StorageSystemTables.cpp +++ b/src/Storages/System/StorageSystemTables.cpp @@ -27,12 +27,6 @@ #include #include #include -<<<<<<< HEAD -======= -#include -#include -#include ->>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) #include #include #include @@ -704,107 +698,18 @@ class TablesBlockSource final : public ISource ASTPtr expression_ptr; if (columns_mask[src_index++]) { -<<<<<<< HEAD if (metadata_snapshot && (expression_ptr = metadata_snapshot->getPartitionKeyAST())) res_columns[res_index++]->insert(format({context, *expression_ptr})); else res_columns[res_index++]->insertDefault(); -======= - bool inserted = false; - - try - { - // Extract from specific DataLake metadata if suitable - if (auto * obj = dynamic_cast(table.get())) - { - if (auto * dl_meta = obj->getExternalMetadata(context)) - { - if (auto p = dl_meta->partitionKey(context); p.has_value()) - { - res_columns[res_index++]->insert(*p); - inserted = true; - } - } - } - else if (auto * clobj = dynamic_cast(table.get())) - { - if (auto * dl_meta = clobj->getExternalMetadata(context)) - { - if (auto p = dl_meta->partitionKey(context); p.has_value()) - { - res_columns[res_index++]->insert(*p); - inserted = true; - } - } - - } - } - catch (const Exception &) - { - /// Failed to get info. It's not critical, just log it. - tryLogCurrentException("StorageSystemTables"); - } - - if (!inserted) - { - if (metadata_snapshot && (expression_ptr = metadata_snapshot->getPartitionKeyAST())) - res_columns[res_index++]->insert(format({context, *expression_ptr})); - else - res_columns[res_index++]->insertDefault(); - } ->>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) } if (columns_mask[src_index++]) { -<<<<<<< HEAD if (metadata_snapshot && (expression_ptr = metadata_snapshot->getSortingKey().expression_list_ast)) res_columns[res_index++]->insert(format({context, *expression_ptr})); else res_columns[res_index++]->insertDefault(); -======= - bool inserted = false; - - try - { - // Extract from specific DataLake metadata if suitable - if (auto * obj = dynamic_cast(table.get())) - { - if (auto * dl_meta = obj->getExternalMetadata(context)) - { - if (auto p = dl_meta->sortingKey(context); p.has_value()) - { - res_columns[res_index++]->insert(*p); - inserted = true; - } - } - } - else if (auto * clobj = dynamic_cast(table.get())) - { - if (auto * dl_meta = clobj->getExternalMetadata(context)) - { - if (auto p = dl_meta->sortingKey(context); p.has_value()) - { - res_columns[res_index++]->insert(*p); - inserted = true; - } - } - } - } - catch (const Exception &) - { - /// Failed to get info. It's not critical, just log it. - tryLogCurrentException("StorageSystemTables"); - } - - if (!inserted) - { - if (metadata_snapshot && (expression_ptr = metadata_snapshot->getSortingKey().expression_list_ast)) - res_columns[res_index++]->insert(format({context, *expression_ptr})); - else - res_columns[res_index++]->insertDefault(); - } ->>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) } if (columns_mask[src_index++]) diff --git a/src/TableFunctions/TableFunctionObjectStorage.cpp b/src/TableFunctions/TableFunctionObjectStorage.cpp index 4b3c985ed03c..8af35a52fb8a 100644 --- a/src/TableFunctions/TableFunctionObjectStorage.cpp +++ b/src/TableFunctions/TableFunctionObjectStorage.cpp @@ -415,48 +415,6 @@ template class TableFunctionObjectStorage; #endif -<<<<<<< HEAD -#if USE_AVRO -void registerTableFunctionIceberg(TableFunctionFactory & factory); -void registerTableFunctionIceberg(TableFunctionFactory & factory) -{ -#if USE_AWS_S3 - factory.registerFunction( - {.description = R"(The table function can be used to read the Iceberg table stored on S3 object store. Alias to icebergS3)", - .examples{{IcebergDefinition::name, "SELECT * FROM iceberg(url, access_key_id, secret_access_key)", ""}}, - .category = FunctionDocumentation::Category::TableFunction}, - {.allow_readonly = false}); - factory.registerFunction( - {.description = R"(The table function can be used to read the Iceberg table stored on S3 object store.)", - .examples{{IcebergS3Definition::name, "SELECT * FROM icebergS3(url, access_key_id, secret_access_key)", ""}}, - .category = FunctionDocumentation::Category::TableFunction}, - {.allow_readonly = false}); - -#endif -#if USE_AZURE_BLOB_STORAGE - factory.registerFunction( - {.description = R"(The table function can be used to read the Iceberg table stored on Azure object store.)", - .examples{{IcebergAzureDefinition::name, "SELECT * FROM icebergAzure(url, access_key_id, secret_access_key)", ""}}, - .category = FunctionDocumentation::Category::TableFunction}, - {.allow_readonly = false}); -#endif -#if USE_HDFS - factory.registerFunction( - {.description = R"(The table function can be used to read the Iceberg table stored on HDFS virtual filesystem.)", - .examples{{IcebergHDFSDefinition::name, "SELECT * FROM icebergHDFS(url)", ""}}, - .category = FunctionDocumentation::Category::TableFunction}, - {.allow_readonly = false}); -#endif - factory.registerFunction( - {.description = R"(The table function can be used to read the Iceberg table stored locally.)", - .examples{{IcebergLocalDefinition::name, "SELECT * FROM icebergLocal(filename)", ""}}, - .category = FunctionDocumentation::Category::TableFunction}, - {.allow_readonly = false}); -} -#endif - -======= ->>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) #if USE_AVRO void registerTableFunctionPaimon(TableFunctionFactory & factory); @@ -510,21 +468,6 @@ void registerTableFunctionDeltaLake(TableFunctionFactory & factory) } #endif -<<<<<<< HEAD -#if USE_AWS_S3 -void registerTableFunctionHudi(TableFunctionFactory & factory); -void registerTableFunctionHudi(TableFunctionFactory & factory) -{ - factory.registerFunction( - {.description = R"(The table function can be used to read the Hudi table stored on object store.)", - .examples{{HudiDefinition::name, "SELECT * FROM hudi(url, access_key_id, secret_access_key)", ""}}, - .category = FunctionDocumentation::Category::TableFunction}, - {.allow_readonly = false}); -} -#endif - -======= ->>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) void registerDataLakeTableFunctions(TableFunctionFactory & factory) { UNUSED(factory); diff --git a/tests/integration/compose/docker_compose_iceberg_rest_catalog.yml b/tests/integration/compose/docker_compose_iceberg_rest_catalog.yml index c993bf122a5d..f584a954da8c 100644 --- a/tests/integration/compose/docker_compose_iceberg_rest_catalog.yml +++ b/tests/integration/compose/docker_compose_iceberg_rest_catalog.yml @@ -1,33 +1,8 @@ services: -<<<<<<< HEAD rest: image: tabulario/iceberg-rest:1.6.0 ports: - "${ICEBERG_REST_CATALOG_PORT}:8181" -======= - spark-iceberg: - image: tabulario/spark-iceberg:3.5.5_1.8.1 - build: spark/ - depends_on: - rest: - condition: service_healthy - minio: - condition: service_started - environment: - - AWS_ACCESS_KEY_ID=admin - - AWS_SECRET_ACCESS_KEY=password - - AWS_REGION=us-east-1 - ports: - - ${SPARK_ICEBERG_EXTERNAL_PORT:-8080}:8080 - - ${SPARK_ICEBERG_EXTERNAL_PORT_2:-10002}:10000 - - ${SPARK_ICEBERG_EXTERNAL_PORT_3:-10003}:10001 - stop_grace_period: 5s - cpus: 3 - rest: - image: tabulario/iceberg-rest:1.6.0 - ports: - - ${ICEBERG_REST_EXTERNAL_PORT:-8182}:8181 ->>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) environment: - AWS_ACCESS_KEY_ID=minio - AWS_SECRET_ACCESS_KEY=ClickHouse_Minio_P@ssw0rd diff --git a/tests/integration/helpers/iceberg_utils.py b/tests/integration/helpers/iceberg_utils.py index 136689e3bcec..62975f02b370 100644 --- a/tests/integration/helpers/iceberg_utils.py +++ b/tests/integration/helpers/iceberg_utils.py @@ -447,54 +447,14 @@ def create_iceberg_table( run_on_cluster=False, format="Parquet", order_by="", -<<<<<<< HEAD settings=None, + object_storage_cluster=False, **kwargs, ): node.query( - get_creation_expression(storage_type, table_name, cluster, schema, format_version, partition_by, if_not_exists, compression_method, format, order_by, run_on_cluster=run_on_cluster, **kwargs), + get_creation_expression(storage_type, table_name, cluster, schema, format_version, partition_by, if_not_exists, compression_method, format, order_by, run_on_cluster=run_on_cluster, object_storage_cluster=object_storage_cluster, **kwargs), settings=settings, ) -======= - object_storage_cluster=False, - **kwargs, -): - if 'output_format_parquet_use_custom_encoder' in kwargs: - node.query( - get_creation_expression( - storage_type, - table_name, - cluster, - schema, - format_version, - partition_by, - if_not_exists, - compression_method, - format, - order_by, - run_on_cluster=run_on_cluster, - object_storage_cluster=object_storage_cluster, - **kwargs), - settings={"output_format_parquet_use_custom_encoder" : 0, "output_format_parquet_parallel_encoding" : 0} - ) - else: - node.query( - get_creation_expression( - storage_type, - table_name, - cluster, - schema, - format_version, - partition_by, - if_not_exists, - compression_method, - format, - order_by, - run_on_cluster=run_on_cluster, - object_storage_cluster=object_storage_cluster, - **kwargs), - ) ->>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) def drop_iceberg_table( diff --git a/tests/integration/test_database_delta/test.py b/tests/integration/test_database_delta/test.py index 0385f458bb26..d4c274972df6 100644 --- a/tests/integration/test_database_delta/test.py +++ b/tests/integration/test_database_delta/test.py @@ -956,7 +956,6 @@ def get_table_versions(): ) -<<<<<<< HEAD @pytest.mark.parametrize("use_delta_kernel", ["1", "0"]) def test_varchar_char_types_via_unity_catalog(started_cluster, use_delta_kernel): """ @@ -1030,7 +1029,8 @@ def test_varchar_char_types_via_unity_catalog(started_cluster, use_delta_kernel) .strip() ) assert row == "1\thello varchar\thello char" -======= + + def test_namespace_filter(started_cluster): node = started_cluster.instances["node1"] @@ -1068,4 +1068,3 @@ def create_namespace(suffix): assert node.query(f"SELECT count() FROM {CATALOG_NAME}.`{namespace_prefix}alpha.{table_name}`") == "0\n" assert "is filtered by `namespaces` database parameter." in node.query_and_get_error(f"SELECT count() FROM {CATALOG_NAME}.`{namespace_prefix}bravo.{table_name}`") ->>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) diff --git a/tests/integration/test_database_glue/test.py b/tests/integration/test_database_glue/test.py index b0a40879f302..1018b0d3193d 100644 --- a/tests/integration/test_database_glue/test.py +++ b/tests/integration/test_database_glue/test.py @@ -1069,7 +1069,6 @@ def test_table_without_metadata_location(started_cluster): node.query(f"DROP DATABASE IF EXISTS {db_name} SYNC") -<<<<<<< HEAD def test_check_database(started_cluster): """Test that CHECK DATABASE works with Glue catalog database.""" node = started_cluster.instances["node1"] @@ -1131,7 +1130,7 @@ def test_check_database(started_cluster): "SYSTEM DISABLE FAILPOINT check_database_datalake_negative" ) -======= + def test_namespace_filter(started_cluster): node = started_cluster.instances["node1"] @@ -1169,8 +1168,7 @@ def create_namespace(suffix): node.query(f"DROP TABLE {CATALOG_NAME}.`{namespace_prefix}alpha.{table_name}`") assert "is filtered by `namespaces` database parameter." in node.query_and_get_error(f"DROP TABLE {CATALOG_NAME}.`{namespace_prefix}bravo.{table_name}`") - ->>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) + def test_sts_smoke(started_cluster): """Test that STS authentication works with Glue catalog using role_arn and role_session_name""" node = started_cluster.instances["node1"] diff --git a/tests/integration/test_database_iceberg/test.py b/tests/integration/test_database_iceberg/test.py index 88ea5ae0b204..7c2716dac320 100644 --- a/tests/integration/test_database_iceberg/test.py +++ b/tests/integration/test_database_iceberg/test.py @@ -133,13 +133,7 @@ def create_clickhouse_iceberg_database( node.query( f""" DROP DATABASE IF EXISTS {name}; -<<<<<<< HEAD -CREATE DATABASE {name} ENGINE = DataLakeCatalog('{BASE_URL}', 'minio', '{minio_secret_key}') -======= -SET allow_database_iceberg=true; -SET write_full_path_in_iceberg_metadata=1; CREATE DATABASE {name} ENGINE = {engine}('{BASE_URL}', 'minio', '{minio_secret_key}') ->>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) SETTINGS {",".join((k+"="+repr(v) for k, v in settings.items()))} """, settings={ @@ -291,7 +285,6 @@ def test_list_tables(started_cluster, engine): ) -<<<<<<< HEAD def test_check_database(started_cluster): node = started_cluster.instances["node1"] @@ -356,11 +349,8 @@ def test_check_database(started_cluster): ) -def test_many_namespaces(started_cluster): -======= @pytest.mark.parametrize("engine", AVAILABLE_ENGINES) def test_many_namespaces(started_cluster, engine): ->>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) node = started_cluster.instances["node1"] root_namespace_1 = f"A_{uuid.uuid4()}" root_namespace_2 = f"B_{uuid.uuid4()}" @@ -450,7 +440,6 @@ def test_hide_sensitive_info(started_cluster, engine): create_table(catalog, namespace, table_name) -<<<<<<< HEAD def check_secret_hidden(secret, additional_settings): settings = { "catalog_type": "rest", @@ -462,7 +451,7 @@ def check_secret_hidden(secret, additional_settings): node.query(f"DROP DATABASE IF EXISTS {CATALOG_NAME}") try: node.query( - f"""CREATE DATABASE {CATALOG_NAME} ENGINE = DataLakeCatalog('{BASE_URL}', 'minio', '{minio_secret_key}') + f"""CREATE DATABASE {CATALOG_NAME} ENGINE = {engine}('{BASE_URL}', 'minio', '{minio_secret_key}') SETTINGS {",".join((k + "=" + repr(v) for k, v in settings.items()))}""", settings={ "allow_database_iceberg": 1, @@ -506,23 +495,6 @@ def test_no_secrets_in_logs(started_cluster): "allow_database_iceberg": 1, "write_full_path_in_iceberg_metadata": 1, }, -======= - create_clickhouse_iceberg_database( - started_cluster, - node, - CATALOG_NAME, - additional_settings={"catalog_credential": "SECRET_1"}, - engine=engine, - ) - assert "SECRET_1" not in node.query(f"SHOW CREATE DATABASE {CATALOG_NAME}") - - create_clickhouse_iceberg_database( - started_cluster, - node, - CATALOG_NAME, - additional_settings={"auth_header": "SECRET_2"}, - engine=engine, ->>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) ) qid_table = uuid.uuid4().hex @@ -1227,7 +1199,6 @@ def test_gcs(started_cluster): assert "Google cloud storage converts to S3" in str(err.value) -<<<<<<< HEAD def test_invalid_auth_header_format(started_cluster): node = started_cluster.instances["node1"] @@ -1325,9 +1296,6 @@ def test_iceberg_file_progress_callback(started_cluster): ) -# TODO - turn on after merge alternative syntax -def _test_cluster_joins(started_cluster): -======= def test_namespace_filter(started_cluster): node = started_cluster.instances["node1"] @@ -1394,8 +1362,8 @@ def create_namespace(suffix): assert "is filtered by `namespaces` database parameter." in node.query_and_get_error(f"DROP TABLE {CATALOG_NAME}.`{namespace_prefix}alpha.a2.{table_name}`") -def test_cluster_joins(started_cluster): ->>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) +# TODO - turn on after merge alternative syntax +def _test_cluster_joins(started_cluster): node = started_cluster.instances["node1"] test_ref = f"test_join_tables_{uuid.uuid4()}" diff --git a/tests/integration/test_mask_sensitive_info/test.py b/tests/integration/test_mask_sensitive_info/test.py index b160f9fed188..8161024c87c1 100644 --- a/tests/integration/test_mask_sensitive_info/test.py +++ b/tests/integration/test_mask_sensitive_info/test.py @@ -294,15 +294,7 @@ def test_create_table(): f"AzureQueue('{azure_conn_string}', 'cont', '*', 'CSV') SETTINGS mode = 'unordered', after_processing = 'move', after_processing_move_connection_string = '{azure_sas_conn_string}', after_processing_move_container = 'chprocessed'", f"AzureQueue('{azure_storage_account_url}', 'cont', '*', '{azure_account_name}', '{azure_account_key}', 'CSV') SETTINGS mode = 'unordered'", f"AzureQueue('{azure_storage_account_url}', 'cont', '*', '{azure_account_name}', '{azure_account_key}', 'CSV', 'none') SETTINGS mode = 'unordered'", -<<<<<<< HEAD "AzureBlobStorage('BlobEndpoint=https://my-endpoint/;SharedAccessSignature=sp=r&st=2025-09-29T14:58:11Z&se=2025-09-29T00:00:00Z&spr=https&sv=2022-11-02&sr=c&sig=SECRET%SECRET%SECRET%SECRET', 'exampledatasets', 'example.csv')", - f"S3('https://my-s3-endpoint/bucket/data.csv', 'myaccess', '{password}', 'CSV')", -======= - ( - f"AzureBlobStorage('BlobEndpoint=https://my-endpoint/;SharedAccessSignature=sp=r&st=2025-09-29T14:58:11Z&se=2025-09-29T00:00:00Z&spr=https&sv=2022-11-02&sr=c&sig=SECRET%SECRET%SECRET%SECRET', 'exampledatasets', 'example.csv')", - "STD_EXCEPTION", - ), - f"AzureBlobStorage(named_collection_2, connection_string = '{azure_conn_string}', container = 'cont', blob_path = 'test_simple_7.csv', format = 'CSV')", f"AzureBlobStorage(named_collection_2, storage_account_url = '{azure_storage_account_url}', container = 'cont', blob_path = 'test_simple_8.csv', account_name = '{azure_account_name}', account_key = '{azure_account_key}')", f"AzureBlobStorage('{azure_storage_account_url}', 'cont', 'test_simple_3.csv', '{azure_account_name}', '{azure_account_key}')", @@ -316,8 +308,7 @@ def test_create_table(): f"Iceberg(storage_type='azure', '{azure_storage_account_url}', 'cont', 'test_simple_5_{table_suffix}.csv', '{azure_account_name}', '{azure_account_key}')", f"Iceberg(storage_type='azure', named_collection_2, connection_string = '{azure_conn_string}', container = 'cont', blob_path = 'test_simple_6_{table_suffix}.csv', format = 'CSV')", f"Iceberg(storage_type='azure', named_collection_2, storage_account_url = '{azure_storage_account_url}', container = 'cont', blob_path = 'test_simple_7_{table_suffix}.csv', account_name = '{azure_account_name}', account_key = '{azure_account_key}')", - ->>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) + f"S3('https://my-s3-endpoint/bucket/data.csv', 'myaccess', '{password}', 'CSV')", f"Kafka() SETTINGS kafka_broker_list = '127.0.0.1', kafka_topic_list = 'topic', kafka_group_name = 'group', kafka_format = 'JSONEachRow', kafka_security_protocol = 'sasl_ssl', kafka_sasl_mechanism = 'PLAIN', kafka_sasl_username = 'user', kafka_sasl_password = '{password}', format_avro_schema_registry_url = 'http://schema_user:{password}@'", f"Kafka() SETTINGS kafka_broker_list = '127.0.0.1', kafka_topic_list = 'topic', kafka_group_name = 'group', kafka_format = 'JSONEachRow', kafka_security_protocol = 'sasl_ssl', kafka_sasl_mechanism = 'PLAIN', kafka_sasl_username = 'user', kafka_sasl_password = '{password}', format_avro_schema_registry_url = 'http://schema_user:{password}@domain.com'", f"S3('http://minio1:9001/root/data/test5.csv.gz', 'CSV', access_key_id = 'minio', secret_access_key = '{password}', compression_method = 'gzip')", @@ -420,9 +411,6 @@ def generate_create_table_numbered(tail): generate_create_table_numbered(f"(`x` int) ENGINE = AzureQueue('{azure_storage_account_url}', 'cont', '*', '{azure_account_name}', '[HIDDEN]', 'CSV') SETTINGS mode = 'unordered'"), generate_create_table_numbered(f"(`x` int) ENGINE = AzureQueue('{azure_storage_account_url}', 'cont', '*', '{azure_account_name}', '[HIDDEN]', 'CSV', 'none') SETTINGS mode = 'unordered'"), generate_create_table_numbered(f"(`x` int) ENGINE = AzureBlobStorage('{masked_sas_conn_string}', 'exampledatasets', 'example.csv')"), -<<<<<<< HEAD - generate_create_table_numbered("(`x` int) ENGINE = S3('https://my-s3-endpoint/bucket/data.csv', 'myaccess', '[HIDDEN]', 'CSV')"), -======= generate_create_table_numbered(f"(`x` int) ENGINE = AzureBlobStorage(named_collection_2, connection_string = '{masked_azure_conn_string}', container = 'cont', blob_path = 'test_simple_7.csv', format = 'CSV')"), generate_create_table_numbered(f"(`x` int) ENGINE = AzureBlobStorage(named_collection_2, storage_account_url = '{azure_storage_account_url}', container = 'cont', blob_path = 'test_simple_8.csv', account_name = '{azure_account_name}', account_key = '[HIDDEN]')"), generate_create_table_numbered(f"(`x` int) ENGINE = AzureBlobStorage('{azure_storage_account_url}', 'cont', 'test_simple_3.csv', '{azure_account_name}', '[HIDDEN]')"), @@ -436,7 +424,7 @@ def generate_create_table_numbered(tail): generate_create_table_numbered(f"(`x` int) ENGINE = Iceberg(storage_type = 'azure', '{azure_storage_account_url}', 'cont', 'test_simple_5_{table_suffix}.csv', '{azure_account_name}', '[HIDDEN]')"), generate_create_table_numbered(f"(`x` int) ENGINE = Iceberg(storage_type = 'azure', named_collection_2, connection_string = '{masked_azure_conn_string}', container = 'cont', blob_path = 'test_simple_6_{table_suffix}.csv', format = 'CSV')"), generate_create_table_numbered(f"(`x` int) ENGINE = Iceberg(storage_type = 'azure', named_collection_2, storage_account_url = '{azure_storage_account_url}', container = 'cont', blob_path = 'test_simple_7_{table_suffix}.csv', account_name = '{azure_account_name}', account_key = '[HIDDEN]')"), ->>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) + generate_create_table_numbered("(`x` int) ENGINE = S3('https://my-s3-endpoint/bucket/data.csv', 'myaccess', '[HIDDEN]', 'CSV')"), generate_create_table_numbered("(`x` int) ENGINE = Kafka SETTINGS kafka_broker_list = '127.0.0.1', kafka_topic_list = 'topic', kafka_group_name = 'group', kafka_format = 'JSONEachRow', kafka_security_protocol = 'sasl_ssl', kafka_sasl_mechanism = 'PLAIN', kafka_sasl_username = 'user', kafka_sasl_password = '[HIDDEN]', format_avro_schema_registry_url = 'http://schema_user:[HIDDEN]@'"), generate_create_table_numbered("(`x` int) ENGINE = Kafka SETTINGS kafka_broker_list = '127.0.0.1', kafka_topic_list = 'topic', kafka_group_name = 'group', kafka_format = 'JSONEachRow', kafka_security_protocol = 'sasl_ssl', kafka_sasl_mechanism = 'PLAIN', kafka_sasl_username = 'user', kafka_sasl_password = '[HIDDEN]', format_avro_schema_registry_url = 'http://schema_user:[HIDDEN]@domain.com'"), generate_create_table_numbered("(`x` int) ENGINE = S3('http://minio1:9001/root/data/test5.csv.gz', 'CSV', access_key_id = 'minio', secret_access_key = '[HIDDEN]', compression_method = 'gzip')"), @@ -698,30 +686,6 @@ def make_test_case(i): "CREATE TABLE tablefunc39 (`x` int) AS iceberg('http://minio1:9001/root/data/test11.csv.gz', 'minio', '[HIDDEN]')", "CREATE TABLE tablefunc40 (`x` int) AS iceberg(named_collection_2, url = 'http://minio1:9001/root/data/test4.csv', access_key_id = 'minio', secret_access_key = '[HIDDEN]')", "CREATE TABLE tablefunc41 (`x` int) AS icebergS3('http://minio1:9001/root/data/test11.csv.gz', 'minio', '[HIDDEN]')", -<<<<<<< HEAD - f"CREATE TABLE tablefunc42 (`x` int) AS icebergAzure('{azure_storage_account_url}', 'cont', 'test_simple_6.csv', '{azure_account_name}', '[HIDDEN]', 'CSV', 'none', 'auto')", - f"CREATE TABLE tablefunc43 (`x` int) AS deltaLakeAzure('{azure_storage_account_url}', 'cont', 'test_simple_6.csv', '{azure_account_name}', '[HIDDEN]', 'CSV', 'none', 'auto')", - "CREATE TABLE tablefunc44 (`x` int) AS hudi('http://minio1:9001/root/data/test7.csv', 'minio', '[HIDDEN]')", - "CREATE TABLE tablefunc45 (`x` int) AS arrowFlight('arrowflight1:5006', 'dataset', 'arrowflight_user', '[HIDDEN]')", - "CREATE TABLE tablefunc46 (`x` int) AS arrowFlight(named_collection_1, host = 'arrowflight1', port = 5006, dataset = 'dataset', username = 'arrowflight_user', password = '[HIDDEN]')", - "CREATE TABLE tablefunc47 (`x` int) AS arrowflight(named_collection_1, host = 'arrowflight1', port = 5006, dataset = 'dataset', username = 'arrowflight_user', password = '[HIDDEN]')", - "CREATE TABLE tablefunc48 (`x` int) AS url('https://username:[HIDDEN]@domain.com/path', 'CSV')", - "CREATE TABLE tablefunc49 (`x` int) AS redis('localhost', 'key', 'key Int64', 0, '[HIDDEN]')", - "CREATE TABLE tablefunc50 (`x` int) AS jdbc('[HIDDEN]', 'mydb', 'mytable')", - "CREATE TABLE tablefunc51 (`x` int) AS odbc('[HIDDEN]', 'mydb', 'mytable')", - "CREATE TABLE tablefunc52 (`x` int) AS jdbc('jdbc://user:[HIDDEN]@localhost:5432/mydb', 'mydb', 'mytable')", - "CREATE TABLE tablefunc53 (`x` int) AS odbc('odbc://user:[HIDDEN]@localhost:5432/mydb', 'mydb', 'mytable')", - "CREATE TABLE tablefunc54 (`x` int) AS jdbc(named_collection_1, datasource = '[HIDDEN]')", - "CREATE TABLE tablefunc55 (`x` int) AS odbc(named_collection_1, connection_settings = '[HIDDEN]')", - "CREATE TABLE tablefunc56 (`x` int) AS jdbc(named_collection_1, datasource = 'jdbc://user:[HIDDEN]@localhost:5432/mydb')", - "CREATE TABLE tablefunc57 (`x` int) AS odbc(named_collection_1, connection_settings = 'odbc://user:[HIDDEN]@localhost:5432/mydb')", - "CREATE TABLE tablefunc58 (`x` int) AS jdbc(named_collection_1, datasource = '[HIDDEN]', connection_settings = '[HIDDEN]')", - "CREATE TABLE tablefunc59 (`x` int) AS jdbc(named_collection_1, connection_settings = '[HIDDEN]', external_database = '[HIDDEN]', datasource = '[HIDDEN]')", - "CREATE TABLE tablefunc60 (`x` int) AS deltaLakeS3('http://minio1:9001/root/data/test11.csv.gz', 'minio', '[HIDDEN]')", - "CREATE TABLE tablefunc61 (`x` int) AS paimon('http://minio1:9001/root/data/test11.csv.gz', 'minio', '[HIDDEN]')", - "CREATE TABLE tablefunc62 (`x` int) AS paimonS3('http://minio1:9001/root/data/test11.csv.gz', 'minio', '[HIDDEN]')", - f"CREATE TABLE tablefunc63 (`x` int) AS paimonAzure('{azure_storage_account_url}', 'cont', 'test_simple_6.csv', '{azure_account_name}', '[HIDDEN]', 'CSV', 'none', 'auto')", -======= "CREATE TABLE tablefunc42 (`x` int) AS icebergS3(named_collection_2, url = 'http://minio1:9001/root/data/test4.csv', access_key_id = 'minio', secret_access_key = '[HIDDEN]')", f"CREATE TABLE tablefunc43 (`x` int) AS icebergAzure('{masked_azure_conn_string}', 'cont', 'test_simple.csv')", f"CREATE TABLE tablefunc44 (`x` int) AS icebergAzure('{azure_storage_account_url}', 'cont', 'test_simple.csv', '{azure_account_name}', '[HIDDEN]')", @@ -743,7 +707,20 @@ def make_test_case(i): "CREATE TABLE tablefunc60 (`x` int) AS arrowflight(named_collection_1, host = 'arrowflight1', port = 5006, dataset = 'dataset', username = 'arrowflight_user', password = '[HIDDEN]')", "CREATE TABLE tablefunc61 (`x` int) AS url('https://username:[HIDDEN]@domain.com/path', 'CSV')", "CREATE TABLE tablefunc62 (`x` int) AS redis('localhost', 'key', 'key Int64', 0, '[HIDDEN]')", ->>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) + "CREATE TABLE tablefunc63 (`x` int) AS jdbc('[HIDDEN]', 'mydb', 'mytable')", + "CREATE TABLE tablefunc64 (`x` int) AS odbc('[HIDDEN]', 'mydb', 'mytable')", + "CREATE TABLE tablefunc65 (`x` int) AS jdbc('jdbc://user:[HIDDEN]@localhost:5432/mydb', 'mydb', 'mytable')", + "CREATE TABLE tablefunc66 (`x` int) AS odbc('odbc://user:[HIDDEN]@localhost:5432/mydb', 'mydb', 'mytable')", + "CREATE TABLE tablefunc67 (`x` int) AS jdbc(named_collection_1, datasource = '[HIDDEN]')", + "CREATE TABLE tablefunc68 (`x` int) AS odbc(named_collection_1, connection_settings = '[HIDDEN]')", + "CREATE TABLE tablefunc69 (`x` int) AS jdbc(named_collection_1, datasource = 'jdbc://user:[HIDDEN]@localhost:5432/mydb')", + "CREATE TABLE tablefunc70 (`x` int) AS odbc(named_collection_1, connection_settings = 'odbc://user:[HIDDEN]@localhost:5432/mydb')", + "CREATE TABLE tablefunc71 (`x` int) AS jdbc(named_collection_1, datasource = '[HIDDEN]', connection_settings = '[HIDDEN]')", + "CREATE TABLE tablefunc72 (`x` int) AS jdbc(named_collection_1, connection_settings = '[HIDDEN]', external_database = '[HIDDEN]', datasource = '[HIDDEN]')", + "CREATE TABLE tablefunc73 (`x` int) AS deltaLakeS3('http://minio1:9001/root/data/test11.csv.gz', 'minio', '[HIDDEN]')", + "CREATE TABLE tablefunc74 (`x` int) AS paimon('http://minio1:9001/root/data/test11.csv.gz', 'minio', '[HIDDEN]')", + "CREATE TABLE tablefunc75 (`x` int) AS paimonS3('http://minio1:9001/root/data/test11.csv.gz', 'minio', '[HIDDEN]')", + f"CREATE TABLE tablefunc76 (`x` int) AS paimonAzure('{azure_storage_account_url}', 'cont', 'test_simple_6.csv', '{azure_account_name}', '[HIDDEN]', 'CSV', 'none', 'auto')", ], must_not_contain=[password], ) diff --git a/tests/integration/test_s3_cluster/test.py b/tests/integration/test_s3_cluster/test.py index 7c021a5384a7..bf0c51a0cff7 100644 --- a/tests/integration/test_s3_cluster/test.py +++ b/tests/integration/test_s3_cluster/test.py @@ -1352,7 +1352,6 @@ def test_hive_partitioning(started_cluster, allow_experimental_analyzer, use_par cluster_full_traffic = int(cluster_full_traffic) assert cluster_full_traffic == full_traffic -<<<<<<< HEAD cluster_optimized_traffic = node.query( f""" SELECT sum(ProfileEvents['{event}']) @@ -1434,8 +1433,6 @@ def test_iceberg_s3_cluster_read_task_failpoint(started_cluster): ) node.query(f"DROP TABLE IF EXISTS {dst_table}") node.query(f"DROP TABLE IF EXISTS {iceberg_table}") -======= - assert errors == 0 def test_object_storage_remote_initiator_without_cluster_function(started_cluster): @@ -1532,4 +1529,3 @@ def test_object_storage_remote_initiator_without_cluster_function(started_cluste assert users[1:] == ["s0_0_0\tdefault", "s0_0_1\tfoo", "s0_1_0\tfoo"] ->>>>>>> ff71e89ea9e (Merge pull request #1640 from Altinity/frontport/antalya-26.3/alternative_syntax) From e8c2701587d2684d8808e26f320de8b7f2dbdb9a Mon Sep 17 00:00:00 2001 From: Andrey Zvonov <32552679+zvonand@users.noreply.github.com> Date: Sun, 10 May 2026 16:25:51 +0200 Subject: [PATCH 07/43] Merge pull request #1744 from Altinity/feature/antalya-26.3/pr-1564 Antalya 26.3: Fix export task not being killed during s3 outage Source-PR: #1744 (https://github.com/Altinity/ClickHouse/pull/1744) --- src/Common/ThreadStatus.h | 10 +++++++++ src/Storages/MergeTree/ExportPartTask.cpp | 26 +++++++++++++++++++++-- src/Storages/MergeTree/ExportPartTask.h | 2 +- 3 files changed, 35 insertions(+), 3 deletions(-) diff --git a/src/Common/ThreadStatus.h b/src/Common/ThreadStatus.h index 975349e2dde5..42ace6a4f088 100644 --- a/src/Common/ThreadStatus.h +++ b/src/Common/ThreadStatus.h @@ -118,6 +118,16 @@ class ThreadGroup void attachQueryForLog(const String & query_, UInt64 normalized_hash = 0); void attachInternalProfileEventsQueue(const InternalProfileEventsQueuePtr & profile_queue); + /// Override the cancellation predicate. All threads that subsequently attach to this + /// group via ThreadGroupSwitcher inherit the predicate in their local_data, making + /// isQueryCanceled() reflect task-level cancellation without a process-list entry. + /// Required for part and partition export cancellation during S3 outage. + void setCancelPredicate(QueryIsCanceledPredicate predicate) + { + std::lock_guard lock(mutex); + shared_data.query_is_canceled_predicate = std::move(predicate); + } + /// When new query starts, new thread group is created for it, current thread becomes master thread of the query static ThreadGroupPtr createForQuery(ContextPtr query_context_, FatalErrorCallback fatal_error_callback_ = {}); diff --git a/src/Storages/MergeTree/ExportPartTask.cpp b/src/Storages/MergeTree/ExportPartTask.cpp index bb6fc1491afa..9f1d4a773b17 100644 --- a/src/Storages/MergeTree/ExportPartTask.cpp +++ b/src/Storages/MergeTree/ExportPartTask.cpp @@ -175,6 +175,28 @@ bool ExportPartTask::executeStep() manifest.query_id, local_context); + /* + This is a hack to fix the issue where S3 is out, ClickHouse keeps retrying S3 requests deep + in the AWS SDK and never check for the `isCancelled()` flag. That prevents the task from being killed / cancelled. It also prevents the table from being dropped. + + Merges and mutations don't suffer from this problem because they don't make requests to S3 :). Select statements + do make requests to S3, but the cancel predicate is properly setup for regular queries. + + I think this is the first time we have a background operation that makes requests to S3, so we need to connect the dots. + + The simples way is this one, and given the release timeline, I am opting for it. + */ + (*exports_list_entry)->thread_group->setCancelPredicate( + [weak_this = weak_from_this()]() -> bool + { + if (auto shared_this = weak_this.lock()) + { + return shared_this->isCancelled(); + } + + return true; + }); + SinkToStoragePtr sink; const auto new_file_path_callback = [&exports_list_entry](const std::string & file_path) @@ -185,6 +207,8 @@ bool ExportPartTask::executeStep() try { + ThreadGroupSwitcher switcher((*exports_list_entry)->thread_group, ThreadName::EXPORT_PART); + const auto filename = buildDestinationFilename(manifest, storage.getStorageID(), local_context); sink = destination_storage->import( @@ -233,8 +257,6 @@ bool ExportPartTask::executeStep() local_context, getLogger("ExportPartition")); - ThreadGroupSwitcher switcher((*exports_list_entry)->thread_group, ThreadName::EXPORT_PART); - /// We need to support exporting materialized and alias columns to object storage. For some reason, object storage engines don't support them. /// This is a hack that materializes the columns before the export so they can be exported to tables that have matching columns materializeSpecialColumns(plan_for_part.getCurrentHeader(), metadata_snapshot, local_context, plan_for_part); diff --git a/src/Storages/MergeTree/ExportPartTask.h b/src/Storages/MergeTree/ExportPartTask.h index a3f1635c4902..1596f2bf23c9 100644 --- a/src/Storages/MergeTree/ExportPartTask.h +++ b/src/Storages/MergeTree/ExportPartTask.h @@ -7,7 +7,7 @@ namespace DB { -class ExportPartTask : public IExecutableTask +class ExportPartTask : public IExecutableTask, public std::enable_shared_from_this { public: explicit ExportPartTask( From 1906f2017f8a51780b2420909806f6a6242301ad Mon Sep 17 00:00:00 2001 From: Andrey Zvonov <32552679+zvonand@users.noreply.github.com> Date: Mon, 18 May 2026 19:44:00 +0200 Subject: [PATCH 08/43] Merge pull request #1713 from Altinity/export_partition_remove_lock_in_task Remove lock inside the task export partition Source-PR: #1713 (https://github.com/Altinity/ClickHouse/pull/1713) --- src/Core/Settings.cpp | 4 - src/Core/SettingsChangesHistory.cpp | 1 - ...portReplicatedMergeTreePartitionManifest.h | 4 - .../ExportPartFromPartitionExportTask.cpp | 75 --------------- .../ExportPartFromPartitionExportTask.h | 36 ------- .../ExportPartitionTaskScheduler.cpp | 96 +++++-------------- src/Storages/MergeTree/MergeTreeData.h | 1 - src/Storages/StorageReplicatedMergeTree.cpp | 2 - src/Storages/StorageReplicatedMergeTree.h | 1 - 9 files changed, 25 insertions(+), 195 deletions(-) delete mode 100644 src/Storages/MergeTree/ExportPartFromPartitionExportTask.cpp delete mode 100644 src/Storages/MergeTree/ExportPartFromPartitionExportTask.h diff --git a/src/Core/Settings.cpp b/src/Core/Settings.cpp index 67241001c1b0..2543b236ffca 100644 --- a/src/Core/Settings.cpp +++ b/src/Core/Settings.cpp @@ -8075,10 +8075,6 @@ Throw an error if there are pending mutations when exporting a merge tree part. )", 0) \ DECLARE(Bool, export_merge_tree_part_throw_on_pending_patch_parts, true, R"( Throw an error if there are pending patch parts when exporting a merge tree part. -)", 0) \ - DECLARE(Bool, export_merge_tree_partition_lock_inside_the_task, false, R"( -Only lock a part when the task is already running. This might help with busy waiting where the scheduler locks a part, but the task ends in the pending list. -On the other hand, there is a chance once the task executes that part has already been locked by another replica and the task will simply early exit. )", 0) \ DECLARE(Bool, export_merge_tree_partition_system_table_prefer_remote_information, false, R"( Controls whether the system.replicated_partition_exports will prefer to query ZooKeeper to get the most up to date information or use the local information. diff --git a/src/Core/SettingsChangesHistory.cpp b/src/Core/SettingsChangesHistory.cpp index 080d15b05918..49cbdb18b557 100644 --- a/src/Core/SettingsChangesHistory.cpp +++ b/src/Core/SettingsChangesHistory.cpp @@ -494,7 +494,6 @@ const VersionToSettingsChangesMap & getSettingsChangesHistory() {"export_merge_tree_part_file_already_exists_policy", "skip", "skip", "New setting."}, {"export_merge_tree_part_max_bytes_per_file", 0, 0, "New setting."}, {"export_merge_tree_part_max_rows_per_file", 0, 0, "New setting."}, - {"export_merge_tree_partition_lock_inside_the_task", false, false, "New setting."}, {"export_merge_tree_partition_system_table_prefer_remote_information", true, true, "New setting."}, {"export_merge_tree_part_throw_on_pending_mutations", true, true, "New setting."}, {"export_merge_tree_part_throw_on_pending_patch_parts", true, true, "New setting."}, diff --git a/src/Storages/ExportReplicatedMergeTreePartitionManifest.h b/src/Storages/ExportReplicatedMergeTreePartitionManifest.h index f1c96b120a28..dd5ef9886ded 100644 --- a/src/Storages/ExportReplicatedMergeTreePartitionManifest.h +++ b/src/Storages/ExportReplicatedMergeTreePartitionManifest.h @@ -119,7 +119,6 @@ struct ExportReplicatedMergeTreePartitionManifest size_t max_rows_per_file; MergeTreePartExportManifest::FileAlreadyExistsPolicy file_already_exists_policy; String filename_pattern; - bool lock_inside_the_task; /// todo temporary bool write_full_path_in_iceberg_metadata = false; String iceberg_metadata_json; @@ -154,7 +153,6 @@ struct ExportReplicatedMergeTreePartitionManifest json.set("max_retries", max_retries); json.set("ttl_seconds", ttl_seconds); json.set("task_timeout_seconds", task_timeout_seconds); - json.set("lock_inside_the_task", lock_inside_the_task); json.set("write_full_path_in_iceberg_metadata", write_full_path_in_iceberg_metadata); std::ostringstream oss; // STYLE_CHECK_ALLOW_STD_STRING_STREAM oss.exceptions(std::ios::failbit); @@ -208,8 +206,6 @@ struct ExportReplicatedMergeTreePartitionManifest /// what to do if it's not a valid value? } - manifest.lock_inside_the_task = json->getValue("lock_inside_the_task"); - manifest.write_full_path_in_iceberg_metadata = json->getValue("write_full_path_in_iceberg_metadata"); return manifest; diff --git a/src/Storages/MergeTree/ExportPartFromPartitionExportTask.cpp b/src/Storages/MergeTree/ExportPartFromPartitionExportTask.cpp deleted file mode 100644 index 2a3343e73f13..000000000000 --- a/src/Storages/MergeTree/ExportPartFromPartitionExportTask.cpp +++ /dev/null @@ -1,75 +0,0 @@ -#include -#include -#include - -namespace ProfileEvents -{ - extern const Event ExportPartitionZooKeeperRequests; - extern const Event ExportPartitionZooKeeperGetChildren; - extern const Event ExportPartitionZooKeeperCreate; -} -namespace DB -{ - -ExportPartFromPartitionExportTask::ExportPartFromPartitionExportTask( - StorageReplicatedMergeTree & storage_, - const std::string & key_, - const MergeTreePartExportManifest & manifest_) - : storage(storage_), - key(key_), - manifest(manifest_) -{ - export_part_task = std::make_shared(storage, manifest); -} - -bool ExportPartFromPartitionExportTask::executeStep() -{ - /// Runs on a MergeTreeBackgroundExecutor thread, so it does not inherit any component set by the scheduling task. - auto component_guard = Coordination::setCurrentComponent("ExportPartFromPartitionExportTask::executeStep"); - - const auto zk = storage.getZooKeeper(); - const auto part_name = manifest.data_part->name; - - LOG_INFO(storage.log, "ExportPartFromPartitionExportTask: Attempting to lock part: {}", part_name); - - ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRequests); - ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperCreate); - if (Coordination::Error::ZOK == zk->tryCreate(fs::path(storage.zookeeper_path) / "exports" / key / "locks" / part_name, storage.replica_name, zkutil::CreateMode::Ephemeral)) - { - LOG_INFO(storage.log, "ExportPartFromPartitionExportTask: Locked part: {}", part_name); - export_part_task->executeStep(); - return false; - } - - std::lock_guard inner_lock(storage.export_manifests_mutex); - storage.export_manifests.erase(manifest); - - LOG_INFO(storage.log, "ExportPartFromPartitionExportTask: Failed to lock part {}, skipping", part_name); - return false; -} - -void ExportPartFromPartitionExportTask::cancel() noexcept -{ - export_part_task->cancel(); -} - -void ExportPartFromPartitionExportTask::onCompleted() -{ - export_part_task->onCompleted(); -} - -StorageID ExportPartFromPartitionExportTask::getStorageID() const -{ - return export_part_task->getStorageID(); -} - -Priority ExportPartFromPartitionExportTask::getPriority() const -{ - return export_part_task->getPriority(); -} - -String ExportPartFromPartitionExportTask::getQueryId() const -{ - return export_part_task->getQueryId(); -} -} diff --git a/src/Storages/MergeTree/ExportPartFromPartitionExportTask.h b/src/Storages/MergeTree/ExportPartFromPartitionExportTask.h deleted file mode 100644 index e170b22b470d..000000000000 --- a/src/Storages/MergeTree/ExportPartFromPartitionExportTask.h +++ /dev/null @@ -1,36 +0,0 @@ -#pragma once - -#include -#include -#include -#include - -namespace DB -{ - -/* - Decorator around the ExportPartTask to lock the part inside the task -*/ -class ExportPartFromPartitionExportTask : public IExecutableTask -{ -public: - explicit ExportPartFromPartitionExportTask( - StorageReplicatedMergeTree & storage_, - const std::string & key_, - const MergeTreePartExportManifest & manifest_); - bool executeStep() override; - void onCompleted() override; - StorageID getStorageID() const override; - Priority getPriority() const override; - String getQueryId() const override; - - void cancel() noexcept override; - -private: - StorageReplicatedMergeTree & storage; - std::string key; - MergeTreePartExportManifest manifest; - std::shared_ptr export_part_task; -}; - -} diff --git a/src/Storages/MergeTree/ExportPartitionTaskScheduler.cpp b/src/Storages/MergeTree/ExportPartitionTaskScheduler.cpp index a77e2894b4a1..5ee5c42e3412 100644 --- a/src/Storages/MergeTree/ExportPartitionTaskScheduler.cpp +++ b/src/Storages/MergeTree/ExportPartitionTaskScheduler.cpp @@ -8,7 +8,6 @@ #include #include "Storages/MergeTree/ExportPartitionUtils.h" #include "Storages/MergeTree/MergeTreePartExportManifest.h" -#include "Storages/MergeTree/ExportPartFromPartitionExportTask.h" #include "Formats/FormatFactory.h" #include @@ -177,90 +176,45 @@ void ExportPartitionTaskScheduler::run() auto context = ExportPartitionUtils::getContextCopyWithTaskSettings(storage.getContext(), manifest); - /// todo arthur this code path does not perform all the validations a simple part export does because we are not calling exportPartToTable directly. - /// the schema and everything else has been validated when the export partition task was created, but nothing prevents the destination table from being - /// recreated with a new schema before the export task is scheduled. - if (manifest.lock_inside_the_task) + try { - LOG_INFO(storage.log, "ExportPartition scheduler task: Locking part export inside the task"); - std::lock_guard part_export_lock(storage.export_manifests_mutex); + LOG_INFO(storage.log, "ExportPartition scheduler task: Exporting part to table"); - MergeTreePartExportManifest part_export_manifest( - destination_storage, - part, + LOG_INFO(storage.log, "ExportPartition scheduler task: Attempting to lock part: {}", zk_part_name); + + ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRequests); + ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperCreate); + if (Coordination::Error::ZOK != zk->tryCreate(fs::path(storage.zookeeper_path) / "exports" / key / "locks" / zk_part_name, storage.replica_name, zkutil::CreateMode::Ephemeral)) + { + LOG_INFO(storage.log, "ExportPartition scheduler task: Failed to lock part {}, skipping", zk_part_name); + continue; + } + + LOG_INFO(storage.log, "ExportPartition scheduler task: Locked part: {}", zk_part_name); + + storage.exportPartToTable( + part->name, + destination_storage_id, manifest.transaction_id, - manifest.query_id, - context->getSettingsRef()[Setting::export_merge_tree_part_file_already_exists_policy].value, - context->getSettingsCopy(), - storage.getInMemoryMetadataPtr(), + context, manifest.iceberg_metadata_json, + /*allow_outdated_parts*/ true, [this, key, zk_part_name, manifest, destination_storage] (MergeTreePartExportManifest::CompletionCallbackResult result) { handlePartExportCompletion(key, zk_part_name, manifest, destination_storage, result); }); - part_export_manifest.task = std::make_shared(storage, key, part_export_manifest); - - /// todo arthur this might conflict with the standalone export part. what to do in this case? - if (!storage.export_manifests.emplace(part_export_manifest).second) - { - LOG_INFO(storage.log, "ExportPartition scheduler task: Part {} is already being exported, skipping", zk_part_name); - continue; - } - - if (!storage.background_moves_assignee.scheduleMoveTask(part_export_manifest.task)) - { - storage.export_manifests.erase(part_export_manifest); - LOG_INFO(storage.log, "ExportPartition scheduler task: Failed to schedule export part task, skipping"); - return; - } - scheduled_exports_count++; } - else + catch (const Exception &) { - try - { - LOG_INFO(storage.log, "ExportPartition scheduler task: Exporting part to table"); - - LOG_INFO(storage.log, "ExportPartition scheduler task: Attempting to lock part: {}", zk_part_name); - - ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRequests); - ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperCreate); - if (Coordination::Error::ZOK != zk->tryCreate(fs::path(storage.zookeeper_path) / "exports" / key / "locks" / zk_part_name, storage.replica_name, zkutil::CreateMode::Ephemeral)) - { - LOG_INFO(storage.log, "ExportPartition scheduler task: Failed to lock part {}, skipping", zk_part_name); - continue; - } - - LOG_INFO(storage.log, "ExportPartition scheduler task: Locked part: {}", zk_part_name); - - storage.exportPartToTable( - part->name, - destination_storage_id, - manifest.transaction_id, - context, - manifest.iceberg_metadata_json, - /*allow_outdated_parts*/ true, - [this, key, zk_part_name, manifest, destination_storage] - (MergeTreePartExportManifest::CompletionCallbackResult result) - { - handlePartExportCompletion(key, zk_part_name, manifest, destination_storage, result); - }); - - scheduled_exports_count++; - } - catch (const Exception &) - { - tryLogCurrentException(__PRETTY_FUNCTION__); - ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRequests); - ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRemove); - zk->tryRemove(fs::path(storage.zookeeper_path) / "exports" / key / "locks" / zk_part_name); - /// we should not increment retry_count because the node might just be full - } + tryLogCurrentException(__PRETTY_FUNCTION__); + ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRequests); + ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRemove); + zk->tryRemove(fs::path(storage.zookeeper_path) / "exports" / key / "locks" / zk_part_name); + /// we should not increment retry_count because the node might just be full } - } } } diff --git a/src/Storages/MergeTree/MergeTreeData.h b/src/Storages/MergeTree/MergeTreeData.h index 480226d94fdf..0d29276c6c18 100644 --- a/src/Storages/MergeTree/MergeTreeData.h +++ b/src/Storages/MergeTree/MergeTreeData.h @@ -1504,7 +1504,6 @@ class MergeTreeData : public WithMutableContext, public IStorage, public IBackgr friend class VersionMetadataOnKeeper; // for access to log friend class MutationsState; // for access to log friend class ExportPartTask; - friend class ExportPartFromPartitionExportTask; bool require_part_metadata; diff --git a/src/Storages/StorageReplicatedMergeTree.cpp b/src/Storages/StorageReplicatedMergeTree.cpp index 86119b70df72..32b1e18eec6c 100644 --- a/src/Storages/StorageReplicatedMergeTree.cpp +++ b/src/Storages/StorageReplicatedMergeTree.cpp @@ -235,7 +235,6 @@ namespace Setting extern const SettingsUInt64 export_merge_tree_part_max_rows_per_file; extern const SettingsBool export_merge_tree_part_throw_on_pending_mutations; extern const SettingsBool export_merge_tree_part_throw_on_pending_patch_parts; - extern const SettingsBool export_merge_tree_partition_lock_inside_the_task; extern const SettingsString export_merge_tree_part_filename_pattern; extern const SettingsBool write_full_path_in_iceberg_metadata; extern const SettingsBool allow_insert_into_iceberg; @@ -8673,7 +8672,6 @@ void StorageReplicatedMergeTree::exportPartitionToTable(const PartitionCommand & manifest.parquet_parallel_encoding = query_context->getSettingsRef()[Setting::output_format_parquet_parallel_encoding]; manifest.max_bytes_per_file = query_context->getSettingsRef()[Setting::export_merge_tree_part_max_bytes_per_file]; manifest.max_rows_per_file = query_context->getSettingsRef()[Setting::export_merge_tree_part_max_rows_per_file]; - manifest.lock_inside_the_task = query_context->getSettingsRef()[Setting::export_merge_tree_partition_lock_inside_the_task]; manifest.file_already_exists_policy = query_context->getSettingsRef()[Setting::export_merge_tree_part_file_already_exists_policy].value; manifest.filename_pattern = query_context->getSettingsRef()[Setting::export_merge_tree_part_filename_pattern].value; diff --git a/src/Storages/StorageReplicatedMergeTree.h b/src/Storages/StorageReplicatedMergeTree.h index 58e46d3b019b..fc62da110449 100644 --- a/src/Storages/StorageReplicatedMergeTree.h +++ b/src/Storages/StorageReplicatedMergeTree.h @@ -405,7 +405,6 @@ class StorageReplicatedMergeTree final : public MergeTreeData friend class ReplicatedMergeMutateTaskBase; friend class ExportPartitionManifestUpdatingTask; friend class ExportPartitionTaskScheduler; - friend class ExportPartFromPartitionExportTask; using MergeStrategyPicker = ReplicatedMergeTreeMergeStrategyPicker; using LogEntry = ReplicatedMergeTreeLogEntry; From 3c13e2490fb11dc2c447c5bded9ee7e79ce04602 Mon Sep 17 00:00:00 2001 From: Andrey Zvonov <32552679+zvonand@users.noreply.github.com> Date: Mon, 18 May 2026 19:45:51 +0200 Subject: [PATCH 09/43] Merge pull request #1783 from Altinity/bugfix/antalya-26.3/frontports_ai_issues Fix incorrect interaction between object_storage_remote_initiator and object_storage_cluster. Source-PR: #1783 (https://github.com/Altinity/ClickHouse/pull/1783) --- src/Storages/IStorageCluster.cpp | 12 +- .../StorageObjectStorageCluster.cpp | 62 +++---- .../StorageObjectStorageCluster.h | 3 +- .../test_remote_initiator.py | 154 ++++++++++++++++++ 4 files changed, 195 insertions(+), 36 deletions(-) create mode 100644 tests/integration/test_storage_iceberg_with_spark/test_remote_initiator.py diff --git a/src/Storages/IStorageCluster.cpp b/src/Storages/IStorageCluster.cpp index 8749a4dc571a..1d7db845af04 100644 --- a/src/Storages/IStorageCluster.cpp +++ b/src/Storages/IStorageCluster.cpp @@ -222,6 +222,10 @@ void IStorageCluster::updateQueryWithJoinToSendIfNeeded( { case ObjectStorageClusterJoinMode::LOCAL: { + if (!context->getSettingsRef()[Setting::allow_experimental_analyzer]) + throw Exception(ErrorCodes::NOT_IMPLEMENTED, + "object_storage_cluster_join_mode!='allow' is not supported without allow_experimental_analyzer=true"); + auto info = getQueryTreeInfo(query_tree, context); if (info.has_join || info.has_cross_join || info.has_local_columns_in_where) @@ -332,16 +336,16 @@ void IStorageCluster::read( { if (settings[Setting::object_storage_remote_initiator]) { + auto remote_initiator_cluster_name = settings[Setting::object_storage_remote_initiator_cluster].value; + if (remote_initiator_cluster_name.empty()) + throw Exception(ErrorCodes::BAD_ARGUMENTS, "Setting 'object_storage_remote_initiator' can be used only with 'object_storage_remote_initiator_cluster' or 'object_storage_cluster'"); + /// rewrite query to execute `remote('remote_host', s3(...))` /// remote_host can execute query itself or make on-cluster query depends on own `object_storage_cluster` setting updateConfigurationIfNeeded(context); updateQueryWithJoinToSendIfNeeded(query_to_send, query_info.query_tree, context); updateQueryToSendIfNeeded(query_to_send, storage_snapshot, context, /*make_cluster_function*/ false); - auto remote_initiator_cluster_name = settings[Setting::object_storage_remote_initiator_cluster].value; - if (remote_initiator_cluster_name.empty()) - throw Exception(ErrorCodes::BAD_ARGUMENTS, "Setting 'object_storage_remote_initiator' can be used only with 'object_storage_remote_initiator_cluster' or 'object_storage_cluster'"); - auto remote_initiator_cluster = getClusterImpl(context, remote_initiator_cluster_name); auto storage_and_context = convertToRemote(remote_initiator_cluster, context, remote_initiator_cluster_name, query_to_send); auto src_distributed = std::dynamic_pointer_cast(storage_and_context.storage); diff --git a/src/Storages/ObjectStorage/StorageObjectStorageCluster.cpp b/src/Storages/ObjectStorage/StorageObjectStorageCluster.cpp index 26acb15c7f68..66c7b4de7f2b 100644 --- a/src/Storages/ObjectStorage/StorageObjectStorageCluster.cpp +++ b/src/Storages/ObjectStorage/StorageObjectStorageCluster.cpp @@ -294,7 +294,7 @@ std::optional StorageObjectStorageCluster::totalBytes(ContextPtr query_c return configuration->totalBytes(query_context); } -void StorageObjectStorageCluster::updateQueryForDistributedEngineIfNeeded(ASTPtr & query, ContextPtr context, bool make_cluster_function) +bool StorageObjectStorageCluster::updateQueryForDistributedEngineIfNeeded(ASTPtr & query, ContextPtr context, bool make_cluster_function) { // Change table engine on table function for distributed request // CREATE TABLE t (...) ENGINE=IcebergS3(...) @@ -305,7 +305,7 @@ void StorageObjectStorageCluster::updateQueryForDistributedEngineIfNeeded(ASTPtr auto * select_query = query->as(); if (!select_query || !select_query->tables()) - return; + return false; auto * tables = select_query->tables()->as(); @@ -318,10 +318,10 @@ void StorageObjectStorageCluster::updateQueryForDistributedEngineIfNeeded(ASTPtr auto * table_expression = tables->children[0]->as()->table_expression->as(); if (!table_expression) - return; + return false; if (!table_expression->database_and_table_name) - return; + return false; auto & table_identifier_typed = table_expression->database_and_table_name->as(); @@ -394,34 +394,34 @@ void StorageObjectStorageCluster::updateQueryForDistributedEngineIfNeeded(ASTPtr table_expression->table_function = function_ast_ptr; table_expression->children[0] = function_ast_ptr; - if (make_cluster_function) - { - auto cluster_name = getClusterName(context); + if (!make_cluster_function) + return false; - if (cluster_name.empty()) - { - throw Exception( - ErrorCodes::LOGICAL_ERROR, - "Can't be here without cluster name, no cluster name in query {}", - query->formatForLogging()); - } + auto cluster_name = getClusterName(context); - auto settings = select_query->settings(); - if (settings) - { - auto & settings_ast = settings->as(); - settings_ast.changes.insertSetting("object_storage_cluster", cluster_name); - } - else - { - auto settings_ast_ptr = make_intrusive(); - settings_ast_ptr->is_standalone = false; - settings_ast_ptr->changes.setSetting("object_storage_cluster", cluster_name); - select_query->setExpression(ASTSelectQuery::Expression::SETTINGS, std::move(settings_ast_ptr)); - } + if (cluster_name.empty()) + { + throw Exception( + ErrorCodes::LOGICAL_ERROR, + "Can't be here without cluster name, no cluster name in query {}", + query->formatForLogging()); + } - cluster_name_in_settings = true; + auto settings = select_query->settings(); + if (settings) + { + auto & settings_ast = settings->as(); + settings_ast.changes.insertSetting("object_storage_cluster", cluster_name); } + else + { + auto settings_ast_ptr = make_intrusive(); + settings_ast_ptr->is_standalone = false; + settings_ast_ptr->changes.setSetting("object_storage_cluster", cluster_name); + select_query->setExpression(ASTSelectQuery::Expression::SETTINGS, std::move(settings_ast_ptr)); + } + + return true; } void StorageObjectStorageCluster::updateQueryToSendIfNeeded( @@ -430,7 +430,7 @@ void StorageObjectStorageCluster::updateQueryToSendIfNeeded( const ContextPtr & context, bool make_cluster_function) { - updateQueryForDistributedEngineIfNeeded(query, context, make_cluster_function); + bool cluster_name_added_to_settings = updateQueryForDistributedEngineIfNeeded(query, context, make_cluster_function); auto * table_function = extractTableFunctionFromSelectQuery(query); if (!table_function) @@ -455,7 +455,7 @@ void StorageObjectStorageCluster::updateQueryToSendIfNeeded( } ASTPtr object_storage_type_arg; - configuration->extractDynamicStorageType(args, context, &object_storage_type_arg, !cluster_name_in_settings); + configuration->extractDynamicStorageType(args, context, &object_storage_type_arg, !cluster_name_in_settings && !cluster_name_added_to_settings); ASTPtr settings_temporary_storage = nullptr; for (auto it = args.begin(); it != args.end(); ++it) @@ -469,7 +469,7 @@ void StorageObjectStorageCluster::updateQueryToSendIfNeeded( } } - if (cluster_name_in_settings || !endsWith(table_function->name, "Cluster")) + if (cluster_name_in_settings || cluster_name_added_to_settings || !endsWith(table_function->name, "Cluster")) { configuration->addStructureAndFormatToArgsIfNeeded(args, structure, configuration->getFormat(), context, /*with_structure=*/true); diff --git a/src/Storages/ObjectStorage/StorageObjectStorageCluster.h b/src/Storages/ObjectStorage/StorageObjectStorageCluster.h index c168af53c28b..52c7d5951855 100644 --- a/src/Storages/ObjectStorage/StorageObjectStorageCluster.h +++ b/src/Storages/ObjectStorage/StorageObjectStorageCluster.h @@ -207,8 +207,9 @@ class StorageObjectStorageCluster : public IStorageCluster to SELECT * FROM s3(...) SETTINGS object_storage_cluster='cluster' to make distributed request over cluster 'cluster'. + Returns true if cluster name was added to settings. */ - void updateQueryForDistributedEngineIfNeeded(ASTPtr & query, ContextPtr context, bool make_cluster_function); + bool updateQueryForDistributedEngineIfNeeded(ASTPtr & query, ContextPtr context, bool make_cluster_function); const String engine_name; StorageObjectStorageConfigurationPtr configuration; diff --git a/tests/integration/test_storage_iceberg_with_spark/test_remote_initiator.py b/tests/integration/test_storage_iceberg_with_spark/test_remote_initiator.py new file mode 100644 index 000000000000..ba0a61f9a998 --- /dev/null +++ b/tests/integration/test_storage_iceberg_with_spark/test_remote_initiator.py @@ -0,0 +1,154 @@ +import pytest +import uuid + +from helpers.iceberg_utils import ( + get_uuid_str, + create_iceberg_table, + execute_spark_query_general, +) + + +@pytest.mark.parametrize("storage_type", ["s3"]) +def test_remote_initiator_after_non_remote(started_cluster_iceberg_with_spark, storage_type): + instance = started_cluster_iceberg_with_spark.instances["node1"] + spark = started_cluster_iceberg_with_spark.spark_session + TABLE_NAME = "test_remote_initiator_after_non_remote_table_" + get_uuid_str() + + def execute_spark_query(query: str): + return execute_spark_query_general( + spark, + started_cluster_iceberg_with_spark, + storage_type, + TABLE_NAME, + query, + ) + + execute_spark_query( + f""" + CREATE TABLE {TABLE_NAME} ( + tag INT, + number INT + ) + USING iceberg + PARTITIONED BY (identity(tag)) + OPTIONS('format-version'='2') + """ + ) + + execute_spark_query( + f""" + INSERT INTO {TABLE_NAME} VALUES + (1, 1) + """ + ) + + create_iceberg_table(storage_type, instance, TABLE_NAME, started_cluster_iceberg_with_spark) + + def flush_logs(): + for node in started_cluster_iceberg_with_spark.instances.values(): + node.query("SYSTEM FLUSH LOGS") + + query_id = uuid.uuid4().hex + res = instance.query(f""" + SELECT * + FROM {TABLE_NAME} + WHERE number=1 + SETTINGS + object_storage_cluster='cluster_simple' + """, + query_id = query_id) + assert res == "1\t1\n" + flush_logs() + queries = instance.query(f""" + SELECT count() + FROM clusterAllReplicas('cluster_simple', system.query_log) + WHERE type='QueryFinish' AND initial_query_id='{query_id}' + """) + assert queries == "4\n" + + query_id = uuid.uuid4().hex + res = instance.query(f""" + SELECT * + FROM {TABLE_NAME} + WHERE number=1 + SETTINGS + object_storage_remote_initiator=1, + object_storage_cluster='cluster_simple' + """, + query_id = query_id) + assert res == "1\t1\n" + flush_logs() + queries = instance.query(f""" + SELECT count() + FROM clusterAllReplicas('cluster_simple', system.query_log) + WHERE type='QueryFinish' AND initial_query_id='{query_id}' + """) + assert queries == "6\n" + + query_id = uuid.uuid4().hex + res = instance.query(f""" + SELECT * + FROM {TABLE_NAME} + WHERE number=1 + """, + query_id = query_id) + assert res == "1\t1\n" + flush_logs() + queries = instance.query(f""" + SELECT count() + FROM clusterAllReplicas('cluster_simple', system.query_log) + WHERE type='QueryFinish' AND initial_query_id='{query_id}' + """) + assert queries == "1\n" + + +@pytest.mark.parametrize("storage_type", ["s3"]) +def test_remote_initiator_after_with_join_old_analyzer(started_cluster_iceberg_with_spark, storage_type): + instance = started_cluster_iceberg_with_spark.instances["node1"] + spark = started_cluster_iceberg_with_spark.spark_session + TABLE_NAME = "test_remote_initiator_after_with_join_old_analyzer_table_" + get_uuid_str() + TABLE2_NAME = "test_remote_initiator_after_with_join_old_analyzer_table_2_" + get_uuid_str() + + def execute_spark_query(query: str): + return execute_spark_query_general( + spark, + started_cluster_iceberg_with_spark, + storage_type, + TABLE_NAME, + query, + ) + + execute_spark_query( + f""" + CREATE TABLE {TABLE_NAME} ( + tag INT, + number INT + ) + USING iceberg + PARTITIONED BY (identity(tag)) + OPTIONS('format-version'='2') + """ + ) + + execute_spark_query( + f""" + INSERT INTO {TABLE_NAME} VALUES + (1, 1) + """ + ) + + create_iceberg_table(storage_type, instance, TABLE_NAME, started_cluster_iceberg_with_spark) + + instance.query(f"CREATE TABLE {TABLE2_NAME} (tag INT, number2 INT) ENGINE=Memory") + instance.query(f"INSERT INTO {TABLE2_NAME} VALUES (1, 2)") + + assert "object_storage_cluster_join_mode!='allow' is not supported without allow_experimental_analyzer=true" in instance.query_and_get_error(f""" + SELECT * + FROM {TABLE_NAME} AS t1 + JOIN {TABLE2_NAME} AS t2 USING (tag) + SETTINGS + object_storage_remote_initiator=1, + object_storage_remote_initiator_cluster='cluster_simple', + object_storage_cluster_join_mode='local', + allow_experimental_analyzer=0 + """) From 98ed773e3c399d70618f27104655b95692c0de16 Mon Sep 17 00:00:00 2001 From: Andrey Zvonov <32552679+zvonand@users.noreply.github.com> Date: Thu, 21 May 2026 11:13:00 +0200 Subject: [PATCH 10/43] Cherry-pick of https://github.com/Altinity/ClickHouse/pull/1741 with unresolved conflict markers (resolution in next commit) --- Original cherry-pick message follows: Merge pull request #1741 from Altinity/export_partition_all Export partition all # Conflicts: # src/Common/ErrorCodes.cpp # src/Core/Settings.h --- src/Common/ErrorCodes.cpp | 10 ++ src/Core/Settings.cpp | 8 ++ src/Core/Settings.h | 4 + src/Core/SettingsChangesHistory.cpp | 1 + src/Core/SettingsEnums.cpp | 2 + src/Core/SettingsEnums.h | 8 ++ src/Storages/MergeTree/MergeTreeData.cpp | 7 +- src/Storages/StorageReplicatedMergeTree.cpp | 83 +++++++++++++++- .../test.py | 26 +++++ .../test.py | 97 +++++++++++++++++++ 10 files changed, 242 insertions(+), 4 deletions(-) diff --git a/src/Common/ErrorCodes.cpp b/src/Common/ErrorCodes.cpp index 4664dff73f1c..cafc88ef114c 100644 --- a/src/Common/ErrorCodes.cpp +++ b/src/Common/ErrorCodes.cpp @@ -672,11 +672,17 @@ M(1002, UNKNOWN_EXCEPTION) \ M(1003, SSH_EXCEPTION) \ M(1004, STARTUP_SCRIPTS_ERROR) \ +<<<<<<< HEAD M(1005, STALE_VERSION) \ M(1006, INVALID_CURSOR_LOOKUP) \ M(1007, ILLEGAL_STREAM) \ M(1008, TEMPORARY_DATA_NOT_IN_CACHE) \ M(1009, PENDING_MUTATIONS_NOT_ALLOWED) \ +======= + M(1005, PENDING_MUTATIONS_NOT_ALLOWED) \ + M(1006, EXPORT_PARTITION_ALREADY_EXPORTED) \ + M(1007, PARTITION_EXPORT_FAILED) \ +>>>>>>> 12992af66fa (Merge pull request #1741 from Altinity/export_partition_all) /* See END */ #ifdef APPLY_FOR_EXTERNAL_ERROR_CODES @@ -693,7 +699,11 @@ namespace ErrorCodes APPLY_FOR_ERROR_CODES(M) #undef M +<<<<<<< HEAD constexpr ErrorCode END = 1009; +======= + constexpr ErrorCode END = 1007; +>>>>>>> 12992af66fa (Merge pull request #1741 from Altinity/export_partition_all) ErrorPairHolder values[END + 1]{}; struct ErrorCodesNames diff --git a/src/Core/Settings.cpp b/src/Core/Settings.cpp index 2543b236ffca..6f0e6c87f934 100644 --- a/src/Core/Settings.cpp +++ b/src/Core/Settings.cpp @@ -8079,6 +8079,14 @@ Throw an error if there are pending patch parts when exporting a merge tree part DECLARE(Bool, export_merge_tree_partition_system_table_prefer_remote_information, false, R"( Controls whether the system.replicated_partition_exports will prefer to query ZooKeeper to get the most up to date information or use the local information. Querying ZooKeeper is expensive, and only available if the ZooKeeper feature flag MULTI_READ is enabled. +)", 0) \ + DECLARE(ExportPartitionAllOnError, export_merge_tree_partition_all_on_error, ExportPartitionAllOnError::throw_first, R"( +Failure handling for `ALTER TABLE ... EXPORT PARTITION ALL ...`. +Possible values: +- `throw_first` (default) - stop at the first failed partition; partitions already scheduled remain scheduled. +- `collect` - try every partition and throw a single aggregated exception at the end if any failed; partitions that succeeded remain scheduled. +- `skip_conflicts` - silently skip partitions that are already exported / being exported (errors with code EXPORT_PARTITION_ALREADY_EXPORTED); fail-fast on every other error. +Has no effect on `EXPORT PARTITION ` (single-partition export). )", 0) \ DECLARE(String, export_merge_tree_part_filename_pattern, "{part_name}_{checksum}", R"( Pattern for the filename of the exported merge tree part. The `part_name` and `checksum` are calculated and replaced on the fly. Additional macros are supported. diff --git a/src/Core/Settings.h b/src/Core/Settings.h index 277b1dfda86d..3883f06e4326 100644 --- a/src/Core/Settings.h +++ b/src/Core/Settings.h @@ -122,7 +122,11 @@ class WriteBuffer; M(CLASS_NAME, JoinOrderAlgorithm) \ M(CLASS_NAME, DeduplicateInsertSelectMode) \ M(CLASS_NAME, DeduplicateInsertMode) \ +<<<<<<< HEAD M(CLASS_NAME, FileLikeEngineDefaultPartitionStrategy) +======= + M(CLASS_NAME, ExportPartitionAllOnError) \ +>>>>>>> 12992af66fa (Merge pull request #1741 from Altinity/export_partition_all) COMMON_SETTINGS_SUPPORTED_TYPES(Settings, DECLARE_SETTING_TRAIT) diff --git a/src/Core/SettingsChangesHistory.cpp b/src/Core/SettingsChangesHistory.cpp index 49cbdb18b557..8ab42f498c2e 100644 --- a/src/Core/SettingsChangesHistory.cpp +++ b/src/Core/SettingsChangesHistory.cpp @@ -495,6 +495,7 @@ const VersionToSettingsChangesMap & getSettingsChangesHistory() {"export_merge_tree_part_max_bytes_per_file", 0, 0, "New setting."}, {"export_merge_tree_part_max_rows_per_file", 0, 0, "New setting."}, {"export_merge_tree_partition_system_table_prefer_remote_information", true, true, "New setting."}, + {"export_merge_tree_partition_all_on_error", "throw_first", "throw_first", "New setting."}, {"export_merge_tree_part_throw_on_pending_mutations", true, true, "New setting."}, {"export_merge_tree_part_throw_on_pending_patch_parts", true, true, "New setting."}, {"object_storage_cluster", "", "", "Antalya: New setting"}, diff --git a/src/Core/SettingsEnums.cpp b/src/Core/SettingsEnums.cpp index dcb252f1d26d..065f7f65b9af 100644 --- a/src/Core/SettingsEnums.cpp +++ b/src/Core/SettingsEnums.cpp @@ -522,4 +522,6 @@ IMPLEMENT_SETTING_ENUM( IMPLEMENT_SETTING_AUTO_ENUM(MergeTreePartExportFileAlreadyExistsPolicy, ErrorCodes::BAD_ARGUMENTS); +IMPLEMENT_SETTING_AUTO_ENUM(ExportPartitionAllOnError, ErrorCodes::BAD_ARGUMENTS); + } diff --git a/src/Core/SettingsEnums.h b/src/Core/SettingsEnums.h index 1af70b226b84..7ec7f88492be 100644 --- a/src/Core/SettingsEnums.h +++ b/src/Core/SettingsEnums.h @@ -619,5 +619,13 @@ enum class MergeTreePartExportFileAlreadyExistsPolicy : uint8_t DECLARE_SETTING_ENUM(MergeTreePartExportFileAlreadyExistsPolicy) +enum class ExportPartitionAllOnError : uint8_t +{ + throw_first, + collect, + skip_conflicts, +}; + +DECLARE_SETTING_ENUM(ExportPartitionAllOnError) } diff --git a/src/Storages/MergeTree/MergeTreeData.cpp b/src/Storages/MergeTree/MergeTreeData.cpp index a20a916c68a8..1e28352bec5c 100644 --- a/src/Storages/MergeTree/MergeTreeData.cpp +++ b/src/Storages/MergeTree/MergeTreeData.cpp @@ -6780,8 +6780,11 @@ void MergeTreeData::checkAlterPartitionIsPossible( const auto * partition_ast = command.partition->as(); if (partition_ast && partition_ast->all) { - if (command.type != PartitionCommand::DROP_PARTITION && command.type != PartitionCommand::ATTACH_PARTITION && !(command.type == PartitionCommand::REPLACE_PARTITION && !command.replace)) - throw DB::Exception(ErrorCodes::SUPPORT_IS_DISABLED, "Only support DROP/DETACH/ATTACH PARTITION ALL currently"); + if (command.type != PartitionCommand::DROP_PARTITION + && command.type != PartitionCommand::ATTACH_PARTITION + && command.type != PartitionCommand::EXPORT_PARTITION + && !(command.type == PartitionCommand::REPLACE_PARTITION && !command.replace)) + throw DB::Exception(ErrorCodes::SUPPORT_IS_DISABLED, "Only support DROP/DETACH/ATTACH/EXPORT PARTITION ALL currently"); } else { diff --git a/src/Storages/StorageReplicatedMergeTree.cpp b/src/Storages/StorageReplicatedMergeTree.cpp index 32b1e18eec6c..ad8614412983 100644 --- a/src/Storages/StorageReplicatedMergeTree.cpp +++ b/src/Storages/StorageReplicatedMergeTree.cpp @@ -235,6 +235,7 @@ namespace Setting extern const SettingsUInt64 export_merge_tree_part_max_rows_per_file; extern const SettingsBool export_merge_tree_part_throw_on_pending_mutations; extern const SettingsBool export_merge_tree_part_throw_on_pending_patch_parts; + extern const SettingsExportPartitionAllOnError export_merge_tree_partition_all_on_error; extern const SettingsString export_merge_tree_part_filename_pattern; extern const SettingsBool write_full_path_in_iceberg_metadata; extern const SettingsBool allow_insert_into_iceberg; @@ -355,6 +356,8 @@ namespace ErrorCodes extern const int TIMEOUT_EXCEEDED; extern const int INVALID_SETTING_VALUE; extern const int PENDING_MUTATIONS_NOT_ALLOWED; + extern const int EXPORT_PARTITION_ALREADY_EXPORTED; + extern const int PARTITION_EXPORT_FAILED; } namespace ServerSetting @@ -8504,6 +8507,82 @@ void StorageReplicatedMergeTree::exportPartitionToTable(const PartitionCommand & "If you are exporting to an Apache Iceberg table, you also need to enable the setting `allow_experimental_insert_into_iceberg` on all replicas. The same goes for `allow_experimental_export_merge_tree_part`"); } + /// EXPORT PARTITION ALL: expand into one sub-call per active partition id. + /// Failure handling is controlled by `export_merge_tree_partition_all_on_error`. + if (const auto * partition_ast = command.partition->as(); partition_ast && partition_ast->all) + { + auto partition_id_set = getAllPartitionIds(); + if (partition_id_set.empty()) + throw Exception(ErrorCodes::BAD_ARGUMENTS, + "Table {} has no active partitions to export", + getStorageID().getNameForLogs()); + + /// Sort for deterministic ordering (so failure messages and tests are stable). + std::vector partition_ids(partition_id_set.begin(), partition_id_set.end()); + std::sort(partition_ids.begin(), partition_ids.end()); + + const auto & on_error_setting = query_context->getSettingsRef()[Setting::export_merge_tree_partition_all_on_error]; + const ExportPartitionAllOnError on_error = on_error_setting.value; + + LOG_INFO(log, "EXPORT PARTITION ALL: scheduling export for {} partitions, on_error={}", + partition_ids.size(), on_error_setting.toString()); + + std::vector> failures; /// (partition_id, message) + size_t skipped_conflicts = 0; + + for (const auto & partition_id : partition_ids) + { + PartitionCommand sub = command; + auto synthetic = make_intrusive(); + synthetic->setPartitionID(make_intrusive(partition_id)); + sub.partition = synthetic; + + try + { + exportPartitionToTable(sub, query_context); + } + catch (const Exception & e) + { + switch (on_error) + { + case ExportPartitionAllOnError::throw_first: + throw; + case ExportPartitionAllOnError::skip_conflicts: + if (e.code() == ErrorCodes::EXPORT_PARTITION_ALREADY_EXPORTED) + { + ++skipped_conflicts; + LOG_INFO(log, + "EXPORT PARTITION ALL: skipping partition {} (already exported / concurrent): {}", + partition_id, e.message()); + break; + } + throw; + case ExportPartitionAllOnError::collect: + LOG_WARNING(log, "EXPORT PARTITION ALL: partition {} failed: {}", + partition_id, e.message()); + failures.emplace_back(partition_id, e.message()); + break; + } + } + } + + if (!failures.empty()) + { + String aggregated = fmt::format( + "EXPORT PARTITION ALL: {}/{} partitions failed to schedule. Per-partition errors:", + failures.size(), partition_ids.size()); + for (const auto & [pid, msg] : failures) + aggregated += fmt::format("\n {}: {}", pid, msg); + throw Exception(ErrorCodes::PARTITION_EXPORT_FAILED, "{}", aggregated); + } + + if (skipped_conflicts > 0) + LOG_INFO(log, "EXPORT PARTITION ALL: skipped {} partitions due to existing exports", + skipped_conflicts); + + return; + } + const auto dest_database = query_context->resolveDatabase(command.to_database); const auto dest_table = command.to_table; const auto dest_storage_id = StorageID(dest_database, dest_table); @@ -8583,7 +8662,7 @@ void StorageReplicatedMergeTree::exportPartitionToTable(const PartitionCommand & if (!has_expired && !query_context->getSettingsRef()[Setting::export_merge_tree_partition_force_export]) { - throw Exception(ErrorCodes::BAD_ARGUMENTS, "Export with key {} already exported or it is being exported, and it has not expired. Set `export_merge_tree_partition_force_export` to overwrite it.", export_key); + throw Exception(ErrorCodes::EXPORT_PARTITION_ALREADY_EXPORTED, "Export with key {} already exported or it is being exported, and it has not expired. Set `export_merge_tree_partition_force_export` to overwrite it.", export_key); } LOG_INFO(log, "Overwriting export with key {}", export_key); @@ -8780,7 +8859,7 @@ void StorageReplicatedMergeTree::exportPartitionToTable(const PartitionCommand & { /// Lost the race on the root export node. Current code already /// validated (exists / expired / force) — so this is *always* a race. - throw Exception(ErrorCodes::BAD_ARGUMENTS, + throw Exception(ErrorCodes::EXPORT_PARTITION_ALREADY_EXPORTED, "Export with key {} was created concurrently by another replica. Retry if needed", export_key); } diff --git a/tests/integration/test_export_replicated_mt_partition_to_iceberg/test.py b/tests/integration/test_export_replicated_mt_partition_to_iceberg/test.py index af6ca75bafd7..574296f90c49 100644 --- a/tests/integration/test_export_replicated_mt_partition_to_iceberg/test.py +++ b/tests/integration/test_export_replicated_mt_partition_to_iceberg/test.py @@ -196,6 +196,32 @@ def test_export_two_partitions_to_iceberg(cluster): assert count_2021 == 1, f"Expected 1 row for year=2021, got {count_2021}" +def test_export_partition_all_to_iceberg(cluster): + """ + `ALTER TABLE ... EXPORT PARTITION ALL TO TABLE ...` schedules every active partition + in one statement and exercises the Iceberg-specific destination compatibility checks + (which are repeated per sub-call inside the loop). + """ + node = cluster.instances["replica1"] + + uid = unique_suffix() + mt_table = f"mt_{uid}" + iceberg_table = f"iceberg_{uid}" + + setup_tables(cluster, mt_table, iceberg_table, nodes=["replica1"]) + + node.query(f"ALTER TABLE {mt_table} EXPORT PARTITION ALL TO TABLE {iceberg_table}") + + wait_for_export_status(node, mt_table, iceberg_table, "2020", "COMPLETED") + wait_for_export_status(node, mt_table, iceberg_table, "2021", "COMPLETED") + + count_2020 = int(node.query(f"SELECT count() FROM {iceberg_table} WHERE year = 2020").strip()) + count_2021 = int(node.query(f"SELECT count() FROM {iceberg_table} WHERE year = 2021").strip()) + + assert count_2020 == 3, f"Expected 3 rows for year=2020, got {count_2020}" + assert count_2021 == 1, f"Expected 1 row for year=2021, got {count_2021}" + + def test_failure_is_logged_in_system_table(cluster): """ When S3 is unreachable the export must be marked FAILED in diff --git a/tests/integration/test_export_replicated_mt_partition_to_object_storage/test.py b/tests/integration/test_export_replicated_mt_partition_to_object_storage/test.py index a5e60f81194e..852e8cf78c72 100644 --- a/tests/integration/test_export_replicated_mt_partition_to_object_storage/test.py +++ b/tests/integration/test_export_replicated_mt_partition_to_object_storage/test.py @@ -1506,3 +1506,100 @@ def test_export_partition_resumes_after_stop_moves_during_export(cluster): row_count = int(node.query(f"SELECT count() FROM {s3_table} WHERE year = 2020").strip()) assert row_count == 3, f"Expected 3 rows in S3 after export completed, got {row_count}" + + +def test_export_partition_all(cluster): + """Happy path for `ALTER TABLE ... EXPORT PARTITION ALL TO TABLE ...`. + + Schedules one export task per active partition in a single ALTER, then + verifies every partition lands in the destination S3 table. + """ + node = cluster.instances["replica1"] + + uid = str(uuid.uuid4()).replace("-", "_") + mt_table = f"export_all_mt_{uid}" + s3_table = f"export_all_s3_{uid}" + + node.query( + f"CREATE TABLE {mt_table} (id UInt64, year UInt16)" + f" ENGINE = ReplicatedMergeTree('/clickhouse/tables/{mt_table}', 'replica1')" + f" PARTITION BY year ORDER BY tuple()" + ) + node.query(f"INSERT INTO {mt_table} VALUES (1, 2020), (2, 2021), (3, 2022)") + create_s3_table(node, s3_table) + + node.query(f"ALTER TABLE {mt_table} EXPORT PARTITION ALL TO TABLE {s3_table}") + + for partition_id in ("2020", "2021", "2022"): + wait_for_export_status(node, mt_table, s3_table, partition_id, "COMPLETED", timeout=60) + + row_count = int(node.query(f"SELECT count() FROM {s3_table}").strip()) + assert row_count == 3, f"Expected 3 rows in S3 after EXPORT PARTITION ALL, got {row_count}" + + +def test_export_partition_all_failure_modes(cluster): + """Cover the three values of `export_merge_tree_partition_all_on_error`. + + Set up an already-fully-exported source table, then re-run EXPORT PARTITION ALL + with each failure mode and assert the documented behavior. + """ + node = cluster.instances["replica1"] + + uid = str(uuid.uuid4()).replace("-", "_") + mt_table = f"export_all_modes_mt_{uid}" + s3_table = f"export_all_modes_s3_{uid}" + empty_mt = f"export_all_empty_mt_{uid}" + + node.query( + f"CREATE TABLE {mt_table} (id UInt64, year UInt16)" + f" ENGINE = ReplicatedMergeTree('/clickhouse/tables/{mt_table}', 'replica1')" + f" PARTITION BY year ORDER BY tuple()" + ) + node.query(f"INSERT INTO {mt_table} VALUES (1, 2020), (2, 2021), (3, 2022)") + create_s3_table(node, s3_table) + + # First run: schedule + wait for all partitions to complete. + node.query(f"ALTER TABLE {mt_table} EXPORT PARTITION ALL TO TABLE {s3_table}") + for partition_id in ("2020", "2021", "2022"): + wait_for_export_status(node, mt_table, s3_table, partition_id, "COMPLETED", timeout=60) + + # Empty table: throws BAD_ARGUMENTS (no active partitions). + node.query( + f"CREATE TABLE {empty_mt} (id UInt64, year UInt16)" + f" ENGINE = ReplicatedMergeTree('/clickhouse/tables/{empty_mt}', 'replica1')" + f" PARTITION BY year ORDER BY tuple()" + ) + error = node.query_and_get_error( + f"ALTER TABLE {empty_mt} EXPORT PARTITION ALL TO TABLE {s3_table}" + ) + assert "no active partitions to export" in error, ( + f"Expected 'no active partitions' error, got: {error}" + ) + + # throw_first (default): re-run aborts on the first conflicting partition. + error = node.query_and_get_error( + f"ALTER TABLE {mt_table} EXPORT PARTITION ALL TO TABLE {s3_table}" + f" SETTINGS export_merge_tree_partition_all_on_error = 'throw_first'" + ) + assert "EXPORT_PARTITION_ALREADY_EXPORTED" in error, ( + f"Expected EXPORT_PARTITION_ALREADY_EXPORTED in error, got: {error}" + ) + + # collect: aggregated PARTITION_EXPORT_FAILED message lists every conflicting partition. + error = node.query_and_get_error( + f"ALTER TABLE {mt_table} EXPORT PARTITION ALL TO TABLE {s3_table}" + f" SETTINGS export_merge_tree_partition_all_on_error = 'collect'" + ) + assert "PARTITION_EXPORT_FAILED" in error, ( + f"Expected PARTITION_EXPORT_FAILED in error, got: {error}" + ) + for partition_id in ("2020", "2021", "2022"): + assert partition_id in error, ( + f"Expected aggregated error to mention partition {partition_id}, got: {error}" + ) + + # skip_conflicts: succeeds silently because every partition conflicts and is skipped. + node.query( + f"ALTER TABLE {mt_table} EXPORT PARTITION ALL TO TABLE {s3_table}" + f" SETTINGS export_merge_tree_partition_all_on_error = 'skip_conflicts'" + ) From 15f4159a961c88aceb2e6e741cafb7a9e8642769 Mon Sep 17 00:00:00 2001 From: Andrey Zvonov <32552679+zvonand@users.noreply.github.com> Date: Wed, 5 Aug 2026 17:43:03 +0200 Subject: [PATCH 11/43] Resolve conflicts in cherry-pick of #1741 MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Renumbered the two new error codes (EXPORT_PARTITION_ALREADY_EXPORTED, PARTITION_EXPORT_FAILED) to the next free values on antalya-26.6, which already occupies 1005-1009, and bumped END accordingly. Adapted: error code numbering — antalya-26.6 already uses 1005..1009 (STALE_VERSION..PENDING_MUTATIONS_NOT_ALLOWED), so the PR's 1006/1007 became 1010/1011 and END 1007 became 1011 Source-PR: #1741 (https://github.com/Altinity/ClickHouse/pull/1741) --- src/Common/ErrorCodes.cpp | 14 +++----------- src/Core/Settings.h | 5 +---- 2 files changed, 4 insertions(+), 15 deletions(-) diff --git a/src/Common/ErrorCodes.cpp b/src/Common/ErrorCodes.cpp index cafc88ef114c..f9ecd77e997c 100644 --- a/src/Common/ErrorCodes.cpp +++ b/src/Common/ErrorCodes.cpp @@ -672,17 +672,13 @@ M(1002, UNKNOWN_EXCEPTION) \ M(1003, SSH_EXCEPTION) \ M(1004, STARTUP_SCRIPTS_ERROR) \ -<<<<<<< HEAD M(1005, STALE_VERSION) \ M(1006, INVALID_CURSOR_LOOKUP) \ M(1007, ILLEGAL_STREAM) \ M(1008, TEMPORARY_DATA_NOT_IN_CACHE) \ M(1009, PENDING_MUTATIONS_NOT_ALLOWED) \ -======= - M(1005, PENDING_MUTATIONS_NOT_ALLOWED) \ - M(1006, EXPORT_PARTITION_ALREADY_EXPORTED) \ - M(1007, PARTITION_EXPORT_FAILED) \ ->>>>>>> 12992af66fa (Merge pull request #1741 from Altinity/export_partition_all) + M(1010, EXPORT_PARTITION_ALREADY_EXPORTED) \ + M(1011, PARTITION_EXPORT_FAILED) \ /* See END */ #ifdef APPLY_FOR_EXTERNAL_ERROR_CODES @@ -699,11 +695,7 @@ namespace ErrorCodes APPLY_FOR_ERROR_CODES(M) #undef M -<<<<<<< HEAD - constexpr ErrorCode END = 1009; -======= - constexpr ErrorCode END = 1007; ->>>>>>> 12992af66fa (Merge pull request #1741 from Altinity/export_partition_all) + constexpr ErrorCode END = 1011; ErrorPairHolder values[END + 1]{}; struct ErrorCodesNames diff --git a/src/Core/Settings.h b/src/Core/Settings.h index 3883f06e4326..2214ab32aae6 100644 --- a/src/Core/Settings.h +++ b/src/Core/Settings.h @@ -122,11 +122,8 @@ class WriteBuffer; M(CLASS_NAME, JoinOrderAlgorithm) \ M(CLASS_NAME, DeduplicateInsertSelectMode) \ M(CLASS_NAME, DeduplicateInsertMode) \ -<<<<<<< HEAD - M(CLASS_NAME, FileLikeEngineDefaultPartitionStrategy) -======= + M(CLASS_NAME, FileLikeEngineDefaultPartitionStrategy) \ M(CLASS_NAME, ExportPartitionAllOnError) \ ->>>>>>> 12992af66fa (Merge pull request #1741 from Altinity/export_partition_all) COMMON_SETTINGS_SUPPORTED_TYPES(Settings, DECLARE_SETTING_TRAIT) From 1e8399c1f2d360dca97ddeb1485822d22408cf9a Mon Sep 17 00:00:00 2001 From: Andrey Zvonov <32552679+zvonand@users.noreply.github.com> Date: Thu, 21 May 2026 11:13:20 +0200 Subject: [PATCH 12/43] Cherry-pick of https://github.com/Altinity/ClickHouse/pull/1728 with unresolved conflict markers (resolution in next commit) --- Original cherry-pick message follows: Merge pull request #1728 from Altinity/export_part_respect_background_memory_limit Make export part and partition respect background tasks memory limit # Conflicts: # src/Storages/MergeTree/MergeTreeData.cpp --- src/Common/ProfileEvents.cpp | 1 + .../ExportPartitionTaskScheduler.cpp | 17 +++++++++++ src/Storages/MergeTree/MergeTreeData.cpp | 29 +++++++++++++++++++ 3 files changed, 47 insertions(+) diff --git a/src/Common/ProfileEvents.cpp b/src/Common/ProfileEvents.cpp index bcc334491632..95b64697f705 100644 --- a/src/Common/ProfileEvents.cpp +++ b/src/Common/ProfileEvents.cpp @@ -381,6 +381,7 @@ M(ExportPartitionZooKeeperRemoveRecursive, "Number of 'removeRecursive' requests to ZooKeeper made by the export partition feature.", ValueType::Number) \ M(ExportPartitionZooKeeperMulti, "Number of 'multi' requests to ZooKeeper made by the export partition feature.", ValueType::Number) \ M(ExportPartitionZooKeeperExists, "Number of 'exists' requests to ZooKeeper made by the export partition feature.", ValueType::Number) \ + M(ExportPartsRejectedByMemoryLimit, "Number of background export part tasks rejected due to background memory limit.", ValueType::Number) \ \ M(DistributedConnectionTries, "Total count of distributed connection attempts.", ValueType::Number) \ M(DistributedConnectionUsable, "Total count of successful distributed connections to a usable server (with required table, but maybe stale).", ValueType::Number) \ diff --git a/src/Storages/MergeTree/ExportPartitionTaskScheduler.cpp b/src/Storages/MergeTree/ExportPartitionTaskScheduler.cpp index 5ee5c42e3412..48f2c0e987e9 100644 --- a/src/Storages/MergeTree/ExportPartitionTaskScheduler.cpp +++ b/src/Storages/MergeTree/ExportPartitionTaskScheduler.cpp @@ -3,9 +3,11 @@ #include #include #include +#include #include #include #include +#include #include "Storages/MergeTree/ExportPartitionUtils.h" #include "Storages/MergeTree/MergeTreePartExportManifest.h" #include "Formats/FormatFactory.h" @@ -20,6 +22,7 @@ namespace ProfileEvents extern const Event ExportPartitionZooKeeperSet; extern const Event ExportPartitionZooKeeperRemove; extern const Event ExportPartitionZooKeeperMulti; + extern const Event ExportPartsRejectedByMemoryLimit; } @@ -53,6 +56,20 @@ void ExportPartitionTaskScheduler::run() return; } + /// Respect the background memory soft-limit: refuse to schedule new export-part tasks when + /// background tasks are already pressing the limit. The task is rescheduled by the parent + /// background pool a few seconds later, so this just defers work without losing it. + if (!canEnqueueBackgroundTask()) + { + ProfileEvents::increment(ProfileEvents::ExportPartsRejectedByMemoryLimit); + LOG_TRACE(storage.log, + "ExportPartition scheduler task: Reached memory limit for the background tasks ({}), " + "so won't select new parts to export. Current background tasks memory usage: {}.", + formatReadableSizeWithBinarySuffix(background_memory_tracker.getSoftLimit()), + formatReadableSizeWithBinarySuffix(background_memory_tracker.get())); + return; + } + LOG_INFO(storage.log, "ExportPartition scheduler task: Available move executors: {}", available_move_executors); std::size_t scheduled_exports_count = 0; diff --git a/src/Storages/MergeTree/MergeTreeData.cpp b/src/Storages/MergeTree/MergeTreeData.cpp index 1e28352bec5c..35fcc1a0165f 100644 --- a/src/Storages/MergeTree/MergeTreeData.cpp +++ b/src/Storages/MergeTree/MergeTreeData.cpp @@ -19,8 +19,25 @@ #include #include #include +<<<<<<< HEAD #include #include +======= +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +>>>>>>> 9fb50c7029a (Merge pull request #1728 from Altinity/export_part_respect_background_memory_limit) #include #include #include @@ -204,6 +221,7 @@ namespace ProfileEvents extern const Event PartsExportTotalMilliseconds; extern const Event PartsExportFailures; extern const Event PartsExportDuplicated; + extern const Event ExportPartsRejectedByMemoryLimit; } namespace CurrentMetrics @@ -7229,6 +7247,17 @@ void MergeTreeData::exportPartToTable( } { + if (!canEnqueueBackgroundTask()) + { + ProfileEvents::increment(ProfileEvents::ExportPartsRejectedByMemoryLimit); + throw Exception(ErrorCodes::ABORTED, + "Failed to schedule export part task for data part '{}'. " + "Reached memory limit for the background tasks ({}). Current background tasks memory usage: {}.", + part_name, + formatReadableSizeWithBinarySuffix(background_memory_tracker.getSoftLimit()), + formatReadableSizeWithBinarySuffix(background_memory_tracker.get())); + } + MergeTreePartExportManifest manifest( dest_storage, part, From bdc7adf23b31074fb6f0527e47ac010334836b4b Mon Sep 17 00:00:00 2001 From: Andrey Zvonov <32552679+zvonand@users.noreply.github.com> Date: Wed, 5 Aug 2026 17:44:44 +0200 Subject: [PATCH 13/43] Resolve conflicts in cherry-pick of #1728 Kept ours' include layout; the only source-PR include addition (Common/MemoryTracker.h) was placed into the relocated include block on antalya-26.6. Source-PR: #1728 (https://github.com/Altinity/ClickHouse/pull/1728) --- src/Storages/MergeTree/MergeTreeData.cpp | 18 +----------------- 1 file changed, 1 insertion(+), 17 deletions(-) diff --git a/src/Storages/MergeTree/MergeTreeData.cpp b/src/Storages/MergeTree/MergeTreeData.cpp index 35fcc1a0165f..f04ccb6e1bd9 100644 --- a/src/Storages/MergeTree/MergeTreeData.cpp +++ b/src/Storages/MergeTree/MergeTreeData.cpp @@ -19,25 +19,8 @@ #include #include #include -<<<<<<< HEAD #include #include -======= -#include -#include -#include -#include -#include -#include -#include -#include -#include -#include -#include -#include -#include -#include ->>>>>>> 9fb50c7029a (Merge pull request #1728 from Altinity/export_part_respect_background_memory_limit) #include #include #include @@ -138,6 +121,7 @@ #include #include #include +#include #include #include #include From cda76d4bebf389c8d6e66886686a6e1e77fcbcd6 Mon Sep 17 00:00:00 2001 From: Andrey Zvonov <32552679+zvonand@users.noreply.github.com> Date: Fri, 22 May 2026 13:50:30 +0200 Subject: [PATCH 14/43] Merge pull request #1730 from Altinity/export_partition_refactor_exceptions_observability improve observability over exceptions Source-PR: #1730 (https://github.com/Altinity/ClickHouse/pull/1730) --- docs/en/antalya/partition_export.md | 26 +- src/Core/Settings.cpp | 4 - src/Core/SettingsChangesHistory.cpp | 1 - ...portReplicatedMergeTreePartitionManifest.h | 54 ++ ...ortReplicatedMergeTreePartitionTaskEntry.h | 8 + .../ExportPartitionManifestUpdatingTask.cpp | 478 +++++------------- .../ExportPartitionManifestUpdatingTask.h | 5 +- .../ExportPartitionTaskScheduler.cpp | 4 + .../MergeTree/ExportPartitionUtils.cpp | 70 +-- src/Storages/MergeTree/ExportPartitionUtils.h | 28 +- .../tests/gtest_export_partition_ordering.cpp | 6 +- src/Storages/StorageReplicatedMergeTree.cpp | 18 +- src/Storages/StorageReplicatedMergeTree.h | 2 +- ...torageSystemReplicatedPartitionExports.cpp | 38 +- .../StorageSystemReplicatedPartitionExports.h | 11 +- .../helpers/export_partition_helpers.py | 15 +- .../test.py | 43 +- .../test.py | 25 +- 18 files changed, 362 insertions(+), 474 deletions(-) diff --git a/docs/en/antalya/partition_export.md b/docs/en/antalya/partition_export.md index 975915859482..5365ed90e353 100644 --- a/docs/en/antalya/partition_export.md +++ b/docs/en/antalya/partition_export.md @@ -6,7 +6,7 @@ The `ALTER TABLE EXPORT PARTITION` command exports entire partitions from Replic The set of parts that are exported is based on the list of parts the replica that received the export command sees. The other replicas will assist in the export process if they have those parts locally. Otherwise they will ignore it. -The partition export tasks can be observed through `system.replicated_partition_exports`. Querying this table results in a query to ZooKeeper, so it must be used with care. Individual part export progress can be observed as usual through `system.exports`. +The partition export tasks can be observed through `system.replicated_partition_exports`. The table is served from each replica's in-memory mirror, so queries do not contact ZooKeeper and are cheap to run. The mirror is refreshed on the manifest-updater poll cycle and on every status change, so a freshly written exception or terminal state may take up to one poll interval to appear. Individual part export progress can be observed as usual through `system.exports`. The same partition can not be exported to the same destination more than once. There are two ways to override this behavior: either by setting the `export_merge_tree_partition_force_export` setting or waiting for the task to expire. @@ -105,7 +105,7 @@ TO TABLE [destination_database.]destination_table - **Type**: `UInt64` - **Default**: `3600` - **Description**: The timeout is measured from the manifest's create_time. Set to 0 to disable the timeout. -When the timeout is exceeded the task transitions to KILLED (same terminal state as `KILL QUERY ... EXPORT PARTITION`), and `last_exception` is populated with a timeout reason. +When the timeout is exceeded the task transitions to KILLED (same terminal state as `KILL QUERY ... EXPORT PARTITION`), and a `last_exception_per_replica` entry on the replica that fires the timeout is populated with a timeout reason. Notes: - Enforcement is best-effort: actual kill latency is bounded by one manifest-updater poll cycle (~30s) plus ZooKeeper watch propagation. @@ -170,9 +170,7 @@ parts: ['2022_0_0_0','2022_1_1_0','2022_2_2_0'] parts_count: 3 parts_to_do: 0 status: COMPLETED -exception_replica: -last_exception: -exception_part: +last_exception_per_replica: [] exception_count: 0 Row 2: @@ -189,9 +187,7 @@ parts: ['2021_0_0_0'] parts_count: 1 parts_to_do: 0 status: COMPLETED -exception_replica: -last_exception: -exception_part: +last_exception_per_replica: [] exception_count: 0 2 rows in set. Elapsed: 0.019 sec. @@ -205,6 +201,20 @@ Status values include: - `FAILED` - Export failed - `KILLED` - Export was cancelled +### Exception columns + +- `last_exception_per_replica` is an `Array(Tuple(replica String, message String, part String, time DateTime, count UInt64))`. Each tuple is the most recent exception observed by a single replica plus a best-effort within-replica `count`. Replicas that have never reported an exception are omitted. +- `exception_count` is the sum of every `count` in `last_exception_per_replica`. Each replica owns its own counter, so cross-replica updates do not race; the sum is exact w.r.t. the snapshot returned. Within a single replica concurrent failing writers may under-count by one. + +To pick the latest exception across replicas: + +```sql +SELECT + arraySort(x -> -x.time, last_exception_per_replica)[1] AS latest_exception +FROM system.replicated_partition_exports +WHERE source_table = 'rmt_table' AND destination_table = 's3_table'; +``` + ## Related Features - [ALTER TABLE EXPORT PART](/docs/en/engines/table-engines/mergetree-family/part_export.md) - Export individual parts (non-replicated) diff --git a/src/Core/Settings.cpp b/src/Core/Settings.cpp index 6f0e6c87f934..2dc709872bbb 100644 --- a/src/Core/Settings.cpp +++ b/src/Core/Settings.cpp @@ -8075,10 +8075,6 @@ Throw an error if there are pending mutations when exporting a merge tree part. )", 0) \ DECLARE(Bool, export_merge_tree_part_throw_on_pending_patch_parts, true, R"( Throw an error if there are pending patch parts when exporting a merge tree part. -)", 0) \ - DECLARE(Bool, export_merge_tree_partition_system_table_prefer_remote_information, false, R"( -Controls whether the system.replicated_partition_exports will prefer to query ZooKeeper to get the most up to date information or use the local information. -Querying ZooKeeper is expensive, and only available if the ZooKeeper feature flag MULTI_READ is enabled. )", 0) \ DECLARE(ExportPartitionAllOnError, export_merge_tree_partition_all_on_error, ExportPartitionAllOnError::throw_first, R"( Failure handling for `ALTER TABLE ... EXPORT PARTITION ALL ...`. diff --git a/src/Core/SettingsChangesHistory.cpp b/src/Core/SettingsChangesHistory.cpp index 8ab42f498c2e..061c5076a022 100644 --- a/src/Core/SettingsChangesHistory.cpp +++ b/src/Core/SettingsChangesHistory.cpp @@ -494,7 +494,6 @@ const VersionToSettingsChangesMap & getSettingsChangesHistory() {"export_merge_tree_part_file_already_exists_policy", "skip", "skip", "New setting."}, {"export_merge_tree_part_max_bytes_per_file", 0, 0, "New setting."}, {"export_merge_tree_part_max_rows_per_file", 0, 0, "New setting."}, - {"export_merge_tree_partition_system_table_prefer_remote_information", true, true, "New setting."}, {"export_merge_tree_partition_all_on_error", "throw_first", "throw_first", "New setting."}, {"export_merge_tree_part_throw_on_pending_mutations", true, true, "New setting."}, {"export_merge_tree_part_throw_on_pending_patch_parts", true, true, "New setting."}, diff --git a/src/Storages/ExportReplicatedMergeTreePartitionManifest.h b/src/Storages/ExportReplicatedMergeTreePartitionManifest.h index dd5ef9886ded..acfabc28ca61 100644 --- a/src/Storages/ExportReplicatedMergeTreePartitionManifest.h +++ b/src/Storages/ExportReplicatedMergeTreePartitionManifest.h @@ -60,6 +60,60 @@ struct ExportReplicatedMergeTreePartitionProcessingPartEntry } }; +/// Per-task "last exception" record persisted at /last_exception. +/// +/// Single znode per export task. Updated atomically with the surrounding state +/// transition (status flip / lock release / retry counter bump) via a single +/// `tryMulti` Set op. The `count` field is best-effort and non-atomic: writers +/// `tryGet` the current value and write `count + 1` back without a version +/// check, so concurrent writers may under-count. This matches the semantics +/// used by `commit_attempts` and is documented in the system table. +struct LastExceptionEntry +{ + String message; + String part; /// empty for task-level exceptions (commit failure, timeout) + String replica; + time_t time = 0; + size_t count = 0; + + std::string toJsonString() const + { + Poco::JSON::Object json; + json.set("message", message); + json.set("part", part); + json.set("replica", replica); + json.set("time", time); + json.set("count", count); + std::ostringstream oss; // STYLE_CHECK_ALLOW_STD_STRING_STREAM + oss.exceptions(std::ios::failbit); + Poco::JSON::Stringifier::stringify(json, oss); + return oss.str(); + } + + static LastExceptionEntry fromJsonString(const std::string & json_string) + { + LastExceptionEntry entry; + if (json_string.empty()) + return entry; + + Poco::JSON::Parser parser; + auto json = parser.parse(json_string).extract(); + chassert(json); + + if (json->has("message")) + entry.message = json->getValue("message"); + if (json->has("part")) + entry.part = json->getValue("part"); + if (json->has("replica")) + entry.replica = json->getValue("replica"); + if (json->has("time")) + entry.time = json->getValue("time"); + if (json->has("count")) + entry.count = json->getValue("count"); + return entry; + } +}; + struct ExportReplicatedMergeTreePartitionProcessedPartEntry { String part_name; diff --git a/src/Storages/ExportReplicatedMergeTreePartitionTaskEntry.h b/src/Storages/ExportReplicatedMergeTreePartitionTaskEntry.h index e62f7de99bed..8af873e0b89c 100644 --- a/src/Storages/ExportReplicatedMergeTreePartitionTaskEntry.h +++ b/src/Storages/ExportReplicatedMergeTreePartitionTaskEntry.h @@ -1,5 +1,6 @@ #pragma once +#include #include #include #include "Core/QualifiedTableName.h" @@ -32,6 +33,13 @@ struct ExportReplicatedMergeTreePartitionTaskEntry /// There is also a chance this replica does not contain a given part and it is totally ok. mutable std::vector part_references; + /// In-memory mirror of /last_exception/ leaves in ZK, + /// keyed by replica name (verbatim, not escaped). Refreshed on every poll() cycle + /// and on every status-change handler invocation; served verbatim to + /// system.replicated_partition_exports without any extra ZK read. + /// An empty map means no replica has recorded an exception yet for this task. + mutable std::map last_exception_per_replica; + std::string getCompositeKey() const { const auto qualified_table_name = QualifiedTableName {manifest.destination_database, manifest.destination_table}; diff --git a/src/Storages/MergeTree/ExportPartitionManifestUpdatingTask.cpp b/src/Storages/MergeTree/ExportPartitionManifestUpdatingTask.cpp index 0ed8d1033135..106dff071188 100644 --- a/src/Storages/MergeTree/ExportPartitionManifestUpdatingTask.cpp +++ b/src/Storages/MergeTree/ExportPartitionManifestUpdatingTask.cpp @@ -7,6 +7,7 @@ #include #include #include +#include #include #include @@ -36,6 +37,86 @@ namespace FailPoints namespace { + /// Fetch all per-replica last_exception leaves under /last_exception and build + /// a fresh map keyed by replica name. The map key prefers the unescaped `replica` field + /// embedded in the JSON payload; if it is missing or empty, the leaf name is unescaped as + /// a fallback. + /// + /// An empty result means "nothing actionable": either the parent getChildren failed (ZK + /// glitch), the container has no children yet (no replica has reported), or every leaf + /// fetch came back ZNONODE / malformed. Callers MUST skip the assignment in that case to + /// preserve the in-memory mirror across transient errors. This is safe because per-replica + /// leaves are never individually removed — the entire entry path is wiped recursively when + /// a task is cleaned up, which is handled separately by removeStaleEntries. + std::map readLastExceptionPerReplica( + const zkutil::ZooKeeperPtr & zk, + const std::filesystem::path & entry_path, + const std::string & log_key, + const LoggerPtr & log) + { + std::map out; + + const auto container_path = entry_path / "last_exception"; + + Strings children; + ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRequests); + ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperGetChildren); + if (Coordination::Error::ZOK != zk->tryGetChildren(container_path, children)) + { + LOG_INFO(log, "ExportPartition Manifest Updating Task: failed to list last_exception leaves for {}, leaving in-memory copy untouched", log_key); + return out; + } + + if (children.empty()) + return out; + + std::vector paths; + paths.reserve(children.size()); + for (const auto & child : children) + paths.emplace_back(container_path / child); + + /// One MULTI_READ when supported, parallel async gets otherwise. See + /// ZooKeeper::multiRead in src/Common/ZooKeeper/ZooKeeper.h. + ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRequests); + ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperGet, paths.size()); + auto responses = zk->tryGet(paths); + responses.waitForResponses(); + + for (size_t i = 0; i < paths.size(); ++i) + { + Coordination::GetResponse response; + try + { + /// MultiTryGetResponse::operator[] swallows ZNONODE but rethrows on + /// other errors; treat any unexpected Keeper error as "skip this + /// leaf, retry on the next poll". Matches the lenient semantics of + /// the previous per-leaf tryGet implementation. + response = responses[i]; + } + catch (...) + { + LOG_WARNING(log, "ExportPartition Manifest Updating Task: ZK error fetching last_exception leaf {} for {}, skipping", children[i], log_key); + continue; + } + + if (response.error != Coordination::Error::ZOK) + continue; /// ZNONODE: child concurrently removed (recursive cleanup race). + + try + { + auto entry = LastExceptionEntry::fromJsonString(response.data); + String replica = entry.replica.empty() ? unescapeForFileName(children[i]) : entry.replica; + out.emplace(std::move(replica), std::move(entry)); + } + catch (...) + { + LOG_WARNING(log, "ExportPartition Manifest Updating Task: malformed last_exception JSON for {} (leaf {}), ignoring", log_key, children[i]); + } + } + + return out; + } + /* Remove expired entries and fix non-committed exports that have already exported all parts. @@ -178,10 +259,13 @@ namespace /// Bump commit-attempts counter; transition to FAILED once the budget is exhausted. /// This is the primary retry path for the commit phase — handlePartExportSuccess /// only fires once (on the last part's completion); subsequent retries come from here. + /// The exception is recorded in /last_exception inside the same multi. const bool became_failed = ExportPartitionUtils::handleCommitFailure( zk, entry_path, metadata.max_retries, + storage.getReplicaName(), + e.message(), log); if (became_failed) @@ -214,348 +298,13 @@ ExportPartitionManifestUpdatingTask::ExportPartitionManifestUpdatingTask(Storage std::vector ExportPartitionManifestUpdatingTask::getPartitionExportsInfo() const { - std::vector infos; - const auto zk = storage.getZooKeeper(); - - const auto exports_path = fs::path(storage.zookeeper_path) / "exports"; - - ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRequests); - ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperGetChildren); - - std::vector children; - if (Coordination::Error::ZOK != zk->tryGetChildren(exports_path, children)) - { - LOG_INFO(storage.log, "Failed to get children from exports path, returning empty export info list"); - return infos; - } - - if (children.empty()) - return infos; - - /// Batch all metadata.json, status gets, and getChildren operations in a single multi request - Coordination::Requests requests; - requests.reserve(children.size() * 4); // metadata, status, processing, exceptions_per_replica - - // Track response indices for each child - struct ChildResponseIndices - { - size_t metadata_idx; - size_t status_idx; - size_t processing_idx; - size_t exceptions_per_replica_idx; - }; - std::vector response_indices; - response_indices.reserve(children.size()); - - for (const auto & child : children) - { - const auto export_partition_path = fs::path(exports_path) / child; - - ChildResponseIndices indices; - indices.metadata_idx = requests.size(); - requests.push_back(zkutil::makeGetRequest(export_partition_path / "metadata.json")); - - indices.status_idx = requests.size(); - requests.push_back(zkutil::makeGetRequest(export_partition_path / "status")); - - indices.processing_idx = requests.size(); - requests.push_back(zkutil::makeListRequest(export_partition_path / "processing")); - - indices.exceptions_per_replica_idx = requests.size(); - requests.push_back(zkutil::makeListRequest(export_partition_path / "exceptions_per_replica")); - - response_indices.push_back(indices); - } - - ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRequests); - ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperMulti); - - Coordination::Responses responses; - Coordination::Error code = zk->tryMulti(requests, responses); - - if (code != Coordination::Error::ZOK) - { - LOG_INFO(storage.log, "Failed to execute multi request for export partition info, error: {}", code); - return infos; - } - - // Helper to extract GetResponse data - auto getGetResponseData = [&responses](size_t idx) -> std::pair - { - if (idx >= responses.size()) - return {Coordination::Error::ZRUNTIMEINCONSISTENCY, ""}; - - const auto * get_response = dynamic_cast(responses[idx].get()); - if (!get_response) - return {Coordination::Error::ZRUNTIMEINCONSISTENCY, ""}; - - return {get_response->error, get_response->data}; - }; - - // Helper to extract ListResponse data - auto getListResponseData = [&responses](size_t idx) -> std::pair - { - if (idx >= responses.size()) - return {Coordination::Error::ZRUNTIMEINCONSISTENCY, Strings{}}; - - const auto * list_response = dynamic_cast(responses[idx].get()); - if (!list_response) - return {Coordination::Error::ZRUNTIMEINCONSISTENCY, Strings{}}; - - return {list_response->error, list_response->names}; - }; - - // Create response wrappers matching the MultiTryGetResponse/MultiTryGetChildrenResponse interface - struct ResponseWrapper - { - Coordination::Error error; - std::string data; - Strings names; - - ResponseWrapper(Coordination::Error err, const std::string & d, const Strings & n) - : error(err), data(d), names(n) {} - }; - - std::vector metadata_responses_wrapper; - std::vector status_responses_wrapper; - std::vector processing_responses_wrapper; - std::vector exceptions_per_replica_responses_wrapper; - - metadata_responses_wrapper.reserve(children.size()); - status_responses_wrapper.reserve(children.size()); - processing_responses_wrapper.reserve(children.size()); - exceptions_per_replica_responses_wrapper.reserve(children.size()); - - for (size_t child_idx = 0; child_idx < children.size(); ++child_idx) - { - const auto & indices = response_indices[child_idx]; - - // Extract metadata response - auto [metadata_error, metadata_data] = getGetResponseData(indices.metadata_idx); - metadata_responses_wrapper.emplace_back(metadata_error, metadata_data, Strings{}); - - // Extract status response - auto [status_error, status_data] = getGetResponseData(indices.status_idx); - status_responses_wrapper.emplace_back(status_error, status_data, Strings{}); - - // Extract processing response - auto [processing_error, processing_names] = getListResponseData(indices.processing_idx); - processing_responses_wrapper.emplace_back(processing_error, "", processing_names); - - // Extract exceptions_per_replica response - auto [exceptions_error, exceptions_names] = getListResponseData(indices.exceptions_per_replica_idx); - exceptions_per_replica_responses_wrapper.emplace_back(exceptions_error, "", exceptions_names); - } - - // Use wrapper vectors directly - they match the interface expected by the code below - auto & metadata_responses = metadata_responses_wrapper; - auto & status_responses = status_responses_wrapper; - auto & processing_responses = processing_responses_wrapper; - auto & exceptions_per_replica_responses = exceptions_per_replica_responses_wrapper; - - /// Collect all exception replica paths for batching - struct ExceptionReplicaPath - { - size_t child_idx; - std::string replica; - std::string count_path; - std::string exception_path; - std::string part_path; - }; - - std::vector exception_replica_paths; - for (size_t child_idx = 0; child_idx < children.size(); ++child_idx) - { - const auto & child = children[child_idx]; - const auto export_partition_path = fs::path(exports_path) / child; - /// Check if we got valid responses - if (metadata_responses[child_idx].error != Coordination::Error::ZOK) - { - LOG_INFO(storage.log, "Skipping {}: missing metadata.json", child); - continue; - } - if (status_responses[child_idx].error != Coordination::Error::ZOK) - { - LOG_INFO(storage.log, "Skipping {}: missing status", child); - continue; - } - if (processing_responses[child_idx].error != Coordination::Error::ZOK) - { - LOG_INFO(storage.log, "Skipping {}: missing processing parts", child); - continue; - } - if (exceptions_per_replica_responses[child_idx].error != Coordination::Error::ZOK) - { - LOG_INFO(storage.log, "Skipping {}: missing exceptions_per_replica", export_partition_path); - continue; - } - const auto exceptions_per_replica_path = export_partition_path / "exceptions_per_replica"; - const auto & exception_replicas = exceptions_per_replica_responses[child_idx].names; - for (const auto & replica : exception_replicas) - { - const auto last_exception_path = exceptions_per_replica_path / replica / "last_exception"; - exception_replica_paths.push_back({ - child_idx, - replica, - (exceptions_per_replica_path / replica / "count").string(), - (last_exception_path / "exception").string(), - (last_exception_path / "part").string() - }); - } - } - /// Batch get all exception data in a single multi request - std::map>> exception_data_by_child; - - if (!exception_replica_paths.empty()) - { - Coordination::Requests exception_requests; - exception_requests.reserve(exception_replica_paths.size() * 3); // count, exception, part for each - - // Track response indices for each exception replica path - struct ExceptionResponseIndices - { - size_t count_idx; - size_t exception_idx; - size_t part_idx; - }; - std::vector exception_response_indices; - exception_response_indices.reserve(exception_replica_paths.size()); - - for (const auto & erp : exception_replica_paths) - { - ExceptionResponseIndices indices; - indices.count_idx = exception_requests.size(); - exception_requests.push_back(zkutil::makeGetRequest(erp.count_path)); - - indices.exception_idx = exception_requests.size(); - exception_requests.push_back(zkutil::makeGetRequest(erp.exception_path)); - - indices.part_idx = exception_requests.size(); - exception_requests.push_back(zkutil::makeGetRequest(erp.part_path)); - - exception_response_indices.push_back(indices); - } - - // Execute single multi request for all exception data - ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRequests); - ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperMulti); - - Coordination::Responses exception_responses; - Coordination::Error exception_code = zk->tryMulti(exception_requests, exception_responses); - - if (exception_code != Coordination::Error::ZOK) - { - LOG_INFO(storage.log, "Failed to execute multi request for exception data, error: {}", exception_code); - } - else - { - // Parse exception responses - for (size_t exception_path_idx = 0; exception_path_idx < exception_replica_paths.size(); ++exception_path_idx) - { - const auto & erp = exception_replica_paths[exception_path_idx]; - const auto & indices = exception_response_indices[exception_path_idx]; - - std::string count_str; - std::string exception_str; - std::string part_str; - - // Extract count response - if (indices.count_idx < exception_responses.size()) - { - const auto * count_response = dynamic_cast(exception_responses[indices.count_idx].get()); - if (count_response && count_response->error == Coordination::Error::ZOK) - count_str = count_response->data; - } - - // Extract exception response - if (indices.exception_idx < exception_responses.size()) - { - const auto * exception_response = dynamic_cast(exception_responses[indices.exception_idx].get()); - if (exception_response && exception_response->error == Coordination::Error::ZOK) - exception_str = exception_response->data; - } - - // Extract part response - if (indices.part_idx < exception_responses.size()) - { - const auto * part_response = dynamic_cast(exception_responses[indices.part_idx].get()); - if (part_response && part_response->error == Coordination::Error::ZOK) - part_str = part_response->data; - } - - exception_data_by_child[erp.child_idx].emplace_back(erp.replica, count_str, exception_str, part_str); - } - } - } - - /// Build the result - for (size_t child_idx = 0; child_idx < children.size(); ++child_idx) - { - /// Skip if we already determined this child is invalid - if (metadata_responses[child_idx].error != Coordination::Error::ZOK - || status_responses[child_idx].error != Coordination::Error::ZOK - || processing_responses[child_idx].error != Coordination::Error::ZOK - || exceptions_per_replica_responses[child_idx].error != Coordination::Error::ZOK) - { - continue; - } - - ReplicatedPartitionExportInfo info; - const auto metadata_json = metadata_responses[child_idx].data; - const auto status = status_responses[child_idx].data; - const auto processing_parts = processing_responses[child_idx].names; - const auto parts_to_do = processing_parts.size(); - std::string exception_replica; - std::string last_exception; - std::string exception_part; - std::size_t exception_count = 0; - /// Process exception data - auto exception_data_it = exception_data_by_child.find(child_idx); - if (exception_data_it != exception_data_by_child.end()) - { - for (const auto & [replica, count_str, exception_str, part_str] : exception_data_it->second) - { - if (!count_str.empty()) - { - exception_count += parse(count_str); - } - if (last_exception.empty() && !exception_str.empty() && !part_str.empty()) - { - exception_replica = replica; - last_exception = exception_str; - exception_part = part_str; - } - } - } - - const auto metadata = ExportReplicatedMergeTreePartitionManifest::fromJsonString(metadata_json); - - info.destination_database = metadata.destination_database; - info.destination_table = metadata.destination_table; - info.partition_id = metadata.partition_id; - info.transaction_id = metadata.transaction_id; - info.query_id = metadata.query_id; - info.create_time = metadata.create_time; - info.source_replica = metadata.source_replica; - info.parts_count = metadata.number_of_parts; - info.parts_to_do = parts_to_do; - info.parts = metadata.parts; - info.status = status; - info.exception_replica = exception_replica; - info.last_exception = last_exception; - info.exception_part = exception_part; - info.exception_count = exception_count; - infos.emplace_back(std::move(info)); - } - - return infos; -} - -std::vector ExportPartitionManifestUpdatingTask::getPartitionExportsInfoLocal() const -{ + /// Strictly read from the in-memory mirror; no ZooKeeper traffic. The mirror is + /// kept up to date by poll() (periodic + parent-children watch) and by the existing + /// status-change handler. See the class header comment for the convergence guarantee. std::lock_guard lock(storage.export_merge_tree_partition_mutex); std::vector infos; + infos.reserve(storage.export_merge_tree_partition_task_entries_by_key.size()); for (const auto & entry : storage.export_merge_tree_partition_task_entries_by_key) { @@ -573,6 +322,15 @@ std::vector ExportPartitionManifestUpdatingTask:: info.parts = entry.manifest.parts; info.status = magic_enum::enum_name(entry.status); + info.last_exception_per_replica.reserve(entry.last_exception_per_replica.size()); + size_t total_exception_count = 0; + for (const auto & [_, ex] : entry.last_exception_per_replica) + { + total_exception_count += ex.count; + info.last_exception_per_replica.push_back(ex); + } + info.exception_count = total_exception_count; + infos.emplace_back(std::move(info)); } @@ -625,6 +383,15 @@ void ExportPartitionManifestUpdatingTask::poll() const auto metadata = ExportReplicatedMergeTreePartitionManifest::fromJsonString(metadata_json); + /// Read last_exception leaves (no watch). Surfacing exceptions in the system table relies + /// on this read being part of every poll cycle: per-part failures during PENDING do not + /// trigger a status watch, so the only refresh path while the task is still in-flight is + /// the periodic poll. An empty result collapses every "nothing actionable" case + /// (transient ZK error, no children, all leaves ZNONODE/malformed) into a no-op so the + /// in-memory copy stays intact. + auto last_exception_per_replica = readLastExceptionPerReplica( + zk, fs::path(entry_path), key, storage.log.load()); + const auto local_entry = entries_by_key.find(key); /// If the zk entry has been replaced with export_merge_tree_partition_force_export, checking only for the export key is not enough @@ -632,9 +399,16 @@ void ExportPartitionManifestUpdatingTask::poll() bool has_local_entry_and_is_up_to_date = local_entry != entries_by_key.end() && local_entry->manifest.transaction_id == metadata.transaction_id; - /// If the entry is up to date and we don't have the cleanup lock, early exit, nothing to be done. + /// If the entry is up to date and we don't have the cleanup lock, refresh the in-memory + /// last_exception (surfaced by system.replicated_partition_exports) and early exit. + /// Direct mutation of the `mutable` field is safe under export_merge_tree_partition_mutex, + /// which is held throughout poll(). if (!cleanup_lock && has_local_entry_and_is_up_to_date) + { + if (!last_exception_per_replica.empty()) + local_entry->last_exception_per_replica = std::move(last_exception_per_replica); continue; + } std::weak_ptr weak_manifest_updater = storage.export_merge_tree_partition_manifest_updater; @@ -687,11 +461,15 @@ void ExportPartitionManifestUpdatingTask::poll() if (has_local_entry_and_is_up_to_date) { + /// Same refresh as the early-exit branch above; we also reach this point when + /// holding the cleanup lock (cleanup did not consume the entry). + if (!last_exception_per_replica.empty()) + local_entry->last_exception_per_replica = std::move(last_exception_per_replica); LOG_INFO(storage.log, "ExportPartition Manifest Updating Task: Skipping {}: already exists", key); continue; } - addTask(metadata, *status, key, entries_by_key); + addTask(metadata, *status, std::move(last_exception_per_replica), key, entries_by_key); } /// Remove entries that were deleted by someone else @@ -705,6 +483,7 @@ void ExportPartitionManifestUpdatingTask::poll() void ExportPartitionManifestUpdatingTask::addTask( const ExportReplicatedMergeTreePartitionManifest & metadata, ExportReplicatedMergeTreePartitionTaskEntry::Status status, + std::map last_exception_per_replica, const std::string & key, auto & entries_by_key ) @@ -714,7 +493,7 @@ void ExportPartitionManifestUpdatingTask::addTask( /// If the status is PENDING, we grab references to the data parts to prevent them from being deleted from the disk /// Otherwise, the operation has already been completed and there is no need to keep the data parts alive /// You might also ask: why bother adding tasks that have already been completed (i.e, status != PENDING)? - /// The reason is the `replicated_partition_exports` table in the local only mode might miss entries if they are not added here. + /// The reason is the `replicated_partition_exports` table might miss entries if they are not added here. if (status == ExportReplicatedMergeTreePartitionTaskEntry::Status::PENDING) { for (const auto & part_name : metadata.parts) @@ -727,7 +506,7 @@ void ExportPartitionManifestUpdatingTask::addTask( } /// Insert or update entry. The multi_index container automatically maintains both indexes. - auto entry = ExportReplicatedMergeTreePartitionTaskEntry {metadata, status, std::move(part_references)}; + ExportReplicatedMergeTreePartitionTaskEntry entry {metadata, status, std::move(part_references), std::move(last_exception_per_replica)}; auto it = entries_by_key.find(key); if (it != entries_by_key.end()) entries_by_key.replace(it, entry); @@ -826,6 +605,19 @@ void ExportPartitionManifestUpdatingTask::handleStatusChanges() LOG_INFO(storage.log, "ExportPartition Manifest Updating task: status changed for task {}. New status: {}", key, magic_enum::enum_name(*new_status).data()); + /// Refresh last_exception leaves too. Status transitions to FAILED (via commit budget) + /// and KILLED (via timeout) atomically write a per-replica leaf in the same multi, so + /// reading them here ensures the system table surfaces the cause together with the + /// visible state change. No new watch is added — this piggybacks on the existing + /// status watch. An empty result means "nothing actionable" and leaves the previous + /// snapshot intact. + if (auto fetched = readLastExceptionPerReplica( + zk, fs::path(storage.zookeeper_path) / "exports" / key, key, storage.log.load()); + !fetched.empty()) + { + it->last_exception_per_replica = std::move(fetched); + } + /// If status changed to KILLED, cancel local export operations if (*new_status == ExportReplicatedMergeTreePartitionTaskEntry::Status::KILLED) { diff --git a/src/Storages/MergeTree/ExportPartitionManifestUpdatingTask.h b/src/Storages/MergeTree/ExportPartitionManifestUpdatingTask.h index 855ecc334c09..32487f2dc68c 100644 --- a/src/Storages/MergeTree/ExportPartitionManifestUpdatingTask.h +++ b/src/Storages/MergeTree/ExportPartitionManifestUpdatingTask.h @@ -23,16 +23,17 @@ class ExportPartitionManifestUpdatingTask void addStatusChange(const std::string & key); + /// Returns a snapshot of every replicated partition export task tracked by this + /// replica's in-memory mirror. No ZooKeeper traffic; safe to call from query threads. std::vector getPartitionExportsInfo() const; - std::vector getPartitionExportsInfoLocal() const; - private: StorageReplicatedMergeTree & storage; void addTask( const ExportReplicatedMergeTreePartitionManifest & metadata, ExportReplicatedMergeTreePartitionTaskEntry::Status status, + std::map last_exception_per_replica, const std::string & key, auto & entries_by_key ); diff --git a/src/Storages/MergeTree/ExportPartitionTaskScheduler.cpp b/src/Storages/MergeTree/ExportPartitionTaskScheduler.cpp index 48f2c0e987e9..ef5ddbdcdbff 100644 --- a/src/Storages/MergeTree/ExportPartitionTaskScheduler.cpp +++ b/src/Storages/MergeTree/ExportPartitionTaskScheduler.cpp @@ -306,10 +306,14 @@ void ExportPartitionTaskScheduler::handlePartExportSuccess( /// Bump commit-attempts counter; transition to FAILED once the budget is exhausted. /// Prevents the task from remaining stuck in PENDING if commit() fails persistently /// (e.g. schema/spec mismatch, prolonged destination outage). + /// The exception is recorded in /last_exception via appendExceptionOps + /// inside the same multi as the commit_attempts bump and the (possible) FAILED set. const bool became_failed = ExportPartitionUtils::handleCommitFailure( zk, export_path, manifest.max_retries, + storage.replica_name, + e.message(), storage.log.load()); if (became_failed) diff --git a/src/Storages/MergeTree/ExportPartitionUtils.cpp b/src/Storages/MergeTree/ExportPartitionUtils.cpp index da11b5a11bbd..81df09c86523 100644 --- a/src/Storages/MergeTree/ExportPartitionUtils.cpp +++ b/src/Storages/MergeTree/ExportPartitionUtils.cpp @@ -2,6 +2,7 @@ #include #include #include +#include #include #include "Storages/ExportReplicatedMergeTreePartitionManifest.h" #include "Storages/ExportReplicatedMergeTreePartitionTaskEntry.h" @@ -22,7 +23,6 @@ namespace ProfileEvents extern const Event ExportPartitionZooKeeperGetChildren; extern const Event ExportPartitionZooKeeperSet; extern const Event ExportPartitionZooKeeperMulti; - extern const Event ExportPartitionZooKeeperExists; } namespace DB @@ -217,6 +217,8 @@ namespace ExportPartitionUtils const zkutil::ZooKeeperPtr & zk, const std::string & entry_path, size_t max_attempts, + const std::string & replica_name, + const std::string & exception_message, const LoggerPtr & log) { const std::string status_path = fs::path(entry_path) / "status"; @@ -257,11 +259,15 @@ namespace ExportPartitionUtils Coordination::Requests ops; + /// Record the exception in the same multi as the commit-attempts bump and the + /// (possible) FAILED transition, so the user-visible last_exception znode is + /// updated atomically with the state change that exposes it. + appendExceptionOps(ops, zk, fs::path(entry_path), replica_name, /*part_name=*/"", exception_message, log); + /// Bump the global commit_attempts counter (shared across replicas). - /// Non-atomic get+set(-1), matching exceptions_per_replica/count semantics. - /// Under a race, two replicas may see the same value and write the same +1, - /// under-counting by one. FAILED then fires one retry later than the threshold, - /// which is acceptable (we always converge to FAILED, never "never"). + /// Non-atomic get+set(-1). Under a race, two replicas may see the same value + /// and write the same +1, under-counting by one. FAILED then fires one retry + /// later than the threshold, which is acceptable. const std::string commit_attempts_path = fs::path(entry_path) / "commit_attempts"; size_t attempts = 0; @@ -336,42 +342,42 @@ namespace ExportPartitionUtils const std::string & exception_message, const LoggerPtr & log) { - const auto exceptions_per_replica_path = entry_path / "exceptions_per_replica" / replica_name; - const auto count_path = exceptions_per_replica_path / "count"; - const auto last_exception_path = exceptions_per_replica_path / "last_exception"; + /// Per-replica leaf under the `last_exception/` container created at task setup. + /// Each replica only ever writes its own leaf, so cross-replica updates never + /// race on the count. Concurrent writers within the same replica still race + /// on read+1+write (best-effort), matching the documented column semantics. + const auto last_exception_path + = entry_path / "last_exception" / escapeForFileName(replica_name); + + LastExceptionEntry entry; + std::string current_data; ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRequests); - ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperExists); - if (zk->exists(exceptions_per_replica_path)) + ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperGet); + const bool leaf_exists = zk->tryGet(last_exception_path, current_data); + if (leaf_exists) { - LOG_INFO(log, "ExportPartition: Exceptions per replica path exists, no need to create it"); - std::string num_exceptions_string; - - ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRequests); - ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperGet); - if (zk->tryGet(count_path, num_exceptions_string)) + try { - const auto num_exceptions = parse(num_exceptions_string) + 1; - ops.emplace_back(zkutil::makeSetRequest(count_path, std::to_string(num_exceptions), -1)); + entry = LastExceptionEntry::fromJsonString(current_data); } - else + catch (...) { - /// TODO maybe we should find a better way to handle this case, not urgent - LOG_INFO(log, "ExportPartition: Failed to get number of exceptions, will not increment it"); + LOG_WARNING(log, "ExportPartition: last_exception JSON at {} is malformed, resetting", last_exception_path.string()); + entry = LastExceptionEntry{}; } - - ops.emplace_back(zkutil::makeSetRequest(last_exception_path / "part", part_name, -1)); - ops.emplace_back(zkutil::makeSetRequest(last_exception_path / "exception", exception_message, -1)); } + + entry.message = exception_message; + entry.part = part_name; + entry.replica = replica_name; + entry.time = ::time(nullptr); + entry.count += 1; + + if (leaf_exists) + ops.emplace_back(zkutil::makeSetRequest(last_exception_path, entry.toJsonString(), -1)); else - { - LOG_INFO(log, "ExportPartition: Exceptions per replica path does not exist, will create it"); - ops.emplace_back(zkutil::makeCreateRequest(exceptions_per_replica_path, "", zkutil::CreateMode::Persistent)); - ops.emplace_back(zkutil::makeCreateRequest(count_path, "1", zkutil::CreateMode::Persistent)); - ops.emplace_back(zkutil::makeCreateRequest(last_exception_path, "", zkutil::CreateMode::Persistent)); - ops.emplace_back(zkutil::makeCreateRequest(last_exception_path / "part", part_name, zkutil::CreateMode::Persistent)); - ops.emplace_back(zkutil::makeCreateRequest(last_exception_path / "exception", exception_message, zkutil::CreateMode::Persistent)); - } + ops.emplace_back(zkutil::makeCreateRequest(last_exception_path, entry.toJsonString(), zkutil::CreateMode::Persistent)); } #if USE_AVRO diff --git a/src/Storages/MergeTree/ExportPartitionUtils.h b/src/Storages/MergeTree/ExportPartitionUtils.h index 411f3b5224be..d7bb83224755 100644 --- a/src/Storages/MergeTree/ExportPartitionUtils.h +++ b/src/Storages/MergeTree/ExportPartitionUtils.h @@ -47,31 +47,33 @@ namespace ExportPartitionUtils ); /// Handles a commit-phase failure for a replicated partition export: + /// - records the exception via appendExceptionOps in the same multi /// - increments /commit_attempts (lazy-created) /// - sets /status to FAILED once attempts >= max_attempts /// - /// The counter is a best-effort, non-atomic get+set(-1), matching - /// exceptions_per_replica/count. Concurrent failing commits may under-count by one - /// (FAILED may fire one retry later than the threshold), which is acceptable. - /// - /// `replica_name` and `exception` are currently unused and reserved for future - /// integration with per-replica diagnostics. + /// The counter is a best-effort, non-atomic get+set(-1). Concurrent failing + /// commits may under-count by one (FAILED may fire one retry later than the + /// threshold), which is acceptable. /// /// Returns true if this call transitioned the task to FAILED. bool handleCommitFailure( const zkutil::ZooKeeperPtr & zk, const std::string & entry_path, size_t max_attempts, + const std::string & replica_name, + const std::string & exception_message, const LoggerPtr & log); - /// Appends ZK ops to `ops` that record a per-replica exception under - /// /exceptions_per_replica//last_exception/{exception,part} - /// and increment /exceptions_per_replica//count, - /// creating the subtree if absent. + /// Appends a single ZK op to `ops` that writes the per-replica leaf + /// /last_exception/ + /// with a JSON-encoded LastExceptionEntry containing the message, part, + /// replica, time, and an incremented count. If the leaf does not yet exist + /// the op is a Create; otherwise it is a Set with version -1. /// - /// The count increment is non-atomic (synchronous tryGet + set with version -1). - /// Concurrent failing writers may under-count by one, which is accepted in this - /// subsystem and matches the pre-existing behaviour. + /// Cross-replica updates do not race: each replica only writes its own + /// leaf. Within a single replica the count increment is best-effort and + /// non-atomic (synchronous tryGet + Set with version -1); concurrent + /// failing writers may under-count by one, which is accepted. /// /// Intended to be combined with additional ops (for example a version-guarded /// status set) and executed as a single `tryMulti` so the exception record and diff --git a/src/Storages/MergeTree/tests/gtest_export_partition_ordering.cpp b/src/Storages/MergeTree/tests/gtest_export_partition_ordering.cpp index c9e3ffd9eef9..c4669a9bf55c 100644 --- a/src/Storages/MergeTree/tests/gtest_export_partition_ordering.cpp +++ b/src/Storages/MergeTree/tests/gtest_export_partition_ordering.cpp @@ -43,9 +43,9 @@ TEST_F(ExportPartitionOrderingTest, IterationOrderMatchesCreateTime) manifest3.transaction_id = "tx3"; manifest3.create_time = base_time; // Oldest - ExportReplicatedMergeTreePartitionTaskEntry entry1{manifest1, ExportReplicatedMergeTreePartitionTaskEntry::Status::PENDING, {}}; - ExportReplicatedMergeTreePartitionTaskEntry entry2{manifest2, ExportReplicatedMergeTreePartitionTaskEntry::Status::PENDING, {}}; - ExportReplicatedMergeTreePartitionTaskEntry entry3{manifest3, ExportReplicatedMergeTreePartitionTaskEntry::Status::PENDING, {}}; + ExportReplicatedMergeTreePartitionTaskEntry entry1{manifest1, ExportReplicatedMergeTreePartitionTaskEntry::Status::PENDING, {}, {}}; + ExportReplicatedMergeTreePartitionTaskEntry entry2{manifest2, ExportReplicatedMergeTreePartitionTaskEntry::Status::PENDING, {}, {}}; + ExportReplicatedMergeTreePartitionTaskEntry entry3{manifest3, ExportReplicatedMergeTreePartitionTaskEntry::Status::PENDING, {}, {}}; // Insert in reverse order by_key.insert(entry1); diff --git a/src/Storages/StorageReplicatedMergeTree.cpp b/src/Storages/StorageReplicatedMergeTree.cpp index ad8614412983..32920ac265a7 100644 --- a/src/Storages/StorageReplicatedMergeTree.cpp +++ b/src/Storages/StorageReplicatedMergeTree.cpp @@ -4784,17 +4784,9 @@ void StorageReplicatedMergeTree::exportMergeTreePartitionStatusHandlingTask() } } -std::vector StorageReplicatedMergeTree::getPartitionExportsInfo(bool prefer_remote_information) const +std::vector StorageReplicatedMergeTree::getPartitionExportsInfo() const { - /// Called from a query thread (system.replicated_partition_exports), which does not have a component set. - auto component_guard = Coordination::setCurrentComponent("StorageReplicatedMergeTree::getPartitionExportsInfo"); - - if (prefer_remote_information && getZooKeeper()->isFeatureEnabled(DB::KeeperFeatureFlag::MULTI_READ)) - { - return export_merge_tree_partition_manifest_updater->getPartitionExportsInfo(); - } - - return export_merge_tree_partition_manifest_updater->getPartitionExportsInfoLocal(); + return export_merge_tree_partition_manifest_updater->getPartitionExportsInfo(); } StorageReplicatedMergeTree::CreateMergeEntryResult StorageReplicatedMergeTree::createLogEntryToMergeParts( @@ -8808,9 +8800,11 @@ void StorageReplicatedMergeTree::exportPartitionToTable(const PartitionCommand & manifest.toJsonString(), zkutil::CreateMode::Persistent)); + /// Container for per-replica last_exception leaves; children are created lazily by the + /// first writer per replica (see ExportPartitionUtils::appendExceptionOps). ops.emplace_back(zkutil::makeCreateRequest( - fs::path(partition_exports_path) / "exceptions_per_replica", - "", + fs::path(partition_exports_path) / "last_exception", + "", zkutil::CreateMode::Persistent)); ops.emplace_back(zkutil::makeCreateRequest( diff --git a/src/Storages/StorageReplicatedMergeTree.h b/src/Storages/StorageReplicatedMergeTree.h index fc62da110449..7487b3f0a125 100644 --- a/src/Storages/StorageReplicatedMergeTree.h +++ b/src/Storages/StorageReplicatedMergeTree.h @@ -377,7 +377,7 @@ class StorageReplicatedMergeTree final : public MergeTreeData using ShutdownDeadline = std::chrono::time_point; void waitForUniquePartsToBeFetchedByOtherReplicas(ShutdownDeadline shutdown_deadline); - std::vector getPartitionExportsInfo(bool prefer_remote_information) const; + std::vector getPartitionExportsInfo() const; private: std::atomic_bool are_restoring_replica {false}; diff --git a/src/Storages/System/StorageSystemReplicatedPartitionExports.cpp b/src/Storages/System/StorageSystemReplicatedPartitionExports.cpp index e088e4f77214..9e8faab689d6 100644 --- a/src/Storages/System/StorageSystemReplicatedPartitionExports.cpp +++ b/src/Storages/System/StorageSystemReplicatedPartitionExports.cpp @@ -6,23 +6,28 @@ #include #include #include +#include #include #include #include "Columns/ColumnString.h" #include "Storages/VirtualColumnUtils.h" -#include namespace DB { -namespace Setting -{ - extern const SettingsBool export_merge_tree_partition_system_table_prefer_remote_information; -} - ColumnsDescription StorageSystemReplicatedPartitionExports::getColumnsDescription() { + auto last_exception_tuple = std::make_shared( + DataTypes{ + std::make_shared(), + std::make_shared(), + std::make_shared(), + std::make_shared(), + std::make_shared(), + }, + Names{"replica", "message", "part", "time", "count"}); + return ColumnsDescription { {"source_database", std::make_shared(), "Name of the source database."}, @@ -38,10 +43,10 @@ ColumnsDescription StorageSystemReplicatedPartitionExports::getColumnsDescriptio {"parts_count", std::make_shared(), "Number of parts in the export."}, {"parts_to_do", std::make_shared(), "Number of parts pending to be exported."}, {"status", std::make_shared(), "Status of the export."}, - {"exception_replica", std::make_shared(), "Replica that caused the last exception"}, - {"last_exception", std::make_shared(), "Last exception message of any part (not necessarily the last global exception)"}, - {"exception_part", std::make_shared(), "Part that caused the last exception"}, - {"exception_count", std::make_shared(), "Number of global exceptions"}, + {"last_exception_per_replica", std::make_shared(last_exception_tuple), + "Per-replica last exception entries. Each tuple records the most recent exception observed by that replica plus a best-effort within-replica count. Empty array if no replica has reported an exception for this task."}, + {"exception_count", std::make_shared(), + "Sum of per-replica exception counts. Each replica owns its own count, so the sum is exact w.r.t. the in-memory snapshot; within-replica updates remain best-effort and may under-count by one under concurrent failures."}, }; } @@ -117,7 +122,7 @@ void StorageSystemReplicatedPartitionExports::fillData(MutableColumns & res_colu { const IStorage * storage = replicated_merge_tree_tables[database][table].get(); if (const auto * replicated_merge_tree = dynamic_cast(storage)) - partition_exports_info = replicated_merge_tree->getPartitionExportsInfo(context->getSettingsRef()[Setting::export_merge_tree_partition_system_table_prefer_remote_information]); + partition_exports_info = replicated_merge_tree->getPartitionExportsInfo(); } for (const ReplicatedPartitionExportInfo & info : partition_exports_info) @@ -135,14 +140,17 @@ void StorageSystemReplicatedPartitionExports::fillData(MutableColumns & res_colu Array parts_array; parts_array.reserve(info.parts.size()); for (const auto & part : info.parts) - parts_array.push_back(part); + parts_array.push_back(part); res_columns[i++]->insert(parts_array); res_columns[i++]->insert(info.parts_count); res_columns[i++]->insert(info.parts_to_do); res_columns[i++]->insert(info.status); - res_columns[i++]->insert(info.exception_replica); - res_columns[i++]->insert(info.last_exception); - res_columns[i++]->insert(info.exception_part); + + Array per_replica; + per_replica.reserve(info.last_exception_per_replica.size()); + for (const auto & ex : info.last_exception_per_replica) + per_replica.push_back(Tuple{ex.replica, ex.message, ex.part, ex.time, ex.count}); + res_columns[i++]->insert(per_replica); res_columns[i++]->insert(info.exception_count); } } diff --git a/src/Storages/System/StorageSystemReplicatedPartitionExports.h b/src/Storages/System/StorageSystemReplicatedPartitionExports.h index 15eb54f38c0e..a8666374a7f0 100644 --- a/src/Storages/System/StorageSystemReplicatedPartitionExports.h +++ b/src/Storages/System/StorageSystemReplicatedPartitionExports.h @@ -1,5 +1,6 @@ #pragma once +#include #include namespace DB @@ -20,9 +21,13 @@ struct ReplicatedPartitionExportInfo size_t parts_to_do; std::vector parts; String status; - std::string exception_replica; - std::string last_exception; - std::string exception_part; + /// One entry per replica that has recorded at least one exception for this task. + /// Sourced verbatim from the in-memory mirror; no ZooKeeper traffic. + std::vector last_exception_per_replica; + /// Sum of per-replica counts. Each replica owns its own count, so cross-replica + /// updates do not race; the sum is exact w.r.t. the in-memory snapshot. Within a + /// single replica the count is best-effort (concurrent failing writers may under- + /// count by one), matching the documented column semantics. size_t exception_count = 0; }; diff --git a/tests/integration/helpers/export_partition_helpers.py b/tests/integration/helpers/export_partition_helpers.py index d6bf78df0998..46e73e8a04e6 100644 --- a/tests/integration/helpers/export_partition_helpers.py +++ b/tests/integration/helpers/export_partition_helpers.py @@ -86,10 +86,20 @@ def wait_for_exception_count( dest_table, partition_id, min_exception_count=1, - timeout=30, + timeout=60, poll_interval=0.5, ): - """Wait for exception_count to reach at least *min_exception_count*.""" + """Wait for exception_count to reach at least *min_exception_count*. + + The default timeout is intentionally larger than one manifest-updater poll + cycle (~30s, see StorageReplicatedMergeTree::exportMergeTreePartitionUpdatingTask). + system.replicated_partition_exports is served from the in-memory mirror, which + is refreshed on (a) the periodic poll tick and (b) status changes. While the + task is still PENDING (e.g. transient part-export failures with a generous + max_retries), no status watch fires, so newly written per-replica exception + leaves only become visible on the next poll. Allow at least one full cycle + plus headroom so the test is not racing the cadence. + """ start_time = time.time() last_exception_count = None while time.time() - start_time < timeout: @@ -98,7 +108,6 @@ def wait_for_exception_count( f" WHERE source_table = '{source_table}'" f" AND destination_table = '{dest_table}'" f" AND partition_id = '{partition_id}'" - f" SETTINGS export_merge_tree_partition_system_table_prefer_remote_information = 1" ).strip() if exception_count_str: diff --git a/tests/integration/test_export_replicated_mt_partition_to_iceberg/test.py b/tests/integration/test_export_replicated_mt_partition_to_iceberg/test.py index 574296f90c49..471d8d36fe73 100644 --- a/tests/integration/test_export_replicated_mt_partition_to_iceberg/test.py +++ b/tests/integration/test_export_replicated_mt_partition_to_iceberg/test.py @@ -268,7 +268,6 @@ def test_failure_is_logged_in_system_table(cluster): WHERE source_table = '{mt_table}' AND destination_table = '{iceberg_table}' AND partition_id = '2020' - SETTINGS export_merge_tree_partition_system_table_prefer_remote_information = 1 """ ).strip() assert status == "FAILED", f"Expected FAILED status, got: {status!r}" @@ -279,7 +278,6 @@ def test_failure_is_logged_in_system_table(cluster): WHERE source_table = '{mt_table}' AND destination_table = '{iceberg_table}' AND partition_id = '2020' - SETTINGS export_merge_tree_partition_system_table_prefer_remote_information = 1 """ ).strip()) assert exception_count > 0, "Expected non-zero exception_count in system.replicated_partition_exports" @@ -347,7 +345,6 @@ def test_inject_short_living_failures(cluster): WHERE source_table = '{mt_table}' AND destination_table = '{iceberg_table}' AND partition_id = '2020' - SETTINGS export_merge_tree_partition_system_table_prefer_remote_information = 1 """ ).strip()) assert exception_count >= 1, "Expected at least one transient exception to be recorded" @@ -883,23 +880,31 @@ def test_export_task_timeout_kills_stuck_pending_task(cluster): timeout=90, ) - # TODO: system.replicated_partition_exports does not currently surface - # last_exception / exception_count reliably (the engine's aggregation - # from exceptions_per_replica is incomplete). Read the raw znode via - # system.zookeeper until that is fixed. - export_key = f"2020_default.{iceberg_table}" - last_exception_path = ( - f"/clickhouse/tables/{mt_table}/exports/{export_key}" - f"/exceptions_per_replica/replica1/last_exception" - ) - last_exception = node.query( - f""" - SELECT value FROM system.zookeeper - WHERE path = '{last_exception_path}' AND name = 'exception' - """ - ).strip() + # The KILL transition writes a per-replica last_exception leaf in the same + # ZK multi as the status flip; handleStatusChanges then mirrors it into + # memory together with the status. Poll briefly to allow that watch -> + # mirror hop. We use arrayJoin to flatten the per-replica array column; + # any replica reporting the timeout reason is sufficient. + deadline = time.time() + 30 + last_exception = "" + while time.time() < deadline: + last_exception = node.query( + f""" + SELECT arrayStringConcat( + arrayMap(x -> x.message, last_exception_per_replica), + '\\n' + ) + FROM system.replicated_partition_exports + WHERE source_table = '{mt_table}' + AND destination_table = '{iceberg_table}' + AND partition_id = '2020' + """ + ).strip() + if "timed out" in last_exception: + break + time.sleep(0.5) assert "timed out" in last_exception, ( - f"Expected last_exception znode to mention the timeout reason, got: {last_exception!r}" + f"Expected last_exception_per_replica column to mention the timeout reason, got: {last_exception!r}" ) finally: node.query("SYSTEM DISABLE FAILPOINT export_partition_commit_always_throw") diff --git a/tests/integration/test_export_replicated_mt_partition_to_object_storage/test.py b/tests/integration/test_export_replicated_mt_partition_to_object_storage/test.py index 852e8cf78c72..3161e3b67100 100644 --- a/tests/integration/test_export_replicated_mt_partition_to_object_storage/test.py +++ b/tests/integration/test_export_replicated_mt_partition_to_object_storage/test.py @@ -228,18 +228,15 @@ def test_restart_nodes_during_export(cluster): assert node.query(f"SELECT count() FROM {s3_table} WHERE year = 2021") != f'0\n', "Export of partition 2021 did not resume after crash" -@pytest.mark.parametrize( - "system_table_prefer_remote_information", ['0', '1'] -) -def test_kill_export(cluster, system_table_prefer_remote_information): +def test_kill_export(cluster): skip_if_remote_database_disk_enabled(cluster) node = cluster.instances["replica1"] node2 = cluster.instances["replica2"] watcher_node = cluster.instances["watcher_node"] postfix = str(uuid.uuid4()).replace("-", "_") - mt_table = f"kill_export_mt_table_{system_table_prefer_remote_information}_{postfix}" - s3_table = f"kill_export_s3_table_{system_table_prefer_remote_information}_{postfix}" + mt_table = f"kill_export_mt_table_{postfix}" + s3_table = f"kill_export_s3_table_{postfix}" create_tables_and_insert_data(node, mt_table, s3_table, "replica1") create_tables_and_insert_data(node2, mt_table, s3_table, "replica2") @@ -316,8 +313,8 @@ def test_kill_export(cluster, system_table_prefer_remote_information): assert node.query(f"SELECT count() FROM s3(s3_conn, filename='{s3_table}/commit_2021_*', format=LineAsString)") != f'0\n', "Partition 2021 was not written to S3, but it should have been" # check system.replicated_partition_exports for the export, status should be KILLED - assert node.query(f"SELECT status FROM system.replicated_partition_exports WHERE partition_id = '2020' and source_table = '{mt_table}' and destination_table = '{s3_table}' SETTINGS export_merge_tree_partition_system_table_prefer_remote_information = {system_table_prefer_remote_information}") == 'KILLED\n', "Partition 2020 was not killed as expected" - assert node.query(f"SELECT status FROM system.replicated_partition_exports WHERE partition_id = '2021' and source_table = '{mt_table}' and destination_table = '{s3_table}' SETTINGS export_merge_tree_partition_system_table_prefer_remote_information = {system_table_prefer_remote_information}") == 'COMPLETED\n', "Partition 2021 was not completed, this is unexpected" + assert node.query(f"SELECT status FROM system.replicated_partition_exports WHERE partition_id = '2020' and source_table = '{mt_table}' and destination_table = '{s3_table}'") == 'KILLED\n', "Partition 2020 was not killed as expected" + assert node.query(f"SELECT status FROM system.replicated_partition_exports WHERE partition_id = '2021' and source_table = '{mt_table}' and destination_table = '{s3_table}'") == 'COMPLETED\n', "Partition 2021 was not completed, this is unexpected" # check the data did not land on s3 assert node.query(f"SELECT count() FROM {s3_table} WHERE year = 2020") == '0\n', "Partition 2020 was written to S3, it was not killed as expected" @@ -376,14 +373,12 @@ def test_kill_export_resilient_to_status_handling_failure(cluster): # Wait up to 15 s (5 s retry delay + margin) for the kill to propagate. wait_for_export_status(node, mt_table, s3_table, "2020", "KILLED", timeout=15) - # query the local status export_merge_tree_partition_system_table_prefer_remote_information=0 assert ( node.query( f"SELECT status FROM system.replicated_partition_exports" f" WHERE partition_id = '2020'" f" AND source_table = '{mt_table}'" f" AND destination_table = '{s3_table}'" - f" SETTINGS export_merge_tree_partition_system_table_prefer_remote_information = 0" ).strip() == "KILLED" ), "Export was not killed — status change was lost after the injected failure" @@ -537,7 +532,6 @@ def test_failure_is_logged_in_system_table(cluster): WHERE source_table = '{mt_table}' AND destination_table = '{s3_table}' AND partition_id = '2020' - SETTINGS export_merge_tree_partition_system_table_prefer_remote_information = 1 """ ) @@ -549,7 +543,6 @@ def test_failure_is_logged_in_system_table(cluster): WHERE source_table = '{mt_table}' AND destination_table = '{s3_table}' AND partition_id = '2020' - SETTINGS export_merge_tree_partition_system_table_prefer_remote_information = 1 """ ) assert int(exception_count.strip()) > 0, "Expected non-zero exception_count in system.replicated_partition_exports" @@ -600,8 +593,11 @@ def test_inject_short_living_failures(cluster): f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {s3_table} SETTINGS export_merge_tree_partition_max_retries=100;" ) - # wait for at least one exception to occur, but not enough to finish the export - wait_for_exception_count(node, mt_table, s3_table, "2020", min_exception_count=1, timeout=30) + # wait for at least one exception to occur, but not enough to finish the export. + # Use the helper default (>= one manifest-updater poll cycle): system.replicated_partition_exports + # is served from the in-memory mirror, and while the task stays PENDING the mirror only + # picks up new exception leaves on the next poll tick (~30s) — see helper docstring. + wait_for_exception_count(node, mt_table, s3_table, "2020", min_exception_count=1) # wait for the export to finish wait_for_export_status(node, mt_table, s3_table, "2020", "COMPLETED") @@ -626,7 +622,6 @@ def test_inject_short_living_failures(cluster): WHERE source_table = '{mt_table}' AND destination_table = '{s3_table}' AND partition_id = '2020' - SETTINGS export_merge_tree_partition_system_table_prefer_remote_information = 1 """ ) assert int(exception_count.strip()) >= 1, "Expected at least one exception" From c5f4ee9419aa03ce9fdbf4664f1dc73c7da02347 Mon Sep 17 00:00:00 2001 From: Andrey Zvonov <32552679+zvonand@users.noreply.github.com> Date: Thu, 28 May 2026 10:40:21 +0200 Subject: [PATCH 15/43] Merge pull request #1813 from Altinity/export_partition_commit_no_lock dont commit while holding the lock on cleanup thread Source-PR: #1813 (https://github.com/Altinity/ClickHouse/pull/1813) --- .../ExportPartitionManifestUpdatingTask.cpp | 419 +++++++++++------- 1 file changed, 251 insertions(+), 168 deletions(-) diff --git a/src/Storages/MergeTree/ExportPartitionManifestUpdatingTask.cpp b/src/Storages/MergeTree/ExportPartitionManifestUpdatingTask.cpp index 106dff071188..3ce171e5fa3e 100644 --- a/src/Storages/MergeTree/ExportPartitionManifestUpdatingTask.cpp +++ b/src/Storages/MergeTree/ExportPartitionManifestUpdatingTask.cpp @@ -37,6 +37,16 @@ namespace FailPoints namespace { + /// Work item describing a commit-recovery attempt that has been deferred out of + /// the `poll()` critical section. Captures everything by value so it can be + /// executed safely after `export_merge_tree_partition_mutex` has been released. + struct CommitRecoveryWork + { + ExportReplicatedMergeTreePartitionManifest metadata; + std::string entry_path; + StoragePtr destination_storage; + ContextPtr context; + }; /// Fetch all per-replica last_exception leaves under /last_exception and build /// a fresh map keyed by replica name. The map key prefers the unescaped `replica` field /// embedded in the JSON payload; if it is missing or empty, the leaf name is unescaped as @@ -123,6 +133,15 @@ namespace Return values: - true: the cleanup was successful, the entry is removed from the entries_by_key container and the function returns true. Proceed to the next entry. - false: the cleanup was not successful, the entry is not removed from the entries_by_key container and the function returns false. + + Side outputs: + - `deferred_commits`: when a PENDING entry has all parts processed but the export was + never committed, this function appends a CommitRecoveryWork item to be executed by + the caller after releasing the storage-wide mutex. The actual commit() call (which + performs network I/O to the destination catalog and S3) MUST NOT run under the lock. + The function still returns `false` in that case so the outer poll() loop falls through + to `addTask`, keeping the in-memory entry consistent regardless of whether the + deferred commit ultimately succeeds. */ bool tryCleanup( const zkutil::ZooKeeperPtr & zk, @@ -134,7 +153,8 @@ namespace const ExportReplicatedMergeTreePartitionManifest & metadata, const time_t now, const bool is_pending, - auto & entries_by_key + auto & entries_by_key, + std::vector & deferred_commits ) { bool has_expired = metadata.create_time < now - static_cast(metadata.ttl_seconds); @@ -234,8 +254,8 @@ namespace if (parts_in_processing_or_pending.empty()) { - LOG_INFO(log, "ExportPartition Manifest Updating Task: Cleanup found PENDING for {} with all parts exported, try to fix it by committing the export", entry_path); - + LOG_INFO(log, "ExportPartition Manifest Updating Task: Cleanup found PENDING for {} with all parts exported, deferring commit recovery to post-lock phase", entry_path); + const auto destination_storage_id = StorageID(QualifiedTableName {metadata.destination_database, metadata.destination_table}); const auto destination_storage = DatabaseCatalog::instance().tryGetTable(destination_storage_id, context); if (!destination_storage) @@ -244,46 +264,24 @@ namespace return false; } - /// it sounds like a replica exported the last part, but was not able to commit the export. Try to fix it - try - { - ExportPartitionUtils::commit(metadata, destination_storage, zk, log, entry_path, context, storage); - } - catch (const Exception & e) - { - LOG_WARNING(log, - "ExportPartition Manifest Updating Task: " - "Caught exception while committing export for {}: {}", - entry_path, e.message()); - - /// Bump commit-attempts counter; transition to FAILED once the budget is exhausted. - /// This is the primary retry path for the commit phase — handlePartExportSuccess - /// only fires once (on the last part's completion); subsequent retries come from here. - /// The exception is recorded in /last_exception inside the same multi. - const bool became_failed = ExportPartitionUtils::handleCommitFailure( - zk, - entry_path, - metadata.max_retries, - storage.getReplicaName(), - e.message(), - log); - - if (became_failed) - { - LOG_WARNING(log, - "ExportPartition Manifest Updating Task: " - "Commit for {} transitioned to FAILED after exhausting max_retries={}", - entry_path, metadata.max_retries); - } - - /// Return false so the next poll re-enters the cleanup path: - /// - if FAILED: status != PENDING on re-read, cleanup is a no-op - /// until the entry expires (handled by the first tryCleanup branch). - /// - if still PENDING: next poll increments the counter again. - return false; - } - - return true; + /// A replica exported the last part but the commit never landed. Capture everything + /// needed to run commit() outside `export_merge_tree_partition_mutex`. The + /// commit path performs network I/O (REST catalog + S3) with up to + /// MAX_TRANSACTION_RETRIES = 100 retries; holding the storage-wide mutex across + /// that work is what caused `system.replicated_partition_exports` to hang. + /// + /// Returning false here keeps the outer poll() loop on the normal path: it will + /// call addTask() so the in-memory container reflects the PENDING entry. The + /// status watch registered by poll() will transition the local entry to + /// COMPLETED/FAILED once the deferred commit (or a peer's commit) updates + /// /status in ZooKeeper. + deferred_commits.push_back(CommitRecoveryWork{ + .metadata = metadata, + .entry_path = entry_path, + .destination_storage = destination_storage, + .context = context, + }); + return false; } } @@ -298,33 +296,49 @@ ExportPartitionManifestUpdatingTask::ExportPartitionManifestUpdatingTask(Storage std::vector ExportPartitionManifestUpdatingTask::getPartitionExportsInfo() const { - /// Strictly read from the in-memory mirror; no ZooKeeper traffic. The mirror is - /// kept up to date by poll() (periodic + parent-children watch) and by the existing - /// status-change handler. See the class header comment for the convergence guarantee. - std::lock_guard lock(storage.export_merge_tree_partition_mutex); + /// Snapshot just the fields we need under the lock, then build the public + /// ReplicatedPartitionExportInfo vector after releasing it. This keeps the critical + /// section O(entries) of cheap struct copies (manifest + enum) rather than also + /// covering the formatting / Array construction performed by the caller. + struct EntrySnapshot + { + ExportReplicatedMergeTreePartitionManifest manifest; + ExportReplicatedMergeTreePartitionTaskEntry::Status status; + std::map last_exception_per_replica; + }; + + std::vector snapshots; + + { + std::lock_guard lock(storage.export_merge_tree_partition_mutex); + + snapshots.reserve(storage.export_merge_tree_partition_task_entries_by_key.size()); + for (const auto & entry : storage.export_merge_tree_partition_task_entries_by_key) + snapshots.push_back(EntrySnapshot{entry.manifest, entry.status, entry.last_exception_per_replica}); + } std::vector infos; - infos.reserve(storage.export_merge_tree_partition_task_entries_by_key.size()); + infos.reserve(snapshots.size()); - for (const auto & entry : storage.export_merge_tree_partition_task_entries_by_key) + for (auto & snapshot : snapshots) { ReplicatedPartitionExportInfo info; - info.destination_database = entry.manifest.destination_database; - info.destination_table = entry.manifest.destination_table; - info.partition_id = entry.manifest.partition_id; - info.transaction_id = entry.manifest.transaction_id; - info.query_id = entry.manifest.query_id; - info.create_time = entry.manifest.create_time; - info.source_replica = entry.manifest.source_replica; - info.parts_count = entry.manifest.number_of_parts; - info.parts_to_do = entry.manifest.parts.size(); - info.parts = entry.manifest.parts; - info.status = magic_enum::enum_name(entry.status); - - info.last_exception_per_replica.reserve(entry.last_exception_per_replica.size()); + info.destination_database = snapshot.manifest.destination_database; + info.destination_table = snapshot.manifest.destination_table; + info.partition_id = snapshot.manifest.partition_id; + info.transaction_id = snapshot.manifest.transaction_id; + info.query_id = snapshot.manifest.query_id; + info.create_time = snapshot.manifest.create_time; + info.source_replica = snapshot.manifest.source_replica; + info.parts_count = snapshot.manifest.number_of_parts; + info.parts_to_do = snapshot.manifest.parts.size(); + info.parts = std::move(snapshot.manifest.parts); + info.status = magic_enum::enum_name(snapshot.status); + + info.last_exception_per_replica.reserve(snapshot.last_exception_per_replica.size()); size_t total_exception_count = 0; - for (const auto & [_, ex] : entry.last_exception_per_replica) + for (const auto & [_, ex] : snapshot.last_exception_per_replica) { total_exception_count += ex.count; info.last_exception_per_replica.push_back(ex); @@ -339,143 +353,212 @@ std::vector ExportPartitionManifestUpdatingTask:: void ExportPartitionManifestUpdatingTask::poll() { - std::lock_guard lock(storage.export_merge_tree_partition_mutex); - - LOG_INFO(storage.log, "ExportPartition Manifest Updating Task: Polling for new entries for table {}. Current number of entries: {}", storage.getStorageID().getNameForLogs(), storage.export_merge_tree_partition_task_entries_by_key.size()); + /// Commit-recovery work collected while the storage-wide mutex is held. + /// Executed AFTER the mutex is released - committing to Iceberg/REST-catalog can take + /// many seconds (up to MAX_TRANSACTION_RETRIES=100 catalog round-trips) and blocking + /// `system.replicated_partition_exports` for that long is what we are fixing here. + std::vector deferred_commits; auto zk = storage.getZooKeeper(); - + const std::string exports_path = fs::path(storage.zookeeper_path) / "exports"; const std::string cleanup_lock_path = fs::path(storage.zookeeper_path) / "exports_cleanup_lock"; + /// The `exports_cleanup_lock` is an ephemeral ZK node that serializes cleanup work + /// across replicas: only the replica holding it walks `tryCleanup` (entry expiry + + /// commit recovery). It MUST outlive the deferred-commit loop below; otherwise a peer + /// replica's next poll() could acquire it and race us on the same commit-recovery work, + /// duplicating REST-catalog round-trips and snapshot writes. The EphemeralNodeHolder + /// destructor removes the node, so we declare it at function scope and let it die + /// at the end of poll() - after all deferred commits have completed. + /// Acquired here (no mutex needed - it is just a ZK ephemeral create). auto cleanup_lock = zkutil::EphemeralNodeHolder::tryCreate(cleanup_lock_path, *zk, storage.replica_name); if (cleanup_lock) { LOG_INFO(storage.log, "ExportPartition Manifest Updating Task: Cleanup lock acquired, will remove stale entries"); } - ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRequests); - ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperGetChildrenWatch); + { + std::lock_guard lock(storage.export_merge_tree_partition_mutex); - Coordination::Stat stat; - const auto children = zk->getChildrenWatch(exports_path, &stat, storage.export_merge_tree_partition_watch_callback); - const std::unordered_set zk_children(children.begin(), children.end()); + LOG_INFO(storage.log, "ExportPartition Manifest Updating Task: Polling for new entries for table {}. Current number of entries: {}", storage.getStorageID().getNameForLogs(), storage.export_merge_tree_partition_task_entries_by_key.size()); - const auto now = time(nullptr); + ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRequests); + ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperGetChildrenWatch); - auto & entries_by_key = storage.export_merge_tree_partition_task_entries_by_key; + Coordination::Stat stat; + const auto children = zk->getChildrenWatch(exports_path, &stat, storage.export_merge_tree_partition_watch_callback); + const std::unordered_set zk_children(children.begin(), children.end()); - /// Load new entries - /// If we have the cleanup lock, also remove stale entries from zk and local - /// Upload dangling commit files if any - for (const auto & key : zk_children) - { - const std::string entry_path = fs::path(exports_path) / key; + const auto now = time(nullptr); - ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRequests); - ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperGet); - std::string metadata_json; - if (!zk->tryGet(fs::path(entry_path) / "metadata.json", metadata_json)) - { - LOG_INFO(storage.log, "ExportPartition Manifest Updating Task: Skipping {}: missing metadata.json", key); - continue; - } + auto & entries_by_key = storage.export_merge_tree_partition_task_entries_by_key; - const auto metadata = ExportReplicatedMergeTreePartitionManifest::fromJsonString(metadata_json); - - /// Read last_exception leaves (no watch). Surfacing exceptions in the system table relies - /// on this read being part of every poll cycle: per-part failures during PENDING do not - /// trigger a status watch, so the only refresh path while the task is still in-flight is - /// the periodic poll. An empty result collapses every "nothing actionable" case - /// (transient ZK error, no children, all leaves ZNONODE/malformed) into a no-op so the - /// in-memory copy stays intact. - auto last_exception_per_replica = readLastExceptionPerReplica( - zk, fs::path(entry_path), key, storage.log.load()); - - const auto local_entry = entries_by_key.find(key); - - /// If the zk entry has been replaced with export_merge_tree_partition_force_export, checking only for the export key is not enough - /// we need to make sure it is the same transaction id. If it is not, it needs to be replaced. - bool has_local_entry_and_is_up_to_date = local_entry != entries_by_key.end() - && local_entry->manifest.transaction_id == metadata.transaction_id; - - /// If the entry is up to date and we don't have the cleanup lock, refresh the in-memory - /// last_exception (surfaced by system.replicated_partition_exports) and early exit. - /// Direct mutation of the `mutable` field is safe under export_merge_tree_partition_mutex, - /// which is held throughout poll(). - if (!cleanup_lock && has_local_entry_and_is_up_to_date) + /// Load new entries + /// If we have the cleanup lock, also remove stale entries from zk and local + /// Upload dangling commit files if any + for (const auto & key : zk_children) { - if (!last_exception_per_replica.empty()) - local_entry->last_exception_per_replica = std::move(last_exception_per_replica); - continue; - } + const std::string entry_path = fs::path(exports_path) / key; - std::weak_ptr weak_manifest_updater = storage.export_merge_tree_partition_manifest_updater; + ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRequests); + ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperGet); + std::string metadata_json; + if (!zk->tryGet(fs::path(entry_path) / "metadata.json", metadata_json)) + { + LOG_INFO(storage.log, "ExportPartition Manifest Updating Task: Skipping {}: missing metadata.json", key); + continue; + } - auto status_watch_callback = std::make_shared([weak_manifest_updater, key](const Coordination::WatchResponse &) - { - /// If the table is dropped but the watch is not removed, we need to prevent use after free - /// below code assumes that if manifest updater is still alive, the status handling task is also alive - if (auto manifest_updater = weak_manifest_updater.lock()) + const auto metadata = ExportReplicatedMergeTreePartitionManifest::fromJsonString(metadata_json); + + /// Read last_exception leaves (no watch). Surfacing exceptions in the system table relies + /// on this read being part of every poll cycle: per-part failures during PENDING do not + /// trigger a status watch, so the only refresh path while the task is still in-flight is + /// the periodic poll. An empty result collapses every "nothing actionable" case + /// (transient ZK error, no children, all leaves ZNONODE/malformed) into a no-op so the + /// in-memory copy stays intact. + auto last_exception_per_replica = readLastExceptionPerReplica( + zk, fs::path(entry_path), key, storage.log.load()); + + const auto local_entry = entries_by_key.find(key); + + /// If the zk entry has been replaced with export_merge_tree_partition_force_export, checking only for the export key is not enough + /// we need to make sure it is the same transaction id. If it is not, it needs to be replaced. + bool has_local_entry_and_is_up_to_date = local_entry != entries_by_key.end() + && local_entry->manifest.transaction_id == metadata.transaction_id; + + /// If the entry is up to date and we don't have the cleanup lock, refresh the in-memory + /// last_exception (surfaced by system.replicated_partition_exports) and early exit. + /// Direct mutation of the `mutable` field is safe under export_merge_tree_partition_mutex, + /// which is held throughout poll(). + if (!cleanup_lock && has_local_entry_and_is_up_to_date) { - manifest_updater->addStatusChange(key); - manifest_updater->storage.export_merge_tree_partition_status_handling_task->schedule(); + if (!last_exception_per_replica.empty()) + local_entry->last_exception_per_replica = std::move(last_exception_per_replica); + continue; } - }); - ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRequests); - ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperGetWatch); - std::string status_string; - if (!zk->tryGetWatch(fs::path(entry_path) / "status", status_string, nullptr, status_watch_callback)) - { - LOG_INFO(storage.log, "ExportPartition Manifest Updating Task: Skipping {}: missing status", key); - continue; - } + std::weak_ptr weak_manifest_updater = storage.export_merge_tree_partition_manifest_updater; - const auto status = magic_enum::enum_cast(status_string); - if (!status) - { - LOG_INFO(storage.log, "ExportPartition Manifest Updating Task: Invalid status {} for task {}, skipping", status_string, key); - continue; - } + auto status_watch_callback = std::make_shared([weak_manifest_updater, key](const Coordination::WatchResponse &) + { + /// If the table is dropped but the watch is not removed, we need to prevent use after free + /// below code assumes that if manifest updater is still alive, the status handling task is also alive + if (auto manifest_updater = weak_manifest_updater.lock()) + { + manifest_updater->addStatusChange(key); + manifest_updater->storage.export_merge_tree_partition_status_handling_task->schedule(); + } + }); - /// if we have the cleanup lock, try to cleanup - /// if we successfully cleaned it up, early exit - if (cleanup_lock) - { - bool cleanup_successful = tryCleanup( - zk, - entry_path, - storage.log.load(), - storage.getContext(), - storage, - key, - metadata, - now, - *status == ExportReplicatedMergeTreePartitionTaskEntry::Status::PENDING, - entries_by_key); - - if (cleanup_successful) + ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRequests); + ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperGetWatch); + std::string status_string; + if (!zk->tryGetWatch(fs::path(entry_path) / "status", status_string, nullptr, status_watch_callback)) + { + LOG_INFO(storage.log, "ExportPartition Manifest Updating Task: Skipping {}: missing status", key); continue; - } + } - if (has_local_entry_and_is_up_to_date) - { - /// Same refresh as the early-exit branch above; we also reach this point when - /// holding the cleanup lock (cleanup did not consume the entry). - if (!last_exception_per_replica.empty()) - local_entry->last_exception_per_replica = std::move(last_exception_per_replica); - LOG_INFO(storage.log, "ExportPartition Manifest Updating Task: Skipping {}: already exists", key); - continue; + const auto status = magic_enum::enum_cast(status_string); + if (!status) + { + LOG_INFO(storage.log, "ExportPartition Manifest Updating Task: Invalid status {} for task {}, skipping", status_string, key); + continue; + } + + /// if we have the cleanup lock, try to cleanup + /// if we successfully cleaned it up, early exit + if (cleanup_lock) + { + bool cleanup_successful = tryCleanup( + zk, + entry_path, + storage.log.load(), + storage.getContext(), + storage, + key, + metadata, + now, + *status == ExportReplicatedMergeTreePartitionTaskEntry::Status::PENDING, + entries_by_key, + deferred_commits); + + if (cleanup_successful) + continue; + } + + if (has_local_entry_and_is_up_to_date) + { + /// Same refresh as the early-exit branch above; we also reach this point when + /// holding the cleanup lock (cleanup did not consume the entry). + if (!last_exception_per_replica.empty()) + local_entry->last_exception_per_replica = std::move(last_exception_per_replica); + LOG_INFO(storage.log, "ExportPartition Manifest Updating Task: Skipping {}: already exists", key); + continue; + } + + addTask(metadata, *status, std::move(last_exception_per_replica), key, entries_by_key); } - addTask(metadata, *status, std::move(last_exception_per_replica), key, entries_by_key); + /// Remove entries that were deleted by someone else + removeStaleEntries(zk_children, entries_by_key); + + LOG_INFO(storage.log, "ExportPartition Manifest Updating task: finished polling for new entries. Number of entries: {}", entries_by_key.size()); } + /// `export_merge_tree_partition_mutex` released here. Everything below runs without it + /// so concurrent readers of `system.replicated_partition_exports` and other writers are + /// not blocked by the (potentially slow) catalog round-trips below. + /// + /// `cleanup_lock` (the ZK ephemeral node) is INTENTIONALLY still held here and is only + /// destructed at end of function. This preserves the existing cross-replica invariant: + /// at any moment only one replica is performing commit recovery for a given table, so + /// peer replicas will not race us on the same `commit()` calls below. + /// + /// Shutdown safety: this function runs on a BackgroundSchedulePool task that + /// `StorageReplicatedMergeTree::shutdown()` deactivates before clearing the entry + /// container. Deactivation waits for the currently-running invocation (this very call) + /// to return before proceeding, so the deferred commits below complete (or throw) before + /// any teardown observes empty `export_merge_tree_partition_task_entries`. All work + /// items capture their inputs by value, so they are independent from container state. + + const auto log_ptr = storage.log.load(); - /// Remove entries that were deleted by someone else - removeStaleEntries(zk_children, entries_by_key); + for (const auto & work : deferred_commits) + { + /// A replica exported the last part but the commit never landed. Try to fix it. + try + { + ExportPartitionUtils::commit(work.metadata, work.destination_storage, zk, log_ptr, work.entry_path, work.context, storage); + } + catch (const Exception & e) + { + LOG_WARNING(log_ptr, + "ExportPartition Manifest Updating Task: " + "Caught exception while committing export for {}: {}", + work.entry_path, e.message()); + + /// Bump commit-attempts counter; transition to FAILED once the budget is exhausted. + /// This is the primary retry path for the commit phase — handlePartExportSuccess + /// only fires once (on the last part's completion); subsequent retries come from here. + const bool became_failed = ExportPartitionUtils::handleCommitFailure( + zk, + work.entry_path, + work.metadata.max_retries, + storage.getReplicaName(), + e.message(), + log_ptr); - LOG_INFO(storage.log, "ExportPartition Manifest Updating task: finished polling for new entries. Number of entries: {}", entries_by_key.size()); + if (became_failed) + { + LOG_WARNING(log_ptr, + "ExportPartition Manifest Updating Task: " + "Commit for {} transitioned to FAILED after exhausting max_retries={}", + work.entry_path, work.metadata.max_retries); + } + } + } storage.export_merge_tree_partition_select_task->schedule(); } From 55ef77e4c6084edc4b9b2d3c4a9e9caa785ed537 Mon Sep 17 00:00:00 2001 From: Andrey Zvonov <32552679+zvonand@users.noreply.github.com> Date: Thu, 28 May 2026 15:35:26 +0200 Subject: [PATCH 16/43] Merge pull request #1836 from Altinity/export_partition_drop_allow_setting_madness Do not require iceberg insert settings on replicas Source-PR: #1836 (https://github.com/Altinity/ClickHouse/pull/1836) --- .../MergeTree/ExportPartitionUtils.cpp | 6 ++ src/Storages/MergeTree/MergeTreeData.cpp | 2 +- src/Storages/StorageReplicatedMergeTree.cpp | 4 +- .../configs/users.d/profile.xml | 1 - .../test.py | 61 +++++++++++++------ .../users.d/allow_export_partition.xml | 1 - .../test_export_partition_iceberg.py | 27 +++++--- .../test_export_partition_iceberg_catalog.py | 12 ++-- 8 files changed, 76 insertions(+), 38 deletions(-) diff --git a/src/Storages/MergeTree/ExportPartitionUtils.cpp b/src/Storages/MergeTree/ExportPartitionUtils.cpp index 81df09c86523..3badc49d65dd 100644 --- a/src/Storages/MergeTree/ExportPartitionUtils.cpp +++ b/src/Storages/MergeTree/ExportPartitionUtils.cpp @@ -86,6 +86,12 @@ namespace ExportPartitionUtils context_copy->setSetting("export_merge_tree_part_filename_pattern", manifest.filename_pattern); context_copy->setSetting("write_full_path_in_iceberg_metadata", manifest.write_full_path_in_iceberg_metadata); + /// The request-time call to exportPartitionToTable has already validated allow_insert_into_iceberg + /// against the initiator's settings. Once the manifest is in ZooKeeper, every replica must be + /// able to execute the task regardless of its own profile - otherwise an export silently + /// stalls when the setting is only set at the query level. + context_copy->setSetting("allow_insert_into_iceberg", true); + return context_copy; } diff --git a/src/Storages/MergeTree/MergeTreeData.cpp b/src/Storages/MergeTree/MergeTreeData.cpp index f04ccb6e1bd9..10414f70e663 100644 --- a/src/Storages/MergeTree/MergeTreeData.cpp +++ b/src/Storages/MergeTree/MergeTreeData.cpp @@ -7127,7 +7127,7 @@ void MergeTreeData::exportPartToTable( { throw Exception(ErrorCodes::SUPPORT_IS_DISABLED, "Iceberg writes are experimental. " - "To allow its usage, enable the setting allow_experimental_insert_into_iceberg"); + "To allow its usage, enable the setting `allow_insert_into_iceberg`."); } #if USE_AVRO diff --git a/src/Storages/StorageReplicatedMergeTree.cpp b/src/Storages/StorageReplicatedMergeTree.cpp index 32920ac265a7..2c19c4e7fec5 100644 --- a/src/Storages/StorageReplicatedMergeTree.cpp +++ b/src/Storages/StorageReplicatedMergeTree.cpp @@ -8496,7 +8496,7 @@ void StorageReplicatedMergeTree::exportPartitionToTable(const PartitionCommand & { throw Exception(ErrorCodes::SUPPORT_IS_DISABLED, "Exporting merge tree partition is experimental. Set the server setting `allow_experimental_export_merge_tree_partition` to enable it (on all replicas).\n" - "If you are exporting to an Apache Iceberg table, you also need to enable the setting `allow_experimental_insert_into_iceberg` on all replicas. The same goes for `allow_experimental_export_merge_tree_part`"); + "If you are exporting to an Apache Iceberg table, you also need to enable the setting `allow_insert_into_iceberg` on the initiator (query, session or profile) - replicas inherit it from the scheduled task."); } /// EXPORT PARTITION ALL: expand into one sub-call per active partition id. @@ -8774,7 +8774,7 @@ void StorageReplicatedMergeTree::exportPartitionToTable(const PartitionCommand & { throw Exception(ErrorCodes::SUPPORT_IS_DISABLED, "Iceberg writes are experimental. " - "To allow its usage, enable the setting allow_experimental_insert_into_iceberg (on all replicas). The same goes for `allow_experimental_export_merge_tree_partition` and `allow_experimental_export_merge_tree_part`"); + "To allow its usage, enable the setting `allow_insert_into_iceberg` on the initiator (query, session or profile) - replicas inherit it from the scheduled task."); } const auto metadata_object = iceberg_metadata->getMetadataJSON(query_context); diff --git a/tests/integration/test_export_replicated_mt_partition_to_iceberg/configs/users.d/profile.xml b/tests/integration/test_export_replicated_mt_partition_to_iceberg/configs/users.d/profile.xml index 1b50dfbdd310..518f29708929 100644 --- a/tests/integration/test_export_replicated_mt_partition_to_iceberg/configs/users.d/profile.xml +++ b/tests/integration/test_export_replicated_mt_partition_to_iceberg/configs/users.d/profile.xml @@ -2,7 +2,6 @@ 3 - 1 diff --git a/tests/integration/test_export_replicated_mt_partition_to_iceberg/test.py b/tests/integration/test_export_replicated_mt_partition_to_iceberg/test.py index 471d8d36fe73..8b49589a3005 100644 --- a/tests/integration/test_export_replicated_mt_partition_to_iceberg/test.py +++ b/tests/integration/test_export_replicated_mt_partition_to_iceberg/test.py @@ -152,7 +152,8 @@ def test_export_partition_to_iceberg(cluster): setup_tables(cluster, mt_table, iceberg_table, nodes=["replica1"]) node.query( - f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {iceberg_table}" + f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {iceberg_table}", + settings={"allow_insert_into_iceberg": 1}, ) wait_for_export_status(node, mt_table, iceberg_table, "2020", "COMPLETED") @@ -183,7 +184,8 @@ def test_export_two_partitions_to_iceberg(cluster): ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {iceberg_table}, EXPORT PARTITION ID '2021' TO TABLE {iceberg_table} - """ + """, + settings={"allow_insert_into_iceberg": 1}, ) wait_for_export_status(node, mt_table, iceberg_table, "2020", "COMPLETED") @@ -210,7 +212,10 @@ def test_export_partition_all_to_iceberg(cluster): setup_tables(cluster, mt_table, iceberg_table, nodes=["replica1"]) - node.query(f"ALTER TABLE {mt_table} EXPORT PARTITION ALL TO TABLE {iceberg_table}") + node.query( + f"ALTER TABLE {mt_table} EXPORT PARTITION ALL TO TABLE {iceberg_table}", + settings={"allow_insert_into_iceberg": 1}, + ) wait_for_export_status(node, mt_table, iceberg_table, "2020", "COMPLETED") wait_for_export_status(node, mt_table, iceberg_table, "2021", "COMPLETED") @@ -240,7 +245,7 @@ def test_failure_is_logged_in_system_table(cluster): node.query(f"SYSTEM STOP MOVES {mt_table}") - node.query(f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {iceberg_table} SETTINGS export_merge_tree_partition_max_retries = 1") + node.query(f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {iceberg_table} SETTINGS export_merge_tree_partition_max_retries = 1, allow_insert_into_iceberg = 1") with PartitionManager() as pm: pm.add_rule({ @@ -301,7 +306,7 @@ def test_inject_short_living_failures(cluster): node.query(f"SYSTEM STOP MOVES {mt_table}") - node.query(f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {iceberg_table} SETTINGS export_merge_tree_partition_max_retries = 100") + node.query(f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {iceberg_table} SETTINGS export_merge_tree_partition_max_retries = 100, allow_insert_into_iceberg = 1") with PartitionManager() as pm: pm.add_rule({ @@ -370,7 +375,8 @@ def test_export_partition_scheduler_skipped_when_moves_stopped(cluster): node.query(f"SYSTEM STOP MOVES {mt_table}") node.query( - f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {iceberg_table}" + f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {iceberg_table}", + settings={"allow_insert_into_iceberg": 1}, ) wait_for_export_to_start(node, mt_table, iceberg_table, "2020") @@ -421,7 +427,7 @@ def test_export_partition_resumes_after_stop_moves(cluster): node.query( f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {iceberg_table}" - f" SETTINGS export_merge_tree_partition_max_retries = 50" + f" SETTINGS export_merge_tree_partition_max_retries = 50, allow_insert_into_iceberg = 1" ) wait_for_export_to_start(node, mt_table, iceberg_table, "2020") @@ -466,7 +472,7 @@ def test_export_partition_resumes_after_stop_moves_during_export(cluster): node.query( f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {iceberg_table}" - f" SETTINGS export_merge_tree_partition_max_retries = 50") + f" SETTINGS export_merge_tree_partition_max_retries = 50, allow_insert_into_iceberg = 1") wait_for_export_to_start(node, mt_table, iceberg_table, "2020") @@ -527,7 +533,8 @@ def test_partition_transform_compatibility_accepted(cluster): def check_accepted(mt, iceberg, description): pid = first_partition_id(node, mt) node.query( - f"ALTER TABLE {mt} EXPORT PARTITION ID '{pid}' TO TABLE {iceberg}" + f"ALTER TABLE {mt} EXPORT PARTITION ID '{pid}' TO TABLE {iceberg}", + settings={"allow_insert_into_iceberg": 1}, ) # 1. Compound identity: (year, region) @@ -598,7 +605,8 @@ def assert_rejected(mt, iceberg, description): # The compatibility check fires synchronously; any partition ID works here. pid = first_partition_id(node, mt) error = node.query_and_get_error( - f"ALTER TABLE {mt} EXPORT PARTITION ID '{pid}' TO TABLE {iceberg}" + f"ALTER TABLE {mt} EXPORT PARTITION ID '{pid}' TO TABLE {iceberg}", + settings={"allow_insert_into_iceberg": 1}, ) assert "BAD_ARGUMENTS" in error, ( f"[{description}] Expected BAD_ARGUMENTS, got: {error!r}" @@ -688,7 +696,8 @@ def test_partition_key_compatibility_check(cluster): """ ) error = node.query_and_get_error( - f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {iceberg_col_mismatch}" + f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {iceberg_col_mismatch}", + settings={"allow_insert_into_iceberg": 1}, ) assert "BAD_ARGUMENTS" in error, ( f"Expected BAD_ARGUMENTS for partition column mismatch, got: {error!r}" @@ -709,7 +718,8 @@ def test_partition_key_compatibility_check(cluster): """ ) error = node.query_and_get_error( - f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {iceberg_count_mismatch}" + f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {iceberg_count_mismatch}", + settings={"allow_insert_into_iceberg": 1}, ) assert "BAD_ARGUMENTS" in error, ( f"Expected BAD_ARGUMENTS for partition count mismatch, got: {error!r}" @@ -731,7 +741,8 @@ def test_partition_key_compatibility_check(cluster): ) # Should not raise — the check passes so the export is accepted synchronously node.query( - f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {iceberg_match}" + f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {iceberg_match}", + settings={"allow_insert_into_iceberg": 1}, ) @@ -752,12 +763,13 @@ def test_export_ttl(cluster): # First export. node.query( f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {iceberg_table} " - f"SETTINGS export_merge_tree_partition_manifest_ttl = {ttl_seconds}" + f"SETTINGS export_merge_tree_partition_manifest_ttl = {ttl_seconds}, allow_insert_into_iceberg = 1" ) # A second export before the TTL expires must be rejected. error = node.query_and_get_error( - f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {iceberg_table}" + f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {iceberg_table}", + settings={"allow_insert_into_iceberg": 1}, ) assert "Export with key" in error, f"Expected duplicate-export error before TTL, got: {error}" @@ -771,7 +783,8 @@ def test_export_ttl(cluster): # Second export must be accepted now. node.query( - f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {iceberg_table}" + f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {iceberg_table}", + settings={"allow_insert_into_iceberg": 1}, ) wait_for_export_status(node, mt_table, iceberg_table, "2020", "COMPLETED") @@ -791,7 +804,10 @@ def test_export_data_files_are_not_cleaned_up_on_commit_failure(cluster): node.query("SYSTEM ENABLE FAILPOINT iceberg_writes_non_retry_cleanup") - node.query(f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {iceberg_table}") + node.query( + f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {iceberg_table}", + settings={"allow_insert_into_iceberg": 1}, + ) wait_for_export_status(node, mt_table, iceberg_table, "2020", "COMPLETED") count = int(node.query(f"SELECT count() FROM {iceberg_table} WHERE year = 2020").strip()) @@ -826,7 +842,10 @@ def test_post_publish_exception_preserves_snapshot(cluster): node.query("SYSTEM ENABLE FAILPOINT iceberg_writes_post_publish_throw") - node.query(f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {iceberg_table}") + node.query( + f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {iceberg_table}", + settings={"allow_insert_into_iceberg": 1}, + ) wait_for_export_status(node, mt_table, iceberg_table, "2020", "COMPLETED") count = int(node.query(f"SELECT count() FROM {iceberg_table} WHERE year = 2020").strip()) @@ -869,7 +888,8 @@ def test_export_task_timeout_kills_stuck_pending_task(cluster): f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {iceberg_table}" f" SETTINGS export_merge_tree_partition_task_timeout_seconds = 5," f" export_merge_tree_partition_max_retries = 1000000," - f" export_merge_tree_partition_manifest_ttl = 3600" + f" export_merge_tree_partition_manifest_ttl = 3600," + f" allow_insert_into_iceberg = 1" ) # Timeout budget must cover: the 5s task timeout + one manifest-updating @@ -947,7 +967,8 @@ def test_export_partition_writes_column_statistics(cluster): setup_stats_tables(node, mt_table, iceberg_table) node.query( - f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {iceberg_table}" + f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {iceberg_table}", + settings={"allow_insert_into_iceberg": 1}, ) wait_for_export_status(node, mt_table, iceberg_table, "2020", "COMPLETED") diff --git a/tests/integration/test_storage_iceberg_with_spark/configs/users.d/allow_export_partition.xml b/tests/integration/test_storage_iceberg_with_spark/configs/users.d/allow_export_partition.xml index b129913efd18..db0dd71de565 100644 --- a/tests/integration/test_storage_iceberg_with_spark/configs/users.d/allow_export_partition.xml +++ b/tests/integration/test_storage_iceberg_with_spark/configs/users.d/allow_export_partition.xml @@ -1,7 +1,6 @@ - 1 diff --git a/tests/integration/test_storage_iceberg_with_spark/test_export_partition_iceberg.py b/tests/integration/test_storage_iceberg_with_spark/test_export_partition_iceberg.py index d289639eac12..5466fe543275 100644 --- a/tests/integration/test_storage_iceberg_with_spark/test_export_partition_iceberg.py +++ b/tests/integration/test_storage_iceberg_with_spark/test_export_partition_iceberg.py @@ -183,7 +183,10 @@ def run_accepted(export_cluster, label, spark_ddl, ch_schema, rmt_columns, rmt_p node.query(f"INSERT INTO {source} VALUES {insert_values}") pid = first_partition_id(node, source) - node.query(f"ALTER TABLE {source} EXPORT PARTITION ID '{pid}' TO TABLE {iceberg}") + node.query( + f"ALTER TABLE {source} EXPORT PARTITION ID '{pid}' TO TABLE {iceberg}", + settings={"allow_insert_into_iceberg": 1}, + ) wait_for_export_status(node, source, iceberg, pid) return node, source, iceberg, pid @@ -208,7 +211,8 @@ def run_rejected(export_cluster, label, spark_ddl, ch_schema, rmt_columns, rmt_p pid = first_partition_id(node, source) error = node.query_and_get_error( - f"ALTER TABLE {source} EXPORT PARTITION ID '{pid}' TO TABLE {iceberg}" + f"ALTER TABLE {source} EXPORT PARTITION ID '{pid}' TO TABLE {iceberg}", + settings={"allow_insert_into_iceberg": 1}, ) return error @@ -691,7 +695,10 @@ def test_idempotency_after_commit_crash(export_cluster): # injection point (after a successful Iceberg commit), std::terminate() is called # and the process exits immediately without setting ZK COMPLETED. node.query("SYSTEM ENABLE FAILPOINT iceberg_export_after_commit_before_zk_completed") - node.query(f"ALTER TABLE {source} EXPORT PARTITION ID '{pid}' TO TABLE {iceberg}") + node.query( + f"ALTER TABLE {source} EXPORT PARTITION ID '{pid}' TO TABLE {iceberg}", + settings={"allow_insert_into_iceberg": 1}, + ) # the fail point will sleep for 10 seconds. Wait for 5 and then re-start clickhouse. time.sleep(5) # Restart ClickHouse. The ZK task is still PENDING; the scheduler will pick it up. @@ -757,7 +764,8 @@ def test_commit_attempts_budget_transitions_to_failed(export_cluster): # to exhaust the budget and flip the task to FAILED. node.query( f"ALTER TABLE {source} EXPORT PARTITION ID '{pid}' TO TABLE {iceberg}" - f" SETTINGS export_merge_tree_partition_max_retries = 2" + f" SETTINGS export_merge_tree_partition_max_retries = 2," + f" allow_insert_into_iceberg = 1" ) # Timeout must cover: at least one manifest-updating poll cycle (30s) @@ -807,7 +815,10 @@ def test_export_initiated_from_replica2(export_cluster): r1.query(f"INSERT INTO {mt_table} VALUES (1, 2020), (2, 2020), (3, 2020)") r2.query(f"SYSTEM SYNC REPLICA {mt_table}") - r2.query(f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {iceberg_table}") + r2.query( + f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {iceberg_table}", + settings={"allow_insert_into_iceberg": 1}, + ) wait_for_export_status(r2, mt_table, iceberg_table, "2020") count_r1 = int(r1.query(f"SELECT count() FROM {iceberg_table}").strip()) @@ -846,7 +857,8 @@ def test_concurrent_exports_different_partitions_across_replicas(export_cluster) def export_from(node, pid): try: node.query( - f"ALTER TABLE {mt_table} EXPORT PARTITION ID '{pid}' TO TABLE {iceberg_table}" + f"ALTER TABLE {mt_table} EXPORT PARTITION ID '{pid}' TO TABLE {iceberg_table}", + settings={"allow_insert_into_iceberg": 1}, ) wait_for_export_status(node, mt_table, iceberg_table, pid) except Exception as exc: @@ -895,7 +907,8 @@ def test_three_replica_concurrent_exports(export_cluster): def export_fn(node_pid): node, pid = node_pid node.query( - f"ALTER TABLE {mt_table} EXPORT PARTITION ID '{pid}' TO TABLE {iceberg_table}" + f"ALTER TABLE {mt_table} EXPORT PARTITION ID '{pid}' TO TABLE {iceberg_table}", + settings={"allow_insert_into_iceberg": 1}, ) wait_for_export_status(node, mt_table, iceberg_table, pid) diff --git a/tests/integration/test_storage_iceberg_with_spark/test_export_partition_iceberg_catalog.py b/tests/integration/test_storage_iceberg_with_spark/test_export_partition_iceberg_catalog.py index 9d0cc557dc18..8a216860d029 100644 --- a/tests/integration/test_storage_iceberg_with_spark/test_export_partition_iceberg_catalog.py +++ b/tests/integration/test_storage_iceberg_with_spark/test_export_partition_iceberg_catalog.py @@ -231,7 +231,7 @@ def test_catalog_basic_export(catalog_export_cluster): node.query( f"ALTER TABLE {source} EXPORT PARTITION ID '{pid}' TO TABLE {dest_ch}", - settings={"write_full_path_in_iceberg_metadata": 1}, + settings={"write_full_path_in_iceberg_metadata": 1, "allow_insert_into_iceberg": 1}, ) wait_for_export_status(node, source, None, pid) @@ -279,7 +279,7 @@ def export_partition(pid: str) -> None: try: node.query( f"ALTER TABLE {source} EXPORT PARTITION ID '{pid}' TO TABLE {dest_ch}", - settings={"write_full_path_in_iceberg_metadata": 1}, + settings={"write_full_path_in_iceberg_metadata": 1, "allow_insert_into_iceberg": 1}, ) wait_for_export_status(node, source, None, pid, timeout=120) except Exception as exc: @@ -340,7 +340,7 @@ def test_catalog_idempotent_retry(catalog_export_cluster): node.query("SYSTEM ENABLE FAILPOINT iceberg_export_after_commit_before_zk_completed") node.query( f"ALTER TABLE {source} EXPORT PARTITION ID '{pid}' TO TABLE {dest_ch}", - settings={"write_full_path_in_iceberg_metadata": 1}, + settings={"write_full_path_in_iceberg_metadata": 1, "allow_insert_into_iceberg": 1}, ) # Give the background scheduler time to export the data files and reach the @@ -401,7 +401,7 @@ def test_catalog_export_two_replicas_basic(catalog_export_cluster): r1.query( f"ALTER TABLE {source} EXPORT PARTITION ID '{pid}' TO TABLE {dest_ch}", - settings={"write_full_path_in_iceberg_metadata": 1}, + settings={"write_full_path_in_iceberg_metadata": 1, "allow_insert_into_iceberg": 1}, ) wait_for_export_status(r1, source, None, pid) @@ -447,7 +447,7 @@ def export_partition(node, pid): try: node.query( f"ALTER TABLE {source} EXPORT PARTITION ID '{pid}' TO TABLE {dest_ch}", - settings={"write_full_path_in_iceberg_metadata": 1}, + settings={"write_full_path_in_iceberg_metadata": 1, "allow_insert_into_iceberg": 1}, ) wait_for_export_status(node, source, None, pid, timeout=120) except Exception as exc: @@ -505,7 +505,7 @@ def export_partition(node, pid): # try: # node.query( # f"ALTER TABLE {source} EXPORT PARTITION ID '{pid}' TO TABLE {dest_ch}", -# settings={"write_full_path_in_iceberg_metadata": 1}, +# settings={"write_full_path_in_iceberg_metadata": 1, "allow_insert_into_iceberg": 1}, # ) # wait_for_export_status(node, source, None, pid, timeout=120) # except Exception as exc: From 79f0cfd35281fa5b7d252a8a74899043e5bb1910 Mon Sep 17 00:00:00 2001 From: Andrey Zvonov <32552679+zvonand@users.noreply.github.com> Date: Mon, 1 Jun 2026 14:52:39 +0200 Subject: [PATCH 17/43] Merge pull request #1847 from Altinity/export-partition-commit-lock Export partition - Introduce commit lock Source-PR: #1847 (https://github.com/Altinity/ClickHouse/pull/1847) --- .../ExportPartitionManifestUpdatingTask.cpp | 2 +- .../MergeTree/ExportPartitionTaskScheduler.cpp | 2 +- src/Storages/MergeTree/ExportPartitionUtils.cpp | 16 +++++++++++++++- src/Storages/MergeTree/ExportPartitionUtils.h | 3 ++- 4 files changed, 19 insertions(+), 4 deletions(-) diff --git a/src/Storages/MergeTree/ExportPartitionManifestUpdatingTask.cpp b/src/Storages/MergeTree/ExportPartitionManifestUpdatingTask.cpp index 3ce171e5fa3e..b45f3667be87 100644 --- a/src/Storages/MergeTree/ExportPartitionManifestUpdatingTask.cpp +++ b/src/Storages/MergeTree/ExportPartitionManifestUpdatingTask.cpp @@ -530,7 +530,7 @@ void ExportPartitionManifestUpdatingTask::poll() /// A replica exported the last part but the commit never landed. Try to fix it. try { - ExportPartitionUtils::commit(work.metadata, work.destination_storage, zk, log_ptr, work.entry_path, work.context, storage); + ExportPartitionUtils::commit(work.metadata, work.destination_storage, zk, log_ptr, work.entry_path, work.context, storage, storage.getReplicaName()); } catch (const Exception & e) { diff --git a/src/Storages/MergeTree/ExportPartitionTaskScheduler.cpp b/src/Storages/MergeTree/ExportPartitionTaskScheduler.cpp index ef5ddbdcdbff..d722fb77b2c0 100644 --- a/src/Storages/MergeTree/ExportPartitionTaskScheduler.cpp +++ b/src/Storages/MergeTree/ExportPartitionTaskScheduler.cpp @@ -297,7 +297,7 @@ void ExportPartitionTaskScheduler::handlePartExportSuccess( try { auto context = ExportPartitionUtils::getContextCopyWithTaskSettings(storage.getContext(), manifest); - ExportPartitionUtils::commit(manifest, destination_storage, zk, storage.log.load(), export_path, context, storage); + ExportPartitionUtils::commit(manifest, destination_storage, zk, storage.log.load(), export_path, context, storage, storage.replica_name); } catch (const Exception & e) { diff --git a/src/Storages/MergeTree/ExportPartitionUtils.cpp b/src/Storages/MergeTree/ExportPartitionUtils.cpp index 3badc49d65dd..4a447c8de3ab 100644 --- a/src/Storages/MergeTree/ExportPartitionUtils.cpp +++ b/src/Storages/MergeTree/ExportPartitionUtils.cpp @@ -158,7 +158,8 @@ namespace ExportPartitionUtils const LoggerPtr & log, const std::string & entry_path, const ContextPtr & context_in, - MergeTreeData & source_storage) + MergeTreeData & source_storage, + const String & replica_name) { auto context = Context::createCopy(context_in); context->setSetting("write_full_path_in_iceberg_metadata", manifest.write_full_path_in_iceberg_metadata); @@ -171,6 +172,19 @@ namespace ExportPartitionUtils "Failpoint: export_partition_commit_always_throw"); }); + /// Per-task ephemeral lock that serializes the commit phase across replicas. + /// Without it, `handlePartExportSuccess` (post-last-part path) and `tryCleanup` + /// (poll/recovery path) can drive `commitExportPartitionTransaction` concurrently + /// for the same task. + const auto commit_lock_path = fs::path(entry_path) / "commit_lock"; + auto commit_lock = zkutil::EphemeralNodeHolder::tryCreate(commit_lock_path, *zk, replica_name); + if (!commit_lock) + { + LOG_INFO(log, "ExportPartition: commit_lock for {} is held by another replica, skipping commit on this replica", entry_path); + return; + } + LOG_INFO(log, "ExportPartition: commit_lock for {} acquired by replica {}", entry_path, replica_name); + const auto exported_paths = ExportPartitionUtils::getExportedPaths(log, zk, entry_path); if (exported_paths.empty()) diff --git a/src/Storages/MergeTree/ExportPartitionUtils.h b/src/Storages/MergeTree/ExportPartitionUtils.h index d7bb83224755..eb67d288d71e 100644 --- a/src/Storages/MergeTree/ExportPartitionUtils.h +++ b/src/Storages/MergeTree/ExportPartitionUtils.h @@ -43,7 +43,8 @@ namespace ExportPartitionUtils const LoggerPtr & log, const std::string & entry_path, const ContextPtr & context, - MergeTreeData & source_storage + MergeTreeData & source_storage, + const String & replica_name ); /// Handles a commit-phase failure for a replicated partition export: From eb45735724167273b4b9be745e32aa0687764053 Mon Sep 17 00:00:00 2001 From: Andrey Zvonov <32552679+zvonand@users.noreply.github.com> Date: Mon, 1 Jun 2026 17:55:40 +0200 Subject: [PATCH 18/43] Cherry-pick of https://github.com/Altinity/ClickHouse/pull/1782 with unresolved conflict markers (resolution in next commit) --- Original cherry-pick message follows: Merge pull request #1782 from Altinity/frontport/antalya-26.3/json_part2 Antalya 26.3: Cluster Joins part 2 - global mode # Conflicts: # src/Storages/buildQueryTreeForShard.cpp # src/Storages/buildQueryTreeForShard.h # tests/integration/test_database_iceberg/test.py # tests/integration/test_s3_cluster/test.py --- src/Core/Settings.cpp | 2 +- src/Storages/IStorageCluster.cpp | 88 +++++++-- src/Storages/IStorageCluster.h | 7 +- src/Storages/StorageDistributed.cpp | 52 +---- src/Storages/buildQueryTreeForShard.cpp | 132 ++++++++++++- src/Storages/buildQueryTreeForShard.h | 11 +- .../integration/test_database_iceberg/test.py | 15 +- tests/integration/test_s3_cluster/test.py | 178 +++++++++++++++++- .../conftest.py | 3 + .../test_cluster_joins.py | 106 ++++++++++- 10 files changed, 491 insertions(+), 103 deletions(-) diff --git a/src/Core/Settings.cpp b/src/Core/Settings.cpp index 2dc709872bbb..3dd6dd861157 100644 --- a/src/Core/Settings.cpp +++ b/src/Core/Settings.cpp @@ -2133,7 +2133,7 @@ ClickHouse applies this setting when the query contains the product of object st Possible values: - `local` — Replaces the database and table in the subquery with local ones for the destination server (shard), leaving the normal `IN`/`JOIN.` -- `global` — Unsupported for now. Replaces the `IN`/`JOIN` query with `GLOBAL IN`/`GLOBAL JOIN.` +- `global` — Replaces the `IN`/`JOIN` query with `GLOBAL IN`/`GLOBAL JOIN.` Right table executes first and is added to the secondary query as temporay table. - `allow` — Default value. Allows the use of these types of subqueries. )", 0) \ \ diff --git a/src/Storages/IStorageCluster.cpp b/src/Storages/IStorageCluster.cpp index 1d7db845af04..5f22c64fb3dd 100644 --- a/src/Storages/IStorageCluster.cpp +++ b/src/Storages/IStorageCluster.cpp @@ -17,12 +17,14 @@ #include #include #include +#include #include #include #include #include #include #include +#include #include #include #include @@ -118,11 +120,14 @@ class SearcherVisitor : public InDepthQueryTreeVisitorWithContext; using Base::Base; - explicit SearcherVisitor(std::unordered_set types_, ContextPtr context) : Base(context), types(types_) {} + explicit SearcherVisitor(std::unordered_set types_, size_t entry_, ContextPtr context) + : Base(context) + , types(types_) + , entry(entry_) {} bool needChildVisit(QueryTreeNodePtr & /*parent*/, QueryTreeNodePtr & /*child*/) { - return getSubqueryDepth() <= 2 && !passed_node; + return getSubqueryDepth() <= 2 && !passed_node && !current_entry; } void enterImpl(QueryTreeNodePtr & node) @@ -133,13 +138,19 @@ class SearcherVisitor : public InDepthQueryTreeVisitorWithContextgetNodeType(); if (types.contains(node_type)) - passed_node = node; + { + ++current_entry; + if (current_entry == entry) + passed_node = node; + } } QueryTreeNodePtr getNode() const { return passed_node; } private: std::unordered_set types; + size_t entry; + size_t current_entry = 0; QueryTreeNodePtr passed_node; }; @@ -206,15 +217,24 @@ Converts localtable as t ON s3.key == t.key -to +to (object_storage_cluster_join_mode='local') SELECT s3.c1, s3.c2, s3.key FROM s3Cluster(...) AS s3 + +or (object_storage_cluster_join_mode='global') + + SELECT s3.c1, s3.c2, t.c3 + FROM + s3Cluster(...) as s3 + JOIN + values('key UInt32, data String', (1, 'one'), (2, 'two'), ...) as t + ON s3.key == t.key */ void IStorageCluster::updateQueryWithJoinToSendIfNeeded( ASTPtr & query_to_send, - QueryTreeNodePtr query_tree, + SelectQueryInfo query_info, const ContextPtr & context) { auto object_storage_cluster_join_mode = context->getSettingsRef()[Setting::object_storage_cluster_join_mode]; @@ -226,17 +246,17 @@ void IStorageCluster::updateQueryWithJoinToSendIfNeeded( throw Exception(ErrorCodes::NOT_IMPLEMENTED, "object_storage_cluster_join_mode!='allow' is not supported without allow_experimental_analyzer=true"); - auto info = getQueryTreeInfo(query_tree, context); + auto info = getQueryTreeInfo(query_info.query_tree, context); if (info.has_join || info.has_cross_join || info.has_local_columns_in_where) { - auto modified_query_tree = query_tree->clone(); + auto modified_query_tree = query_info.query_tree->clone(); - SearcherVisitor left_table_expression_searcher({QueryTreeNodeType::TABLE, QueryTreeNodeType::TABLE_FUNCTION}, context); + SearcherVisitor left_table_expression_searcher({QueryTreeNodeType::TABLE, QueryTreeNodeType::TABLE_FUNCTION}, 1, context); left_table_expression_searcher.visit(modified_query_tree); auto table_function_node = left_table_expression_searcher.getNode(); if (!table_function_node) - throw Exception(ErrorCodes::LOGICAL_ERROR, "Can't find table function node"); + throw Exception(ErrorCodes::LOGICAL_ERROR, "Can't find left table function node"); QueryTreeNodePtr query_tree_distributed; @@ -249,7 +269,7 @@ void IStorageCluster::updateQueryWithJoinToSendIfNeeded( } else if (info.has_cross_join) { - SearcherVisitor join_searcher({QueryTreeNodeType::CROSS_JOIN}, context); + SearcherVisitor join_searcher({QueryTreeNodeType::CROSS_JOIN}, 1, context); join_searcher.visit(modified_query_tree); auto cross_join_node = join_searcher.getNode(); if (!cross_join_node) @@ -304,8 +324,28 @@ void IStorageCluster::updateQueryWithJoinToSendIfNeeded( return; } case ObjectStorageClusterJoinMode::GLOBAL: - // TODO - throw Exception(ErrorCodes::NOT_IMPLEMENTED, "`Global` mode for `object_storage_cluster_join_mode` setting is unimplemented for now"); + { + auto info = getQueryTreeInfo(query_info.query_tree, context); + + if (info.has_join || info.has_cross_join || info.has_local_columns_in_where) + { + auto modified_query_tree = query_info.query_tree->clone(); + + rewriteJoinToGlobalJoin(modified_query_tree, context, /*force_prefer_global_join*/ true); + + if (info.has_local_columns_in_where) + rewriteInToGlobalIn(modified_query_tree, context, /*rewrite_for_distributed*/ true); + + modified_query_tree = buildQueryTreeForShard( + query_info.planner_context, + modified_query_tree, + /*allow_global_join_for_right_table*/ false, + /*find_cross_join*/ true); + query_to_send = queryNodeToDistributedSelectQuery(modified_query_tree); + } + + return; + } case ObjectStorageClusterJoinMode::ALLOW: // Do nothing special return; } @@ -343,7 +383,7 @@ void IStorageCluster::read( /// rewrite query to execute `remote('remote_host', s3(...))` /// remote_host can execute query itself or make on-cluster query depends on own `object_storage_cluster` setting updateConfigurationIfNeeded(context); - updateQueryWithJoinToSendIfNeeded(query_to_send, query_info.query_tree, context); + updateQueryWithJoinToSendIfNeeded(query_to_send, query_info, context); updateQueryToSendIfNeeded(query_to_send, storage_snapshot, context, /*make_cluster_function*/ false); auto remote_initiator_cluster = getClusterImpl(context, remote_initiator_cluster_name); @@ -368,7 +408,7 @@ void IStorageCluster::read( SharedHeader sample_block; - updateQueryWithJoinToSendIfNeeded(query_to_send, query_info.query_tree, context); + updateQueryWithJoinToSendIfNeeded(query_to_send, query_info, context); if (settings[Setting::allow_experimental_analyzer]) { @@ -420,6 +460,10 @@ void IStorageCluster::read( auto this_ptr = std::static_pointer_cast(shared_from_this()); + std::optional external_tables = std::nullopt; + if (query_info.planner_context && query_info.planner_context->getMutableQueryContext()) + external_tables = query_info.planner_context->getMutableQueryContext()->getExternalTables(); + auto reading = std::make_unique( column_names, query_info, @@ -430,7 +474,8 @@ void IStorageCluster::read( std::move(query_to_send), processed_stage, cluster, - log); + log, + external_tables); query_plan.addStep(std::move(reading)); } @@ -572,7 +617,7 @@ void ReadFromCluster::initializePipeline(QueryPipelineBuilder & pipeline, const new_context, /*throttler=*/nullptr, scalars, - Tables(), + external_tables.has_value() ? *external_tables : Tables(), processed_stage, nullptr, RemoteQueryExecutor::Extension{.task_iterator = extension->task_iterator, .replica_info = std::move(replica_info)}, @@ -611,7 +656,7 @@ IStorageCluster::QueryTreeInfo IStorageCluster::getQueryTreeInfo(QueryTreeNodePt info.has_cross_join = true; } - SearcherVisitor left_table_expression_searcher({QueryTreeNodeType::TABLE, QueryTreeNodeType::TABLE_FUNCTION}, context); + SearcherVisitor left_table_expression_searcher({QueryTreeNodeType::TABLE, QueryTreeNodeType::TABLE_FUNCTION}, 1, context); left_table_expression_searcher.visit(query_tree); auto table_function_node = left_table_expression_searcher.getNode(); if (!table_function_node) @@ -646,9 +691,12 @@ QueryProcessingStage::Enum IStorageCluster::getQueryProcessingStage( throw Exception(ErrorCodes::NOT_IMPLEMENTED, "object_storage_cluster_join_mode!='allow' is not supported without allow_experimental_analyzer=true"); - auto info = getQueryTreeInfo(query_info.query_tree, context); - if (info.has_join || info.has_cross_join || info.has_local_columns_in_where) - return QueryProcessingStage::Enum::FetchColumns; + if (object_storage_cluster_join_mode == ObjectStorageClusterJoinMode::LOCAL) + { + auto info = getQueryTreeInfo(query_info.query_tree, context); + if (info.has_join || info.has_cross_join || info.has_local_columns_in_where) + return QueryProcessingStage::Enum::FetchColumns; + } } /// Initiator executes query on remote node. diff --git a/src/Storages/IStorageCluster.h b/src/Storages/IStorageCluster.h index e42c841baf14..d4d7c9da7f29 100644 --- a/src/Storages/IStorageCluster.h +++ b/src/Storages/IStorageCluster.h @@ -69,7 +69,7 @@ class IStorageCluster : public IStorage const StorageSnapshotPtr & /*storage_snapshot*/, const ContextPtr & /*context*/, bool /*make_cluster_function*/) {} - void updateQueryWithJoinToSendIfNeeded(ASTPtr & query_to_send, QueryTreeNodePtr query_tree, const ContextPtr & context); + void updateQueryWithJoinToSendIfNeeded(ASTPtr & query_to_send, SelectQueryInfo query_info, const ContextPtr & context); virtual void updateConfigurationIfNeeded(ContextPtr /* context */) {} @@ -143,7 +143,8 @@ class ReadFromCluster : public SourceStepWithFilter ASTPtr query_to_send_, QueryProcessingStage::Enum processed_stage_, ClusterPtr cluster_, - LoggerPtr log_) + LoggerPtr log_, + std::optional external_tables_) : SourceStepWithFilter( std::move(sample_block), column_names_, @@ -155,6 +156,7 @@ class ReadFromCluster : public SourceStepWithFilter , processed_stage(processed_stage_) , cluster(std::move(cluster_)) , log(log_) + , external_tables(external_tables_) { } @@ -166,6 +168,7 @@ class ReadFromCluster : public SourceStepWithFilter LoggerPtr log; std::optional extension; + std::optional external_tables; void createExtension(const ActionsDAG::Node * predicate); ContextPtr updateSettings(const Settings & settings); diff --git a/src/Storages/StorageDistributed.cpp b/src/Storages/StorageDistributed.cpp index 6dab3173be2d..3d8df6e34f85 100644 --- a/src/Storages/StorageDistributed.cpp +++ b/src/Storages/StorageDistributed.cpp @@ -784,53 +784,6 @@ class ReplaseAliasColumnsVisitor : public InDepthQueryTreeVisitor -{ -public: - using Base = InDepthQueryTreeVisitorWithContext; - using Base::Base; - - void enterImpl(QueryTreeNodePtr & node) - { - if (auto * function_node = node->as(); function_node && isNameOfLocalInFunction(function_node->getFunctionName())) - { - auto * query = function_node->getArguments().getNodes()[1]->as(); - if (!query) - return; - bool no_replace = true; - for (const auto & table_node : extractTableExpressions(query->getJoinTree(), false, true)) - { - const StorageDistributed * storage_distributed = nullptr; - if (const TableNode * table_node_typed = table_node->as()) - storage_distributed = typeid_cast(table_node_typed->getStorage().get()); - else if (const TableFunctionNode * table_function_node_typed = table_node->as()) - storage_distributed = typeid_cast(table_function_node_typed->getStorage().get()); - - if (!storage_distributed) - { - no_replace = false; - break; - } - } - if (no_replace) - return; - - auto result_function = std::make_shared(getGlobalInFunctionNameForLocalInFunctionName(function_node->getFunctionName())); - result_function->getArguments().getNodes() = std::move(function_node->getArguments().getNodes()); - resolveOrdinaryFunctionNodeByName(*result_function, result_function->getFunctionName(), getContext()); - node = result_function; - } - } - - static bool needChildVisit(QueryTreeNodePtr & parent, QueryTreeNodePtr &) - { - if (auto * function_node = parent->as(); function_node && function_node->getFunctionName().starts_with("global")) - return false; - - return true; - } -}; - bool rewriteJoinToGlobalJoinIfNeeded(QueryTreeNodePtr join_tree) { bool rewrite = false; @@ -943,10 +896,7 @@ QueryTreeNodePtr buildQueryTreeDistributed(SelectQueryInfo & query_info, { auto & query_node = query_tree_to_modify->as(); if (query_node.hasWhere()) - { - RewriteInToGlobalInVisitor visitor(query_context); - visitor.visit(query_node.getWhere()); - } + rewriteInToGlobalIn(query_node.getWhere(), query_context); rewriteJoinToGlobalJoinIfNeeded(query_node.getJoinTree()); } diff --git a/src/Storages/buildQueryTreeForShard.cpp b/src/Storages/buildQueryTreeForShard.cpp index 762a82ecb72f..5f6d4efb9956 100644 --- a/src/Storages/buildQueryTreeForShard.cpp +++ b/src/Storages/buildQueryTreeForShard.cpp @@ -57,7 +57,11 @@ namespace Setting extern const SettingsBool prefer_global_in_and_join; extern const SettingsBool enable_add_distinct_to_in_subqueries; extern const SettingsInt64 optimize_const_name_size; +<<<<<<< HEAD extern const SettingsOverflowMode transfer_overflow_mode; +======= + extern const SettingsObjectStorageClusterJoinMode object_storage_cluster_join_mode; +>>>>>>> e9beb426145 (Merge pull request #1782 from Altinity/frontport/antalya-26.3/json_part2) } namespace ErrorCodes @@ -136,8 +140,9 @@ class DistributedProductModeRewriteInJoinVisitor : public InDepthQueryTreeVisito using Base = InDepthQueryTreeVisitorWithContext; using Base::Base; - explicit DistributedProductModeRewriteInJoinVisitor(const ContextPtr & context_) + explicit DistributedProductModeRewriteInJoinVisitor(const ContextPtr & context_, bool find_cross_join_) : Base(context_) + , find_cross_join(find_cross_join_) {} struct InFunctionOrJoin @@ -173,9 +178,11 @@ class DistributedProductModeRewriteInJoinVisitor : public InDepthQueryTreeVisito { auto * function_node = node->as(); auto * join_node = node->as(); + CrossJoinNode * cross_join_node = find_cross_join ? node->as() : nullptr; if ((function_node && isNameOfGlobalInFunction(function_node->getFunctionName())) || - (join_node && join_node->getLocality() == JoinLocality::Global)) + (join_node && join_node->getLocality() == JoinLocality::Global) || + cross_join_node) { InFunctionOrJoin in_function_or_join_entry; in_function_or_join_entry.query_node = node; @@ -239,7 +246,9 @@ class DistributedProductModeRewriteInJoinVisitor : public InDepthQueryTreeVisito replacement_table_expression->setTableExpressionModifiers(*table_expression_modifiers); replacement_map.emplace(table_node.get(), std::move(replacement_table_expression)); } - else if ((distributed_product_mode == DistributedProductMode::GLOBAL || getSettings()[Setting::prefer_global_in_and_join]) && + else if ((distributed_product_mode == DistributedProductMode::GLOBAL || + getSettings()[Setting::prefer_global_in_and_join] || + (find_cross_join && getSettings()[Setting::object_storage_cluster_join_mode] == ObjectStorageClusterJoinMode::GLOBAL)) && !in_function_or_join_stack.empty()) { auto * in_or_join_node_to_modify = in_function_or_join_stack.back().query_node.get(); @@ -273,6 +282,8 @@ class DistributedProductModeRewriteInJoinVisitor : public InDepthQueryTreeVisito std::vector in_function_or_join_stack; std::unordered_map replacement_map; std::vector global_in_or_join_nodes; + + bool find_cross_join = false; }; /** Replaces large constant values with `__getScalar` function calls to avoid @@ -600,14 +611,18 @@ QueryTreeNodePtr getSubqueryFromTableExpression( } -QueryTreeNodePtr buildQueryTreeForShard(const PlannerContextPtr & planner_context, QueryTreeNodePtr query_tree_to_modify, bool allow_global_join_for_right_table) +QueryTreeNodePtr buildQueryTreeForShard( + const PlannerContextPtr & planner_context, + QueryTreeNodePtr query_tree_to_modify, + bool allow_global_join_for_right_table, + bool find_cross_join) { CollectColumnSourceToColumnsVisitor collect_column_source_to_columns_visitor; collect_column_source_to_columns_visitor.visit(query_tree_to_modify); const auto & column_source_to_columns = collect_column_source_to_columns_visitor.getColumnSourceToColumns(); - DistributedProductModeRewriteInJoinVisitor visitor(planner_context->getQueryContext()); + DistributedProductModeRewriteInJoinVisitor visitor(planner_context->getQueryContext(), find_cross_join); visitor.visit(query_tree_to_modify); auto replacement_map = visitor.getReplacementMap(); @@ -669,6 +684,41 @@ QueryTreeNodePtr buildQueryTreeForShard(const PlannerContextPtr & planner_contex replacement_map.emplace(join_table_expression.get(), std::move(temporary_table_expression_node)); continue; } + if (auto * cross_join_node = global_in_or_join_node.query_node->as()) + { + auto tables_count = cross_join_node->getTableExpressions().size(); + for (size_t i = 1; i < tables_count; ++i) + { + QueryTreeNodePtr join_table_expression = cross_join_node->getTableExpressions()[i]; + + auto subquery_node = getSubqueryFromTableExpression(join_table_expression, column_source_to_columns, planner_context->getQueryContext()); + + auto temporary_table_expression_node = executeSubqueryNode(subquery_node, + planner_context->getMutableQueryContext(), + global_in_or_join_node.subquery_depth); + temporary_table_expression_node->setAlias(join_table_expression->getAlias()); + + std::vector descendants_to_map; + for (const auto & child : join_table_expression->getChildren()) + if (child) + descendants_to_map.push_back(child.get()); + + while (!descendants_to_map.empty()) + { + const auto * descendant = descendants_to_map.back(); + descendants_to_map.pop_back(); + + replacement_map.emplace(descendant, temporary_table_expression_node); + + for (const auto & child : descendant->getChildren()) + if (child) + descendants_to_map.push_back(child.get()); + } + + replacement_map.emplace(join_table_expression.get(), std::move(temporary_table_expression_node)); + } + continue; + } if (auto * in_function_node = global_in_or_join_node.query_node->as()) { auto & in_function_subquery_node = in_function_node->getArguments().getNodes().at(1); @@ -778,7 +828,7 @@ class RewriteJoinToGlobalJoinVisitor : public InDepthQueryTreeVisitorWithContext { if (auto * join_node = node->as()) { - bool prefer_local_join = getContext()->getSettingsRef()[Setting::parallel_replicas_prefer_local_join]; + bool prefer_local_join = !force_prefer_local_join && getContext()->getSettingsRef()[Setting::parallel_replicas_prefer_local_join]; bool should_use_global_join = !prefer_local_join || !allStoragesAreMergeTree(join_node->getRightTableExpression()); if (should_use_global_join) join_node->setLocality(JoinLocality::Global); @@ -793,11 +843,79 @@ class RewriteJoinToGlobalJoinVisitor : public InDepthQueryTreeVisitorWithContext return true; } + + void setForcePreferLocalJoin(bool force_prefer_local_join_) { force_prefer_local_join = force_prefer_local_join_; } + +private: + bool force_prefer_local_join = false; }; -void rewriteJoinToGlobalJoin(QueryTreeNodePtr query_tree_to_modify, ContextPtr context) +void rewriteJoinToGlobalJoin(QueryTreeNodePtr query_tree_to_modify, ContextPtr context, bool force_prefer_local_join) { RewriteJoinToGlobalJoinVisitor visitor(context); + visitor.setForcePreferLocalJoin(force_prefer_local_join); + visitor.visit(query_tree_to_modify); +} + +class RewriteInToGlobalInVisitor : public InDepthQueryTreeVisitorWithContext +{ +public: + using Base = InDepthQueryTreeVisitorWithContext; + using Base::Base; + + void enterImpl(QueryTreeNodePtr & node) + { + if (auto * function_node = node->as(); function_node && isNameOfLocalInFunction(function_node->getFunctionName())) + { + auto * query = function_node->getArguments().getNodes()[1]->as(); + if (!query) + return; + if (!rewrite_for_distributed) + { + bool no_replace = true; + for (const auto & table_node : extractTableExpressions(query->getJoinTree(), false, true)) + { + const StorageDistributed * storage_distributed = nullptr; + if (const TableNode * table_node_typed = table_node->as()) + storage_distributed = typeid_cast(table_node_typed->getStorage().get()); + else if (const TableFunctionNode * table_function_node_typed = table_node->as()) + storage_distributed = typeid_cast(table_function_node_typed->getStorage().get()); + + if (!storage_distributed) + { + no_replace = false; + break; + } + } + if (no_replace) + return; + } + + auto result_function = std::make_shared(getGlobalInFunctionNameForLocalInFunctionName(function_node->getFunctionName())); + result_function->getArguments().getNodes() = std::move(function_node->getArguments().getNodes()); + resolveOrdinaryFunctionNodeByName(*result_function, result_function->getFunctionName(), getContext()); + node = result_function; + } + } + + static bool needChildVisit(QueryTreeNodePtr & parent, QueryTreeNodePtr &) + { + if (auto * function_node = parent->as(); function_node && function_node->getFunctionName().starts_with("global")) + return false; + + return true; + } + + void setRewriteForDistributed(bool rewrite_for_distributed_) { rewrite_for_distributed = rewrite_for_distributed_; } + +private: + bool rewrite_for_distributed = false; +}; + +void rewriteInToGlobalIn(QueryTreeNodePtr & query_tree_to_modify, ContextPtr context, bool rewrite_for_distributed) +{ + RewriteInToGlobalInVisitor visitor(context); + visitor.setRewriteForDistributed(rewrite_for_distributed); visitor.visit(query_tree_to_modify); } diff --git a/src/Storages/buildQueryTreeForShard.h b/src/Storages/buildQueryTreeForShard.h index bc950213f5df..92ef4d1649db 100644 --- a/src/Storages/buildQueryTreeForShard.h +++ b/src/Storages/buildQueryTreeForShard.h @@ -19,11 +19,20 @@ using PlannerContextPtr = std::shared_ptr; class Context; using ContextPtr = std::shared_ptr; +<<<<<<< HEAD class Block; QueryTreeNodePtr buildQueryTreeForShard(const PlannerContextPtr & planner_context, QueryTreeNodePtr query_tree_to_modify, bool allow_global_join_for_right_table); +======= +QueryTreeNodePtr buildQueryTreeForShard( + const PlannerContextPtr & planner_context, + QueryTreeNodePtr query_tree_to_modify, + bool allow_global_join_for_right_table, + bool find_cross_join = false); +>>>>>>> e9beb426145 (Merge pull request #1782 from Altinity/frontport/antalya-26.3/json_part2) -void rewriteJoinToGlobalJoin(QueryTreeNodePtr query_tree_to_modify, ContextPtr context); +void rewriteJoinToGlobalJoin(QueryTreeNodePtr query_tree_to_modify, ContextPtr context, bool force_prefer_global_join = false); +void rewriteInToGlobalIn(QueryTreeNodePtr & query_tree_to_modify, ContextPtr context, bool rewrite_for_distributed = false); /** When a Distributed/parallel-replicas query is executed up to `WithMergeableState`, the shard's query tree has its * `ALIAS` columns inlined into their defining expressions. If several projection (or sort/group/...) items expand to the diff --git a/tests/integration/test_database_iceberg/test.py b/tests/integration/test_database_iceberg/test.py index 7c2716dac320..af3a55858f66 100644 --- a/tests/integration/test_database_iceberg/test.py +++ b/tests/integration/test_database_iceberg/test.py @@ -1362,8 +1362,13 @@ def create_namespace(suffix): assert "is filtered by `namespaces` database parameter." in node.query_and_get_error(f"DROP TABLE {CATALOG_NAME}.`{namespace_prefix}alpha.a2.{table_name}`") +<<<<<<< HEAD # TODO - turn on after merge alternative syntax def _test_cluster_joins(started_cluster): +======= +@pytest.mark.parametrize("join_mode", ["local", "global"]) +def test_cluster_joins(started_cluster, join_mode): +>>>>>>> e9beb426145 (Merge pull request #1782 from Altinity/frontport/antalya-26.3/json_part2) node = started_cluster.instances["node1"] test_ref = f"test_join_tables_{uuid.uuid4()}" @@ -1431,7 +1436,7 @@ def _test_cluster_joins(started_cluster): ORDER BY ALL SETTINGS object_storage_cluster='cluster_simple', - object_storage_cluster_join_mode='local' + object_storage_cluster_join_mode='{join_mode}' """ ) @@ -1448,7 +1453,7 @@ def _test_cluster_joins(started_cluster): ORDER BY ALL SETTINGS object_storage_cluster='cluster_simple', - object_storage_cluster_join_mode='local' + object_storage_cluster_join_mode='{join_mode}' """ ) @@ -1464,7 +1469,7 @@ def _test_cluster_joins(started_cluster): ORDER BY ALL SETTINGS object_storage_cluster='cluster_simple', - object_storage_cluster_join_mode='local' + object_storage_cluster_join_mode='{join_mode}' """ ) @@ -1481,7 +1486,7 @@ def _test_cluster_joins(started_cluster): ORDER BY ALL SETTINGS object_storage_cluster='cluster_simple', - object_storage_cluster_join_mode='local' + object_storage_cluster_join_mode='{join_mode}' """ ) @@ -1496,7 +1501,7 @@ def _test_cluster_joins(started_cluster): ORDER BY ALL SETTINGS object_storage_cluster='cluster_simple', - object_storage_cluster_join_mode='local' + object_storage_cluster_join_mode='{join_mode}' """ ) diff --git a/tests/integration/test_s3_cluster/test.py b/tests/integration/test_s3_cluster/test.py index bf0c51a0cff7..6d09d1c5474f 100644 --- a/tests/integration/test_s3_cluster/test.py +++ b/tests/integration/test_s3_cluster/test.py @@ -156,10 +156,15 @@ def started_cluster(): yield cluster finally: +<<<<<<< HEAD # Only this worker's directory; never touches other workers' data. shutil.rmtree(_generated_host_dir(), ignore_errors=True) if cluster is not None: cluster.shutdown() +======= + shutil.rmtree(os.path.join(SCRIPT_DIR, "data/generated/"), ignore_errors=True) + cluster.shutdown() +>>>>>>> e9beb426145 (Merge pull request #1782 from Altinity/frontport/antalya-26.3/json_part2) def test_select_all(started_cluster): @@ -1055,7 +1060,146 @@ def test_remote_no_hedged(started_cluster): assert TSV(pure_s3) == TSV(s3_distributed) +<<<<<<< HEAD def test_joins(started_cluster): +======= +@pytest.mark.parametrize("allow_experimental_analyzer", [0, 1]) +def test_hive_partitioning(started_cluster, allow_experimental_analyzer): + node = started_cluster.instances["s0_0_0"] + + node.query(f"SET allow_experimental_analyzer = {allow_experimental_analyzer}") + + for i in range(1, 5): + exists = node.query( + f""" + SELECT + count() + FROM s3('http://minio1:9001/root/data/hive/key={i}/*', 'minio', '{minio_secret_key}', 'Parquet', 'key Int32, value Int32') + GROUP BY ALL + FORMAT TSV + """ + ) + if int(exists) == 0: + node.query( + f""" + INSERT + INTO FUNCTION s3('http://minio1:9001/root/data/hive/key={i}/data.parquet', 'minio', '{minio_secret_key}', 'Parquet', 'key Int32, value Int32') + SELECT {i}, {i} + SETTINGS use_hive_partitioning = 0 + """ + ) + + settings = "enable_filesystem_cache = 0, use_query_cache = 0, use_cache_for_count_from_files = 0, use_iceberg_metadata_files_cache = 0, use_parquet_metadata_cache = 0, use_page_cache_for_object_storage = 0" + + query_id_full = str(uuid.uuid4()) + result = node.query( + f""" + SELECT count() + FROM s3('http://minio1:9001/root/data/hive/key=**.parquet', 'minio', '{minio_secret_key}', 'Parquet', 'key Int32, value Int32') + WHERE key <= 2 + FORMAT TSV + SETTINGS {settings}, use_hive_partitioning = 0 + """, + query_id=query_id_full, + ) + result = int(result) + assert result == 2 + + query_id_optimized = str(uuid.uuid4()) + result = node.query( + f""" + SELECT count() + FROM s3('http://minio1:9001/root/data/hive/key=**.parquet', 'minio', '{minio_secret_key}', 'Parquet', 'key Int32, value Int32') + WHERE key <= 2 + FORMAT TSV + SETTINGS {settings}, use_hive_partitioning = 1 + """, + query_id=query_id_optimized, + ) + result = int(result) + assert result == 2 + + query_id_cluster_full = str(uuid.uuid4()) + result = node.query( + f""" + SELECT count() + FROM s3Cluster(cluster_simple, 'http://minio1:9001/root/data/hive/key=**.parquet', 'minio', '{minio_secret_key}', 'Parquet', 'key Int32, value Int32') + WHERE key <= 2 + FORMAT TSV + SETTINGS {settings}, use_hive_partitioning = 0 + """, + query_id=query_id_cluster_full, + ) + result = int(result) + assert result == 2 + + query_id_cluster_optimized = str(uuid.uuid4()) + result = node.query( + f""" + SELECT count() + FROM s3Cluster(cluster_simple, 'http://minio1:9001/root/data/hive/key=**.parquet', 'minio', '{minio_secret_key}', 'Parquet', 'key Int32, value Int32') + WHERE key <= 2 + FORMAT TSV + SETTINGS {settings}, use_hive_partitioning = 1 + """, + query_id=query_id_cluster_optimized, + ) + result = int(result) + assert result == 2 + + node.query("SYSTEM FLUSH LOGS ON CLUSTER 'cluster_simple'") + + full_traffic = node.query( + f""" + SELECT sum(ProfileEvents['ReadBufferFromS3Bytes']) + FROM clusterAllReplicas(cluster_simple, system.query_log) + WHERE type='QueryFinish' AND initial_query_id='{query_id_full}' + FORMAT TSV + """ + ) + full_traffic = int(full_traffic) + assert full_traffic > 0 # 612*4 + + optimized_traffic = node.query( + f""" + SELECT sum(ProfileEvents['ReadBufferFromS3Bytes']) + FROM clusterAllReplicas(cluster_simple, system.query_log) + WHERE type='QueryFinish' AND initial_query_id='{query_id_optimized}' + FORMAT TSV + """ + ) + optimized_traffic = int(optimized_traffic) + assert optimized_traffic > 0 # 612*2 + assert full_traffic > optimized_traffic + + cluster_full_traffic = node.query( + f""" + SELECT sum(ProfileEvents['ReadBufferFromS3Bytes']) + FROM clusterAllReplicas(cluster_simple, system.query_log) + WHERE type='QueryFinish' AND initial_query_id='{query_id_cluster_full}' + FORMAT TSV + """ + ) + cluster_full_traffic = int(cluster_full_traffic) + assert cluster_full_traffic == full_traffic + + cluster_optimized_traffic = node.query( + f""" + SELECT sum(ProfileEvents['ReadBufferFromS3Bytes']) + FROM clusterAllReplicas(cluster_simple, system.query_log) + WHERE type='QueryFinish' AND initial_query_id='{query_id_cluster_optimized}' + FORMAT TSV + """ + ) + cluster_optimized_traffic = int(cluster_optimized_traffic) + assert cluster_optimized_traffic == optimized_traffic + + node.query("SET allow_experimental_analyzer = DEFAULT") + + +@pytest.mark.parametrize("join_mode", ["local", "global"]) +def test_joins(started_cluster, join_mode): +>>>>>>> e9beb426145 (Merge pull request #1782 from Altinity/frontport/antalya-26.3/json_part2) node = started_cluster.instances["s0_0_0"] # Table join_table only exists on the node 's0_0_0'. @@ -1089,7 +1233,7 @@ def test_joins(started_cluster): join_table AS t2 ON t1.value = t2.id ORDER BY t1.name - SETTINGS object_storage_cluster_join_mode='local'; + SETTINGS object_storage_cluster_join_mode='{join_mode}'; """ ) @@ -1112,7 +1256,7 @@ def test_joins(started_cluster): 'name String, value UInt32, polygon Array(Array(Tuple(Float64, Float64)))') AS t1 ON t1.value = t2.id ORDER BY t1.name - SETTINGS object_storage_cluster_join_mode='local'; + SETTINGS object_storage_cluster_join_mode='{join_mode}'; """ ) @@ -1130,7 +1274,7 @@ def test_joins(started_cluster): ON t1.value = t2.id WHERE (t1.value % 2) ORDER BY t1.name - SETTINGS object_storage_cluster_join_mode='local'; + SETTINGS object_storage_cluster_join_mode='{join_mode}'; """ ) @@ -1149,7 +1293,7 @@ def test_joins(started_cluster): ON t1.value = t2.id WHERE (t2.id % 2) ORDER BY t1.name - SETTINGS object_storage_cluster_join_mode='local'; + SETTINGS object_storage_cluster_join_mode='{join_mode}'; """ ) @@ -1167,13 +1311,14 @@ def test_joins(started_cluster): ON t1.value = t2.id WHERE (t1.value % 2) AND ((t2.id % 3) == 2) ORDER BY t1.name - SETTINGS object_storage_cluster_join_mode='local'; + SETTINGS object_storage_cluster_join_mode='{join_mode}'; """ ) res = list(map(str.split, result5.splitlines())) assert len(res) == 6 + # With WHERE clause with global subquery result6 = node.query( f""" SELECT name FROM @@ -1182,12 +1327,28 @@ def test_joins(started_cluster): 'name String, value UInt32, polygon Array(Array(Tuple(Float64, Float64)))') WHERE value IN (SELECT id FROM join_table) ORDER BY name - SETTINGS object_storage_cluster_join_mode='local'; + SETTINGS object_storage_cluster_join_mode='{join_mode}'; + """ + ) + res = list(map(str.split, result6.splitlines())) + assert len(res) == 25 + + # With WHERE clause with global subquery + result6 = node.query( + f""" + SELECT name FROM + s3Cluster('cluster_simple', + 'http://minio1:9001/root/data/{{clickhouse,database}}/*', 'minio', '{minio_secret_key}', 'CSV', + 'name String, value UInt32, polygon Array(Array(Tuple(Float64, Float64)))') + WHERE value GLOBAL IN (SELECT id FROM join_table) + ORDER BY name + SETTINGS object_storage_cluster_join_mode='{join_mode}'; """ ) res = list(map(str.split, result6.splitlines())) assert len(res) == 25 + # With WHERE clause without columns in condition result7 = node.query( f""" SELECT count() FROM @@ -1198,11 +1359,12 @@ def test_joins(started_cluster): join_table AS t2 ON 1 GROUP BY ALL - SETTINGS object_storage_cluster_join_mode='local'; + SETTINGS object_storage_cluster_join_mode='{join_mode}'; """ ) assert result7.strip() == "625" + # With WHERE clause without columns in condition and with local column in SELECT result8 = node.query( f""" SELECT count(), t2.id FROM @@ -1213,7 +1375,7 @@ def test_joins(started_cluster): join_table AS t2 ON 1 GROUP BY ALL - SETTINGS object_storage_cluster_join_mode='local'; + SETTINGS object_storage_cluster_join_mode='{join_mode}'; """ ) res = list(map(str.split, result8.splitlines())) diff --git a/tests/integration/test_storage_iceberg_with_spark/conftest.py b/tests/integration/test_storage_iceberg_with_spark/conftest.py index ea5282607f7d..e0e76b9e9228 100644 --- a/tests/integration/test_storage_iceberg_with_spark/conftest.py +++ b/tests/integration/test_storage_iceberg_with_spark/conftest.py @@ -82,6 +82,7 @@ def started_cluster_iceberg_with_spark(): with_minio=True, with_azurite=True, stay_alive=True, + with_zookeeper=True, ) cluster.add_instance( "node2", @@ -94,6 +95,7 @@ def started_cluster_iceberg_with_spark(): ], user_configs=["configs/users.d/users.xml"], stay_alive=True, + with_zookeeper=True, ) cluster.add_instance( "node3", @@ -106,6 +108,7 @@ def started_cluster_iceberg_with_spark(): ], user_configs=["configs/users.d/users.xml"], stay_alive=True, + with_zookeeper=True, ) logging.info("Starting cluster...") diff --git a/tests/integration/test_storage_iceberg_with_spark/test_cluster_joins.py b/tests/integration/test_storage_iceberg_with_spark/test_cluster_joins.py index c04940c1eeb8..82e9d6c3c572 100644 --- a/tests/integration/test_storage_iceberg_with_spark/test_cluster_joins.py +++ b/tests/integration/test_storage_iceberg_with_spark/test_cluster_joins.py @@ -6,13 +6,16 @@ execute_spark_query_general, ) +@pytest.mark.parametrize("join_mode", ["local", "global"]) @pytest.mark.parametrize("storage_type", ["s3", "azure"]) -def test_cluster_joins(started_cluster_iceberg_with_spark, storage_type): +def test_cluster_joins(started_cluster_iceberg_with_spark, storage_type, join_mode): instance = started_cluster_iceberg_with_spark.instances["node1"] spark = started_cluster_iceberg_with_spark.spark_session TABLE_NAME = "test_cluster_joins_" + storage_type + "_" + get_uuid_str() TABLE_NAME_2 = "test_cluster_joins_2_" + storage_type + "_" + get_uuid_str() TABLE_NAME_LOCAL = "test_cluster_joins_local_" + storage_type + "_" + get_uuid_str() + TABLE_NAME_SOURCE = "test_cluster_joins_source_" + storage_type + "_" + get_uuid_str() + TABLE_NAME_DISTRIBUTED = "test_cluster_joins_distributed_" + storage_type + "_" + get_uuid_str() def execute_spark_query(query: str, table_name): return execute_spark_query_general( @@ -72,6 +75,10 @@ def execute_spark_query(query: str, table_name): instance.query(f"CREATE TABLE `{TABLE_NAME_LOCAL}` (id Int64, second_name String) ENGINE = Memory()") instance.query(f"INSERT INTO `{TABLE_NAME_LOCAL}` VALUES (1, 'silver'), (2, 'black')") + instance.query(f"CREATE TABLE `{TABLE_NAME_SOURCE}` ON CLUSTER 'cluster_simple' (id Int64, second_name String) ENGINE = Memory()") + instance.query(f"CREATE TABLE `{TABLE_NAME_DISTRIBUTED}` (id Int64, second_name String) ENGINE = Distributed('cluster_simple', currentDatabase(), '{TABLE_NAME_SOURCE}')") + instance.query(f"INSERT INTO `{TABLE_NAME_DISTRIBUTED}` VALUES (1, 'smith'), (2, 'wesson')") + res = instance.query( f""" SELECT t1.name,t2.second_name @@ -81,7 +88,7 @@ def execute_spark_query(query: str, table_name): ORDER BY ALL SETTINGS object_storage_cluster='cluster_simple', - object_storage_cluster_join_mode='local' + object_storage_cluster_join_mode='{join_mode}' """ ) @@ -91,14 +98,31 @@ def execute_spark_query(query: str, table_name): f""" SELECT name FROM {creation_expression} - WHERE tag in ( + WHERE tag IN ( + SELECT id + FROM {creation_expression_2} + ) + ORDER BY ALL + SETTINGS + object_storage_cluster='cluster_simple', + object_storage_cluster_join_mode='{join_mode}' + """ + ) + + assert res == "jack\njohn\n" + + res = instance.query( + f""" + SELECT name + FROM {creation_expression} + WHERE tag GLOBAL IN ( SELECT id FROM {creation_expression_2} ) ORDER BY ALL SETTINGS object_storage_cluster='cluster_simple', - object_storage_cluster_join_mode='local' + object_storage_cluster_join_mode='{join_mode}' """ ) @@ -114,7 +138,7 @@ def execute_spark_query(query: str, table_name): ORDER BY ALL SETTINGS object_storage_cluster='cluster_simple', - object_storage_cluster_join_mode='local' + object_storage_cluster_join_mode='{join_mode}' """ ) @@ -124,14 +148,31 @@ def execute_spark_query(query: str, table_name): f""" SELECT name FROM {creation_expression} - WHERE tag in ( + WHERE tag IN ( + SELECT id + FROM `{TABLE_NAME_LOCAL}` + ) + ORDER BY ALL + SETTINGS + object_storage_cluster='cluster_simple', + object_storage_cluster_join_mode='{join_mode}' + """ + ) + + assert res == "jack\njohn\n" + + res = instance.query( + f""" + SELECT name + FROM {creation_expression} + WHERE tag GLOBAL IN ( SELECT id FROM `{TABLE_NAME_LOCAL}` ) ORDER BY ALL SETTINGS object_storage_cluster='cluster_simple', - object_storage_cluster_join_mode='local' + object_storage_cluster_join_mode='{join_mode}' """ ) @@ -146,8 +187,57 @@ def execute_spark_query(query: str, table_name): ORDER BY ALL SETTINGS object_storage_cluster='cluster_simple', - object_storage_cluster_join_mode='local' + object_storage_cluster_join_mode='{join_mode}' """ ) assert res == "jack\tblack\njack\tsilver\njohn\tblack\njohn\tsilver\n" + + res = instance.query( + f""" + SELECT t1.name,t2.second_name + FROM {creation_expression} AS t1 + JOIN `{TABLE_NAME_DISTRIBUTED}` AS t2 + ON t1.tag=t2.id + ORDER BY ALL + SETTINGS + object_storage_cluster='cluster_simple', + object_storage_cluster_join_mode='{join_mode}' + """ + ) + + assert res == "jack\twesson\njohn\tsmith\n" + + res = instance.query( + f""" + SELECT name + FROM {creation_expression} + WHERE tag GLOBAL IN ( + SELECT id + FROM `{TABLE_NAME_DISTRIBUTED}` + ) + ORDER BY ALL + SETTINGS + object_storage_cluster='cluster_simple', + object_storage_cluster_join_mode='{join_mode}' + """ + ) + + assert res == "jack\njohn\n" + + res = instance.query( + f""" + SELECT name + FROM {creation_expression} + WHERE tag IN ( + SELECT id + FROM `{TABLE_NAME_DISTRIBUTED}` + ) + ORDER BY ALL + SETTINGS + object_storage_cluster='cluster_simple', + object_storage_cluster_join_mode='{join_mode}' + """ + ) + + assert res == "jack\njohn\n" From ce062b31d118c70ec2974927c8865afc5f8170c4 Mon Sep 17 00:00:00 2001 From: Andrey Zvonov <32552679+zvonand@users.noreply.github.com> Date: Wed, 5 Aug 2026 17:48:29 +0200 Subject: [PATCH 19/43] Resolve conflicts in cherry-pick of #1782 Kept antalya-26.6 additions outside the PR scope (transfer_overflow_mode setting declaration, forward-declared Block for buildShardCollapseFanOut, per-worker generated-data cleanup in test_s3_cluster, disabled _test_cluster_joins in test_database_iceberg) and applied the PR's changes on top; dropped the unchanged test_hive_partitioning context block the merge commit carried into the conflict region. Source-PR: #1782 (https://github.com/Altinity/ClickHouse/pull/1782) --- src/Storages/buildQueryTreeForShard.cpp | 3 - src/Storages/buildQueryTreeForShard.h | 4 - .../integration/test_database_iceberg/test.py | 6 +- tests/integration/test_s3_cluster/test.py | 143 ------------------ 4 files changed, 1 insertion(+), 155 deletions(-) diff --git a/src/Storages/buildQueryTreeForShard.cpp b/src/Storages/buildQueryTreeForShard.cpp index 5f6d4efb9956..76c979bdae53 100644 --- a/src/Storages/buildQueryTreeForShard.cpp +++ b/src/Storages/buildQueryTreeForShard.cpp @@ -57,11 +57,8 @@ namespace Setting extern const SettingsBool prefer_global_in_and_join; extern const SettingsBool enable_add_distinct_to_in_subqueries; extern const SettingsInt64 optimize_const_name_size; -<<<<<<< HEAD extern const SettingsOverflowMode transfer_overflow_mode; -======= extern const SettingsObjectStorageClusterJoinMode object_storage_cluster_join_mode; ->>>>>>> e9beb426145 (Merge pull request #1782 from Altinity/frontport/antalya-26.3/json_part2) } namespace ErrorCodes diff --git a/src/Storages/buildQueryTreeForShard.h b/src/Storages/buildQueryTreeForShard.h index 92ef4d1649db..5e3ae678dcf6 100644 --- a/src/Storages/buildQueryTreeForShard.h +++ b/src/Storages/buildQueryTreeForShard.h @@ -19,17 +19,13 @@ using PlannerContextPtr = std::shared_ptr; class Context; using ContextPtr = std::shared_ptr; -<<<<<<< HEAD class Block; -QueryTreeNodePtr buildQueryTreeForShard(const PlannerContextPtr & planner_context, QueryTreeNodePtr query_tree_to_modify, bool allow_global_join_for_right_table); -======= QueryTreeNodePtr buildQueryTreeForShard( const PlannerContextPtr & planner_context, QueryTreeNodePtr query_tree_to_modify, bool allow_global_join_for_right_table, bool find_cross_join = false); ->>>>>>> e9beb426145 (Merge pull request #1782 from Altinity/frontport/antalya-26.3/json_part2) void rewriteJoinToGlobalJoin(QueryTreeNodePtr query_tree_to_modify, ContextPtr context, bool force_prefer_global_join = false); void rewriteInToGlobalIn(QueryTreeNodePtr & query_tree_to_modify, ContextPtr context, bool rewrite_for_distributed = false); diff --git a/tests/integration/test_database_iceberg/test.py b/tests/integration/test_database_iceberg/test.py index af3a55858f66..a20c986af9e7 100644 --- a/tests/integration/test_database_iceberg/test.py +++ b/tests/integration/test_database_iceberg/test.py @@ -1362,13 +1362,9 @@ def create_namespace(suffix): assert "is filtered by `namespaces` database parameter." in node.query_and_get_error(f"DROP TABLE {CATALOG_NAME}.`{namespace_prefix}alpha.a2.{table_name}`") -<<<<<<< HEAD # TODO - turn on after merge alternative syntax -def _test_cluster_joins(started_cluster): -======= @pytest.mark.parametrize("join_mode", ["local", "global"]) -def test_cluster_joins(started_cluster, join_mode): ->>>>>>> e9beb426145 (Merge pull request #1782 from Altinity/frontport/antalya-26.3/json_part2) +def _test_cluster_joins(started_cluster, join_mode): node = started_cluster.instances["node1"] test_ref = f"test_join_tables_{uuid.uuid4()}" diff --git a/tests/integration/test_s3_cluster/test.py b/tests/integration/test_s3_cluster/test.py index 6d09d1c5474f..5985547713f2 100644 --- a/tests/integration/test_s3_cluster/test.py +++ b/tests/integration/test_s3_cluster/test.py @@ -156,15 +156,10 @@ def started_cluster(): yield cluster finally: -<<<<<<< HEAD # Only this worker's directory; never touches other workers' data. shutil.rmtree(_generated_host_dir(), ignore_errors=True) if cluster is not None: cluster.shutdown() -======= - shutil.rmtree(os.path.join(SCRIPT_DIR, "data/generated/"), ignore_errors=True) - cluster.shutdown() ->>>>>>> e9beb426145 (Merge pull request #1782 from Altinity/frontport/antalya-26.3/json_part2) def test_select_all(started_cluster): @@ -1060,146 +1055,8 @@ def test_remote_no_hedged(started_cluster): assert TSV(pure_s3) == TSV(s3_distributed) -<<<<<<< HEAD -def test_joins(started_cluster): -======= -@pytest.mark.parametrize("allow_experimental_analyzer", [0, 1]) -def test_hive_partitioning(started_cluster, allow_experimental_analyzer): - node = started_cluster.instances["s0_0_0"] - - node.query(f"SET allow_experimental_analyzer = {allow_experimental_analyzer}") - - for i in range(1, 5): - exists = node.query( - f""" - SELECT - count() - FROM s3('http://minio1:9001/root/data/hive/key={i}/*', 'minio', '{minio_secret_key}', 'Parquet', 'key Int32, value Int32') - GROUP BY ALL - FORMAT TSV - """ - ) - if int(exists) == 0: - node.query( - f""" - INSERT - INTO FUNCTION s3('http://minio1:9001/root/data/hive/key={i}/data.parquet', 'minio', '{minio_secret_key}', 'Parquet', 'key Int32, value Int32') - SELECT {i}, {i} - SETTINGS use_hive_partitioning = 0 - """ - ) - - settings = "enable_filesystem_cache = 0, use_query_cache = 0, use_cache_for_count_from_files = 0, use_iceberg_metadata_files_cache = 0, use_parquet_metadata_cache = 0, use_page_cache_for_object_storage = 0" - - query_id_full = str(uuid.uuid4()) - result = node.query( - f""" - SELECT count() - FROM s3('http://minio1:9001/root/data/hive/key=**.parquet', 'minio', '{minio_secret_key}', 'Parquet', 'key Int32, value Int32') - WHERE key <= 2 - FORMAT TSV - SETTINGS {settings}, use_hive_partitioning = 0 - """, - query_id=query_id_full, - ) - result = int(result) - assert result == 2 - - query_id_optimized = str(uuid.uuid4()) - result = node.query( - f""" - SELECT count() - FROM s3('http://minio1:9001/root/data/hive/key=**.parquet', 'minio', '{minio_secret_key}', 'Parquet', 'key Int32, value Int32') - WHERE key <= 2 - FORMAT TSV - SETTINGS {settings}, use_hive_partitioning = 1 - """, - query_id=query_id_optimized, - ) - result = int(result) - assert result == 2 - - query_id_cluster_full = str(uuid.uuid4()) - result = node.query( - f""" - SELECT count() - FROM s3Cluster(cluster_simple, 'http://minio1:9001/root/data/hive/key=**.parquet', 'minio', '{minio_secret_key}', 'Parquet', 'key Int32, value Int32') - WHERE key <= 2 - FORMAT TSV - SETTINGS {settings}, use_hive_partitioning = 0 - """, - query_id=query_id_cluster_full, - ) - result = int(result) - assert result == 2 - - query_id_cluster_optimized = str(uuid.uuid4()) - result = node.query( - f""" - SELECT count() - FROM s3Cluster(cluster_simple, 'http://minio1:9001/root/data/hive/key=**.parquet', 'minio', '{minio_secret_key}', 'Parquet', 'key Int32, value Int32') - WHERE key <= 2 - FORMAT TSV - SETTINGS {settings}, use_hive_partitioning = 1 - """, - query_id=query_id_cluster_optimized, - ) - result = int(result) - assert result == 2 - - node.query("SYSTEM FLUSH LOGS ON CLUSTER 'cluster_simple'") - - full_traffic = node.query( - f""" - SELECT sum(ProfileEvents['ReadBufferFromS3Bytes']) - FROM clusterAllReplicas(cluster_simple, system.query_log) - WHERE type='QueryFinish' AND initial_query_id='{query_id_full}' - FORMAT TSV - """ - ) - full_traffic = int(full_traffic) - assert full_traffic > 0 # 612*4 - - optimized_traffic = node.query( - f""" - SELECT sum(ProfileEvents['ReadBufferFromS3Bytes']) - FROM clusterAllReplicas(cluster_simple, system.query_log) - WHERE type='QueryFinish' AND initial_query_id='{query_id_optimized}' - FORMAT TSV - """ - ) - optimized_traffic = int(optimized_traffic) - assert optimized_traffic > 0 # 612*2 - assert full_traffic > optimized_traffic - - cluster_full_traffic = node.query( - f""" - SELECT sum(ProfileEvents['ReadBufferFromS3Bytes']) - FROM clusterAllReplicas(cluster_simple, system.query_log) - WHERE type='QueryFinish' AND initial_query_id='{query_id_cluster_full}' - FORMAT TSV - """ - ) - cluster_full_traffic = int(cluster_full_traffic) - assert cluster_full_traffic == full_traffic - - cluster_optimized_traffic = node.query( - f""" - SELECT sum(ProfileEvents['ReadBufferFromS3Bytes']) - FROM clusterAllReplicas(cluster_simple, system.query_log) - WHERE type='QueryFinish' AND initial_query_id='{query_id_cluster_optimized}' - FORMAT TSV - """ - ) - cluster_optimized_traffic = int(cluster_optimized_traffic) - assert cluster_optimized_traffic == optimized_traffic - - node.query("SET allow_experimental_analyzer = DEFAULT") - - @pytest.mark.parametrize("join_mode", ["local", "global"]) def test_joins(started_cluster, join_mode): ->>>>>>> e9beb426145 (Merge pull request #1782 from Altinity/frontport/antalya-26.3/json_part2) node = started_cluster.instances["s0_0_0"] # Table join_table only exists on the node 's0_0_0'. From dcb3d94cfafab32959efd101225701172bfa0ac1 Mon Sep 17 00:00:00 2001 From: Andrey Zvonov <32552679+zvonand@users.noreply.github.com> Date: Mon, 1 Jun 2026 17:55:52 +0200 Subject: [PATCH 20/43] Cherry-pick of https://github.com/Altinity/ClickHouse/pull/1845 with unresolved conflict markers (resolution in next commit) --- Original cherry-pick message follows: Merge pull request #1845 from Altinity/feature/antalya-26.3/no_useless_describe Do not make 'describe table' query when schema is known # Conflicts: # src/TableFunctions/TableFunctionRemote.cpp --- src/Storages/IStorageCluster.cpp | 5 ++++ src/TableFunctions/CMakeLists.txt | 1 + src/TableFunctions/TableFunctionRemote.cpp | 7 ++++++ src/TableFunctions/TableFunctionRemote.h | 3 +++ tests/integration/test_s3_cluster/test.py | 24 +++++++++---------- .../test_remote_initiator.py | 4 +++- 6 files changed, 31 insertions(+), 13 deletions(-) diff --git a/src/Storages/IStorageCluster.cpp b/src/Storages/IStorageCluster.cpp index 5f22c64fb3dd..ac89945a0b7c 100644 --- a/src/Storages/IStorageCluster.cpp +++ b/src/Storages/IStorageCluster.cpp @@ -27,6 +27,7 @@ #include #include #include +#include #include #include #include @@ -553,6 +554,10 @@ IStorageCluster::RemoteCallVariables IStorageCluster::convertToRemote( auto remote_function = TableFunctionFactory::instance().get(remote_query, new_context); + std::shared_ptr remote_table_function = std::dynamic_pointer_cast(remote_function); + if (remote_table_function) + remote_table_function->setActualTableStructure(getInMemoryMetadata().columns); + auto storage = remote_function->execute(query_to_send, new_context, remote_function_name); return RemoteCallVariables{storage, new_context}; diff --git a/src/TableFunctions/CMakeLists.txt b/src/TableFunctions/CMakeLists.txt index ccdab2fc41b2..eb1c67d2018d 100644 --- a/src/TableFunctions/CMakeLists.txt +++ b/src/TableFunctions/CMakeLists.txt @@ -12,6 +12,7 @@ extract_into_parent_list(clickhouse_table_functions_sources dbms_sources ITableFunction.cpp TableFunctionView.cpp TableFunctionFactory.cpp + TableFunctionRemote.cpp ) extract_into_parent_list(clickhouse_table_functions_headers dbms_headers ITableFunction.h diff --git a/src/TableFunctions/TableFunctionRemote.cpp b/src/TableFunctions/TableFunctionRemote.cpp index 816f0d4b8714..be360603a864 100644 --- a/src/TableFunctions/TableFunctionRemote.cpp +++ b/src/TableFunctions/TableFunctionRemote.cpp @@ -394,7 +394,14 @@ StoragePtr TableFunctionRemote::executeImpl(const ASTPtr & /*ast_function*/, Con ColumnsDescription TableFunctionRemote::getActualTableStructure(ContextPtr context, bool /*is_insert_query*/) const { +<<<<<<< HEAD chassert(cluster); +======= + if (!remote_table_columns.empty()) + return remote_table_columns; + + assert(cluster); +>>>>>>> 289ca733946 (Merge pull request #1845 from Altinity/feature/antalya-26.3/no_useless_describe) return getStructureOfRemoteTable(*cluster, remote_table_id, context, remote_table_function_ptr); } diff --git a/src/TableFunctions/TableFunctionRemote.h b/src/TableFunctions/TableFunctionRemote.h index 498339231153..47e8f1c27efa 100644 --- a/src/TableFunctions/TableFunctionRemote.h +++ b/src/TableFunctions/TableFunctionRemote.h @@ -28,6 +28,8 @@ class TableFunctionRemote : public ITableFunction void setRemoteTableFunction(ASTPtr remote_table_function_ptr_) { remote_table_function_ptr = remote_table_function_ptr_; } + void setActualTableStructure(ColumnsDescription remote_table_columns_) { remote_table_columns = remote_table_columns_; } + private: StoragePtr executeImpl(const ASTPtr & ast_function, ContextPtr context, const std::string & table_name, ColumnsDescription cached_columns, bool is_insert_query) const override; @@ -44,6 +46,7 @@ class TableFunctionRemote : public ITableFunction StorageID remote_table_id = StorageID::createEmpty(); ASTPtr remote_table_function_ptr; ASTPtr sharding_key = nullptr; + ColumnsDescription remote_table_columns; }; } diff --git a/tests/integration/test_s3_cluster/test.py b/tests/integration/test_s3_cluster/test.py index 5985547713f2..29544e404855 100644 --- a/tests/integration/test_s3_cluster/test.py +++ b/tests/integration/test_s3_cluster/test.py @@ -850,8 +850,8 @@ def test_object_storage_remote_initiator(started_cluster): """ ).splitlines() - # initial node + describe table + remote initiator + 2 subqueries on replicas - assert queries == ["5"] + # initial node + remote initiator + 2 subqueries on replicas + assert queries == ["4"] # Cluster with dots in the host names query_id = uuid.uuid4().hex @@ -878,8 +878,8 @@ def test_object_storage_remote_initiator(started_cluster): """ ).splitlines() - # initial node + describe table + remote initiator + 2 subqueries on replicas - assert queries == ["5"] + # initial node + remote initiator + 2 subqueries on replicas + assert queries == ["4"] users = node.query( f""" @@ -920,8 +920,8 @@ def test_object_storage_remote_initiator(started_cluster): """ ).splitlines() - # initial node + describe table + remote initiator + 2 subqueries on replicas - assert queries == ["5"] + # initial node + remote initiator + 2 subqueries on replicas + assert queries == ["4"] users = node.query( f""" @@ -981,8 +981,8 @@ def test_object_storage_remote_initiator(started_cluster): """ ).splitlines() - # initial node + describe table + remote initiator + 2 subqueries on replicas - assert queries == ["5"] + # initial node + remote initiator + 2 subqueries on replicas + assert queries == ["4"] users = node.query( f""" @@ -1485,8 +1485,8 @@ def test_object_storage_remote_initiator_without_cluster_function(started_cluste """ ).splitlines() - # initial node + describe table + remote initiator - assert queries == ["3"] + # initial node + remote initiator + assert queries == ["2"] users = node.query( f""" @@ -1530,8 +1530,8 @@ def test_object_storage_remote_initiator_without_cluster_function(started_cluste """ ).splitlines() - # initial node + describe table + remote initiator + 2 subqueries on replicas - assert queries == ["5"] + # initial node + remote initiator + 2 subqueries on replicas + assert queries == ["4"] users = node.query( f""" diff --git a/tests/integration/test_storage_iceberg_with_spark/test_remote_initiator.py b/tests/integration/test_storage_iceberg_with_spark/test_remote_initiator.py index ba0a61f9a998..a5b833c8a2ea 100644 --- a/tests/integration/test_storage_iceberg_with_spark/test_remote_initiator.py +++ b/tests/integration/test_storage_iceberg_with_spark/test_remote_initiator.py @@ -64,6 +64,7 @@ def flush_logs(): FROM clusterAllReplicas('cluster_simple', system.query_log) WHERE type='QueryFinish' AND initial_query_id='{query_id}' """) + # initial node + 3 subqueries on replicas assert queries == "4\n" query_id = uuid.uuid4().hex @@ -83,7 +84,8 @@ def flush_logs(): FROM clusterAllReplicas('cluster_simple', system.query_log) WHERE type='QueryFinish' AND initial_query_id='{query_id}' """) - assert queries == "6\n" + # initial node + remote initiator + 3 subqueries on replicas + assert queries == "5\n" query_id = uuid.uuid4().hex res = instance.query(f""" From 19d48829f61b2b6ef8d7f98c80c2b0ec22876f5c Mon Sep 17 00:00:00 2001 From: Andrey Zvonov <32552679+zvonand@users.noreply.github.com> Date: Wed, 5 Aug 2026 17:49:56 +0200 Subject: [PATCH 21/43] Resolve conflicts in cherry-pick of #1845 Source-PR: #1845 (https://github.com/Altinity/ClickHouse/pull/1845) --- src/TableFunctions/TableFunctionRemote.cpp | 6 +----- 1 file changed, 1 insertion(+), 5 deletions(-) diff --git a/src/TableFunctions/TableFunctionRemote.cpp b/src/TableFunctions/TableFunctionRemote.cpp index be360603a864..2c9e44bdc66c 100644 --- a/src/TableFunctions/TableFunctionRemote.cpp +++ b/src/TableFunctions/TableFunctionRemote.cpp @@ -394,14 +394,10 @@ StoragePtr TableFunctionRemote::executeImpl(const ASTPtr & /*ast_function*/, Con ColumnsDescription TableFunctionRemote::getActualTableStructure(ContextPtr context, bool /*is_insert_query*/) const { -<<<<<<< HEAD - chassert(cluster); -======= if (!remote_table_columns.empty()) return remote_table_columns; - assert(cluster); ->>>>>>> 289ca733946 (Merge pull request #1845 from Altinity/feature/antalya-26.3/no_useless_describe) + chassert(cluster); return getStructureOfRemoteTable(*cluster, remote_table_id, context, remote_table_function_ptr); } From 850318316bbcb692ab3e9f98699425e037b89b0f Mon Sep 17 00:00:00 2001 From: Andrey Zvonov <32552679+zvonand@users.noreply.github.com> Date: Tue, 2 Jun 2026 13:22:31 +0200 Subject: [PATCH 22/43] Merge pull request #1856 from Altinity/bugfix/antalya-26.3/fix_test_remote_initiator_after_non_remote Fix test_remote_initiator_after_non_remote Source-PR: #1856 (https://github.com/Altinity/ClickHouse/pull/1856) --- .../configs/config.d/cluster.xml | 12 ++++++++++++ .../test_remote_initiator.py | 12 ++++++------ 2 files changed, 18 insertions(+), 6 deletions(-) diff --git a/tests/integration/test_storage_iceberg_with_spark/configs/config.d/cluster.xml b/tests/integration/test_storage_iceberg_with_spark/configs/config.d/cluster.xml index 54c08b27abe8..5835f155a998 100644 --- a/tests/integration/test_storage_iceberg_with_spark/configs/config.d/cluster.xml +++ b/tests/integration/test_storage_iceberg_with_spark/configs/config.d/cluster.xml @@ -16,5 +16,17 @@ + + + + node2 + 9000 + + + node3 + 9000 + + + diff --git a/tests/integration/test_storage_iceberg_with_spark/test_remote_initiator.py b/tests/integration/test_storage_iceberg_with_spark/test_remote_initiator.py index a5b833c8a2ea..763836d21f60 100644 --- a/tests/integration/test_storage_iceberg_with_spark/test_remote_initiator.py +++ b/tests/integration/test_storage_iceberg_with_spark/test_remote_initiator.py @@ -54,7 +54,7 @@ def flush_logs(): FROM {TABLE_NAME} WHERE number=1 SETTINGS - object_storage_cluster='cluster_simple' + object_storage_cluster='cluster_23' """, query_id = query_id) assert res == "1\t1\n" @@ -64,8 +64,8 @@ def flush_logs(): FROM clusterAllReplicas('cluster_simple', system.query_log) WHERE type='QueryFinish' AND initial_query_id='{query_id}' """) - # initial node + 3 subqueries on replicas - assert queries == "4\n" + # initial node + 2 subqueries on replicas + assert queries == "3\n" query_id = uuid.uuid4().hex res = instance.query(f""" @@ -74,7 +74,7 @@ def flush_logs(): WHERE number=1 SETTINGS object_storage_remote_initiator=1, - object_storage_cluster='cluster_simple' + object_storage_cluster='cluster_23' """, query_id = query_id) assert res == "1\t1\n" @@ -84,8 +84,8 @@ def flush_logs(): FROM clusterAllReplicas('cluster_simple', system.query_log) WHERE type='QueryFinish' AND initial_query_id='{query_id}' """) - # initial node + remote initiator + 3 subqueries on replicas - assert queries == "5\n" + # initial node + remote initiator + 2 subqueries on replicas + assert queries == "4\n" query_id = uuid.uuid4().hex res = instance.query(f""" From 703270a38fbc5a31c72941c9fc28be783daf826e Mon Sep 17 00:00:00 2001 From: Andrey Zvonov <32552679+zvonand@users.noreply.github.com> Date: Thu, 11 Jun 2026 14:03:27 +0200 Subject: [PATCH 23/43] Cherry-pick of https://github.com/Altinity/ClickHouse/pull/1863 with unresolved conflict markers (resolution in next commit) --- Original cherry-pick message follows: Merge pull request #1863 from Altinity/bugfix/antalya-26.3/1855_s3cluster_hive Fix cluster functions with hive partitioning # Conflicts: # src/Storages/ObjectStorage/StorageObjectStorageCluster.cpp # src/Storages/StorageFileCluster.cpp # src/Storages/StorageURLCluster.cpp # tests/integration/test_file_cluster/test.py --- src/Storages/IStorageCluster.cpp | 14 +++++ src/Storages/IStorageCluster.h | 4 ++ .../StorageObjectStorageCluster.cpp | 5 ++ .../StorageObjectStorageCluster.h | 1 - src/Storages/StorageFileCluster.cpp | 33 +++++++++- src/Storages/StorageFileCluster.h | 1 - src/Storages/StorageURLCluster.cpp | 20 ++++++ src/Storages/StorageURLCluster.h | 1 - tests/integration/test_file_cluster/test.py | 62 +++++++++++++++++++ tests/integration/test_s3_cluster/test.py | 35 +++++++++++ tests/integration/test_storage_url/test.py | 32 ++++++++++ 11 files changed, 204 insertions(+), 4 deletions(-) diff --git a/src/Storages/IStorageCluster.cpp b/src/Storages/IStorageCluster.cpp index ac89945a0b7c..2d2fc907ec55 100644 --- a/src/Storages/IStorageCluster.cpp +++ b/src/Storages/IStorageCluster.cpp @@ -713,6 +713,20 @@ QueryProcessingStage::Enum IStorageCluster::getQueryProcessingStage( return QueryProcessingStage::Enum::FetchColumns; } +NamesAndTypesList IStorageCluster::getHivePartitionColumnsWithoutVirtuals() const +{ + // Virtual columns can contain hive columns, so we remove these hive coulmns to avoid duplicates. + // In non-cluster case these columns are filtered in DB::prepareReadingFromFormat function. + auto virtual_columns = getVirtualsList(); + NamesAndTypesList hive_partition_filtered; + for (const auto & hive_name_and_type : hive_partition_columns_to_read_from_file_path) + { + if (!virtual_columns.contains(hive_name_and_type.name)) + hive_partition_filtered.emplace_back(hive_name_and_type); + } + return hive_partition_filtered; +} + ContextPtr ReadFromCluster::updateSettings(const Settings & settings) { Settings new_settings{settings}; diff --git a/src/Storages/IStorageCluster.h b/src/Storages/IStorageCluster.h index d4d7c9da7f29..882a3d375d5c 100644 --- a/src/Storages/IStorageCluster.h +++ b/src/Storages/IStorageCluster.h @@ -107,6 +107,10 @@ class IStorageCluster : public IStorage throw Exception(ErrorCodes::NOT_IMPLEMENTED, "Method writeFallBackToPure is not supported by storage {}", getName()); } + NamesAndTypesList getHivePartitionColumnsWithoutVirtuals() const; + + NamesAndTypesList hive_partition_columns_to_read_from_file_path; + private: static ClusterPtr getClusterImpl(ContextPtr context, const String & cluster_name_, size_t max_hosts = 0); diff --git a/src/Storages/ObjectStorage/StorageObjectStorageCluster.cpp b/src/Storages/ObjectStorage/StorageObjectStorageCluster.cpp index 66c7b4de7f2b..4791514cc0b8 100644 --- a/src/Storages/ObjectStorage/StorageObjectStorageCluster.cpp +++ b/src/Storages/ObjectStorage/StorageObjectStorageCluster.cpp @@ -595,8 +595,13 @@ RemoteQueryExecutor::Extension StorageObjectStorageCluster::getTaskIteratorExten local_context, predicate, filter, +<<<<<<< HEAD storage_metadata_snapshot->virtuals.getSampleBlock(VirtualsKind::All, VirtualsMaterializationPlace::Reader).getNamesAndTypesList(), hive_partition_columns_to_read_from_file_path, +======= + getVirtualsList(), + getHivePartitionColumnsWithoutVirtuals(), +>>>>>>> e884b9beef0 (Merge pull request #1863 from Altinity/bugfix/antalya-26.3/1855_s3cluster_hive) nullptr, local_context->getFileProgressCallback(), /*ignore_archive_globs=*/false, diff --git a/src/Storages/ObjectStorage/StorageObjectStorageCluster.h b/src/Storages/ObjectStorage/StorageObjectStorageCluster.h index 52c7d5951855..bcaf293ad4e7 100644 --- a/src/Storages/ObjectStorage/StorageObjectStorageCluster.h +++ b/src/Storages/ObjectStorage/StorageObjectStorageCluster.h @@ -214,7 +214,6 @@ class StorageObjectStorageCluster : public IStorageCluster const String engine_name; StorageObjectStorageConfigurationPtr configuration; const ObjectStoragePtr object_storage; - NamesAndTypesList hive_partition_columns_to_read_from_file_path; bool cluster_name_in_settings; /// non-clustered storage to fall back on pure realisation if needed diff --git a/src/Storages/StorageFileCluster.cpp b/src/Storages/StorageFileCluster.cpp index 6cd6ec0ec2cd..472772a68a54 100644 --- a/src/Storages/StorageFileCluster.cpp +++ b/src/Storages/StorageFileCluster.cpp @@ -68,16 +68,27 @@ StorageFileCluster::StorageFileCluster( auto & storage_columns = storage_metadata.columns; + const auto sample_path = paths.empty() ? "" : paths.front(); + /// Not grabbing the file_columns because it is not necessary to do it here. std::tie(hive_partition_columns_to_read_from_file_path, std::ignore) = HivePartitioningUtils::setupHivePartitioningForFileURLLikeStorage( storage_columns, - paths.empty() ? "" : paths.front(), + sample_path, columns_.empty(), std::nullopt, context); storage_metadata.setConstraints(constraints_); +<<<<<<< HEAD storage_metadata.setVirtuals(VirtualColumnUtils::getVirtualsForFileLikeStorage(storage_metadata.columns, context)); +======= + setVirtuals(VirtualColumnUtils::getVirtualsForFileLikeStorage( + storage_metadata.columns, + context, + std::nullopt, + PartitionStrategyFactory::StrategyType::NONE, + sample_path)); +>>>>>>> e884b9beef0 (Merge pull request #1863 from Altinity/bugfix/antalya-26.3/1855_s3cluster_hive) setInMemoryMetadata(storage_metadata); } @@ -109,8 +120,28 @@ RemoteQueryExecutor::Extension StorageFileCluster::getTaskIteratorExtension( if (file.empty()) return std::make_shared(); return std::make_shared(std::move(file)); +<<<<<<< HEAD }; auto callback = std::make_shared(std::move(next_callback)); +======= + } + +private: + mutable StorageFileSource::FilesIterator iterator; +}; + +RemoteQueryExecutor::Extension StorageFileCluster::getTaskIteratorExtension( + const ActionsDAG::Node * predicate, const ActionsDAG * /* filter */, const ContextPtr & context, ClusterPtr, StorageMetadataPtr) const +{ + auto callback = std::make_shared( + paths, + std::nullopt, + predicate, + getVirtualsList(), + getHivePartitionColumnsWithoutVirtuals(), + context + ); +>>>>>>> e884b9beef0 (Merge pull request #1863 from Altinity/bugfix/antalya-26.3/1855_s3cluster_hive) return RemoteQueryExecutor::Extension{.task_iterator = std::move(callback)}; } diff --git a/src/Storages/StorageFileCluster.h b/src/Storages/StorageFileCluster.h index eb2a70f60b89..dc4c49573623 100644 --- a/src/Storages/StorageFileCluster.h +++ b/src/Storages/StorageFileCluster.h @@ -45,7 +45,6 @@ class StorageFileCluster : public IStorageCluster Strings paths; String filename; String format_name; - NamesAndTypesList hive_partition_columns_to_read_from_file_path; }; } diff --git a/src/Storages/StorageURLCluster.cpp b/src/Storages/StorageURLCluster.cpp index 2232645e5a7f..46edd1138298 100644 --- a/src/Storages/StorageURLCluster.cpp +++ b/src/Storages/StorageURLCluster.cpp @@ -154,8 +154,28 @@ RemoteQueryExecutor::Extension StorageURLCluster::getTaskIteratorExtension( if (url.empty()) return std::make_shared(); return std::make_shared(std::move(url)); +<<<<<<< HEAD }; auto callback = std::make_shared(std::move(next_callback)); +======= + } + +private: + mutable StorageURLSource::DisclosedGlobIterator iterator; +}; + +RemoteQueryExecutor::Extension StorageURLCluster::getTaskIteratorExtension( + const ActionsDAG::Node * predicate, const ActionsDAG * /* filter */, const ContextPtr & context, ClusterPtr, StorageMetadataPtr) const +{ + auto callback = std::make_shared( + uri, + context->getSettingsRef()[Setting::glob_expansion_max_elements], + predicate, + getVirtualsList(), + getHivePartitionColumnsWithoutVirtuals(), + context + ); +>>>>>>> e884b9beef0 (Merge pull request #1863 from Altinity/bugfix/antalya-26.3/1855_s3cluster_hive) return RemoteQueryExecutor::Extension{.task_iterator = std::move(callback)}; } diff --git a/src/Storages/StorageURLCluster.h b/src/Storages/StorageURLCluster.h index 2ed1e28922f1..5dd81d755f65 100644 --- a/src/Storages/StorageURLCluster.h +++ b/src/Storages/StorageURLCluster.h @@ -46,7 +46,6 @@ class StorageURLCluster : public IStorageCluster String uri; String format_name; - NamesAndTypesList hive_partition_columns_to_read_from_file_path; }; diff --git a/tests/integration/test_file_cluster/test.py b/tests/integration/test_file_cluster/test.py index f8922cbb9a5d..2f3211de364b 100644 --- a/tests/integration/test_file_cluster/test.py +++ b/tests/integration/test_file_cluster/test.py @@ -1,4 +1,9 @@ import logging +<<<<<<< HEAD +======= +import time +import uuid +>>>>>>> e884b9beef0 (Merge pull request #1863 from Altinity/bugfix/antalya-26.3/1855_s3cluster_hive) import pytest @@ -209,3 +214,60 @@ def test_format_detection(started_cluster): "select * from fileCluster('my_cluster', 'file_for_format_detection*', auto, 's String, i UInt32', auto) ORDER BY (i, s)" ) assert result == expected_result + + +def test_hive_partitioning_with_where_condition(started_cluster): + test_id = uuid.uuid4().hex[:8] + hive_glob = f"hive_file_cluster_{test_id}/date=*/data.csv" + + for node_name in ("s0_0_0", "s0_0_1", "s0_1_0"): + node = started_cluster.instances[node_name] + for i in range(1, 5): + node.query( + f""" + INSERT INTO TABLE FUNCTION file( + 'hive_file_cluster_{test_id}/date=2000-01-0{i}/data.csv', 'CSVWithNames', 'd UInt64') + SELECT number FROM numbers(10) + SETTINGS engine_file_truncate_on_insert=1 + """ + ) + + node = started_cluster.instances["s0_0_0"] + + result = node.query( + f""" + SELECT count() FROM file('{hive_glob}', 'CSVWithNames', 'd UInt64') + WHERE date='2000-01-02' + SETTINGS use_hive_partitioning=1 + """ + ) + assert result.strip() == "10" + + result = node.query( + f""" + SELECT date, d FROM file('{hive_glob}', 'CSVWithNames', 'd UInt64') + WHERE date='2000-01-02' + LIMIT 1 + SETTINGS use_hive_partitioning=1 + """ + ) + assert "2000-01-02" in result + + result = node.query( + f""" + SELECT count() FROM fileCluster('my_cluster', '{hive_glob}', 'CSVWithNames', 'd UInt64') + WHERE date='2000-01-02' + SETTINGS use_hive_partitioning=1 + """ + ) + assert result.strip() == "10" + + result = node.query( + f""" + SELECT date, d FROM fileCluster('my_cluster', '{hive_glob}', 'CSVWithNames', 'd UInt64') + WHERE date='2000-01-02' + LIMIT 1 + SETTINGS use_hive_partitioning=1 + """ + ) + assert "2000-01-02" in result diff --git a/tests/integration/test_s3_cluster/test.py b/tests/integration/test_s3_cluster/test.py index 29544e404855..b9b6978d309e 100644 --- a/tests/integration/test_s3_cluster/test.py +++ b/tests/integration/test_s3_cluster/test.py @@ -1548,3 +1548,38 @@ def test_object_storage_remote_initiator_without_cluster_function(started_cluste assert users[1:] == ["s0_0_0\tdefault", "s0_0_1\tfoo", "s0_1_0\tfoo"] + + +def test_hive_partitioning_with_where_condition(started_cluster): + node = started_cluster.instances["s0_0_0"] + test_id = uuid.uuid4().hex[:8] + + for i in range(1, 5): + node.query( + f""" + INSERT INTO FUNCTION s3('http://minio1:9001/root/hive/{test_id}/date=2000-01-0{i}/data.csv', + 'minio','{minio_secret_key}','CSVWithNames','d UInt64') + SELECT number FROM numbers(10) + SETTINGS s3_truncate_on_insert=1 + """) + + # Direct query + result = node.query( + f""" + SELECT count() FROM s3('http://minio1:9001/root/hive/{test_id}/date=*/data.csv', + 'minio','{minio_secret_key}','CSVWithNames','d UInt64') + WHERE date='2000-01-02' + SETTINGS use_hive_partitioning=1 + """ + ) + assert result.strip() == "10" + + result = node.query( + f""" + SELECT count() FROM s3Cluster('cluster_simple', 'http://minio1:9001/root/hive/{test_id}/date=*/data.csv', + 'minio','{minio_secret_key}','CSVWithNames','d UInt64') + WHERE date='2000-01-02' + SETTINGS use_hive_partitioning=1 + """ + ) + assert result.strip() == "10" diff --git a/tests/integration/test_storage_url/test.py b/tests/integration/test_storage_url/test.py index 9526873570fc..7cafe0f861a0 100644 --- a/tests/integration/test_storage_url/test.py +++ b/tests/integration/test_storage_url/test.py @@ -40,6 +40,38 @@ def test_partition_by(): assert result.strip() == "1\t2\t3" +def test_hive_partitioning_with_where_condition(): + test_id = uuid.uuid4().hex[:8] + base_url = f"http://nginx:80/hive_url_cluster_{test_id}" + + node1.query( + f""" + INSERT INTO FUNCTION url(url_file, url='{base_url}/date=2000-01-01/data.csv', format='CSVWithNames', structure='d UInt64') + SELECT number FROM numbers(10) + """ + ) + + # 'ur' table function does not work with globs, so we have to test hive partitioning with a single file. + result = node1.query( + f""" + SELECT count() FROM url('{base_url}/date=2000-01-01/data.csv', 'CSVWithNames', 'd UInt64') + WHERE date='2000-01-01' + SETTINGS use_hive_partitioning=1 + """ + ) + assert result.strip() == "10" + + result = node1.query( + f""" + SELECT count() FROM urlCluster( + 'test_cluster_two_shards', '{base_url}/date=2000-01-01/data.csv', 'CSVWithNames', 'd UInt64') + WHERE date='2000-01-01' + SETTINGS use_hive_partitioning=1 + """ + ) + assert result.strip() == "10" + + def test_url_cluster(): result = node1.query( "select * from urlCluster('test_cluster_two_shards', 'http://nginx:80/test_1', 'TSV', 'column1 UInt32, column2 UInt32, column3 UInt32')" From ce8eb6d64c6a81ed6da0a854a65c288775f393a5 Mon Sep 17 00:00:00 2001 From: Andrey Zvonov <32552679+zvonand@users.noreply.github.com> Date: Wed, 5 Aug 2026 18:55:53 +0200 Subject: [PATCH 24/43] Resolve conflicts in cherry-pick of #1863 Kept antalya-26.6's metadata-snapshot based virtuals plumbing and applied the PR's switch to the filtered hive partition column list. Adapted: IStorageCluster::getHivePartitionColumnsWithoutVirtuals now takes the metadata snapshot, because IStorage::getVirtualsList() no longer exists on antalya-26.6 (virtuals live in StorageInMemoryMetadata::virtuals and are read via metadata->virtuals.getSampleBlock(VirtualsKind::All, VirtualsMaterializationPlace::Reader).getNamesAndTypesList()). Adapted: StorageFileCluster sets virtuals through storage_metadata.setVirtuals() instead of the removed IStorage::setVirtuals(), keeping the PR's new sample_path / PartitionStrategyFactory::StrategyType::NONE arguments. Source-PR: #1863 (https://github.com/Altinity/ClickHouse/pull/1863) --- src/Storages/IStorageCluster.cpp | 4 +-- src/Storages/IStorageCluster.h | 2 +- .../StorageObjectStorageCluster.cpp | 7 +---- src/Storages/StorageFileCluster.cpp | 28 ++----------------- src/Storages/StorageURLCluster.cpp | 22 +-------------- tests/integration/test_file_cluster/test.py | 4 --- 6 files changed, 7 insertions(+), 60 deletions(-) diff --git a/src/Storages/IStorageCluster.cpp b/src/Storages/IStorageCluster.cpp index 2d2fc907ec55..505d817cc204 100644 --- a/src/Storages/IStorageCluster.cpp +++ b/src/Storages/IStorageCluster.cpp @@ -713,11 +713,11 @@ QueryProcessingStage::Enum IStorageCluster::getQueryProcessingStage( return QueryProcessingStage::Enum::FetchColumns; } -NamesAndTypesList IStorageCluster::getHivePartitionColumnsWithoutVirtuals() const +NamesAndTypesList IStorageCluster::getHivePartitionColumnsWithoutVirtuals(const StorageMetadataPtr & metadata_snapshot) const { // Virtual columns can contain hive columns, so we remove these hive coulmns to avoid duplicates. // In non-cluster case these columns are filtered in DB::prepareReadingFromFormat function. - auto virtual_columns = getVirtualsList(); + auto virtual_columns = metadata_snapshot->virtuals.getSampleBlock(VirtualsKind::All, VirtualsMaterializationPlace::Reader).getNamesAndTypesList(); NamesAndTypesList hive_partition_filtered; for (const auto & hive_name_and_type : hive_partition_columns_to_read_from_file_path) { diff --git a/src/Storages/IStorageCluster.h b/src/Storages/IStorageCluster.h index 882a3d375d5c..9613f9549562 100644 --- a/src/Storages/IStorageCluster.h +++ b/src/Storages/IStorageCluster.h @@ -107,7 +107,7 @@ class IStorageCluster : public IStorage throw Exception(ErrorCodes::NOT_IMPLEMENTED, "Method writeFallBackToPure is not supported by storage {}", getName()); } - NamesAndTypesList getHivePartitionColumnsWithoutVirtuals() const; + NamesAndTypesList getHivePartitionColumnsWithoutVirtuals(const StorageMetadataPtr & metadata_snapshot) const; NamesAndTypesList hive_partition_columns_to_read_from_file_path; diff --git a/src/Storages/ObjectStorage/StorageObjectStorageCluster.cpp b/src/Storages/ObjectStorage/StorageObjectStorageCluster.cpp index 4791514cc0b8..95b7e61410f0 100644 --- a/src/Storages/ObjectStorage/StorageObjectStorageCluster.cpp +++ b/src/Storages/ObjectStorage/StorageObjectStorageCluster.cpp @@ -595,13 +595,8 @@ RemoteQueryExecutor::Extension StorageObjectStorageCluster::getTaskIteratorExten local_context, predicate, filter, -<<<<<<< HEAD storage_metadata_snapshot->virtuals.getSampleBlock(VirtualsKind::All, VirtualsMaterializationPlace::Reader).getNamesAndTypesList(), - hive_partition_columns_to_read_from_file_path, -======= - getVirtualsList(), - getHivePartitionColumnsWithoutVirtuals(), ->>>>>>> e884b9beef0 (Merge pull request #1863 from Altinity/bugfix/antalya-26.3/1855_s3cluster_hive) + getHivePartitionColumnsWithoutVirtuals(storage_metadata_snapshot), nullptr, local_context->getFileProgressCallback(), /*ignore_archive_globs=*/false, diff --git a/src/Storages/StorageFileCluster.cpp b/src/Storages/StorageFileCluster.cpp index 472772a68a54..420276fe0e48 100644 --- a/src/Storages/StorageFileCluster.cpp +++ b/src/Storages/StorageFileCluster.cpp @@ -79,16 +79,12 @@ StorageFileCluster::StorageFileCluster( context); storage_metadata.setConstraints(constraints_); -<<<<<<< HEAD - storage_metadata.setVirtuals(VirtualColumnUtils::getVirtualsForFileLikeStorage(storage_metadata.columns, context)); -======= - setVirtuals(VirtualColumnUtils::getVirtualsForFileLikeStorage( + storage_metadata.setVirtuals(VirtualColumnUtils::getVirtualsForFileLikeStorage( storage_metadata.columns, context, std::nullopt, PartitionStrategyFactory::StrategyType::NONE, sample_path)); ->>>>>>> e884b9beef0 (Merge pull request #1863 from Altinity/bugfix/antalya-26.3/1855_s3cluster_hive) setInMemoryMetadata(storage_metadata); } @@ -113,35 +109,15 @@ void StorageFileCluster::updateQueryToSendIfNeeded( RemoteQueryExecutor::Extension StorageFileCluster::getTaskIteratorExtension( const ActionsDAG::Node * predicate, const ActionsDAG * /* filter */, const ContextPtr & context, ClusterPtr, StorageMetadataPtr metadata) const { - auto iterator = std::make_shared(paths, std::nullopt, predicate, metadata->virtuals.getSampleBlock(VirtualsKind::All, VirtualsMaterializationPlace::Reader).getNamesAndTypesList(), hive_partition_columns_to_read_from_file_path, context); + auto iterator = std::make_shared(paths, std::nullopt, predicate, metadata->virtuals.getSampleBlock(VirtualsKind::All, VirtualsMaterializationPlace::Reader).getNamesAndTypesList(), getHivePartitionColumnsWithoutVirtuals(metadata), context); auto next_callback = [iter = std::move(iterator)](size_t) mutable -> ClusterFunctionReadTaskResponsePtr { auto file = iter->next(); if (file.empty()) return std::make_shared(); return std::make_shared(std::move(file)); -<<<<<<< HEAD }; auto callback = std::make_shared(std::move(next_callback)); -======= - } - -private: - mutable StorageFileSource::FilesIterator iterator; -}; - -RemoteQueryExecutor::Extension StorageFileCluster::getTaskIteratorExtension( - const ActionsDAG::Node * predicate, const ActionsDAG * /* filter */, const ContextPtr & context, ClusterPtr, StorageMetadataPtr) const -{ - auto callback = std::make_shared( - paths, - std::nullopt, - predicate, - getVirtualsList(), - getHivePartitionColumnsWithoutVirtuals(), - context - ); ->>>>>>> e884b9beef0 (Merge pull request #1863 from Altinity/bugfix/antalya-26.3/1855_s3cluster_hive) return RemoteQueryExecutor::Extension{.task_iterator = std::move(callback)}; } diff --git a/src/Storages/StorageURLCluster.cpp b/src/Storages/StorageURLCluster.cpp index 46edd1138298..9942bb428bc1 100644 --- a/src/Storages/StorageURLCluster.cpp +++ b/src/Storages/StorageURLCluster.cpp @@ -146,7 +146,7 @@ RemoteQueryExecutor::Extension StorageURLCluster::getTaskIteratorExtension( const ActionsDAG::Node * predicate, const ActionsDAG * /* filter */, const ContextPtr & context, ClusterPtr, StorageMetadataPtr metadata) const { auto iterator = std::make_shared( - uri, context->getSettingsRef()[Setting::glob_expansion_max_elements], predicate, metadata->virtuals.getSampleBlock(VirtualsKind::All, VirtualsMaterializationPlace::Reader).getNamesAndTypesList(), hive_partition_columns_to_read_from_file_path, context); + uri, context->getSettingsRef()[Setting::glob_expansion_max_elements], predicate, metadata->virtuals.getSampleBlock(VirtualsKind::All, VirtualsMaterializationPlace::Reader).getNamesAndTypesList(), getHivePartitionColumnsWithoutVirtuals(metadata), context); auto next_callback = [iter = std::move(iterator)](size_t) mutable -> ClusterFunctionReadTaskResponsePtr { @@ -154,28 +154,8 @@ RemoteQueryExecutor::Extension StorageURLCluster::getTaskIteratorExtension( if (url.empty()) return std::make_shared(); return std::make_shared(std::move(url)); -<<<<<<< HEAD }; auto callback = std::make_shared(std::move(next_callback)); -======= - } - -private: - mutable StorageURLSource::DisclosedGlobIterator iterator; -}; - -RemoteQueryExecutor::Extension StorageURLCluster::getTaskIteratorExtension( - const ActionsDAG::Node * predicate, const ActionsDAG * /* filter */, const ContextPtr & context, ClusterPtr, StorageMetadataPtr) const -{ - auto callback = std::make_shared( - uri, - context->getSettingsRef()[Setting::glob_expansion_max_elements], - predicate, - getVirtualsList(), - getHivePartitionColumnsWithoutVirtuals(), - context - ); ->>>>>>> e884b9beef0 (Merge pull request #1863 from Altinity/bugfix/antalya-26.3/1855_s3cluster_hive) return RemoteQueryExecutor::Extension{.task_iterator = std::move(callback)}; } diff --git a/tests/integration/test_file_cluster/test.py b/tests/integration/test_file_cluster/test.py index 2f3211de364b..9f843d864171 100644 --- a/tests/integration/test_file_cluster/test.py +++ b/tests/integration/test_file_cluster/test.py @@ -1,9 +1,5 @@ import logging -<<<<<<< HEAD -======= -import time import uuid ->>>>>>> e884b9beef0 (Merge pull request #1863 from Altinity/bugfix/antalya-26.3/1855_s3cluster_hive) import pytest From b3b5ed4861e60eb3ae2c7a07f5963219bced24b0 Mon Sep 17 00:00:00 2001 From: Andrey Zvonov <32552679+zvonand@users.noreply.github.com> Date: Thu, 11 Jun 2026 14:40:46 +0200 Subject: [PATCH 25/43] Cherry-pick of https://github.com/Altinity/ClickHouse/pull/1872 with unresolved conflict markers (resolution in next commit) --- Original cherry-pick message follows: Merge pull request #1872 from Altinity/bugfix/antalya-26.3/fix_aggregation_with_remote_initiator Fix aggregation flow with remote initiator # Conflicts: # src/Storages/ObjectStorage/StorageObjectStorageCluster.cpp --- src/Storages/IStorageCluster.cpp | 3 +- .../StorageObjectStorageCluster.cpp | 19 ++++- ...leFunctionObjectStorageClusterFallback.cpp | 13 +++- tests/integration/test_s3_cluster/test.py | 75 +++++++++++++++++++ 4 files changed, 107 insertions(+), 3 deletions(-) diff --git a/src/Storages/IStorageCluster.cpp b/src/Storages/IStorageCluster.cpp index 505d817cc204..408641f9e417 100644 --- a/src/Storages/IStorageCluster.cpp +++ b/src/Storages/IStorageCluster.cpp @@ -379,7 +379,8 @@ void IStorageCluster::read( { auto remote_initiator_cluster_name = settings[Setting::object_storage_remote_initiator_cluster].value; if (remote_initiator_cluster_name.empty()) - throw Exception(ErrorCodes::BAD_ARGUMENTS, "Setting 'object_storage_remote_initiator' can be used only with 'object_storage_remote_initiator_cluster' or 'object_storage_cluster'"); + throw Exception(ErrorCodes::BAD_ARGUMENTS, + "Setting 'object_storage_remote_initiator' can be used only with 'object_storage_remote_initiator_cluster', 'object_storage_cluster', or cluster name in arguments"); /// rewrite query to execute `remote('remote_host', s3(...))` /// remote_host can execute query itself or make on-cluster query depends on own `object_storage_cluster` setting diff --git a/src/Storages/ObjectStorage/StorageObjectStorageCluster.cpp b/src/Storages/ObjectStorage/StorageObjectStorageCluster.cpp index 95b7e61410f0..92a278f9d0e0 100644 --- a/src/Storages/ObjectStorage/StorageObjectStorageCluster.cpp +++ b/src/Storages/ObjectStorage/StorageObjectStorageCluster.cpp @@ -39,6 +39,8 @@ namespace Setting extern const SettingsObjectStorageGranularityLevel cluster_table_function_split_granularity; extern const SettingsBool parallel_replicas_for_cluster_engines; extern const SettingsString object_storage_cluster; + extern const SettingsBool object_storage_remote_initiator; + extern const SettingsString object_storage_remote_initiator_cluster; extern const SettingsInt64 delta_lake_snapshot_start_version; extern const SettingsInt64 delta_lake_snapshot_end_version; } @@ -46,11 +48,16 @@ namespace Setting namespace ErrorCodes { extern const int LOGICAL_ERROR; +<<<<<<< HEAD } namespace FailPoints { extern const char storage_cluster_read_sleep[]; +======= + extern const int INVALID_SETTING_VALUE; + extern const int BAD_ARGUMENTS; +>>>>>>> 41e66548fd0 (Merge pull request #1872 from Altinity/bugfix/antalya-26.3/fix_aggregation_with_remote_initiator) } String StorageObjectStorageCluster::getPathSample(ContextPtr context) @@ -693,9 +700,19 @@ String StorageObjectStorageCluster::getClusterName(ContextPtr context) const QueryProcessingStage::Enum StorageObjectStorageCluster::getQueryProcessingStage( ContextPtr context, QueryProcessingStage::Enum to_stage, const StorageSnapshotPtr & storage_snapshot, SelectQueryInfo & query_info) const { + if (!isClusterSupported()) + return QueryProcessingStage::Enum::FetchColumns; + /// Full query if fall back to pure storage. - if (getClusterName(context).empty()) + if (getClusterName(context).empty() // Not cluster request + && context->getSettingsRef()[Setting::object_storage_remote_initiator_cluster].value.empty()) // Not request with remote initiator + { + if (context->getSettingsRef()[Setting::object_storage_remote_initiator]) + throw Exception(ErrorCodes::BAD_ARGUMENTS, + "Setting 'object_storage_remote_initiator' can be used only with 'object_storage_remote_initiator_cluster', 'object_storage_cluster', or cluster name in arguments"); + return QueryProcessingStage::Enum::FetchColumns; + } /// Distributed storage. return IStorageCluster::getQueryProcessingStage(context, to_stage, storage_snapshot, query_info); diff --git a/src/TableFunctions/TableFunctionObjectStorageClusterFallback.cpp b/src/TableFunctions/TableFunctionObjectStorageClusterFallback.cpp index 8b2819e42d46..98d8dc43c85a 100644 --- a/src/TableFunctions/TableFunctionObjectStorageClusterFallback.cpp +++ b/src/TableFunctions/TableFunctionObjectStorageClusterFallback.cpp @@ -117,7 +117,18 @@ void TableFunctionObjectStorageClusterFallback::parseArguments const auto & settings = context->getSettingsRef(); is_cluster_function = !settings[Setting::object_storage_cluster].value.empty() && typename Base::Configuration().isClusterSupported(); - is_remote = settings[Setting::object_storage_remote_initiator]; + // Remote initiator requires 'object_storage_cluster' or 'object_storage_remote_initiator_cluster' + if (settings[Setting::object_storage_remote_initiator]) + { + if (settings[Setting::object_storage_cluster].value.empty() + && settings[Setting::object_storage_remote_initiator_cluster].value.empty()) + { + throw Exception(ErrorCodes::BAD_ARGUMENTS, + "Setting 'object_storage_remote_initiator' can be used only with 'object_storage_remote_initiator_cluster', 'object_storage_cluster', or cluster name in arguments"); + } + + is_remote = true; + } if (is_cluster_function) { diff --git a/tests/integration/test_s3_cluster/test.py b/tests/integration/test_s3_cluster/test.py index b9b6978d309e..51aa7809e44e 100644 --- a/tests/integration/test_s3_cluster/test.py +++ b/tests/integration/test_s3_cluster/test.py @@ -1550,6 +1550,81 @@ def test_object_storage_remote_initiator_without_cluster_function(started_cluste "s0_1_0\tfoo"] +def test_object_storage_remote_initiator_aggregation(started_cluster): + node = started_cluster.instances["s0_0_0"] + + # Remove initiator without cluster request + # Check that aggregation works on nodes + query_id = uuid.uuid4().hex + + result = node.query( + f""" + SELECT sum(value) from s3( + 'http://minio1:9001/root/data/{{clickhouse,database}}/*', 'minio', '{minio_secret_key}', 'CSV', + 'name String, value UInt32, polygon Array(Array(Tuple(Float64, Float64)))') + SETTINGS + object_storage_remote_initiator=1, + object_storage_remote_initiator_cluster='cluster_with_dots_and_user' + """, + query_id = query_id, + ) + + assert result == "67802152770\n" + + node.query("SYSTEM FLUSH LOGS ON CLUSTER 'cluster_all'") + result_rows = node.query( + f""" + SELECT sum(result_rows) + FROM clusterAllReplicas('cluster_all', system.query_log) + WHERE type='QueryFinish' AND initial_query_id='{query_id}' + AND is_initial_query = 0 + ORDER BY ALL + FORMAT TSV + """ + ).splitlines() + + # Data processed on cluster 'hidden_cluster_with_username_and_password'. + # Cluster contains two nodes, each returns one row. + assert result_rows == ["2"] + + # Remove initiator without cluster request + # Check that aggregation works on nodes + query_id = uuid.uuid4().hex + + result = node.query( + f""" + SELECT value % 2 as bit, sum(value) from s3( + 'http://minio1:9001/root/data/{{clickhouse,database}}/*', 'minio', '{minio_secret_key}', 'CSV', + 'name String, value UInt32, polygon Array(Array(Tuple(Float64, Float64)))') + GROUP BY bit + ORDER BY bit + SETTINGS + object_storage_remote_initiator=1, + object_storage_remote_initiator_cluster='cluster_with_dots_and_user' + """, + query_id = query_id, + ) + + assert result == "0\t41117771522\n1\t26684381248\n" + + node.query("SYSTEM FLUSH LOGS ON CLUSTER 'cluster_all'") + result_rows = node.query( + f""" + SELECT sum(result_rows) + FROM clusterAllReplicas('cluster_all', system.query_log) + WHERE type='QueryFinish' AND initial_query_id='{query_id}' + AND is_initial_query = 0 + ORDER BY ALL + FORMAT TSV + """ + ).splitlines() + + # Data processed on cluster 'hidden_cluster_with_username_and_password'. + # Cluster contains two nodes, each returns up to two rows, at least two rows totaly. + result_rows = int(result_rows[0]) + assert result_rows >= 2 and result_rows <= 4 + + def test_hive_partitioning_with_where_condition(started_cluster): node = started_cluster.instances["s0_0_0"] test_id = uuid.uuid4().hex[:8] From 95771a6c606b37ede7276ba0b8448c1173fce639 Mon Sep 17 00:00:00 2001 From: Andrey Zvonov <32552679+zvonand@users.noreply.github.com> Date: Wed, 5 Aug 2026 18:57:37 +0200 Subject: [PATCH 26/43] Resolve conflicts in cherry-pick of #1872 Source-PR: #1872 (https://github.com/Altinity/ClickHouse/pull/1872) --- src/Storages/ObjectStorage/StorageObjectStorageCluster.cpp | 6 +----- 1 file changed, 1 insertion(+), 5 deletions(-) diff --git a/src/Storages/ObjectStorage/StorageObjectStorageCluster.cpp b/src/Storages/ObjectStorage/StorageObjectStorageCluster.cpp index 92a278f9d0e0..d924b38ca25f 100644 --- a/src/Storages/ObjectStorage/StorageObjectStorageCluster.cpp +++ b/src/Storages/ObjectStorage/StorageObjectStorageCluster.cpp @@ -48,16 +48,12 @@ namespace Setting namespace ErrorCodes { extern const int LOGICAL_ERROR; -<<<<<<< HEAD + extern const int BAD_ARGUMENTS; } namespace FailPoints { extern const char storage_cluster_read_sleep[]; -======= - extern const int INVALID_SETTING_VALUE; - extern const int BAD_ARGUMENTS; ->>>>>>> 41e66548fd0 (Merge pull request #1872 from Altinity/bugfix/antalya-26.3/fix_aggregation_with_remote_initiator) } String StorageObjectStorageCluster::getPathSample(ContextPtr context) From ebc7ad5d8f6a16f7aa72c89da690a5e9c1d50067 Mon Sep 17 00:00:00 2001 From: Andrey Zvonov <32552679+zvonand@users.noreply.github.com> Date: Mon, 22 Jun 2026 22:41:42 +0200 Subject: [PATCH 27/43] Cherry-pick of https://github.com/Altinity/ClickHouse/pull/1917 with unresolved conflict markers (resolution in next commit) --- Original cherry-pick message follows: Merge pull request #1917 from Altinity/do_not_evict_entries_from_replicated_partition_exports_table Turn system.replicated_partition_exports table into a history table by removing the TTL # Conflicts: # antalya/docs/design/alter-table-export-part-partition.md # src/Core/Settings.cpp --- .../alter-table-export-part-partition.md | 752 ++++++++++++++++++ docs/en/antalya/partition_export.md | 9 +- src/Core/Settings.cpp | 9 +- ...portReplicatedMergeTreePartitionManifest.h | 3 - .../ExportPartitionManifestUpdatingTask.cpp | 72 +- src/Storages/StorageReplicatedMergeTree.cpp | 33 +- .../test.py | 44 - .../test.py | 36 - 8 files changed, 786 insertions(+), 172 deletions(-) create mode 100644 antalya/docs/design/alter-table-export-part-partition.md diff --git a/antalya/docs/design/alter-table-export-part-partition.md b/antalya/docs/design/alter-table-export-part-partition.md new file mode 100644 index 000000000000..8ebe237c634a --- /dev/null +++ b/antalya/docs/design/alter-table-export-part-partition.md @@ -0,0 +1,752 @@ +Feature Design: `ALTER TABLE EXPORT PART` and `ALTER TABLE EXPORT PARTITION` +============================================================================ + +**Status:** draft +**Author(s):** Arthur Passos +**Related issues/PRs:** +https://github.com/Altinity/ClickHouse/pull/1618 + +**Last updated:** 2026-04-21 + +--- + +## 1. Requirements + +### Motivation + +Cost of storing data is a growing problem for large analytic systems +that use open source ClickHouse and replicated block storage. The core +problem is that block storage is (a) expensive and (b) replication makes +multiple copies. Project Antalya solves the storage cost problem using +the Hybrid Table Engine. Hybrid tables allow users to split tables +into segments, placing hot data on replicated block storage and +cold data on shared Iceberg tables using Parquet data files. + +The hybrid table approach requires a robust mechanism to export table +data from `MergeTree` tables into shared storage. The mechanism must +be fast, use machine resources efficiently, handle failures +automatically, and be easy to monitor. This design covers two new +ClickHouse commands to export data so users can populate hybrid +tables and move data to them at regular intervals. + +* `ALTER TABLE EXPORT PART` -- Exports a single part to a destination. + +* `ALTER TABLE EXPORT PARTITION` -- Exports one or more partitions to a destination. + +Both commands accept two destination families: + +1. **Plain object storage** (`S3`, `AzureBlobStorage`, equivalents). + Output is Parquet laid out in hive-partitioned directories. Atomicity + is provided by a sidecar *commit file* that enumerates the data files + written in the transaction; readers that want atomicity filter by + commit. An external catalog (Glue, REST, Nessie, Lakekeeper, ...) can + register this layout as an Iceberg table afterward, but the export + commands themselves do not interact with any catalog in this mode. + +2. **Apache Iceberg tables, with or without a catalog** (`Iceberg*` + engines, `iceberg*` table functions, `DatabaseIceberg`). Output is + Parquet data files plus per-file Avro *statistics sidecars*; on + commit the EXPORT process assembles a new Iceberg manifest, + writes a new `metadata.json`, and swaps the catalog pointer (or the + warehouse `metadata.json` pointer when catalog-less). Atomicity is + native — the snapshot either exists or it does not. + +These commands replace `INSERT INTO ... SELECT FROM` pipelines that +select rows and write them out to one or more Parquet files. That +approach uses resources for sorting, does not coordinate across replicas, +and does not take advantage of existing partitioning and sorting in +`MergeTree`. + +### Requirements + +1. **SQL only.** All operations related to export are available in SQL. + There should be no need to use non-SQL tools or directly access + storage to run exports or clean up problems. + +2. **Efficient, order-preserving writes.** Write a specified `MergeTree` + part (or every part of a specified partition) to an object-storage + destination in Parquet, preserving the source part's sort order, + without using a `SELECT ORDER BY` pass. Exporting a part should use the + same or less RAM than doing an `INSERT...SELECT...ORDER BY` on the same + data. (For example, it's not uncommon for the latter command to run + out of memory on large parts when the part ordering is added to the + SELECT ORDER BY.) + +3. **Output file management.** Allow users to break exported parts + into smaller Parquet files, which helps ensure good performance + when scanning Iceberg data. + +4. **Data type equivalence.** Map ClickHouse types to Iceberg types + that cast back without data loss to the original ClickHouse types + when selecting data. Applications that access exported data through + a Hybrid table should be able to read the data back from Iceberg + without requiring changes. + +5. **Atomic transfer.** Readers should never see a partial export. + The mechanism depends on the destination family: + - **Plain object storage.** Each transaction writes a sidecar + commit file that lists every data file produced by the transaction. + Readers that want atomicity filter by commit; a crash before the + commit file lands leaves only orphaned data files. + - **Iceberg destinations.** Atomicity is provided by Iceberg's own + snapshot-commit protocol — the new snapshot either becomes the + current metadata pointer or it does not. A crash before the + pointer swap leaves only orphaned data files and sidecars. + +6. **Distributed operation.** + - `EXPORT PART` always runs locally on the ClickHouse host where + it is invoked. + - `EXPORT PARTITION` from non-replicated `MergeTree` tables runs + locally on the ClickHouse host where it is invoked. + - `EXPORT PARTITION` is cluster-coordinated on `Replicated*MergeTree` + tables: any replica that has a given part contributes to the + export; the task is persistent and resumes after restarts. + +7. **Observability.** + It must be possible for users to track the following from system tables: + - Export part request status. + - Export partition request status. + - Relevant profile events related to export. + +8. **Error recovery.** + - **Idempotence.** Re-issuing the same export to the same + destination is a no-op while the export is running. (There should + be a way to track 'recent' exports so that they are idempotent as + well.) + - **Clean-up.** If file or metadata clean-up is required before resubmitting a failed + export, it must be possible to do so using only SQL commands. + - **Automatic restart.** `EXPORT PARTITION` task is persistent and + resumes after restarts. + +9. **Killable.** It must be possible to terminate any `ALTER TABLE EXPORT` command. + The command should be idempotent and must throw a clear exception on failure + rather than hanging. + +### Open questions and future requirements + +The design should address the following topics in the near future. + +- EXPORT PARTITION for MergeTree tables. Must work without Keeper installation. +- Export history. Provide a system table to track the history of part exports. +- Flexible casting that addresses issues like the following. + - Handling potentially lossy casts like INSERT SELECT: int64 -> int32. + - Export to tables that are missing columns. + - How to map column names--by position or by name? (e.g, is id, name, age compatible with id, age, name)? + +### Out of scope requirements + +- Non-Parquet output file formats. Only `Parquet` is targeted in this iteration. Later + iterations may add new output file formats. +- Exporting to arbitrary table functions. Only those backed by an object-storage engine that + supports exports (e.g. `s3`, `azure`) are valid; others throw `NOT_IMPLEMENTED`. +- Non-matching Iceberg schema, sorting or partitioning. Not supported. The source + `MergeTree` schema and partition keys must be compatible with the destination + Iceberg table's current `schema-id` and `partition-spec-id`. Destination partition values + are derived directly from the source part's partition key; we do not recompute them + from row data. +- Any read/query path over exported files — consumption happens via normal `S3` / `s3` / + external-engine reads. +- Synchronous exports. Not supported. EXPORT commands return immediately to client after + starting the export task; completion is polled via system tables. +- Importing parts back from object storage (that is tracked separately). + +### Constraints + +- Experimental gate: `allow_experimental_export_merge_tree_part` (query-level) for `EXPORT PART`; + `allow_experimental_export_merge_tree_partition_feature` (server-level) for `EXPORT PARTITION`. +- For best results `EXPORT PARTITION` requires a ZooKeeper / `clickhouse-keeper` ensemble with + the `multi_read` feature flag. This reduces API calls and ensures transactional consistency + when reading multiple fields. + governed by the destination table's Iceberg partition spec instead. +- No change to `MergeTree` on-disk part format; only the Keeper schema under the table's + replication path is extended. The extension is tranparent to users. + +### References + +- `docs/en/antalya/part_export.md` +- `docs/en/antalya/partition_export.md` +- `tests/queries/0_stateless/03572_export_merge_tree_part_basic.sh` +- `tests/queries/0_stateless/03572_export_merge_tree_part_to_object_storage_simple.sql` +- `tests/queries/0_stateless/03572_export_merge_tree_part_limits_and_table_functions.sh` +- `tests/queries/0_stateless/03572_export_merge_tree_part_special_columns.sh` +- `tests/queries/0_stateless/03572_export_replicated_merge_tree_part_to_object_storage.sh` +- `tests/queries/0_stateless/03572_export_replicated_merge_tree_part_to_object_storage_simple.sql` +- `tests/queries/0_stateless/03604_export_merge_tree_partition.sh` +- `tests/queries/0_stateless/03608_export_merge_tree_part_filename_pattern.sh` +- `tests/integration/test_export_merge_tree_part_to_object_storage/test.py` +- `tests/integration/test_export_replicated_mt_partition_to_object_storage/test.py` + +--- + +## 2. Functional specification + +### User-facing behavior +A user points `ALTER TABLE` at a `MergeTree` source and a destination (table or table function). +The command returns immediately with no rows. The export runs in the background; progress lives +in `system.exports` and, for partition exports, `system.replicated_partition_exports`. Successful +exports append to `system.part_log` with `event_type = 'ExportPart'`. + +Output shape depends on the destination family: + +- **Plain object storage.** One Parquet data file per part (or per chunk when split by + size/rows) plus one commit file per transaction — `//_..parquet` + and `/commit_<...>`. Readers that want atomicity filter by commit. This is the default path, + but it can be customized to handle sharding, which is not covered by the default case. See the + `` + +- **Iceberg destination.** One Parquet data file per part (or per chunk) plus a sibling + Avro statistics sidecar `_clickhouse_export_part_sidecar.avro` carrying + `record_count`, `file_size_in_bytes`, `column_sizes`, `null_value_counts`, `lower_bounds`, + and `upper_bounds`. The per-part task does not modify Iceberg metadata. On final commit + the initiating replica reads every sidecar, assembles a new manifest and manifest list, + writes a new `metadata.json`, and atomically swaps the pointer (via the catalog when one + is configured; otherwise via the warehouse `metadata.json` pointer). The manifest summary + contains `clickhouse.export-partition-transaction-id`, checked before every commit attempt + to prevent a double-commit after a post-commit / pre-status-update crash. Sidecar files + are not referenced from any Iceberg manifest and can be deleted safely after the commit + lands; ClickHouse does not reap them. + +### SQL syntax / API + +```sql +-- Export a single part to a destination table +ALTER TABLE [db.]table + EXPORT PART 'part_name' + TO TABLE [dest_db.]dest_table + [SETTINGS ...]; + +-- Export a single part to a destination table function +ALTER TABLE [db.]table + EXPORT PART 'part_name' + TO TABLE FUNCTION s3(...) PARTITION BY + [SETTINGS ...]; + +-- Export every active part of a partition (Replicated*MergeTree only) +ALTER TABLE [db.]table + EXPORT PARTITION ID 'partition_id' + TO TABLE [dest_db.]dest_table + [SETTINGS ...]; + +-- Export every active part of all partitions (Replicated*MergeTree only) +ALTER TABLE [db.]table + EXPORT PARTITION ALL + TO TABLE [dest_db.]dest_table + [SETTINGS ...]; + +-- Cancel one or more partition exports +KILL EXPORT PARTITION WHERE ; +``` + +### Individual command examples +These are derived from `tests/queries/0_stateless/03572_*` and `03604_export_merge_tree_partition.sh`. + +```sql +-- Part export to S3 table +ALTER TABLE mt_table EXPORT PART '2020_1_1_0' TO TABLE s3_table +SETTINGS allow_experimental_export_merge_tree_part = 1; + +-- Part export to S3 table function (schema inferred from source) +ALTER TABLE mt_table EXPORT PART '2020_1_1_0' +TO TABLE FUNCTION s3(s3_conn, filename='tf', format='Parquet', partition_strategy='hive') +PARTITION BY year +SETTINGS allow_experimental_export_merge_tree_part = 1; + +-- Split large part across multiple Parquet files +ALTER TABLE big EXPORT PART '2025_0_32_3' TO TABLE big_dest +SETTINGS allow_experimental_export_merge_tree_part = 1, + export_merge_tree_part_max_bytes_per_file = 10000000, + output_format_parquet_row_group_size_bytes = 5000000; +-- (See note on settings below. Iceberg table engine now has built-in +-- settings for Parquet files.) + +-- Partition export across a Replicated cluster. This currently +-- selects the parts on the replica that receives the plan. This +-- means the result may vary if new parts are arriving on other +-- replicas. +ALTER TABLE rmt_table EXPORT PARTITION ID '2020' TO TABLE s3_table; + +-- Cancel by filter. The WHERE uses the same filter used to read from `system.replicated_partition_exports`. +KILL EXPORT PARTITION +WHERE partition_id = '2020' + AND source_table = 'rmt_table' + AND destination_table = 's3_table'; +``` + +### End-to-end examples + +Two parallel walkthroughs illustrate each destination family. Both +begin from the same `ReplicatedMergeTree` source. + +```sql +-- Source table (shared by both examples). +CREATE TABLE events +( + id UInt64, + ts DateTime, + year UInt16 +) +ENGINE = ReplicatedMergeTree('/clickhouse/tables/{database}/events', 'r1') +PARTITION BY year +ORDER BY (year, id); + +-- Seed two partitions (2024, 2025) straight from `system.numbers`. +INSERT INTO events +SELECT + number AS id, + toDateTime('2024-01-01 00:00:00') + INTERVAL number SECOND AS ts, + 2024 AS year +FROM system.numbers +LIMIT 1000000; + +INSERT INTO events +SELECT + number AS id, + toDateTime('2025-01-01 00:00:00') + INTERVAL number SECOND AS ts, + 2025 AS year +FROM system.numbers +LIMIT 1000000; +``` + +#### Plain object storage (hive layout) + +Export to a hive-partitioned S3 destination. The on-disk shape is what +an external Iceberg catalog (Glue, REST, Nessie, Lakekeeper, ...) would +register as an Iceberg table; the `EXPORT PARTITION` command itself +does not touch any catalog in this mode. + +```sql +-- 1. Destination: S3 with hive partition layout. +CREATE TABLE events_s3 +( + id UInt64, + ts DateTime, + year UInt16 +) +ENGINE = S3(s3_conn, filename='warehouse/events', format = Parquet, partition_strategy = 'hive') +PARTITION BY year; + +-- 2. Export the 2024 partition. Returns immediately; runs in the background. +ALTER TABLE events EXPORT PARTITION ID '2024' TO TABLE events_s3; + +-- 3. Watch progress (Keeper round-trip — use sparingly). +SELECT status, parts_count, parts_to_do, last_exception +FROM system.replicated_partition_exports +WHERE source_table = 'events' AND partition_id = '2024'; + +-- 4. When status = 'COMPLETED', the destination bucket contains: +-- warehouse/events/year=2024/_.1.parquet (one per part) +-- warehouse/events/commit_2024_ (atomicity manifest) +-- Readers that filter by commit see either the full partition or nothing. +SELECT count() FROM events_s3 WHERE year = 2024; + +-- 5. Or inspect the layout directly. +SELECT _path +FROM s3(s3_conn, filename = 'warehouse/events/year=2024/**', format = 'One') +ORDER BY _path; +``` + +#### Iceberg destination + +Export directly to an Iceberg table. Unlike the plain-object-storage +flow, `EXPORT PARTITION` writes native Iceberg metadata on commit, so +the result is queryable as an Iceberg table immediately — no external +registration step. The example uses `IcebergS3` without a catalog; +swap in `DatabaseIceberg` or an `iceberg(...)` catalog-backed table +function to route through REST / Glue / Unity. + +**Open Issue** We need to specify how to commit via a catalog. + +```sql +-- 1. Destination: an Iceberg table backed by S3. No catalog required +-- for this form; the warehouse metadata.json pointer is managed +-- by ClickHouse directly. +CREATE TABLE events_iceberg +( + id UInt64, + ts DateTime, + year UInt16 +) +ENGINE = IcebergS3(s3_conn, filename='warehouse/events_iceberg/') +PARTITION BY year; + +-- 2. Export the 2024 partition. Returns immediately; runs in the background. +ALTER TABLE events EXPORT PARTITION ID '2024' TO TABLE events_iceberg; + +-- 3. Watch progress. You can use the system.exports table for this. +SELECT status, parts_count, parts_to_do, last_exception +FROM system.replicated_partition_exports +WHERE source_table = 'events' AND partition_id = '2024'; + +-- 4. When status = 'COMPLETED', the destination contains a fully-formed +-- Iceberg table. Readers see the snapshot atomically — either the +-- full partition or nothing. +SELECT count() FROM events_iceberg WHERE year = 2024; + +-- 5. Object layout: each data file has a sidecar carrying per-file stats +-- used at commit time. These are ClickHouse-private and unreferenced +-- from any Iceberg manifest; safe to delete after COMPLETED. +SELECT _path +FROM s3(s3_conn, filename = 'warehouse/events_iceberg/data/year=2024/**', format = 'One') +ORDER BY _path; +-- warehouse/events_iceberg/data/year=2024/_.1.parquet +-- warehouse/events_iceberg/data/year=2024/_.1.parquet_clickhouse_export_part_sidecar.avro +-- ... +-- warehouse/events_iceberg/metadata/v.metadata.json +-- warehouse/events_iceberg/metadata/snap--*.avro +-- warehouse/events_iceberg/metadata/.avro + +-- 6. The commit carries an idempotency marker in the manifest summary +-- (clickhouse.export-partition-transaction-id). If the initiator +-- crashes post-commit / pre-status-update, a retry sees its own +-- transaction id in the target's latest manifest and skips instead +-- of double-committing. +``` + +The EXPORT PARTITION flow initially only works on ReplicatedMergeTree tables and +requires Keeper. Future iterations will support MergeTree tables as a source. + +### Operational notes + +The following notes expand on expected behavior of commands. + +1. When writing to object storage using `partition_strategy = 'wildcard'`, either wildcard + or 'hive' arguments are permitted. (This setting has no impact on Iceberg, + which records file partitions using metadata.) + +2. By default `ALTER TABLE t EXPORT PART 'p' TO TABLE s3_t` writes + `//_.1.parquet` plus + `/commit__`. You can read this using + `SELECT * FROM s3(...)`. + +3. `ALTER TABLE rmt EXPORT PARTITION ID 'p' TO TABLE s3_t` exports + every active part of partition `p` across all replicas that host + it; `system.replicated_partition_exports` converges to `COMPLETED`. + +4. Re-issuing the same `EXPORT PARTITION` is rejected (no duplicate + files) unless `export_merge_tree_partition_force_export = 1`. Export + entries are kept indefinitely in `system.replicated_partition_exports` + as history; they never expire. This behavior avoids accidentally + exporting the same data twice. Note, however that forcing the + operation is dangerous if ClickHouse can't clean up the previous + operation. In this case you'll potentially commit files twice. + +5. Killing an in-flight partition export via `KILL EXPORT PARTITION` + transitions status to `KILLED` and stops all replicas' contributions. + +6. Exception during part export is counted in `PartsExportFailures`; + retry behavior honors `export_merge_tree_partition_max_retries`. The + same budget also bounds per-task commit retries for Iceberg + destinations; the task fails terminally if commit retries alone + exceed the budget. + +7. A partition export that remains in `PENDING` longer than + `export_merge_tree_partition_task_timeout_seconds` (default 3600s; + `0` disables) is auto-killed by the background cleanup loop and + transitions to `KILLED` with a timeout reason recorded in + `last_exception`. Enforcement is best-effort: actual kill latency is + bounded by one manifest-updater poll cycle (~30s) plus ZooKeeper + watch propagation. This is primarily a backstop against tasks stuck + on missing parts, a missing destination table, or an infinite + commit-retry loop against Iceberg. + +8. **Third-party Iceberg catalog manifest cleanup.** If the catalog + reaps old manifest files, its retention window MUST exceed + `export_merge_tree_partition_task_timeout_seconds`. Otherwise the + rare "sole node commits to Iceberg, crashes before the `COMPLETED` + status reaches Keeper, boots back up after the reaper has deleted + its commit manifest" scenario can produce duplicate data when the + recovered node retries the commit. ClickHouse does not reap its own + export artifacts; the Iceberg sidecars (`*_clickhouse_export_part_sidecar.avro`) + are safe to delete once the commit has landed. + +9. **Settings in `PART` vs `PARTIION` export.** There is a subtle difference in + how settings are handled between PART and PARTITION export. + - For part export, the settings used at the moment of the query are preserved + and re-stored in the background export task. + - For partition export, it is slightly harder to preserve the settings because + we need to serialize them in ZooKeeper. This is done for a few very important + settings including export_merge_tree_part_max_bytes_per_file. This will be + cleaned up in future iterations. + +### Settings + +| Setting | Scope | Default | Range / Values | Applies to | Description | +| --- | --- | --- | --- | --- | --- | +| `allow_experimental_export_merge_tree_part` | query | `false` | `Bool` | `EXPORT PART` | Experimental gate; required. | +| `allow_experimental_export_merge_tree_partition_feature` | server | `false` | `Bool` | `EXPORT PARTITION` | Experimental gate; required. | +| `export_merge_tree_part_overwrite_file_if_exists` | query | `false` | `Bool` | `EXPORT PART` | Overwrite existing destination file; otherwise throws. | +| `export_merge_tree_part_max_bytes_per_file` | query | `0` | `UInt64` (`0`=unlimited) | both | Soft cap per output file. Non-zero values can break idempotency. (SEE NOTE 1 below.)| +| `export_merge_tree_part_max_rows_per_file` | query | `0` | `UInt64` (`0`=unlimited) | both | Soft cap per output file. Non-zero values can break idempotency. | +| `export_merge_tree_part_throw_on_pending_mutations` | query | `true` | `Bool` | both | Refuse to export parts with pending mutations (unless mutation was `IN PARTITION`). | +| `export_merge_tree_part_throw_on_pending_patch_parts` | query | `true` | `Bool` | both | Refuse to export parts with pending patch parts. | +| `export_merge_tree_part_filename_pattern` | query | `{part_name}_{checksum}` | `String` | both | Filename template; supports `{part_name}`, `{checksum}`, `{database}`, `{table}`, server macros. | +| `export_merge_tree_partition_force_export` | query | `false` | `Bool` | `EXPORT PARTITION` | Overwrite a live Keeper manifest for the same `(source, destination, partition_id)`. Dangerous — can produce duplicate data on the destination; use with caution. | +| `export_merge_tree_partition_max_retries` | query | `3` | `UInt64` | `EXPORT PARTITION` | Retry budget applied to both per-part export attempts and per-task commit attempts (Iceberg). The task fails terminally if commit retries alone exceed the budget. | +| `export_merge_tree_partition_task_timeout_seconds` | query | `3600` (seconds) | `UInt64` (`0`=disable) | `EXPORT PARTITION` | Wall-clock cap for `PENDING` tasks; on expiry transitions to `KILLED` with a timeout reason. Measured from manifest `create_time`. Enforcement latency ≈ one manifest-updater poll cycle (~30s) plus Keeper watch propagation. | +| `export_merge_tree_partition_system_table_prefer_remote_information` | query | `false` | `Bool` | `EXPORT PARTITION` | When `true`, `system.replicated_partition_exports` fetches fresh state from Keeper (requires the `MULTI_READ` feature flag); when `false`, uses local cached state. **Default flipped from `true` to `false` in this release** — Keeper round-trips were more expensive than warranted for the typical observability workload. (See NOTE 2.)| +| `export_merge_tree_part_file_already_exists_policy` | query | `skip` | `skip` / `error` / `overwrite` | `EXPORT PARTITION` | Per-file policy during partition export. | + +Default-value impact: all new settings default to "off" or to +conservative values (pending-mutation guards default to throwing). One +default *changed* in this release: +`export_merge_tree_partition_system_table_prefer_remote_information` +flipped from `true` to `false`, so `system.replicated_partition_exports` +now serves local cached state by default instead of querying Keeper. +Users who relied on always-fresh results must set it back to `true` +explicitly. + +**NOTE 1:** `export_merge_tree_part_max_bytes_per_file` overrides `iceberg_insert_max_bytes_in_data_file` or +other more specific file size parameters. + +EXPORT should also observe the following existing settings for export to Iceberg / Parquet: +- `iceberg_insert_max_bytes_in_data_file` (as above, overridden by `export_merge_tree_part_max_bytes_per_file` if specified) +- `iceberg_insert_max_rows_in_data_file` +- `output_format_parquet_row_group_size` +- `output_format_parquet_row_group_size_bytes` +- `output_format_parquet_data_page_size` +- `output_format_parquet_compression_method` +- `output_format_parquet_version` + +**NOTE 2:** `export_merge_tree_partition_system_table_prefer_remote_information` may be dropped. +Querying Keeper from the user path is complex and has side effects. + +### System tables / metrics / log messages / observability + +- `system.exports` — rows for currently-executing part exports (source/destination tables, + `part_name`, destination paths, `elapsed`, rows/bytes counters, memory counters). Dropped when + the export completes. +- `system.replicated_partition_exports` — rows for `EXPORT PARTITION` tasks. Backed by Keeper; + querying it is a Keeper round-trip and should be used sparingly. Columns include + `source_database`, `source_table`, `destination_database`, `destination_table`, `create_time`, + `partition_id`, `transaction_id`, `source_replica`, `parts`, `parts_count`, `parts_to_do`, + `status`, `exception_replica`, `last_exception`, `exception_part`, `exception_count`. +- `system.part_log` — each completed part export appends one row with `event_type = 'ExportPart'`, + filled `remote_file_paths`, `merged_from = [part_name]`, plus standard timing / size / error + fields. +- `ProfileEvents`: + - `PartsExports` — successful part-export completions. + - `PartsExportFailures` — failures. + - `PartsExportDuplicated` — skipped because destination file already existed. + - `PartsExportTotalMilliseconds` — cumulative wall time. +- Status enum on `system.replicated_partition_exports`: `PENDING`, `COMPLETED`, `FAILED`, `KILLED`. + +### Error behavior + +- Missing experimental flag: `SUPPORT_IS_DISABLED` (exact code TBD — confirm against + `src/Common/ErrorCodes.cpp`). +- Destination schema mismatch (columns / types / order; source `EPHEMERAL` column present in + destination): `INCOMPATIBLE_COLUMNS`. +- Destination engine doesn't support exports (e.g. `url`, unknown engine): `NOT_IMPLEMENTED`. +- Destination is an unknown table function: `UNKNOWN_FUNCTION`. +- Pending mutations or patch parts when the guard is enabled: `BAD_ARGUMENTS` (TBD — confirm). +- Part not found on any replica: `NO_SUCH_DATA_PART` (TBD — confirm). +- Destination file already exists and policy is `error` / `export_merge_tree_part_overwrite_file_if_exists = 0`: + `FILE_ALREADY_EXISTS` (TBD — confirm). +- Duplicate live manifest without `..._force_export`: `DUPLICATE_EXPORT_TASK` or equivalent + (TBD — confirm). +- Commit-retry budget exhausted (Iceberg destination): terminal `FAILED` with + `last_exception` populated; the task is not retried further. +- Task timeout exceeded (`export_merge_tree_partition_task_timeout_seconds`): + terminal `KILLED` with a timeout reason in `last_exception`. + +All of the above **throw exceptions** or surface through `last_exception` on the +task row; they do not crash the server. + +### Backward compatibility +- **Older client → newer server:** harmless — the client issues the new `ALTER` text; the server + parses it. No wire-protocol change. +- **Newer client → older server:** older server fails parse on `EXPORT PART` / `EXPORT PARTITION` + with `SYNTAX_ERROR`. Acceptable. +- **Mixed-version cluster (replication):** `EXPORT PART` is local to one replica; no cross-replica + effect. `EXPORT PARTITION` stores its manifest under the table's Keeper path in a new + subtree; replicas on older versions ignore unknown nodes but will NOT contribute parts to the + export — the initiating replica (which must be on the new version) completes alone if it holds + all the parts; otherwise the task stalls on `parts_to_do > 0`. An upgrade-ordering note for + operators is required (section 4 / Rollout). +- **On-disk format:** unchanged. Parts are read as-is; Parquet is produced on the fly. +- **Default-value changes:** `export_merge_tree_partition_system_table_prefer_remote_information` + flipped from `true` to `false`. Users running dashboards that read + `system.replicated_partition_exports` and require always-fresh state must set this + back to `true` explicitly (and ensure the Keeper `MULTI_READ` feature flag is enabled). + No other defaults that affect existing workloads (see settings table). + +--- + +## 3. Implementation + +This design only covers user visible behavior. It does not internal implementatation +details. The implementation section is omitted. + +## 4. Test plan + +### Functional tests — `tests/queries/0_stateless` + +Existing coverage to retain: + +- `03572_export_merge_tree_part_basic.sh` — golden path, idempotent re-export, wildcard + + hive partition strategies. +- `03572_export_merge_tree_part_to_object_storage_simple.sql` — error cases + (`INCOMPATIBLE_COLUMNS`, `NOT_IMPLEMENTED`, `UNKNOWN_FUNCTION`, `EPHEMERAL` collision). +- `03572_export_merge_tree_part_limits_and_table_functions.sh` — `max_bytes_per_file`, + `max_rows_per_file`, table-function destination with schema inheritance / explicit structure. +- `03572_export_merge_tree_part_special_columns.sh` — `ALIAS`, `MATERIALIZED`, `EPHEMERAL`, + mixed / complex expressions. +- `03572_export_replicated_merge_tree_part_to_object_storage.sh` + + `03572_export_replicated_merge_tree_part_to_object_storage_simple.sql` — part-level export + from `ReplicatedMergeTree`. +- `03604_export_merge_tree_partition.sh` — basic `EXPORT PARTITION ID`. +- `03608_export_merge_tree_part_filename_pattern.sh` — default and custom + `export_merge_tree_part_filename_pattern` including `{database}` / `{table}` macros. + +New tests to add: + +- `NNNN_export_merge_tree_part_pending_mutations.sh` — with/without the + `..._throw_on_pending_mutations` / `..._throw_on_pending_patch_parts` guards and `IN PARTITION` + mutations. +- `NNNN_export_merge_tree_part_commit_file.sh` — verify a `commit__` file exists + alongside every successful export and references every written data file; verify that a + partial run (simulated by killing before commit) does not produce a commit file. +- `NNNN_export_merge_tree_part_overwrite_policy.sh` — all three values of + `export_merge_tree_part_file_already_exists_policy` (`skip`, `error`, `overwrite`) plus + `export_merge_tree_part_overwrite_file_if_exists`. +- `NNNN_export_merge_tree_part_profile_events.sh` — assert `PartsExports`, + `PartsExportFailures`, `PartsExportDuplicated`, `PartsExportTotalMilliseconds` move as + expected. + +Note: Iceberg-destination coverage is deliberately in integration (next section), not in +`tests/queries/0_stateless`, because it requires a live warehouse and — for the catalog +path — a REST / Glue fixture. The commit-file test above only exercises plain +object-storage atomicity. + +Do not add `no-parallel` to any new test unless explicitly required by shared S3 bucket paths; +`03604` currently has the tag and should be re-examined to see whether unique per-run paths +remove the need. + +### Integration tests — `tests/integration` + +**Keep (modified in this PR):** + +- `test_export_merge_tree_part_to_object_storage/` — part export in a multi-node setup. + PR 1618 makes minor adjustments. +- `test_export_replicated_mt_partition_to_object_storage/` — partition export across + replicas, including `wait_for_export_status`, retry counting, and replica failure + scenarios. PR 1618 removes the `s3_retries.xml` config and reshapes several test cases + against the new shared helpers. + +**New in PR 1618:** + +- `test_export_merge_tree_part_to_iceberg/` — per-part export to an Iceberg destination, + covering golden path, sidecar emission, manifest shape, and error paths. +- `test_export_replicated_mt_partition_to_iceberg/` — distributed partition export to + Iceberg across replicas, including `test_export_task_timeout_kills_stuck_pending_task` + (uses the `export_partition_commit_always_throw` failpoint to exhaust the commit path, + then asserts the timeout transitions the task to `KILLED`). +- `test_storage_iceberg_with_spark/test_export_partition_iceberg.py` — catalog-less + Iceberg round-trip; Spark reads ClickHouse-written data and verifies schema, partition + layout, and snapshot atomicity. +- `test_storage_iceberg_with_spark/test_export_partition_iceberg_catalog.py` — + catalog-backed (REST) round-trip; exercises the `commitExportPartitionTransaction` + path against a real catalog. +- Shared helpers: + - `tests/integration/helpers/export_partition_helpers.py` — shared + `wait_for_export_status`, manifest-inspection utilities. + - `tests/integration/helpers/iceberg_export_stats.py` — sidecar decoders and stats + assertion helpers. + +**Remaining gaps to add:** + +- Initiating-replica dies mid-commit (post-data-file-write, pre-catalog-CAS) — asserts a + surviving replica completes via the Keeper-stashed `metadata.json` snapshot, and the + `clickhouse.export-partition-transaction-id` idempotency check prevents double-commit. +- Experimental feature disabled on one replica + (`disable_experimental_export_partition.xml` config) and enabled on the rest — task + still completes via the enabled replicas. +- `KILL EXPORT PARTITION` against an Iceberg task in mid-commit: status transitions to + `KILLED`, no dangling half-written `metadata.json`, data files and sidecars remain as + orphans (cleanup is user responsibility per Operational note 7 in § 2). +- Mixed-version cluster: upgrade scenario where only some replicas know about + `EXPORT PARTITION` or the new Keeper manifest fields. + +Invocation: +`python -m ci.praktika run "integration" --test test_export_merge_tree_part_to_object_storage,test_export_replicated_mt_partition_to_object_storage,test_export_merge_tree_part_to_iceberg,test_export_replicated_mt_partition_to_iceberg,test_storage_iceberg_with_spark`. + +### Failpoints + +The following failpoints are registered for deterministic testing of crash windows and +retry logic. Enable via `SYSTEM ENABLE FAILPOINT ` from test harnesses. + +| Failpoint | Kind | What it tests | +| --- | --- | --- | +| `iceberg_writes_non_retry_cleanup` | ONCE | Cleanup path when an Iceberg write fails in a non-retryable way. | +| `iceberg_writes_post_publish_throw` | ONCE | Commit succeeded in object storage but the publish step throws — exercises the recovery path that must not double-commit. | +| `iceberg_export_after_commit_before_zk_completed` | ONCE | Crash window between a successful Iceberg commit and the `COMPLETED` Keeper status update — the idempotency marker (`clickhouse.export-partition-transaction-id`) must prevent a second commit on recovery. | +| `export_partition_commit_always_throw` | REGULAR | Every commit attempt throws — used to exhaust `max_retries` and drive the task-timeout path. | +| `export_partition_status_change_throw` | ONCE | Throws during a manifest status transition — exercises the status-drain lock invariant (decision #11 in § 3) and the manifest-updating task's retry logic. | + +These failpoints replace the "simulate a crash before commit" phrasing in the +functional-tests section above; prefer them over process kills for deterministic CI +behavior. + +### Performance tests — `tests/performance` + +Add `export_merge_tree_part.xml`: compare `ALTER TABLE ... EXPORT PART` vs. +`INSERT INTO s3_t SELECT * FROM mt WHERE _part = ...` on a ~1 GB Wide part; track wall time and +peak memory. Hot path is the Parquet encoder, which warrants a guard against regressions. + +- Iceberg-destination benchmark: the Iceberg commit is O(data files), not per-row, so the + per-part write path performance should match plain object storage within noise. A + secondary benchmark comparing `EXPORT PARTITION` to Iceberg vs. to hive-layout S3 over + a ≥1000-part partition would catch regressions in the sidecar / manifest-assembly code + path specifically. + +### Manual verification + +- Roundtrip: export partition → read via `SELECT * FROM s3(...)` → create new + `ReplicatedMergeTree` from the S3 data → row counts / checksums match the source. +- `system.replicated_partition_exports` behaviour under a crashed initiator (cluster restart). +- Object-storage layout inspection via `s3(..., format=One)` listing: exactly N data files + 1 + commit file per transaction. +- Iceberg roundtrip with an external reader: export a partition to an Iceberg destination + → read it back through the same catalog → row counts and column checksums match the + source. DuckDB is the preferred external reader here (lightweight, fast to stand up, + mature Iceberg support); Spark or Trino may be substituted where a specific catalog + integration needs to be exercised. Confirms the on-disk metadata we write is actually + interoperable, not just self-consistent. + +### Rollout / risk + +- **Risk (Keeper schema extension):** the `partition_exports` subtree is write-once; a + partially-rolled-out cluster where only some replicas understand the subtree — or the + new manifest fields (`task_timeout_seconds`, `commit_attempts`, Iceberg `metadata.json` + snapshot, `write_full_path_in_iceberg_metadata`) — will stall partition exports + (`parts_to_do > 0`) rather than corrupt data. Acceptable but must be documented in the + upgrade notes. +- **Risk (object-storage cost / accidental large exports):** mitigated by the experimental + gates (default off) and the duplicate-export rejection (an existing export key is refused + unless `export_merge_tree_partition_force_export` is set). +- **Risk (Iceberg catalog manifest retention):** if the catalog reaps old manifest files + with a retention window shorter than `export_merge_tree_partition_task_timeout_seconds`, + the rare "sole-node commits, crashes, recovers after reaper deleted the commit manifest" + scenario can duplicate data. Operators MUST verify their catalog's retention before + enabling Iceberg destinations (see Operational note 7 in § 2). +- **Risk (default flip on `_prefer_remote_information`):** dashboards that read + `system.replicated_partition_exports` now see local cached state by default instead of + Keeper-fresh state. Existing users must set the flag back to `true` explicitly if they + rely on always-fresh results (and ensure `MULTI_READ` is enabled). +- **Flag strategy:** ship with `allow_experimental_export_merge_tree_part` (query, default + `false`) and `allow_experimental_export_merge_tree_partition_feature` (server, default + `false`). Both destination families ride on these gates — no separate Iceberg gate. + Flip defaults to `true` only after: (a) the open questions in § 3 are resolved, (b) the + remaining integration gaps listed above are closed, (c) one release cycle of customer + feedback. +- **Watch in production:** + - `PartsExportFailures` (existing). + - `exception_count` on `system.replicated_partition_exports` (existing). + - Keeper watch counts under the `partition_exports` subtree (existing). + - Object-storage request-error rates (existing). + - `commit_attempts` values approaching `max_retries` — signal of a catalog or network + issue throttling commits. + - Rate of `KILLED` transitions with timeout reason — signal of tasks stuck on missing + parts, missing destination, or commit backpressure. + - For Iceberg destinations: `metadata/` prefix write-error rate (CAS contention, + vended-credential expiry) and catalog-API error rate. diff --git a/docs/en/antalya/partition_export.md b/docs/en/antalya/partition_export.md index 5365ed90e353..ca465a8dea82 100644 --- a/docs/en/antalya/partition_export.md +++ b/docs/en/antalya/partition_export.md @@ -59,7 +59,7 @@ TO TABLE [destination_database.]destination_table - **Type**: `Bool` - **Default**: `false` -- **Description**: Ignore existing partition export and overwrite the ZooKeeper entry. Allows re-exporting a partition to the same destination before the manifest expires. **IMPORTANT:** this is dangerous because it can lead to duplicated data, use it with caution. +- **Description**: Ignore existing partition export and overwrite the ZooKeeper entry. Allows re-exporting a partition that was already exported to the same destination. **IMPORTANT:** this is dangerous because it can lead to duplicated data, use it with caution. #### `export_merge_tree_partition_max_retries` (Optional) @@ -67,12 +67,6 @@ TO TABLE [destination_database.]destination_table - **Default**: `3` - **Description**: Maximum number of retries for exporting a merge tree part in an export partition task. If it exceeds, the entire task fails. -#### `export_merge_tree_partition_manifest_ttl` (Optional) - -- **Type**: `UInt64` -- **Default**: `180` (seconds) -- **Description**: Determines how long the manifest will live in ZooKeeper. It prevents the same partition from being exported twice to the same destination. This setting does not affect or delete in-progress tasks; it only cleans up completed ones. - #### `export_merge_tree_part_file_already_exists_policy` (Optional) - **Type**: `MergeTreePartExportFileAlreadyExistsPolicy` @@ -109,7 +103,6 @@ When the timeout is exceeded the task transitions to KILLED (same terminal state Notes: - Enforcement is best-effort: actual kill latency is bounded by one manifest-updater poll cycle (~30s) plus ZooKeeper watch propagation. -- Since both this timeout and `export_merge_tree_partition_manifest_ttl` are measured from `create_time`, keep `export_merge_tree_partition_manifest_ttl` greater than `export_merge_tree_partition_task_timeout_seconds` if you want the KILLED entry to remain visible in `system.replicated_partition_exports` after the timeout fires. ## Examples diff --git a/src/Core/Settings.cpp b/src/Core/Settings.cpp index 3dd6dd861157..6a20d23a8a6c 100644 --- a/src/Core/Settings.cpp +++ b/src/Core/Settings.cpp @@ -8043,18 +8043,21 @@ Ignore existing partition export and overwrite the zookeeper entry DECLARE(UInt64, export_merge_tree_partition_max_retries, 3, R"( Maximum number of retries for exporting a merge tree part in an export partition task )", 0) \ +<<<<<<< HEAD DECLARE(UInt64, export_merge_tree_partition_manifest_ttl, 86400, R"( Determines how long the manifest will live in ZooKeeper. It prevents the same partition from being exported twice to the same destination. This setting does not affect / delete in progress tasks. It'll only cleanup the completed ones. )", 0) \ DECLARE(UInt64, export_merge_tree_partition_task_timeout_seconds, 3600, R"( +======= + DECLARE(UInt64, export_merge_tree_partition_task_timeout_seconds, 86400, R"( +>>>>>>> 1e5a11cb4eb (Merge pull request #1917 from Altinity/do_not_evict_entries_from_replicated_partition_exports_table) Maximum wall-clock duration (in seconds) an export partition task is allowed to remain in the PENDING state before it is auto-killed by the background cleanup loop. The timeout is measured from the manifest's create_time. Set to 0 to disable the timeout. When the timeout is exceeded the task transitions to KILLED (same terminal state as `KILL QUERY ... EXPORT PARTITION`), and `last_exception` is populated with a timeout reason. Notes: - Enforcement is best-effort: actual kill latency is bounded by one manifest-updater poll cycle (~30s) plus ZooKeeper watch propagation. -- Since both this timeout and `export_merge_tree_partition_manifest_ttl` are measured from `create_time`, keep `export_merge_tree_partition_manifest_ttl` greater than `export_merge_tree_partition_task_timeout_seconds` if you want the KILLED entry to remain visible in `system.replicated_partition_exports` after the timeout fires. )", 0) \ DECLARE(MergeTreePartExportFileAlreadyExistsPolicy, export_merge_tree_part_file_already_exists_policy, MergeTreePartExportFileAlreadyExistsPolicy::skip, R"( Possible values: @@ -8491,7 +8494,11 @@ Maximum number of texts to include in a single HTTP request made by `aiEmbed`. T #define OBSOLETE_SETTINGS(M, ALIAS) \ /** Obsolete settings which are kept around for compatibility reasons. They have no effect anymore. */ \ +<<<<<<< HEAD MAKE_OBSOLETE(M, Bool, allow_experimental_query_deduplication, false) \ +======= + MAKE_OBSOLETE(M, UInt64, export_merge_tree_partition_manifest_ttl, 86400) \ +>>>>>>> 1e5a11cb4eb (Merge pull request #1917 from Altinity/do_not_evict_entries_from_replicated_partition_exports_table) MAKE_OBSOLETE(M, Bool, query_condition_cache_store_conditions_as_plaintext, false) \ MAKE_OBSOLETE(M, Bool, update_insert_deduplication_token_in_dependent_materialized_views, 0) \ MAKE_OBSOLETE(M, UInt64, max_memory_usage_for_all_queries, 0) \ diff --git a/src/Storages/ExportReplicatedMergeTreePartitionManifest.h b/src/Storages/ExportReplicatedMergeTreePartitionManifest.h index acfabc28ca61..bdebc63565d1 100644 --- a/src/Storages/ExportReplicatedMergeTreePartitionManifest.h +++ b/src/Storages/ExportReplicatedMergeTreePartitionManifest.h @@ -164,7 +164,6 @@ struct ExportReplicatedMergeTreePartitionManifest std::vector parts; time_t create_time; size_t max_retries; - size_t ttl_seconds; size_t task_timeout_seconds; size_t max_threads; bool parallel_formatting; @@ -205,7 +204,6 @@ struct ExportReplicatedMergeTreePartitionManifest json.set("filename_pattern", filename_pattern); json.set("create_time", create_time); json.set("max_retries", max_retries); - json.set("ttl_seconds", ttl_seconds); json.set("task_timeout_seconds", task_timeout_seconds); json.set("write_full_path_in_iceberg_metadata", write_full_path_in_iceberg_metadata); std::ostringstream oss; // STYLE_CHECK_ALLOW_STD_STRING_STREAM @@ -240,7 +238,6 @@ struct ExportReplicatedMergeTreePartitionManifest manifest.parts.push_back(parts_array->getElement(static_cast(i))); manifest.create_time = json->getValue("create_time"); - manifest.ttl_seconds = json->getValue("ttl_seconds"); manifest.task_timeout_seconds = json->getValue("task_timeout_seconds"); manifest.max_threads = json->getValue("max_threads"); manifest.parallel_formatting = json->getValue("parallel_formatting"); diff --git a/src/Storages/MergeTree/ExportPartitionManifestUpdatingTask.cpp b/src/Storages/MergeTree/ExportPartitionManifestUpdatingTask.cpp index b45f3667be87..8ba8d8fc64a0 100644 --- a/src/Storages/MergeTree/ExportPartitionManifestUpdatingTask.cpp +++ b/src/Storages/MergeTree/ExportPartitionManifestUpdatingTask.cpp @@ -128,54 +128,35 @@ namespace } /* - Remove expired entries and fix non-committed exports that have already exported all parts. - - Return values: - - true: the cleanup was successful, the entry is removed from the entries_by_key container and the function returns true. Proceed to the next entry. - - false: the cleanup was not successful, the entry is not removed from the entries_by_key container and the function returns false. + Enforce the PENDING task timeout and recover non-committed exports that have already + exported all parts. Entries are never removed for age — `system.replicated_partition_exports` + is append-only history, so the entry always stays in the in-memory container: a KILLED + transition is driven by the status watch, and a deferred commit is handled by the caller + after the lock is released. Side outputs: - `deferred_commits`: when a PENDING entry has all parts processed but the export was never committed, this function appends a CommitRecoveryWork item to be executed by the caller after releasing the storage-wide mutex. The actual commit() call (which performs network I/O to the destination catalog and S3) MUST NOT run under the lock. - The function still returns `false` in that case so the outer poll() loop falls through - to `addTask`, keeping the in-memory entry consistent regardless of whether the - deferred commit ultimately succeeds. */ - bool tryCleanup( + void tryCleanup( const zkutil::ZooKeeperPtr & zk, const std::string & entry_path, const LoggerPtr & log, const ContextPtr & storage_context, StorageReplicatedMergeTree & storage, - const std::string & key, const ExportReplicatedMergeTreePartitionManifest & metadata, const time_t now, const bool is_pending, - auto & entries_by_key, std::vector & deferred_commits ) { - bool has_expired = metadata.create_time < now - static_cast(metadata.ttl_seconds); - bool task_timed_out = is_pending && metadata.task_timeout_seconds > 0 && metadata.create_time + static_cast(metadata.task_timeout_seconds) < now; - if (has_expired && !is_pending) - { - zk->tryRemoveRecursive(fs::path(entry_path)); - ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRequests); - ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRemoveRecursive); - auto it = entries_by_key.find(key); - if (it != entries_by_key.end()) - entries_by_key.erase(it); - LOG_INFO(log, "ExportPartition Manifest Updating Task: Removed {}: expired", key); - - return true; - } - else if (task_timed_out) + if (task_timed_out) { const std::string status_path = fs::path(entry_path) / "status"; @@ -187,14 +168,14 @@ namespace if (!zk->tryGet(status_path, status_string, &status_stat)) { LOG_INFO(log, "ExportPartition Manifest Updating Task: Failed to read status for {} while enforcing task timeout, skipping", entry_path); - return false; + return; } const auto current_status = magic_enum::enum_cast(status_string); if (!current_status || *current_status != ExportReplicatedMergeTreePartitionTaskEntry::Status::PENDING) { LOG_INFO(log, "ExportPartition Manifest Updating Task: Task {} is not PENDING, can't set to KILLED, skipping", entry_path); - return false; + return; } const auto timeout_message = fmt::format( @@ -234,9 +215,9 @@ namespace entry_path, rc); } - /// Return false so the entry remains in entries_by_key; the status watch will drive + /// The entry remains in entries_by_key; the status watch will drive /// handleStatusChanges -> killExportPart on every replica, mirroring user-initiated KILL. - return false; + return; } else if (is_pending) { @@ -249,7 +230,7 @@ namespace { LOG_INFO(log, "ExportPartition Manifest Updating Task: Failed to get parts in processing or pending, skipping"); - return false; + return; } if (parts_in_processing_or_pending.empty()) @@ -261,7 +242,7 @@ namespace if (!destination_storage) { LOG_INFO(log, "ExportPartition Manifest Updating Task: Failed to reconstruct destination storage: {}, skipping", destination_storage_id.getNameForLogs()); - return false; + return; } /// A replica exported the last part but the commit never landed. Capture everything @@ -270,22 +251,18 @@ namespace /// MAX_TRANSACTION_RETRIES = 100 retries; holding the storage-wide mutex across /// that work is what caused `system.replicated_partition_exports` to hang. /// - /// Returning false here keeps the outer poll() loop on the normal path: it will - /// call addTask() so the in-memory container reflects the PENDING entry. The - /// status watch registered by poll() will transition the local entry to - /// COMPLETED/FAILED once the deferred commit (or a peer's commit) updates - /// /status in ZooKeeper. + /// The outer poll() loop stays on the normal path: it will call addTask() so the + /// in-memory container reflects the PENDING entry. The status watch registered by + /// poll() will transition the local entry to COMPLETED/FAILED once the deferred + /// commit (or a peer's commit) updates /status in ZooKeeper. deferred_commits.push_back(CommitRecoveryWork{ .metadata = metadata, .entry_path = entry_path, .destination_storage = destination_storage, .context = context, }); - return false; } } - - return false; } } @@ -365,8 +342,8 @@ void ExportPartitionManifestUpdatingTask::poll() const std::string cleanup_lock_path = fs::path(storage.zookeeper_path) / "exports_cleanup_lock"; /// The `exports_cleanup_lock` is an ephemeral ZK node that serializes cleanup work - /// across replicas: only the replica holding it walks `tryCleanup` (entry expiry + - /// commit recovery). It MUST outlive the deferred-commit loop below; otherwise a peer + /// across replicas: only the replica holding it walks `tryCleanup` (task-timeout + /// enforcement + commit recovery). It MUST outlive the deferred-commit loop below; otherwise a peer /// replica's next poll() could acquire it and race us on the same commit-recovery work, /// duplicating REST-catalog round-trips and snapshot writes. The EphemeralNodeHolder /// destructor removes the node, so we declare it at function scope and let it die @@ -468,25 +445,20 @@ void ExportPartitionManifestUpdatingTask::poll() continue; } - /// if we have the cleanup lock, try to cleanup - /// if we successfully cleaned it up, early exit + /// If we hold the cleanup lock, enforce the task timeout and recover uncommitted exports. + /// Entries are never removed here, so we always fall through to refresh / addTask below. if (cleanup_lock) { - bool cleanup_successful = tryCleanup( + tryCleanup( zk, entry_path, storage.log.load(), storage.getContext(), storage, - key, metadata, now, *status == ExportReplicatedMergeTreePartitionTaskEntry::Status::PENDING, - entries_by_key, deferred_commits); - - if (cleanup_successful) - continue; } if (has_local_entry_and_is_up_to_date) diff --git a/src/Storages/StorageReplicatedMergeTree.cpp b/src/Storages/StorageReplicatedMergeTree.cpp index 2c19c4e7fec5..cd17e609d9e7 100644 --- a/src/Storages/StorageReplicatedMergeTree.cpp +++ b/src/Storages/StorageReplicatedMergeTree.cpp @@ -225,7 +225,6 @@ namespace Setting extern const SettingsBool allow_experimental_export_merge_tree_part; extern const SettingsBool export_merge_tree_partition_force_export; extern const SettingsUInt64 export_merge_tree_partition_max_retries; - extern const SettingsUInt64 export_merge_tree_partition_manifest_ttl; extern const SettingsUInt64 export_merge_tree_partition_task_timeout_seconds; extern const SettingsBool output_format_parallel_formatting; extern const SettingsBool output_format_parquet_parallel_encoding; @@ -8625,36 +8624,11 @@ void StorageReplicatedMergeTree::exportPartitionToTable(const PartitionCommand & ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperExists); if (zookeeper->exists(partition_exports_path)) { - LOG_INFO(log, "Export with key {} is already exported or it is being exported. Checking if it has expired so that we can overwrite it", export_key); + LOG_INFO(log, "Export with key {} is already exported or it is being exported", export_key); - bool has_expired = false; - - ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRequests); - ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperExists); - if (zookeeper->exists(fs::path(partition_exports_path) / "metadata.json")) - { - std::string metadata_json; - ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRequests); - ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperGet); - if (zookeeper->tryGet(fs::path(partition_exports_path) / "metadata.json", metadata_json)) - { - const auto manifest = ExportReplicatedMergeTreePartitionManifest::fromJsonString(metadata_json); - - const auto now = time(nullptr); - const auto expiration_time = manifest.create_time + manifest.ttl_seconds; - - LOG_INFO(log, "Export with key {} has expiration time {}, now is {}", export_key, expiration_time, now); - - if (static_cast(expiration_time) < now) - { - has_expired = true; - } - } - } - - if (!has_expired && !query_context->getSettingsRef()[Setting::export_merge_tree_partition_force_export]) + if (!query_context->getSettingsRef()[Setting::export_merge_tree_partition_force_export]) { - throw Exception(ErrorCodes::EXPORT_PARTITION_ALREADY_EXPORTED, "Export with key {} already exported or it is being exported, and it has not expired. Set `export_merge_tree_partition_force_export` to overwrite it.", export_key); + throw Exception(ErrorCodes::EXPORT_PARTITION_ALREADY_EXPORTED, "Export with key {} already exported or it is being exported. Set `export_merge_tree_partition_force_export` to overwrite it.", export_key); } LOG_INFO(log, "Overwriting export with key {}", export_key); @@ -8736,7 +8710,6 @@ void StorageReplicatedMergeTree::exportPartitionToTable(const PartitionCommand & manifest.parts = part_names; manifest.create_time = time(nullptr); manifest.max_retries = query_context->getSettingsRef()[Setting::export_merge_tree_partition_max_retries]; - manifest.ttl_seconds = query_context->getSettingsRef()[Setting::export_merge_tree_partition_manifest_ttl]; manifest.task_timeout_seconds = query_context->getSettingsRef()[Setting::export_merge_tree_partition_task_timeout_seconds]; manifest.max_threads = query_context->getSettingsRef()[Setting::max_threads]; manifest.parallel_formatting = query_context->getSettingsRef()[Setting::output_format_parallel_formatting]; diff --git a/tests/integration/test_export_replicated_mt_partition_to_iceberg/test.py b/tests/integration/test_export_replicated_mt_partition_to_iceberg/test.py index 8b49589a3005..3f0947dcfa1a 100644 --- a/tests/integration/test_export_replicated_mt_partition_to_iceberg/test.py +++ b/tests/integration/test_export_replicated_mt_partition_to_iceberg/test.py @@ -746,49 +746,6 @@ def test_partition_key_compatibility_check(cluster): ) -def test_export_ttl(cluster): - """ - After a manifest TTL expires the same partition can be re-exported, and the - new data is appended to (or replaces) what is in the Iceberg table. - """ - node = cluster.instances["replica1"] - ttl_seconds = 3 - - uid = unique_suffix() - mt_table = f"mt_{uid}" - iceberg_table = f"iceberg_{uid}" - - setup_tables(cluster, mt_table, iceberg_table, nodes=["replica1"]) - - # First export. - node.query( - f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {iceberg_table} " - f"SETTINGS export_merge_tree_partition_manifest_ttl = {ttl_seconds}, allow_insert_into_iceberg = 1" - ) - - # A second export before the TTL expires must be rejected. - error = node.query_and_get_error( - f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {iceberg_table}", - settings={"allow_insert_into_iceberg": 1}, - ) - assert "Export with key" in error, f"Expected duplicate-export error before TTL, got: {error}" - - wait_for_export_status(node, mt_table, iceberg_table, "2020", "COMPLETED") - - count_after_first = int(node.query(f"SELECT count() FROM {iceberg_table} WHERE year = 2020").strip()) - assert count_after_first == 3, f"Expected 3 rows after first export, got {count_after_first}" - - # Wait for the manifest TTL to expire. - time.sleep(ttl_seconds * 2) - - # Second export must be accepted now. - node.query( - f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {iceberg_table}", - settings={"allow_insert_into_iceberg": 1}, - ) - wait_for_export_status(node, mt_table, iceberg_table, "2020", "COMPLETED") - - def test_export_data_files_are_not_cleaned_up_on_commit_failure(cluster): """ Verify that the data files are not cleaned up on commit failure and the export is retried. @@ -888,7 +845,6 @@ def test_export_task_timeout_kills_stuck_pending_task(cluster): f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {iceberg_table}" f" SETTINGS export_merge_tree_partition_task_timeout_seconds = 5," f" export_merge_tree_partition_max_retries = 1000000," - f" export_merge_tree_partition_manifest_ttl = 3600," f" allow_insert_into_iceberg = 1" ) diff --git a/tests/integration/test_export_replicated_mt_partition_to_object_storage/test.py b/tests/integration/test_export_replicated_mt_partition_to_object_storage/test.py index 3161e3b67100..93eb070b3876 100644 --- a/tests/integration/test_export_replicated_mt_partition_to_object_storage/test.py +++ b/tests/integration/test_export_replicated_mt_partition_to_object_storage/test.py @@ -627,42 +627,6 @@ def test_inject_short_living_failures(cluster): assert int(exception_count.strip()) >= 1, "Expected at least one exception" -def test_export_ttl(cluster): - node = cluster.instances["replica1"] - - postfix = str(uuid.uuid4()).replace("-", "_") - mt_table = f"export_ttl_mt_table_{postfix}" - s3_table = f"export_ttl_s3_table_{postfix}" - - expiration_time = 3 - - create_tables_and_insert_data(node, mt_table, s3_table, "replica1") - - # start export - node.query(f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {s3_table} SETTINGS export_merge_tree_partition_manifest_ttl={expiration_time};") - - # assert that I get an error when trying to export the same partition again, query_and_get_error - error = node.query_and_get_error(f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {s3_table};") - assert "Export with key" in error, "Expected error about expired export" - - # wait for the export to finish and for the manifest to expire - wait_for_export_status(node, mt_table, s3_table, "2020", "COMPLETED") - time.sleep(expiration_time * 2) - - # assert that the export succeeded, check the commit file - assert node.query(f"SELECT count() FROM s3(s3_conn, filename='{s3_table}/commit_2020_*', format=LineAsString)") == '1\n', "Export did not succeed" - - # start export again - node.query(f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {s3_table}") - - # wait for the export to finish - wait_for_export_status(node, mt_table, s3_table, "2020", "COMPLETED") - - # assert that the export succeeded, check the commit file - # there should be two commit files now, one for the first export and one for the second export - assert node.query(f"SELECT count() FROM s3(s3_conn, filename='{s3_table}/commit_2020_*', format=LineAsString)") == '2\n', "Export did not succeed" - - def test_export_partition_file_already_exists_policy(cluster): node = cluster.instances["replica1"] From 40072662e54c415c39d391350d4c66a58b201eda Mon Sep 17 00:00:00 2001 From: Andrey Zvonov <32552679+zvonand@users.noreply.github.com> Date: Wed, 5 Aug 2026 19:00:25 +0200 Subject: [PATCH 28/43] Resolve conflicts in cherry-pick of #1917 src/Core/Settings.cpp: - Removed the DECLARE of export_merge_tree_partition_manifest_ttl (source PR) while keeping antalya-26.6's export_merge_tree_partition_task_timeout_seconds default of 3600 (the 86400 on "theirs" is unrelated base-branch context). - Added the MAKE_OBSOLETE row for export_merge_tree_partition_manifest_ttl from the source PR, keeping the pre-existing allow_experimental_query_deduplication row that "ours" already had in that append-only list. antalya/docs/design/alter-table-export-part-partition.md does not exist on antalya-26.6 (introduced by the separate design-doc PR #1673, not ported); the cherry-pick recreated the whole 752-line file, of which only ~15 lines belong to #1917. Kept "ours" (file absent) instead of importing another PR's document. Source-PR: #1917 (https://github.com/Altinity/ClickHouse/pull/1917) --- .../alter-table-export-part-partition.md | 752 ------------------ src/Core/Settings.cpp | 13 +- 2 files changed, 1 insertion(+), 764 deletions(-) delete mode 100644 antalya/docs/design/alter-table-export-part-partition.md diff --git a/antalya/docs/design/alter-table-export-part-partition.md b/antalya/docs/design/alter-table-export-part-partition.md deleted file mode 100644 index 8ebe237c634a..000000000000 --- a/antalya/docs/design/alter-table-export-part-partition.md +++ /dev/null @@ -1,752 +0,0 @@ -Feature Design: `ALTER TABLE EXPORT PART` and `ALTER TABLE EXPORT PARTITION` -============================================================================ - -**Status:** draft -**Author(s):** Arthur Passos -**Related issues/PRs:** -https://github.com/Altinity/ClickHouse/pull/1618 - -**Last updated:** 2026-04-21 - ---- - -## 1. Requirements - -### Motivation - -Cost of storing data is a growing problem for large analytic systems -that use open source ClickHouse and replicated block storage. The core -problem is that block storage is (a) expensive and (b) replication makes -multiple copies. Project Antalya solves the storage cost problem using -the Hybrid Table Engine. Hybrid tables allow users to split tables -into segments, placing hot data on replicated block storage and -cold data on shared Iceberg tables using Parquet data files. - -The hybrid table approach requires a robust mechanism to export table -data from `MergeTree` tables into shared storage. The mechanism must -be fast, use machine resources efficiently, handle failures -automatically, and be easy to monitor. This design covers two new -ClickHouse commands to export data so users can populate hybrid -tables and move data to them at regular intervals. - -* `ALTER TABLE EXPORT PART` -- Exports a single part to a destination. - -* `ALTER TABLE EXPORT PARTITION` -- Exports one or more partitions to a destination. - -Both commands accept two destination families: - -1. **Plain object storage** (`S3`, `AzureBlobStorage`, equivalents). - Output is Parquet laid out in hive-partitioned directories. Atomicity - is provided by a sidecar *commit file* that enumerates the data files - written in the transaction; readers that want atomicity filter by - commit. An external catalog (Glue, REST, Nessie, Lakekeeper, ...) can - register this layout as an Iceberg table afterward, but the export - commands themselves do not interact with any catalog in this mode. - -2. **Apache Iceberg tables, with or without a catalog** (`Iceberg*` - engines, `iceberg*` table functions, `DatabaseIceberg`). Output is - Parquet data files plus per-file Avro *statistics sidecars*; on - commit the EXPORT process assembles a new Iceberg manifest, - writes a new `metadata.json`, and swaps the catalog pointer (or the - warehouse `metadata.json` pointer when catalog-less). Atomicity is - native — the snapshot either exists or it does not. - -These commands replace `INSERT INTO ... SELECT FROM` pipelines that -select rows and write them out to one or more Parquet files. That -approach uses resources for sorting, does not coordinate across replicas, -and does not take advantage of existing partitioning and sorting in -`MergeTree`. - -### Requirements - -1. **SQL only.** All operations related to export are available in SQL. - There should be no need to use non-SQL tools or directly access - storage to run exports or clean up problems. - -2. **Efficient, order-preserving writes.** Write a specified `MergeTree` - part (or every part of a specified partition) to an object-storage - destination in Parquet, preserving the source part's sort order, - without using a `SELECT ORDER BY` pass. Exporting a part should use the - same or less RAM than doing an `INSERT...SELECT...ORDER BY` on the same - data. (For example, it's not uncommon for the latter command to run - out of memory on large parts when the part ordering is added to the - SELECT ORDER BY.) - -3. **Output file management.** Allow users to break exported parts - into smaller Parquet files, which helps ensure good performance - when scanning Iceberg data. - -4. **Data type equivalence.** Map ClickHouse types to Iceberg types - that cast back without data loss to the original ClickHouse types - when selecting data. Applications that access exported data through - a Hybrid table should be able to read the data back from Iceberg - without requiring changes. - -5. **Atomic transfer.** Readers should never see a partial export. - The mechanism depends on the destination family: - - **Plain object storage.** Each transaction writes a sidecar - commit file that lists every data file produced by the transaction. - Readers that want atomicity filter by commit; a crash before the - commit file lands leaves only orphaned data files. - - **Iceberg destinations.** Atomicity is provided by Iceberg's own - snapshot-commit protocol — the new snapshot either becomes the - current metadata pointer or it does not. A crash before the - pointer swap leaves only orphaned data files and sidecars. - -6. **Distributed operation.** - - `EXPORT PART` always runs locally on the ClickHouse host where - it is invoked. - - `EXPORT PARTITION` from non-replicated `MergeTree` tables runs - locally on the ClickHouse host where it is invoked. - - `EXPORT PARTITION` is cluster-coordinated on `Replicated*MergeTree` - tables: any replica that has a given part contributes to the - export; the task is persistent and resumes after restarts. - -7. **Observability.** - It must be possible for users to track the following from system tables: - - Export part request status. - - Export partition request status. - - Relevant profile events related to export. - -8. **Error recovery.** - - **Idempotence.** Re-issuing the same export to the same - destination is a no-op while the export is running. (There should - be a way to track 'recent' exports so that they are idempotent as - well.) - - **Clean-up.** If file or metadata clean-up is required before resubmitting a failed - export, it must be possible to do so using only SQL commands. - - **Automatic restart.** `EXPORT PARTITION` task is persistent and - resumes after restarts. - -9. **Killable.** It must be possible to terminate any `ALTER TABLE EXPORT` command. - The command should be idempotent and must throw a clear exception on failure - rather than hanging. - -### Open questions and future requirements - -The design should address the following topics in the near future. - -- EXPORT PARTITION for MergeTree tables. Must work without Keeper installation. -- Export history. Provide a system table to track the history of part exports. -- Flexible casting that addresses issues like the following. - - Handling potentially lossy casts like INSERT SELECT: int64 -> int32. - - Export to tables that are missing columns. - - How to map column names--by position or by name? (e.g, is id, name, age compatible with id, age, name)? - -### Out of scope requirements - -- Non-Parquet output file formats. Only `Parquet` is targeted in this iteration. Later - iterations may add new output file formats. -- Exporting to arbitrary table functions. Only those backed by an object-storage engine that - supports exports (e.g. `s3`, `azure`) are valid; others throw `NOT_IMPLEMENTED`. -- Non-matching Iceberg schema, sorting or partitioning. Not supported. The source - `MergeTree` schema and partition keys must be compatible with the destination - Iceberg table's current `schema-id` and `partition-spec-id`. Destination partition values - are derived directly from the source part's partition key; we do not recompute them - from row data. -- Any read/query path over exported files — consumption happens via normal `S3` / `s3` / - external-engine reads. -- Synchronous exports. Not supported. EXPORT commands return immediately to client after - starting the export task; completion is polled via system tables. -- Importing parts back from object storage (that is tracked separately). - -### Constraints - -- Experimental gate: `allow_experimental_export_merge_tree_part` (query-level) for `EXPORT PART`; - `allow_experimental_export_merge_tree_partition_feature` (server-level) for `EXPORT PARTITION`. -- For best results `EXPORT PARTITION` requires a ZooKeeper / `clickhouse-keeper` ensemble with - the `multi_read` feature flag. This reduces API calls and ensures transactional consistency - when reading multiple fields. - governed by the destination table's Iceberg partition spec instead. -- No change to `MergeTree` on-disk part format; only the Keeper schema under the table's - replication path is extended. The extension is tranparent to users. - -### References - -- `docs/en/antalya/part_export.md` -- `docs/en/antalya/partition_export.md` -- `tests/queries/0_stateless/03572_export_merge_tree_part_basic.sh` -- `tests/queries/0_stateless/03572_export_merge_tree_part_to_object_storage_simple.sql` -- `tests/queries/0_stateless/03572_export_merge_tree_part_limits_and_table_functions.sh` -- `tests/queries/0_stateless/03572_export_merge_tree_part_special_columns.sh` -- `tests/queries/0_stateless/03572_export_replicated_merge_tree_part_to_object_storage.sh` -- `tests/queries/0_stateless/03572_export_replicated_merge_tree_part_to_object_storage_simple.sql` -- `tests/queries/0_stateless/03604_export_merge_tree_partition.sh` -- `tests/queries/0_stateless/03608_export_merge_tree_part_filename_pattern.sh` -- `tests/integration/test_export_merge_tree_part_to_object_storage/test.py` -- `tests/integration/test_export_replicated_mt_partition_to_object_storage/test.py` - ---- - -## 2. Functional specification - -### User-facing behavior -A user points `ALTER TABLE` at a `MergeTree` source and a destination (table or table function). -The command returns immediately with no rows. The export runs in the background; progress lives -in `system.exports` and, for partition exports, `system.replicated_partition_exports`. Successful -exports append to `system.part_log` with `event_type = 'ExportPart'`. - -Output shape depends on the destination family: - -- **Plain object storage.** One Parquet data file per part (or per chunk when split by - size/rows) plus one commit file per transaction — `//_..parquet` - and `/commit_<...>`. Readers that want atomicity filter by commit. This is the default path, - but it can be customized to handle sharding, which is not covered by the default case. See the - `` - -- **Iceberg destination.** One Parquet data file per part (or per chunk) plus a sibling - Avro statistics sidecar `_clickhouse_export_part_sidecar.avro` carrying - `record_count`, `file_size_in_bytes`, `column_sizes`, `null_value_counts`, `lower_bounds`, - and `upper_bounds`. The per-part task does not modify Iceberg metadata. On final commit - the initiating replica reads every sidecar, assembles a new manifest and manifest list, - writes a new `metadata.json`, and atomically swaps the pointer (via the catalog when one - is configured; otherwise via the warehouse `metadata.json` pointer). The manifest summary - contains `clickhouse.export-partition-transaction-id`, checked before every commit attempt - to prevent a double-commit after a post-commit / pre-status-update crash. Sidecar files - are not referenced from any Iceberg manifest and can be deleted safely after the commit - lands; ClickHouse does not reap them. - -### SQL syntax / API - -```sql --- Export a single part to a destination table -ALTER TABLE [db.]table - EXPORT PART 'part_name' - TO TABLE [dest_db.]dest_table - [SETTINGS ...]; - --- Export a single part to a destination table function -ALTER TABLE [db.]table - EXPORT PART 'part_name' - TO TABLE FUNCTION s3(...) PARTITION BY - [SETTINGS ...]; - --- Export every active part of a partition (Replicated*MergeTree only) -ALTER TABLE [db.]table - EXPORT PARTITION ID 'partition_id' - TO TABLE [dest_db.]dest_table - [SETTINGS ...]; - --- Export every active part of all partitions (Replicated*MergeTree only) -ALTER TABLE [db.]table - EXPORT PARTITION ALL - TO TABLE [dest_db.]dest_table - [SETTINGS ...]; - --- Cancel one or more partition exports -KILL EXPORT PARTITION WHERE ; -``` - -### Individual command examples -These are derived from `tests/queries/0_stateless/03572_*` and `03604_export_merge_tree_partition.sh`. - -```sql --- Part export to S3 table -ALTER TABLE mt_table EXPORT PART '2020_1_1_0' TO TABLE s3_table -SETTINGS allow_experimental_export_merge_tree_part = 1; - --- Part export to S3 table function (schema inferred from source) -ALTER TABLE mt_table EXPORT PART '2020_1_1_0' -TO TABLE FUNCTION s3(s3_conn, filename='tf', format='Parquet', partition_strategy='hive') -PARTITION BY year -SETTINGS allow_experimental_export_merge_tree_part = 1; - --- Split large part across multiple Parquet files -ALTER TABLE big EXPORT PART '2025_0_32_3' TO TABLE big_dest -SETTINGS allow_experimental_export_merge_tree_part = 1, - export_merge_tree_part_max_bytes_per_file = 10000000, - output_format_parquet_row_group_size_bytes = 5000000; --- (See note on settings below. Iceberg table engine now has built-in --- settings for Parquet files.) - --- Partition export across a Replicated cluster. This currently --- selects the parts on the replica that receives the plan. This --- means the result may vary if new parts are arriving on other --- replicas. -ALTER TABLE rmt_table EXPORT PARTITION ID '2020' TO TABLE s3_table; - --- Cancel by filter. The WHERE uses the same filter used to read from `system.replicated_partition_exports`. -KILL EXPORT PARTITION -WHERE partition_id = '2020' - AND source_table = 'rmt_table' - AND destination_table = 's3_table'; -``` - -### End-to-end examples - -Two parallel walkthroughs illustrate each destination family. Both -begin from the same `ReplicatedMergeTree` source. - -```sql --- Source table (shared by both examples). -CREATE TABLE events -( - id UInt64, - ts DateTime, - year UInt16 -) -ENGINE = ReplicatedMergeTree('/clickhouse/tables/{database}/events', 'r1') -PARTITION BY year -ORDER BY (year, id); - --- Seed two partitions (2024, 2025) straight from `system.numbers`. -INSERT INTO events -SELECT - number AS id, - toDateTime('2024-01-01 00:00:00') + INTERVAL number SECOND AS ts, - 2024 AS year -FROM system.numbers -LIMIT 1000000; - -INSERT INTO events -SELECT - number AS id, - toDateTime('2025-01-01 00:00:00') + INTERVAL number SECOND AS ts, - 2025 AS year -FROM system.numbers -LIMIT 1000000; -``` - -#### Plain object storage (hive layout) - -Export to a hive-partitioned S3 destination. The on-disk shape is what -an external Iceberg catalog (Glue, REST, Nessie, Lakekeeper, ...) would -register as an Iceberg table; the `EXPORT PARTITION` command itself -does not touch any catalog in this mode. - -```sql --- 1. Destination: S3 with hive partition layout. -CREATE TABLE events_s3 -( - id UInt64, - ts DateTime, - year UInt16 -) -ENGINE = S3(s3_conn, filename='warehouse/events', format = Parquet, partition_strategy = 'hive') -PARTITION BY year; - --- 2. Export the 2024 partition. Returns immediately; runs in the background. -ALTER TABLE events EXPORT PARTITION ID '2024' TO TABLE events_s3; - --- 3. Watch progress (Keeper round-trip — use sparingly). -SELECT status, parts_count, parts_to_do, last_exception -FROM system.replicated_partition_exports -WHERE source_table = 'events' AND partition_id = '2024'; - --- 4. When status = 'COMPLETED', the destination bucket contains: --- warehouse/events/year=2024/_.1.parquet (one per part) --- warehouse/events/commit_2024_ (atomicity manifest) --- Readers that filter by commit see either the full partition or nothing. -SELECT count() FROM events_s3 WHERE year = 2024; - --- 5. Or inspect the layout directly. -SELECT _path -FROM s3(s3_conn, filename = 'warehouse/events/year=2024/**', format = 'One') -ORDER BY _path; -``` - -#### Iceberg destination - -Export directly to an Iceberg table. Unlike the plain-object-storage -flow, `EXPORT PARTITION` writes native Iceberg metadata on commit, so -the result is queryable as an Iceberg table immediately — no external -registration step. The example uses `IcebergS3` without a catalog; -swap in `DatabaseIceberg` or an `iceberg(...)` catalog-backed table -function to route through REST / Glue / Unity. - -**Open Issue** We need to specify how to commit via a catalog. - -```sql --- 1. Destination: an Iceberg table backed by S3. No catalog required --- for this form; the warehouse metadata.json pointer is managed --- by ClickHouse directly. -CREATE TABLE events_iceberg -( - id UInt64, - ts DateTime, - year UInt16 -) -ENGINE = IcebergS3(s3_conn, filename='warehouse/events_iceberg/') -PARTITION BY year; - --- 2. Export the 2024 partition. Returns immediately; runs in the background. -ALTER TABLE events EXPORT PARTITION ID '2024' TO TABLE events_iceberg; - --- 3. Watch progress. You can use the system.exports table for this. -SELECT status, parts_count, parts_to_do, last_exception -FROM system.replicated_partition_exports -WHERE source_table = 'events' AND partition_id = '2024'; - --- 4. When status = 'COMPLETED', the destination contains a fully-formed --- Iceberg table. Readers see the snapshot atomically — either the --- full partition or nothing. -SELECT count() FROM events_iceberg WHERE year = 2024; - --- 5. Object layout: each data file has a sidecar carrying per-file stats --- used at commit time. These are ClickHouse-private and unreferenced --- from any Iceberg manifest; safe to delete after COMPLETED. -SELECT _path -FROM s3(s3_conn, filename = 'warehouse/events_iceberg/data/year=2024/**', format = 'One') -ORDER BY _path; --- warehouse/events_iceberg/data/year=2024/_.1.parquet --- warehouse/events_iceberg/data/year=2024/_.1.parquet_clickhouse_export_part_sidecar.avro --- ... --- warehouse/events_iceberg/metadata/v.metadata.json --- warehouse/events_iceberg/metadata/snap--*.avro --- warehouse/events_iceberg/metadata/.avro - --- 6. The commit carries an idempotency marker in the manifest summary --- (clickhouse.export-partition-transaction-id). If the initiator --- crashes post-commit / pre-status-update, a retry sees its own --- transaction id in the target's latest manifest and skips instead --- of double-committing. -``` - -The EXPORT PARTITION flow initially only works on ReplicatedMergeTree tables and -requires Keeper. Future iterations will support MergeTree tables as a source. - -### Operational notes - -The following notes expand on expected behavior of commands. - -1. When writing to object storage using `partition_strategy = 'wildcard'`, either wildcard - or 'hive' arguments are permitted. (This setting has no impact on Iceberg, - which records file partitions using metadata.) - -2. By default `ALTER TABLE t EXPORT PART 'p' TO TABLE s3_t` writes - `//_.1.parquet` plus - `/commit__`. You can read this using - `SELECT * FROM s3(...)`. - -3. `ALTER TABLE rmt EXPORT PARTITION ID 'p' TO TABLE s3_t` exports - every active part of partition `p` across all replicas that host - it; `system.replicated_partition_exports` converges to `COMPLETED`. - -4. Re-issuing the same `EXPORT PARTITION` is rejected (no duplicate - files) unless `export_merge_tree_partition_force_export = 1`. Export - entries are kept indefinitely in `system.replicated_partition_exports` - as history; they never expire. This behavior avoids accidentally - exporting the same data twice. Note, however that forcing the - operation is dangerous if ClickHouse can't clean up the previous - operation. In this case you'll potentially commit files twice. - -5. Killing an in-flight partition export via `KILL EXPORT PARTITION` - transitions status to `KILLED` and stops all replicas' contributions. - -6. Exception during part export is counted in `PartsExportFailures`; - retry behavior honors `export_merge_tree_partition_max_retries`. The - same budget also bounds per-task commit retries for Iceberg - destinations; the task fails terminally if commit retries alone - exceed the budget. - -7. A partition export that remains in `PENDING` longer than - `export_merge_tree_partition_task_timeout_seconds` (default 3600s; - `0` disables) is auto-killed by the background cleanup loop and - transitions to `KILLED` with a timeout reason recorded in - `last_exception`. Enforcement is best-effort: actual kill latency is - bounded by one manifest-updater poll cycle (~30s) plus ZooKeeper - watch propagation. This is primarily a backstop against tasks stuck - on missing parts, a missing destination table, or an infinite - commit-retry loop against Iceberg. - -8. **Third-party Iceberg catalog manifest cleanup.** If the catalog - reaps old manifest files, its retention window MUST exceed - `export_merge_tree_partition_task_timeout_seconds`. Otherwise the - rare "sole node commits to Iceberg, crashes before the `COMPLETED` - status reaches Keeper, boots back up after the reaper has deleted - its commit manifest" scenario can produce duplicate data when the - recovered node retries the commit. ClickHouse does not reap its own - export artifacts; the Iceberg sidecars (`*_clickhouse_export_part_sidecar.avro`) - are safe to delete once the commit has landed. - -9. **Settings in `PART` vs `PARTIION` export.** There is a subtle difference in - how settings are handled between PART and PARTITION export. - - For part export, the settings used at the moment of the query are preserved - and re-stored in the background export task. - - For partition export, it is slightly harder to preserve the settings because - we need to serialize them in ZooKeeper. This is done for a few very important - settings including export_merge_tree_part_max_bytes_per_file. This will be - cleaned up in future iterations. - -### Settings - -| Setting | Scope | Default | Range / Values | Applies to | Description | -| --- | --- | --- | --- | --- | --- | -| `allow_experimental_export_merge_tree_part` | query | `false` | `Bool` | `EXPORT PART` | Experimental gate; required. | -| `allow_experimental_export_merge_tree_partition_feature` | server | `false` | `Bool` | `EXPORT PARTITION` | Experimental gate; required. | -| `export_merge_tree_part_overwrite_file_if_exists` | query | `false` | `Bool` | `EXPORT PART` | Overwrite existing destination file; otherwise throws. | -| `export_merge_tree_part_max_bytes_per_file` | query | `0` | `UInt64` (`0`=unlimited) | both | Soft cap per output file. Non-zero values can break idempotency. (SEE NOTE 1 below.)| -| `export_merge_tree_part_max_rows_per_file` | query | `0` | `UInt64` (`0`=unlimited) | both | Soft cap per output file. Non-zero values can break idempotency. | -| `export_merge_tree_part_throw_on_pending_mutations` | query | `true` | `Bool` | both | Refuse to export parts with pending mutations (unless mutation was `IN PARTITION`). | -| `export_merge_tree_part_throw_on_pending_patch_parts` | query | `true` | `Bool` | both | Refuse to export parts with pending patch parts. | -| `export_merge_tree_part_filename_pattern` | query | `{part_name}_{checksum}` | `String` | both | Filename template; supports `{part_name}`, `{checksum}`, `{database}`, `{table}`, server macros. | -| `export_merge_tree_partition_force_export` | query | `false` | `Bool` | `EXPORT PARTITION` | Overwrite a live Keeper manifest for the same `(source, destination, partition_id)`. Dangerous — can produce duplicate data on the destination; use with caution. | -| `export_merge_tree_partition_max_retries` | query | `3` | `UInt64` | `EXPORT PARTITION` | Retry budget applied to both per-part export attempts and per-task commit attempts (Iceberg). The task fails terminally if commit retries alone exceed the budget. | -| `export_merge_tree_partition_task_timeout_seconds` | query | `3600` (seconds) | `UInt64` (`0`=disable) | `EXPORT PARTITION` | Wall-clock cap for `PENDING` tasks; on expiry transitions to `KILLED` with a timeout reason. Measured from manifest `create_time`. Enforcement latency ≈ one manifest-updater poll cycle (~30s) plus Keeper watch propagation. | -| `export_merge_tree_partition_system_table_prefer_remote_information` | query | `false` | `Bool` | `EXPORT PARTITION` | When `true`, `system.replicated_partition_exports` fetches fresh state from Keeper (requires the `MULTI_READ` feature flag); when `false`, uses local cached state. **Default flipped from `true` to `false` in this release** — Keeper round-trips were more expensive than warranted for the typical observability workload. (See NOTE 2.)| -| `export_merge_tree_part_file_already_exists_policy` | query | `skip` | `skip` / `error` / `overwrite` | `EXPORT PARTITION` | Per-file policy during partition export. | - -Default-value impact: all new settings default to "off" or to -conservative values (pending-mutation guards default to throwing). One -default *changed* in this release: -`export_merge_tree_partition_system_table_prefer_remote_information` -flipped from `true` to `false`, so `system.replicated_partition_exports` -now serves local cached state by default instead of querying Keeper. -Users who relied on always-fresh results must set it back to `true` -explicitly. - -**NOTE 1:** `export_merge_tree_part_max_bytes_per_file` overrides `iceberg_insert_max_bytes_in_data_file` or -other more specific file size parameters. - -EXPORT should also observe the following existing settings for export to Iceberg / Parquet: -- `iceberg_insert_max_bytes_in_data_file` (as above, overridden by `export_merge_tree_part_max_bytes_per_file` if specified) -- `iceberg_insert_max_rows_in_data_file` -- `output_format_parquet_row_group_size` -- `output_format_parquet_row_group_size_bytes` -- `output_format_parquet_data_page_size` -- `output_format_parquet_compression_method` -- `output_format_parquet_version` - -**NOTE 2:** `export_merge_tree_partition_system_table_prefer_remote_information` may be dropped. -Querying Keeper from the user path is complex and has side effects. - -### System tables / metrics / log messages / observability - -- `system.exports` — rows for currently-executing part exports (source/destination tables, - `part_name`, destination paths, `elapsed`, rows/bytes counters, memory counters). Dropped when - the export completes. -- `system.replicated_partition_exports` — rows for `EXPORT PARTITION` tasks. Backed by Keeper; - querying it is a Keeper round-trip and should be used sparingly. Columns include - `source_database`, `source_table`, `destination_database`, `destination_table`, `create_time`, - `partition_id`, `transaction_id`, `source_replica`, `parts`, `parts_count`, `parts_to_do`, - `status`, `exception_replica`, `last_exception`, `exception_part`, `exception_count`. -- `system.part_log` — each completed part export appends one row with `event_type = 'ExportPart'`, - filled `remote_file_paths`, `merged_from = [part_name]`, plus standard timing / size / error - fields. -- `ProfileEvents`: - - `PartsExports` — successful part-export completions. - - `PartsExportFailures` — failures. - - `PartsExportDuplicated` — skipped because destination file already existed. - - `PartsExportTotalMilliseconds` — cumulative wall time. -- Status enum on `system.replicated_partition_exports`: `PENDING`, `COMPLETED`, `FAILED`, `KILLED`. - -### Error behavior - -- Missing experimental flag: `SUPPORT_IS_DISABLED` (exact code TBD — confirm against - `src/Common/ErrorCodes.cpp`). -- Destination schema mismatch (columns / types / order; source `EPHEMERAL` column present in - destination): `INCOMPATIBLE_COLUMNS`. -- Destination engine doesn't support exports (e.g. `url`, unknown engine): `NOT_IMPLEMENTED`. -- Destination is an unknown table function: `UNKNOWN_FUNCTION`. -- Pending mutations or patch parts when the guard is enabled: `BAD_ARGUMENTS` (TBD — confirm). -- Part not found on any replica: `NO_SUCH_DATA_PART` (TBD — confirm). -- Destination file already exists and policy is `error` / `export_merge_tree_part_overwrite_file_if_exists = 0`: - `FILE_ALREADY_EXISTS` (TBD — confirm). -- Duplicate live manifest without `..._force_export`: `DUPLICATE_EXPORT_TASK` or equivalent - (TBD — confirm). -- Commit-retry budget exhausted (Iceberg destination): terminal `FAILED` with - `last_exception` populated; the task is not retried further. -- Task timeout exceeded (`export_merge_tree_partition_task_timeout_seconds`): - terminal `KILLED` with a timeout reason in `last_exception`. - -All of the above **throw exceptions** or surface through `last_exception` on the -task row; they do not crash the server. - -### Backward compatibility -- **Older client → newer server:** harmless — the client issues the new `ALTER` text; the server - parses it. No wire-protocol change. -- **Newer client → older server:** older server fails parse on `EXPORT PART` / `EXPORT PARTITION` - with `SYNTAX_ERROR`. Acceptable. -- **Mixed-version cluster (replication):** `EXPORT PART` is local to one replica; no cross-replica - effect. `EXPORT PARTITION` stores its manifest under the table's Keeper path in a new - subtree; replicas on older versions ignore unknown nodes but will NOT contribute parts to the - export — the initiating replica (which must be on the new version) completes alone if it holds - all the parts; otherwise the task stalls on `parts_to_do > 0`. An upgrade-ordering note for - operators is required (section 4 / Rollout). -- **On-disk format:** unchanged. Parts are read as-is; Parquet is produced on the fly. -- **Default-value changes:** `export_merge_tree_partition_system_table_prefer_remote_information` - flipped from `true` to `false`. Users running dashboards that read - `system.replicated_partition_exports` and require always-fresh state must set this - back to `true` explicitly (and ensure the Keeper `MULTI_READ` feature flag is enabled). - No other defaults that affect existing workloads (see settings table). - ---- - -## 3. Implementation - -This design only covers user visible behavior. It does not internal implementatation -details. The implementation section is omitted. - -## 4. Test plan - -### Functional tests — `tests/queries/0_stateless` - -Existing coverage to retain: - -- `03572_export_merge_tree_part_basic.sh` — golden path, idempotent re-export, wildcard + - hive partition strategies. -- `03572_export_merge_tree_part_to_object_storage_simple.sql` — error cases - (`INCOMPATIBLE_COLUMNS`, `NOT_IMPLEMENTED`, `UNKNOWN_FUNCTION`, `EPHEMERAL` collision). -- `03572_export_merge_tree_part_limits_and_table_functions.sh` — `max_bytes_per_file`, - `max_rows_per_file`, table-function destination with schema inheritance / explicit structure. -- `03572_export_merge_tree_part_special_columns.sh` — `ALIAS`, `MATERIALIZED`, `EPHEMERAL`, - mixed / complex expressions. -- `03572_export_replicated_merge_tree_part_to_object_storage.sh` + - `03572_export_replicated_merge_tree_part_to_object_storage_simple.sql` — part-level export - from `ReplicatedMergeTree`. -- `03604_export_merge_tree_partition.sh` — basic `EXPORT PARTITION ID`. -- `03608_export_merge_tree_part_filename_pattern.sh` — default and custom - `export_merge_tree_part_filename_pattern` including `{database}` / `{table}` macros. - -New tests to add: - -- `NNNN_export_merge_tree_part_pending_mutations.sh` — with/without the - `..._throw_on_pending_mutations` / `..._throw_on_pending_patch_parts` guards and `IN PARTITION` - mutations. -- `NNNN_export_merge_tree_part_commit_file.sh` — verify a `commit__` file exists - alongside every successful export and references every written data file; verify that a - partial run (simulated by killing before commit) does not produce a commit file. -- `NNNN_export_merge_tree_part_overwrite_policy.sh` — all three values of - `export_merge_tree_part_file_already_exists_policy` (`skip`, `error`, `overwrite`) plus - `export_merge_tree_part_overwrite_file_if_exists`. -- `NNNN_export_merge_tree_part_profile_events.sh` — assert `PartsExports`, - `PartsExportFailures`, `PartsExportDuplicated`, `PartsExportTotalMilliseconds` move as - expected. - -Note: Iceberg-destination coverage is deliberately in integration (next section), not in -`tests/queries/0_stateless`, because it requires a live warehouse and — for the catalog -path — a REST / Glue fixture. The commit-file test above only exercises plain -object-storage atomicity. - -Do not add `no-parallel` to any new test unless explicitly required by shared S3 bucket paths; -`03604` currently has the tag and should be re-examined to see whether unique per-run paths -remove the need. - -### Integration tests — `tests/integration` - -**Keep (modified in this PR):** - -- `test_export_merge_tree_part_to_object_storage/` — part export in a multi-node setup. - PR 1618 makes minor adjustments. -- `test_export_replicated_mt_partition_to_object_storage/` — partition export across - replicas, including `wait_for_export_status`, retry counting, and replica failure - scenarios. PR 1618 removes the `s3_retries.xml` config and reshapes several test cases - against the new shared helpers. - -**New in PR 1618:** - -- `test_export_merge_tree_part_to_iceberg/` — per-part export to an Iceberg destination, - covering golden path, sidecar emission, manifest shape, and error paths. -- `test_export_replicated_mt_partition_to_iceberg/` — distributed partition export to - Iceberg across replicas, including `test_export_task_timeout_kills_stuck_pending_task` - (uses the `export_partition_commit_always_throw` failpoint to exhaust the commit path, - then asserts the timeout transitions the task to `KILLED`). -- `test_storage_iceberg_with_spark/test_export_partition_iceberg.py` — catalog-less - Iceberg round-trip; Spark reads ClickHouse-written data and verifies schema, partition - layout, and snapshot atomicity. -- `test_storage_iceberg_with_spark/test_export_partition_iceberg_catalog.py` — - catalog-backed (REST) round-trip; exercises the `commitExportPartitionTransaction` - path against a real catalog. -- Shared helpers: - - `tests/integration/helpers/export_partition_helpers.py` — shared - `wait_for_export_status`, manifest-inspection utilities. - - `tests/integration/helpers/iceberg_export_stats.py` — sidecar decoders and stats - assertion helpers. - -**Remaining gaps to add:** - -- Initiating-replica dies mid-commit (post-data-file-write, pre-catalog-CAS) — asserts a - surviving replica completes via the Keeper-stashed `metadata.json` snapshot, and the - `clickhouse.export-partition-transaction-id` idempotency check prevents double-commit. -- Experimental feature disabled on one replica - (`disable_experimental_export_partition.xml` config) and enabled on the rest — task - still completes via the enabled replicas. -- `KILL EXPORT PARTITION` against an Iceberg task in mid-commit: status transitions to - `KILLED`, no dangling half-written `metadata.json`, data files and sidecars remain as - orphans (cleanup is user responsibility per Operational note 7 in § 2). -- Mixed-version cluster: upgrade scenario where only some replicas know about - `EXPORT PARTITION` or the new Keeper manifest fields. - -Invocation: -`python -m ci.praktika run "integration" --test test_export_merge_tree_part_to_object_storage,test_export_replicated_mt_partition_to_object_storage,test_export_merge_tree_part_to_iceberg,test_export_replicated_mt_partition_to_iceberg,test_storage_iceberg_with_spark`. - -### Failpoints - -The following failpoints are registered for deterministic testing of crash windows and -retry logic. Enable via `SYSTEM ENABLE FAILPOINT ` from test harnesses. - -| Failpoint | Kind | What it tests | -| --- | --- | --- | -| `iceberg_writes_non_retry_cleanup` | ONCE | Cleanup path when an Iceberg write fails in a non-retryable way. | -| `iceberg_writes_post_publish_throw` | ONCE | Commit succeeded in object storage but the publish step throws — exercises the recovery path that must not double-commit. | -| `iceberg_export_after_commit_before_zk_completed` | ONCE | Crash window between a successful Iceberg commit and the `COMPLETED` Keeper status update — the idempotency marker (`clickhouse.export-partition-transaction-id`) must prevent a second commit on recovery. | -| `export_partition_commit_always_throw` | REGULAR | Every commit attempt throws — used to exhaust `max_retries` and drive the task-timeout path. | -| `export_partition_status_change_throw` | ONCE | Throws during a manifest status transition — exercises the status-drain lock invariant (decision #11 in § 3) and the manifest-updating task's retry logic. | - -These failpoints replace the "simulate a crash before commit" phrasing in the -functional-tests section above; prefer them over process kills for deterministic CI -behavior. - -### Performance tests — `tests/performance` - -Add `export_merge_tree_part.xml`: compare `ALTER TABLE ... EXPORT PART` vs. -`INSERT INTO s3_t SELECT * FROM mt WHERE _part = ...` on a ~1 GB Wide part; track wall time and -peak memory. Hot path is the Parquet encoder, which warrants a guard against regressions. - -- Iceberg-destination benchmark: the Iceberg commit is O(data files), not per-row, so the - per-part write path performance should match plain object storage within noise. A - secondary benchmark comparing `EXPORT PARTITION` to Iceberg vs. to hive-layout S3 over - a ≥1000-part partition would catch regressions in the sidecar / manifest-assembly code - path specifically. - -### Manual verification - -- Roundtrip: export partition → read via `SELECT * FROM s3(...)` → create new - `ReplicatedMergeTree` from the S3 data → row counts / checksums match the source. -- `system.replicated_partition_exports` behaviour under a crashed initiator (cluster restart). -- Object-storage layout inspection via `s3(..., format=One)` listing: exactly N data files + 1 - commit file per transaction. -- Iceberg roundtrip with an external reader: export a partition to an Iceberg destination - → read it back through the same catalog → row counts and column checksums match the - source. DuckDB is the preferred external reader here (lightweight, fast to stand up, - mature Iceberg support); Spark or Trino may be substituted where a specific catalog - integration needs to be exercised. Confirms the on-disk metadata we write is actually - interoperable, not just self-consistent. - -### Rollout / risk - -- **Risk (Keeper schema extension):** the `partition_exports` subtree is write-once; a - partially-rolled-out cluster where only some replicas understand the subtree — or the - new manifest fields (`task_timeout_seconds`, `commit_attempts`, Iceberg `metadata.json` - snapshot, `write_full_path_in_iceberg_metadata`) — will stall partition exports - (`parts_to_do > 0`) rather than corrupt data. Acceptable but must be documented in the - upgrade notes. -- **Risk (object-storage cost / accidental large exports):** mitigated by the experimental - gates (default off) and the duplicate-export rejection (an existing export key is refused - unless `export_merge_tree_partition_force_export` is set). -- **Risk (Iceberg catalog manifest retention):** if the catalog reaps old manifest files - with a retention window shorter than `export_merge_tree_partition_task_timeout_seconds`, - the rare "sole-node commits, crashes, recovers after reaper deleted the commit manifest" - scenario can duplicate data. Operators MUST verify their catalog's retention before - enabling Iceberg destinations (see Operational note 7 in § 2). -- **Risk (default flip on `_prefer_remote_information`):** dashboards that read - `system.replicated_partition_exports` now see local cached state by default instead of - Keeper-fresh state. Existing users must set the flag back to `true` explicitly if they - rely on always-fresh results (and ensure `MULTI_READ` is enabled). -- **Flag strategy:** ship with `allow_experimental_export_merge_tree_part` (query, default - `false`) and `allow_experimental_export_merge_tree_partition_feature` (server, default - `false`). Both destination families ride on these gates — no separate Iceberg gate. - Flip defaults to `true` only after: (a) the open questions in § 3 are resolved, (b) the - remaining integration gaps listed above are closed, (c) one release cycle of customer - feedback. -- **Watch in production:** - - `PartsExportFailures` (existing). - - `exception_count` on `system.replicated_partition_exports` (existing). - - Keeper watch counts under the `partition_exports` subtree (existing). - - Object-storage request-error rates (existing). - - `commit_attempts` values approaching `max_retries` — signal of a catalog or network - issue throttling commits. - - Rate of `KILLED` transitions with timeout reason — signal of tasks stuck on missing - parts, missing destination, or commit backpressure. - - For Iceberg destinations: `metadata/` prefix write-error rate (CAS contention, - vended-credential expiry) and catalog-API error rate. diff --git a/src/Core/Settings.cpp b/src/Core/Settings.cpp index 6a20d23a8a6c..2123c3843b29 100644 --- a/src/Core/Settings.cpp +++ b/src/Core/Settings.cpp @@ -8042,16 +8042,8 @@ Ignore existing partition export and overwrite the zookeeper entry )", 0) \ DECLARE(UInt64, export_merge_tree_partition_max_retries, 3, R"( Maximum number of retries for exporting a merge tree part in an export partition task -)", 0) \ -<<<<<<< HEAD - DECLARE(UInt64, export_merge_tree_partition_manifest_ttl, 86400, R"( -Determines how long the manifest will live in ZooKeeper. It prevents the same partition from being exported twice to the same destination. -This setting does not affect / delete in progress tasks. It'll only cleanup the completed ones. )", 0) \ DECLARE(UInt64, export_merge_tree_partition_task_timeout_seconds, 3600, R"( -======= - DECLARE(UInt64, export_merge_tree_partition_task_timeout_seconds, 86400, R"( ->>>>>>> 1e5a11cb4eb (Merge pull request #1917 from Altinity/do_not_evict_entries_from_replicated_partition_exports_table) Maximum wall-clock duration (in seconds) an export partition task is allowed to remain in the PENDING state before it is auto-killed by the background cleanup loop. The timeout is measured from the manifest's create_time. Set to 0 to disable the timeout. When the timeout is exceeded the task transitions to KILLED (same terminal state as `KILL QUERY ... EXPORT PARTITION`), and `last_exception` is populated with a timeout reason. @@ -8494,11 +8486,8 @@ Maximum number of texts to include in a single HTTP request made by `aiEmbed`. T #define OBSOLETE_SETTINGS(M, ALIAS) \ /** Obsolete settings which are kept around for compatibility reasons. They have no effect anymore. */ \ -<<<<<<< HEAD - MAKE_OBSOLETE(M, Bool, allow_experimental_query_deduplication, false) \ -======= MAKE_OBSOLETE(M, UInt64, export_merge_tree_partition_manifest_ttl, 86400) \ ->>>>>>> 1e5a11cb4eb (Merge pull request #1917 from Altinity/do_not_evict_entries_from_replicated_partition_exports_table) + MAKE_OBSOLETE(M, Bool, allow_experimental_query_deduplication, false) \ MAKE_OBSOLETE(M, Bool, query_condition_cache_store_conditions_as_plaintext, false) \ MAKE_OBSOLETE(M, Bool, update_insert_deduplication_token_in_dependent_materialized_views, 0) \ MAKE_OBSOLETE(M, UInt64, max_memory_usage_for_all_queries, 0) \ From 506a733d71013bf433d85c94c1027d70bf37cc95 Mon Sep 17 00:00:00 2001 From: Andrey Zvonov <32552679+zvonand@users.noreply.github.com> Date: Mon, 22 Jun 2026 15:36:03 +0200 Subject: [PATCH 29/43] Merge pull request #1779 from Altinity/feature/antalya-26.3/export-cast-compatibility Auto cast values on export part and partition just like insert select Source-PR: #1779 (https://github.com/Altinity/ClickHouse/pull/1779) --- docs/en/antalya/part_export.md | 10 + docs/en/antalya/partition_export.md | 10 + src/Core/Settings.cpp | 5 + src/Core/SettingsChangesHistory.cpp | 3 + ...portReplicatedMergeTreePartitionManifest.h | 7 + src/Storages/IStorage.h | 10 +- src/Storages/MergeTree/ExportPartTask.cpp | 31 +- .../MergeTree/ExportPartitionUtils.cpp | 71 +- src/Storages/MergeTree/ExportPartitionUtils.h | 24 +- src/Storages/MergeTree/MergeTreeData.cpp | 20 +- .../DataLakes/IDataLakeMetadata.h | 2 +- .../DataLakes/Iceberg/IcebergMetadata.cpp | 58 +- .../DataLakes/Iceberg/IcebergMetadata.h | 2 +- .../ObjectStorage/StorageObjectStorage.cpp | 5 +- src/Storages/StorageReplicatedMergeTree.cpp | 18 +- .../test.py | 282 +++++++- .../test.py | 658 ++++++++++++++++++ .../test.py | 80 +++ ...rge_tree_part_to_object_storage_simple.sql | 41 +- ...rge_tree_part_to_object_storage_simple.sql | 37 +- 20 files changed, 1320 insertions(+), 54 deletions(-) diff --git a/docs/en/antalya/part_export.md b/docs/en/antalya/part_export.md index 03f9f479991b..73d467c5d9b1 100644 --- a/docs/en/antalya/part_export.md +++ b/docs/en/antalya/part_export.md @@ -107,6 +107,16 @@ In case a table function is used as the destination, the schema can be omitted a - **Default**: `{part_name}_{checksum}` - **Description**: Pattern for the filename of the exported merge tree part. The `part_name` and `checksum` are calculated and replaced on the fly. Additional macros are supported. +### `export_merge_tree_part_allow_lossy_cast` (Optional) + +- **Type**: `Bool` +- **Default**: `false` +- **Description**: Allow `EXPORT PART`/`EXPORT PARTITION` to apply lossy (non-value-preserving) casts when the source and destination column types differ. When disabled, an export that would require a lossy cast throws instead. + + When exporting to Apache Iceberg, the partition value written to the metadata is derived from the source partition columns by casting them to the destination partition-field types and applying the destination partition transform — the same computation the exported data files use. This keeps the Iceberg metadata consistent with the data files. + + **Warning:** A lossy cast on a partition column remains semantically truncating. For example, if a table is partitioned by an `Int64` column and some partition values do not fit into a destination `Int32` partition column, both the data files and the Iceberg metadata will contain the truncated `Int32` value (they agree with each other, but the original `Int64` value is lost). Such casts require `export_merge_tree_part_allow_lossy_cast = 1`. + ## Examples diff --git a/docs/en/antalya/partition_export.md b/docs/en/antalya/partition_export.md index ca465a8dea82..f4c45f612699 100644 --- a/docs/en/antalya/partition_export.md +++ b/docs/en/antalya/partition_export.md @@ -104,6 +104,16 @@ When the timeout is exceeded the task transitions to KILLED (same terminal state Notes: - Enforcement is best-effort: actual kill latency is bounded by one manifest-updater poll cycle (~30s) plus ZooKeeper watch propagation. +### `export_merge_tree_part_allow_lossy_cast` (Optional) + +- **Type**: `Bool` +- **Default**: `false` +- **Description**: Allow `EXPORT PART`/`EXPORT PARTITION` to apply lossy (non-value-preserving) casts when the source and destination column types differ. When disabled, an export that would require a lossy cast throws instead. + + When exporting to Apache Iceberg, the partition value written to the metadata is derived from the source partition columns by casting them to the destination partition-field types and applying the destination partition transform — the same computation the exported data files use. This keeps the Iceberg metadata consistent with the data files. + + **Warning:** A lossy cast on a partition column remains semantically truncating. For example, if a table is partitioned by an `Int64` column and some partition values do not fit into a destination `Int32` partition column, both the data files and the Iceberg metadata will contain the truncated `Int32` value (they agree with each other, but the original `Int64` value is lost). Such casts require `export_merge_tree_part_allow_lossy_cast = 1`. + ## Examples ### Basic Export to S3 diff --git a/src/Core/Settings.cpp b/src/Core/Settings.cpp index 2123c3843b29..7a8ca857a25a 100644 --- a/src/Core/Settings.cpp +++ b/src/Core/Settings.cpp @@ -8081,6 +8081,11 @@ Has no effect on `EXPORT PARTITION ` (single-partition export). )", 0) \ DECLARE(String, export_merge_tree_part_filename_pattern, "{part_name}_{checksum}", R"( Pattern for the filename of the exported merge tree part. The `part_name` and `checksum` are calculated and replaced on the fly. Additional macros are supported. +)", 0) \ + DECLARE(Bool, export_merge_tree_part_allow_lossy_cast, false, R"( +Allow `EXPORT PART`/`EXPORT PARTITION` to apply lossy (non-value-preserving) casts when the source and destination column types differ. When disabled, an export that would require a lossy cast throws instead. + +When exporting to Apache Iceberg, the partition value written to the metadata is derived from the source partition columns by casting them to the destination partition-field types and applying the destination partition transform — the same computation the exported data files use, so the metadata stays consistent with the data. A lossy cast on a partition column remains semantically truncating: both the data files and the metadata contain the truncated value, and such casts require this setting to be enabled. )", 0) \ \ /* ####################################################### */ \ diff --git a/src/Core/SettingsChangesHistory.cpp b/src/Core/SettingsChangesHistory.cpp index 061c5076a022..9cfda40eee66 100644 --- a/src/Core/SettingsChangesHistory.cpp +++ b/src/Core/SettingsChangesHistory.cpp @@ -86,6 +86,9 @@ const VersionToSettingsChangesMap & getSettingsChangesHistory() {"allow_experimental_query_deduplication", false, false, "The setting is obsolete, the feature has been removed."}, {"query_plan_min_columns_for_join_lazy_indexing", 0, 3, "Control the minimum number of payload columns from the left side required for enabling lazy indexing optimization in JOIN"}, {"query_plan_max_limit_for_join_lazy_indexing", 1000, 1000, "Added new setting to control maximum limit value that allows to use query plan for lazy join indexing optimization. If zero, there is no limit"}, + {"object_storage_cluster_join_mode", "allow", "allow", "New setting"}, + {"export_merge_tree_partition_task_timeout_seconds", "3600", "86400", "Increase default value to make it more realistic"}, + {"export_merge_tree_part_allow_lossy_cast", false, false, "New setting to gate lossy casts in EXPORT PART/PARTITION behind explicit acknowledgment"}, }); addSettingsChanges(settings_changes_history, "26.5", diff --git a/src/Storages/ExportReplicatedMergeTreePartitionManifest.h b/src/Storages/ExportReplicatedMergeTreePartitionManifest.h index bdebc63565d1..4e28e5e4f505 100644 --- a/src/Storages/ExportReplicatedMergeTreePartitionManifest.h +++ b/src/Storages/ExportReplicatedMergeTreePartitionManifest.h @@ -173,6 +173,7 @@ struct ExportReplicatedMergeTreePartitionManifest MergeTreePartExportManifest::FileAlreadyExistsPolicy file_already_exists_policy; String filename_pattern; bool write_full_path_in_iceberg_metadata = false; + bool allow_lossy_cast = false; String iceberg_metadata_json; std::string toJsonString() const @@ -206,6 +207,7 @@ struct ExportReplicatedMergeTreePartitionManifest json.set("max_retries", max_retries); json.set("task_timeout_seconds", task_timeout_seconds); json.set("write_full_path_in_iceberg_metadata", write_full_path_in_iceberg_metadata); + json.set("allow_lossy_cast", allow_lossy_cast); std::ostringstream oss; // STYLE_CHECK_ALLOW_STD_STRING_STREAM oss.exceptions(std::ios::failbit); Poco::JSON::Stringifier::stringify(json, oss); @@ -259,6 +261,11 @@ struct ExportReplicatedMergeTreePartitionManifest manifest.write_full_path_in_iceberg_metadata = json->getValue("write_full_path_in_iceberg_metadata"); + /// Default to true for tasks created before this field existed, so an in-flight + /// export scheduled with the old permissive worker behavior is not wrongly rejected + /// on upgrade. New tasks always persist the initiator's actual choice. + manifest.allow_lossy_cast = json->has("allow_lossy_cast") ? json->getValue("allow_lossy_cast") : true; + return manifest; } }; diff --git a/src/Storages/IStorage.h b/src/Storages/IStorage.h index 890cae49c236..815f38d68ad4 100644 --- a/src/Storages/IStorage.h +++ b/src/Storages/IStorage.h @@ -468,11 +468,11 @@ It is currently only implemented in StorageObjectStorage. struct IcebergCommitExportPartitionArguments { std::string metadata_json_string; - /// Partition column values (after transforms). Callers are responsible for - /// populating this: the partition-export path parses them from the persisted - /// JSON string, while the direct EXPORT PART path reads them from the part's - /// partition key. - std::vector partition_values; + /// Representative source partition-key columns from one exported part (the part's + /// minmax block). The destination derives the Iceberg partition tuple from a row of + /// this block by casting to the destination column types and applying the partition + /// transform, so the metadata partition value matches the exported data files. + Block partition_source_block; }; virtual void commitExportPartitionTransaction( diff --git a/src/Storages/MergeTree/ExportPartTask.cpp b/src/Storages/MergeTree/ExportPartTask.cpp index 9f1d4a773b17..25d6e7ffcadd 100644 --- a/src/Storages/MergeTree/ExportPartTask.cpp +++ b/src/Storages/MergeTree/ExportPartTask.cpp @@ -99,6 +99,31 @@ namespace } } + /// Mirrors `InterpreterInsertQuery::addInsertToSelectPipeline`: positional match, + /// destination header = `getSampleBlockNonMaterialized()`, all type bridging is done + /// by the CAST inside `makeConvertingActions`. No pre-validation, no per-column + /// lossy/non-lossy classification — restrictions are exactly what INSERT SELECT enforces. + void addExportConvertingActions( + QueryPlan & plan_for_part, + const IStorage & destination_storage, + const ContextPtr & local_context) + { + const auto destination_header + = destination_storage.getInMemoryMetadataPtr()->getSampleBlockNonMaterialized(); + + auto dag = ActionsDAG::makeConvertingActions( + plan_for_part.getCurrentHeader()->getColumnsWithTypeAndName(), + destination_header.getColumnsWithTypeAndName(), + ActionsDAG::MatchColumnsMode::Position, + local_context); + + auto expression_step = std::make_unique( + plan_for_part.getCurrentHeader(), + std::move(dag)); + expression_step->setStepDescription("Convert source columns to destination types for export"); + plan_for_part.addStep(std::move(expression_step)); + } + String buildDestinationFilename( const MergeTreePartExportManifest & manifest, const StorageID & storage_id, @@ -261,6 +286,10 @@ bool ExportPartTask::executeStep() /// This is a hack that materializes the columns before the export so they can be exported to tables that have matching columns materializeSpecialColumns(plan_for_part.getCurrentHeader(), metadata_snapshot, local_context, plan_for_part); + /// Align the pipeline header with the destination's non-materialized sample block, + /// using the same `makeConvertingActions(Position)` call INSERT SELECT performs. + addExportConvertingActions(plan_for_part, *destination_storage, local_context); + QueryPlanOptimizationSettings optimization_settings(local_context); auto pipeline_settings = BuildQueryPipelineSettings(local_context); auto builder = plan_for_part.buildQueryPipeline(optimization_settings, pipeline_settings); @@ -303,7 +332,7 @@ bool ExportPartTask::executeStep() { IStorage::IcebergCommitExportPartitionArguments iceberg_args; iceberg_args.metadata_json_string = manifest.iceberg_metadata_json; - iceberg_args.partition_values = manifest.data_part->partition.value; + iceberg_args.partition_source_block = block_with_partition_values; destination_storage->commitExportPartitionTransaction( manifest.transaction_id, diff --git a/src/Storages/MergeTree/ExportPartitionUtils.cpp b/src/Storages/MergeTree/ExportPartitionUtils.cpp index 4a447c8de3ab..679e07d0b132 100644 --- a/src/Storages/MergeTree/ExportPartitionUtils.cpp +++ b/src/Storages/MergeTree/ExportPartitionUtils.cpp @@ -9,7 +9,12 @@ #include #include #include +#include +#include +#include +#include #include +#include #if USE_AVRO #include @@ -37,6 +42,11 @@ namespace ErrorCodes extern const int NETWORK_ERROR; } +namespace Setting +{ + extern const SettingsBool export_merge_tree_part_allow_lossy_cast; +} + namespace FailPoints { extern const char iceberg_export_after_commit_before_zk_completed[]; @@ -47,14 +57,13 @@ namespace fs = std::filesystem; namespace ExportPartitionUtils { - std::vector getPartitionValuesForIcebergCommit( + Block getPartitionSourceBlockForIcebergCommit( MergeTreeData & storage, const String & partition_id) { auto lock = storage.readLockParts(); const auto parts = storage.getDataPartsVectorInPartitionForInternalUsage( MergeTreeDataPartState::Active, partition_id, lock); - - /// todo arthur: bad arguments for now, pick a better one + if (parts.empty()) throw Exception(ErrorCodes::NO_SUCH_DATA_PART, "Cannot find active part for partition_id '{}' to derive Iceberg partition " @@ -62,7 +71,7 @@ namespace ExportPartitionUtils "or this replica has not yet received any part for this partition. " "The commit will be retried.", partition_id); - return parts.front()->partition.value; + return parts.front()->minmax_idx->getBlock(storage); } ContextPtr getContextCopyWithTaskSettings(const ContextPtr & context, const ExportReplicatedMergeTreePartitionManifest & manifest) @@ -92,6 +101,12 @@ namespace ExportPartitionUtils /// stalls when the setting is only set at the query level. context_copy->setSetting("allow_insert_into_iceberg", true); + /// Reapply the initiator's lossy-cast decision (persisted in the manifest) so the + /// worker's schema revalidation honors the user's choice. Without this, a task + /// scheduled without the opt-in could still apply a lossy cast if the destination + /// schema drifts to a lossy target between scheduling and execution. + context_copy->setSetting("export_merge_tree_part_allow_lossy_cast", manifest.allow_lossy_cast); + return context_copy; } @@ -204,8 +219,8 @@ namespace ExportPartitionUtils { iceberg_args.metadata_json_string = manifest.iceberg_metadata_json; if (source_storage.getInMemoryMetadataPtr()->hasPartitionKey()) - iceberg_args.partition_values = - getPartitionValuesForIcebergCommit(source_storage, manifest.partition_id); + iceberg_args.partition_source_block = + getPartitionSourceBlockForIcebergCommit(source_storage, manifest.partition_id); } destination_storage->commitExportPartitionTransaction(manifest.transaction_id, manifest.partition_id, exported_paths, iceberg_args, context); @@ -519,6 +534,50 @@ namespace ExportPartitionUtils } } #endif + + void verifyExportSchemaCastable( + const StorageMetadataPtr & source_metadata, + const StorageMetadataPtr & destination_metadata, + const StorageID & destination_storage_id, + const ContextPtr & context) + { + /// Build (and discard) the same converting DAG the export worker will build + /// later, to surface structural mismatches (column count, untyped casts) early. + Block source_sample_block; + for (const auto & column : source_metadata->getColumns().getReadable()) + source_sample_block.insert({column.type->createColumn(), column.type, column.name}); + + const auto destination_sample_block = destination_metadata->getSampleBlockNonMaterialized(); + + const auto source_columns = source_sample_block.getColumnsWithTypeAndName(); + const auto destination_columns = destination_sample_block.getColumnsWithTypeAndName(); + + (void) ActionsDAG::makeConvertingActions( + source_columns, + destination_columns, + ActionsDAG::MatchColumnsMode::Position, + context); + + /// Lossy casts may silently change values, so reject them unless the user opts in. + if (context->getSettingsRef()[Setting::export_merge_tree_part_allow_lossy_cast]) + return; + + const size_t num_columns = std::min(source_columns.size(), destination_columns.size()); + for (size_t i = 0; i < num_columns; ++i) + { + const auto & source_column = source_columns[i]; + const auto & destination_column = destination_columns[i]; + if (!canBeSafelyCast(source_column.type, destination_column.type)) + throw Exception(ErrorCodes::BAD_ARGUMENTS, + "Cannot export to {}: column '{}' requires a lossy cast from {} to {}, " + "which may change values. Set `export_merge_tree_part_allow_lossy_cast = 1` " + "to allow lossy casts during export.", + destination_storage_id.getFullTableName(), + destination_column.name, + source_column.type->getName(), + destination_column.type->getName()); + } + } } } diff --git a/src/Storages/MergeTree/ExportPartitionUtils.h b/src/Storages/MergeTree/ExportPartitionUtils.h index eb67d288d71e..0434bc59a2cb 100644 --- a/src/Storages/MergeTree/ExportPartitionUtils.h +++ b/src/Storages/MergeTree/ExportPartitionUtils.h @@ -7,6 +7,7 @@ #include #include #include "Storages/IStorage.h" +#include #include #if USE_AVRO @@ -26,14 +27,15 @@ namespace ExportPartitionUtils ContextPtr getContextCopyWithTaskSettings(const ContextPtr & context, const ExportReplicatedMergeTreePartitionManifest & manifest); - /// Returns the partition key values for the given partition_id by reading from - /// the first active local part. Throws LOGICAL_ERROR if no such part is found. + /// Returns the representative source partition-key columns (the first active local part's + /// minmax block) for the given partition_id. The destination recomputes the Iceberg partition + /// tuple from this block by casting to its column types and applying the partition transform. /// /// Edge case: if the partition was dropped after export started, or this replica /// has not yet received any part for this partition (extreme replication lag on a /// recovery path), no active part will be found and the commit will fail. The task /// will be retried on the next poll cycle or picked up by a different replica. - std::vector getPartitionValuesForIcebergCommit( + Block getPartitionSourceBlockForIcebergCommit( MergeTreeData & storage, const String & partition_id); void commit( @@ -88,10 +90,20 @@ namespace ExportPartitionUtils const std::string & exception_message, const LoggerPtr & log); + /// Validates that source columns can be exported into the destination with the + /// same positional CAST matching as `INSERT INTO dest SELECT * FROM src`. Lossy + /// casts are rejected unless `export_merge_tree_part_allow_lossy_cast` is set. + /// Throws BAD_ARGUMENTS on any violation. + void verifyExportSchemaCastable( + const StorageMetadataPtr & source_metadata, + const StorageMetadataPtr & destination_metadata, + const StorageID & destination_storage_id, + const ContextPtr & context); + #if USE_AVRO - /// Verifies that the source MergeTree partition key is compatible with the - /// destination Iceberg partition spec by comparing field source-ids and - /// transforms in order. Throws BAD_ARGUMENTS if they do not match. + /// Verifies the source MergeTree partition key matches the destination Iceberg + /// partition spec (source-ids and transforms in order). Throws BAD_ARGUMENTS on + /// mismatch. void verifyIcebergPartitionCompatibility( const Poco::JSON::Object::Ptr & metadata_object, const ASTPtr & partition_key_ast); diff --git a/src/Storages/MergeTree/MergeTreeData.cpp b/src/Storages/MergeTree/MergeTreeData.cpp index 10414f70e663..8ad1f1907266 100644 --- a/src/Storages/MergeTree/MergeTreeData.cpp +++ b/src/Storages/MergeTree/MergeTreeData.cpp @@ -23,6 +23,7 @@ #include #include #include +#include #include #include #include @@ -7162,7 +7163,9 @@ void MergeTreeData::exportPartToTable( metadata_object->stringify(oss); iceberg_metadata_json = oss.str(); - ExportPartitionUtils::verifyIcebergPartitionCompatibility(metadata_object, source_metadata_ptr->getPartitionKeyAST()); + ExportPartitionUtils::verifyIcebergPartitionCompatibility( + metadata_object, + source_metadata_ptr->getPartitionKeyAST()); } #else (void)iceberg_metadata_json_; @@ -7170,17 +7173,12 @@ void MergeTreeData::exportPartToTable( #endif } - const auto & source_columns = source_metadata_ptr->getColumns(); + /// Positional CAST matching, like `INSERT INTO dest SELECT * FROM src`. + ExportPartitionUtils::verifyExportSchemaCastable( + source_metadata_ptr, destination_metadata_ptr, dest_storage->getStorageID(), query_context); - const auto & destination_columns = destination_metadata_ptr->getColumns(); - - /// compare all source readable columns with all destination insertable columns - /// this allows us to skip ephemeral columns - if (source_columns.getReadable().sizeOfDifference(destination_columns.getInsertable())) - throw Exception(ErrorCodes::INCOMPATIBLE_COLUMNS, "Tables have different structure"); - - /// for data lakes this check is performed differently. It is a bit more complex as we need to convert the iceberg partition spec - /// to the MergeTree partition spec and compare the two. + /// Iceberg partition compatibility is checked above; here we only need the + /// partition-key ASTs to match (partition-column types follow the lossy-cast gate). if (!dest_storage->isDataLake()) { if (query_to_string(source_metadata_ptr->getPartitionKeyAST()) != query_to_string(destination_metadata_ptr->getPartitionKeyAST())) diff --git a/src/Storages/ObjectStorage/DataLakes/IDataLakeMetadata.h b/src/Storages/ObjectStorage/DataLakes/IDataLakeMetadata.h index 29fe0ceb420e..94e0a8cb20ec 100644 --- a/src/Storages/ObjectStorage/DataLakes/IDataLakeMetadata.h +++ b/src/Storages/ObjectStorage/DataLakes/IDataLakeMetadata.h @@ -159,7 +159,7 @@ class IDataLakeMetadata : boost::noncopyable const String & /* transaction_id */, Int64 /* original_schema_id */, Int64 /* partition_spec_id */, - const std::vector & /* partition_values */, + const Block & /* partition_source_block */, SharedHeader /* sample_block */, const std::vector & /* data_file_paths */, StorageObjectStorageConfigurationPtr /* configuration */, diff --git a/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergMetadata.cpp b/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergMetadata.cpp index 5b554cc0e03a..9efa8ca1bc04 100644 --- a/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergMetadata.cpp +++ b/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergMetadata.cpp @@ -34,6 +34,7 @@ #include #include +#include #include #include @@ -1483,6 +1484,57 @@ Poco::JSON::Object::Ptr lookupSchema(const Poco::JSON::Object::Ptr & meta, Int64 "Schema with id {} not found in table metadata", schema_id); } +/// Derive the Iceberg partition tuple for an exported part from a representative source row. +/// The MergeTree `partition.value` is the source partition-key expression result; it is neither +/// cast to the destination column types nor expressed through the Iceberg transform, so it must +/// not be written to metadata directly. Within a MergeTree partition the transform result is +/// constant, so a single representative value per partition-source column (taken from the part's +/// minmax block) suffices: cast it to the destination column type and run the same transform the +/// data uses. The result is transform-correct and consistent with the exported data files. +std::vector recomputeExportPartitionValues( + ChunkPartitioner & partitioner, + const SharedHeader & sample_block, + const Block & partition_source_block) +{ + const auto & partition_columns = partitioner.getColumns(); + if (partition_columns.empty()) + return {}; + + for (const auto & column_name : partition_columns) + if (!partition_source_block.has(column_name)) + throw Exception(ErrorCodes::LOGICAL_ERROR, + "Partition source column '{}' required by the Iceberg partition transform is missing " + "from the representative source block while committing an export.", column_name); + + Columns columns; + columns.reserve(sample_block->columns()); + for (size_t i = 0; i < sample_block->columns(); ++i) + { + const auto & dest_column = sample_block->getByPosition(i); + if (partition_source_block.has(dest_column.name)) + { + const auto & source = partition_source_block.getByName(dest_column.name); + ColumnWithTypeAndName representative{source.column->cut(0, 1), source.type, source.name}; + columns.push_back(castColumn(representative, dest_column.type)); + } + else + { + auto column = dest_column.type->createColumn(); + column->insertDefault(); + columns.push_back(std::move(column)); + } + } + + auto partitioned = partitioner.partitionChunk(Chunk(std::move(columns), 1)); + if (partitioned.size() != 1) + throw Exception(ErrorCodes::LOGICAL_ERROR, + "Recomputing Iceberg partition values produced {} partitions for a single representative row; " + "a MergeTree partition must map to exactly one Iceberg partition.", partitioned.size()); + + const auto & key = partitioned.front().first; + return std::vector(key.begin(), key.end()); +} + } bool IcebergMetadata::commitImportPartitionTransactionImpl( @@ -1788,7 +1840,7 @@ void IcebergMetadata::commitExportPartitionTransaction( const String & transaction_id, Int64 original_schema_id, Int64 partition_spec_id, - const std::vector & partition_values, + const Block & partition_source_block, SharedHeader sample_block, const std::vector & data_file_paths, StorageObjectStorageConfigurationPtr configuration, @@ -1852,6 +1904,10 @@ void IcebergMetadata::commitExportPartitionTransaction( const auto partition_columns = partitioner.getColumns(); const auto partition_types = partitioner.getResultTypes(); + /// Recompute the partition tuple via the destination transform so the metadata partition + /// value matches the exported data (rather than the raw source MergeTree partition value). + const auto partition_values = recomputeExportPartitionValues(partitioner, sample_block, partition_source_block); + const auto metadata_compression_method = persistent_components.metadata_compression_method; auto config_path = persistent_components.table_path; if (config_path.empty() || config_path.back() != '/') diff --git a/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergMetadata.h b/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergMetadata.h index a60b6784f519..ac4cb9a9dfee 100644 --- a/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergMetadata.h +++ b/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergMetadata.h @@ -167,7 +167,7 @@ class IcebergMetadata : public IDataLakeMetadata const String & transaction_id, Int64 original_schema_id, Int64 partition_spec_id, - const std::vector & partition_values, + const Block & partition_source_block, SharedHeader sample_block, const std::vector & data_file_paths, StorageObjectStorageConfigurationPtr configuration, diff --git a/src/Storages/ObjectStorage/StorageObjectStorage.cpp b/src/Storages/ObjectStorage/StorageObjectStorage.cpp index 6cb98ae5dfca..46012ed387bc 100644 --- a/src/Storages/ObjectStorage/StorageObjectStorage.cpp +++ b/src/Storages/ObjectStorage/StorageObjectStorage.cpp @@ -731,7 +731,8 @@ void StorageObjectStorage::commitExportPartitionTransaction( /// Parse the Iceberg metadata snapshot (stored in ZooKeeper at export-start time) only to /// extract the schema-id and partition-spec-id that were current when the export began. /// partition_columns and partition_types are derived inside commitExportPartitionTransaction - /// from the same JSON, so only partition_values need to be carried here. + /// from the same JSON; the representative source partition columns are carried here so the + /// partition tuple can be recomputed through the destination transform. Poco::JSON::Parser iceberg_parser; Poco::JSON::Object::Ptr iceberg_metadata = iceberg_parser.parse(iceberg_commit_export_partition_arguments.metadata_json_string).extract(); @@ -746,7 +747,7 @@ void StorageObjectStorage::commitExportPartitionTransaction( transaction_id, original_schema_id, partition_spec_id, - iceberg_commit_export_partition_arguments.partition_values, + iceberg_commit_export_partition_arguments.partition_source_block, std::make_shared(getInMemoryMetadataPtr()->getSampleBlock()), exported_paths, configuration, diff --git a/src/Storages/StorageReplicatedMergeTree.cpp b/src/Storages/StorageReplicatedMergeTree.cpp index cd17e609d9e7..6d6bf12c0375 100644 --- a/src/Storages/StorageReplicatedMergeTree.cpp +++ b/src/Storages/StorageReplicatedMergeTree.cpp @@ -85,6 +85,7 @@ #include #include #include +#include #include #include @@ -234,6 +235,7 @@ namespace Setting extern const SettingsUInt64 export_merge_tree_part_max_rows_per_file; extern const SettingsBool export_merge_tree_part_throw_on_pending_mutations; extern const SettingsBool export_merge_tree_part_throw_on_pending_patch_parts; + extern const SettingsBool export_merge_tree_part_allow_lossy_cast; extern const SettingsExportPartitionAllOnError export_merge_tree_partition_all_on_error; extern const SettingsString export_merge_tree_part_filename_pattern; extern const SettingsBool write_full_path_in_iceberg_metadata; @@ -8595,19 +8597,17 @@ void StorageReplicatedMergeTree::exportPartitionToTable(const PartitionCommand & auto src_snapshot = getInMemoryMetadataPtr(); auto destination_snapshot = dest_storage->getInMemoryMetadataPtr(); - /// compare all source readable columns with all destination insertable columns - /// this allows us to skip ephemeral columns - if (src_snapshot->getColumns().getReadable().sizeOfDifference(destination_snapshot->getColumns().getInsertable())) - throw Exception(ErrorCodes::INCOMPATIBLE_COLUMNS, "Tables have different structure"); + /// Positional CAST matching, like `INSERT INTO dest SELECT * FROM src`. + ExportPartitionUtils::verifyExportSchemaCastable( + src_snapshot, destination_snapshot, dest_storage->getStorageID(), query_context); - /// for data lakes this check is performed later. It is a bit more complex as we need to convert the iceberg partition spec - /// to the MergeTree partition spec and compare the two. + /// Iceberg partition compatibility is checked below; here we only need the + /// partition-key ASTs to match (partition-column types follow the lossy-cast gate). if (!dest_storage->isDataLake()) { if (query_to_string(src_snapshot->getPartitionKeyAST()) != query_to_string(destination_snapshot->getPartitionKeyAST())) throw Exception(ErrorCodes::BAD_ARGUMENTS, "Tables have different partition key"); } - zkutil::ZooKeeperPtr zookeeper = getZooKeeperAndAssertNotReadonly(); @@ -8720,6 +8720,7 @@ void StorageReplicatedMergeTree::exportPartitionToTable(const PartitionCommand & manifest.file_already_exists_policy = query_context->getSettingsRef()[Setting::export_merge_tree_part_file_already_exists_policy].value; manifest.filename_pattern = query_context->getSettingsRef()[Setting::export_merge_tree_part_filename_pattern].value; manifest.write_full_path_in_iceberg_metadata = query_context->getSettingsRef()[Setting::write_full_path_in_iceberg_metadata]; + manifest.allow_lossy_cast = query_context->getSettingsRef()[Setting::export_merge_tree_part_allow_lossy_cast]; if (dest_storage->isDataLake()) { @@ -8753,7 +8754,8 @@ void StorageReplicatedMergeTree::exportPartitionToTable(const PartitionCommand & const auto metadata_object = iceberg_metadata->getMetadataJSON(query_context); ExportPartitionUtils::verifyIcebergPartitionCompatibility( - metadata_object, src_snapshot->getPartitionKeyAST()); + metadata_object, + src_snapshot->getPartitionKeyAST()); std::ostringstream oss; // STYLE_CHECK_ALLOW_STD_STRING_STREAM oss.exceptions(std::ios::failbit); diff --git a/tests/integration/test_export_merge_tree_part_to_iceberg/test.py b/tests/integration/test_export_merge_tree_part_to_iceberg/test.py index b371727a08cd..486adf1f2b17 100644 --- a/tests/integration/test_export_merge_tree_part_to_iceberg/test.py +++ b/tests/integration/test_export_merge_tree_part_to_iceberg/test.py @@ -69,11 +69,15 @@ def get_part(node, table: str, partition_id: str) -> str: ).strip() -def export_part(node, table: str, part: str, dest: str) -> None: +def export_part(node, table: str, part: str, dest: str, extra_settings: str = "") -> None: + settings = ( + "allow_experimental_export_merge_tree_part = 1, " + "allow_experimental_insert_into_iceberg = 1" + ) + if extra_settings: + settings += ", " + extra_settings node.query( - f"ALTER TABLE {table} EXPORT PART '{part}' TO TABLE {dest} " - f"SETTINGS allow_experimental_export_merge_tree_part = 1, " - f"allow_experimental_insert_into_iceberg = 1" + f"ALTER TABLE {table} EXPORT PART '{part}' TO TABLE {dest} SETTINGS {settings}" ) @@ -121,6 +125,42 @@ def assert_part_log(node, table: str, part: str) -> None: ) +def wait_for_failed_export_part( + node, + table: str, + part: str, + timeout: int = 60, + poll_interval: float = 0.5, +) -> str: + """Poll system.part_log until a failed ExportPart event appears for *part*. + + Returns the exception text recorded on the log entry, which lets callers + assert on the specific runtime error that propagated from the export worker. + """ + deadline = time.time() + timeout + last_seen = "" + while time.time() < deadline: + node.query("SYSTEM FLUSH LOGS") + row = node.query( + f"SELECT error, exception FROM system.part_log " + f"WHERE event_type = 'ExportPart' " + f"AND database = currentDatabase() " + f"AND table = '{table}' " + f"AND part_name = '{part}' " + f"AND error != 0 " + f"ORDER BY event_time DESC LIMIT 1" + ).strip() + if row: + _error, exception = row.split("\t", 1) + return exception + last_seen = row + time.sleep(poll_interval) + raise TimeoutError( + f"Failed ExportPart event for part {part!r} in table {table!r} " + f"did not appear in system.part_log within {timeout}s (last row: {last_seen!r})" + ) + + # --------------------------------------------------------------------------- # Tests # --------------------------------------------------------------------------- @@ -349,6 +389,38 @@ def test_export_part_with_year_transform_partition(cluster): node.query(f"DROP TABLE IF EXISTS {iceberg}") +def test_export_part_partition_column_lossless_widening(cluster): + """A lossless widening of a partition column (year Int32 -> Int64) round-trips.""" + node = cluster.instances["node1"] + sfx = unique_suffix() + mt = f"mt_pcol_widening_{sfx}" + iceberg = f"iceberg_pcol_widening_{sfx}" + + make_mt(node, mt, "id Int32, year Int32", "year") + make_iceberg_s3(node, iceberg, "id Int32, year Int64", "year") + + node.query(f"INSERT INTO {mt} VALUES (1, 2020), (2, 2020), (3, 2020)") + + part_2020 = get_part(node, mt, "2020") + export_part(node, mt, part_2020, iceberg) + wait_for_export_part(node, mt, part_2020) + + count = int(node.query(f"SELECT count() FROM {iceberg}").strip()) + assert count == 3, f"Expected 3 rows in Iceberg table after export, got {count}" + + result = node.query( + f"SELECT id, toTypeName(year), year FROM {iceberg} ORDER BY id" + ).strip() + assert result == "1\tInt64\t2020\n2\tInt64\t2020\n3\tInt64\t2020", ( + f"Unexpected widened partition-column data:\n{result}" + ) + + assert_part_log(node, mt, part_2020) + + node.query(f"DROP TABLE IF EXISTS {mt} SYNC") + node.query(f"DROP TABLE IF EXISTS {iceberg}") + + def test_export_part_partition_key_mismatch_is_rejected(cluster): """ EXPORT PART must synchronously reject (BAD_ARGUMENTS) when the source @@ -479,3 +551,205 @@ def test_export_part_writes_column_statistics(cluster): node.query(f"DROP TABLE IF EXISTS {mt} SYNC") node.query(f"DROP TABLE IF EXISTS {iceberg}") + + +def test_export_part_column_count_mismatch_source_more_is_rejected(cluster): + """ + Source has 3 columns (id, year, extra), destination has 2 (id, year). + The ALTER must be rejected synchronously with NUMBER_OF_COLUMNS_DOESNT_MATCH + and the Iceberg table must remain empty. + """ + node = cluster.instances["node1"] + sfx = unique_suffix() + mt = f"mt_count_more_{sfx}" + iceberg = f"iceberg_count_more_{sfx}" + + make_mt(node, mt, "id Int32, year Int32, extra String", "year") + make_iceberg_s3(node, iceberg, "id Int32, year Int32", "year") + + node.query(f"INSERT INTO {mt} VALUES (1, 2020, 'foo'), (2, 2020, 'bar')") + part_2020 = get_part(node, mt, "2020") + + error = node.query_and_get_error( + f"ALTER TABLE {mt} EXPORT PART '{part_2020}' TO TABLE {iceberg} " + f"SETTINGS allow_experimental_export_merge_tree_part = 1, " + f"allow_experimental_insert_into_iceberg = 1" + ) + assert "NUMBER_OF_COLUMNS_DOESNT_MATCH" in error, ( + f"Expected NUMBER_OF_COLUMNS_DOESNT_MATCH for source>dest column count, " + f"got: {error!r}" + ) + + node.query(f"DROP TABLE IF EXISTS {mt} SYNC") + node.query(f"DROP TABLE IF EXISTS {iceberg}") + + +def test_export_part_column_count_mismatch_source_fewer_is_rejected(cluster): + """ + Source has 2 columns (id, year), destination has 3 (id, year, extra). + Same expected synchronous rejection as the source>dest case. + """ + node = cluster.instances["node1"] + sfx = unique_suffix() + mt = f"mt_count_fewer_{sfx}" + iceberg = f"iceberg_count_fewer_{sfx}" + + make_mt(node, mt, "id Int32, year Int32", "year") + make_iceberg_s3(node, iceberg, "id Int32, year Int32, extra String", "year") + + node.query(f"INSERT INTO {mt} VALUES (1, 2020), (2, 2020)") + part_2020 = get_part(node, mt, "2020") + + error = node.query_and_get_error( + f"ALTER TABLE {mt} EXPORT PART '{part_2020}' TO TABLE {iceberg} " + f"SETTINGS allow_experimental_export_merge_tree_part = 1, " + f"allow_experimental_insert_into_iceberg = 1" + ) + assert "NUMBER_OF_COLUMNS_DOESNT_MATCH" in error, ( + f"Expected NUMBER_OF_COLUMNS_DOESNT_MATCH for source Int32) succeeds once the user opts in via + export_merge_tree_part_allow_lossy_cast.""" + node = cluster.instances["node1"] + sfx = unique_suffix() + mt = f"mt_narrow_fit_{sfx}" + iceberg = f"iceberg_narrow_fit_{sfx}" + + make_mt(node, mt, "id Int64, year Int32", "year") + make_iceberg_s3(node, iceberg, "id Int32, year Int32", "year") + + node.query(f"INSERT INTO {mt} VALUES (1, 2020), (2, 2020)") + part_2020 = get_part(node, mt, "2020") + + export_part(node, mt, part_2020, iceberg, "export_merge_tree_part_allow_lossy_cast = 1") + wait_for_export_part(node, mt, part_2020) + + count = int(node.query(f"SELECT count() FROM {iceberg}").strip()) + assert count == 2, f"Expected 2 rows in Iceberg table after export, got {count}" + + result = node.query( + f"SELECT id, toTypeName(id), year FROM {iceberg} ORDER BY id" + ).strip() + assert result == "1\tInt32\t2020\n2\tInt32\t2020", ( + f"Unexpected narrowed data:\n{result}" + ) + + assert_part_log(node, mt, part_2020) + + node.query(f"DROP TABLE IF EXISTS {mt} SYNC") + node.query(f"DROP TABLE IF EXISTS {iceberg}") + + +def test_export_part_runtime_cast_failure_propagates_async(cluster): + """A String value that cannot be parsed as the destination Int32 passes the + synchronous lossy-cast gate (with export_merge_tree_part_allow_lossy_cast = 1) but + fails at runtime in the async worker; the failure surfaces in system.part_log and + Iceberg is left empty. + + (Integer overflow is not used because the internal cast uses CastType::nonAccurate, + which wraps rather than throwing.) + """ + node = cluster.instances["node1"] + sfx = unique_suffix() + mt = f"mt_runtime_cast_fail_{sfx}" + iceberg = f"iceberg_runtime_cast_fail_{sfx}" + + make_mt(node, mt, "id String, year Int32", "year") + make_iceberg_s3(node, iceberg, "id Int32, year Int32", "year") + + node.query(f"INSERT INTO {mt} VALUES ('not a number', 2020)") + part_2020 = get_part(node, mt, "2020") + + export_part(node, mt, part_2020, iceberg, "export_merge_tree_part_allow_lossy_cast = 1") + + exception = wait_for_failed_export_part(node, mt, part_2020) + assert exception, ( + f"Expected non-empty exception text on failed ExportPart entry, got {exception!r}" + ) + + count = int(node.query(f"SELECT count() FROM {iceberg}").strip()) + assert count == 0, ( + f"Expected 0 rows in Iceberg table after failed export, got {count}" + ) + + node.query(f"DROP TABLE IF EXISTS {mt} SYNC") + node.query(f"DROP TABLE IF EXISTS {iceberg}") diff --git a/tests/integration/test_export_replicated_mt_partition_to_iceberg/test.py b/tests/integration/test_export_replicated_mt_partition_to_iceberg/test.py index 3f0947dcfa1a..62a8a2196313 100644 --- a/tests/integration/test_export_replicated_mt_partition_to_iceberg/test.py +++ b/tests/integration/test_export_replicated_mt_partition_to_iceberg/test.py @@ -940,3 +940,661 @@ def test_export_partition_writes_column_statistics(cluster): entries = fetch_manifest_entries(node, query_id) assert_exported_stats(entries) + + +def test_export_partition_column_count_mismatch_source_more_is_rejected(cluster): + """ + Source has 3 columns (id, year, extra), destination has 2 (id, year). + The ALTER must be rejected synchronously with NUMBER_OF_COLUMNS_DOESNT_MATCH, + nothing must be scheduled in system.replicated_partition_exports, and the + Iceberg table must remain empty. + """ + node = cluster.instances["replica1"] + + uid = unique_suffix() + mt_table = f"mt_count_more_{uid}" + iceberg_table = f"iceberg_count_more_{uid}" + + make_rmt(node, mt_table, "id Int64, year Int32, extra String", "year", + replica_name="replica1") + node.query(f"INSERT INTO {mt_table} VALUES (1, 2020, 'foo'), (2, 2020, 'bar')") + + make_iceberg_s3(node, iceberg_table, "id Int64, year Int32", partition_by="year") + + error = node.query_and_get_error( + f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {iceberg_table}", + settings={"allow_insert_into_iceberg": 1}, + ) + assert "NUMBER_OF_COLUMNS_DOESNT_MATCH" in error, ( + f"Expected NUMBER_OF_COLUMNS_DOESNT_MATCH for source>dest column count, " + f"got: {error!r}" + ) + + rows_in_system_view = node.query( + f"SELECT count() FROM system.replicated_partition_exports " + f"WHERE source_table = '{mt_table}' " + f" AND destination_table = '{iceberg_table}' " + f" AND partition_id = '2020'" + ).strip() + assert rows_in_system_view == "0", ( + f"Expected no row in system.replicated_partition_exports after a " + f"synchronously-rejected export, got {rows_in_system_view}." + ) + + count = int(node.query(f"SELECT count() FROM {iceberg_table}").strip()) + assert count == 0, ( + f"Expected 0 rows in Iceberg table after rejected export, got {count}" + ) + + +def test_export_partition_column_count_mismatch_source_fewer_is_rejected(cluster): + """ + Source has 2 columns (id, year), destination has 3 (id, year, extra). + Same expected synchronous rejection as the source>dest case. + """ + node = cluster.instances["replica1"] + + uid = unique_suffix() + mt_table = f"mt_count_fewer_{uid}" + iceberg_table = f"iceberg_count_fewer_{uid}" + + make_rmt(node, mt_table, "id Int64, year Int32", "year", replica_name="replica1") + node.query(f"INSERT INTO {mt_table} VALUES (1, 2020), (2, 2020)") + + make_iceberg_s3(node, iceberg_table, "id Int64, year Int32, extra String", + partition_by="year") + + error = node.query_and_get_error( + f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {iceberg_table}", + settings={"allow_insert_into_iceberg": 1}, + ) + assert "NUMBER_OF_COLUMNS_DOESNT_MATCH" in error, ( + f"Expected NUMBER_OF_COLUMNS_DOESNT_MATCH for source Int64) and the + partition column (year Int32 -> Int64) round-trips.""" + node = cluster.instances["replica1"] + + uid = unique_suffix() + mt_table = f"mt_widen_{uid}" + iceberg_table = f"iceberg_widen_{uid}" + + make_rmt(node, mt_table, "id Int32, year Int32", "year", replica_name="replica1") + node.query(f"INSERT INTO {mt_table} VALUES (1, 2020), (2, 2020)") + + make_iceberg_s3(node, iceberg_table, "id Int64, year Int64", partition_by="year") + + node.query( + f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {iceberg_table}", + settings={"allow_insert_into_iceberg": 1}, + ) + wait_for_export_status(node, mt_table, iceberg_table, "2020", "COMPLETED") + + count = int(node.query(f"SELECT count() FROM {iceberg_table}").strip()) + assert count == 2, f"Expected 2 rows in Iceberg table after export, got {count}" + + result = node.query( + f"SELECT id, toTypeName(id), year, toTypeName(year) FROM {iceberg_table} ORDER BY id" + ).strip() + assert result == "1\tInt64\t2020\tInt64\n2\tInt64\t2020\tInt64", ( + f"Unexpected widened data:\n{result}" + ) + + +def test_export_partition_with_castable_narrowing_values_fit(cluster): + """A lossy narrowing (id Int64 -> Int32) succeeds once the user opts in via + export_merge_tree_part_allow_lossy_cast.""" + node = cluster.instances["replica1"] + + uid = unique_suffix() + mt_table = f"mt_narrow_fit_{uid}" + iceberg_table = f"iceberg_narrow_fit_{uid}" + + make_rmt(node, mt_table, "id Int64, year Int32", "year", replica_name="replica1") + node.query(f"INSERT INTO {mt_table} VALUES (1, 2020), (2, 2020)") + + make_iceberg_s3(node, iceberg_table, "id Int32, year Int32", partition_by="year") + + node.query( + f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {iceberg_table}", + settings={ + "allow_insert_into_iceberg": 1, + "export_merge_tree_part_allow_lossy_cast": 1, + }, + ) + wait_for_export_status(node, mt_table, iceberg_table, "2020", "COMPLETED") + + count = int(node.query(f"SELECT count() FROM {iceberg_table}").strip()) + assert count == 2, f"Expected 2 rows in Iceberg table after export, got {count}" + + result = node.query( + f"SELECT id, toTypeName(id), year FROM {iceberg_table} ORDER BY id" + ).strip() + assert result == "1\tInt32\t2020\n2\tInt32\t2020", ( + f"Unexpected narrowed data:\n{result}" + ) + + +def test_export_partition_lossy_cast_rejected_without_optin(cluster): + """A lossy narrowing (id Int64 -> Int32) is rejected synchronously with + BAD_ARGUMENTS unless export_merge_tree_part_allow_lossy_cast is set.""" + node = cluster.instances["replica1"] + + uid = unique_suffix() + mt_table = f"mt_lossy_reject_{uid}" + iceberg_table = f"iceberg_lossy_reject_{uid}" + + make_rmt(node, mt_table, "id Int64, year Int32", "year", replica_name="replica1") + node.query(f"INSERT INTO {mt_table} VALUES (1, 2020)") + + make_iceberg_s3(node, iceberg_table, "id Int32, year Int32", partition_by="year") + + error = node.query_and_get_error( + f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {iceberg_table} " + f"SETTINGS allow_insert_into_iceberg = 1" + ) + assert "BAD_ARGUMENTS" in error, f"Expected BAD_ARGUMENTS, got: {error!r}" + assert "lossy cast" in error, f"Expected 'lossy cast' in error, got: {error!r}" + + count = int(node.query(f"SELECT count() FROM {iceberg_table}").strip()) + assert count == 0, f"Expected no rows after a rejected export, got {count}" + + +def test_export_partition_runtime_cast_failure_propagates_async(cluster): + """A String value that cannot be parsed as the destination Int32 passes the + synchronous lossy-cast gate (with export_merge_tree_part_allow_lossy_cast = 1) but + fails at runtime in the async worker, marking the export FAILED and leaving Iceberg + empty. + + (Integer overflow is not used because the internal cast uses CastType::nonAccurate, + which wraps rather than throwing.) + """ + node = cluster.instances["replica1"] + + uid = unique_suffix() + mt_table = f"mt_runtime_cast_fail_{uid}" + iceberg_table = f"iceberg_runtime_cast_fail_{uid}" + + make_rmt(node, mt_table, "id String, year Int32", "year", replica_name="replica1") + node.query(f"INSERT INTO {mt_table} VALUES ('not a number', 2020)") + + make_iceberg_s3(node, iceberg_table, "id Int32, year Int32", partition_by="year") + + node.query( + f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {iceberg_table} " + f"SETTINGS export_merge_tree_partition_max_retries = 1, allow_insert_into_iceberg = 1, " + f"export_merge_tree_part_allow_lossy_cast = 1" + ) + + wait_for_export_status(node, mt_table, iceberg_table, "2020", "FAILED", timeout=60) + + exception_count = int(node.query( + f"SELECT any(exception_count) FROM system.replicated_partition_exports " + f"WHERE source_table = '{mt_table}' " + f" AND destination_table = '{iceberg_table}' " + f" AND partition_id = '2020'" + ).strip()) + assert exception_count > 0, ( + "Expected non-zero exception_count after a failed runtime cast" + ) + + count = int(node.query(f"SELECT count() FROM {iceberg_table}").strip()) + assert count == 0, ( + f"Expected 0 rows in Iceberg table after failed export, got {count}" + ) + + +def test_export_partition_all_iceberg_types(cluster): + """Every getIcebergType-supported type round-trips through an EXPORT PARTITION: + scalars use narrower source types (explicit lossless widening CASTs), plus + Array/Map/Tuple nested columns.""" + node = cluster.instances["replica1"] + + uid = unique_suffix() + mt_table = f"mt_all_types_{uid}" + iceberg_table = f"iceberg_all_types_{uid}" + + # Scalar source types are strictly narrower than the destination; the export inserts + # a positional widening CAST per column (Int8->Int16, UInt32->UInt64, ...). Nested + # columns keep the same type on both sides. + source_columns = ( + "i16 Int8, u16 UInt8, u32 UInt16, u64 UInt32, " + "id Int16, big Int32, f32 Float32, f64 Float64, " + "d Date, d32 Date32, dt DateTime, dt64 DateTime64(6), " + "s String, uid UUID, " + "arr Array(Int32), m Map(String, Int64), tup Tuple(a Int32, b String), " + "year Int32" + ) + dest_columns = ( + "i16 Int16, u16 UInt16, u32 UInt32, u64 UInt64, " + "id Int32, big Int64, f32 Float32, f64 Float64, " + "d Date, d32 Date32, dt DateTime, dt64 DateTime64(6), " + "s String, uid UUID, " + "arr Array(Int32), m Map(String, Int64), tup Tuple(a Int32, b String), " + "year Int32" + ) + + make_rmt(node, mt_table, source_columns, "year", replica_name="replica1") + make_iceberg_s3(node, iceberg_table, dest_columns, partition_by="year") + + node.query( + f""" + INSERT INTO {mt_table} + (i16, u16, u32, u64, id, big, f32, f64, d, d32, dt, dt64, s, uid, arr, m, tup, year) + VALUES ( + -100, 200, 50000, 4000000000, + 12345, 1000000000, 3.14, 2.718281828459045, + '2024-01-15', '2024-01-15', '2024-01-15 12:30:45', '2024-01-15 12:30:45.123456', + 'hello iceberg', '550e8400-e29b-41d4-a716-446655440000', + [1, 2, 3], {{'a': 10, 'b': 20}}, (7, 'seven'), 2024 + ) + """ + ) + + node.query( + f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2024' TO TABLE {iceberg_table}", + settings={"allow_insert_into_iceberg": 1}, + ) + wait_for_export_status(node, mt_table, iceberg_table, "2024", "COMPLETED") + + count = int(node.query(f"SELECT count() FROM {iceberg_table}").strip()) + assert count == 1, f"Expected 1 row in Iceberg table, got {count}" + + result = node.query( + f""" + SELECT + i16, u16, u32, u64, id, big, + toString(d), toString(d32), toString(dt), + s, toString(uid), + arr, m['a'], m['b'], tup.a, tup.b, year + FROM {iceberg_table} + """ + ).strip() + expected = "\t".join([ + "-100", "200", "50000", "4000000000", + "12345", "1000000000", + "2024-01-15", "2024-01-15", "2024-01-15 12:30:45.000000", + "hello iceberg", "550e8400-e29b-41d4-a716-446655440000", + "[1,2,3]", "10", "20", "7", "seven", "2024", + ]) + assert result == expected, f"Unexpected round-trip data:\n{result!r}\nexpected:\n{expected!r}" + + # Floats compared with a tolerance to avoid formatting flakiness. + floats_ok = node.query( + f"SELECT abs(f32 - 3.14) < 1e-4 AND abs(f64 - 2.718281828459045) < 1e-12 FROM {iceberg_table}" + ).strip() + assert floats_ok == "1", f"Float round-trip outside tolerance: {floats_ok!r}" + + # DateTime64 sub-second component: assert the date part is preserved (exact format varies). + ts_result = node.query(f"SELECT dt64 FROM {iceberg_table}").strip() + assert "2024-01-15" in ts_result, f"DateTime64 date component missing: {ts_result!r}" + + +def test_export_partition_all_iceberg_types_lossy(cluster): + """Lossy narrowing casts across types succeed with the opt-in flag: values that + fit round-trip, Float64 -> Float32 loses precision, and Nullable columns carry + both NULL and non-NULL (the latter via a lossy Nullable(Int64) -> Nullable(Int32)).""" + node = cluster.instances["replica1"] + + uid = unique_suffix() + mt_table = f"mt_lossy_types_{uid}" + iceberg_table = f"iceberg_lossy_types_{uid}" + + # Each source column is wider than the destination, so the export inserts a lossy + # narrowing CAST (allowed only because export_merge_tree_part_allow_lossy_cast=1). + # Int8/UInt8 are not Iceberg-representable, so the narrowest integer dest is Int16. + source_columns = ( + "big Int64, ubig UInt64, mid Int32, " + "f Float64, dt DateTime64(6), d Date32, " + "opt_s Nullable(String), opt_i Nullable(Int64), year Int32" + ) + dest_columns = ( + "big Int32, ubig UInt32, mid Int16, " + "f Float32, dt DateTime, d Date, " + "opt_s Nullable(String), opt_i Nullable(Int32), year Int32" + ) + + make_rmt(node, mt_table, source_columns, "year", replica_name="replica1") + make_iceberg_s3(node, iceberg_table, dest_columns, partition_by="year") + + # Values chosen to fit the destination types (the async cast wraps on overflow + # rather than throwing, so out-of-range values would silently corrupt instead). + # opt_s is NULL and opt_i is set, covering both nullable paths in one row. + node.query( + f""" + INSERT INTO {mt_table} (big, ubig, mid, f, dt, d, opt_s, opt_i, year) + VALUES ( + 1000000, 2000000000, 30000, + 2.718281828459045, '2024-01-15 12:30:45.123456', '2024-01-15', + NULL, 100, 2024 + ) + """ + ) + + node.query( + f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2024' TO TABLE {iceberg_table}", + settings={ + "allow_insert_into_iceberg": 1, + "export_merge_tree_part_allow_lossy_cast": 1, + }, + ) + wait_for_export_status(node, mt_table, iceberg_table, "2024", "COMPLETED") + + count = int(node.query(f"SELECT count() FROM {iceberg_table}").strip()) + assert count == 1, f"Expected 1 row in Iceberg table, got {count}" + + result = node.query( + f"SELECT big, ubig, mid, toString(d), toString(dt), opt_s, opt_i, year FROM {iceberg_table}" + ).strip() + expected = "\t".join([ + "1000000", "2000000000", "30000", + "2024-01-15", "2024-01-15 12:30:45.000000", "\\N", "100", "2024", + ]) + assert result == expected, f"Unexpected lossy round-trip data:\n{result!r}\nexpected:\n{expected!r}" + + # Float64 -> Float32 stays within Float32 precision but is no longer exact. + f_checks = node.query( + f"SELECT abs(f - 2.718281828459045) < 1e-6, abs(f - 2.718281828459045) > 1e-9 FROM {iceberg_table}" + ).strip() + assert f_checks == "1\t1", f"Expected Float32 precision loss within tolerance, got: {f_checks!r}" + + +def _data_file_partition_records(entries): + """Partition dicts of the non-delete data files described by manifest entries.""" + records = [] + for entry in entries: + data_file = entry.get("data_file") or {} + if data_file.get("content", 0) not in (0, None): + continue + partition = data_file.get("partition") + if partition is not None: + records.append(partition) + return records + + +def _partition_scalar(partition, field): + """Read a partition field value, tolerating an Avro-union ``{type: value}`` wrapper.""" + value = partition.get(field) + if isinstance(value, dict): + assert len(value) == 1, f"Unexpected partition union shape for {field!r}: {value!r}" + value = next(iter(value.values())) + return value + + +def test_export_partition_bucket_transform_metadata_matches_data(cluster): + """A bucket[N] partition column whose type changes Int64 -> String records the + destination murmur(String) bucket in the Iceberg metadata, matching the exported + data rather than the source hashLong bucket.""" + node = cluster.instances["replica1"] + + uid = unique_suffix() + mt_table = f"mt_bucket_xform_{uid}" + iceberg_table = f"iceberg_bucket_xform_{uid}" + + # N=16, key=42 diverges: icebergBucket(16, 42::Int64)=14 (source/old hashLong) but + # icebergBucket(16, '42')=6 (destination/new murmur over the exported String). + make_rmt(node, mt_table, "id Int64, key Int64", "icebergBucket(16, key)", + replica_name="replica1") + node.query(f"INSERT INTO {mt_table} VALUES (1, 42), (2, 42)") + + make_iceberg_s3(node, iceberg_table, "id Int64, key String", + partition_by="icebergBucket(16, key)") + + pid = first_partition_id(node, mt_table) + node.query( + f"ALTER TABLE {mt_table} EXPORT PARTITION ID '{pid}' TO TABLE {iceberg_table}", + settings={"allow_insert_into_iceberg": 1}, + ) + wait_for_export_status(node, mt_table, iceberg_table, pid, "COMPLETED") + + count = int(node.query(f"SELECT count() FROM {iceberg_table}").strip()) + assert count == 2, f"Expected 2 rows after export, got {count}" + + string_bucket = int(node.query( + f"SELECT DISTINCT icebergBucket(16, key) FROM {iceberg_table}" + ).strip()) + long_bucket = int(node.query( + f"SELECT DISTINCT icebergBucket(16, toInt64(key)) FROM {iceberg_table}" + ).strip()) + assert string_bucket != long_bucket, ( + f"Test setup invalid: String and Int64 buckets coincide ({string_bucket}); " + f"pick a different N/key so the transform diverges." + ) + + query_id = f"bucket_xform_{uid}" + node.query( + f"SELECT * FROM {iceberg_table}", + query_id=query_id, + settings={"iceberg_metadata_log_level": "manifest_file_entry"}, + ) + entries = fetch_manifest_entries(node, query_id) + partitions = _data_file_partition_records(entries) + assert partitions, "No data-file partition records found in manifest entries" + meta_values = {int(_partition_scalar(p, "key")) for p in partitions} + assert meta_values == {string_bucket}, ( + f"Metadata bucket {meta_values} must equal the destination String bucket " + f"{string_bucket} (not the source Int64 bucket {long_bucket})." + ) + + +def test_export_partition_month_transform_metadata_matches_data(cluster): + """A month-transform partition records a months-since-epoch value in metadata that + matches the value derived from the exported data, and a transform-filtered read + returns the rows.""" + node = cluster.instances["replica1"] + + uid = unique_suffix() + mt_table = f"mt_month_xform_{uid}" + iceberg_table = f"iceberg_month_xform_{uid}" + + make_rmt(node, mt_table, "id Int64, event_date Date", + "toMonthNumSinceEpoch(event_date)", replica_name="replica1") + node.query( + f"INSERT INTO {mt_table} VALUES " + f"(1, '2024-03-05'), (2, '2024-03-20'), (3, '2024-03-31')" + ) + + make_iceberg_s3(node, iceberg_table, "id Int64, event_date Date", + partition_by="toMonthNumSinceEpoch(event_date)") + + pid = first_partition_id(node, mt_table) + node.query( + f"ALTER TABLE {mt_table} EXPORT PARTITION ID '{pid}' TO TABLE {iceberg_table}", + settings={"allow_insert_into_iceberg": 1}, + ) + wait_for_export_status(node, mt_table, iceberg_table, pid, "COMPLETED") + + count = int(node.query(f"SELECT count() FROM {iceberg_table}").strip()) + assert count == 3, f"Expected 3 rows after export, got {count}" + + month_num = int(node.query( + f"SELECT DISTINCT toMonthNumSinceEpoch(event_date) FROM {iceberg_table}" + ).strip()) + + query_id = f"month_xform_{uid}" + node.query( + f"SELECT * FROM {iceberg_table}", + query_id=query_id, + settings={"iceberg_metadata_log_level": "manifest_file_entry"}, + ) + entries = fetch_manifest_entries(node, query_id) + partitions = _data_file_partition_records(entries) + assert partitions, "No data-file partition records found in manifest entries" + meta_values = {int(_partition_scalar(p, "event_date")) for p in partitions} + assert meta_values == {month_num}, ( + f"Metadata month {meta_values} must equal toMonthNumSinceEpoch over the data " + f"({month_num})." + ) + + filtered = int(node.query( + f"SELECT count() FROM {iceberg_table} " + f"WHERE toMonthNumSinceEpoch(event_date) = {month_num}" + ).strip()) + assert filtered == 3, f"Transform-filtered read expected 3 rows, got {filtered}" + + +def test_export_partition_identity_type_change_metadata_matches_data(cluster): + """An identity partition column whose type changes UInt16 -> String records the + destination String value in the Iceberg metadata, matching the exported data.""" + node = cluster.instances["replica1"] + + uid = unique_suffix() + mt_table = f"mt_identity_xform_{uid}" + iceberg_table = f"iceberg_identity_xform_{uid}" + + make_rmt(node, mt_table, "id Int32, year UInt16", "year", replica_name="replica1") + node.query(f"INSERT INTO {mt_table} VALUES (1, 2024), (2, 2024)") + + make_iceberg_s3(node, iceberg_table, "id Int32, year String", partition_by="year") + + pid = first_partition_id(node, mt_table) + node.query( + f"ALTER TABLE {mt_table} EXPORT PARTITION ID '{pid}' TO TABLE {iceberg_table}", + settings={"allow_insert_into_iceberg": 1}, + ) + wait_for_export_status(node, mt_table, iceberg_table, pid, "COMPLETED") + + count = int(node.query(f"SELECT count() FROM {iceberg_table}").strip()) + assert count == 2, f"Expected 2 rows after export, got {count}" + + data_year = node.query(f"SELECT DISTINCT year FROM {iceberg_table}").strip() + assert data_year == "2024", f"Expected exported year '2024' (String), got {data_year!r}" + + query_id = f"identity_xform_{uid}" + node.query( + f"SELECT * FROM {iceberg_table}", + query_id=query_id, + settings={"iceberg_metadata_log_level": "manifest_file_entry"}, + ) + entries = fetch_manifest_entries(node, query_id) + partitions = _data_file_partition_records(entries) + assert partitions, "No data-file partition records found in manifest entries" + meta_values = {str(_partition_scalar(p, "year")) for p in partitions} + assert meta_values == {"2024"}, ( + f"Metadata partition {meta_values} must equal the destination String value " + f"'2024' (not the source integer representation)." + ) + + +def test_export_partition_multicolumn_identity_metadata_matches_data(cluster): + """A multi-column identity partition (event_date Date, retention UInt64 -> Int64) + records per-column values in the Iceberg metadata that match the exported data.""" + node = cluster.instances["replica1"] + + uid = unique_suffix() + mt_table = f"mt_multicol_{uid}" + iceberg_table = f"iceberg_multicol_{uid}" + + # Iceberg has no unsigned types, so retention widens UInt64 -> Int64; the cast is + # not value-preserving per canBeSafelyCast, hence the lossy opt-in below. + make_rmt(node, mt_table, "id Int64, event_date Date, retention UInt64", + "(event_date, retention)", replica_name="replica1") + node.query( + f"INSERT INTO {mt_table} VALUES " + f"(1, '2024-03-05', 30), (2, '2024-03-05', 30), (3, '2024-03-05', 30)" + ) + + make_iceberg_s3(node, iceberg_table, "id Int64, event_date Date, retention Int64", + partition_by="(event_date, retention)") + + pid = first_partition_id(node, mt_table) + node.query( + f"ALTER TABLE {mt_table} EXPORT PARTITION ID '{pid}' TO TABLE {iceberg_table}", + settings={ + "allow_insert_into_iceberg": 1, + "export_merge_tree_part_allow_lossy_cast": 1, + }, + ) + wait_for_export_status(node, mt_table, iceberg_table, pid, "COMPLETED") + + count = int(node.query(f"SELECT count() FROM {iceberg_table}").strip()) + assert count == 3, f"Expected 3 rows after export, got {count}" + + data_retention = int(node.query( + f"SELECT DISTINCT retention FROM {iceberg_table}" + ).strip()) + assert data_retention == 30, f"Expected exported retention 30, got {data_retention}" + + days = int(node.query( + f"SELECT DISTINCT toInt64(event_date) FROM {iceberg_table}" + ).strip()) + + query_id = f"multicol_{uid}" + node.query( + f"SELECT * FROM {iceberg_table}", + query_id=query_id, + settings={"iceberg_metadata_log_level": "manifest_file_entry"}, + ) + entries = fetch_manifest_entries(node, query_id) + partitions = _data_file_partition_records(entries) + assert partitions, "No data-file partition records found in manifest entries" + + meta_dates = {int(_partition_scalar(p, "event_date")) for p in partitions} + assert meta_dates == {days}, ( + f"Metadata event_date {meta_dates} must equal days-since-epoch {days}." + ) + meta_retentions = {int(_partition_scalar(p, "retention")) for p in partitions} + assert meta_retentions == {30}, ( + f"Metadata retention {meta_retentions} must equal the exported value 30." + ) + + filtered = int(node.query( + f"SELECT count() FROM {iceberg_table} " + f"WHERE event_date = '2024-03-05' AND retention = 30" + ).strip()) + assert filtered == 3, f"Partition-filtered read expected 3 rows, got {filtered}" diff --git a/tests/integration/test_export_replicated_mt_partition_to_object_storage/test.py b/tests/integration/test_export_replicated_mt_partition_to_object_storage/test.py index 93eb070b3876..8ad265a375ad 100644 --- a/tests/integration/test_export_replicated_mt_partition_to_object_storage/test.py +++ b/tests/integration/test_export_replicated_mt_partition_to_object_storage/test.py @@ -1496,6 +1496,86 @@ def test_export_partition_all(cluster): assert row_count == 3, f"Expected 3 rows in S3 after EXPORT PARTITION ALL, got {row_count}" +def test_export_partition_partition_column_castable_type_mismatch(cluster): + """A lossy partition-column cast (year String -> UInt16) is rejected synchronously + when export_merge_tree_part_allow_lossy_cast is off, scheduling nothing.""" + skip_if_remote_database_disk_enabled(cluster) + node = cluster.instances["replica1"] + + postfix = str(uuid.uuid4()).replace("-", "_") + mt_table = f"pkey_cast_mismatch_partition_mt_{postfix}" + s3_table = f"pkey_cast_mismatch_partition_s3_{postfix}" + + # Source: year String; destination: year UInt16. PARTITION BY year on + # both sides — same AST text — to defeat the AST equivalence check. + node.query( + f"CREATE TABLE {mt_table} (id UInt64, year String) " + f"ENGINE = ReplicatedMergeTree('/clickhouse/tables/{mt_table}', 'replica1') " + f"PARTITION BY year " + f"ORDER BY tuple()" + ) + node.query( + f"CREATE TABLE {s3_table} (id UInt64, year UInt16) " + f"ENGINE = S3(s3_conn, filename='{s3_table}', " + f"format=Parquet, partition_strategy='hive') " + f"PARTITION BY year" + ) + + node.query( + f"INSERT INTO {mt_table} VALUES (1, '2020'), (2, '2020'), (3, '2020')" + ) + + # With a String partition column the partition_id is the SipHash of the + # value rather than the textual representation — look it up so we can + # reference the partition explicitly in EXPORT PARTITION ID and in + # subsequent system.replicated_partition_exports queries. + partition_id = node.query( + f"SELECT partition_id FROM system.parts " + f"WHERE database = currentDatabase() AND table = '{mt_table}' " + f" AND active " + f"ORDER BY name LIMIT 1" + ).strip() + assert partition_id, ( + "Expected one active part on the source table after INSERT; " + "system.parts returned nothing." + ) + + error = node.query_and_get_error( + f"ALTER TABLE {mt_table} EXPORT PARTITION ID '{partition_id}' " + f"TO TABLE {s3_table}" + ) + assert "BAD_ARGUMENTS" in error, ( + f"Expected BAD_ARGUMENTS for a lossy partition-column cast, " + f"got: {error!r}" + ) + assert "requires a lossy cast" in error and "'year'" in error, ( + f"Expected the error message to report the lossy cast on column " + f"'year', got: {error!r}" + ) + + # Nothing scheduled: no row in system.replicated_partition_exports. + rows_in_system_view = node.query( + f"SELECT count() FROM system.replicated_partition_exports " + f"WHERE source_table = '{mt_table}' " + f" AND destination_table = '{s3_table}' " + f" AND partition_id = '{partition_id}'" + ).strip() + assert rows_in_system_view == "0", ( + f"Expected no row in system.replicated_partition_exports after a " + f"synchronously-rejected export, got {rows_in_system_view}." + ) + + # Nothing written: no parquet file under any year=*/ partition prefix. + files_in_s3 = node.query( + f"SELECT count() FROM s3(s3_conn, " + f"filename='{s3_table}/year=*/*.parquet', format='One')" + ).strip() + assert files_in_s3 == "0", ( + f"Expected no Parquet files in S3 after a synchronously-rejected " + f"export, found {files_in_s3}." + ) + + def test_export_partition_all_failure_modes(cluster): """Cover the three values of `export_merge_tree_partition_all_on_error`. diff --git a/tests/queries/0_stateless/03572_export_merge_tree_part_to_object_storage_simple.sql b/tests/queries/0_stateless/03572_export_merge_tree_part_to_object_storage_simple.sql index 7cb70af024a2..8200233a7322 100644 --- a/tests/queries/0_stateless/03572_export_merge_tree_part_to_object_storage_simple.sql +++ b/tests/queries/0_stateless/03572_export_merge_tree_part_to_object_storage_simple.sql @@ -1,6 +1,6 @@ -- Tags: no-parallel, no-fasttest -DROP TABLE IF EXISTS 03572_mt_table, 03572_invalid_schema_table, 03572_ephemeral_mt_table, 03572_matching_ephemeral_s3_table; +DROP TABLE IF EXISTS 03572_mt_table, 03572_invalid_schema_table, 03572_ephemeral_mt_table, 03572_matching_ephemeral_s3_table, 03572_partition_type_mismatch_mt, 03572_partition_type_mismatch_s3, 03572_lossy_mt, 03572_lossy_s3, 03572_lossless_mt, 03572_lossless_s3; SET allow_experimental_export_merge_tree_part=1; @@ -9,10 +9,12 @@ CREATE TABLE 03572_mt_table (id UInt64, year UInt16) ENGINE = MergeTree() PARTIT INSERT INTO 03572_mt_table VALUES (1, 2020); -- Create a table with a different partition key and export a partition to it. It should throw +-- on the partition-key AST mismatch (schema compat now follows INSERT SELECT positional semantics, +-- so the column shape matches and the partition-key check is what fires). CREATE TABLE 03572_invalid_schema_table (id UInt64, x UInt16) ENGINE = S3(s3_conn, filename='03572_invalid_schema_table', format='Parquet', partition_strategy='hive') PARTITION BY x; ALTER TABLE 03572_mt_table EXPORT PART '2020_1_1_0' TO TABLE 03572_invalid_schema_table -SETTINGS allow_experimental_export_merge_tree_part = 1; -- {serverError INCOMPATIBLE_COLUMNS} +SETTINGS allow_experimental_export_merge_tree_part = 1; -- {serverError BAD_ARGUMENTS} DROP TABLE 03572_invalid_schema_table; @@ -27,13 +29,42 @@ ALTER TABLE 03572_mt_table EXPORT PART '2020_1_1_0' TO TABLE FUNCTION extractKey -- It is a table function, but the engine does not support exports/imports, should throw ALTER TABLE 03572_mt_table EXPORT PART '2020_1_1_0' TO TABLE FUNCTION url('a.parquet'); -- {serverError NOT_IMPLEMENTED} --- Test that destination table can not have a column that matches the source ephemeral +-- Source-side ephemeral columns are not readable, so the destination must not declare a matching +-- ordinary column or the column count will not align under positional matching. CREATE TABLE 03572_ephemeral_mt_table (id UInt64, year UInt16, name String EPHEMERAL) ENGINE = MergeTree() PARTITION BY year ORDER BY tuple(); CREATE TABLE 03572_matching_ephemeral_s3_table (id UInt64, year UInt16, name String) ENGINE = S3(s3_conn, filename='03572_matching_ephemeral_s3_table', format='Parquet', partition_strategy='hive') PARTITION BY year; INSERT INTO 03572_ephemeral_mt_table (id, year, name) VALUES (1, 2020, 'alice'); -ALTER TABLE 03572_ephemeral_mt_table EXPORT PART '2020_1_1_0' TO TABLE 03572_matching_ephemeral_s3_table; -- {serverError INCOMPATIBLE_COLUMNS} +ALTER TABLE 03572_ephemeral_mt_table EXPORT PART '2020_1_1_0' TO TABLE 03572_matching_ephemeral_s3_table; -- {serverError NUMBER_OF_COLUMNS_DOESNT_MATCH} -DROP TABLE IF EXISTS 03572_mt_table, 03572_invalid_schema_table, 03572_ephemeral_mt_table, 03572_matching_ephemeral_s3_table; +-- Partition columns follow the same lossy-cast gate as any other column (no special +-- exact-type guard). String -> UInt16 is a lossy cast, so with the default +-- export_merge_tree_part_allow_lossy_cast = 0 it is rejected synchronously. +CREATE TABLE 03572_partition_type_mismatch_mt (id UInt64, year String) ENGINE = MergeTree() PARTITION BY year ORDER BY tuple(); +CREATE TABLE 03572_partition_type_mismatch_s3 (id UInt64, year UInt16) ENGINE = S3(s3_conn, filename='03572_partition_type_mismatch_s3', format='Parquet', partition_strategy='hive') PARTITION BY year; + +ALTER TABLE 03572_partition_type_mismatch_mt EXPORT PART '2020_1_1_0' TO TABLE 03572_partition_type_mismatch_s3 +SETTINGS allow_experimental_export_merge_tree_part = 1; -- {serverError BAD_ARGUMENTS} + +CREATE TABLE 03572_lossy_mt (id Int64, year UInt16) ENGINE = MergeTree() PARTITION BY year ORDER BY tuple(); +CREATE TABLE 03572_lossy_s3 (id Int32, year UInt16) ENGINE = S3(s3_conn, filename='03572_lossy_s3', format='Parquet', partition_strategy='hive') PARTITION BY year; + +ALTER TABLE 03572_lossy_mt EXPORT PART '2020_1_1_0' TO TABLE 03572_lossy_s3 +SETTINGS allow_experimental_export_merge_tree_part = 1; -- {serverError BAD_ARGUMENTS} + +-- With the acknowledgment setting enabled, the lossy cast passes validation and reaches the +-- part lookup, which fails because the part does not exist. +ALTER TABLE 03572_lossy_mt EXPORT PART '2020_1_1_0' TO TABLE 03572_lossy_s3 +SETTINGS allow_experimental_export_merge_tree_part = 1, export_merge_tree_part_allow_lossy_cast = 1; -- {serverError NO_SUCH_DATA_PART} + +-- A lossless widening cast (Int32 -> Int64) passes validation without the setting and reaches +-- the part lookup, which fails because the part does not exist. +CREATE TABLE 03572_lossless_mt (id Int32, year UInt16) ENGINE = MergeTree() PARTITION BY year ORDER BY tuple(); +CREATE TABLE 03572_lossless_s3 (id Int64, year UInt16) ENGINE = S3(s3_conn, filename='03572_lossless_s3', format='Parquet', partition_strategy='hive') PARTITION BY year; + +ALTER TABLE 03572_lossless_mt EXPORT PART '2020_1_1_0' TO TABLE 03572_lossless_s3 +SETTINGS allow_experimental_export_merge_tree_part = 1; -- {serverError NO_SUCH_DATA_PART} + +DROP TABLE IF EXISTS 03572_mt_table, 03572_invalid_schema_table, 03572_ephemeral_mt_table, 03572_matching_ephemeral_s3_table, 03572_partition_type_mismatch_mt, 03572_partition_type_mismatch_s3, 03572_lossy_mt, 03572_lossy_s3, 03572_lossless_mt, 03572_lossless_s3; diff --git a/tests/queries/0_stateless/03572_export_replicated_merge_tree_part_to_object_storage_simple.sql b/tests/queries/0_stateless/03572_export_replicated_merge_tree_part_to_object_storage_simple.sql index f8f23532f0a7..0a199755c40a 100644 --- a/tests/queries/0_stateless/03572_export_replicated_merge_tree_part_to_object_storage_simple.sql +++ b/tests/queries/0_stateless/03572_export_replicated_merge_tree_part_to_object_storage_simple.sql @@ -1,16 +1,18 @@ -- Tags: no-parallel, no-fasttest -DROP TABLE IF EXISTS 03572_rmt_table, 03572_invalid_schema_table; +DROP TABLE IF EXISTS 03572_rmt_table, 03572_invalid_schema_table, 03572_rmt_partition_type_mismatch_mt, 03572_rmt_partition_type_mismatch_s3, 03572_rmt_lossy_mt, 03572_rmt_lossy_s3, 03572_rmt_lossless_mt, 03572_rmt_lossless_s3; CREATE TABLE 03572_rmt_table (id UInt64, year UInt16) ENGINE = ReplicatedMergeTree('/clickhouse/{database}/test_03572_rmt/03572_rmt_table', 'replica1') PARTITION BY year ORDER BY tuple(); INSERT INTO 03572_rmt_table VALUES (1, 2020); -- Create a table with a different partition key and export a partition to it. It should throw +-- on the partition-key AST mismatch (schema compat now follows INSERT SELECT positional semantics, +-- so the column shape matches and the partition-key check is what fires). CREATE TABLE 03572_invalid_schema_table (id UInt64, x UInt16) ENGINE = S3(s3_conn, filename='03572_invalid_schema_table', format='Parquet', partition_strategy='hive') PARTITION BY x; ALTER TABLE 03572_rmt_table EXPORT PART '2020_0_0_0' TO TABLE 03572_invalid_schema_table -SETTINGS allow_experimental_export_merge_tree_part = 1; -- {serverError INCOMPATIBLE_COLUMNS} +SETTINGS allow_experimental_export_merge_tree_part = 1; -- {serverError BAD_ARGUMENTS} DROP TABLE 03572_invalid_schema_table; @@ -19,4 +21,33 @@ CREATE TABLE 03572_invalid_schema_table (id UInt64, year UInt16) ENGINE = S3(s3_ ALTER TABLE 03572_rmt_table EXPORT PART '2020_0_0_0' TO TABLE 03572_invalid_schema_table SETTINGS allow_experimental_export_merge_tree_part = 1; -- {serverError NOT_IMPLEMENTED} -DROP TABLE IF EXISTS 03572_rmt_table, 03572_invalid_schema_table; +-- Partition columns follow the same lossy-cast gate as any other column (no special +-- exact-type guard). String -> UInt16 is a lossy cast, so with the default +-- export_merge_tree_part_allow_lossy_cast = 0 it is rejected synchronously. +CREATE TABLE 03572_rmt_partition_type_mismatch_mt (id UInt64, year String) ENGINE = ReplicatedMergeTree('/clickhouse/{database}/test_03572_rmt_pcol_type/03572_rmt_partition_type_mismatch_mt', 'replica1') PARTITION BY year ORDER BY tuple(); +CREATE TABLE 03572_rmt_partition_type_mismatch_s3 (id UInt64, year UInt16) ENGINE = S3(s3_conn, filename='03572_rmt_partition_type_mismatch_s3', format='Parquet', partition_strategy='hive') PARTITION BY year; + +ALTER TABLE 03572_rmt_partition_type_mismatch_mt EXPORT PART '2020_0_0_0' TO TABLE 03572_rmt_partition_type_mismatch_s3 +SETTINGS allow_experimental_export_merge_tree_part = 1; -- {serverError BAD_ARGUMENTS} + +-- A lossy cast on a non-partition column (Int64 -> Int32) is rejected synchronously by default. +CREATE TABLE 03572_rmt_lossy_mt (id Int64, year UInt16) ENGINE = ReplicatedMergeTree('/clickhouse/{database}/test_03572_rmt_lossy/03572_rmt_lossy_mt', 'replica1') PARTITION BY year ORDER BY tuple(); +CREATE TABLE 03572_rmt_lossy_s3 (id Int32, year UInt16) ENGINE = S3(s3_conn, filename='03572_rmt_lossy_s3', format='Parquet', partition_strategy='hive') PARTITION BY year; + +ALTER TABLE 03572_rmt_lossy_mt EXPORT PART '2020_0_0_0' TO TABLE 03572_rmt_lossy_s3 +SETTINGS allow_experimental_export_merge_tree_part = 1; -- {serverError BAD_ARGUMENTS} + +-- With the acknowledgment setting enabled, the lossy cast passes validation and reaches the +-- part lookup, which fails because the part does not exist. +ALTER TABLE 03572_rmt_lossy_mt EXPORT PART '2020_0_0_0' TO TABLE 03572_rmt_lossy_s3 +SETTINGS allow_experimental_export_merge_tree_part = 1, export_merge_tree_part_allow_lossy_cast = 1; -- {serverError NO_SUCH_DATA_PART} + +-- A lossless widening cast (Int32 -> Int64) passes validation without the setting and reaches +-- the part lookup, which fails because the part does not exist. +CREATE TABLE 03572_rmt_lossless_mt (id Int32, year UInt16) ENGINE = ReplicatedMergeTree('/clickhouse/{database}/test_03572_rmt_lossless/03572_rmt_lossless_mt', 'replica1') PARTITION BY year ORDER BY tuple(); +CREATE TABLE 03572_rmt_lossless_s3 (id Int64, year UInt16) ENGINE = S3(s3_conn, filename='03572_rmt_lossless_s3', format='Parquet', partition_strategy='hive') PARTITION BY year; + +ALTER TABLE 03572_rmt_lossless_mt EXPORT PART '2020_0_0_0' TO TABLE 03572_rmt_lossless_s3 +SETTINGS allow_experimental_export_merge_tree_part = 1; -- {serverError NO_SUCH_DATA_PART} + +DROP TABLE IF EXISTS 03572_rmt_table, 03572_invalid_schema_table, 03572_rmt_partition_type_mismatch_mt, 03572_rmt_partition_type_mismatch_s3, 03572_rmt_lossy_mt, 03572_rmt_lossy_s3, 03572_rmt_lossless_mt, 03572_rmt_lossless_s3; From b3c4759f32a9bc5b4b9237141935fdfabb191d19 Mon Sep 17 00:00:00 2001 From: Andrey Zvonov <32552679+zvonand@users.noreply.github.com> Date: Tue, 30 Jun 2026 15:49:32 +0200 Subject: [PATCH 30/43] Merge pull request #1912 from Altinity/rw_lock_for_partition_export Replace the export partition local manifests lock by a multiversion container Source-PR: #1912 (https://github.com/Altinity/ClickHouse/pull/1912) --- .../ExportPartitionManifestUpdatingTask.cpp | 335 +++++++++--------- .../ExportPartitionManifestUpdatingTask.h | 5 + .../ExportPartitionTaskScheduler.cpp | 45 ++- .../MergeTree/ExportPartitionUtils.cpp | 12 + src/Storages/StorageReplicatedMergeTree.cpp | 48 ++- src/Storages/StorageReplicatedMergeTree.h | 12 +- 6 files changed, 251 insertions(+), 206 deletions(-) diff --git a/src/Storages/MergeTree/ExportPartitionManifestUpdatingTask.cpp b/src/Storages/MergeTree/ExportPartitionManifestUpdatingTask.cpp index 8ba8d8fc64a0..6aa9a15cfe0f 100644 --- a/src/Storages/MergeTree/ExportPartitionManifestUpdatingTask.cpp +++ b/src/Storages/MergeTree/ExportPartitionManifestUpdatingTask.cpp @@ -37,9 +37,7 @@ namespace FailPoints namespace { - /// Work item describing a commit-recovery attempt that has been deferred out of - /// the `poll()` critical section. Captures everything by value so it can be - /// executed safely after `export_merge_tree_partition_mutex` has been released. + /// Describes pending commits struct CommitRecoveryWork { ExportReplicatedMergeTreePartitionManifest metadata; @@ -47,17 +45,9 @@ namespace StoragePtr destination_storage; ContextPtr context; }; + /// Fetch all per-replica last_exception leaves under /last_exception and build - /// a fresh map keyed by replica name. The map key prefers the unescaped `replica` field - /// embedded in the JSON payload; if it is missing or empty, the leaf name is unescaped as - /// a fallback. - /// - /// An empty result means "nothing actionable": either the parent getChildren failed (ZK - /// glitch), the container has no children yet (no replica has reported), or every leaf - /// fetch came back ZNONODE / malformed. Callers MUST skip the assignment in that case to - /// preserve the in-memory mirror across transient errors. This is safe because per-replica - /// leaves are never individually removed — the entire entry path is wiped recursively when - /// a task is cleaned up, which is handled separately by removeStaleEntries. + /// a fresh map keyed by replica name. std::map readLastExceptionPerReplica( const zkutil::ZooKeeperPtr & zk, const std::filesystem::path & entry_path, @@ -85,8 +75,6 @@ namespace for (const auto & child : children) paths.emplace_back(container_path / child); - /// One MULTI_READ when supported, parallel async gets otherwise. See - /// ZooKeeper::multiRead in src/Common/ZooKeeper/ZooKeeper.h. ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRequests); ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperGet, paths.size()); auto responses = zk->tryGet(paths); @@ -127,19 +115,8 @@ namespace return out; } - /* - Enforce the PENDING task timeout and recover non-committed exports that have already - exported all parts. Entries are never removed for age — `system.replicated_partition_exports` - is append-only history, so the entry always stays in the in-memory container: a KILLED - transition is driven by the status watch, and a deferred commit is handled by the caller - after the lock is released. - - Side outputs: - - `deferred_commits`: when a PENDING entry has all parts processed but the export was - never committed, this function appends a CommitRecoveryWork item to be executed by - the caller after releasing the storage-wide mutex. The actual commit() call (which - performs network I/O to the destination catalog and S3) MUST NOT run under the lock. - */ + + /// collects pending commits and kills tasks that have timed out void tryCleanup( const zkutil::ZooKeeperPtr & zk, const std::string & entry_path, @@ -158,6 +135,15 @@ namespace if (task_timed_out) { + /// Serialize against commit(): don't kill a task whose commit is in progress. + auto commit_lock = zkutil::EphemeralNodeHolder::tryCreate( + fs::path(entry_path) / "commit_lock", *zk, storage.getReplicaName()); + if (!commit_lock) + { + LOG_INFO(log, "ExportPartition Manifest Updating Task: commit in progress for {}, skipping timeout kill", entry_path); + return; + } + const std::string status_path = fs::path(entry_path) / "status"; Coordination::Stat status_stat; @@ -245,16 +231,7 @@ namespace return; } - /// A replica exported the last part but the commit never landed. Capture everything - /// needed to run commit() outside `export_merge_tree_partition_mutex`. The - /// commit path performs network I/O (REST catalog + S3) with up to - /// MAX_TRANSACTION_RETRIES = 100 retries; holding the storage-wide mutex across - /// that work is what caused `system.replicated_partition_exports` to hang. - /// - /// The outer poll() loop stays on the normal path: it will call addTask() so the - /// in-memory container reflects the PENDING entry. The status watch registered by - /// poll() will transition the local entry to COMPLETED/FAILED once the deferred - /// commit (or a peer's commit) updates /status in ZooKeeper. + /// A replica exported the last part but the commit never landed deferred_commits.push_back(CommitRecoveryWork{ .metadata = metadata, .entry_path = entry_path, @@ -273,49 +250,35 @@ ExportPartitionManifestUpdatingTask::ExportPartitionManifestUpdatingTask(Storage std::vector ExportPartitionManifestUpdatingTask::getPartitionExportsInfo() const { - /// Snapshot just the fields we need under the lock, then build the public - /// ReplicatedPartitionExportInfo vector after releasing it. This keeps the critical - /// section O(entries) of cheap struct copies (manifest + enum) rather than also - /// covering the formatting / Array construction performed by the caller. - struct EntrySnapshot - { - ExportReplicatedMergeTreePartitionManifest manifest; - ExportReplicatedMergeTreePartitionTaskEntry::Status status; - std::map last_exception_per_replica; - }; - - std::vector snapshots; + const auto model = storage.export_partition_manifests.get(); - { - std::lock_guard lock(storage.export_merge_tree_partition_mutex); - - snapshots.reserve(storage.export_merge_tree_partition_task_entries_by_key.size()); - for (const auto & entry : storage.export_merge_tree_partition_task_entries_by_key) - snapshots.push_back(EntrySnapshot{entry.manifest, entry.status, entry.last_exception_per_replica}); - } + if (!model) + return {}; std::vector infos; - infos.reserve(snapshots.size()); + infos.reserve(model->size()); - for (auto & snapshot : snapshots) + for (const auto & entry : model->get()) { + const auto & manifest = entry.manifest; + ReplicatedPartitionExportInfo info; - info.destination_database = snapshot.manifest.destination_database; - info.destination_table = snapshot.manifest.destination_table; - info.partition_id = snapshot.manifest.partition_id; - info.transaction_id = snapshot.manifest.transaction_id; - info.query_id = snapshot.manifest.query_id; - info.create_time = snapshot.manifest.create_time; - info.source_replica = snapshot.manifest.source_replica; - info.parts_count = snapshot.manifest.number_of_parts; - info.parts_to_do = snapshot.manifest.parts.size(); - info.parts = std::move(snapshot.manifest.parts); - info.status = magic_enum::enum_name(snapshot.status); - - info.last_exception_per_replica.reserve(snapshot.last_exception_per_replica.size()); + info.destination_database = manifest.destination_database; + info.destination_table = manifest.destination_table; + info.partition_id = manifest.partition_id; + info.transaction_id = manifest.transaction_id; + info.query_id = manifest.query_id; + info.create_time = manifest.create_time; + info.source_replica = manifest.source_replica; + info.parts_count = manifest.number_of_parts; + info.parts_to_do = manifest.parts.size(); + info.parts = manifest.parts; + info.status = magic_enum::enum_name(entry.status); + + info.last_exception_per_replica.reserve(entry.last_exception_per_replica.size()); size_t total_exception_count = 0; - for (const auto & [_, ex] : snapshot.last_exception_per_replica) + for (const auto & [_, ex] : entry.last_exception_per_replica) { total_exception_count += ex.count; info.last_exception_per_replica.push_back(ex); @@ -345,10 +308,7 @@ void ExportPartitionManifestUpdatingTask::poll() /// across replicas: only the replica holding it walks `tryCleanup` (task-timeout /// enforcement + commit recovery). It MUST outlive the deferred-commit loop below; otherwise a peer /// replica's next poll() could acquire it and race us on the same commit-recovery work, - /// duplicating REST-catalog round-trips and snapshot writes. The EphemeralNodeHolder - /// destructor removes the node, so we declare it at function scope and let it die - /// at the end of poll() - after all deferred commits have completed. - /// Acquired here (no mutex needed - it is just a ZK ephemeral create). + /// duplicating REST-catalog round-trips and snapshot writes. auto cleanup_lock = zkutil::EphemeralNodeHolder::tryCreate(cleanup_lock_path, *zk, storage.replica_name); if (cleanup_lock) { @@ -356,9 +316,20 @@ void ExportPartitionManifestUpdatingTask::poll() } { - std::lock_guard lock(storage.export_merge_tree_partition_mutex); + /// M_task: serializes poll() vs handleStatusChanges(). We copy the current read-model into a + /// private mutable container, mutate that copy across the ZooKeeper reads below, and publish + /// it atomically via export_read_model.set() at the end. Readers never see partial updates. + std::lock_guard task_guard(background_task_serialization_mutex); + + const auto current_model = storage.export_partition_manifests.get(); - LOG_INFO(storage.log, "ExportPartition Manifest Updating Task: Polling for new entries for table {}. Current number of entries: {}", storage.getStorageID().getNameForLogs(), storage.export_merge_tree_partition_task_entries_by_key.size()); + auto working_model = current_model + ? std::make_unique(*current_model) + : std::make_unique(); + + auto & entries_by_key = working_model->get(); + + LOG_INFO(storage.log, "ExportPartition Manifest Updating Task: Polling for new entries for table {}. Current number of entries: {}", storage.getStorageID().getNameForLogs(), entries_by_key.size()); ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRequests); ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperGetChildrenWatch); @@ -369,8 +340,6 @@ void ExportPartitionManifestUpdatingTask::poll() const auto now = time(nullptr); - auto & entries_by_key = storage.export_merge_tree_partition_task_entries_by_key; - /// Load new entries /// If we have the cleanup lock, also remove stale entries from zk and local /// Upload dangling commit files if any @@ -389,50 +358,49 @@ void ExportPartitionManifestUpdatingTask::poll() const auto metadata = ExportReplicatedMergeTreePartitionManifest::fromJsonString(metadata_json); - /// Read last_exception leaves (no watch). Surfacing exceptions in the system table relies - /// on this read being part of every poll cycle: per-part failures during PENDING do not - /// trigger a status watch, so the only refresh path while the task is still in-flight is - /// the periodic poll. An empty result collapses every "nothing actionable" case - /// (transient ZK error, no children, all leaves ZNONODE/malformed) into a no-op so the - /// in-memory copy stays intact. auto last_exception_per_replica = readLastExceptionPerReplica( zk, fs::path(entry_path), key, storage.log.load()); - const auto local_entry = entries_by_key.find(key); - /// If the zk entry has been replaced with export_merge_tree_partition_force_export, checking only for the export key is not enough /// we need to make sure it is the same transaction id. If it is not, it needs to be replaced. - bool has_local_entry_and_is_up_to_date = local_entry != entries_by_key.end() + const auto local_entry = entries_by_key.find(key); + const bool has_local_entry = local_entry != entries_by_key.end() && local_entry->manifest.transaction_id == metadata.transaction_id; - /// If the entry is up to date and we don't have the cleanup lock, refresh the in-memory - /// last_exception (surfaced by system.replicated_partition_exports) and early exit. - /// Direct mutation of the `mutable` field is safe under export_merge_tree_partition_mutex, - /// which is held throughout poll(). - if (!cleanup_lock && has_local_entry_and_is_up_to_date) - { - if (!last_exception_per_replica.empty()) - local_entry->last_exception_per_replica = std::move(last_exception_per_replica); - continue; - } + std::string status_string; - std::weak_ptr weak_manifest_updater = storage.export_merge_tree_partition_manifest_updater; + /// In theory, we should be notified when the status changes by the status watch + /// but in practice, the watch is not always reliable (e.g. if the ZooKeeper session is lost) + /// so we need to read the status from the ZK node directly. + if (has_local_entry) + { + ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRequests); + ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperGet); - auto status_watch_callback = std::make_shared([weak_manifest_updater, key](const Coordination::WatchResponse &) + zk->tryGet(fs::path(entry_path) / "status", status_string); + } + else { - /// If the table is dropped but the watch is not removed, we need to prevent use after free - /// below code assumes that if manifest updater is still alive, the status handling task is also alive - if (auto manifest_updater = weak_manifest_updater.lock()) + /// If we don't have a local entry, we need to arm a status watch to be notified when the status changes + std::weak_ptr weak_manifest_updater = storage.export_merge_tree_partition_manifest_updater; + auto status_watch_callback = std::make_shared([weak_manifest_updater, key](const Coordination::WatchResponse &) { - manifest_updater->addStatusChange(key); - manifest_updater->storage.export_merge_tree_partition_status_handling_task->schedule(); - } - }); + /// If the table is dropped but the watch is not removed, we need to prevent use after free + /// below code assumes that if manifest updater is still alive, the status handling task is also alive + if (auto manifest_updater = weak_manifest_updater.lock()) + { + manifest_updater->addStatusChange(key); + manifest_updater->storage.export_merge_tree_partition_status_handling_task->schedule(); + } + }); - ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRequests); - ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperGetWatch); - std::string status_string; - if (!zk->tryGetWatch(fs::path(entry_path) / "status", status_string, nullptr, status_watch_callback)) + ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRequests); + ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperGetWatch); + + zk->tryGetWatch(fs::path(entry_path) / "status", status_string, nullptr, status_watch_callback); + } + + if (status_string.empty()) { LOG_INFO(storage.log, "ExportPartition Manifest Updating Task: Skipping {}: missing status", key); continue; @@ -461,42 +429,53 @@ void ExportPartitionManifestUpdatingTask::poll() deferred_commits); } - if (has_local_entry_and_is_up_to_date) + if (!has_local_entry) { - /// Same refresh as the early-exit branch above; we also reach this point when - /// holding the cleanup lock (cleanup did not consume the entry). - if (!last_exception_per_replica.empty()) - local_entry->last_exception_per_replica = std::move(last_exception_per_replica); - LOG_INFO(storage.log, "ExportPartition Manifest Updating Task: Skipping {}: already exists", key); + addTask(metadata, *status, std::move(last_exception_per_replica), key, entries_by_key); + LOG_INFO(storage.log, "ExportPartition Manifest Updating Task: Added new entry for task {}", key); continue; } - addTask(metadata, *status, std::move(last_exception_per_replica), key, entries_by_key); + /// If we already have the local entry, we need to update it if the status has changed or if there are new last exceptions. + const bool status_changed = local_entry->status != *status; + if (!last_exception_per_replica.empty() || status_changed) + { + if (!last_exception_per_replica.empty()) + local_entry->last_exception_per_replica = std::move(last_exception_per_replica); + if (status_changed) + { + local_entry->status = *status; + if (local_entry->status != ExportReplicatedMergeTreePartitionTaskEntry::Status::PENDING) + { + /// terminal now - we no longer need to keep the data parts alive + local_entry->part_references.clear(); + + /// looks like we missed a status change event, we should kill local operations. + if (local_entry->status == ExportReplicatedMergeTreePartitionTaskEntry::Status::KILLED) + { + storage.killExportPart(local_entry->manifest.transaction_id); + } + } + } + } + LOG_INFO(storage.log, "ExportPartition Manifest Updating Task: Skipping {}: already exists", key); + } - /// Remove entries that were deleted by someone else removeStaleEntries(zk_children, entries_by_key); - LOG_INFO(storage.log, "ExportPartition Manifest Updating task: finished polling for new entries. Number of entries: {}", entries_by_key.size()); + const auto entries_count = entries_by_key.size(); + + /// Publish the updated copy atomically. `working_model` is moved out here, so + /// `entries_by_key` (a reference into it) must not be used afterwards. + storage.export_partition_manifests.set(std::move(working_model)); + + LOG_INFO(storage.log, "ExportPartition Manifest Updating task: finished polling for new entries. Number of entries: {}", entries_count); } - /// `export_merge_tree_partition_mutex` released here. Everything below runs without it - /// so concurrent readers of `system.replicated_partition_exports` and other writers are - /// not blocked by the (potentially slow) catalog round-trips below. - /// - /// `cleanup_lock` (the ZK ephemeral node) is INTENTIONALLY still held here and is only - /// destructed at end of function. This preserves the existing cross-replica invariant: - /// at any moment only one replica is performing commit recovery for a given table, so - /// peer replicas will not race us on the same `commit()` calls below. - /// - /// Shutdown safety: this function runs on a BackgroundSchedulePool task that - /// `StorageReplicatedMergeTree::shutdown()` deactivates before clearing the entry - /// container. Deactivation waits for the currently-running invocation (this very call) - /// to return before proceeding, so the deferred commits below complete (or throw) before - /// any teardown observes empty `export_merge_tree_partition_task_entries`. All work - /// items capture their inputs by value, so they are independent from container state. const auto log_ptr = storage.log.load(); + /// Execute pending commits for (const auto & work : deferred_commits) { /// A replica exported the last part but the commit never landed. Try to fix it. @@ -511,10 +490,7 @@ void ExportPartitionManifestUpdatingTask::poll() "Caught exception while committing export for {}: {}", work.entry_path, e.message()); - /// Bump commit-attempts counter; transition to FAILED once the budget is exhausted. - /// This is the primary retry path for the commit phase — handlePartExportSuccess - /// only fires once (on the last part's completion); subsequent retries come from here. - const bool became_failed = ExportPartitionUtils::handleCommitFailure( + const bool exceeded_commimt_max_retries = ExportPartitionUtils::handleCommitFailure( zk, work.entry_path, work.metadata.max_retries, @@ -522,7 +498,7 @@ void ExportPartitionManifestUpdatingTask::poll() e.message(), log_ptr); - if (became_failed) + if (exceeded_commimt_max_retries) { LOG_WARNING(log_ptr, "ExportPartition Manifest Updating Task: " @@ -560,7 +536,7 @@ void ExportPartitionManifestUpdatingTask::addTask( } } - /// Insert or update entry. The multi_index container automatically maintains both indexes. + /// Called from poll() under M_task (sole mutator), so no extra locking is required. ExportReplicatedMergeTreePartitionTaskEntry entry {metadata, status, std::move(part_references), std::move(last_exception_per_replica)}; auto it = entries_by_key.find(key); if (it != entries_by_key.end()) @@ -576,19 +552,18 @@ void ExportPartitionManifestUpdatingTask::removeStaleEntries( { for (auto it = entries_by_key.begin(); it != entries_by_key.end();) { - const auto & key = it->getCompositeKey(); + const auto key = it->getCompositeKey(); if (zk_children.contains(key)) { ++it; continue; } - const auto & transaction_id = it->manifest.transaction_id; - LOG_INFO(storage.log, "ExportPartition Manifest Updating Task: Export task {} was deleted, calling killExportPartition for transaction {}", key, transaction_id); - + LOG_INFO(storage.log, "ExportPartition Manifest Updating Task: Export task {} was deleted, calling killExportPartition for transaction {}", key, it->manifest.transaction_id); + try { - storage.killExportPart(transaction_id); + storage.killExportPart(it->manifest.transaction_id); } catch (...) { @@ -607,20 +582,35 @@ void ExportPartitionManifestUpdatingTask::addStatusChange(const std::string & ke void ExportPartitionManifestUpdatingTask::handleStatusChanges() { - /// copy the events to a local queue to avoid holding the status_changes_mutex while also holding export_merge_tree_partition_mutex + /// copy the events to a local queue to avoid holding status_changes_mutex under M_task std::queue local_status_changes; { std::lock_guard lock(status_changes_mutex); std::swap(status_changes, local_status_changes); } + /// Take a snapshot of all status changes. If an exception is thrown, we will requeue the whole batch. + const std::queue batch = local_status_changes; + try { - std::lock_guard task_entries_lock(storage.export_merge_tree_partition_mutex); + /// M_task: serializes this against poll(). We copy the current read-model into a private + /// mutable container, apply this batch's status transitions to that copy across the ZooKeeper + /// reads below, and publish it atomically via export_read_model.set() at the end. Readers + /// never see partial updates. + std::lock_guard task_guard(background_task_serialization_mutex); auto zk = storage.getZooKeeper(); + const bool had_changes = !local_status_changes.empty(); + LOG_INFO(storage.log, "ExportPartition Manifest Updating task: handling status changes. Number of status changes: {}", local_status_changes.size()); + const auto current_model = storage.export_partition_manifests.get(); + auto working_model = current_model + ? std::make_unique(*current_model) + : std::make_unique(); + auto & entries_by_key = working_model->get(); + while (!local_status_changes.empty()) { const auto & key = local_status_changes.front(); @@ -632,8 +622,8 @@ void ExportPartitionManifestUpdatingTask::handleStatusChanges() "Failpoint: simulating exception during status change handling for key {}", key); }); - auto it = storage.export_merge_tree_partition_task_entries_by_key.find(key); - if (it == storage.export_merge_tree_partition_task_entries_by_key.end()) + const auto it = entries_by_key.find(key); + if (it == entries_by_key.end()) { local_status_changes.pop(); continue; @@ -660,20 +650,10 @@ void ExportPartitionManifestUpdatingTask::handleStatusChanges() LOG_INFO(storage.log, "ExportPartition Manifest Updating task: status changed for task {}. New status: {}", key, magic_enum::enum_name(*new_status).data()); - /// Refresh last_exception leaves too. Status transitions to FAILED (via commit budget) - /// and KILLED (via timeout) atomically write a per-replica leaf in the same multi, so - /// reading them here ensures the system table surfaces the cause together with the - /// visible state change. No new watch is added — this piggybacks on the existing - /// status watch. An empty result means "nothing actionable" and leaves the previous - /// snapshot intact. - if (auto fetched = readLastExceptionPerReplica( - zk, fs::path(storage.zookeeper_path) / "exports" / key, key, storage.log.load()); - !fetched.empty()) - { - it->last_exception_per_replica = std::move(fetched); - } + auto fetched = readLastExceptionPerReplica( + zk, fs::path(storage.zookeeper_path) / "exports" / key, key, storage.log.load()); - /// If status changed to KILLED, cancel local export operations + /// If status changed to KILLED, cancel local export operations. if (*new_status == ExportReplicatedMergeTreePartitionTaskEntry::Status::KILLED) { try @@ -687,6 +667,10 @@ void ExportPartitionManifestUpdatingTask::handleStatusChanges() } } + /// Apply the in-memory updates directly (poll() cannot run concurrently under M_task). + if (!fetched.empty()) + it->last_exception_per_replica = std::move(fetched); + it->status = *new_status; if (it->status != ExportReplicatedMergeTreePartitionTaskEntry::Status::PENDING) @@ -697,31 +681,34 @@ void ExportPartitionManifestUpdatingTask::handleStatusChanges() local_status_changes.pop(); } + + /// Publish this batch's status transitions to readers. `working_model` is moved out here, + /// so `entries_by_key` (a reference into it) must not be used afterwards. + if (had_changes) + storage.export_partition_manifests.set(std::move(working_model)); } catch (...) { tryLogCurrentException(storage.log, __PRETTY_FUNCTION__); - LOG_INFO(storage.log, "ExportPartition Manifest Updating task: exception thrown while handling status changes, enqueuing remaining status changes back to the status_changes queue. Number of remaining status changes: {}", local_status_changes.size()); + LOG_INFO(storage.log, "ExportPartition Manifest Updating task: exception thrown while handling status changes; nothing was published, requeuing the whole batch. Batch size: {}", batch.size()); std::lock_guard lock(status_changes_mutex); - /// It is possible that an exception is thrown while handling the status. In this scenario - /// we need to enqueue the remaining status changes back to the status_changes queue not to lose them. - /// The other solution to this problem would be to ignore it and schedule a poll - maybe it is simpler? - if (!local_status_changes.empty()) + /// upon exception, requeue the whole batch + if (!batch.empty()) { - // Prepend remaining items before any newly-arrived items + std::queue requeued = batch; while (!status_changes.empty()) { - local_status_changes.push(std::move(status_changes.front())); + requeued.push(std::move(status_changes.front())); status_changes.pop(); } - std::swap(status_changes, local_status_changes); + std::swap(status_changes, requeued); } - LOG_INFO(storage.log, "ExportPartition Manifest Updating task: The new number of pending status after enqueueing unprocessed ones is {}", status_changes.size()); + LOG_INFO(storage.log, "ExportPartition Manifest Updating task: pending status changes after requeue: {}", status_changes.size()); throw; } diff --git a/src/Storages/MergeTree/ExportPartitionManifestUpdatingTask.h b/src/Storages/MergeTree/ExportPartitionManifestUpdatingTask.h index 32487f2dc68c..3bb4e7ac92ab 100644 --- a/src/Storages/MergeTree/ExportPartitionManifestUpdatingTask.h +++ b/src/Storages/MergeTree/ExportPartitionManifestUpdatingTask.h @@ -45,6 +45,11 @@ class ExportPartitionManifestUpdatingTask std::mutex status_changes_mutex; std::queue status_changes; + + /// M_task: serializes poll() and handleStatusChanges(). Each builds a private mutable copy of + /// the current read-model, mutates it, and atomically publishes it via export_read_model.set(). + /// Held across ZooKeeper I/O; no reader takes it (readers use export_read_model.get()). + std::mutex background_task_serialization_mutex; }; } diff --git a/src/Storages/MergeTree/ExportPartitionTaskScheduler.cpp b/src/Storages/MergeTree/ExportPartitionTaskScheduler.cpp index d722fb77b2c0..89b97460ec08 100644 --- a/src/Storages/MergeTree/ExportPartitionTaskScheduler.cpp +++ b/src/Storages/MergeTree/ExportPartitionTaskScheduler.cpp @@ -77,12 +77,17 @@ void ExportPartitionTaskScheduler::run() const uint32_t seed = uint32_t(std::hash{}(storage.replica_name)) ^ uint32_t(scheduled_exports_count); pcg64_fast rng(seed); - std::lock_guard lock(storage.export_merge_tree_partition_mutex); + /// Hold the published snapshot for the whole pass and iterate it directly (sorted by + /// create_time). It is immutable and the shared_ptr copy never blocks the writer. The scheduler + /// is a pure reader; status converges via the status watch -> handleStatusChanges and poll(). + const auto model = storage.export_partition_manifests.get(); + if (!model) + return; auto zk = storage.getZooKeeper(); // Iterate sorted by create_time - for (auto & entry : storage.export_merge_tree_partition_task_entries_by_create_time) + for (const auto & entry : model->get()) { if (scheduled_exports_count >= available_move_executors) { @@ -90,11 +95,6 @@ void ExportPartitionTaskScheduler::run() break; } - const auto & manifest = entry.manifest; - const auto key = entry.getCompositeKey(); - const auto database = storage.getContext()->resolveDatabase(manifest.destination_database); - const auto & table = manifest.destination_table; - /// No need to query zk for status if the local one is not PENDING if (entry.status != ExportReplicatedMergeTreePartitionTaskEntry::Status::PENDING) { @@ -102,6 +102,11 @@ void ExportPartitionTaskScheduler::run() continue; } + const auto & manifest = entry.manifest; + const auto key = entry.getCompositeKey(); + const auto database = storage.getContext()->resolveDatabase(manifest.destination_database); + const auto & table = manifest.destination_table; + const auto destination_storage_id = StorageID(QualifiedTableName {database, table}); const auto destination_storage = DatabaseCatalog::instance().tryGetTable(destination_storage_id, storage.getContext()); @@ -131,15 +136,14 @@ void ExportPartitionTaskScheduler::run() if (status_in_zk.value() != ExportReplicatedMergeTreePartitionTaskEntry::Status::PENDING) { - entry.status = status_in_zk.value(); - LOG_INFO(storage.log, "ExportPartition scheduler task: Skipping {}... Status from zk is {}", key, magic_enum::enum_name(entry.status).data()); + LOG_INFO(storage.log, "ExportPartition scheduler task: Skipping {}... Status from zk is {}", key, magic_enum::enum_name(status_in_zk.value()).data()); continue; } ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRequests); ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperGetChildren); std::vector parts_in_processing_or_pending; - + if (Coordination::Error::ZOK != zk->tryGetChildren(fs::path(storage.zookeeper_path) / "exports" / key / "processing", parts_in_processing_or_pending)) { LOG_INFO(storage.log, "ExportPartition scheduler task: Failed to get parts in processing or pending, skipping"); @@ -392,6 +396,25 @@ void ExportPartitionTaskScheduler::handlePartExportFailure( return; } + const std::string status_path = export_path / "status"; + Coordination::Stat status_stat; + std::string current_status; + + ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRequests); + ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperGet); + if (!zk->tryGet(status_path, current_status, &status_stat)) + { + LOG_INFO(storage.log, "ExportPartition scheduler task: /status missing for {}, skipping failure bookkeeping", export_path.string()); + return; + } + + const auto status = magic_enum::enum_cast(current_status); + if (!status || *status != ExportReplicatedMergeTreePartitionTaskEntry::Status::PENDING) + { + LOG_INFO(storage.log, "ExportPartition scheduler task: /status for {} is {} (not PENDING), skipping failure bookkeeping", export_path.string(), current_status); + return; + } + Coordination::Requests ops; const auto processing_part_path = processing_parts_path / part_name; @@ -422,7 +445,7 @@ void ExportPartitionTaskScheduler::handlePartExportFailure( processing_part_entry.status = ExportReplicatedMergeTreePartitionProcessingPartEntry::Status::FAILED; processing_part_entry.finished_by = storage.replica_name; - ops.emplace_back(zkutil::makeSetRequest(export_path / "status", String(magic_enum::enum_name(ExportReplicatedMergeTreePartitionTaskEntry::Status::FAILED)).data(), -1)); + ops.emplace_back(zkutil::makeSetRequest(status_path, String(magic_enum::enum_name(ExportReplicatedMergeTreePartitionTaskEntry::Status::FAILED)).data(), status_stat.version)); LOG_INFO(storage.log, "ExportPartition scheduler task: Retry count limit exceeded for part {}, will try to fail the entire task", part_name); } else diff --git a/src/Storages/MergeTree/ExportPartitionUtils.cpp b/src/Storages/MergeTree/ExportPartitionUtils.cpp index 679e07d0b132..622fa6af92df 100644 --- a/src/Storages/MergeTree/ExportPartitionUtils.cpp +++ b/src/Storages/MergeTree/ExportPartitionUtils.cpp @@ -200,6 +200,18 @@ namespace ExportPartitionUtils } LOG_INFO(log, "ExportPartition: commit_lock for {} acquired by replica {}", entry_path, replica_name); + /// Honor a concurrent KILL: commit_lock serializes us against killExportPartition, + /// so a non-PENDING status here means cancel won the race. + std::string status_str; + if (!zk->tryGet(fs::path(entry_path) / "status", status_str)) + return; + const auto status = magic_enum::enum_cast(status_str); + if (!status || *status != ExportReplicatedMergeTreePartitionTaskEntry::Status::PENDING) + { + LOG_INFO(log, "ExportPartition: {} not PENDING, skipping commit", entry_path); + return; + } + const auto exported_paths = ExportPartitionUtils::getExportedPaths(log, zk, entry_path); if (exported_paths.empty()) diff --git a/src/Storages/StorageReplicatedMergeTree.cpp b/src/Storages/StorageReplicatedMergeTree.cpp index 6d6bf12c0375..049bcaf7fcfe 100644 --- a/src/Storages/StorageReplicatedMergeTree.cpp +++ b/src/Storages/StorageReplicatedMergeTree.cpp @@ -498,9 +498,7 @@ StorageReplicatedMergeTree::StorageReplicatedMergeTree( , merge_strategy_picker(*this) , queue(*this, merge_strategy_picker) , fetcher(*this) - , export_merge_tree_partition_task_entries_by_key(export_merge_tree_partition_task_entries.get()) - , export_merge_tree_partition_task_entries_by_transaction_id(export_merge_tree_partition_task_entries.get()) - , export_merge_tree_partition_task_entries_by_create_time(export_merge_tree_partition_task_entries.get()) + , export_partition_manifests(std::make_unique()) , cleanup_thread(*this) , deduplication_hashes_cache(*this, "deduplication_hashes") , async_block_ids_cache(*this, "async_blocks") @@ -6272,10 +6270,7 @@ void StorageReplicatedMergeTree::shutdown(bool) std::lock_guard lock(data_parts_exchange_ptr->rwlock); } - { - std::lock_guard lock(export_merge_tree_partition_mutex); - export_merge_tree_partition_task_entries.clear(); - } + export_partition_manifests.set(std::make_unique()); { std::lock_guard lock(export_manifests_mutex); @@ -10295,8 +10290,20 @@ CancellationCode StorageReplicatedMergeTree::killExportPartition(const String & /// Called from a query thread (KILL EXPORT PARTITION via InterpreterKillQueryQuery), which does not have a component set. auto component_guard = Coordination::setCurrentComponent("StorageReplicatedMergeTree::killExportPartition"); + /// KILL is serialized against the commit phase via commit_lock (see below), so a kill that + /// succeeds cannot be overwritten by a concurrent commit. + auto try_set_status_to_killed = [this](const zkutil::ZooKeeperPtr & zk, const std::string & status_path) { + /// Serialize against commit(): if a commit holds the lock, it is too late to cancel. + auto commit_lock = zkutil::EphemeralNodeHolder::tryCreate( + fs::path(status_path).parent_path() / "commit_lock", *zk, replica_name); + if (!commit_lock) + { + LOG_INFO(log, "Commit in progress, can not cancel export partition task"); + return CancellationCode::CancelCannotBeSent; + } + Coordination::Stat stat; std::string status_from_zk_string; @@ -10330,23 +10337,38 @@ CancellationCode StorageReplicatedMergeTree::killExportPartition(const String & return CancellationCode::CancelSent; }; - std::lock_guard lock(export_merge_tree_partition_mutex); - const auto zk = getZooKeeper(); + /// Read the published snapshot (shared_ptr copy, no lock, no ZooKeeper). The KILLED status set + /// below propagates back into the mirror via the status watch -> handleStatusChanges. + bool local_entry_found = false; + bool local_entry_pending = false; + std::string local_composite_key; + + if (const auto model = export_partition_manifests.get()) + { + const auto & by_transaction_id = model->get(); + const auto entry = by_transaction_id.find(transaction_id); + if (entry != by_transaction_id.end()) + { + local_entry_found = true; + local_entry_pending = entry->status == ExportReplicatedMergeTreePartitionTaskEntry::Status::PENDING; + local_composite_key = entry->getCompositeKey(); + } + } + /// if we have the entry locally, no need to list from zk. we can save some requests. - const auto & entry = export_merge_tree_partition_task_entries_by_transaction_id.find(transaction_id); - if (entry != export_merge_tree_partition_task_entries_by_transaction_id.end()) + if (local_entry_found) { LOG_INFO(log, "Export partition task found locally, trying to cancel it"); /// found locally, no need to get children on zk - if (entry->status != ExportReplicatedMergeTreePartitionTaskEntry::Status::PENDING) + if (!local_entry_pending) { LOG_INFO(log, "Export partition task is not pending, can not cancel it"); return CancellationCode::CancelCannotBeSent; } - return try_set_status_to_killed(zk, fs::path(zookeeper_path) / "exports" / entry->getCompositeKey() / "status"); + return try_set_status_to_killed(zk, fs::path(zookeeper_path) / "exports" / local_composite_key / "status"); } else { diff --git a/src/Storages/StorageReplicatedMergeTree.h b/src/Storages/StorageReplicatedMergeTree.h index 7487b3f0a125..73cbb0e1f594 100644 --- a/src/Storages/StorageReplicatedMergeTree.h +++ b/src/Storages/StorageReplicatedMergeTree.h @@ -36,6 +36,7 @@ #include #include #include +#include #include #include #include @@ -527,16 +528,11 @@ class StorageReplicatedMergeTree final : public MergeTreeData Coordination::WatchCallbackPtr export_merge_tree_partition_watch_callback; - std::mutex export_merge_tree_partition_mutex; - BackgroundSchedulePoolTaskHolder export_merge_tree_partition_select_task; - ExportPartitionTaskEntriesContainer export_merge_tree_partition_task_entries; - - // Convenience references to indexes - ExportPartitionTaskEntriesContainer::index::type & export_merge_tree_partition_task_entries_by_key; - ExportPartitionTaskEntriesContainer::index::type & export_merge_tree_partition_task_entries_by_transaction_id; - ExportPartitionTaskEntriesContainer::index::type & export_merge_tree_partition_task_entries_by_create_time; + /// Immutable snapshot republished after each writer batch (part_references stripped). Readers + /// (system table, scheduler, KILL) get() a consistent version with no lock and no ZooKeeper. + MultiVersion export_partition_manifests; /// A thread that removes old parts, log entries, and blocks. ReplicatedMergeTreeCleanupThread cleanup_thread; From 799819f2eff04bf8fcd3dd25b6c11fbaa7191f94 Mon Sep 17 00:00:00 2001 From: Arthur Passos Date: Fri, 3 Jul 2026 09:12:11 -0300 Subject: [PATCH 31/43] Merge pull request #1990 from Altinity/preserve_few_settings_to_partition_export Preserve a few settings to partition export Source-PR: #1990 (https://github.com/Altinity/ClickHouse/pull/1990) --- .../ExportReplicatedMergeTreePartitionManifest.h | 13 +++++++++++++ src/Storages/MergeTree/ExportPartitionUtils.cpp | 4 ++++ src/Storages/StorageReplicatedMergeTree.cpp | 8 ++++++++ 3 files changed, 25 insertions(+) diff --git a/src/Storages/ExportReplicatedMergeTreePartitionManifest.h b/src/Storages/ExportReplicatedMergeTreePartitionManifest.h index 4e28e5e4f505..3c2b5da61a18 100644 --- a/src/Storages/ExportReplicatedMergeTreePartitionManifest.h +++ b/src/Storages/ExportReplicatedMergeTreePartitionManifest.h @@ -175,6 +175,10 @@ struct ExportReplicatedMergeTreePartitionManifest bool write_full_path_in_iceberg_metadata = false; bool allow_lossy_cast = false; String iceberg_metadata_json; + String parquet_compression_method; + UInt64 output_format_compression_level; + UInt64 parquet_row_group_size; + UInt64 parquet_row_group_size_bytes; std::string toJsonString() const { @@ -208,6 +212,10 @@ struct ExportReplicatedMergeTreePartitionManifest json.set("task_timeout_seconds", task_timeout_seconds); json.set("write_full_path_in_iceberg_metadata", write_full_path_in_iceberg_metadata); json.set("allow_lossy_cast", allow_lossy_cast); + json.set("parquet_compression_method", parquet_compression_method); + json.set("output_format_compression_level", output_format_compression_level); + json.set("parquet_row_group_size", parquet_row_group_size); + json.set("parquet_row_group_size_bytes", parquet_row_group_size_bytes); std::ostringstream oss; // STYLE_CHECK_ALLOW_STD_STRING_STREAM oss.exceptions(std::ios::failbit); Poco::JSON::Stringifier::stringify(json, oss); @@ -266,6 +274,11 @@ struct ExportReplicatedMergeTreePartitionManifest /// on upgrade. New tasks always persist the initiator's actual choice. manifest.allow_lossy_cast = json->has("allow_lossy_cast") ? json->getValue("allow_lossy_cast") : true; + manifest.parquet_compression_method = json->getValue("parquet_compression_method"); + manifest.output_format_compression_level = json->getValue("output_format_compression_level"); + manifest.parquet_row_group_size = json->getValue("parquet_row_group_size"); + manifest.parquet_row_group_size_bytes = json->getValue("parquet_row_group_size_bytes"); + return manifest; } }; diff --git a/src/Storages/MergeTree/ExportPartitionUtils.cpp b/src/Storages/MergeTree/ExportPartitionUtils.cpp index 622fa6af92df..c9fe33744b12 100644 --- a/src/Storages/MergeTree/ExportPartitionUtils.cpp +++ b/src/Storages/MergeTree/ExportPartitionUtils.cpp @@ -81,6 +81,10 @@ namespace ExportPartitionUtils context_copy->setCurrentQueryId(manifest.query_id); context_copy->setSetting("output_format_parallel_formatting", manifest.parallel_formatting); context_copy->setSetting("output_format_parquet_parallel_encoding", manifest.parquet_parallel_encoding); + context_copy->setSetting("output_format_parquet_compression_method", manifest.parquet_compression_method); + context_copy->setSetting("output_format_compression_level", manifest.output_format_compression_level); + context_copy->setSetting("output_format_parquet_row_group_size", manifest.parquet_row_group_size); + context_copy->setSetting("output_format_parquet_row_group_size_bytes", manifest.parquet_row_group_size_bytes); context_copy->setSetting("max_threads", manifest.max_threads); context_copy->setSetting("export_merge_tree_part_file_already_exists_policy", String(magic_enum::enum_name(manifest.file_already_exists_policy))); context_copy->setSetting("export_merge_tree_part_max_bytes_per_file", manifest.max_bytes_per_file); diff --git a/src/Storages/StorageReplicatedMergeTree.cpp b/src/Storages/StorageReplicatedMergeTree.cpp index 049bcaf7fcfe..e49e671d064f 100644 --- a/src/Storages/StorageReplicatedMergeTree.cpp +++ b/src/Storages/StorageReplicatedMergeTree.cpp @@ -229,6 +229,10 @@ namespace Setting extern const SettingsUInt64 export_merge_tree_partition_task_timeout_seconds; extern const SettingsBool output_format_parallel_formatting; extern const SettingsBool output_format_parquet_parallel_encoding; + extern const SettingsParquetCompression output_format_parquet_compression_method; + extern const SettingsUInt64 output_format_compression_level; + extern const SettingsUInt64 output_format_parquet_row_group_size; + extern const SettingsUInt64 output_format_parquet_row_group_size_bytes; extern const SettingsMaxThreads max_threads; extern const SettingsMergeTreePartExportFileAlreadyExistsPolicy export_merge_tree_part_file_already_exists_policy; extern const SettingsUInt64 export_merge_tree_part_max_bytes_per_file; @@ -8709,6 +8713,10 @@ void StorageReplicatedMergeTree::exportPartitionToTable(const PartitionCommand & manifest.max_threads = query_context->getSettingsRef()[Setting::max_threads]; manifest.parallel_formatting = query_context->getSettingsRef()[Setting::output_format_parallel_formatting]; manifest.parquet_parallel_encoding = query_context->getSettingsRef()[Setting::output_format_parquet_parallel_encoding]; + manifest.parquet_compression_method = query_context->getSettingsRef()[Setting::output_format_parquet_compression_method].toString(); + manifest.output_format_compression_level = query_context->getSettingsRef()[Setting::output_format_compression_level]; + manifest.parquet_row_group_size = query_context->getSettingsRef()[Setting::output_format_parquet_row_group_size]; + manifest.parquet_row_group_size_bytes = query_context->getSettingsRef()[Setting::output_format_parquet_row_group_size_bytes]; manifest.max_bytes_per_file = query_context->getSettingsRef()[Setting::export_merge_tree_part_max_bytes_per_file]; manifest.max_rows_per_file = query_context->getSettingsRef()[Setting::export_merge_tree_part_max_rows_per_file]; From 93a5c3f8a4bb1455dbc116314fc751a50ab70fc6 Mon Sep 17 00:00:00 2001 From: Arthur Passos Date: Tue, 7 Jul 2026 10:41:55 -0300 Subject: [PATCH 32/43] Merge pull request #2004 from Altinity/change_export_partition_log_levels Change export partition log levels Source-PR: #2004 (https://github.com/Altinity/ClickHouse/pull/2004) --- .../ExportPartitionManifestUpdatingTask.cpp | 40 ++++++------ .../ExportPartitionTaskScheduler.cpp | 62 +++++++++---------- .../MergeTree/ExportPartitionUtils.cpp | 18 +++--- 3 files changed, 60 insertions(+), 60 deletions(-) diff --git a/src/Storages/MergeTree/ExportPartitionManifestUpdatingTask.cpp b/src/Storages/MergeTree/ExportPartitionManifestUpdatingTask.cpp index 6aa9a15cfe0f..876be1113a31 100644 --- a/src/Storages/MergeTree/ExportPartitionManifestUpdatingTask.cpp +++ b/src/Storages/MergeTree/ExportPartitionManifestUpdatingTask.cpp @@ -63,7 +63,7 @@ namespace ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperGetChildren); if (Coordination::Error::ZOK != zk->tryGetChildren(container_path, children)) { - LOG_INFO(log, "ExportPartition Manifest Updating Task: failed to list last_exception leaves for {}, leaving in-memory copy untouched", log_key); + LOG_WARNING(log, "ExportPartition Manifest Updating Task: failed to list last_exception leaves for {}, leaving in-memory copy untouched", log_key); return out; } @@ -140,7 +140,7 @@ namespace fs::path(entry_path) / "commit_lock", *zk, storage.getReplicaName()); if (!commit_lock) { - LOG_INFO(log, "ExportPartition Manifest Updating Task: commit in progress for {}, skipping timeout kill", entry_path); + LOG_DEBUG(log, "ExportPartition Manifest Updating Task: commit in progress for {}, skipping timeout kill", entry_path); return; } @@ -153,14 +153,14 @@ namespace ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperGet); if (!zk->tryGet(status_path, status_string, &status_stat)) { - LOG_INFO(log, "ExportPartition Manifest Updating Task: Failed to read status for {} while enforcing task timeout, skipping", entry_path); + LOG_WARNING(log, "ExportPartition Manifest Updating Task: Failed to read status for {} while enforcing task timeout, skipping", entry_path); return; } const auto current_status = magic_enum::enum_cast(status_string); if (!current_status || *current_status != ExportReplicatedMergeTreePartitionTaskEntry::Status::PENDING) { - LOG_INFO(log, "ExportPartition Manifest Updating Task: Task {} is not PENDING, can't set to KILLED, skipping", entry_path); + LOG_DEBUG(log, "ExportPartition Manifest Updating Task: Task {} is not PENDING, can't set to KILLED, skipping", entry_path); return; } @@ -195,7 +195,7 @@ namespace /// ZBADVERSION (status changed), ZNODEEXISTS (lazy-create race with the scheduler), /// counter race, or ZNONODE (entry concurrently removed). In all cases the batch /// was rolled back atomically and the task will be re-evaluated on the next poll. - LOG_INFO(log, + LOG_DEBUG(log, "ExportPartition Manifest Updating Task: atomic kill for {} failed (rc={}); " "status was concurrently updated or a ZK op conflicted, will retry on next poll", entry_path, rc); @@ -215,19 +215,19 @@ namespace if (Coordination::Error::ZOK != zk->tryGetChildren(fs::path(entry_path) / "processing", parts_in_processing_or_pending)) { - LOG_INFO(log, "ExportPartition Manifest Updating Task: Failed to get parts in processing or pending, skipping"); + LOG_WARNING(log, "ExportPartition Manifest Updating Task: Failed to get parts in processing or pending, skipping"); return; } if (parts_in_processing_or_pending.empty()) { - LOG_INFO(log, "ExportPartition Manifest Updating Task: Cleanup found PENDING for {} with all parts exported, deferring commit recovery to post-lock phase", entry_path); + LOG_DEBUG(log, "ExportPartition Manifest Updating Task: Cleanup found PENDING for {} with all parts exported, deferring commit recovery to post-lock phase", entry_path); const auto destination_storage_id = StorageID(QualifiedTableName {metadata.destination_database, metadata.destination_table}); const auto destination_storage = DatabaseCatalog::instance().tryGetTable(destination_storage_id, context); if (!destination_storage) { - LOG_INFO(log, "ExportPartition Manifest Updating Task: Failed to reconstruct destination storage: {}, skipping", destination_storage_id.getNameForLogs()); + LOG_WARNING(log, "ExportPartition Manifest Updating Task: Failed to reconstruct destination storage: {}, skipping", destination_storage_id.getNameForLogs()); return; } @@ -312,7 +312,7 @@ void ExportPartitionManifestUpdatingTask::poll() auto cleanup_lock = zkutil::EphemeralNodeHolder::tryCreate(cleanup_lock_path, *zk, storage.replica_name); if (cleanup_lock) { - LOG_INFO(storage.log, "ExportPartition Manifest Updating Task: Cleanup lock acquired, will remove stale entries"); + LOG_DEBUG(storage.log, "ExportPartition Manifest Updating Task: Cleanup lock acquired, will remove stale entries"); } { @@ -329,7 +329,7 @@ void ExportPartitionManifestUpdatingTask::poll() auto & entries_by_key = working_model->get(); - LOG_INFO(storage.log, "ExportPartition Manifest Updating Task: Polling for new entries for table {}. Current number of entries: {}", storage.getStorageID().getNameForLogs(), entries_by_key.size()); + LOG_DEBUG(storage.log, "ExportPartition Manifest Updating Task: Polling for new entries for table {}. Current number of entries: {}", storage.getStorageID().getNameForLogs(), entries_by_key.size()); ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRequests); ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperGetChildrenWatch); @@ -352,7 +352,7 @@ void ExportPartitionManifestUpdatingTask::poll() std::string metadata_json; if (!zk->tryGet(fs::path(entry_path) / "metadata.json", metadata_json)) { - LOG_INFO(storage.log, "ExportPartition Manifest Updating Task: Skipping {}: missing metadata.json", key); + LOG_WARNING(storage.log, "ExportPartition Manifest Updating Task: Skipping {}: missing metadata.json", key); continue; } @@ -402,14 +402,14 @@ void ExportPartitionManifestUpdatingTask::poll() if (status_string.empty()) { - LOG_INFO(storage.log, "ExportPartition Manifest Updating Task: Skipping {}: missing status", key); + LOG_WARNING(storage.log, "ExportPartition Manifest Updating Task: Skipping {}: missing status", key); continue; } const auto status = magic_enum::enum_cast(status_string); if (!status) { - LOG_INFO(storage.log, "ExportPartition Manifest Updating Task: Invalid status {} for task {}, skipping", status_string, key); + LOG_WARNING(storage.log, "ExportPartition Manifest Updating Task: Invalid status {} for task {}, skipping", status_string, key); continue; } @@ -458,7 +458,7 @@ void ExportPartitionManifestUpdatingTask::poll() } } } - LOG_INFO(storage.log, "ExportPartition Manifest Updating Task: Skipping {}: already exists", key); + LOG_DEBUG(storage.log, "ExportPartition Manifest Updating Task: Skipping {}: already exists", key); } @@ -470,7 +470,7 @@ void ExportPartitionManifestUpdatingTask::poll() /// `entries_by_key` (a reference into it) must not be used afterwards. storage.export_partition_manifests.set(std::move(working_model)); - LOG_INFO(storage.log, "ExportPartition Manifest Updating task: finished polling for new entries. Number of entries: {}", entries_count); + LOG_DEBUG(storage.log, "ExportPartition Manifest Updating task: finished polling for new entries. Number of entries: {}", entries_count); } const auto log_ptr = storage.log.load(); @@ -603,7 +603,7 @@ void ExportPartitionManifestUpdatingTask::handleStatusChanges() const bool had_changes = !local_status_changes.empty(); - LOG_INFO(storage.log, "ExportPartition Manifest Updating task: handling status changes. Number of status changes: {}", local_status_changes.size()); + LOG_DEBUG(storage.log, "ExportPartition Manifest Updating task: handling status changes. Number of status changes: {}", local_status_changes.size()); const auto current_model = storage.export_partition_manifests.get(); auto working_model = current_model @@ -635,7 +635,7 @@ void ExportPartitionManifestUpdatingTask::handleStatusChanges() std::string new_status_string; if (!zk->tryGet(fs::path(storage.zookeeper_path) / "exports" / key / "status", new_status_string)) { - LOG_INFO(storage.log, "ExportPartition Manifest Updating Task: Failed to get new status for task {}, skipping", key); + LOG_WARNING(storage.log, "ExportPartition Manifest Updating Task: Failed to get new status for task {}, skipping", key); local_status_changes.pop(); continue; } @@ -643,7 +643,7 @@ void ExportPartitionManifestUpdatingTask::handleStatusChanges() const auto new_status = magic_enum::enum_cast(new_status_string); if (!new_status) { - LOG_INFO(storage.log, "ExportPartition Manifest Updating Task: Invalid status {} for task {}, skipping", new_status_string, key); + LOG_WARNING(storage.log, "ExportPartition Manifest Updating Task: Invalid status {} for task {}, skipping", new_status_string, key); local_status_changes.pop(); continue; } @@ -691,7 +691,7 @@ void ExportPartitionManifestUpdatingTask::handleStatusChanges() { tryLogCurrentException(storage.log, __PRETTY_FUNCTION__); - LOG_INFO(storage.log, "ExportPartition Manifest Updating task: exception thrown while handling status changes; nothing was published, requeuing the whole batch. Batch size: {}", batch.size()); + LOG_WARNING(storage.log, "ExportPartition Manifest Updating task: exception thrown while handling status changes; nothing was published, requeuing the whole batch. Batch size: {}", batch.size()); std::lock_guard lock(status_changes_mutex); @@ -708,7 +708,7 @@ void ExportPartitionManifestUpdatingTask::handleStatusChanges() std::swap(status_changes, requeued); } - LOG_INFO(storage.log, "ExportPartition Manifest Updating task: pending status changes after requeue: {}", status_changes.size()); + LOG_DEBUG(storage.log, "ExportPartition Manifest Updating task: pending status changes after requeue: {}", status_changes.size()); throw; } diff --git a/src/Storages/MergeTree/ExportPartitionTaskScheduler.cpp b/src/Storages/MergeTree/ExportPartitionTaskScheduler.cpp index 89b97460ec08..c2018382530a 100644 --- a/src/Storages/MergeTree/ExportPartitionTaskScheduler.cpp +++ b/src/Storages/MergeTree/ExportPartitionTaskScheduler.cpp @@ -52,7 +52,7 @@ void ExportPartitionTaskScheduler::run() /// this is subject to TOCTOU - but for now we choose to live with it. if (available_move_executors == 0) { - LOG_INFO(storage.log, "ExportPartition scheduler task: No available move executors, skipping"); + LOG_DEBUG(storage.log, "ExportPartition scheduler task: No available move executors, skipping"); return; } @@ -70,7 +70,7 @@ void ExportPartitionTaskScheduler::run() return; } - LOG_INFO(storage.log, "ExportPartition scheduler task: Available move executors: {}", available_move_executors); + LOG_DEBUG(storage.log, "ExportPartition scheduler task: Available move executors: {}", available_move_executors); std::size_t scheduled_exports_count = 0; @@ -91,14 +91,14 @@ void ExportPartitionTaskScheduler::run() { if (scheduled_exports_count >= available_move_executors) { - LOG_INFO(storage.log, "ExportPartition scheduler task: Scheduled exports count is greater than available move executors, skipping"); + LOG_DEBUG(storage.log, "ExportPartition scheduler task: Scheduled exports count is greater than available move executors, skipping"); break; } /// No need to query zk for status if the local one is not PENDING if (entry.status != ExportReplicatedMergeTreePartitionTaskEntry::Status::PENDING) { - LOG_INFO(storage.log, "ExportPartition scheduler task: Skipping... Local status is {}", magic_enum::enum_name(entry.status).data()); + LOG_DEBUG(storage.log, "ExportPartition scheduler task: Skipping... Local status is {}", magic_enum::enum_name(entry.status).data()); continue; } @@ -113,7 +113,7 @@ void ExportPartitionTaskScheduler::run() if (!destination_storage) { - LOG_INFO(storage.log, "ExportPartition scheduler task: Failed to reconstruct destination storage: {}, skipping", destination_storage_id.getNameForLogs()); + LOG_WARNING(storage.log, "ExportPartition scheduler task: Failed to reconstruct destination storage: {}, skipping", destination_storage_id.getNameForLogs()); continue; } @@ -122,7 +122,7 @@ void ExportPartitionTaskScheduler::run() std::string status_in_zk_string; if (!zk->tryGet(fs::path(storage.zookeeper_path) / "exports" / key / "status", status_in_zk_string)) { - LOG_INFO(storage.log, "ExportPartition scheduler task: Failed to get status, skipping"); + LOG_WARNING(storage.log, "ExportPartition scheduler task: Failed to get status, skipping"); continue; } @@ -130,13 +130,13 @@ void ExportPartitionTaskScheduler::run() if (!status_in_zk) { - LOG_INFO(storage.log, "ExportPartition scheduler task: Failed to get status from zk, skipping"); + LOG_WARNING(storage.log, "ExportPartition scheduler task: Failed to get status from zk, skipping"); continue; } if (status_in_zk.value() != ExportReplicatedMergeTreePartitionTaskEntry::Status::PENDING) { - LOG_INFO(storage.log, "ExportPartition scheduler task: Skipping {}... Status from zk is {}", key, magic_enum::enum_name(status_in_zk.value()).data()); + LOG_DEBUG(storage.log, "ExportPartition scheduler task: Skipping {}... Status from zk is {}", key, magic_enum::enum_name(status_in_zk.value()).data()); continue; } @@ -146,14 +146,14 @@ void ExportPartitionTaskScheduler::run() if (Coordination::Error::ZOK != zk->tryGetChildren(fs::path(storage.zookeeper_path) / "exports" / key / "processing", parts_in_processing_or_pending)) { - LOG_INFO(storage.log, "ExportPartition scheduler task: Failed to get parts in processing or pending, skipping"); + LOG_WARNING(storage.log, "ExportPartition scheduler task: Failed to get parts in processing or pending, skipping"); continue; } if (parts_in_processing_or_pending.empty()) { - LOG_INFO(storage.log, "ExportPartition scheduler task: No parts in processing or pending, skipping"); + LOG_DEBUG(storage.log, "ExportPartition scheduler task: No parts in processing or pending, skipping"); continue; } @@ -166,7 +166,7 @@ void ExportPartitionTaskScheduler::run() if (Coordination::Error::ZOK != zk->tryGetChildren(fs::path(storage.zookeeper_path) / "exports" / key / "locks", locked_parts)) { - LOG_INFO(storage.log, "ExportPartition scheduler task: Failed to get locked parts, skipping"); + LOG_WARNING(storage.log, "ExportPartition scheduler task: Failed to get locked parts, skipping"); continue; } @@ -176,20 +176,20 @@ void ExportPartitionTaskScheduler::run() { if (scheduled_exports_count >= available_move_executors) { - LOG_INFO(storage.log, "ExportPartition scheduler task: Scheduled exports count is greater than available move executors, skipping"); + LOG_DEBUG(storage.log, "ExportPartition scheduler task: Scheduled exports count is greater than available move executors, skipping"); break; } if (locked_parts_set.contains(zk_part_name)) { - LOG_INFO(storage.log, "ExportPartition scheduler task: Part {} is locked, skipping", zk_part_name); + LOG_DEBUG(storage.log, "ExportPartition scheduler task: Part {} is locked, skipping", zk_part_name); continue; } const auto part = storage.getPartIfExists(zk_part_name, {MergeTreeDataPartState::Active, MergeTreeDataPartState::Outdated}); if (!part) { - LOG_INFO(storage.log, "ExportPartition scheduler task: Part {} not found locally, skipping", zk_part_name); + LOG_DEBUG(storage.log, "ExportPartition scheduler task: Part {} not found locally, skipping", zk_part_name); continue; } @@ -199,7 +199,7 @@ void ExportPartitionTaskScheduler::run() try { - LOG_INFO(storage.log, "ExportPartition scheduler task: Exporting part to table"); + LOG_DEBUG(storage.log, "ExportPartition scheduler task: Exporting part to table"); LOG_INFO(storage.log, "ExportPartition scheduler task: Attempting to lock part: {}", zk_part_name); @@ -280,12 +280,12 @@ void ExportPartitionTaskScheduler::handlePartExportSuccess( for (const auto & relative_path_in_destination_storage : relative_paths_in_destination_storage) { - LOG_INFO(storage.log, "ExportPartition scheduler task: {}", relative_path_in_destination_storage); + LOG_DEBUG(storage.log, "ExportPartition scheduler task: {}", relative_path_in_destination_storage); } if (!tryToMovePartToProcessed(export_path, processing_parts_path, processed_part_path, part_name, relative_paths_in_destination_storage, zk)) { - LOG_INFO(storage.log, "ExportPartition scheduler task: Failed to move part to processed, will not commit export partition"); + LOG_WARNING(storage.log, "ExportPartition scheduler task: Failed to move part to processed, will not commit export partition"); return; } @@ -352,13 +352,13 @@ void ExportPartitionTaskScheduler::handlePartExportFailure( ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperGet); if (!zk->tryGet(export_path / "locks" / part_name, locked_by, &locked_by_stat)) { - LOG_INFO(storage.log, "ExportPartition scheduler task: Part {} is not locked by any replica, will not increment error counts", part_name); + LOG_DEBUG(storage.log, "ExportPartition scheduler task: Part {} is not locked by any replica, will not increment error counts", part_name); return; } if (locked_by != storage.replica_name) { - LOG_INFO(storage.log, "ExportPartition scheduler task: Part {} is locked by another replica, will not increment error counts", part_name); + LOG_DEBUG(storage.log, "ExportPartition scheduler task: Part {} is locked by another replica, will not increment error counts", part_name); return; } @@ -385,7 +385,7 @@ void ExportPartitionTaskScheduler::handlePartExportFailure( if (Coordination::Error::ZBADVERSION == removal_code) { - LOG_INFO(storage.log, "ExportPartition scheduler task: Part {} lock version mismatch, will not increment error counts", part_name); + LOG_DEBUG(storage.log, "ExportPartition scheduler task: Part {} lock version mismatch, will not increment error counts", part_name); break; } @@ -404,14 +404,14 @@ void ExportPartitionTaskScheduler::handlePartExportFailure( ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperGet); if (!zk->tryGet(status_path, current_status, &status_stat)) { - LOG_INFO(storage.log, "ExportPartition scheduler task: /status missing for {}, skipping failure bookkeeping", export_path.string()); + LOG_DEBUG(storage.log, "ExportPartition scheduler task: /status missing for {}, skipping failure bookkeeping", export_path.string()); return; } const auto status = magic_enum::enum_cast(current_status); if (!status || *status != ExportReplicatedMergeTreePartitionTaskEntry::Status::PENDING) { - LOG_INFO(storage.log, "ExportPartition scheduler task: /status for {} is {} (not PENDING), skipping failure bookkeeping", export_path.string(), current_status); + LOG_DEBUG(storage.log, "ExportPartition scheduler task: /status for {} is {} (not PENDING), skipping failure bookkeeping", export_path.string(), current_status); return; } @@ -425,7 +425,7 @@ void ExportPartitionTaskScheduler::handlePartExportFailure( if (!zk->tryGet(processing_part_path, processing_part_string)) { - LOG_INFO(storage.log, "ExportPartition scheduler task: Failed to get processing part, will not increment error counts"); + LOG_WARNING(storage.log, "ExportPartition scheduler task: Failed to get processing part, will not increment error counts"); return; } @@ -446,11 +446,11 @@ void ExportPartitionTaskScheduler::handlePartExportFailure( processing_part_entry.finished_by = storage.replica_name; ops.emplace_back(zkutil::makeSetRequest(status_path, String(magic_enum::enum_name(ExportReplicatedMergeTreePartitionTaskEntry::Status::FAILED)).data(), status_stat.version)); - LOG_INFO(storage.log, "ExportPartition scheduler task: Retry count limit exceeded for part {}, will try to fail the entire task", part_name); + LOG_WARNING(storage.log, "ExportPartition scheduler task: Retry count limit exceeded for part {}, will try to fail the entire task", part_name); } else { - LOG_INFO(storage.log, "ExportPartition scheduler task: Retry count limit not exceeded for part {}, will increment retry count", part_name); + LOG_DEBUG(storage.log, "ExportPartition scheduler task: Retry count limit not exceeded for part {}, will increment retry count", part_name); } ExportPartitionUtils::appendExceptionOps( @@ -462,7 +462,7 @@ void ExportPartitionTaskScheduler::handlePartExportFailure( Coordination::Responses responses; if (Coordination::Error::ZOK != zk->tryMulti(ops, responses)) { - LOG_INFO(storage.log, "ExportPartition scheduler task: All failure mechanism failed, will not try to update it"); + LOG_WARNING(storage.log, "ExportPartition scheduler task: All failure mechanism failed, will not try to update it"); return; } @@ -485,7 +485,7 @@ bool ExportPartitionTaskScheduler::tryToMovePartToProcessed( ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperGet); if (!zk->tryGet(export_path / "locks" / part_name, locked_by, &locked_by_stat)) { - LOG_INFO(storage.log, "ExportPartition scheduler task: Part {} is not locked by any replica, will not commit or set it as completed", part_name); + LOG_DEBUG(storage.log, "ExportPartition scheduler task: Part {} is not locked by any replica, will not commit or set it as completed", part_name); return false; } @@ -493,7 +493,7 @@ bool ExportPartitionTaskScheduler::tryToMovePartToProcessed( /// I guess we should not throw if file already exists for export partition, hard coded. if (locked_by != storage.replica_name) { - LOG_INFO(storage.log, "ExportPartition scheduler task: Part {} is locked by another replica, will not commit or set it as completed", part_name); + LOG_DEBUG(storage.log, "ExportPartition scheduler task: Part {} is locked by another replica, will not commit or set it as completed", part_name); return false; } @@ -515,7 +515,7 @@ bool ExportPartitionTaskScheduler::tryToMovePartToProcessed( { /// todo arthur remember what to do here - LOG_INFO(storage.log, "ExportPartition scheduler task: Failed to update export path, skipping"); + LOG_WARNING(storage.log, "ExportPartition scheduler task: Failed to update export path, skipping"); return false; } @@ -531,13 +531,13 @@ bool ExportPartitionTaskScheduler::areAllPartsProcessed( Strings parts_in_processing_or_pending; if (Coordination::Error::ZOK != zk->tryGetChildren(export_path / "processing", parts_in_processing_or_pending)) { - LOG_INFO(storage.log, "ExportPartition scheduler task: Failed to get parts in processing or pending, will not try to commit export partition"); + LOG_WARNING(storage.log, "ExportPartition scheduler task: Failed to get parts in processing or pending, will not try to commit export partition"); return false; } if (!parts_in_processing_or_pending.empty()) { - LOG_INFO(storage.log, "ExportPartition scheduler task: There are still parts in processing or pending, will not try to commit export partition"); + LOG_DEBUG(storage.log, "ExportPartition scheduler task: There are still parts in processing or pending, will not try to commit export partition"); return false; } diff --git a/src/Storages/MergeTree/ExportPartitionUtils.cpp b/src/Storages/MergeTree/ExportPartitionUtils.cpp index c9fe33744b12..d922ebc42371 100644 --- a/src/Storages/MergeTree/ExportPartitionUtils.cpp +++ b/src/Storages/MergeTree/ExportPartitionUtils.cpp @@ -121,7 +121,7 @@ namespace ExportPartitionUtils { std::vector exported_paths; - LOG_INFO(log, "ExportPartition: Getting exported paths for {}", export_path); + LOG_DEBUG(log, "ExportPartition: Getting exported paths for {}", export_path); const auto processed_parts_path = fs::path(export_path) / "processed"; @@ -131,7 +131,7 @@ namespace ExportPartitionUtils if (Coordination::Error::ZOK != zk->tryGetChildren(processed_parts_path, processed_parts)) { /// todo arthur do something here - LOG_INFO(log, "ExportPartition: Failed to get parts children, exiting"); + LOG_WARNING(log, "ExportPartition: Failed to get parts children, exiting"); return {}; } @@ -155,7 +155,7 @@ namespace ExportPartitionUtils /// todo arthur what to do in this case? /// It could be that zk is corrupt, in that case we should fail the task /// but it can also be some temporary network issue? not sure - LOG_INFO(log, "ExportPartition: Failed to get exported path, exiting"); + LOG_WARNING(log, "ExportPartition: Failed to get exported path, exiting"); return {}; } @@ -199,7 +199,7 @@ namespace ExportPartitionUtils auto commit_lock = zkutil::EphemeralNodeHolder::tryCreate(commit_lock_path, *zk, replica_name); if (!commit_lock) { - LOG_INFO(log, "ExportPartition: commit_lock for {} is held by another replica, skipping commit on this replica", entry_path); + LOG_DEBUG(log, "ExportPartition: commit_lock for {} is held by another replica, skipping commit on this replica", entry_path); return; } LOG_INFO(log, "ExportPartition: commit_lock for {} acquired by replica {}", entry_path, replica_name); @@ -212,7 +212,7 @@ namespace ExportPartitionUtils const auto status = magic_enum::enum_cast(status_str); if (!status || *status != ExportReplicatedMergeTreePartitionTaskEntry::Status::PENDING) { - LOG_INFO(log, "ExportPartition: {} not PENDING, skipping commit", entry_path); + LOG_DEBUG(log, "ExportPartition: {} not PENDING, skipping commit", entry_path); return; } @@ -287,14 +287,14 @@ namespace ExportPartitionUtils if (!zk->tryGet(status_path, current_status, &status_stat)) { /// Task was removed (TTL cleanup or force-overwrite). Nothing to do. - LOG_INFO(log, "ExportPartition: /status missing for {}, skipping commit-failure bookkeeping", entry_path); + LOG_DEBUG(log, "ExportPartition: /status missing for {}, skipping commit-failure bookkeeping", entry_path); return false; } const auto status = magic_enum::enum_cast(current_status); if (!status) { - LOG_INFO(log, "ExportPartition: Invalid status {} for task {}, skipping commit-failure bookkeeping", current_status, entry_path); + LOG_WARNING(log, "ExportPartition: Invalid status {} for task {}, skipping commit-failure bookkeeping", current_status, entry_path); return false; } @@ -302,7 +302,7 @@ namespace ExportPartitionUtils { /// Another replica already reached a terminal state (COMPLETED or FAILED). /// Do NOT overwrite — a successful commit by a peer must win. - LOG_INFO(log, + LOG_DEBUG(log, "ExportPartition: /status for {} is {} (not PENDING), skipping commit-failure bookkeeping", entry_path, current_status); return false; @@ -372,7 +372,7 @@ namespace ExportPartitionUtils /// non-fatal: the next attempt re-reads /status and either skips (terminal /// state won) or retries the bookkeeping. Worst case we delay FAILED by one /// poll cycle, which matches the best-effort property of the existing counters. - LOG_INFO(log, "ExportPartition: Failed to persist commit failure bookkeeping for {}: {}", entry_path, rc); + LOG_WARNING(log, "ExportPartition: Failed to persist commit failure bookkeeping for {}: {}", entry_path, rc); return false; } From 5a72ad38fac3a86e2a7f83c301c13bb908fdace4 Mon Sep 17 00:00:00 2001 From: Arthur Passos Date: Wed, 8 Jul 2026 09:27:33 -0300 Subject: [PATCH 33/43] Cherry-pick of https://github.com/Altinity/ClickHouse/pull/1984 with unresolved conflict markers (resolution in next commit) --- Original cherry-pick message follows: Merge pull request #1984 from Altinity/export-partition-retry-backoff Partition export per part local backoff policy # Conflicts: # src/Core/Settings.cpp # src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergMetadata.cpp --- docs/en/antalya/partition_export.md | 12 +- src/Common/FailPoint.cpp | 2 + src/Core/Settings.cpp | 14 +- src/Core/SettingsChangesHistory.cpp | 4 +- ...portReplicatedMergeTreePartitionManifest.h | 20 +- src/Storages/MergeTree/ExportPartTask.cpp | 25 ++ .../ExportPartitionManifestUpdatingTask.cpp | 34 +- .../ExportPartitionTaskScheduler.cpp | 213 +++++++++--- .../MergeTree/ExportPartitionTaskScheduler.h | 58 +++- .../MergeTree/ExportPartitionUtils.cpp | 164 +++++---- src/Storages/MergeTree/ExportPartitionUtils.h | 19 +- .../DataLakes/Iceberg/IcebergMetadata.cpp | 22 +- src/Storages/StorageReplicatedMergeTree.cpp | 25 +- ...torageSystemReplicatedPartitionExports.cpp | 16 + .../StorageSystemReplicatedPartitionExports.h | 9 + .../test.py | 315 +++++++++++++++--- .../test.py | 132 ++++++-- .../test_export_partition_iceberg.py | 80 ----- ...rge_tree_part_to_object_storage_simple.sql | 4 +- ...rge_tree_part_to_object_storage_simple.sql | 4 +- 20 files changed, 864 insertions(+), 308 deletions(-) diff --git a/docs/en/antalya/partition_export.md b/docs/en/antalya/partition_export.md index f4c45f612699..9a964af11581 100644 --- a/docs/en/antalya/partition_export.md +++ b/docs/en/antalya/partition_export.md @@ -61,11 +61,17 @@ TO TABLE [destination_database.]destination_table - **Default**: `false` - **Description**: Ignore existing partition export and overwrite the ZooKeeper entry. Allows re-exporting a partition that was already exported to the same destination. **IMPORTANT:** this is dangerous because it can lead to duplicated data, use it with caution. -#### `export_merge_tree_partition_max_retries` (Optional) +#### `export_merge_tree_partition_retry_initial_backoff_seconds` (Optional) - **Type**: `UInt64` -- **Default**: `3` -- **Description**: Maximum number of retries for exporting a merge tree part in an export partition task. If it exceeds, the entire task fails. +- **Default**: `5` +- **Description**: Initial delay (in seconds) before retrying a failed part export. The delay grows exponentially with the per-replica retry count (`delay = min(initial << (attempts - 1), max)`). The back-off is per-replica in-memory state: it only spaces this replica's retries out in time and never prevents another replica from attempting the same part. Retryable failures (transient memory/network/object-storage/Keeper errors) are retried until the task succeeds or `export_merge_tree_partition_task_timeout_seconds` elapses, while non-retryable failures (e.g. schema/type incompatibilities) fail the task immediately. + +#### `export_merge_tree_partition_retry_max_backoff_seconds` (Optional) + +- **Type**: `UInt64` +- **Default**: `300` +- **Description**: Maximum delay (in seconds) between retries of a failed part export. Caps the exponential growth controlled by `export_merge_tree_partition_retry_initial_backoff_seconds`. #### `export_merge_tree_part_file_already_exists_policy` (Optional) diff --git a/src/Common/FailPoint.cpp b/src/Common/FailPoint.cpp index bf5ae1bea887..30b4b0c2c75e 100644 --- a/src/Common/FailPoint.cpp +++ b/src/Common/FailPoint.cpp @@ -168,6 +168,8 @@ static struct InitFiu ONCE(iceberg_export_after_commit_before_zk_completed) \ REGULAR(export_partition_commit_always_throw) \ ONCE(export_partition_status_change_throw) \ + REGULAR(export_part_non_retryable_throw) \ + REGULAR(export_part_retryable_throw) \ ONCE(backup_add_empty_memory_table) \ PAUSEABLE_ONCE(backup_pause_on_start) \ PAUSEABLE_ONCE(restore_pause_on_start) \ diff --git a/src/Core/Settings.cpp b/src/Core/Settings.cpp index 7a8ca857a25a..39cdc0498c21 100644 --- a/src/Core/Settings.cpp +++ b/src/Core/Settings.cpp @@ -8040,8 +8040,14 @@ Overwrite file if it already exists when exporting a merge tree part DECLARE(Bool, export_merge_tree_partition_force_export, false, R"( Ignore existing partition export and overwrite the zookeeper entry )", 0) \ - DECLARE(UInt64, export_merge_tree_partition_max_retries, 3, R"( -Maximum number of retries for exporting a merge tree part in an export partition task + DECLARE(UInt64, export_merge_tree_partition_retry_initial_backoff_seconds, 5, R"( +Initial delay (in seconds) before retrying a failed part export in an export partition task. +The delay grows exponentially with the per-replica retry count (capped doubling): `delay = min(initial << (attempts - 1), max)`, where `max` is `export_merge_tree_partition_retry_max_backoff_seconds`. +The back-off is per-replica in-memory state: it only spaces this replica's retries out in time and never prevents another replica from attempting the same part. Retryable failures are retried until the task succeeds or `export_merge_tree_partition_task_timeout_seconds` elapses. +To survive a long transient outage (e.g. object storage downtime), raise `export_merge_tree_partition_task_timeout_seconds`. +)", 0) \ + DECLARE(UInt64, export_merge_tree_partition_retry_max_backoff_seconds, 300, R"( +Maximum delay (in seconds) between retries of a failed part export in an export partition task. Caps the exponential growth controlled by `export_merge_tree_partition_retry_initial_backoff_seconds`. )", 0) \ DECLARE(UInt64, export_merge_tree_partition_task_timeout_seconds, 3600, R"( Maximum wall-clock duration (in seconds) an export partition task is allowed to remain in the PENDING state before it is auto-killed by the background cleanup loop. @@ -8492,7 +8498,11 @@ Maximum number of texts to include in a single HTTP request made by `aiEmbed`. T #define OBSOLETE_SETTINGS(M, ALIAS) \ /** Obsolete settings which are kept around for compatibility reasons. They have no effect anymore. */ \ MAKE_OBSOLETE(M, UInt64, export_merge_tree_partition_manifest_ttl, 86400) \ +<<<<<<< HEAD MAKE_OBSOLETE(M, Bool, allow_experimental_query_deduplication, false) \ +======= + MAKE_OBSOLETE(M, UInt64, export_merge_tree_partition_max_retries, 3) \ +>>>>>>> 091776b7b65 (Merge pull request #1984 from Altinity/export-partition-retry-backoff) MAKE_OBSOLETE(M, Bool, query_condition_cache_store_conditions_as_plaintext, false) \ MAKE_OBSOLETE(M, Bool, update_insert_deduplication_token_in_dependent_materialized_views, 0) \ MAKE_OBSOLETE(M, UInt64, max_memory_usage_for_all_queries, 0) \ diff --git a/src/Core/SettingsChangesHistory.cpp b/src/Core/SettingsChangesHistory.cpp index 9cfda40eee66..657fb5a1ece0 100644 --- a/src/Core/SettingsChangesHistory.cpp +++ b/src/Core/SettingsChangesHistory.cpp @@ -89,6 +89,9 @@ const VersionToSettingsChangesMap & getSettingsChangesHistory() {"object_storage_cluster_join_mode", "allow", "allow", "New setting"}, {"export_merge_tree_partition_task_timeout_seconds", "3600", "86400", "Increase default value to make it more realistic"}, {"export_merge_tree_part_allow_lossy_cast", false, false, "New setting to gate lossy casts in EXPORT PART/PARTITION behind explicit acknowledgment"}, + {"export_merge_tree_partition_retry_initial_backoff_seconds", 5, 5, "New setting for exponential back-off between failed part export retries in an export partition task"}, + {"export_merge_tree_partition_retry_max_backoff_seconds", 300, 300, "New setting capping the exponential back-off between failed part export retries in an export partition task"}, + {"export_merge_tree_partition_max_retries", 3, 3, "Obsolete and ignored: export partition tasks now retry retryable failures until the task timeout and fail immediately on non-retryable errors, instead of using a fixed retry budget"}, }); addSettingsChanges(settings_changes_history, "26.5", @@ -492,7 +495,6 @@ const VersionToSettingsChangesMap & getSettingsChangesHistory() {"allow_experimental_export_merge_tree_part", false, true, "Turned ON by default for Antalya."}, {"export_merge_tree_part_overwrite_file_if_exists", false, false, "New setting."}, {"export_merge_tree_partition_force_export", false, false, "New setting."}, - {"export_merge_tree_partition_max_retries", 3, 3, "New setting."}, {"export_merge_tree_partition_manifest_ttl", 180, 180, "New setting."}, {"export_merge_tree_part_file_already_exists_policy", "skip", "skip", "New setting."}, {"export_merge_tree_part_max_bytes_per_file", 0, 0, "New setting."}, diff --git a/src/Storages/ExportReplicatedMergeTreePartitionManifest.h b/src/Storages/ExportReplicatedMergeTreePartitionManifest.h index 3c2b5da61a18..49da8eff4895 100644 --- a/src/Storages/ExportReplicatedMergeTreePartitionManifest.h +++ b/src/Storages/ExportReplicatedMergeTreePartitionManifest.h @@ -23,7 +23,6 @@ struct ExportReplicatedMergeTreePartitionProcessingPartEntry String part_name; Status status; - size_t retry_count; String finished_by; std::string toJsonString() const @@ -32,7 +31,6 @@ struct ExportReplicatedMergeTreePartitionProcessingPartEntry json.set("part_name", part_name); json.set("status", String(magic_enum::enum_name(status))); - json.set("retry_count", retry_count); json.set("finished_by", finished_by); std::ostringstream oss; // STYLE_CHECK_ALLOW_STD_STRING_STREAM oss.exceptions(std::ios::failbit); @@ -51,7 +49,6 @@ struct ExportReplicatedMergeTreePartitionProcessingPartEntry entry.part_name = json->getValue("part_name"); entry.status = magic_enum::enum_cast(json->getValue("status")).value(); - entry.retry_count = json->getValue("retry_count"); if (json->has("finished_by")) { entry.finished_by = json->getValue("finished_by"); @@ -163,7 +160,8 @@ struct ExportReplicatedMergeTreePartitionManifest size_t number_of_parts; std::vector parts; time_t create_time; - size_t max_retries; + size_t retry_initial_backoff_seconds = 5; + size_t retry_max_backoff_seconds = 300; size_t task_timeout_seconds; size_t max_threads; bool parallel_formatting; @@ -208,7 +206,8 @@ struct ExportReplicatedMergeTreePartitionManifest json.set("file_already_exists_policy", String(magic_enum::enum_name(file_already_exists_policy))); json.set("filename_pattern", filename_pattern); json.set("create_time", create_time); - json.set("max_retries", max_retries); + json.set("retry_initial_backoff_seconds", retry_initial_backoff_seconds); + json.set("retry_max_backoff_seconds", retry_max_backoff_seconds); json.set("task_timeout_seconds", task_timeout_seconds); json.set("write_full_path_in_iceberg_metadata", write_full_path_in_iceberg_metadata); json.set("allow_lossy_cast", allow_lossy_cast); @@ -236,7 +235,16 @@ struct ExportReplicatedMergeTreePartitionManifest manifest.destination_table = json->getValue("destination_table"); manifest.source_replica = json->getValue("source_replica"); manifest.number_of_parts = json->getValue("number_of_parts"); - manifest.max_retries = json->getValue("max_retries"); + + if (json->has("retry_initial_backoff_seconds")) + { + manifest.retry_initial_backoff_seconds = json->getValue("retry_initial_backoff_seconds"); + } + + if (json->has("retry_max_backoff_seconds")) + { + manifest.retry_max_backoff_seconds = json->getValue("retry_max_backoff_seconds"); + } if (json->has("iceberg_metadata_json")) { diff --git a/src/Storages/MergeTree/ExportPartTask.cpp b/src/Storages/MergeTree/ExportPartTask.cpp index 25d6e7ffcadd..3e12fa0049bb 100644 --- a/src/Storages/MergeTree/ExportPartTask.cpp +++ b/src/Storages/MergeTree/ExportPartTask.cpp @@ -18,6 +18,7 @@ #include #include "Common/setThreadName.h" #include +#include #include #include #include @@ -43,6 +44,18 @@ namespace ErrorCodes extern const int FILE_ALREADY_EXISTS; extern const int LOGICAL_ERROR; extern const int QUERY_WAS_CANCELLED; + extern const int BAD_ARGUMENTS; + extern const int FAULT_INJECTED; +} + +namespace FailPoints +{ + /// Throw a non-retryable (denylisted) error from the part-export worker, so the whole + /// export task transitions to FAILED immediately regardless of any timeout. + extern const char export_part_non_retryable_throw[]; + /// Throw a retryable error from the part-export worker, so the part is retried with the + /// per-replica back-off until the task succeeds or the absolute timeout fires. + extern const char export_part_retryable_throw[]; } namespace Setting @@ -234,6 +247,18 @@ bool ExportPartTask::executeStep() { ThreadGroupSwitcher switcher((*exports_list_entry)->thread_group, ThreadName::EXPORT_PART); + fiu_do_on(FailPoints::export_part_non_retryable_throw, + { + throw Exception(ErrorCodes::BAD_ARGUMENTS, + "Failpoint: export_part_non_retryable_throw"); + }); + + fiu_do_on(FailPoints::export_part_retryable_throw, + { + throw Exception(ErrorCodes::FAULT_INJECTED, + "Failpoint: export_part_retryable_throw"); + }); + const auto filename = buildDestinationFilename(manifest, storage.getStorageID(), local_context); sink = destination_storage->import( diff --git a/src/Storages/MergeTree/ExportPartitionManifestUpdatingTask.cpp b/src/Storages/MergeTree/ExportPartitionManifestUpdatingTask.cpp index 876be1113a31..f020f6da6386 100644 --- a/src/Storages/MergeTree/ExportPartitionManifestUpdatingTask.cpp +++ b/src/Storages/MergeTree/ExportPartitionManifestUpdatingTask.cpp @@ -255,6 +255,8 @@ std::vector ExportPartitionManifestUpdatingTask:: if (!model) return {}; + const auto backoff = storage.export_merge_tree_partition_task_scheduler->getLocalBackoffSnapshot(); + std::vector infos; infos.reserve(model->size()); @@ -285,6 +287,13 @@ std::vector ExportPartitionManifestUpdatingTask:: } info.exception_count = total_exception_count; + if (const auto it = backoff.find(entry.getTransactionId()); it != backoff.end()) + { + info.backoff_per_part.reserve(it->second.size()); + for (const auto & [part_name, state] : it->second) + info.backoff_per_part.push_back({part_name, state.attempts, state.next_retry_time}); + } + infos.emplace_back(std::move(info)); } @@ -356,7 +365,20 @@ void ExportPartitionManifestUpdatingTask::poll() continue; } - const auto metadata = ExportReplicatedMergeTreePartitionManifest::fromJsonString(metadata_json); + ExportReplicatedMergeTreePartitionManifest metadata; + try + { + metadata = ExportReplicatedMergeTreePartitionManifest::fromJsonString(metadata_json); + } + catch (...) + { + /// A single unparseable metadata.json (e.g. genuinely corrupt, or written by a + /// future incompatible format) must not abort the whole poll and stall discovery, + /// cleanup and status convergence for every other task. Skip just this entry. + tryLogCurrentException(storage.log, __PRETTY_FUNCTION__); + LOG_WARNING(storage.log, "ExportPartition Manifest Updating Task: Skipping {}: could not parse metadata.json", key); + continue; + } auto last_exception_per_replica = readLastExceptionPerReplica( zk, fs::path(entry_path), key, storage.log.load()); @@ -490,20 +512,20 @@ void ExportPartitionManifestUpdatingTask::poll() "Caught exception while committing export for {}: {}", work.entry_path, e.message()); - const bool exceeded_commimt_max_retries = ExportPartitionUtils::handleCommitFailure( + const bool became_failed = ExportPartitionUtils::handleCommitFailure( zk, work.entry_path, - work.metadata.max_retries, + e.code(), storage.getReplicaName(), e.message(), log_ptr); - if (exceeded_commimt_max_retries) + if (became_failed) { LOG_WARNING(log_ptr, "ExportPartition Manifest Updating Task: " - "Commit for {} transitioned to FAILED after exhausting max_retries={}", - work.entry_path, work.metadata.max_retries); + "Commit for {} transitioned to FAILED due to non-retryable error (code {})", + work.entry_path, e.code()); } } } diff --git a/src/Storages/MergeTree/ExportPartitionTaskScheduler.cpp b/src/Storages/MergeTree/ExportPartitionTaskScheduler.cpp index c2018382530a..de271bc01ec7 100644 --- a/src/Storages/MergeTree/ExportPartitionTaskScheduler.cpp +++ b/src/Storages/MergeTree/ExportPartitionTaskScheduler.cpp @@ -12,6 +12,7 @@ #include "Storages/MergeTree/MergeTreePartExportManifest.h" #include "Formats/FormatFactory.h" #include +#include namespace ProfileEvents { @@ -40,20 +41,50 @@ namespace ErrorCodes extern const int LOGICAL_ERROR; } +namespace +{ + /// Capped exponential back-off, matching the standard ClickHouse convention + /// (see ZooKeeperRetriesControl): delay = min(initial << (retry_count - 1), max). + /// `retry_count` is the number of failures so far (>= 1 when a retry is pending). + /// The shift is guarded against overflow by saturating to `max_backoff_seconds`. + size_t computeRetryBackoffSeconds(size_t retry_count, size_t initial_backoff_seconds, size_t max_backoff_seconds) + { + const size_t initial = std::min(initial_backoff_seconds, max_backoff_seconds); + + if (retry_count <= 1 || initial == 0) + return initial; + + const size_t shift = retry_count - 1; + + /// If shifting would overflow size_t, the result is certainly clamped to the cap. + static constexpr size_t bits = sizeof(size_t) * 8; + if (shift >= bits) + return max_backoff_seconds; + + const size_t headroom = std::numeric_limits::max() >> shift; + if (initial > headroom) + return max_backoff_seconds; + + return std::min(initial << shift, max_backoff_seconds); + } +} + ExportPartitionTaskScheduler::ExportPartitionTaskScheduler(StorageReplicatedMergeTree & storage_) : storage(storage_) { } -void ExportPartitionTaskScheduler::run() +std::optional ExportPartitionTaskScheduler::run() { + std::optional earliest_backoff_retry; + const auto available_move_executors = storage.background_moves_assignee.getAvailableMoveExecutors(); /// this is subject to TOCTOU - but for now we choose to live with it. if (available_move_executors == 0) { LOG_DEBUG(storage.log, "ExportPartition scheduler task: No available move executors, skipping"); - return; + return earliest_backoff_retry; } /// Respect the background memory soft-limit: refuse to schedule new export-part tasks when @@ -67,7 +98,7 @@ void ExportPartitionTaskScheduler::run() "so won't select new parts to export. Current background tasks memory usage: {}.", formatReadableSizeWithBinarySuffix(background_memory_tracker.getSoftLimit()), formatReadableSizeWithBinarySuffix(background_memory_tracker.get())); - return; + return earliest_backoff_retry; } LOG_DEBUG(storage.log, "ExportPartition scheduler task: Available move executors: {}", available_move_executors); @@ -82,10 +113,12 @@ void ExportPartitionTaskScheduler::run() /// is a pure reader; status converges via the status watch -> handleStatusChanges and poll(). const auto model = storage.export_partition_manifests.get(); if (!model) - return; + return earliest_backoff_retry; auto zk = storage.getZooKeeper(); + pruneLocalBackoff(model->get()); + // Iterate sorted by create_time for (const auto & entry : model->get()) { @@ -172,6 +205,8 @@ void ExportPartitionTaskScheduler::run() std::unordered_set locked_parts_set(locked_parts.begin(), locked_parts.end()); + const auto now = time(nullptr); + for (const auto & zk_part_name : parts_in_processing_or_pending) { if (scheduled_exports_count >= available_move_executors) @@ -186,6 +221,11 @@ void ExportPartitionTaskScheduler::run() continue; } + if (shouldBackOff(entry.getTransactionId(), zk_part_name, now, earliest_backoff_retry)) + { + continue; + } + const auto part = storage.getPartIfExists(zk_part_name, {MergeTreeDataPartState::Active, MergeTreeDataPartState::Outdated}); if (!part) { @@ -234,10 +274,101 @@ void ExportPartitionTaskScheduler::run() ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRequests); ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRemove); zk->tryRemove(fs::path(storage.zookeeper_path) / "exports" / key / "locks" / zk_part_name); - /// we should not increment retry_count because the node might just be full + /// Dispatch-time failure (e.g. Keeper node full). We do not arm the local + /// back-off here: the export never started, so the part stays immediately + /// eligible for this or another replica on the next tick. } } } + + return earliest_backoff_retry; +} + +bool ExportPartitionTaskScheduler::shouldBackOff( + const std::string & transaction_id, + const std::string & part_name, + time_t now, + std::optional & earliest_backoff_retry) const +{ + std::lock_guard lock(local_backoff_mutex); + const auto task_it = local_backoff.find(transaction_id); + if (task_it == local_backoff.end()) + return false; + + const auto part_it = task_it->second.find(part_name); + if (part_it == task_it->second.end() || now >= part_it->second.next_retry_time) + return false; + + const auto next_retry_time = part_it->second.next_retry_time; + LOG_TRACE(storage.log, "ExportPartition scheduler task: Part {} is backing off locally, next retry at {} (now {}), skipping", part_name, next_retry_time, now); + earliest_backoff_retry = earliest_backoff_retry + ? std::min(*earliest_backoff_retry, next_retry_time) : next_retry_time; + return true; +} + +time_t ExportPartitionTaskScheduler::registerLocalBackoff( + const std::string & transaction_id, + const std::string & part_name, + const ExportReplicatedMergeTreePartitionManifest & manifest) +{ + std::lock_guard lock(local_backoff_mutex); + + /// First retryable failure for (transaction_id, part_name): create the map entries. + auto & parts = local_backoff.try_emplace(transaction_id).first->second; + auto & backoff = parts.try_emplace(part_name).first->second; + + ++backoff.attempts; + const auto backoff_seconds = computeRetryBackoffSeconds( + backoff.attempts, manifest.retry_initial_backoff_seconds, manifest.retry_max_backoff_seconds); + const auto now = time(nullptr); + /// Clamp so a huge configured back-off cannot overflow time_t (now is a normal wall-clock value). + const size_t headroom = static_cast(std::numeric_limits::max() - now); + backoff.next_retry_time = now + static_cast(std::min(backoff_seconds, headroom)); + return backoff.next_retry_time; +} + +void ExportPartitionTaskScheduler::clearLocalBackoff(const std::string & transaction_id, const std::string & part_name) +{ + std::lock_guard lock(local_backoff_mutex); + if (const auto task_it = local_backoff.find(transaction_id); task_it != local_backoff.end()) + { + task_it->second.erase(part_name); + if (task_it->second.empty()) + local_backoff.erase(task_it); + } +} + +void ExportPartitionTaskScheduler::pruneLocalBackoff(const ExportPartitionTaskEntriesContainer::index::type & model) +{ + std::lock_guard lock(local_backoff_mutex); + for (auto it = local_backoff.begin(); it != local_backoff.end();) + { + const auto found = model.find(it->first); + if (found != model.end() && found->status == ExportReplicatedMergeTreePartitionTaskEntry::Status::PENDING) + { + ++it; + continue; + } + + it = local_backoff.erase(it); + } +} + +ExportPartitionTaskScheduler::LocalBackoffMap ExportPartitionTaskScheduler::getLocalBackoffSnapshot() const +{ + LocalBackoffMap snapshot; + + std::lock_guard lock(local_backoff_mutex); + snapshot.reserve(local_backoff.size()); + for (const auto & [transaction_id, parts] : local_backoff) + { + auto & out_parts = snapshot[transaction_id]; + out_parts.reserve(parts.size()); + for (const auto & [part_name, backoff] : parts) + out_parts.emplace(part_name, LocalBackoff{backoff.attempts, backoff.next_retry_time}); + } + + return snapshot; } void ExportPartitionTaskScheduler::handlePartExportCompletion( @@ -261,7 +392,7 @@ void ExportPartitionTaskScheduler::handlePartExportCompletion( } else { - handlePartExportFailure(processing_parts_path, part_name, export_path, zk, result.exception, manifest.max_retries); + handlePartExportFailure(part_name, export_path, zk, result.exception, manifest); } } @@ -289,6 +420,9 @@ void ExportPartitionTaskScheduler::handlePartExportSuccess( return; } + /// Part is done on this replica; drop any local back-off state we held for it. + clearLocalBackoff(manifest.transaction_id, part_name); + LOG_INFO(storage.log, "ExportPartition scheduler task: Marked part export {} as completed", part_name); if (!areAllPartsProcessed(export_path, zk)) @@ -307,15 +441,16 @@ void ExportPartitionTaskScheduler::handlePartExportSuccess( { LOG_INFO(storage.log, "ExportPartition scheduler task: Caught exception while committing export partition, {}", e.message()); - /// Bump commit-attempts counter; transition to FAILED once the budget is exhausted. - /// Prevents the task from remaining stuck in PENDING if commit() fails persistently - /// (e.g. schema/spec mismatch, prolonged destination outage). + /// Classify the commit failure: a non-retryable error (e.g. schema/spec mismatch) + /// transitions the task to FAILED immediately; a retryable one (transient catalog or + /// destination outage) only records the exception and leaves the task PENDING so the + /// commit is retried until the absolute task timeout. /// The exception is recorded in /last_exception via appendExceptionOps - /// inside the same multi as the commit_attempts bump and the (possible) FAILED set. + /// inside the same multi as the (possible) FAILED set. const bool became_failed = ExportPartitionUtils::handleCommitFailure( zk, export_path, - manifest.max_retries, + e.code(), storage.replica_name, e.message(), storage.log.load()); @@ -323,19 +458,18 @@ void ExportPartitionTaskScheduler::handlePartExportSuccess( if (became_failed) { LOG_WARNING(storage.log, - "ExportPartition scheduler task: Commit for {} transitioned to FAILED after exhausting max_retries={}", - export_path.string(), manifest.max_retries); + "ExportPartition scheduler task: Commit for {} transitioned to FAILED due to non-retryable error (code {})", + export_path.string(), e.code()); } } } void ExportPartitionTaskScheduler::handlePartExportFailure( - const std::filesystem::path & processing_parts_path, const std::string & part_name, const std::filesystem::path & export_path, const zkutil::ZooKeeperPtr & zk, const std::optional & exception, - size_t max_retries + const ExportReplicatedMergeTreePartitionManifest & manifest ) { LOG_INFO(storage.log, "ExportPartition scheduler task: Part {} export failed", part_name); @@ -415,42 +549,25 @@ void ExportPartitionTaskScheduler::handlePartExportFailure( return; } - Coordination::Requests ops; - - const auto processing_part_path = processing_parts_path / part_name; - - ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRequests); - ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperGet); - std::string processing_part_string; - - if (!zk->tryGet(processing_part_path, processing_part_string)) - { - LOG_WARNING(storage.log, "ExportPartition scheduler task: Failed to get processing part, will not increment error counts"); - return; - } - - /// todo arthur could this have been cached? - auto processing_part_entry = ExportReplicatedMergeTreePartitionProcessingPartEntry::fromJsonString(processing_part_string); + const bool non_retryable = ExportPartitionUtils::isNonRetryableExportError(exception->code()); - processing_part_entry.retry_count++; + Coordination::Requests ops; ops.emplace_back(zkutil::makeRemoveRequest(export_path / "locks" / part_name, locked_by_stat.version)); - ops.emplace_back(zkutil::makeSetRequest(processing_part_path, processing_part_entry.toJsonString(), -1)); - - LOG_INFO(storage.log, "ExportPartition scheduler task: Updating processing part entry for part {}, retry count: {}, max retries: {}", part_name, processing_part_entry.retry_count, max_retries); - if (processing_part_entry.retry_count >= max_retries) + if (non_retryable) { - /// just set status in processing_part_path and finished_by - processing_part_entry.status = ExportReplicatedMergeTreePartitionProcessingPartEntry::Status::FAILED; - processing_part_entry.finished_by = storage.replica_name; - - ops.emplace_back(zkutil::makeSetRequest(status_path, String(magic_enum::enum_name(ExportReplicatedMergeTreePartitionTaskEntry::Status::FAILED)).data(), status_stat.version)); - LOG_WARNING(storage.log, "ExportPartition scheduler task: Retry count limit exceeded for part {}, will try to fail the entire task", part_name); + /// Deterministic failure (e.g. schema/type incompatibility): retrying cannot help, + /// so fail the whole task immediately instead of waiting for the absolute timeout. + ops.emplace_back(zkutil::makeSetRequest( + status_path, + String(magic_enum::enum_name(ExportReplicatedMergeTreePartitionTaskEntry::Status::FAILED)).data(), + status_stat.version)); + LOG_WARNING(storage.log, "ExportPartition scheduler task: Part {} failed with non-retryable error (code {}), failing the entire task", part_name, exception->code()); } else { - LOG_DEBUG(storage.log, "ExportPartition scheduler task: Retry count limit not exceeded for part {}, will increment retry count", part_name); + LOG_DEBUG(storage.log, "ExportPartition scheduler task: Part {} failed with retryable error (code {}), will back off and retry until the task timeout", part_name, exception->code()); } ExportPartitionUtils::appendExceptionOps( @@ -466,7 +583,15 @@ void ExportPartitionTaskScheduler::handlePartExportFailure( return; } - LOG_INFO(storage.log, "ExportPartition scheduler task: Successfully updated exception counters for part {}", part_name); + /// Only after the lock release + exception record committed do we arm the local back-off, + /// so a Keeper failure above does not leave this replica skipping the part for no reason. + if (!non_retryable) + { + const auto next_retry_time = registerLocalBackoff(manifest.transaction_id, part_name, manifest); + LOG_INFO(storage.log, "ExportPartition scheduler task: Part {} backing off locally, next retry at {}", part_name, next_retry_time); + } + + LOG_INFO(storage.log, "ExportPartition scheduler task: Successfully recorded failure for part {}", part_name); } bool ExportPartitionTaskScheduler::tryToMovePartToProcessed( diff --git a/src/Storages/MergeTree/ExportPartitionTaskScheduler.h b/src/Storages/MergeTree/ExportPartitionTaskScheduler.h index 29a41fde1cb9..038febefd831 100644 --- a/src/Storages/MergeTree/ExportPartitionTaskScheduler.h +++ b/src/Storages/MergeTree/ExportPartitionTaskScheduler.h @@ -1,7 +1,14 @@ #pragma once #include +#include #include +#include +#include +#include +#include +#include +#include namespace DB { @@ -17,7 +24,10 @@ class ExportPartitionTaskScheduler public: ExportPartitionTaskScheduler(StorageReplicatedMergeTree & storage); - void run(); + /// Returns the earliest future back-off deadline (unix seconds) among parts that were skipped + /// this tick purely because they are still backing off, or nullopt if none. The caller can use + /// it to wake the select task sooner than the default tick interval. + std::optional run(); private: StorageReplicatedMergeTree & storage; @@ -41,12 +51,11 @@ class ExportPartitionTaskScheduler ); void handlePartExportFailure( - const std::filesystem::path & processing_parts_path, const std::string & part_name, const std::filesystem::path & export_path, const zkutil::ZooKeeperPtr & zk, const std::optional & exception, - size_t max_retries); + const ExportReplicatedMergeTreePartitionManifest & manifest); bool tryToMovePartToProcessed( const std::filesystem::path & export_path, @@ -61,6 +70,49 @@ class ExportPartitionTaskScheduler const std::filesystem::path & export_path, const zkutil::ZooKeeperPtr & zk ); + + struct LocalBackoff + { + size_t attempts = 0; + time_t next_retry_time = 0; + }; + + /// transaction_id -> part name -> back-off state. Keyed by transaction_id (not composite + /// key) so a reused composite key does not inherit a prior instance's back-off. Guarded by + /// local_backoff_mutex because run() (schedule-pool thread) reads it while part-export + /// completion callbacks (background-executor threads) write it. + using PartNameToBackOffMap = std::unordered_map; + using TransactionID = std::string; + using LocalBackoffMap = std::unordered_map; + + mutable std::mutex local_backoff_mutex; + LocalBackoffMap local_backoff TSA_GUARDED_BY(local_backoff_mutex); + + bool shouldBackOff( + const std::string & transaction_id, + const std::string & part_name, + time_t now, + std::optional & earliest_backoff_retry) const; + + /// Record a retryable failure for (transaction_id, part_name): grow the attempt counter and + /// compute the next eligible time. Returns the new absolute deadline. + time_t registerLocalBackoff( + const std::string & transaction_id, + const std::string & part_name, + const ExportReplicatedMergeTreePartitionManifest & manifest); + + /// Drop any back-off state for parts of (transaction_id) once they succeed or the task ends. + void clearLocalBackoff(const std::string & transaction_id, const std::string & part_name); + + /// Remove back-off state for tasks whose transaction_id is no longer PENDING in the published + /// model, bounding the map to the parts of currently-active tasks. + void pruneLocalBackoff(const ExportPartitionTaskEntriesContainer::index::type & model); + +public: + /// Snapshot of the local back-off map for system.replicated_partition_exports: + /// transaction_id -> part -> (attempts, next_retry_time). Briefly locks local_backoff_mutex; + /// never held across ZooKeeper I/O. + std::unordered_map getLocalBackoffSnapshot() const; }; } diff --git a/src/Storages/MergeTree/ExportPartitionUtils.cpp b/src/Storages/MergeTree/ExportPartitionUtils.cpp index d922ebc42371..ec8046f3e991 100644 --- a/src/Storages/MergeTree/ExportPartitionUtils.cpp +++ b/src/Storages/MergeTree/ExportPartitionUtils.cpp @@ -9,6 +9,7 @@ #include #include #include +#include #include #include #include @@ -27,6 +28,7 @@ namespace ProfileEvents extern const Event ExportPartitionZooKeeperGet; extern const Event ExportPartitionZooKeeperGetChildren; extern const Event ExportPartitionZooKeeperSet; + extern const Event ExportPartitionZooKeeperCreate; extern const Event ExportPartitionZooKeeperMulti; } @@ -40,6 +42,34 @@ namespace ErrorCodes extern const int NO_SUCH_DATA_PART; extern const int CORRUPTED_DATA; extern const int NETWORK_ERROR; + extern const int LOGICAL_ERROR; + extern const int NOT_IMPLEMENTED; + extern const int SUPPORT_IS_DISABLED; + extern const int TYPE_MISMATCH; + extern const int CANNOT_CONVERT_TYPE; + extern const int ILLEGAL_TYPE_OF_ARGUMENT; + extern const int ILLEGAL_COLUMN; + extern const int NUMBER_OF_COLUMNS_DOESNT_MATCH; + extern const int INCOMPATIBLE_COLUMNS; + extern const int NO_SUCH_COLUMN_IN_TABLE; + extern const int FILE_ALREADY_EXISTS; + extern const int METADATA_MISMATCH; + extern const int CANNOT_PARSE_TEXT; + extern const int CANNOT_PARSE_NUMBER; + extern const int CANNOT_PARSE_DATE; + extern const int CANNOT_PARSE_DATETIME; + extern const int CANNOT_PARSE_BOOL; + extern const int CANNOT_PARSE_UUID; + extern const int CANNOT_PARSE_IPV4; + extern const int CANNOT_PARSE_IPV6; + extern const int CANNOT_PARSE_QUOTED_STRING; + extern const int CANNOT_PARSE_ESCAPE_SEQUENCE; + extern const int CANNOT_PARSE_INPUT_ASSERTION_FAILED; + extern const int CANNOT_PARSE_DOMAIN_VALUE_FROM_STRING; + extern const int VALUE_IS_OUT_OF_RANGE_OF_DATA_TYPE; + extern const int ATTEMPT_TO_READ_AFTER_EOF; + extern const int CANNOT_READ_ARRAY_FROM_TEXT; + extern const int DECIMAL_OVERFLOW; } namespace Setting @@ -57,6 +87,49 @@ namespace fs = std::filesystem; namespace ExportPartitionUtils { + bool isNonRetryableExportError(int code) + { + /// Deterministic failures where retrying cannot possibly succeed (schema/type + /// incompatibilities, unsupported features, programming errors). Everything else + /// (memory limits, network/object-storage/Keeper transient errors, ...) is retryable. + /// `QUERY_WAS_CANCELLED` is handled separately by the caller and never reaches here. + /// + /// ErrorCodes values are runtime `extern const int`, not constant expressions, so they + /// cannot be used as `switch` labels; compare against a static set instead. + static const std::unordered_set non_retryable_codes = { + ErrorCodes::BAD_ARGUMENTS, + ErrorCodes::TYPE_MISMATCH, + ErrorCodes::CANNOT_CONVERT_TYPE, + ErrorCodes::ILLEGAL_TYPE_OF_ARGUMENT, + ErrorCodes::ILLEGAL_COLUMN, + ErrorCodes::NUMBER_OF_COLUMNS_DOESNT_MATCH, + ErrorCodes::INCOMPATIBLE_COLUMNS, + ErrorCodes::NO_SUCH_COLUMN_IN_TABLE, + ErrorCodes::NOT_IMPLEMENTED, + ErrorCodes::SUPPORT_IS_DISABLED, + ErrorCodes::LOGICAL_ERROR, + ErrorCodes::FILE_ALREADY_EXISTS, + ErrorCodes::METADATA_MISMATCH, + ErrorCodes::CANNOT_PARSE_TEXT, + ErrorCodes::CANNOT_PARSE_NUMBER, + ErrorCodes::CANNOT_PARSE_DATE, + ErrorCodes::CANNOT_PARSE_DATETIME, + ErrorCodes::CANNOT_PARSE_BOOL, + ErrorCodes::CANNOT_PARSE_UUID, + ErrorCodes::CANNOT_PARSE_IPV4, + ErrorCodes::CANNOT_PARSE_IPV6, + ErrorCodes::CANNOT_PARSE_QUOTED_STRING, + ErrorCodes::CANNOT_PARSE_ESCAPE_SEQUENCE, + ErrorCodes::CANNOT_PARSE_INPUT_ASSERTION_FAILED, + ErrorCodes::CANNOT_PARSE_DOMAIN_VALUE_FROM_STRING, + ErrorCodes::VALUE_IS_OUT_OF_RANGE_OF_DATA_TYPE, + ErrorCodes::ATTEMPT_TO_READ_AFTER_EOF, + ErrorCodes::CANNOT_READ_ARRAY_FROM_TEXT, + ErrorCodes::DECIMAL_OVERFLOW, + }; + return non_retryable_codes.contains(code); + } + Block getPartitionSourceBlockForIcebergCommit( MergeTreeData & storage, const String & partition_id) { @@ -267,7 +340,7 @@ namespace ExportPartitionUtils bool handleCommitFailure( const zkutil::ZooKeeperPtr & zk, const std::string & entry_path, - size_t max_attempts, + int exception_code, const std::string & replica_name, const std::string & exception_message, const LoggerPtr & log) @@ -310,51 +383,19 @@ namespace ExportPartitionUtils Coordination::Requests ops; - /// Record the exception in the same multi as the commit-attempts bump and the - /// (possible) FAILED transition, so the user-visible last_exception znode is - /// updated atomically with the state change that exposes it. + /// Record the exception in the same multi as the (possible) FAILED transition, so the + /// user-visible last_exception znode is updated atomically with the state change that + /// exposes it. appendExceptionOps(ops, zk, fs::path(entry_path), replica_name, /*part_name=*/"", exception_message, log); - /// Bump the global commit_attempts counter (shared across replicas). - /// Non-atomic get+set(-1). Under a race, two replicas may see the same value - /// and write the same +1, under-counting by one. FAILED then fires one retry - /// later than the threshold, which is acceptable. - const std::string commit_attempts_path = fs::path(entry_path) / "commit_attempts"; - - size_t attempts = 0; - std::string attempts_string; - - ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRequests); - ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperGet); - if (zk->tryGet(commit_attempts_path, attempts_string)) - { - try - { - attempts = parse(attempts_string); - } - catch (...) - { - LOG_WARNING(log, "ExportPartition: commit_attempts value '{}' at {} is not a valid integer, treating as 0", attempts_string, commit_attempts_path); - attempts = 0; - } - - attempts += 1; - ops.emplace_back(zkutil::makeSetRequest(commit_attempts_path, std::to_string(attempts), -1)); - } - else - { - attempts = 1; - ops.emplace_back(zkutil::makeCreateRequest(commit_attempts_path, "1", zkutil::CreateMode::Persistent)); - } - - /// Transition to FAILED if the commit budget is exhausted. - /// Uses the same setting as per-part retries (manifest.max_retries) per user decision. - /// Version-checked Set: if /status has changed since we read it (e.g. a peer's - /// commit() succeeded and wrote COMPLETED), the whole multi aborts with - /// ZBADVERSION and we safely do nothing — the winning terminal state stands. - const bool exhausted = attempts >= max_attempts; - if (exhausted) + /// A non-retryable error (schema/spec mismatch, ...) can never succeed, + /// so fail the task immediately + const bool non_retryable = isNonRetryableExportError(exception_code); + if (non_retryable) { + /// Version-checked Set: if /status has changed since we read it (e.g. a peer's + /// commit() succeeded and wrote COMPLETED), the whole multi aborts with + /// ZBADVERSION and we safely do nothing — the winning terminal state stands. ops.emplace_back(zkutil::makeSetRequest( status_path, String(magic_enum::enum_name(ExportReplicatedMergeTreePartitionTaskEntry::Status::FAILED)).data(), @@ -367,21 +408,16 @@ namespace ExportPartitionUtils const auto rc = zk->tryMulti(ops, responses); if (rc != Coordination::Error::ZOK) { - /// Any error here (ZBADVERSION on /status race or counter race, ZNODEEXISTS on - /// lazy-create race, ZNONODE if someone removed the task concurrently) is - /// non-fatal: the next attempt re-reads /status and either skips (terminal - /// state won) or retries the bookkeeping. Worst case we delay FAILED by one - /// poll cycle, which matches the best-effort property of the existing counters. LOG_WARNING(log, "ExportPartition: Failed to persist commit failure bookkeeping for {}: {}", entry_path, rc); return false; } LOG_INFO(log, - "ExportPartition: Commit failure recorded for {} (attempt {}/{}){}", - entry_path, attempts, max_attempts, - exhausted ? ", task transitioned to FAILED" : ""); + "ExportPartition: Commit failure recorded for {} (code {}){}", + entry_path, exception_code, + non_retryable ? ", task transitioned to FAILED (non-retryable)" : ", will retry until task timeout"); - return exhausted; + return non_retryable; } void appendExceptionOps( @@ -425,10 +461,24 @@ namespace ExportPartitionUtils entry.time = ::time(nullptr); entry.count += 1; - if (leaf_exists) - ops.emplace_back(zkutil::makeSetRequest(last_exception_path, entry.toJsonString(), -1)); - else - ops.emplace_back(zkutil::makeCreateRequest(last_exception_path, entry.toJsonString(), zkutil::CreateMode::Persistent)); + if (!leaf_exists) + { + /// Materialize the leaf out-of-band (idempotently) so the op we hand back to the + /// caller's atomic multi is always a conflict-free Set. Two failing parts on the + /// same replica whose first failures race would both pick Create here; one of the + /// enclosing multis would then abort with ZNODEEXISTS and roll back its own part + /// lock removal, stranding that part behind its ephemeral lock until session loss + /// or task timeout. A peer thread winning this create (ZNODEEXISTS) is benign. + ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperRequests); + ProfileEvents::increment(ProfileEvents::ExportPartitionZooKeeperCreate); + const auto create_code = zk->tryCreate(last_exception_path, entry.toJsonString(), zkutil::CreateMode::Persistent); + if (create_code != Coordination::Error::ZOK && create_code != Coordination::Error::ZNODEEXISTS) + LOG_INFO(log, "ExportPartition: could not pre-create last_exception leaf {}: {}", last_exception_path.string(), create_code); + } + + /// Always a version -1 Set: it can neither conflict with a peer's create nor abort the + /// enclosing multi, so the lock-release / FAILED-set ops it accompanies always commit. + ops.emplace_back(zkutil::makeSetRequest(last_exception_path, entry.toJsonString(), -1)); } #if USE_AVRO @@ -584,7 +634,7 @@ namespace ExportPartitionUtils const auto & source_column = source_columns[i]; const auto & destination_column = destination_columns[i]; if (!canBeSafelyCast(source_column.type, destination_column.type)) - throw Exception(ErrorCodes::BAD_ARGUMENTS, + throw Exception(ErrorCodes::INCOMPATIBLE_COLUMNS, "Cannot export to {}: column '{}' requires a lossy cast from {} to {}, " "which may change values. Set `export_merge_tree_part_allow_lossy_cast = 1` " "to allow lossy casts during export.", diff --git a/src/Storages/MergeTree/ExportPartitionUtils.h b/src/Storages/MergeTree/ExportPartitionUtils.h index 0434bc59a2cb..0bb8acb9bda4 100644 --- a/src/Storages/MergeTree/ExportPartitionUtils.h +++ b/src/Storages/MergeTree/ExportPartitionUtils.h @@ -23,6 +23,8 @@ struct ExportReplicatedMergeTreePartitionManifest; namespace ExportPartitionUtils { + bool isNonRetryableExportError(int code); + std::vector getExportedPaths(const LoggerPtr & log, const zkutil::ZooKeeperPtr & zk, const std::string & export_path); ContextPtr getContextCopyWithTaskSettings(const ContextPtr & context, const ExportReplicatedMergeTreePartitionManifest & manifest); @@ -51,18 +53,19 @@ namespace ExportPartitionUtils /// Handles a commit-phase failure for a replicated partition export: /// - records the exception via appendExceptionOps in the same multi - /// - increments /commit_attempts (lazy-created) - /// - sets /status to FAILED once attempts >= max_attempts + /// - if `exception_code` is non-retryable (see isNonRetryableExportError), sets + /// /status to FAILED (version-checked against the PENDING read) + /// - otherwise leaves the task PENDING so the commit is retried (by the next + /// last-part success or deferred-commit recovery) until the absolute task timeout /// - /// The counter is a best-effort, non-atomic get+set(-1). Concurrent failing - /// commits may under-count by one (FAILED may fire one retry later than the - /// threshold), which is acceptable. + /// There is no per-task commit-attempt budget: retryable commit failures retry until + /// success or timeout, matching the per-part retry semantics. /// /// Returns true if this call transitioned the task to FAILED. bool handleCommitFailure( const zkutil::ZooKeeperPtr & zk, const std::string & entry_path, - size_t max_attempts, + int exception_code, const std::string & replica_name, const std::string & exception_message, const LoggerPtr & log); @@ -77,10 +80,6 @@ namespace ExportPartitionUtils /// leaf. Within a single replica the count increment is best-effort and /// non-atomic (synchronous tryGet + Set with version -1); concurrent /// failing writers may under-count by one, which is accepted. - /// - /// Intended to be combined with additional ops (for example a version-guarded - /// status set) and executed as a single `tryMulti` so the exception record and - /// the accompanying state transition commit atomically. void appendExceptionOps( Coordination::Requests & ops, const zkutil::ZooKeeperPtr & zk, diff --git a/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergMetadata.cpp b/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergMetadata.cpp index 9efa8ca1bc04..409561506624 100644 --- a/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergMetadata.cpp +++ b/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergMetadata.cpp @@ -117,6 +117,12 @@ extern const int ICEBERG_SPECIFICATION_VIOLATION; extern const int S3_ERROR; extern const int TABLE_ALREADY_EXISTS; extern const int SUPPORT_IS_DISABLED; +<<<<<<< HEAD +======= +extern const int INCORRECT_DATA; +extern const int METADATA_MISMATCH; +extern const int UNFINISHED; +>>>>>>> 091776b7b65 (Merge pull request #1984 from Altinity/export-partition-retry-backoff) } namespace Setting @@ -1456,7 +1462,7 @@ namespace { /// Find the partition spec object with the given spec-id inside a metadata JSON document. -/// Throws BAD_ARGUMENTS if the spec is not found (indicates metadata/spec-id mismatch). +/// Throws METADATA_MISMATCH if the spec is not found (indicates metadata/spec-id mismatch). Poco::JSON::Object::Ptr lookupPartitionSpec(const Poco::JSON::Object::Ptr & meta, Int64 spec_id) { auto specs = meta->getArray(Iceberg::f_partition_specs); @@ -1466,7 +1472,7 @@ Poco::JSON::Object::Ptr lookupPartitionSpec(const Poco::JSON::Object::Ptr & meta if (spec->getValue(Iceberg::f_spec_id) == spec_id) return spec; } - throw Exception(ErrorCodes::BAD_ARGUMENTS, + throw Exception(ErrorCodes::METADATA_MISMATCH, "Partition spec with id {} not found in table metadata", spec_id); } @@ -1480,7 +1486,7 @@ Poco::JSON::Object::Ptr lookupSchema(const Poco::JSON::Object::Ptr & meta, Int64 return schema; } - throw Exception(ErrorCodes::BAD_ARGUMENTS, + throw Exception(ErrorCodes::METADATA_MISMATCH, "Schema with id {} not found in table metadata", schema_id); } @@ -1658,13 +1664,13 @@ bool IcebergMetadata::commitImportPartitionTransactionImpl( /// the caller has to restart the export from scratch. const auto new_schema_id = metadata->getValue(Iceberg::f_current_schema_id); if (new_schema_id != original_schema_id) - throw Exception(ErrorCodes::NOT_IMPLEMENTED, + throw Exception(ErrorCodes::METADATA_MISMATCH, "Table schema changed during export (expected schema {}, got {}). Restart the export operation.", original_schema_id, new_schema_id); const Int64 new_partition_spec_id = metadata->getValue(Iceberg::f_default_spec_id); if (new_partition_spec_id != partition_spec_id) - throw Exception(ErrorCodes::NOT_IMPLEMENTED, + throw Exception(ErrorCodes::METADATA_MISMATCH, "Partition spec changed during export (expected spec {}, got {}). Restart the export operation.", partition_spec_id, new_partition_spec_id); @@ -1879,14 +1885,14 @@ void IcebergMetadata::commitExportPartitionTransaction( /// The exported data files and partition values were produced against the original spec; const auto latest_schema_id = metadata->getValue(Iceberg::f_current_schema_id); if (latest_schema_id != original_schema_id) - throw Exception(ErrorCodes::NOT_IMPLEMENTED, + throw Exception(ErrorCodes::METADATA_MISMATCH, "Table schema changed before export could commit (expected schema {}, got {}). " "Restart the export operation.", original_schema_id, latest_schema_id); const auto latest_spec_id = metadata->getValue(Iceberg::f_default_spec_id); if (latest_spec_id != partition_spec_id) - throw Exception(ErrorCodes::NOT_IMPLEMENTED, + throw Exception(ErrorCodes::METADATA_MISMATCH, "Partition spec changed before export could commit (expected spec {}, got {}). " "Restart the export operation.", partition_spec_id, latest_spec_id); @@ -1978,7 +1984,7 @@ void IcebergMetadata::commitExportPartitionTransaction( ++attempt; } - throw Exception(ErrorCodes::NOT_IMPLEMENTED, + throw Exception(ErrorCodes::UNFINISHED, "Failed to commit export partition transaction after {} attempts due to repeated metadata conflicts.", attempt); } diff --git a/src/Storages/StorageReplicatedMergeTree.cpp b/src/Storages/StorageReplicatedMergeTree.cpp index e49e671d064f..760cfbcb400a 100644 --- a/src/Storages/StorageReplicatedMergeTree.cpp +++ b/src/Storages/StorageReplicatedMergeTree.cpp @@ -225,7 +225,8 @@ namespace Setting extern const SettingsBool update_sequential_consistency; extern const SettingsBool allow_experimental_export_merge_tree_part; extern const SettingsBool export_merge_tree_partition_force_export; - extern const SettingsUInt64 export_merge_tree_partition_max_retries; + extern const SettingsUInt64 export_merge_tree_partition_retry_initial_backoff_seconds; + extern const SettingsUInt64 export_merge_tree_partition_retry_max_backoff_seconds; extern const SettingsUInt64 export_merge_tree_partition_task_timeout_seconds; extern const SettingsBool output_format_parallel_formatting; extern const SettingsBool output_format_parquet_parallel_encoding; @@ -4739,6 +4740,11 @@ void StorageReplicatedMergeTree::exportMergeTreePartitionUpdatingTask() void StorageReplicatedMergeTree::selectPartsToExport() { auto component_guard = Coordination::setCurrentComponent("StorageReplicatedMergeTree::selectPartsToExport"); + + /// Default tick interval; may be shortened below if a part's back-off expires sooner. + static constexpr int64_t default_reschedule_ms = 1000 * 5; + int64_t reschedule_ms = default_reschedule_ms; + try { if (parts_mover.moves_blocker.isCancelled()) @@ -4747,7 +4753,16 @@ void StorageReplicatedMergeTree::selectPartsToExport() } else { - export_merge_tree_partition_task_scheduler->run(); + const auto earliest_backoff_retry = export_merge_tree_partition_task_scheduler->run(); + + /// If a part is only waiting on its back-off deadline and that deadline is sooner than + /// the default tick, wake up earlier so the retry is not delayed by up to a full tick. + if (earliest_backoff_retry) + { + const auto now = time(nullptr); + const int64_t until_ms = (static_cast(*earliest_backoff_retry) - static_cast(now)) * 1000; + reschedule_ms = std::clamp(until_ms, 0, default_reschedule_ms); + } } } catch (...) @@ -4755,7 +4770,7 @@ void StorageReplicatedMergeTree::selectPartsToExport() tryLogCurrentException(log, __PRETTY_FUNCTION__); } - export_merge_tree_partition_select_task->scheduleAfter(1000 * 5); + export_merge_tree_partition_select_task->scheduleAfter(reschedule_ms); } void StorageReplicatedMergeTree::exportMergeTreePartitionStatusHandlingTask() @@ -8708,7 +8723,8 @@ void StorageReplicatedMergeTree::exportPartitionToTable(const PartitionCommand & manifest.number_of_parts = part_names.size(); manifest.parts = part_names; manifest.create_time = time(nullptr); - manifest.max_retries = query_context->getSettingsRef()[Setting::export_merge_tree_partition_max_retries]; + manifest.retry_initial_backoff_seconds = query_context->getSettingsRef()[Setting::export_merge_tree_partition_retry_initial_backoff_seconds]; + manifest.retry_max_backoff_seconds = query_context->getSettingsRef()[Setting::export_merge_tree_partition_retry_max_backoff_seconds]; manifest.task_timeout_seconds = query_context->getSettingsRef()[Setting::export_merge_tree_partition_task_timeout_seconds]; manifest.max_threads = query_context->getSettingsRef()[Setting::max_threads]; manifest.parallel_formatting = query_context->getSettingsRef()[Setting::output_format_parallel_formatting]; @@ -8795,7 +8811,6 @@ void StorageReplicatedMergeTree::exportPartitionToTable(const PartitionCommand & ExportReplicatedMergeTreePartitionProcessingPartEntry entry; entry.status = ExportReplicatedMergeTreePartitionProcessingPartEntry::Status::PENDING; entry.part_name = part; - entry.retry_count = 0; ops.emplace_back(zkutil::makeCreateRequest( fs::path(partition_exports_path) / "processing" / part, diff --git a/src/Storages/System/StorageSystemReplicatedPartitionExports.cpp b/src/Storages/System/StorageSystemReplicatedPartitionExports.cpp index 9e8faab689d6..9aa9d1846273 100644 --- a/src/Storages/System/StorageSystemReplicatedPartitionExports.cpp +++ b/src/Storages/System/StorageSystemReplicatedPartitionExports.cpp @@ -28,6 +28,14 @@ ColumnsDescription StorageSystemReplicatedPartitionExports::getColumnsDescriptio }, Names{"replica", "message", "part", "time", "count"}); + auto backoff_tuple = std::make_shared( + DataTypes{ + std::make_shared(), + std::make_shared(), + std::make_shared(), + }, + Names{"part", "attempts", "next_retry_time"}); + return ColumnsDescription { {"source_database", std::make_shared(), "Name of the source database."}, @@ -47,6 +55,8 @@ ColumnsDescription StorageSystemReplicatedPartitionExports::getColumnsDescriptio "Per-replica last exception entries. Each tuple records the most recent exception observed by that replica plus a best-effort within-replica count. Empty array if no replica has reported an exception for this task."}, {"exception_count", std::make_shared(), "Sum of per-replica exception counts. Each replica owns its own count, so the sum is exact w.r.t. the in-memory snapshot; within-replica updates remain best-effort and may under-count by one under concurrent failures."}, + {"local_backoff_per_part", std::make_shared(backoff_tuple), + "Per-part retry back-off local to this replica: parts currently waiting before their next attempt, with attempt count and the next eligible time. Not shared across replicas; empty if no part is backing off."}, }; } @@ -152,6 +162,12 @@ void StorageSystemReplicatedPartitionExports::fillData(MutableColumns & res_colu per_replica.push_back(Tuple{ex.replica, ex.message, ex.part, ex.time, ex.count}); res_columns[i++]->insert(per_replica); res_columns[i++]->insert(info.exception_count); + + Array backoff_array; + backoff_array.reserve(info.backoff_per_part.size()); + for (const auto & b : info.backoff_per_part) + backoff_array.push_back(Tuple{b.part, b.attempts, b.next_retry_time}); + res_columns[i++]->insert(backoff_array); } } } diff --git a/src/Storages/System/StorageSystemReplicatedPartitionExports.h b/src/Storages/System/StorageSystemReplicatedPartitionExports.h index a8666374a7f0..09d7d3eaf9a2 100644 --- a/src/Storages/System/StorageSystemReplicatedPartitionExports.h +++ b/src/Storages/System/StorageSystemReplicatedPartitionExports.h @@ -29,6 +29,15 @@ struct ReplicatedPartitionExportInfo /// single replica the count is best-effort (concurrent failing writers may under- /// count by one), matching the documented column semantics. size_t exception_count = 0; + + struct PartBackoffEntry + { + String part; + size_t attempts = 0; + time_t next_retry_time = 0; + }; + /// Parts of this task currently backing off (local to this replica). Empty if none. + std::vector backoff_per_part; }; class StorageSystemReplicatedPartitionExports final : public IStorageSystemOneBlock diff --git a/tests/integration/test_export_replicated_mt_partition_to_iceberg/test.py b/tests/integration/test_export_replicated_mt_partition_to_iceberg/test.py index 62a8a2196313..383739b6b0c7 100644 --- a/tests/integration/test_export_replicated_mt_partition_to_iceberg/test.py +++ b/tests/integration/test_export_replicated_mt_partition_to_iceberg/test.py @@ -14,6 +14,7 @@ make_iceberg_s3, make_rmt, unique_suffix, + wait_for_exception_count, wait_for_export_status, wait_for_export_to_start, ) @@ -229,43 +230,31 @@ def test_export_partition_all_to_iceberg(cluster): def test_failure_is_logged_in_system_table(cluster): """ - When S3 is unreachable the export must be marked FAILED in - system.replicated_partition_exports with a non-zero exception_count. + When a part export fails with a non-retryable error the export must be marked + FAILED in system.replicated_partition_exports with a non-zero exception_count. + + Uses the export_part_non_retryable_throw failpoint (throws BAD_ARGUMENTS, a + denylisted code) so the task fails fast without consuming any timeout budget. """ node = cluster.instances["replica1"] - minio_ip = cluster.minio_ip - minio_port = cluster.minio_port uid = unique_suffix() mt_table = f"mt_{uid}" iceberg_table = f"iceberg_{uid}" - setup_tables(cluster, mt_table, iceberg_table, nodes=["replica1"], - s3_retry_attempts=1) - - node.query(f"SYSTEM STOP MOVES {mt_table}") - - node.query(f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {iceberg_table} SETTINGS export_merge_tree_partition_max_retries = 1, allow_insert_into_iceberg = 1") - - with PartitionManager() as pm: - pm.add_rule({ - "instance": node, - "destination": node.ip_address, - "protocol": "tcp", - "source_port": minio_port, - "action": "REJECT --reject-with tcp-reset", - }) - pm.add_rule({ - "instance": node, - "destination": minio_ip, - "protocol": "tcp", - "destination_port": minio_port, - "action": "REJECT --reject-with tcp-reset", - }) + setup_tables(cluster, mt_table, iceberg_table, nodes=["replica1"]) - node.query(f"SYSTEM START MOVES {mt_table}") + node.query("SYSTEM ENABLE FAILPOINT export_part_non_retryable_throw") + try: + node.query( + f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {iceberg_table}", + settings={"allow_insert_into_iceberg": 1}, + ) - wait_for_export_status(node, mt_table, iceberg_table, "2020", "FAILED", timeout=60) + # short timeout to exercise the fast fail path for non retryable errors + wait_for_export_status(node, mt_table, iceberg_table, "2020", "FAILED", timeout=20) + finally: + node.query("SYSTEM DISABLE FAILPOINT export_part_non_retryable_throw") status = node.query( f""" @@ -287,6 +276,9 @@ def test_failure_is_logged_in_system_table(cluster): ).strip()) assert exception_count > 0, "Expected non-zero exception_count in system.replicated_partition_exports" + count = int(node.query(f"SELECT count() FROM {iceberg_table}").strip()) + assert count == 0, f"Expected 0 rows in Iceberg table after a failed export, got {count}" + def test_inject_short_living_failures(cluster): """ @@ -306,7 +298,7 @@ def test_inject_short_living_failures(cluster): node.query(f"SYSTEM STOP MOVES {mt_table}") - node.query(f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {iceberg_table} SETTINGS export_merge_tree_partition_max_retries = 100, allow_insert_into_iceberg = 1") + node.query(f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {iceberg_table} SETTINGS allow_insert_into_iceberg = 1") with PartitionManager() as pm: pm.add_rule({ @@ -355,6 +347,196 @@ def test_inject_short_living_failures(cluster): assert exception_count >= 1, "Expected at least one transient exception to be recorded" +def test_export_partition_retryable_error_killed_on_timeout(cluster): + """ + A retryable part-export error (here FAULT_INJECTED via export_part_retryable_throw) + must NOT fail the task on a retry budget: there is no retry budget anymore, so the + part keeps retrying until the absolute task timeout fires and the task is KILLED. + """ + node = cluster.instances["replica1"] + + uid = unique_suffix() + mt_table = f"mt_{uid}" + iceberg_table = f"iceberg_{uid}" + + setup_tables(cluster, mt_table, iceberg_table, nodes=["replica1"]) + + node.query("SYSTEM ENABLE FAILPOINT export_part_retryable_throw") + try: + # Under the old budget model a small retry budget would fail the task after the + # first retry. With the new model there is no budget and only the 5s timeout fails it. + node.query( + f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {iceberg_table}" + f" SETTINGS export_merge_tree_partition_task_timeout_seconds = 5," + f" allow_insert_into_iceberg = 1" + ) + + # Give the scheduler time to attempt and fail the part several times. The old + # budget would already have transitioned the task to FAILED by now. + time.sleep(15) + status = node.query( + f"SELECT status FROM system.replicated_partition_exports" + f" WHERE source_table = '{mt_table}'" + f" AND destination_table = '{iceberg_table}'" + f" AND partition_id = '2020'" + ).strip() + assert status != "FAILED", ( + f"Retryable failures must not fail the task on a budget, got status {status!r}" + ) + + # The timeout (5s) is past; KILLED fires on the next manifest-updater poll cycle. + wait_for_export_status( + node, mt_table, iceberg_table, "2020", "KILLED", timeout=90 + ) + finally: + node.query("SYSTEM DISABLE FAILPOINT export_part_retryable_throw") + + exception_count = int(node.query( + f"SELECT any(exception_count) FROM system.replicated_partition_exports" + f" WHERE source_table = '{mt_table}'" + f" AND destination_table = '{iceberg_table}'" + f" AND partition_id = '2020'" + ).strip()) + assert exception_count > 0, "Expected at least one retryable exception to be recorded" + + count = int(node.query(f"SELECT count() FROM {iceberg_table}").strip()) + assert count == 0, f"Expected 0 rows in Iceberg table after a killed export, got {count}" + + +def test_export_partition_retryable_error_recovers_after_failpoint_cleared(cluster): + """ + A retryable part-export error must keep the task PENDING (not FAILED) while the + failure persists, applying a per-replica back-off between attempts. Once the + failure clears the export completes successfully — proving the back-off only + spaces retries out and never permanently blocks progress. + """ + node = cluster.instances["replica1"] + + uid = unique_suffix() + mt_table = f"mt_{uid}" + iceberg_table = f"iceberg_{uid}" + + setup_tables(cluster, mt_table, iceberg_table, nodes=["replica1"]) + + node.query("SYSTEM ENABLE FAILPOINT export_part_retryable_throw") + try: + node.query( + f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {iceberg_table}" + f" SETTINGS export_merge_tree_partition_retry_initial_backoff_seconds = 1," + f" export_merge_tree_partition_retry_max_backoff_seconds = 2," + f" allow_insert_into_iceberg = 1" + ) + + # Wait until at least one retryable failure has been recorded; the task must + # still be PENDING (retrying), never FAILED. + wait_for_exception_count(node, mt_table, iceberg_table, "2020", + min_exception_count=1, timeout=60) + status = node.query( + f"SELECT status FROM system.replicated_partition_exports" + f" WHERE source_table = '{mt_table}'" + f" AND destination_table = '{iceberg_table}'" + f" AND partition_id = '2020'" + ).strip() + assert status == "PENDING", ( + f"Retryable failures must keep the task PENDING, got status {status!r}" + ) + finally: + node.query("SYSTEM DISABLE FAILPOINT export_part_retryable_throw") + + # With the failpoint cleared the next retry succeeds and the export completes. + wait_for_export_status(node, mt_table, iceberg_table, "2020", "COMPLETED", timeout=90) + + count = int(node.query(f"SELECT count() FROM {iceberg_table} WHERE year = 2020").strip()) + assert count == 3, f"Expected 3 rows after recovery, got {count}" + + +def test_export_partition_local_backoff_does_not_block_other_replica(cluster): + """ + Back-off is per-replica and in-memory: a part that one replica keeps failing on + (and therefore puts into its local back-off) must NOT be prevented from being + exported by another replica. This is the whole reason the back-off is local + rather than distributed in ZooKeeper. + + replica1 is given a persistent *retryable* failure (export_part_retryable_throw) + and is the only replica scheduling at first (moves are stopped on replica2). Once + replica1 has recorded a failure and a local back-off entry, replica2's scheduler + is enabled. Because the failpoint stays active on replica1 the whole time, the + only way the export can reach COMPLETED is replica2 picking up the very part that + replica1 keeps failing — proving the back-off does not leak across replicas. + """ + replica1 = cluster.instances["replica1"] + replica2 = cluster.instances["replica2"] + + uid = unique_suffix() + mt_table = f"mt_{uid}" + iceberg_table = f"iceberg_{uid}" + + setup_tables(cluster, mt_table, iceberg_table, nodes=["replica1", "replica2"]) + + # Phase 1: only replica1 schedules. Stop the export scheduler on replica2 so the + # part is guaranteed to be attempted (and fail) on replica1 first. + replica2.query(f"SYSTEM STOP MOVES {mt_table}") + + replica1.query("SYSTEM ENABLE FAILPOINT export_part_retryable_throw") + try: + replica1.query( + f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {iceberg_table}" + f" SETTINGS export_merge_tree_partition_retry_initial_backoff_seconds = 1," + f" export_merge_tree_partition_retry_max_backoff_seconds = 2," + f" allow_insert_into_iceberg = 1" + ) + + # replica1 attempts the part, fails (retryable), and enters local back-off. + # The task must stay PENDING — there is no retry budget to fail it. + wait_for_exception_count(replica1, mt_table, iceberg_table, "2020", + min_exception_count=1, timeout=60) + + wait_for_export_status(replica1, mt_table, iceberg_table, "2020", "PENDING", timeout=60) + + # The back-off entry must be observable on replica1 (the failing replica). + deadline = time.time() + 90 + backoff_replica1 = "0" + while time.time() < deadline: + backoff_replica1 = replica1.query( + f"SELECT length(local_backoff_per_part) FROM system.replicated_partition_exports" + f" WHERE source_table = '{mt_table}'" + f" AND destination_table = '{iceberg_table}'" + f" AND partition_id = '2020'" + ).strip() + if backoff_replica1 not in ("", "0"): + break + time.sleep(0.5) + assert backoff_replica1 not in ("", "0"), ( + "Expected replica1 to carry a local back-off entry for the failing part, " + f"got {backoff_replica1!r}" + ) + + # ... and it must NOT have leaked to replica2, which never attempted the part. + # This is the core assertion: local back-off state is not shared across replicas. + backoff_replica2 = replica2.query( + f"SELECT length(local_backoff_per_part) FROM system.replicated_partition_exports" + f" WHERE source_table = '{mt_table}'" + f" AND destination_table = '{iceberg_table}'" + f" AND partition_id = '2020'" + ).strip() + + assert backoff_replica2 in ("", "0"), ( + f"replica2 must not carry replica1's local back-off, got {backoff_replica2!r}" + ) + + # Phase 2: enable replica2's scheduler. replica1 keeps failing (the failpoint + # is still active), so completion can only come from replica2 exporting the + # part that replica1 is backing off on. + replica2.query(f"SYSTEM START MOVES {mt_table}") + + wait_for_export_status(replica2, mt_table, iceberg_table, "2020", "COMPLETED", timeout=60) + finally: + replica1.query("SYSTEM DISABLE FAILPOINT export_part_retryable_throw") + + count = int(replica2.query(f"SELECT count() FROM {iceberg_table} WHERE year = 2020").strip()) + assert count == 3, f"Expected 3 rows after replica2 completed the export, got {count}" + + def test_export_partition_scheduler_skipped_when_moves_stopped(cluster): """ Verify that selectPartsToExport() skips the scheduler entirely when moves @@ -427,7 +609,7 @@ def test_export_partition_resumes_after_stop_moves(cluster): node.query( f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {iceberg_table}" - f" SETTINGS export_merge_tree_partition_max_retries = 50, allow_insert_into_iceberg = 1" + f" SETTINGS allow_insert_into_iceberg = 1" ) wait_for_export_to_start(node, mt_table, iceberg_table, "2020") @@ -472,7 +654,7 @@ def test_export_partition_resumes_after_stop_moves_during_export(cluster): node.query( f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {iceberg_table}" - f" SETTINGS export_merge_tree_partition_max_retries = 50, allow_insert_into_iceberg = 1") + f" SETTINGS allow_insert_into_iceberg = 1") wait_for_export_to_start(node, mt_table, iceberg_table, "2020") @@ -748,10 +930,17 @@ def test_partition_key_compatibility_check(cluster): def test_export_data_files_are_not_cleaned_up_on_commit_failure(cluster): """ - Verify that the data files are not cleaned up on commit failure and the export is retried. - This is to avoid data loss. - - If the data files were cleaned up, a retry would commit a new snapshot that points to dangling references. + Verify that a commit failure does not delete the already-written data files. + `cleanup` only removes the manifest entry / manifest list, never the data files + (a peer replica might still commit the same transaction). This guards against + data loss / dangling references. + + The iceberg_writes_non_retry_cleanup failpoint throws BAD_ARGUMENTS while writing + the manifest entry, after the data files have been written. BAD_ARGUMENTS is a + non-retryable error code, so the task transitions to FAILED; we then confirm the + exported data files are still physically present in object storage by reading + them directly (the Iceberg manifests were removed by cleanup, so we glob the raw + parquet data files instead). """ node = cluster.instances["replica1"] uid = unique_suffix() @@ -760,15 +949,28 @@ def test_export_data_files_are_not_cleaned_up_on_commit_failure(cluster): setup_tables(cluster, mt_table, iceberg_table, nodes=["replica1"]) node.query("SYSTEM ENABLE FAILPOINT iceberg_writes_non_retry_cleanup") - - node.query( - f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {iceberg_table}", - settings={"allow_insert_into_iceberg": 1}, + try: + node.query( + f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {iceberg_table}", + settings={"allow_insert_into_iceberg": 1}, + ) + # BAD_ARGUMENTS from the commit phase is non-retryable -> the task fails fast. + wait_for_export_status(node, mt_table, iceberg_table, "2020", "FAILED", timeout=60) + finally: + node.query("SYSTEM DISABLE FAILPOINT iceberg_writes_non_retry_cleanup") + + # The data files were written before the commit failure; cleanup must have left + # them intact. Read them straight from object storage (bypassing the Iceberg + # metadata, which cleanup removed) and confirm all 3 exported rows survive. + rows = int(node.query( + f"SELECT count() FROM s3(" + f"'http://minio1:9001/root/data/{iceberg_table}/**.parquet', " + f"'minio', 'ClickHouse_Minio_P@ssw0rd', 'Parquet')" + ).strip()) + assert rows == 3, ( + f"Expected the 3 exported rows to still exist as data files after a failed " + f"commit (data files must not be cleaned up), got {rows}" ) - wait_for_export_status(node, mt_table, iceberg_table, "2020", "COMPLETED") - - count = int(node.query(f"SELECT count() FROM {iceberg_table} WHERE year = 2020").strip()) - assert count == 3, f"Expected 3 rows after first export, got {count}" def test_post_publish_exception_preserves_snapshot(cluster): @@ -827,10 +1029,9 @@ def test_export_task_timeout_kills_stuck_pending_task(cluster): descriptive last_exception. The export_partition_commit_always_throw failpoint wedges the task in the - commit retry loop (REGULAR failpoint, fires on every commit attempt). A very - large max_retries budget prevents the commit-attempts path from transitioning - to FAILED before the timeout fires, so the timeout branch in tryCleanup is - the actual mechanism under test. + commit retry loop (REGULAR failpoint, fires on every commit attempt) with a + retryable error, so the task never fails on its own and the timeout branch in + tryCleanup is the actual mechanism under test. """ node = cluster.instances["replica1"] uid = unique_suffix() @@ -844,7 +1045,6 @@ def test_export_task_timeout_kills_stuck_pending_task(cluster): node.query( f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {iceberg_table}" f" SETTINGS export_merge_tree_partition_task_timeout_seconds = 5," - f" export_merge_tree_partition_max_retries = 1000000," f" allow_insert_into_iceberg = 1" ) @@ -1132,7 +1332,7 @@ def test_export_partition_with_castable_narrowing_values_fit(cluster): def test_export_partition_lossy_cast_rejected_without_optin(cluster): """A lossy narrowing (id Int64 -> Int32) is rejected synchronously with - BAD_ARGUMENTS unless export_merge_tree_part_allow_lossy_cast is set.""" + INCOMPATIBLE_COLUMNS unless export_merge_tree_part_allow_lossy_cast is set.""" node = cluster.instances["replica1"] uid = unique_suffix() @@ -1148,7 +1348,7 @@ def test_export_partition_lossy_cast_rejected_without_optin(cluster): f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {iceberg_table} " f"SETTINGS allow_insert_into_iceberg = 1" ) - assert "BAD_ARGUMENTS" in error, f"Expected BAD_ARGUMENTS, got: {error!r}" + assert "INCOMPATIBLE_COLUMNS" in error, f"Expected INCOMPATIBLE_COLUMNS, got: {error!r}" assert "lossy cast" in error, f"Expected 'lossy cast' in error, got: {error!r}" count = int(node.query(f"SELECT count() FROM {iceberg_table}").strip()) @@ -1158,8 +1358,13 @@ def test_export_partition_lossy_cast_rejected_without_optin(cluster): def test_export_partition_runtime_cast_failure_propagates_async(cluster): """A String value that cannot be parsed as the destination Int32 passes the synchronous lossy-cast gate (with export_merge_tree_part_allow_lossy_cast = 1) but - fails at runtime in the async worker, marking the export FAILED and leaving Iceberg - empty. + fails at runtime in the async worker with CANNOT_PARSE_TEXT. That is a deterministic + value-conversion error on the part's immutable data — retrying the same part can + never succeed — so it is classified as non-retryable and fails the whole task fast, + without waiting for the absolute task timeout, leaving Iceberg empty. + + The task timeout is left at its large default, so reaching FAILED quickly proves the + transition is driven by error classification rather than by a timeout. (Integer overflow is not used because the internal cast uses CastType::nonAccurate, which wraps rather than throwing.) @@ -1177,10 +1382,12 @@ def test_export_partition_runtime_cast_failure_propagates_async(cluster): node.query( f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {iceberg_table} " - f"SETTINGS export_merge_tree_partition_max_retries = 1, allow_insert_into_iceberg = 1, " - f"export_merge_tree_part_allow_lossy_cast = 1" + f"SETTINGS allow_insert_into_iceberg = 1, export_merge_tree_part_allow_lossy_cast = 1" ) + # The runtime parse error (CANNOT_PARSE_TEXT) is non-retryable, so the task fails fast. + # No short timeout is set; FAILED within this window can only come from the + # non-retryable classification, not from the (default, ~1 day) task timeout. wait_for_export_status(node, mt_table, iceberg_table, "2020", "FAILED", timeout=60) exception_count = int(node.query( diff --git a/tests/integration/test_export_replicated_mt_partition_to_object_storage/test.py b/tests/integration/test_export_replicated_mt_partition_to_object_storage/test.py index 8ad265a375ad..44f35925f1e7 100644 --- a/tests/integration/test_export_replicated_mt_partition_to_object_storage/test.py +++ b/tests/integration/test_export_replicated_mt_partition_to_object_storage/test.py @@ -196,11 +196,9 @@ def test_restart_nodes_during_export(cluster): export_queries = f""" ALTER TABLE {mt_table} - EXPORT PARTITION ID '2020' TO TABLE {s3_table} - SETTINGS export_merge_tree_partition_max_retries = 50; + EXPORT PARTITION ID '2020' TO TABLE {s3_table}; ALTER TABLE {mt_table} - EXPORT PARTITION ID '2021' TO TABLE {s3_table} - SETTINGS export_merge_tree_partition_max_retries = 50; + EXPORT PARTITION ID '2021' TO TABLE {s3_table}; """ node.query(export_queries) @@ -289,11 +287,9 @@ def test_kill_export(cluster): export_queries = f""" ALTER TABLE {mt_table} - EXPORT PARTITION ID '2020' TO TABLE {s3_table} - SETTINGS export_merge_tree_partition_max_retries = 50; + EXPORT PARTITION ID '2020' TO TABLE {s3_table}; ALTER TABLE {mt_table} - EXPORT PARTITION ID '2021' TO TABLE {s3_table} - SETTINGS export_merge_tree_partition_max_retries = 50; + EXPORT PARTITION ID '2021' TO TABLE {s3_table}; """ node.query(export_queries) @@ -356,7 +352,6 @@ def test_kill_export_resilient_to_status_handling_failure(cluster): node.query( f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {s3_table}" - f" SETTINGS export_merge_tree_partition_max_retries = 50" ) node.query("SYSTEM ENABLE FAILPOINT export_partition_status_change_throw") @@ -425,9 +420,9 @@ def test_drop_source_table_during_export(cluster): export_queries = f""" ALTER TABLE {mt_table} - EXPORT PARTITION ID '2020' TO TABLE {s3_table} SETTINGS s3_retry_attempts = 500, export_merge_tree_partition_max_retries = 50; + EXPORT PARTITION ID '2020' TO TABLE {s3_table} SETTINGS s3_retry_attempts = 500; ALTER TABLE {mt_table} - EXPORT PARTITION ID '2021' TO TABLE {s3_table} SETTINGS s3_retry_attempts = 500, export_merge_tree_partition_max_retries = 50; + EXPORT PARTITION ID '2021' TO TABLE {s3_table} SETTINGS s3_retry_attempts = 500; """ node.query(export_queries) @@ -517,14 +512,22 @@ def test_failure_is_logged_in_system_table(cluster): } pm.add_rule(pm_rule_reject_requests) + # Blocked MinIO produces transient (retryable) S3 errors. There is no retry + # budget anymore, so the task keeps retrying and is only torn down once the + # absolute task timeout fires (transitioning to KILLED). Use a small timeout + # so the test does not wait for the default (a day). node.query( - f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {s3_table} SETTINGS export_merge_tree_partition_max_retries=1;" + f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {s3_table}" + f" SETTINGS export_merge_tree_partition_task_timeout_seconds = 5;" ) - # Wait so that the export fails - wait_for_export_status(node, mt_table, s3_table, "2020", "FAILED", timeout=60) + # Wait for the timeout to kill the stuck task. The KILL is a Keeper operation + # (MinIO being blocked does not affect it); the status mirror needs roughly one + # manifest-updater poll cycle (~30s) plus watch propagation on top of the 5s + # timeout, so allow a generous budget. + wait_for_export_status(node, mt_table, s3_table, "2020", "KILLED", timeout=90) - # Network restored; verify the export is marked as FAILED in the system table + # Network restored; verify the export is marked as KILLED in the system table # Also verify we captured at least one exception and no commit file exists status = node.query( f""" @@ -535,7 +538,7 @@ def test_failure_is_logged_in_system_table(cluster): """ ) - assert status.strip() == "FAILED", f"Expected FAILED status, got: {status!r}" + assert status.strip() == "KILLED", f"Expected KILLED status, got: {status!r}" exception_count = node.query( f""" @@ -588,9 +591,10 @@ def test_inject_short_living_failures(cluster): } pm.add_rule(pm_rule_reject_requests) - # set big max_retries so that the export does not fail completely + # Transient (retryable) failures never fail the task on a budget; it keeps + # retrying until the network is restored and the export completes. node.query( - f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {s3_table} SETTINGS export_merge_tree_partition_max_retries=100;" + f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {s3_table};" ) # wait for at least one exception to occur, but not enough to finish the export. @@ -627,6 +631,85 @@ def test_inject_short_living_failures(cluster): assert int(exception_count.strip()) >= 1, "Expected at least one exception" +def test_export_partition_retry_backoff(cluster): + """Verify the per-replica in-memory exponential back-off between failed part exports. + + The back-off is local in-memory state (no ZooKeeper retry_count / next_retry_time + anymore), so it is not directly observable; instead we observe its effect. With a + large back-off, a part that keeps failing (object storage blocked) is parked for the + back-off window after its first failure and must NOT be retried on every ~5s + scheduler tick. We assert that exception_count stays low across a window that spans + several ticks. Once the network is restored and the back-off elapses, the export + completes (there is no retry budget to exhaust).""" + skip_if_remote_database_disk_enabled(cluster) + node = cluster.instances["replica1"] + + postfix = str(uuid.uuid4()).replace("-", "_") + mt_table = f"retry_backoff_mt_table_{postfix}" + s3_table = f"retry_backoff_s3_table_{postfix}" + + create_tables_and_insert_data(node, mt_table, s3_table, "replica1") + + # Large back-off so a single failed attempt parks the part well beyond the + # ~5s scheduler tick. Kept moderate so the export can still complete promptly + # once the network is restored. + initial_backoff_seconds = 30 + max_backoff_seconds = 30 + + minio_ip = cluster.minio_ip + minio_port = cluster.minio_port + + with PartitionManager() as pm: + # Block responses from MinIO (source_port matches MinIO service) + pm.add_rule({ + "instance": node, + "destination": node.ip_address, + "protocol": "tcp", + "source_port": minio_port, + "action": "REJECT --reject-with tcp-reset", + }) + # Also block requests to MinIO to fail fast + pm.add_rule({ + "instance": node, + "destination": minio_ip, + "protocol": "tcp", + "destination_port": minio_port, + "action": "REJECT --reject-with tcp-reset", + }) + + node.query( + f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {s3_table} " + f"SETTINGS export_merge_tree_partition_retry_initial_backoff_seconds = {initial_backoff_seconds}, " + f"export_merge_tree_partition_retry_max_backoff_seconds = {max_backoff_seconds}" + ) + + # Wait until the first failure is recorded. + count_after_first = wait_for_exception_count( + node, mt_table, s3_table, "2020", min_exception_count=1, timeout=60 + ) + + # While the part is backing off (~30s) it must not be retried again. Observe + # across a window that spans several scheduler ticks: without back-off the + # ~5s tick would add roughly five more failures, so a small increase proves + # the back-off is pacing retries. + time.sleep(25) + count_during_backoff = int(node.query( + f"SELECT exception_count FROM system.replicated_partition_exports" + f" WHERE source_table = '{mt_table}'" + f" AND destination_table = '{s3_table}'" + f" AND partition_id = '2020'" + ).strip()) + assert count_during_backoff - count_after_first <= 2, ( + f"exception_count jumped during the back-off window: " + f"{count_after_first} -> {count_during_backoff}; back-off was not applied" + ) + + # Network restored; once the back-off elapses the export should complete because + # there is no retry budget to exhaust. + wait_for_export_status(node, mt_table, s3_table, "2020", "COMPLETED", timeout=120) + assert node.query(f"SELECT count() FROM {s3_table} WHERE year = 2020") == "3\n", "Export did not succeed" + + def test_export_partition_file_already_exists_policy(cluster): node = cluster.instances["replica1"] @@ -694,10 +777,11 @@ def test_export_partition_file_already_exists_policy(cluster): """ ) == '1\n', "Expected the export to be marked as COMPLETED" - # last but not least, let's try with the error policy - # max retries = 1 so it fails fast + # last but not least, let's try with the error policy. FILE_ALREADY_EXISTS is a + # non-retryable error (retrying always hits the same existing file), so the task + # fails fast without needing a retry budget. node.query( - f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {s3_table} SETTINGS export_merge_tree_partition_force_export=1, export_merge_tree_part_file_already_exists_policy='error', export_merge_tree_partition_max_retries=1", + f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {s3_table} SETTINGS export_merge_tree_partition_force_export=1, export_merge_tree_part_file_already_exists_policy='error'", ) # wait for the export to finish @@ -1376,7 +1460,6 @@ def test_export_partition_resumes_after_stop_moves(cluster): node.query( f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {s3_table}" - f" SETTINGS export_merge_tree_partition_max_retries = 50" ) wait_for_export_to_start(node, mt_table, s3_table, "2020") @@ -1435,7 +1518,6 @@ def test_export_partition_resumes_after_stop_moves_during_export(cluster): node.query( f"ALTER TABLE {mt_table} EXPORT PARTITION ID '2020' TO TABLE {s3_table}" - f" SETTINGS export_merge_tree_partition_max_retries = 50" ) wait_for_export_to_start(node, mt_table, s3_table, "2020") @@ -1544,8 +1626,8 @@ def test_export_partition_partition_column_castable_type_mismatch(cluster): f"ALTER TABLE {mt_table} EXPORT PARTITION ID '{partition_id}' " f"TO TABLE {s3_table}" ) - assert "BAD_ARGUMENTS" in error, ( - f"Expected BAD_ARGUMENTS for a lossy partition-column cast, " + assert "INCOMPATIBLE_COLUMNS" in error, ( + f"Expected INCOMPATIBLE_COLUMNS for a lossy partition-column cast, " f"got: {error!r}" ) assert "requires a lossy cast" in error and "'year'" in error, ( diff --git a/tests/integration/test_storage_iceberg_with_spark/test_export_partition_iceberg.py b/tests/integration/test_storage_iceberg_with_spark/test_export_partition_iceberg.py index 5466fe543275..94d4d6c1a017 100644 --- a/tests/integration/test_storage_iceberg_with_spark/test_export_partition_iceberg.py +++ b/tests/integration/test_storage_iceberg_with_spark/test_export_partition_iceberg.py @@ -713,86 +713,6 @@ def test_idempotency_after_commit_crash(export_cluster): assert count == 3, f"Expected 3 rows (no duplicates), got {count}" -def test_commit_attempts_budget_transitions_to_failed(export_cluster): - """ - Verify that the commit-attempts budget transitions a stuck task to FAILED - instead of leaving it in PENDING forever. - - Reproduction: - - Parts export successfully. - - A REGULAR failpoint (``export_partition_commit_always_throw``) makes every - ``ExportPartitionUtils::commit`` attempt throw before talking to Iceberg. - - ``ExportPartitionUtils::handleCommitFailure`` bumps ``/commit_attempts`` - on each failure and transitions ``/status`` to FAILED once the counter - reaches ``export_merge_tree_partition_max_retries``. - - Expected behaviour: - - The first attempt is made synchronously when the last part completes - (scheduler's ``handlePartExportSuccess``). - - Subsequent attempts come from the manifest-updating task's ``tryCleanup`` - path, polling every 30s. - - With max_retries=2, the task reaches FAILED within roughly one poll cycle. - - The ``commit_attempts`` znode reaches at least max_retries. - """ - node = export_cluster.instances["node1"] - spark = export_cluster.spark_session - - uid = unique_suffix() - source = f"rmt_commit_budget_{uid}" - iceberg = f"spark_commit_budget_{uid}" - - spark_iceberg( - export_cluster, - spark, - iceberg, - f"CREATE TABLE {iceberg} (id BIGINT, year INT)" - f" USING iceberg PARTITIONED BY (identity(year)) OPTIONS('format-version'='2')", - ) - attach_ch_iceberg(node, iceberg, "id Int64, year Int32", export_cluster) - make_rmt(node, source, "id Int64, year Int32", "year") - node.query(f"INSERT INTO {source} VALUES (1, 2024), (2, 2024), (3, 2024)") - pid = first_partition_id(node, source) - - # Force every commit attempt to throw. REGULAR failpoint fires on every hit, - # unlike ONCE which would only fire for the first call. - node.query("SYSTEM ENABLE FAILPOINT export_partition_commit_always_throw") - - # try block exists so we can add a finally that disables the failpoint - try: - # max_retries=2 bounds the test: one attempt from handlePartExportSuccess - # plus one from the manifest-updating task's next poll (~30s) is enough - # to exhaust the budget and flip the task to FAILED. - node.query( - f"ALTER TABLE {source} EXPORT PARTITION ID '{pid}' TO TABLE {iceberg}" - f" SETTINGS export_merge_tree_partition_max_retries = 2," - f" allow_insert_into_iceberg = 1" - ) - - # Timeout must cover: at least one manifest-updating poll cycle (30s) - # plus slack for task scheduling and keeper RTT. - wait_for_export_status( - node, source, iceberg, pid, - expected_status="FAILED", - timeout=90, - ) - - # The commit_attempts znode must have reached (at least) max_retries — the - # counter is the direct mechanism that drove the FAILED transition. - # Locate the export's ZK root via the RMT's zookeeper_path and the - # partition_id_destination_db.destination_table export key convention. - export_key = f"{pid}_default.{iceberg}" - commit_attempts = int(node.query( - f"SELECT value FROM system.zookeeper" - f" WHERE path = '/clickhouse/tables/{source}/exports/{export_key}'" - f" AND name = 'commit_attempts'" - ).strip()) - assert commit_attempts >= 2, ( - f"Expected commit_attempts >= 2 (two commit attempts), got {commit_attempts}" - ) - finally: - node.query("SYSTEM DISABLE FAILPOINT export_partition_commit_always_throw") - - # --------------------------------------------------------------------------- # Replicated tests — IcebergS3, no catalog # --------------------------------------------------------------------------- diff --git a/tests/queries/0_stateless/03572_export_merge_tree_part_to_object_storage_simple.sql b/tests/queries/0_stateless/03572_export_merge_tree_part_to_object_storage_simple.sql index 8200233a7322..d19254dc636f 100644 --- a/tests/queries/0_stateless/03572_export_merge_tree_part_to_object_storage_simple.sql +++ b/tests/queries/0_stateless/03572_export_merge_tree_part_to_object_storage_simple.sql @@ -46,13 +46,13 @@ CREATE TABLE 03572_partition_type_mismatch_mt (id UInt64, year String) ENGINE = CREATE TABLE 03572_partition_type_mismatch_s3 (id UInt64, year UInt16) ENGINE = S3(s3_conn, filename='03572_partition_type_mismatch_s3', format='Parquet', partition_strategy='hive') PARTITION BY year; ALTER TABLE 03572_partition_type_mismatch_mt EXPORT PART '2020_1_1_0' TO TABLE 03572_partition_type_mismatch_s3 -SETTINGS allow_experimental_export_merge_tree_part = 1; -- {serverError BAD_ARGUMENTS} +SETTINGS allow_experimental_export_merge_tree_part = 1; -- {serverError INCOMPATIBLE_COLUMNS} CREATE TABLE 03572_lossy_mt (id Int64, year UInt16) ENGINE = MergeTree() PARTITION BY year ORDER BY tuple(); CREATE TABLE 03572_lossy_s3 (id Int32, year UInt16) ENGINE = S3(s3_conn, filename='03572_lossy_s3', format='Parquet', partition_strategy='hive') PARTITION BY year; ALTER TABLE 03572_lossy_mt EXPORT PART '2020_1_1_0' TO TABLE 03572_lossy_s3 -SETTINGS allow_experimental_export_merge_tree_part = 1; -- {serverError BAD_ARGUMENTS} +SETTINGS allow_experimental_export_merge_tree_part = 1; -- {serverError INCOMPATIBLE_COLUMNS} -- With the acknowledgment setting enabled, the lossy cast passes validation and reaches the -- part lookup, which fails because the part does not exist. diff --git a/tests/queries/0_stateless/03572_export_replicated_merge_tree_part_to_object_storage_simple.sql b/tests/queries/0_stateless/03572_export_replicated_merge_tree_part_to_object_storage_simple.sql index 0a199755c40a..628d1bd4cf46 100644 --- a/tests/queries/0_stateless/03572_export_replicated_merge_tree_part_to_object_storage_simple.sql +++ b/tests/queries/0_stateless/03572_export_replicated_merge_tree_part_to_object_storage_simple.sql @@ -28,14 +28,14 @@ CREATE TABLE 03572_rmt_partition_type_mismatch_mt (id UInt64, year String) ENGIN CREATE TABLE 03572_rmt_partition_type_mismatch_s3 (id UInt64, year UInt16) ENGINE = S3(s3_conn, filename='03572_rmt_partition_type_mismatch_s3', format='Parquet', partition_strategy='hive') PARTITION BY year; ALTER TABLE 03572_rmt_partition_type_mismatch_mt EXPORT PART '2020_0_0_0' TO TABLE 03572_rmt_partition_type_mismatch_s3 -SETTINGS allow_experimental_export_merge_tree_part = 1; -- {serverError BAD_ARGUMENTS} +SETTINGS allow_experimental_export_merge_tree_part = 1; -- {serverError INCOMPATIBLE_COLUMNS} -- A lossy cast on a non-partition column (Int64 -> Int32) is rejected synchronously by default. CREATE TABLE 03572_rmt_lossy_mt (id Int64, year UInt16) ENGINE = ReplicatedMergeTree('/clickhouse/{database}/test_03572_rmt_lossy/03572_rmt_lossy_mt', 'replica1') PARTITION BY year ORDER BY tuple(); CREATE TABLE 03572_rmt_lossy_s3 (id Int32, year UInt16) ENGINE = S3(s3_conn, filename='03572_rmt_lossy_s3', format='Parquet', partition_strategy='hive') PARTITION BY year; ALTER TABLE 03572_rmt_lossy_mt EXPORT PART '2020_0_0_0' TO TABLE 03572_rmt_lossy_s3 -SETTINGS allow_experimental_export_merge_tree_part = 1; -- {serverError BAD_ARGUMENTS} +SETTINGS allow_experimental_export_merge_tree_part = 1; -- {serverError INCOMPATIBLE_COLUMNS} -- With the acknowledgment setting enabled, the lossy cast passes validation and reaches the -- part lookup, which fails because the part does not exist. From 8642d40af753138451ffbd940f8f84829abc5158 Mon Sep 17 00:00:00 2001 From: Andrey Zvonov <32552679+zvonand@users.noreply.github.com> Date: Wed, 5 Aug 2026 19:02:18 +0200 Subject: [PATCH 34/43] Resolve conflicts in cherry-pick of #1984 Settings.cpp: kept both obsolete-setting rows (base branch's allow_experimental_query_deduplication and the PR's export_merge_tree_partition_max_retries). IcebergMetadata.cpp: added only the PR's new error codes METADATA_MISMATCH and UNFINISHED; INCORRECT_DATA was context in the source PR's diff and is unused on antalya-26.6, so it was not imported. Source-PR: #1984 (https://github.com/Altinity/ClickHouse/pull/1984) --- src/Core/Settings.cpp | 5 +- .../DataLake/tests/gtest_rest_catalog.cpp | 1 + src/Interpreters/CancellationCode.h | 2 + src/Storages/IStorageCluster.cpp | 5 +- src/Storages/MergeTree/ExportPartTask.cpp | 7 +- .../MergeTree/ExportPartitionUtils.cpp | 5 +- src/Storages/MergeTree/IMergeTreeDataPart.cpp | 16 ++-- src/Storages/MergeTree/MergeTreeData.cpp | 6 +- src/Storages/MergeTree/MergeTreePartition.cpp | 3 +- .../DataLakes/DataLakeConfiguration.h | 23 ++++-- .../DataLakes/Iceberg/ChunkPartitioner.cpp | 1 + .../DataLakes/Iceberg/IcebergMetadata.cpp | 73 ++++++++----------- .../DataLakes/Iceberg/IcebergWrites.cpp | 31 +++----- .../DataLakes/Iceberg/Mutations.cpp | 2 +- .../ObjectStorage/DataLakes/Iceberg/Utils.cpp | 2 +- .../ObjectStorage/StorageObjectStorage.cpp | 10 ++- .../StorageObjectStorageCluster.cpp | 30 ++++---- .../StorageObjectStorageCluster.h | 4 +- src/Storages/StorageReplicatedMergeTree.cpp | 8 +- .../System/StorageSystemIcebergFiles.cpp | 25 ++++++- .../TableFunctionObjectStorage.cpp | 2 +- ...leFunctionObjectStorageClusterFallback.cpp | 1 + tests/integration/test_database_glue/test.py | 4 +- .../integration/test_database_iceberg/test.py | 21 ++++-- .../test_partition_timezone.py | 22 +++--- .../test_cluster_table_function.py | 1 + .../test_export_partition_iceberg_catalog.py | 20 +++-- 27 files changed, 186 insertions(+), 144 deletions(-) diff --git a/src/Core/Settings.cpp b/src/Core/Settings.cpp index 39cdc0498c21..7dc334e614bd 100644 --- a/src/Core/Settings.cpp +++ b/src/Core/Settings.cpp @@ -8498,11 +8498,8 @@ Maximum number of texts to include in a single HTTP request made by `aiEmbed`. T #define OBSOLETE_SETTINGS(M, ALIAS) \ /** Obsolete settings which are kept around for compatibility reasons. They have no effect anymore. */ \ MAKE_OBSOLETE(M, UInt64, export_merge_tree_partition_manifest_ttl, 86400) \ -<<<<<<< HEAD - MAKE_OBSOLETE(M, Bool, allow_experimental_query_deduplication, false) \ -======= MAKE_OBSOLETE(M, UInt64, export_merge_tree_partition_max_retries, 3) \ ->>>>>>> 091776b7b65 (Merge pull request #1984 from Altinity/export-partition-retry-backoff) + MAKE_OBSOLETE(M, Bool, allow_experimental_query_deduplication, false) \ MAKE_OBSOLETE(M, Bool, query_condition_cache_store_conditions_as_plaintext, false) \ MAKE_OBSOLETE(M, Bool, update_insert_deduplication_token_in_dependent_materialized_views, 0) \ MAKE_OBSOLETE(M, UInt64, max_memory_usage_for_all_queries, 0) \ diff --git a/src/Databases/DataLake/tests/gtest_rest_catalog.cpp b/src/Databases/DataLake/tests/gtest_rest_catalog.cpp index eb1fd0db1a5c..e385d8b0629b 100644 --- a/src/Databases/DataLake/tests/gtest_rest_catalog.cpp +++ b/src/Databases/DataLake/tests/gtest_rest_catalog.cpp @@ -198,6 +198,7 @@ bool restCatalogEmpty(CatalogShape shape) /* auth_header */"", /* oauth_server_uri */"", /* oauth_server_use_request_body */false, + /* namespaces */"*", context); return catalog.empty(); diff --git a/src/Interpreters/CancellationCode.h b/src/Interpreters/CancellationCode.h index e37a7f13105f..571f42c6c5bd 100644 --- a/src/Interpreters/CancellationCode.h +++ b/src/Interpreters/CancellationCode.h @@ -1,5 +1,7 @@ #pragma once +#include + namespace DB { diff --git a/src/Storages/IStorageCluster.cpp b/src/Storages/IStorageCluster.cpp index 408641f9e417..3012c7bff735 100644 --- a/src/Storages/IStorageCluster.cpp +++ b/src/Storages/IStorageCluster.cpp @@ -557,7 +557,10 @@ IStorageCluster::RemoteCallVariables IStorageCluster::convertToRemote( std::shared_ptr remote_table_function = std::dynamic_pointer_cast(remote_function); if (remote_table_function) - remote_table_function->setActualTableStructure(getInMemoryMetadata().columns); + { + auto metadata_snapshot = getInMemoryMetadataPtr(context, false); + remote_table_function->setActualTableStructure(metadata_snapshot->columns); + } auto storage = remote_function->execute(query_to_send, new_context, remote_function_name); diff --git a/src/Storages/MergeTree/ExportPartTask.cpp b/src/Storages/MergeTree/ExportPartTask.cpp index 3e12fa0049bb..2b9e45b43566 100644 --- a/src/Storages/MergeTree/ExportPartTask.cpp +++ b/src/Storages/MergeTree/ExportPartTask.cpp @@ -20,6 +20,7 @@ #include #include #include +#include #include #include #include @@ -121,8 +122,8 @@ namespace const IStorage & destination_storage, const ContextPtr & local_context) { - const auto destination_header - = destination_storage.getInMemoryMetadataPtr()->getSampleBlockNonMaterialized(); + const auto destination_metadata = destination_storage.getInMemoryMetadataPtr(local_context, false); + const auto destination_header = destination_metadata->getSampleBlockNonMaterialized(); auto dag = ActionsDAG::makeConvertingActions( plan_for_part.getCurrentHeader()->getColumnsWithTypeAndName(), @@ -194,7 +195,7 @@ bool ExportPartTask::executeStep() if (metadata_snapshot->hasPartitionKey()) { /// todo arthur do I need to init minmax_idx? - block_with_partition_values = manifest.data_part->minmax_idx->getBlock(storage); + block_with_partition_values = manifest.data_part->getMinMaxIndex()->getBlock(storage); } const auto & destination_storage = manifest.destination_storage_ptr; diff --git a/src/Storages/MergeTree/ExportPartitionUtils.cpp b/src/Storages/MergeTree/ExportPartitionUtils.cpp index ec8046f3e991..5d064373aaf7 100644 --- a/src/Storages/MergeTree/ExportPartitionUtils.cpp +++ b/src/Storages/MergeTree/ExportPartitionUtils.cpp @@ -144,7 +144,7 @@ namespace ExportPartitionUtils "or this replica has not yet received any part for this partition. " "The commit will be retried.", partition_id); - return parts.front()->minmax_idx->getBlock(storage); + return parts.front()->getMinMaxIndex()->getBlock(storage); } ContextPtr getContextCopyWithTaskSettings(const ContextPtr & context, const ExportReplicatedMergeTreePartitionManifest & manifest) @@ -307,7 +307,8 @@ namespace ExportPartitionUtils if (!manifest.iceberg_metadata_json.empty()) { iceberg_args.metadata_json_string = manifest.iceberg_metadata_json; - if (source_storage.getInMemoryMetadataPtr()->hasPartitionKey()) + const auto source_metadata = source_storage.getInMemoryMetadataPtr(context, false); + if (source_metadata->hasPartitionKey()) iceberg_args.partition_source_block = getPartitionSourceBlockForIcebergCommit(source_storage, manifest.partition_id); } diff --git a/src/Storages/MergeTree/IMergeTreeDataPart.cpp b/src/Storages/MergeTree/IMergeTreeDataPart.cpp index 06bb629de89e..0d297e246b11 100644 --- a/src/Storages/MergeTree/IMergeTreeDataPart.cpp +++ b/src/Storages/MergeTree/IMergeTreeDataPart.cpp @@ -378,18 +378,17 @@ Block IMergeTreeDataPart::MinMaxIndex::getBlock(const MergeTreeData & data) cons Block block; - const auto metadata_snapshot = data.getInMemoryMetadataPtr(); + const auto metadata_snapshot = data.getInMemoryMetadataPtr(data.getContext(), false); const auto & partition_key = metadata_snapshot->getPartitionKey(); - const auto minmax_column_names = data.getMinMaxColumnsNames(partition_key); - const auto minmax_column_types = data.getMinMaxColumnsTypes(partition_key); - const auto minmax_idx_size = minmax_column_types.size(); + /// The minmax index may also contain block number/offset columns at the end, + /// they are not part of the partition key, so take only the partition key columns. + const auto minmax_columns = MergeTreeData::getMinMaxColumns( + partition_key, data.getSettings(), MergeTreePartMinMaxIndexColumns::PARTITION_KEY_ONLY); - for (size_t i = 0; i < minmax_idx_size; ++i) + size_t i = 0; + for (const auto & [column_name, data_type] : minmax_columns) { - const auto & data_type = minmax_column_types[i]; - const auto & column_name = minmax_column_names[i]; - const auto column = data_type->createColumn(); auto range = hyperrectangle.at(i); @@ -402,6 +401,7 @@ Block IMergeTreeDataPart::MinMaxIndex::getBlock(const MergeTreeData & data) cons column->insert(max_val); block.insert(ColumnWithTypeAndName(column->getPtr(), data_type, column_name)); + ++i; } return block; diff --git a/src/Storages/MergeTree/MergeTreeData.cpp b/src/Storages/MergeTree/MergeTreeData.cpp index 8ad1f1907266..610a93d67546 100644 --- a/src/Storages/MergeTree/MergeTreeData.cpp +++ b/src/Storages/MergeTree/MergeTreeData.cpp @@ -7053,7 +7053,7 @@ void MergeTreeData::exportPartToTable(const PartitionCommand & command, ContextP if (table_function_ptr->needStructureHint()) { - const auto source_metadata_ptr = getInMemoryMetadataPtr(); + const auto source_metadata_ptr = getInMemoryMetadataPtr(query_context, false); /// Grab only the readable columns from the source metadata to skip ephemeral columns const auto readable_columns = ColumnsDescription(source_metadata_ptr->getColumns().getReadable()); @@ -7117,8 +7117,8 @@ void MergeTreeData::exportPartToTable( return ast ? ast->formatWithSecretsOneLine() : ""; }; - auto source_metadata_ptr = getInMemoryMetadataPtr(); - auto destination_metadata_ptr = dest_storage->getInMemoryMetadataPtr(); + auto source_metadata_ptr = getInMemoryMetadataPtr(query_context, false); + auto destination_metadata_ptr = dest_storage->getInMemoryMetadataPtr(query_context, false); std::string iceberg_metadata_json; diff --git a/src/Storages/MergeTree/MergeTreePartition.cpp b/src/Storages/MergeTree/MergeTreePartition.cpp index 341e9d1a55e3..2cd8b4d73a2a 100644 --- a/src/Storages/MergeTree/MergeTreePartition.cpp +++ b/src/Storages/MergeTree/MergeTreePartition.cpp @@ -10,6 +10,7 @@ #include #include #include +#include #include #include #include @@ -516,7 +517,7 @@ Block MergeTreePartition::getBlockWithPartitionValues(const NamesAndTypesList & std::size_t i = 0; for (const auto & partition_column : partition_columns) { - auto column = partition_column.type->createColumnConst(1, value[i++]); + ColumnPtr column = partition_column.type->createColumnConst(1, value[i++]); result.insert({column, partition_column.type, partition_column.name}); } diff --git a/src/Storages/ObjectStorage/DataLakes/DataLakeConfiguration.h b/src/Storages/ObjectStorage/DataLakes/DataLakeConfiguration.h index 3ad8d3c716a5..04a1f392b2de 100644 --- a/src/Storages/ObjectStorage/DataLakes/DataLakeConfiguration.h +++ b/src/Storages/ObjectStorage/DataLakes/DataLakeConfiguration.h @@ -694,18 +694,29 @@ class StorageIcebergConfiguration : public StorageObjectStorageConfiguration, pu bool supportsDelete() const override { return getImpl().supportsDelete(); } void mutate(const MutationCommands & commands, ContextPtr context, + StoragePtr storage_ptr, const StorageID & storage_id, StorageMetadataPtr metadata_snapshot, std::shared_ptr catalog, const std::optional & format_settings) override { - getImpl().mutate(commands, context, storage_id, metadata_snapshot, catalog, format_settings); + getImpl().mutate(commands, context, storage_ptr, storage_id, metadata_snapshot, catalog, format_settings); } - void checkMutationIsPossible(const MutationCommands & commands) override { getImpl().checkMutationIsPossible(commands); } + void checkMutationIsPossible(ObjectStoragePtr object_storage, ContextPtr context, const MutationCommands & commands) override + { getImpl().checkMutationIsPossible(object_storage, context, commands); } - void checkAlterIsPossible(const AlterCommands & commands) override { getImpl().checkAlterIsPossible(commands); } + void checkAlterIsPossible(ObjectStoragePtr object_storage, ContextPtr context, const AlterCommands & commands) override + { getImpl().checkAlterIsPossible(object_storage, context, commands); } - void alter(const AlterCommands & params, ContextPtr context) override { getImpl().alter(params, context); } + void alter( + ObjectStoragePtr object_storage, + const AlterCommands & params, + ContextPtr context, + const StorageID & storage_id, + std::shared_ptr catalog) override + { + getImpl().alter(object_storage, params, context, storage_id, catalog); + } const DataLakeStorageSettings & getDataLakeSettings() const override { return getImpl().getDataLakeSettings(); } @@ -830,8 +841,8 @@ class StorageIcebergConfiguration : public StorageObjectStorageConfiguration, pu std::shared_ptr getCatalog(ContextPtr context, const StorageID & table_id) const override { return getImpl().getCatalog(context, table_id); } - bool optimize(const StorageMetadataPtr & metadata_snapshot, ContextPtr context, const std::optional & format_settings) override - { return getImpl().optimize(metadata_snapshot, context, format_settings); } + bool optimize(ObjectStoragePtr object_storage, const StorageMetadataPtr & metadata_snapshot, ContextPtr context, const std::optional & format_settings) override + { return getImpl().optimize(object_storage, metadata_snapshot, context, format_settings); } bool supportsPrewhere() const override { return getImpl().supportsPrewhere(); } diff --git a/src/Storages/ObjectStorage/DataLakes/Iceberg/ChunkPartitioner.cpp b/src/Storages/ObjectStorage/DataLakes/Iceberg/ChunkPartitioner.cpp index 502e31463702..924fe8842ee9 100644 --- a/src/Storages/ObjectStorage/DataLakes/Iceberg/ChunkPartitioner.cpp +++ b/src/Storages/ObjectStorage/DataLakes/Iceberg/ChunkPartitioner.cpp @@ -6,6 +6,7 @@ #include #include #include +#include #include #include #include diff --git a/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergMetadata.cpp b/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergMetadata.cpp index 409561506624..76d13a85f142 100644 --- a/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergMetadata.cpp +++ b/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergMetadata.cpp @@ -117,12 +117,8 @@ extern const int ICEBERG_SPECIFICATION_VIOLATION; extern const int S3_ERROR; extern const int TABLE_ALREADY_EXISTS; extern const int SUPPORT_IS_DISABLED; -<<<<<<< HEAD -======= -extern const int INCORRECT_DATA; extern const int METADATA_MISMATCH; extern const int UNFINISHED; ->>>>>>> 091776b7b65 (Merge pull request #1984 from Altinity/export-partition-retry-backoff) } namespace Setting @@ -1574,16 +1570,18 @@ bool IcebergMetadata::commitImportPartitionTransactionImpl( return true; } - CompressionMethod metadata_compression_method = persistent_components.metadata_compression_method; + const auto & resolver = persistent_components.path_resolver; - auto [metadata_name, storage_metadata_name] = filename_generator.generateMetadataName(); + auto metadata_info = filename_generator.generateMetadataPathWithInfo(); + const auto storage_metadata_name = resolver.resolve(metadata_info.path); Int64 parent_snapshot = -1; if (metadata->has(Iceberg::f_current_snapshot_id)) parent_snapshot = metadata->getValue(Iceberg::f_current_snapshot_id); - auto [new_snapshot, manifest_list_name, storage_manifest_list_name] = MetadataGenerator(metadata).generateNextMetadata( - filename_generator, metadata_name, parent_snapshot, total_data_files, total_rows, total_chunks_size, total_data_files, /* added_delete_files */0, /* num_deleted_rows */0); + auto [new_snapshot, manifest_list_path] = MetadataGenerator(metadata).generateNextMetadata( + filename_generator, metadata_info.path, parent_snapshot, total_data_files, total_rows, total_chunks_size, total_data_files, /* added_delete_files */0, /* num_deleted_rows */0); + const auto storage_manifest_list_name = resolver.resolve(manifest_list_path); /// Embed the stable transaction identifier in the snapshot summary so that a retry after crash /// can detect the commit already happened by scanning the live snapshots array, without extra S3 @@ -1591,7 +1589,7 @@ bool IcebergMetadata::commitImportPartitionTransactionImpl( new_snapshot->getObject(Iceberg::f_summary)->set( Iceberg::f_clickhouse_export_partition_transaction_id, transaction_id); - String manifest_entry_name; + Iceberg::IcebergPathFromMetadata manifest_entry_path; String storage_manifest_entry_name; Int64 manifest_lengths = 0; @@ -1640,6 +1638,7 @@ bool IcebergMetadata::commitImportPartitionTransactionImpl( context, getLogger("IcebergWrites").get(), persistent_components.table_uuid, + persistent_components.metadata_compression_method, true); } @@ -1647,8 +1646,8 @@ bool IcebergMetadata::commitImportPartitionTransactionImpl( LOG_DEBUG(log, "Rereading metadata file {} with version {}", metadata_path, last_version); - metadata_compression_method = compression_method; filename_generator.setVersion(last_version + 1); + filename_generator.setCompressionMethod(compression_method); metadata = getMetadataJSONObject( metadata_path, @@ -1683,11 +1682,8 @@ bool IcebergMetadata::commitImportPartitionTransactionImpl( try { - { - auto result = filename_generator.generateManifestEntryName(); - manifest_entry_name = result.path_in_metadata; - storage_manifest_entry_name = result.path_in_storage; - } + manifest_entry_path = filename_generator.generateManifestEntryName(); + storage_manifest_entry_name = resolver.resolve(manifest_entry_path); auto buffer_manifest_entry = object_storage->writeObject( StoredObject(storage_manifest_entry_name), WriteMode::Rewrite, std::nullopt, DBMS_DEFAULT_BUFFER_SIZE, context->getWriteSettings()); @@ -1738,7 +1734,7 @@ bool IcebergMetadata::commitImportPartitionTransactionImpl( try { generateManifestList( - filename_generator, metadata, object_storage, context, {manifest_entry_name}, new_snapshot, manifest_lengths, *buffer_manifest_list, Iceberg::FileContentType::DATA, true); + resolver, metadata, object_storage, context, {manifest_entry_path}, new_snapshot, {manifest_lengths}, *buffer_manifest_list, Iceberg::FileContentType::DATA, true); buffer_manifest_list->finalize(); } catch (...) @@ -1754,15 +1750,14 @@ bool IcebergMetadata::commitImportPartitionTransactionImpl( std::string json_representation = removeEscapedSlashes(oss.str()); LOG_DEBUG(log, "Writing new metadata file {}", storage_metadata_name); - auto hint = filename_generator.generateVersionHint(); + auto hint_path = filename_generator.generateVersionHint(); if (!writeMetadataFileAndVersionHint( - storage_metadata_name, + resolver, + metadata_info, json_representation, - hint.path_in_storage, - storage_metadata_name, + hint_path, object_storage, context, - metadata_compression_method, data_lake_settings[DataLakeStorageSetting::iceberg_use_version_hint])) { LOG_DEBUG(log, "Failed to write metadata {}, retrying", storage_metadata_name); @@ -1774,9 +1769,9 @@ bool IcebergMetadata::commitImportPartitionTransactionImpl( if (catalog) { - String catalog_filename = metadata_name; + String catalog_filename = metadata_info.path.serialize(); if (!catalog_filename.starts_with(blob_storage_type_name)) - catalog_filename = blob_storage_type_name + "://" + blob_storage_namespace_name + "/" + metadata_name; + catalog_filename = blob_storage_type_name + "://" + blob_storage_namespace_name + "/" + catalog_filename; const auto & [namespace_name, table_name] = DataLake::parseTableName(table_id.getTableName()); if (!catalog->updateMetadata(namespace_name, table_name, catalog_filename, new_snapshot)) @@ -1861,6 +1856,7 @@ void IcebergMetadata::commitExportPartitionTransaction( context, getLogger("IcebergMetadata").get(), persistent_components.table_uuid, + persistent_components.metadata_compression_method, true); /// Latest metadata is ALWAYS necessary to commit - but we abort in case schema or partition spec changed @@ -1905,7 +1901,7 @@ void IcebergMetadata::commitExportPartitionTransaction( auto partition_spec = lookupPartitionSpec(metadata, partition_spec_id); - ChunkPartitioner partitioner(partition_spec->getArray(Iceberg::f_fields), schema, context, sample_block); + ChunkPartitioner partitioner(partition_spec->getArray(Iceberg::f_fields), schema->getArray(Iceberg::f_fields), context, sample_block); const auto partition_columns = partitioner.getColumns(); const auto partition_types = partitioner.getResultTypes(); @@ -1915,26 +1911,15 @@ void IcebergMetadata::commitExportPartitionTransaction( const auto partition_values = recomputeExportPartitionValues(partitioner, sample_block, partition_source_block); const auto metadata_compression_method = persistent_components.metadata_compression_method; - auto config_path = persistent_components.table_path; - if (config_path.empty() || config_path.back() != '/') - config_path += "/"; - if (!config_path.starts_with('/')) - config_path = '/' + config_path; - - FileNamesGenerator filename_generator; - if (!context->getSettingsRef()[Setting::write_full_path_in_iceberg_metadata]) - { - filename_generator = FileNamesGenerator( - config_path, config_path, (catalog != nullptr && catalog->isTransactional()), metadata_compression_method, write_format); - } - else - { - auto bucket = metadata->getValue(Iceberg::f_location); - if (bucket.empty() || bucket.back() != '/') - bucket += "/"; - filename_generator = FileNamesGenerator( - bucket, config_path, (catalog != nullptr && catalog->isTransactional()), metadata_compression_method, write_format); - } + + /// Generated paths are always expressed relative to the table location, the conversion + /// to a real storage path is done by `persistent_components.path_resolver`. + FileNamesGenerator filename_generator( + persistent_components.path_resolver.getTableLocation(), + (catalog != nullptr && catalog->isTransactional()), + metadata_compression_method, + write_format); + filename_generator.setVersion(updated_metadata_file_info.version + 1); /// Load per-file sidecar stats, necessary to populate the manifest file stats. diff --git a/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergWrites.cpp b/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergWrites.cpp index 6ec1e4c35a7b..69966433a537 100644 --- a/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergWrites.cpp +++ b/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergWrites.cpp @@ -1366,25 +1366,13 @@ IcebergImportSink::IcebergImportSink( const auto metadata_compression_method = persistent_table_components.metadata_compression_method; - auto config_path = persistent_table_components.table_path; - if (config_path.empty() || config_path.back() != '/') - config_path += "/"; - if (!config_path.starts_with('/')) - config_path = '/' + config_path; - - if (!context_->getSettingsRef()[Setting::write_full_path_in_iceberg_metadata]) - { - filename_generator = FileNamesGenerator( - config_path, config_path, (catalog != nullptr && catalog->isTransactional()), metadata_compression_method, write_format); - } - else - { - auto bucket = metadata_json->getValue(Iceberg::f_location); - if (bucket.empty() || bucket.back() != '/') - bucket += "/"; - filename_generator = FileNamesGenerator( - bucket, config_path, (catalog != nullptr && catalog->isTransactional()), metadata_compression_method, write_format); - } + /// Paths written into Iceberg metadata are always built from the table location, + /// the conversion to the actual storage path is done by the path resolver. + filename_generator = FileNamesGenerator( + persistent_table_components.path_resolver.getTableLocation(), + (catalog != nullptr && catalog->isTransactional()), + metadata_compression_method, + write_format); const auto [last_version, unused_meta_path, unused_compression] = getLatestOrExplicitMetadataFileAndVersion( object_storage, @@ -1393,7 +1381,9 @@ IcebergImportSink::IcebergImportSink( persistent_table_components.metadata_cache, context_, getLogger("IcebergWrites").get(), - persistent_table_components.table_uuid); + persistent_table_components.table_uuid, + metadata_compression_method, + true); (void)unused_meta_path; (void)unused_compression; @@ -1404,6 +1394,7 @@ IcebergImportSink::IcebergImportSink( context->getSettingsRef()[Setting::iceberg_insert_max_bytes_in_data_file], current_schema->getArray(Iceberg::f_fields), filename_generator, + persistent_table_components.path_resolver, object_storage, context, format_settings, diff --git a/src/Storages/ObjectStorage/DataLakes/Iceberg/Mutations.cpp b/src/Storages/ObjectStorage/DataLakes/Iceberg/Mutations.cpp index 8a04b5c83b3e..8fb45abb86aa 100644 --- a/src/Storages/ObjectStorage/DataLakes/Iceberg/Mutations.cpp +++ b/src/Storages/ObjectStorage/DataLakes/Iceberg/Mutations.cpp @@ -753,7 +753,7 @@ void alter( DataLake::TableMetadata table_metadata; table_metadata.withDataLakeSpecificProperties().withLocation(); const auto & [namespace_name, table_name] = DataLake::parseTableName(storage_id.getTableName()); - catalog->getTableMetadata(namespace_name, table_name, table_metadata); + catalog->getTableMetadata(namespace_name, table_name, context, table_metadata); auto specific_properties = table_metadata.getDataLakeSpecificProperties(); if (!specific_properties.has_value() || specific_properties->iceberg_metadata_file_location.empty()) diff --git a/src/Storages/ObjectStorage/DataLakes/Iceberg/Utils.cpp b/src/Storages/ObjectStorage/DataLakes/Iceberg/Utils.cpp index f6165514addd..3c3fb16f507e 100644 --- a/src/Storages/ObjectStorage/DataLakes/Iceberg/Utils.cpp +++ b/src/Storages/ObjectStorage/DataLakes/Iceberg/Utils.cpp @@ -755,7 +755,7 @@ static Poco::JSON::Object::Ptr getPartitionField( throw Exception(ErrorCodes::BAD_ARGUMENTS, "Unsupported function for iceberg partitioning {}", partition_function->name); } -static std::pair getPartitionSpec( +std::pair getPartitionSpec( ASTPtr partition_by, const std::unordered_map & column_name_to_source_id) { diff --git a/src/Storages/ObjectStorage/StorageObjectStorage.cpp b/src/Storages/ObjectStorage/StorageObjectStorage.cpp index 46012ed387bc..637de0a4616d 100644 --- a/src/Storages/ObjectStorage/StorageObjectStorage.cpp +++ b/src/Storages/ObjectStorage/StorageObjectStorage.cpp @@ -682,10 +682,11 @@ SinkToStoragePtr StorageObjectStorage::import( if (isDataLake()) { configuration->lazyInitializeIfNeeded(object_storage, local_context); + auto metadata_snapshot = getInMemoryMetadataPtr(local_context, false); return configuration->getExternalMetadata()->import( catalog, new_file_path_callback, - std::make_shared(getInMemoryMetadataPtr()->getSampleBlock()), + std::make_shared(metadata_snapshot->getSampleBlock()), *iceberg_metadata_json_string, format_settings_ ? format_settings_ : format_settings, local_context); @@ -705,6 +706,8 @@ SinkToStoragePtr StorageObjectStorage::import( const auto base_path = configuration->getPathForWrite(partition_key, file_name).path; + auto metadata_snapshot = getInMemoryMetadataPtr(local_context, false); + return std::make_shared( base_path, /* transaction_id= */ file_name, /// not pretty, but the sink needs some sort of id to generate the commit file name. Using the source part name should be enough @@ -715,7 +718,7 @@ SinkToStoragePtr StorageObjectStorage::import( overwrite_if_exists, new_file_path_callback, format_settings_ ? format_settings_ : format_settings, - std::make_shared(getInMemoryMetadataPtr()->getSampleBlock()), + std::make_shared(metadata_snapshot->getSampleBlock()), local_context); } @@ -741,6 +744,7 @@ void StorageObjectStorage::commitExportPartitionTransaction( const auto partition_spec_id = iceberg_metadata->getValue(Iceberg::f_default_spec_id); configuration->lazyInitializeIfNeeded(object_storage, local_context); + auto metadata_snapshot = getInMemoryMetadataPtr(local_context, false); configuration->getExternalMetadata()->commitExportPartitionTransaction( catalog, storage_id, @@ -748,7 +752,7 @@ void StorageObjectStorage::commitExportPartitionTransaction( original_schema_id, partition_spec_id, iceberg_commit_export_partition_arguments.partition_source_block, - std::make_shared(getInMemoryMetadataPtr()->getSampleBlock()), + std::make_shared(metadata_snapshot->getSampleBlock()), exported_paths, configuration, local_context); diff --git a/src/Storages/ObjectStorage/StorageObjectStorageCluster.cpp b/src/Storages/ObjectStorage/StorageObjectStorageCluster.cpp index d924b38ca25f..3b60d70735d2 100644 --- a/src/Storages/ObjectStorage/StorageObjectStorageCluster.cpp +++ b/src/Storages/ObjectStorage/StorageObjectStorageCluster.cpp @@ -250,8 +250,8 @@ StorageObjectStorageCluster::StorageObjectStorageCluster( object_storage, context_, getStorageID(), - IStorageCluster::getInMemoryMetadata().getColumns(), - IStorageCluster::getInMemoryMetadata().getConstraints(), + metadata.getColumns(), + metadata.getConstraints(), comment_, format_settings_, mode_, @@ -266,10 +266,8 @@ StorageObjectStorageCluster::StorageObjectStorageCluster( updated_configuration, sample_path); - auto virtuals_ = getVirtualsPtr(); - if (virtuals_) - pure_storage->setVirtuals(*virtuals_); - pure_storage->setInMemoryMetadata(IStorageCluster::getInMemoryMetadata()); + /// Virtual columns are a part of StorageInMemoryMetadata, so they are propagated together with it. + pure_storage->setInMemoryMetadata(metadata); } std::string StorageObjectStorageCluster::getName() const @@ -572,14 +570,16 @@ void StorageObjectStorageCluster::updateExternalDynamicMetadataIfExists(ContextP new_metadata = *metadata_snapshot; } - setInMemoryMetadata(new_metadata.withVirtuals(VirtualColumnUtils::getVirtualsForFileLikeStorage( + auto updated_metadata = new_metadata.withVirtuals(VirtualColumnUtils::getVirtualsForFileLikeStorage( new_metadata.columns, query_context, /* format_settings */ std::nullopt, - configuration->getPartitionStrategyType()))); + configuration->getPartitionStrategyType())); + + setInMemoryMetadata(updated_metadata); if (pure_storage) - pure_storage->setInMemoryMetadata(IStorageCluster::getInMemoryMetadata()); + pure_storage->setInMemoryMetadata(updated_metadata); } RemoteQueryExecutor::Extension StorageObjectStorageCluster::getTaskIteratorExtension( @@ -785,11 +785,13 @@ void StorageObjectStorageCluster::alter(const AlterCommands & params, ContextPtr if (getClusterName(context).empty()) { pure_storage->alter(params, context, alter_lock_holder); - setInMemoryMetadata(pure_storage->getInMemoryMetadata()); + auto pure_metadata = pure_storage->getInMemoryMetadataPtr(context, false); + setInMemoryMetadata(*pure_metadata); return; } IStorageCluster::alter(params, context, alter_lock_holder); - pure_storage->setInMemoryMetadata(IStorageCluster::getInMemoryMetadata()); + auto cluster_metadata = IStorageCluster::getInMemoryMetadataPtr(context, false); + pure_storage->setInMemoryMetadata(*cluster_metadata); } void StorageObjectStorageCluster::addInferredEngineArgsToCreateQuery(ASTs & args, const ContextPtr & context) const @@ -797,11 +799,11 @@ void StorageObjectStorageCluster::addInferredEngineArgsToCreateQuery(ASTs & args configuration->addStructureAndFormatToArgsIfNeeded(args, "", configuration->getFormat(), context, /*with_structure=*/false); } -StorageMetadataPtr StorageObjectStorageCluster::getInMemoryMetadataPtr(bool bypass_metadata_cache) const +StorageMetadataHandle StorageObjectStorageCluster::getInMemoryMetadataPtr(ContextPtr context, bool bypass_metadata_cache) const { if (pure_storage) - return pure_storage->getInMemoryMetadataPtr(bypass_metadata_cache); - return IStorageCluster::getInMemoryMetadataPtr(bypass_metadata_cache); + return pure_storage->getInMemoryMetadataPtr(context, bypass_metadata_cache); + return IStorageCluster::getInMemoryMetadataPtr(context, bypass_metadata_cache); } IDataLakeMetadata * StorageObjectStorageCluster::getExternalMetadata(ContextPtr query_context) diff --git a/src/Storages/ObjectStorage/StorageObjectStorageCluster.h b/src/Storages/ObjectStorage/StorageObjectStorageCluster.h index bcaf293ad4e7..1b73b95b20a4 100644 --- a/src/Storages/ObjectStorage/StorageObjectStorageCluster.h +++ b/src/Storages/ObjectStorage/StorageObjectStorageCluster.h @@ -94,7 +94,7 @@ class StorageObjectStorageCluster : public IStorageCluster IDataLakeMetadata * getExternalMetadata(ContextPtr query_context); - StorageMetadataPtr getInMemoryMetadataPtr(bool bypass_metadata_cache = false) const override; + StorageMetadataHandle getInMemoryMetadataPtr(ContextPtr context, bool bypass_metadata_cache) const override; void checkAlterIsPossible(const AlterCommands & commands, ContextPtr context) const override; @@ -171,6 +171,8 @@ class StorageObjectStorageCluster : public IStorageCluster bool isDataLake() const override { return configuration->isDataLakeConfiguration(); } + bool isIcebergStorage() const { return configuration->isIcebergConfiguration(); } + private: void updateQueryToSendIfNeeded( ASTPtr & query, diff --git a/src/Storages/StorageReplicatedMergeTree.cpp b/src/Storages/StorageReplicatedMergeTree.cpp index 760cfbcb400a..074ba4b8c248 100644 --- a/src/Storages/StorageReplicatedMergeTree.cpp +++ b/src/Storages/StorageReplicatedMergeTree.cpp @@ -8608,8 +8608,8 @@ void StorageReplicatedMergeTree::exportPartitionToTable(const PartitionCommand & return ast ? ast->formatWithSecretsOneLine() : ""; }; - auto src_snapshot = getInMemoryMetadataPtr(); - auto destination_snapshot = dest_storage->getInMemoryMetadataPtr(); + auto src_snapshot = getInMemoryMetadataPtr(query_context, false); + auto destination_snapshot = dest_storage->getInMemoryMetadataPtr(query_context, false); /// Positional CAST matching, like `INSERT INTO dest SELECT * FROM src`. ExportPartitionUtils::verifyExportSchemaCastable( @@ -8676,8 +8676,8 @@ void StorageReplicatedMergeTree::exportPartitionToTable(const PartitionCommand & MergeTreeData::IMutationsSnapshot::Params mutations_snapshot_params { - .metadata_version = getInMemoryMetadataPtr()->getMetadataVersion(), - .min_part_metadata_version = MergeTreeData::getMinMetadataVersion(parts), + .metadata_version = src_snapshot->getMetadataVersion(), + .min_part_metadata_version = MergeTreeData::getPartsSnapshotInfo(parts).min_metadata_version, .need_data_mutations = throw_on_pending_mutations, .need_alter_mutations = throw_on_pending_mutations || throw_on_pending_patch_parts, .need_patch_parts = throw_on_pending_patch_parts, diff --git a/src/Storages/System/StorageSystemIcebergFiles.cpp b/src/Storages/System/StorageSystemIcebergFiles.cpp index 929ba6e30ee2..23e95b03775c 100644 --- a/src/Storages/System/StorageSystemIcebergFiles.cpp +++ b/src/Storages/System/StorageSystemIcebergFiles.cpp @@ -24,6 +24,7 @@ #include #include #include +#include #include @@ -220,13 +221,33 @@ class SystemIcebergFilesSource : public ISource if (!lock) return false; + /// Object storage tables created by the storage factory are `StorageObjectStorageCluster` + /// (it falls back to plain object storage reads when no cluster is configured), while some + /// code paths still produce a plain `StorageObjectStorage`. Handle both. auto * object_storage_table = dynamic_cast(storage.get()); - if (!object_storage_table || !object_storage_table->isIcebergStorage()) + auto * object_storage_cluster_table = dynamic_cast(storage.get()); + + if (object_storage_table) + { + if (!object_storage_table->isIcebergStorage()) + return false; + } + else if (object_storage_cluster_table) + { + if (!object_storage_cluster_table->isIcebergStorage()) + return false; + } + else + { return false; + } try { - auto * iceberg_metadata = dynamic_cast(object_storage_table->getExternalMetadata(context_copy)); + auto * external_metadata = object_storage_table + ? object_storage_table->getExternalMetadata(context_copy) + : object_storage_cluster_table->getExternalMetadata(context_copy); + auto * iceberg_metadata = dynamic_cast(external_metadata); if (!iceberg_metadata) return false; diff --git a/src/TableFunctions/TableFunctionObjectStorage.cpp b/src/TableFunctions/TableFunctionObjectStorage.cpp index 8af35a52fb8a..f759c367d2e4 100644 --- a/src/TableFunctions/TableFunctionObjectStorage.cpp +++ b/src/TableFunctions/TableFunctionObjectStorage.cpp @@ -270,7 +270,7 @@ StoragePtr TableFunctionObjectStorage:: /* comment */ String{}, /* format_settings */ std::nullopt, /// No format_settings /* mode */ LoadingStrictnessLevel::CREATE, - configuration->getCatalog(context, /* attach */ false), + configuration->getCatalog(context, StorageID(getDatabaseName(), table_name)), /* if_not_exists */ false, /* is_datalake_query*/ false, /* is_table_function */ true); diff --git a/src/TableFunctions/TableFunctionObjectStorageClusterFallback.cpp b/src/TableFunctions/TableFunctionObjectStorageClusterFallback.cpp index 98d8dc43c85a..bc3d7237f134 100644 --- a/src/TableFunctions/TableFunctionObjectStorageClusterFallback.cpp +++ b/src/TableFunctions/TableFunctionObjectStorageClusterFallback.cpp @@ -1,6 +1,7 @@ #include #include #include +#include #include #include diff --git a/tests/integration/test_database_glue/test.py b/tests/integration/test_database_glue/test.py index 1018b0d3193d..665aee5cd205 100644 --- a/tests/integration/test_database_glue/test.py +++ b/tests/integration/test_database_glue/test.py @@ -1162,8 +1162,8 @@ def create_namespace(suffix): assert node.query(f"SELECT count() FROM {CATALOG_NAME}.`{namespace_prefix}alpha.{table_name}`") == "0\n" assert "is filtered by `namespaces` database parameter." in node.query_and_get_error(f"SELECT count() FROM {CATALOG_NAME}.`{namespace_prefix}bravo.{table_name}`") - node.query(f"CREATE TABLE {CATALOG_NAME}.`{namespace_prefix}alpha.{table2_name}` (x String) ENGINE = IcebergS3('http://minio:9000/warehouse-glue/{namespace_prefix}alpha/a1/{table2_name}/', '{minio_access_key}', '{minio_secret_key}')") - assert "is filtered by `namespaces` database parameter." in node.query_and_get_error(f"CREATE TABLE {CATALOG_NAME}.`{namespace_prefix}bravo.{table2_name}` (x String) ENGINE = IcebergS3('http://minio:9000/warehouse-glue/{namespace_prefix}bravo/{table2_name}/', '{minio_access_key}', '{minio_secret_key}')") + node.query(f"CREATE TABLE {CATALOG_NAME}.`{namespace_prefix}alpha.{table2_name}` (x String) ENGINE = IcebergS3('http://minio1:9001/warehouse-glue/{namespace_prefix}alpha/a1/{table2_name}/', '{minio_access_key}', '{minio_secret_key}')") + assert "is filtered by `namespaces` database parameter." in node.query_and_get_error(f"CREATE TABLE {CATALOG_NAME}.`{namespace_prefix}bravo.{table2_name}` (x String) ENGINE = IcebergS3('http://minio1:9001/warehouse-glue/{namespace_prefix}bravo/{table2_name}/', '{minio_access_key}', '{minio_secret_key}')") node.query(f"DROP TABLE {CATALOG_NAME}.`{namespace_prefix}alpha.{table_name}`") assert "is filtered by `namespaces` database parameter." in node.query_and_get_error(f"DROP TABLE {CATALOG_NAME}.`{namespace_prefix}bravo.{table_name}`") diff --git a/tests/integration/test_database_iceberg/test.py b/tests/integration/test_database_iceberg/test.py index a20c986af9e7..2aae833156a2 100644 --- a/tests/integration/test_database_iceberg/test.py +++ b/tests/integration/test_database_iceberg/test.py @@ -29,6 +29,7 @@ from helpers.cluster import ClickHouseCluster from helpers.config_cluster import minio_secret_key, minio_access_key from helpers.client import QueryRuntimeException +from helpers.test_tools import TSV BASE_URL = "http://rest:8181/v1" @@ -771,8 +772,8 @@ def test_timestamps(started_cluster): SETTINGS iceberg_timezone_for_timestamptz='Foo/Bar' """) - assert node.query(f"SHOW CREATE TABLE {CATALOG_NAME}.`{root_namespace}.{table_name}` SETTINGS iceberg_timezone_for_timestamptz='UTC'") == f"CREATE TABLE {CATALOG_NAME}.`{root_namespace}.{table_name}`\\n(\\n `timestamp` Nullable(DateTime64(6)),\\n `timestamptz` Nullable(DateTime64(6, \\'UTC\\'))\\n)\\nENGINE = Iceberg(\\'http://minio:9000/warehouse-rest/data/\\', \\'minio\\', \\'[HIDDEN]\\')\n" - assert node.query(f"SHOW CREATE TABLE {CATALOG_NAME}.`{root_namespace}.{table_name}` SETTINGS iceberg_timezone_for_timestamptz='Europe/Berlin'") == f"CREATE TABLE {CATALOG_NAME}.`{root_namespace}.{table_name}`\\n(\\n `timestamp` Nullable(DateTime64(6)),\\n `timestamptz` Nullable(DateTime64(6, \\'Europe/Berlin\\'))\\n)\\nENGINE = Iceberg(\\'http://minio:9000/warehouse-rest/data/\\', \\'minio\\', \\'[HIDDEN]\\')\n" + assert node.query(f"SHOW CREATE TABLE {CATALOG_NAME}.`{root_namespace}.{table_name}` SETTINGS iceberg_timezone_for_timestamptz='UTC'") == f"CREATE TABLE {CATALOG_NAME}.`{root_namespace}.{table_name}`\\n(\\n `timestamp` Nullable(DateTime64(6)),\\n `timestamptz` Nullable(DateTime64(6, \\'UTC\\'))\\n)\\nENGINE = Iceberg(\\'http://minio1:9001/warehouse-rest/data/\\', \\'minio\\', \\'[HIDDEN]\\')\n" + assert node.query(f"SHOW CREATE TABLE {CATALOG_NAME}.`{root_namespace}.{table_name}` SETTINGS iceberg_timezone_for_timestamptz='Europe/Berlin'") == f"CREATE TABLE {CATALOG_NAME}.`{root_namespace}.{table_name}`\\n(\\n `timestamp` Nullable(DateTime64(6)),\\n `timestamptz` Nullable(DateTime64(6, \\'Europe/Berlin\\'))\\n)\\nENGINE = Iceberg(\\'http://minio1:9001/warehouse-rest/data/\\', \\'minio\\', \\'[HIDDEN]\\')\n" assert node.query(f"SELECT timezoneOf(timestamptz) FROM {CATALOG_NAME}.`{root_namespace}.{table_name}` LIMIT 1") == "UTC\n" assert node.query(f"SELECT timezoneOf(timestamptz) FROM {CATALOG_NAME}.`{root_namespace}.{table_name}` LIMIT 1 SETTINGS iceberg_timezone_for_timestamptz='UTC'") == "UTC\n" @@ -1353,9 +1354,19 @@ def create_namespace(suffix): assert "is filtered by `namespaces` database parameter." in node.query_and_get_error(f"SELECT count() FROM {CATALOG_NAME}.`{namespace_prefix}echo.{table_name}`") assert node.query(f"SELECT count() FROM {CATALOG_NAME}.`{namespace_prefix}echo.e1.{table_name}`") == "0\n" - node.query(f"CREATE TABLE {CATALOG_NAME}.`{namespace_prefix}alpha.{table2_name}` (x String) ENGINE = IcebergS3('http://minio:9000/warehouse-rest/{namespace_prefix}alpha/{table2_name}/', '{minio_access_key}', '{minio_secret_key}')") - node.query(f"CREATE TABLE {CATALOG_NAME}.`{namespace_prefix}alpha.a1.{table2_name}` (x String) ENGINE = IcebergS3('http://minio:9000/warehouse-rest/{namespace_prefix}alpha/a1/{table2_name}/', '{minio_access_key}', '{minio_secret_key}')") - assert "is filtered by `namespaces` database parameter." in node.query_and_get_error(f"CREATE TABLE {CATALOG_NAME}.`{namespace_prefix}alpha.a2.{table2_name}` (x String) ENGINE = IcebergS3('http://minio:9000/warehouse-rest/{namespace_prefix}alpha/a2/{table2_name}/', '{minio_access_key}', '{minio_secret_key}')") + node.query(f"CREATE TABLE {CATALOG_NAME}.`{namespace_prefix}alpha.{table2_name}` (x String) ENGINE = IcebergS3('http://minio1:9001/warehouse-rest/{namespace_prefix}alpha/{table2_name}/', '{minio_access_key}', '{minio_secret_key}')", + settings={ + "allow_database_iceberg": 1, + "write_full_path_in_iceberg_metadata": 1, + }, + ) + node.query(f"CREATE TABLE {CATALOG_NAME}.`{namespace_prefix}alpha.a1.{table2_name}` (x String) ENGINE = IcebergS3('http://minio1:9001/warehouse-rest/{namespace_prefix}alpha/a1/{table2_name}/', '{minio_access_key}', '{minio_secret_key}')", + settings={ + "allow_database_iceberg": 1, + "write_full_path_in_iceberg_metadata": 1, + }, + ) + assert "is filtered by `namespaces` database parameter." in node.query_and_get_error(f"CREATE TABLE {CATALOG_NAME}.`{namespace_prefix}alpha.a2.{table2_name}` (x String) ENGINE = IcebergS3('http://minio1:9001/warehouse-rest/{namespace_prefix}alpha/a2/{table2_name}/', '{minio_access_key}', '{minio_secret_key}')") node.query(f"DROP TABLE {CATALOG_NAME}.`{namespace_prefix}alpha.{table_name}`") node.query(f"DROP TABLE {CATALOG_NAME}.`{namespace_prefix}alpha.a1.{table_name}`") diff --git a/tests/integration/test_database_iceberg/test_partition_timezone.py b/tests/integration/test_database_iceberg/test_partition_timezone.py index 1a43f3481ad7..e7743aefabbe 100644 --- a/tests/integration/test_database_iceberg/test_partition_timezone.py +++ b/tests/integration/test_database_iceberg/test_partition_timezone.py @@ -36,11 +36,15 @@ from helpers.test_tools import TSV, csv_compare from helpers.config_cluster import minio_secret_key -ICEBERG_PORT = 8183 - BASE_URL = "http://rest:8181/v1" -BASE_URL_LOCAL = f"http://localhost:{ICEBERG_PORT}/v1" -BASE_URL_LOCAL_RAW = f"http://localhost:{ICEBERG_PORT}" +# The REST catalog and MinIO containers are shared with the rest of the suite and their +# host ports are allocated dynamically (`iceberg_rest_catalog_port` / `minio_port`), so +# host-side URLs have to be derived from the cluster object. +WAREHOUSE_ENDPOINT = "http://minio1:9001/warehouse-rest" + + +def get_base_url_local_raw(cluster): + return f"http://localhost:{cluster.iceberg_rest_catalog_port}" CATALOG_NAME = "demo" @@ -60,10 +64,6 @@ def started_cluster(): try: cluster = ClickHouseCluster(__file__) - cluster.iceberg_rest_external_port = ICEBERG_PORT - cluster.spark_iceberg_external_port = 10004 - cluster.spark_iceberg_external_port_2 = 10005 - cluster.spark_iceberg_external_port_3 = 10006 cluster.add_instance( "node1", main_configs=["configs/timezone.xml", "configs/cluster.xml"], @@ -89,9 +89,9 @@ def load_catalog_impl(started_cluster): return load_catalog( CATALOG_NAME, **{ - "uri": BASE_URL_LOCAL_RAW, + "uri": get_base_url_local_raw(started_cluster), "type": "rest", - "s3.endpoint": f"http://{started_cluster.get_instance_ip('minio')}:9000", + "s3.endpoint": f"http://{started_cluster.minio_ip}:{started_cluster.minio_port}", "s3.access-key-id": minio_access_key, "s3.secret-access-key": minio_secret_key, }, @@ -121,7 +121,7 @@ def create_clickhouse_iceberg_database( settings = { "catalog_type": "rest", "warehouse": "demo", - "storage_endpoint": "http://minio:9000/warehouse-rest", + "storage_endpoint": WAREHOUSE_ENDPOINT, } settings.update(additional_settings) diff --git a/tests/integration/test_storage_iceberg_with_spark/test_cluster_table_function.py b/tests/integration/test_storage_iceberg_with_spark/test_cluster_table_function.py index 078e34004ec9..ffbfe3a9c0dd 100644 --- a/tests/integration/test_storage_iceberg_with_spark/test_cluster_table_function.py +++ b/tests/integration/test_storage_iceberg_with_spark/test_cluster_table_function.py @@ -12,6 +12,7 @@ ) import logging +import uuid import pyarrow.parquet as pq from helpers.config_cluster import minio_secret_key diff --git a/tests/integration/test_storage_iceberg_with_spark/test_export_partition_iceberg_catalog.py b/tests/integration/test_storage_iceberg_with_spark/test_export_partition_iceberg_catalog.py index 8a216860d029..b32582829197 100644 --- a/tests/integration/test_storage_iceberg_with_spark/test_export_partition_iceberg_catalog.py +++ b/tests/integration/test_storage_iceberg_with_spark/test_export_partition_iceberg_catalog.py @@ -34,8 +34,14 @@ GLUE_BASE_URL = "http://glue:3000" -GLUE_BASE_URL_LOCAL = "http://localhost:3000" CH_CATALOG_DB = "glue_export_catalog" +# The Glue (Moto) container port is mapped to a dynamically allocated host port, +# see `ClickHouseCluster.glue_catalog_port`, so the host-side URL is per-cluster. +GLUE_WAREHOUSE_ENDPOINT = "http://minio1:9001/warehouse-glue" + + +def get_glue_local_url(cluster): + return f"http://localhost:{cluster.glue_catalog_port}" # --------------------------------------------------------------------------- @@ -104,17 +110,17 @@ def cleanup_tables(catalog_export_cluster): def connect_catalog(cluster): """ - Connect to the Moto Glue mock from the test host via localhost:3000. - MinIO is accessed via the container IP for S3 operations. + Connect to the Moto Glue mock from the test host via its mapped host port. + MinIO is accessed via the container IP for S3 operations. Catalogs share the + standard MinIO container (`minio1`), exposed as `cluster.minio_ip`/`minio_port`. """ - minio_ip = cluster.get_instance_ip("minio") return load_catalog( "glue_test", **{ "type": "glue", - "glue.endpoint": GLUE_BASE_URL_LOCAL, + "glue.endpoint": get_glue_local_url(cluster), "glue.region": "us-east-1", - "s3.endpoint": f"http://{minio_ip}:9000", + "s3.endpoint": f"http://{cluster.minio_ip}:{cluster.minio_port}", "s3.access-key-id": minio_access_key, "s3.secret-access-key": minio_secret_key, }, @@ -132,7 +138,7 @@ def setup_ch_catalog_db(node, db_name: str = CH_CATALOG_DB) -> None: ENGINE = DataLakeCatalog('{GLUE_BASE_URL}', '{minio_access_key}', '{minio_secret_key}') SETTINGS catalog_type = 'glue', warehouse = 'test', - storage_endpoint = 'http://minio:9000/warehouse-glue', + storage_endpoint = '{GLUE_WAREHOUSE_ENDPOINT}', region = 'us-east-1' """ ) From ea0816eea1fc5a67acc778aa031cfa5b130c287b Mon Sep 17 00:00:00 2001 From: Andrey Zvonov <32552679+zvonand@users.noreply.github.com> Date: Wed, 12 Aug 2026 13:00:03 +0200 Subject: [PATCH 35/43] Update SettingsChangesHistory.cpp --- src/Core/SettingsChangesHistory.cpp | 10 ++++++---- 1 file changed, 6 insertions(+), 4 deletions(-) diff --git a/src/Core/SettingsChangesHistory.cpp b/src/Core/SettingsChangesHistory.cpp index 657fb5a1ece0..3e7be887acbd 100644 --- a/src/Core/SettingsChangesHistory.cpp +++ b/src/Core/SettingsChangesHistory.cpp @@ -41,6 +41,7 @@ const VersionToSettingsChangesMap & getSettingsChangesHistory() /// Note: please check if the key already exists to prevent duplicate entries. addSettingsChanges(settings_changes_history, "26.6", { + {"analyzer_compatibility_apply_final_to_all_joined_tables", true, false, "Fixed a bug in the analyzer where FINAL on the left-most table of a JOIN was incorrectly applied to the other joined tables as well. previous_value=true so `compatibility` with versions before 26.6 restores the old behavior."}, {"analyzer_compatibility_allow_non_aggregate_in_having", false, false, "New compatibility setting. When enabled, the new analyzer mimics the legacy `HAVING`-to-`WHERE` rewrite for non-aggregate AND-conjuncts instead of raising `NOT_AN_AGGREGATE`."}, {"reserve_memory", 0, 0, "New setting to reserve memory for specific workload before starting a query."}, {"output_format_image_width", 1024, 1024, "New setting controlling the width of the output image for image output formats such as PNG."}, @@ -50,7 +51,8 @@ const VersionToSettingsChangesMap & getSettingsChangesHistory() {"use_lightweight_primary_key_index_analysis", false, true, "New setting to optimize primary key index analysis for tables with long primary keys"}, {"ai_function_embedding_max_batch_size", 100, 100, "New setting"}, {"enable_nullable_tuple_type", false, false, "Nullable Tuple is now Beta. Added as an alias for 'allow_experimental_nullable_tuple_type'."}, - {"ai_function_credentials", "", "", "New setting"}, + {"ai_function_text_default_credentials", "", "", "New setting"}, + {"ai_function_embedding_default_credentials", "", "", "New setting"}, {"enable_sharding_aggregator", false, false, "New setting to enable sharded `GROUP BY` optimization that distributes rows across threads by hashing the grouping key, so each thread aggregates a disjoint subset of keys without a merge phase; this is efficient for high cardinality keys with evenly distributed data."}, {"allow_experimental_text_index_lazy_apply", false, false, "New setting to gate experimental lazy posting list apply mode"}, {"text_index_posting_list_apply_mode", "materialize", "materialize", "New setting for lazy posting list apply mode"}, @@ -483,9 +485,9 @@ const VersionToSettingsChangesMap & getSettingsChangesHistory() // {"export_merge_tree_partition_max_retries", 3, 3, "New setting."}, // {"export_merge_tree_partition_manifest_ttl", 180, 180, "New setting."}, // {"export_merge_tree_part_file_already_exists_policy", "skip", "skip", "New setting."}, - // {"hybrid_table_auto_cast_columns", true, true, "New setting to automatically cast Hybrid table columns when segments disagree on types. Default enabled."}, - // {"allow_experimental_hybrid_table", false, false, "Added new setting to allow the Hybrid table engine."}, - // {"enable_alias_marker", true, true, "New setting."}, + {"hybrid_table_auto_cast_columns", true, true, "New setting to automatically cast Hybrid table columns when segments disagree on types. Default enabled."}, + {"allow_experimental_hybrid_table", false, false, "Added new setting to allow the Hybrid table engine."}, + {"enable_alias_marker", true, true, "New setting."}, // {"export_merge_tree_part_max_bytes_per_file", 0, 0, "New setting."}, // {"export_merge_tree_part_max_rows_per_file", 0, 0, "New setting."}, // {"export_merge_tree_partition_lock_inside_the_task", false, false, "New setting."}, From 4b71fda9f2e3d552f211306b7a80f86b241a4d02 Mon Sep 17 00:00:00 2001 From: Andrey Zvonov <32552679+zvonand@users.noreply.github.com> Date: Wed, 12 Aug 2026 15:59:12 +0200 Subject: [PATCH 36/43] Fix CI: extend magic_enum range for ASTSystemQuery::Type and register new Antalya settings in the changes history The merge pushed ASTSystemQuery::Type past 127 values, so magic_enum silently dropped RESET_DDL_WORKER and SYSTEM RESET DDL WORKER became unparsable; and three Antalya settings were missing from SettingsChangesHistory.cpp. Addresses 2 failing test(s) in Fast test on https://github.com/Altinity/ClickHouse/pull/2146. Still-failing set shrank from 2 -> 0. --- src/Core/SettingsChangesHistory.cpp | 6 +++--- src/Parsers/ASTSystemQuery.h | 11 +++++++++++ 2 files changed, 14 insertions(+), 3 deletions(-) diff --git a/src/Core/SettingsChangesHistory.cpp b/src/Core/SettingsChangesHistory.cpp index 3e7be887acbd..5cdcb134f4e2 100644 --- a/src/Core/SettingsChangesHistory.cpp +++ b/src/Core/SettingsChangesHistory.cpp @@ -475,10 +475,10 @@ const VersionToSettingsChangesMap & getSettingsChangesHistory() // {"input_format_parquet_use_metadata_cache", true, true, "New setting, turned ON by default"}, // https://github.com/Altinity/ClickHouse/pull/586 {"iceberg_timezone_for_timestamptz", "UTC", "UTC", "New setting."}, {"object_storage_remote_initiator", false, false, "New setting."}, - // {"allow_experimental_iceberg_read_optimization", true, true, "New setting."}, + {"allow_experimental_iceberg_read_optimization", true, true, "New setting."}, {"object_storage_cluster_join_mode", "allow", "allow", "New setting"}, - // {"lock_object_storage_task_distribution_ms", 500, 500, "New setting."}, - // {"allow_retries_in_cluster_requests", false, false, "New setting"}, + {"lock_object_storage_task_distribution_ms", 500, 500, "New setting."}, + {"allow_retries_in_cluster_requests", false, false, "New setting"}, // {"allow_experimental_export_merge_tree_part", false, true, "Turned ON by default for Antalya."}, // {"export_merge_tree_part_overwrite_file_if_exists", false, false, "New setting."}, // {"export_merge_tree_partition_force_export", false, false, "New setting."}, diff --git a/src/Parsers/ASTSystemQuery.h b/src/Parsers/ASTSystemQuery.h index 141f02fc08fd..517b1c6304c9 100644 --- a/src/Parsers/ASTSystemQuery.h +++ b/src/Parsers/ASTSystemQuery.h @@ -4,6 +4,7 @@ #include #include #include +#include #include "config.h" @@ -268,3 +269,13 @@ class ASTSystemQuery : public IAST, public ASTQueryWithOnCluster } + +/// The number of SYSTEM query types exceeds the default magic_enum range [-128, 127]. +/// Without extending the range magic_enum silently ignores the types with values above 127, +/// so such queries cannot be parsed (ParserSystemQuery iterates over magic_enum::enum_values) +/// and cannot be formatted (ASTSystemQuery::typeToString indexes a magic_enum-sized array). +template <> struct magic_enum::customize::enum_range +{ + static constexpr int min = 0; + static constexpr int max = 512; +}; From c1c8cdc1f90b6341b3d3de71e7660899e77a87bb Mon Sep 17 00:00:00 2001 From: Andrey Zvonov <32552679+zvonand@users.noreply.github.com> Date: Thu, 13 Aug 2026 07:41:11 +0200 Subject: [PATCH 37/43] Update SettingsChangesHistory.cpp --- src/Core/SettingsChangesHistory.cpp | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/src/Core/SettingsChangesHistory.cpp b/src/Core/SettingsChangesHistory.cpp index 5cdcb134f4e2..67a4bf17e703 100644 --- a/src/Core/SettingsChangesHistory.cpp +++ b/src/Core/SettingsChangesHistory.cpp @@ -88,6 +88,7 @@ const VersionToSettingsChangesMap & getSettingsChangesHistory() {"allow_experimental_query_deduplication", false, false, "The setting is obsolete, the feature has been removed."}, {"query_plan_min_columns_for_join_lazy_indexing", 0, 3, "Control the minimum number of payload columns from the left side required for enabling lazy indexing optimization in JOIN"}, {"query_plan_max_limit_for_join_lazy_indexing", 1000, 1000, "Added new setting to control maximum limit value that allows to use query plan for lazy join indexing optimization. If zero, there is no limit"}, + {"allow_experimental_database_s3_tables", false, false, "New setting to enable experimental database S3 tables (AWS Iceberg REST catalog)."}, {"object_storage_cluster_join_mode", "allow", "allow", "New setting"}, {"export_merge_tree_partition_task_timeout_seconds", "3600", "86400", "Increase default value to make it more realistic"}, {"export_merge_tree_part_allow_lossy_cast", false, false, "New setting to gate lossy casts in EXPORT PART/PARTITION behind explicit acknowledgment"}, @@ -156,6 +157,7 @@ const VersionToSettingsChangesMap & getSettingsChangesHistory() }); addSettingsChanges(settings_changes_history, "26.4", { + {"analyzer_compatibility_apply_final_to_all_joined_tables", true, true, "New compatibility setting controlling whether FINAL on the left-most table of a JOIN is applied to the other joined tables. Introduced with default true (the old behavior) for backports to versions before 26.6."}, {"max_bytes_before_external_join", 0, 0, "New setting to control automatic spilling of hash joins to disk. Non-zero value enables spilling and sets the byte threshold."}, {"allow_iceberg_remove_orphan_files", false, false, "New setting to gate Iceberg orphan file removal"}, {"iceberg_orphan_files_older_than_seconds", 259200, 259200, "New setting for default orphan file age threshold"}, @@ -474,9 +476,7 @@ const VersionToSettingsChangesMap & getSettingsChangesHistory() {"allow_database_glue_catalog", false, true, "Turned ON by default for Antalya (alias)."}, // {"input_format_parquet_use_metadata_cache", true, true, "New setting, turned ON by default"}, // https://github.com/Altinity/ClickHouse/pull/586 {"iceberg_timezone_for_timestamptz", "UTC", "UTC", "New setting."}, - {"object_storage_remote_initiator", false, false, "New setting."}, {"allow_experimental_iceberg_read_optimization", true, true, "New setting."}, - {"object_storage_cluster_join_mode", "allow", "allow", "New setting"}, {"lock_object_storage_task_distribution_ms", 500, 500, "New setting."}, {"allow_retries_in_cluster_requests", false, false, "New setting"}, // {"allow_experimental_export_merge_tree_part", false, true, "Turned ON by default for Antalya."}, From 75648587fe3e69d17e622a026600bd11f8d11410 Mon Sep 17 00:00:00 2001 From: Andrey Zvonov <32552679+zvonand@users.noreply.github.com> Date: Thu, 13 Aug 2026 08:49:21 +0200 Subject: [PATCH 38/43] Update SettingsChangesHistory.cpp --- src/Core/SettingsChangesHistory.cpp | 1 + 1 file changed, 1 insertion(+) diff --git a/src/Core/SettingsChangesHistory.cpp b/src/Core/SettingsChangesHistory.cpp index 67a4bf17e703..c0c5b6603ea5 100644 --- a/src/Core/SettingsChangesHistory.cpp +++ b/src/Core/SettingsChangesHistory.cpp @@ -476,6 +476,7 @@ const VersionToSettingsChangesMap & getSettingsChangesHistory() {"allow_database_glue_catalog", false, true, "Turned ON by default for Antalya (alias)."}, // {"input_format_parquet_use_metadata_cache", true, true, "New setting, turned ON by default"}, // https://github.com/Altinity/ClickHouse/pull/586 {"iceberg_timezone_for_timestamptz", "UTC", "UTC", "New setting."}, + {"object_storage_remote_initiator", false, false, "New setting."}, {"allow_experimental_iceberg_read_optimization", true, true, "New setting."}, {"lock_object_storage_task_distribution_ms", 500, 500, "New setting."}, {"allow_retries_in_cluster_requests", false, false, "New setting"}, From 2dd6c7bf4a3caea1018c455907913df4eb2f7e71 Mon Sep 17 00:00:00 2001 From: Andrey Zvonov <32552679+zvonand@users.noreply.github.com> Date: Thu, 13 Aug 2026 12:16:15 +0200 Subject: [PATCH 39/43] fix tests --- src/TableFunctions/TableFunctionObjectStorage.cpp | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/src/TableFunctions/TableFunctionObjectStorage.cpp b/src/TableFunctions/TableFunctionObjectStorage.cpp index f759c367d2e4..062a52f735e1 100644 --- a/src/TableFunctions/TableFunctionObjectStorage.cpp +++ b/src/TableFunctions/TableFunctionObjectStorage.cpp @@ -247,11 +247,13 @@ StoragePtr TableFunctionObjectStorage:: /// Only use parallel replicas if the Cluster variant of this table function exists /// (e.g. `s3Cluster` for `s3`). Table functions without a Cluster variant (e.g. `paimonLocal`) /// cannot distribute work via task iterators, so distributing would just read all data on every replica. + /// `getName`, not `Definition::name`: with `TableFunctionObjectStorageClusterFallback` + /// the definition is the *Cluster one, while the invoked function is `s3`. const auto can_use_parallel_replicas = !parallel_replicas_cluster_name.empty() && query_settings[Setting::parallel_replicas_for_cluster_engines] && context->canUseTaskBasedParallelReplicas() && !context->isDistributed() - && TableFunctionFactory::instance().isTableFunctionName(String(name) + "Cluster"); + && TableFunctionFactory::instance().isTableFunctionName(getName() + "Cluster"); const auto is_secondary_query = context->getClientInfo().query_kind == ClientInfo::QueryKind::SECONDARY_QUERY; From 3c7a3efad72b051692e228fd786a4533e5e3a26a Mon Sep 17 00:00:00 2001 From: Andrey Zvonov <32552679+zvonand@users.noreply.github.com> Date: Thu, 13 Aug 2026 19:33:21 +0200 Subject: [PATCH 40/43] Fix CI: only masquerade remote() sub-queries as initial when the remote node runs them to completion Addresses 2 failing test(s) in Stateless tests (arm_binary, parallel) on https://github.com/Altinity/ClickHouse/pull/2146. Still-failing set shrank from 2 -> 1. RemoteQueryExecutor marked queries sent by remote()/cluster() table functions with a single shard as INITIAL_QUERY so the remote node can act as an initiator and distribute further (swarm). For intermediate processing stages the remote node returns a header made of internal column identifiers that the initiator matches by name; INITIAL_QUERY enables query tree optimizations there (they are skipped for SECONDARY_QUERY), which rewrote tupleElement(__table1.t, 1) into the subcolumn __table1.`t.1` and broke 04045_merge_function_missing_columns_remote with prefer_localhost_replica=0. Restrict the masquerade to stage == Complete, which is exactly the single-node proxy case the feature needs. --- src/QueryPipeline/RemoteQueryExecutor.cpp | 6 +++++- 1 file changed, 5 insertions(+), 1 deletion(-) diff --git a/src/QueryPipeline/RemoteQueryExecutor.cpp b/src/QueryPipeline/RemoteQueryExecutor.cpp index 429ec6998d51..7d3394a3157a 100644 --- a/src/QueryPipeline/RemoteQueryExecutor.cpp +++ b/src/QueryPipeline/RemoteQueryExecutor.cpp @@ -452,7 +452,11 @@ void RemoteQueryExecutor::sendQueryUnlocked(ClientInfo::QueryKind query_kind, As ClientInfo modified_client_info = context->getClientInfo(); /// Doesn't support now "remote('1.1.1.{1,2}')"" - if (is_remote_function && (shard_count == 1)) + /// Only when the remote node processes the query to completion (single node acting as a proxy target). + /// For intermediate stages the remote node returns a header made of internal column identifiers that the + /// initiator expects to match exactly; pretending the query is initial there enables query tree optimizations + /// on the remote side (they are skipped for SECONDARY_QUERY), which changes that header and breaks the read. + if (is_remote_function && (shard_count == 1) && (stage == QueryProcessingStage::Complete)) { modified_client_info.setInitialQuery(); modified_client_info.client_name = "ClickHouse server"; From da20658a0eec687ce0bf31db1eabe280be8724a8 Mon Sep 17 00:00:00 2001 From: Andrey Zvonov <32552679+zvonand@users.noreply.github.com> Date: Fri, 14 Aug 2026 05:26:26 +0200 Subject: [PATCH 41/43] Fix CI: masquerade remote() sub-queries as initial only for remote(host, ) Addresses 3 failing test(s) in Integration tests (arm_binary, distributed plan, 3/4) on https://github.com/Altinity/ClickHouse/pull/2146. Still-failing set shrank from 3 -> 0. Commit 3c7a3efad72 restricted the 'remote node acts as initiator' masquerade in RemoteQueryExecutor to stage == Complete to fix 04045_merge_function_missing_columns_remote. That also disabled it for the object_storage_remote_initiator flow (remote(host, s3Cluster(...))) whose queries run at an intermediate stage, so the chosen remote initiator handled the query as a plain worker instead of distributing it over the swarm and the tests counted 2 instead of 4 (1 instead of 2) query_log entries. Restore the original condition in RemoteQueryExecutor and instead enable the masquerade only when remote() wraps a table function (table_func_ptr != nullptr), which is exactly the object_storage_remote_initiator case. For remote(host, db, table) - the shape used by 04045_merge_function_missing_columns_remote - the remote node stays a secondary query, so the header of internal column identifiers is not rewritten by query tree optimizations. --- src/Interpreters/ClusterProxy/executeQuery.cpp | 8 +++++++- src/QueryPipeline/RemoteQueryExecutor.cpp | 6 +----- 2 files changed, 8 insertions(+), 6 deletions(-) diff --git a/src/Interpreters/ClusterProxy/executeQuery.cpp b/src/Interpreters/ClusterProxy/executeQuery.cpp index 5e86d23aec96..89e80bb9cb7a 100644 --- a/src/Interpreters/ClusterProxy/executeQuery.cpp +++ b/src/Interpreters/ClusterProxy/executeQuery.cpp @@ -505,7 +505,13 @@ void executeQuery( std::move(unavailable_shard_tracker)); read_from_remote->setStepDescription("Read from remote replica"); - read_from_remote->setIsRemoteFunction(is_remote_function); + /// The remote node is allowed to act as an initiator (and distribute the query further, e.g. over a swarm) + /// only when `remote()` wraps a table function, like `remote(host, s3Cluster(...))`. + /// For `remote(host, db, table)` the remote node must stay a secondary query: for intermediate processing + /// stages it returns a header made of internal column identifiers which the initiator matches by name, + /// and query tree optimizations enabled for initial queries change those names (see + /// 04045_merge_function_missing_columns_remote). + read_from_remote->setIsRemoteFunction(is_remote_function && table_func_ptr != nullptr); plan->addStep(std::move(read_from_remote)); plan->addInterpreterContext(new_context); plans.emplace_back(std::move(plan)); diff --git a/src/QueryPipeline/RemoteQueryExecutor.cpp b/src/QueryPipeline/RemoteQueryExecutor.cpp index 7d3394a3157a..429ec6998d51 100644 --- a/src/QueryPipeline/RemoteQueryExecutor.cpp +++ b/src/QueryPipeline/RemoteQueryExecutor.cpp @@ -452,11 +452,7 @@ void RemoteQueryExecutor::sendQueryUnlocked(ClientInfo::QueryKind query_kind, As ClientInfo modified_client_info = context->getClientInfo(); /// Doesn't support now "remote('1.1.1.{1,2}')"" - /// Only when the remote node processes the query to completion (single node acting as a proxy target). - /// For intermediate stages the remote node returns a header made of internal column identifiers that the - /// initiator expects to match exactly; pretending the query is initial there enables query tree optimizations - /// on the remote side (they are skipped for SECONDARY_QUERY), which changes that header and breaks the read. - if (is_remote_function && (shard_count == 1) && (stage == QueryProcessingStage::Complete)) + if (is_remote_function && (shard_count == 1)) { modified_client_info.setInitialQuery(); modified_client_info.client_name = "ClickHouse server"; From e3a3deb1b28cc0d72ef0d9e08d84b6866bff1f08 Mon Sep 17 00:00:00 2001 From: Andrey Zvonov <32552679+zvonand@users.noreply.github.com> Date: Fri, 14 Aug 2026 05:38:49 +0200 Subject: [PATCH 42/43] Fix CI: read Iceberg partition/sorting key from StorageObjectStorageCluster in system.tables Addresses 3 failing test(s) in Integration tests (arm_binary, distributed plan, 4/4) on https://github.com/Altinity/ClickHouse/pull/2146. Still-failing set shrank from 3 -> 0. The cherry-picked 'Expose IcebergS3 partition_key and sorting_key in system.tables' feature looks up the data lake metadata with dynamic_cast. In this branch object storage table engines are instantiated as StorageObjectStorageCluster, which is not derived from StorageObjectStorage, so the cast always failed and system.tables reported empty partition_key/sorting_key for Iceberg tables (test_system_tables_partition_sorting_keys). Handle both storage types via a small helper. The other two failures of the shard (test_remote_initiator, test_writes_multiple_threads) are already fixed by da20658a0ee and pass locally. --- src/Storages/System/StorageSystemTables.cpp | 41 +++++++++++++-------- 1 file changed, 26 insertions(+), 15 deletions(-) diff --git a/src/Storages/System/StorageSystemTables.cpp b/src/Storages/System/StorageSystemTables.cpp index 1ff9b0f8c290..fcdcac1848bd 100644 --- a/src/Storages/System/StorageSystemTables.cpp +++ b/src/Storages/System/StorageSystemTables.cpp @@ -28,6 +28,7 @@ #include #include #include +#include #include #include #include @@ -172,6 +173,23 @@ ColumnPtr getFilteredTables( } +namespace +{ + +/// Returns data lake metadata (Iceberg, DeltaLake, ...) of the table, if it has any. +/// Object storage table engines are instantiated as StorageObjectStorageCluster, which is not derived +/// from StorageObjectStorage, so both storage types have to be handled here. +IDataLakeMetadata * tryGetDataLakeMetadata(const StoragePtr & table, ContextPtr context) +{ + if (auto * object_storage = dynamic_cast(table.get())) + return object_storage->getExternalMetadata(context); + if (auto * object_storage_cluster = dynamic_cast(table.get())) + return object_storage_cluster->getExternalMetadata(context); + return nullptr; +} + +} + StorageSystemTables::StorageSystemTables(const StorageID & table_id_) : StorageWithCommonVirtualColumns(table_id_) { @@ -705,17 +723,13 @@ class TablesBlockSource final : public ISource try { // Extract from specific DataLake metadata if suitable - if (auto * obj = dynamic_cast(table.get())) + if (auto * dl_meta = tryGetDataLakeMetadata(table, context)) { - if (auto * dl_meta = obj->getExternalMetadata(context)) + if (auto p = dl_meta->partitionKey(context); p.has_value()) { - if (auto p = dl_meta->partitionKey(context); p.has_value()) - { - res_columns[res_index++]->insert(*p); - inserted = true; - } + res_columns[res_index++]->insert(*p); + inserted = true; } - } } catch (const Exception &) @@ -740,15 +754,12 @@ class TablesBlockSource final : public ISource try { // Extract from specific DataLake metadata if suitable - if (auto * obj = dynamic_cast(table.get())) + if (auto * dl_meta = tryGetDataLakeMetadata(table, context)) { - if (auto * dl_meta = obj->getExternalMetadata(context)) + if (auto p = dl_meta->sortingKey(context); p.has_value()) { - if (auto p = dl_meta->sortingKey(context); p.has_value()) - { - res_columns[res_index++]->insert(*p); - inserted = true; - } + res_columns[res_index++]->insert(*p); + inserted = true; } } } From 7eba7cb7099c40561fc2973a6b06a493d56ab2b4 Mon Sep 17 00:00:00 2001 From: Andrey Zvonov <32552679+zvonand@users.noreply.github.com> Date: Fri, 14 Aug 2026 10:36:13 +0200 Subject: [PATCH 43/43] Fix CI: iceberg read optimization answered 04337 from manifest stats, skipping parquet schema conversion The test imported with the upstream reserved-field-id fix expects ICEBERG_SPECIFICATION_VIOLATION for an unmapped non-reserved field id, but the Antalya-only allow_experimental_iceberg_read_optimization (added by this PR, enabled by default) serves the single-row constant column from the manifest statistics, so the data file is never parsed and the check never runs. Disable that optimization in the two SELECTs of the test so it exercises the parquet schema converter as intended. Addresses 2 failing test(s) in Stateless tests (amd_debug, distributed plan, s3 storage, parallel) on https://github.com/Altinity/ClickHouse/pull/2146. Still-failing set shrank from 2 -> 1 (the other one is an unrelated flake that passes on rerun). --- .../04337_iceberg_v3_row_lineage_reserved_field_id.sh | 8 ++++++-- 1 file changed, 6 insertions(+), 2 deletions(-) diff --git a/tests/queries/0_stateless/04337_iceberg_v3_row_lineage_reserved_field_id.sh b/tests/queries/0_stateless/04337_iceberg_v3_row_lineage_reserved_field_id.sh index 1372f4d60f90..c56f43b2a637 100755 --- a/tests/queries/0_stateless/04337_iceberg_v3_row_lineage_reserved_field_id.sh +++ b/tests/queries/0_stateless/04337_iceberg_v3_row_lineage_reserved_field_id.sh @@ -47,7 +47,11 @@ pq.write_table(table, path) PY # A spec-compliant reader ignores the reserved field id and returns the projected column. -${CLICKHOUSE_CLIENT} --query "SELECT x FROM icebergLocal('${ICEBERG_TABLE_PATH}') ORDER BY x;" +# `allow_experimental_iceberg_read_optimization` is disabled here (and below) on purpose: it can +# answer a query from the manifest statistics alone (e.g. when a projected column is constant in the +# file), in which case the data file is never opened and the parquet schema is never converted, so +# this test would not exercise the reserved-field-id handling it is about. +${CLICKHOUSE_CLIENT} --query "SELECT x FROM icebergLocal('${ICEBERG_TABLE_PATH}') ORDER BY x SETTINGS allow_experimental_iceberg_read_optimization = 0;" # Conversely, 2147483447 (Integer.MAX_VALUE - 200) is the highest field id a table may use, i.e. # NOT reserved. An unmapped column with that id is a genuine schema mismatch and must still be @@ -79,7 +83,7 @@ table = pa.table( pq.write_table(table, path) PY -${CLICKHOUSE_CLIENT} --query "SELECT x FROM icebergLocal('${ICEBERG_TABLE_PATH_UNMAPPED}') ORDER BY x;" 2>&1 | grep -oF "ICEBERG_SPECIFICATION_VIOLATION" | head -1 +${CLICKHOUSE_CLIENT} --query "SELECT x FROM icebergLocal('${ICEBERG_TABLE_PATH_UNMAPPED}') ORDER BY x SETTINGS allow_experimental_iceberg_read_optimization = 0;" 2>&1 | grep -oF "ICEBERG_SPECIFICATION_VIOLATION" | head -1 # Cleanup ${CLICKHOUSE_CLIENT} --query "DROP TABLE IF EXISTS t_v3_row_lineage;"