diff --git a/src/Client/BuzzHouse/Generator/SessionSettings.cpp b/src/Client/BuzzHouse/Generator/SessionSettings.cpp index cdc597a90780..eae9e170ec0e 100644 --- a/src/Client/BuzzHouse/Generator/SessionSettings.cpp +++ b/src/Client/BuzzHouse/Generator/SessionSettings.cpp @@ -638,6 +638,7 @@ std::unordered_map serverSettings = { {"create_index_ignore_unique", trueOrFalseSettingNoOracle}, {"create_table_empty_primary_key_by_default", trueOrFalseSettingNoOracle}, {"cross_to_inner_join_rewrite", CHSetting(zeroOneTwo, {"0", "1", "2"}, false)}, + {"data_lake_delete_data_on_drop", trueOrFalseSettingNoOracle}, {"database_atomic_wait_for_drop_and_detach_synchronously", trueOrFalseSettingNoOracle}, {"database_datalake_require_metadata_access", trueOrFalseSettingNoOracle}, {"database_replicated_allow_explicit_uuid", CHSetting(zeroOneTwo, {}, false)}, @@ -873,7 +874,6 @@ std::unordered_map serverSettings = { [](RandomGenerator & rg, FuzzConfig &) { return std::to_string(rg.thresholdGenerator(0.3, 0.2, 0, 10800)); }, {}, false)}, - {"iceberg_delete_data_on_drop", trueOrFalseSettingNoOracle}, {"iceberg_expire_default_min_snapshots_to_keep", CHSetting( [](RandomGenerator & rg, FuzzConfig &) { return std::to_string(rg.thresholdGenerator(0.2, 0.2, 0, 10)); }, {}, false)}, diff --git a/src/Core/Settings.cpp b/src/Core/Settings.cpp index d0725772d8aa..46ab78c16f90 100644 --- a/src/Core/Settings.cpp +++ b/src/Core/Settings.cpp @@ -5554,9 +5554,9 @@ Possible values: - manifest_file_entry - Everything above + traversed avro manifest files entries. )", 0) \ \ - DECLARE(Bool, iceberg_delete_data_on_drop, false, R"( -Whether to delete all iceberg files on drop or not. -)", 0) \ + DECLARE_WITH_ALIAS(Bool, data_lake_delete_data_on_drop, false, R"( +Whether to delete the underlying data files when dropping a data lake table. For catalog databases the catalog is asked to purge the data (`purgeRequested=true`); for self-managed tables ClickHouse removes the files directly. +)", 0, iceberg_delete_data_on_drop) \ DECLARE(Int64, iceberg_expire_default_min_snapshots_to_keep, 1, R"( Default value for Iceberg table property `history.expire.min-snapshots-to-keep` used by `expire_snapshots` when that property is absent. )", 0) \ @@ -7651,6 +7651,10 @@ Allow experimental database engine DataLakeCatalog with catalog_type = 'iceberg' Cloud default value: `1`. )", BETA, allow_database_iceberg) \ + DECLARE(Bool, datalake_ignore_unsupported_table_properties, false, R"( +Allow `CREATE TABLE`, `CREATE TABLE ... AS`, and `SHOW CREATE TABLE` in a `DataLakeCatalog` database to omit unsupported table properties, including explicitly specified and inherited properties. Supported properties are preserved. By default, properties that cannot be represented cause an exception. +This setting does not suppress invalid expressions, unknown columns, invalid transform arguments, or incompatible storage engines, endpoints, and credentials. +)", BETA) \ DECLARE_WITH_ALIAS(Bool, allow_experimental_database_unity_catalog, true, R"( Allow experimental database engine DataLakeCatalog with catalog_type = 'unity' diff --git a/src/Core/SettingsChangesHistory.cpp b/src/Core/SettingsChangesHistory.cpp index f093c4054cf6..6fba068ccbb4 100644 --- a/src/Core/SettingsChangesHistory.cpp +++ b/src/Core/SettingsChangesHistory.cpp @@ -42,6 +42,8 @@ const VersionToSettingsChangesMap & getSettingsChangesHistory() addSettingsChanges(settings_changes_history, "26.6.2.20001.altinityantalya", { {"use_puffin_files_cache", false, true, "Enables cache of parsed Puffin file content such as deletion vectors."}, + {"data_lake_delete_data_on_drop", false, false, "New setting that unifies dropping of data lake data; the released `iceberg_delete_data_on_drop` is kept as an alias for it."}, + {"datalake_ignore_unsupported_table_properties", false, false, "New setting: allow `CREATE TABLE` and `SHOW CREATE TABLE` in a `DataLakeCatalog` to omit unsupported table properties."}, }); addSettingsChanges(settings_changes_history, "26.6", diff --git a/src/Databases/DataLake/Common.cpp b/src/Databases/DataLake/Common.cpp index 8946d3412d70..cdc26f2c86c2 100644 --- a/src/Databases/DataLake/Common.cpp +++ b/src/Databases/DataLake/Common.cpp @@ -16,6 +16,9 @@ #include +#include +#include + namespace DB::ErrorCodes { extern const int BAD_ARGUMENTS; @@ -121,4 +124,88 @@ std::pair parseTableName(const std::string & name) return {namespace_name, table_name}; } +String constructTableLocation( + const String & location_scheme, + const String & storage_endpoint, + const String & namespace_name, + const String & table_name, + DB::S3UriStyle uri_style) +{ + Poco::URI uri(storage_endpoint); + auto path = uri.getPath(); + while (path.starts_with('/')) + path.erase(0, 1); + while (path.ends_with('/')) + path.pop_back(); + + if (location_scheme == "abfss") + { + /// Azure: `abfss://@/`. `storage_endpoint` is + /// `https:////` or `abfss://@/` + String container = uri.getUserInfo(); + String account_host = uri.getHost(); + String extra_path = path; + + if (container.empty()) + { + auto first_slash = extra_path.find('/'); + if (first_slash == String::npos) + { + container = std::move(extra_path); + extra_path.clear(); + } + else + { + container = extra_path.substr(0, first_slash); + extra_path = extra_path.substr(first_slash + 1); + } + } + + if (account_host.empty() || container.empty()) + throw DB::Exception( + DB::ErrorCodes::BAD_ARGUMENTS, + "`storage_endpoint` ({}) for Azure must include both account host and container " + "(expected https://.dfs.core.windows.net/[/] or " + "abfss://@.dfs.core.windows.net[/])", + storage_endpoint); + + if (extra_path.empty()) + return fmt::format("abfss://{}@{}/{}/{}", container, account_host, namespace_name, table_name); + return fmt::format("abfss://{}@{}/{}/{}/{}", container, account_host, extra_path, namespace_name, table_name); + } + + if (location_scheme == "s3") + { + if (uri_style == DB::S3UriStyle::VIRTUAL_HOSTED) + throw DB::Exception( + DB::ErrorCodes::BAD_ARGUMENTS, + "CREATE TABLE with `storage_uri_style = 'virtual_hosted'` cannot derive the bucket from " + "`storage_endpoint` ({}); set `default_base_location` (a full s3:///) instead.", + storage_endpoint); + + if (path.empty()) + throw DB::Exception( + DB::ErrorCodes::BAD_ARGUMENTS, + "`storage_endpoint` ({}) does not contain a bucket; " + "CREATE TABLE in DataLakeCatalog requires `storage_endpoint` to include a non-empty bucket path.", + storage_endpoint); + return fmt::format("s3://{}/{}/{}", path, namespace_name, table_name); + } + + String authority = uri.getAuthority(); + if (authority.empty()) + { + if (path.empty()) + throw DB::Exception( + DB::ErrorCodes::BAD_ARGUMENTS, + "`storage_endpoint` ({}) does not contain a path", + storage_endpoint); + return fmt::format("{}:///{}/{}/{}", location_scheme, path, namespace_name, table_name); + } + + if (path.empty()) + return fmt::format("{}://{}/{}/{}", location_scheme, authority, namespace_name, table_name); + return fmt::format("{}://{}/{}/{}/{}", location_scheme, authority, path, namespace_name, table_name); +} + } diff --git a/src/Databases/DataLake/Common.h b/src/Databases/DataLake/Common.h index 9b0dd7c626a6..912758f7fae8 100644 --- a/src/Databases/DataLake/Common.h +++ b/src/Databases/DataLake/Common.h @@ -1,6 +1,7 @@ #pragma once #include +#include #include #include @@ -19,4 +20,11 @@ DB::DataTypePtr getType(const String & type_name, bool nullable, DB::ContextPtr /// `E` is a table name. std::pair parseTableName(const std::string & name); +String constructTableLocation( + const String & location_scheme, + const String & storage_endpoint, + const String & namespace_name, + const String & table_name, + DB::S3UriStyle uri_style = DB::S3UriStyle::AUTO); + } diff --git a/src/Databases/DataLake/DatabaseDataLake.cpp b/src/Databases/DataLake/DatabaseDataLake.cpp index 6f2a608398f6..248ba5ea66b1 100644 --- a/src/Databases/DataLake/DatabaseDataLake.cpp +++ b/src/Databases/DataLake/DatabaseDataLake.cpp @@ -17,6 +17,7 @@ #include #include #include +#include #if USE_AVRO && USE_PARQUET @@ -44,6 +45,7 @@ #include #include +#include #include #include @@ -51,8 +53,11 @@ #include #include #include +#include +#include #include #include +#include namespace DB { @@ -64,6 +69,7 @@ namespace DatabaseDataLakeSetting extern const DatabaseDataLakeSettingsString auth_header; extern const DatabaseDataLakeSettingsString auth_scope; extern const DatabaseDataLakeSettingsString storage_endpoint; + extern const DatabaseDataLakeSettingsString default_base_location; extern const DatabaseDataLakeSettingsS3UriStyle storage_uri_style; extern const DatabaseDataLakeSettingsString oauth_server_uri; extern const DatabaseDataLakeSettingsBool oauth_server_use_request_body; @@ -108,8 +114,10 @@ namespace Setting extern const SettingsBool parallel_replicas_for_cluster_engines; extern const SettingsString cluster_for_parallel_replicas; extern const SettingsBool database_datalake_require_metadata_access; + extern const SettingsBool data_lake_delete_data_on_drop; + extern const SettingsBool datalake_ignore_unsupported_table_properties; extern const SettingsBool show_data_lake_catalogs_in_system_tables; - + extern const SettingsString iceberg_metadata_compression_method; } namespace DataLakeStorageSetting @@ -125,6 +133,8 @@ namespace ErrorCodes extern const int DATALAKE_DATABASE_ERROR; extern const int CANNOT_GET_CREATE_TABLE_QUERY; extern const int LOGICAL_ERROR; + extern const int NOT_IMPLEMENTED; + extern const int TABLE_ALREADY_EXISTS; } namespace FailPoints @@ -134,6 +144,23 @@ namespace FailPoints extern const char datalake_get_tables_throw[]; } +namespace +{ + +String getLocationScheme(const std::shared_ptr & catalog) +{ + if (auto storage_type = catalog->getStorageType(); storage_type.has_value()) + return DataLake::storageTypeToScheme(*storage_type); + + throw Exception( + ErrorCodes::BAD_ARGUMENTS, + "Catalog type '{}' does not report a backing storage type, so the location scheme is unknown. " + "Set `default_base_location` on the database or configure the catalog to expose `default-base-location`", + catalog->getCatalogType()); +} + +} + DatabaseDataLake::DatabaseDataLake( const std::string & database_name_, const std::string & url_, @@ -704,9 +731,8 @@ StoragePtr DatabaseDataLake::tryGetTableImpl(const String & name, ContextPtr con if (!metadata_location.empty()) { metadata_location = table_metadata.getMetadataLocation(metadata_location); + (*storage_settings)[DB::DataLakeStorageSetting::iceberg_metadata_file_path] = metadata_location; } - - (*storage_settings)[DB::DataLakeStorageSetting::iceberg_metadata_file_path] = metadata_location; } const auto configuration = getConfiguration(storage_type, storage_settings); @@ -806,16 +832,251 @@ StoragePtr DatabaseDataLake::tryGetTableImpl(const String & name, ContextPtr con return storage_cluster; } -void DatabaseDataLake::dropTable( /// NOLINT +void DatabaseDataLake::validateCreateTableEngine(const ASTStorage & storage) const +{ + const ASTFunction & engine = *storage.engine; + const auto catalog = getCatalog(); + + const String & family_name = table_engine_definition->as().engine->name; + + if (!engine.name.starts_with(family_name)) + throw Exception(ErrorCodes::BAD_ARGUMENTS, + "This DataLakeCatalog stores {}-family tables; got table engine '{}'", + family_name, engine.name); + + if (catalog->managesTableLocation()) + throw Exception( + ErrorCodes::BAD_ARGUMENTS, + "This DataLakeCatalog assigns table locations itself, so an explicit table `ENGINE` with a " + "user-provided location is not supported; omit the `ENGINE` clause"); + + const auto catalog_storage_type = catalog->getStorageType(); + + std::optional engine_backend; + const std::string_view backend_name = std::string_view(engine.name).substr(family_name.size()); + if (backend_name == "S3") + engine_backend = DatabaseDataLakeStorageType::S3; + else if (backend_name == "Azure") + engine_backend = DatabaseDataLakeStorageType::Azure; + else if (backend_name == "HDFS") + engine_backend = DatabaseDataLakeStorageType::HDFS; + else if (backend_name == "Local") + engine_backend = DatabaseDataLakeStorageType::Local; + else if (backend_name.empty()) + { + /// The generic family engine picks its backend from the `disk` setting and defaults to S3, + /// the same way the storage factory does. An unsupported disk type is rejected by the factory. + engine_backend = DatabaseDataLakeStorageType::S3; + const Field * disk_name = storage.settings ? storage.settings->changes.tryGet("disk") : nullptr; + if (disk_name) + { + switch (Context::getGlobalContextInstance()->getDisk(disk_name->safeGet())->getObjectStorage()->getType()) + { + case ObjectStorageType::S3: + break; + case ObjectStorageType::Azure: + engine_backend = DatabaseDataLakeStorageType::Azure; + break; + case ObjectStorageType::Local: + engine_backend = DatabaseDataLakeStorageType::Local; + break; + default: + engine_backend.reset(); + break; + } + } + } + + if (engine_backend.has_value() && catalog_storage_type.has_value() && *catalog_storage_type != *engine_backend) + throw Exception(ErrorCodes::BAD_ARGUMENTS, + "Table engine '{}' uses the {} storage backend, but this DataLakeCatalog stores tables on {}. " + "The table would be reopened with the catalog's storage backend and become unreadable " + "immediately after creation. Use the {}{} engine", + engine.name, *engine_backend, *catalog_storage_type, family_name, *catalog_storage_type); + + validateCreateTableEngineArguments(engine); +} + +void DatabaseDataLake::validateCreateTableEngineArguments(const ASTFunction & engine) const +{ + static const ASTs no_arguments; + const ASTs & engine_args = engine.arguments ? engine.arguments->children : no_arguments; + + const ASTFunction & database_engine = *table_engine_definition->as().engine; + const ASTs & database_args = database_engine.arguments ? database_engine.arguments->children : no_arguments; + + const size_t expected_size = std::max(database_args.size(), 1); + bool arguments_match = engine_args.size() == expected_size; + for (size_t i = 1; arguments_match && i < engine_args.size(); ++i) + arguments_match = engine_args[i]->getTreeHash(/*ignore_aliases=*/true) == database_args[i]->getTreeHash(/*ignore_aliases=*/true); + + if (!arguments_match) + throw Exception(ErrorCodes::BAD_ARGUMENTS, + "Table engine '{}' takes the table location followed by the {} remaining argument(s) of the database " + "engine, got {} argument(s). A DataLakeCatalog table is reopened from its catalog location and the " + "database engine arguments, so per-table storage credentials cannot be preserved", + engine.name, expected_size - 1, engine_args.size()); + + const auto settings_version = database_settings.get(); + const DatabaseDataLakeSettings & settings = *settings_version; + + const auto storage_endpoint = settings[DatabaseDataLakeSetting::storage_endpoint].value; + if (storage_endpoint.empty()) + return; + + if (settings[DatabaseDataLakeSetting::storage_uri_style].value == S3UriStyle::VIRTUAL_HOSTED) + throw Exception( + ErrorCodes::BAD_ARGUMENTS, + "An explicit table `ENGINE` is not supported with `storage_uri_style = 'virtual_hosted'`: " + "the table location cannot be guaranteed to survive catalog round-trip reconstruction. " + "Omit the `ENGINE` clause and use `default_base_location` instead"); + + const auto * location = engine_args[0]->as(); + if (!location || location->value.getType() != Field::Types::String) + throw Exception(ErrorCodes::BAD_ARGUMENTS, + "Table engine '{}' requires the table location as a string literal", engine.name); + + String endpoint_prefix = storage_endpoint; + while (endpoint_prefix.ends_with('/')) + endpoint_prefix.pop_back(); + + const auto & location_str = location->value.safeGet(); + if (!location_str.starts_with(endpoint_prefix + "/")) + throw Exception(ErrorCodes::BAD_ARGUMENTS, + "Table location '{}' is outside the `storage_endpoint` ('{}') of this DataLakeCatalog. " + "The table would be reopened under `storage_endpoint` and become unreadable " + "immediately after creation", + location_str, storage_endpoint); +} + +void DatabaseDataLake::createTable( ContextPtr context_, const String & name, - bool /*sync*/) + const StoragePtr & table, + const ASTPtr & query) { - auto table = tryGetTable(name, context_); + /// Engine-clause path: `IcebergMetadata::createInitial` has already written the metadata and + /// registered the table. if (table) - table->drop(); + return; + + auto catalog = getCatalog(); + + const String & family_name = table_engine_definition->as().engine->name; + if (family_name != "Iceberg") + throw Exception(ErrorCodes::NOT_IMPLEMENTED, + "CREATE TABLE without a table engine is not implemented for a {} DataLakeCatalog; " + "give an explicit {}-family table engine with the table location", + family_name, family_name); + + const auto & create = query->as(); + const auto [namespace_name, table_name] = DataLake::parseTableName(name); + + ColumnsDescription columns; + if (create.columns_list && create.columns_list->columns) + { + for (const auto & child : create.columns_list->columns->children) + { + const auto * col_decl = child->as(); + if (!col_decl || !col_decl->getType()) + throw Exception(ErrorCodes::BAD_ARGUMENTS, "Invalid column declaration in CREATE TABLE"); + + columns.add(ColumnDescription(col_decl->name, DataTypeFactory::instance().get(col_decl->getType()))); + } + } + + if (columns.empty()) + throw Exception(ErrorCodes::BAD_ARGUMENTS, "Cannot create table without columns"); + + ASTPtr partition_by; + ASTPtr order_by; + if (create.storage) + { + if (create.storage->partition_by) + partition_by = create.storage->partition_by->clone(); + if (create.storage->order_by) + order_by = create.storage->order_by->clone(); + } + + const auto settings_version = database_settings.get(); + const DatabaseDataLakeSettings & settings = *settings_version; + + String base_location = catalog->getDefaultBaseLocation(); + if (base_location.empty()) + base_location = settings[DatabaseDataLakeSetting::default_base_location].value; + + String location; + if (!base_location.empty()) + { + if (auto catalog_storage_type = catalog->getStorageType(); catalog_storage_type.has_value() + && DataLake::parseStorageTypeFromLocation(base_location) != *catalog_storage_type) + throw Exception(ErrorCodes::BAD_ARGUMENTS, + "`default_base_location` uses the {} storage backend, but this DataLakeCatalog stores tables on {}. " + "The table would be reopened with the catalog's storage backend and become unreadable " + "immediately after creation", + DataLake::parseStorageTypeFromLocation(base_location), *catalog_storage_type); + + while (base_location.ends_with('/')) + base_location.pop_back(); + location = fmt::format("{}/{}/{}", base_location, namespace_name, table_name); + } else - throw Exception(ErrorCodes::BAD_ARGUMENTS, "Cannot drop table {} because it does not exist", name); + { + const auto storage_endpoint = settings[DatabaseDataLakeSetting::storage_endpoint].value; + if (storage_endpoint.empty()) + throw Exception(ErrorCodes::BAD_ARGUMENTS, + "CREATE TABLE in DataLakeCatalog requires `default_base_location` or `storage_endpoint`"); + + if (catalog->managesTableLocation()) + throw Exception(ErrorCodes::BAD_ARGUMENTS, + "This DataLakeCatalog assigns table locations itself, so `storage_endpoint` cannot be used " + "to place a new table. Set `default_base_location` on the database instead"); + + location = DataLake::constructTableLocation( + getLocationScheme(catalog), storage_endpoint, namespace_name, table_name, + settings[DatabaseDataLakeSetting::storage_uri_style]); + } + + auto metadata_content = Iceberg::createEmptyMetadataFile( + location, + columns, + partition_by, + order_by, + context_, + /*format_version=*/ 2, + /*is_catalog_table=*/ true).first; + + const auto compression_method_str = context_->getSettingsRef()[Setting::iceberg_metadata_compression_method].value; + const auto compression_method = chooseCompressionMethod(compression_method_str, compression_method_str); + + /// The namespace's default location is the namespace base, not this first table's directory. + String namespace_location = location; + if (const String table_suffix = "/" + table_name; namespace_location.ends_with(table_suffix)) + namespace_location.resize(namespace_location.size() - table_suffix.size()); + catalog->createNamespaceIfNotExists(namespace_name, namespace_location); + + const bool created = catalog->createTable( + namespace_name, table_name, /* metadata_path */ "", metadata_content, compression_method, create.if_not_exists); + if (!created) + throw Exception(ErrorCodes::TABLE_ALREADY_EXISTS, + "Table {}.{} already exists in the catalog", namespace_name, table_name); + + LOG_INFO(log, "Created table {}.{}", namespace_name, table_name); +} + +void DatabaseDataLake::dropTable( /// NOLINT + ContextPtr context_, + const String & name, + bool /*sync*/, + bool if_exists) +{ + auto catalog = getCatalog(); + const auto [namespace_name, table_name] = DataLake::parseTableName(name); + + bool purge = context_->getSettingsRef()[Setting::data_lake_delete_data_on_drop]; + catalog->dropTable(namespace_name, table_name, purge, if_exists); + + LOG_INFO(log, "Dropped table {}.{} from DataLakeCatalog (purge={})", namespace_name, table_name, purge); } DatabaseTablesIteratorPtr DatabaseDataLake::getTablesIterator( @@ -1082,7 +1343,7 @@ ASTPtr DatabaseDataLake::getCreateTableQueryImpl( const DatabaseDataLakeSettings & settings = *settings_version; auto catalog = getCatalog(); - auto table_metadata = DataLake::TableMetadata().withLocation().withSchema(); + auto table_metadata = DataLake::TableMetadata().withLocation().withSchema().withPartitionAndSortingKeys(); if (settings[DatabaseDataLakeSetting::force_add_bucket]) table_metadata.withForceAddBucket(); @@ -1095,17 +1356,18 @@ ASTPtr DatabaseDataLake::getCreateTableQueryImpl( return {}; } - auto create_table_query = make_intrusive(); - auto table_storage_define = table_engine_definition->clone(); - - auto * storage = table_storage_define->as(); - storage->engine->setKind(ASTFunction::Kind::TABLE_ENGINE); - if (!table_metadata.isDefaultReadableTable()) - storage->engine->name = DataLake::FAKE_TABLE_ENGINE_NAME_FOR_UNREADABLE_TABLES; - - storage->settings = {}; + if (!table_metadata.getUnsupportedProperties().empty() + && !context_->getSettingsRef()[Setting::datalake_ignore_unsupported_table_properties]) + { + if (!throw_on_error) + return {}; + throw Exception(ErrorCodes::CANNOT_GET_CREATE_TABLE_QUERY, + "Cannot represent {} of table {}.{} in CREATE TABLE. " + "Set datalake_ignore_unsupported_table_properties = 1 to omit unsupported properties", + table_metadata.getUnsupportedProperties(), getDatabaseName(), name); + } - create_table_query->set(create_table_query->storage, table_storage_define); + auto create_table_query = make_intrusive(); auto columns_declare_list = make_intrusive(); auto columns_expression_list = make_intrusive(); @@ -1125,6 +1387,40 @@ ASTPtr DatabaseDataLake::getCreateTableQueryImpl( columns_expression_list->children.emplace_back(column_declaration); } + const auto partition_by = table_metadata.getPartitionBy(); + const auto order_by = table_metadata.getOrderBy(); + + /// The catalog rejects an explicit `ENGINE` in `CREATE TABLE` when it assigns table locations itself, + /// so the `ENGINE` clause is omitted to keep the query replayable. + if (catalog->managesTableLocation()) + { + if (partition_by || order_by) + { + auto storage = make_intrusive(); + if (partition_by) + storage->set(storage->partition_by, partition_by); + if (order_by) + storage->set(storage->order_by, order_by); + create_table_query->set(create_table_query->storage, storage); + } + return create_table_query; + } + + auto table_storage_define = table_engine_definition->clone(); + + auto * storage = table_storage_define->as(); + storage->engine->setKind(ASTFunction::Kind::TABLE_ENGINE); + if (!table_metadata.isDefaultReadableTable()) + storage->engine->name = DataLake::FAKE_TABLE_ENGINE_NAME_FOR_UNREADABLE_TABLES; + + storage->settings = {}; + if (partition_by) + storage->set(storage->partition_by, partition_by); + if (order_by) + storage->set(storage->order_by, order_by); + + create_table_query->set(create_table_query->storage, table_storage_define); + auto storage_engine_arguments = storage->engine->arguments; if (table_metadata.isDefaultReadableTable()) { @@ -1391,6 +1687,7 @@ The following settings are supported: | `auth_header` | Custom HTTP header for authentication with the catalog service | | `auth_scope` | OAuth2 scope for authentication (if using OAuth) | | `storage_endpoint` | Endpoint URL for the underlying storage | +| `default_base_location` | Base URI for new tables when the catalog does not report `default-base-location`. New tables are placed under `//` (e.g. `s3://warehouse/data`) | | `oauth_server_uri` | URI of the OAuth2 authorization server for authentication | | `vended_credentials` | Boolean indicating whether to use vended credentials from the catalog (supports AWS S3 and Azure ADLS Gen2) | | `vended_credentials_cache_ttl` | Maximum cache entry lifetime (in seconds) for vended credentials (REST catalogs only). Default `300`; `0` disables caching. | @@ -1400,6 +1697,109 @@ The following settings are supported: | `dlf_access_key_id` | Access key ID for DLF access | | `dlf_access_key_secret` | Access key Secret for DLF access | +## Creating tables {#creating-tables} + +An Iceberg table in a `DataLakeCatalog` database can be created directly from ClickHouse. + +:::note +`CREATE TABLE` and `DROP TABLE` require a catalog that can perform catalog mutations. They are supported +for Iceberg REST catalogs (including OneLake, BigLake, and Delta Sharing) and for the AWS Glue catalog. +Other catalog types (Hive Metastore, Paimon REST) reject these statements. A Unity catalog stores Delta +tables, which are created with an explicit `DeltaLake` table engine (see below) and cannot be dropped +through the catalog. +::: + +The location of a newly created table comes from `default_base_location` (a full `s3://bucket/prefix`) when +set, otherwise the bucket is derived from `storage_endpoint`. With `storage_uri_style = 'virtual_hosted'` the +bucket cannot be derived from the endpoint unambiguously, so `default_base_location` is required for +`CREATE TABLE`. + +The table name must be quoted with backticks and include the namespace separated by a dot: + +```sql +CREATE TABLE catalog_db.`namespace.table_name` +( + id Int64, + name String, + value Float64 +) +PARTITION BY id +ORDER BY name +SETTINGS allow_database_iceberg = 1; +``` + +Iceberg accepts only a fixed set of partition transforms, so `PARTITION BY` +must use one of the following expressions: + +| Expression | Iceberg transform | +|-------------------------------|-------------------| +| `` | `identity` | +| `toYearNumSinceEpoch()` | `year` | +| `toMonthNumSinceEpoch()` | `month` | +| `toRelativeDayNum()` | `day` | +| `toRelativeHourNum()` | `hour` | +| `icebergTruncate(N, )` | `truncate[N]` | +| `icebergBucket(N, )` | `bucket[N]` | + +Composite partitioning is supported via `PARTITION BY (expr1, expr2, ...)`. +Other expressions (e.g. `toYYYYMM`, `intDiv`) are rejected at `CREATE TABLE`. + +Only the column names and types, `PARTITION BY`, and `ORDER BY` are persisted into the +table metadata. Anything else — the storage clauses `PRIMARY KEY`, `SAMPLE BY`, `TTL`, and +`UNIQUE KEY`; engine `SETTINGS`; indices, constraints, and projections; and the column modifiers +`DEFAULT`, `MATERIALIZED`, `ALIAS`, `EPHEMERAL`, `COMMENT`, `CODEC`, `TTL`, `STATISTICS`, and `SETTINGS` — +is rejected rather than silently dropped. This applies both with and without an explicit +`ENGINE` clause. + +An explicit `ENGINE` clause selects the location of the new table; every other engine argument must +repeat the arguments of the database engine, and the location must be under `storage_endpoint` when it +is set. A table is reopened from its catalog location together with the database engine arguments, so a +per-table endpoint or per-table credentials would apply to the `CREATE` only and are rejected instead. +Explicit table engines are also rejected with `storage_uri_style = 'virtual_hosted'`, because their URLs +cannot be guaranteed to survive reconstruction from the catalog location. Omit the `ENGINE` clause and +set `default_base_location` instead. + +You can also create an Iceberg table that inherits the schema of an existing table: + +```sql +CREATE TABLE catalog_db.`namespace.table_name` +AS other_db.source_table +SETTINGS allow_database_iceberg = 1; +``` + +If the source table's `PARTITION BY` and `ORDER BY` use only the expressions +listed above, they are copied into the new Iceberg table. + +## Dropping tables {#dropping-tables} + +Tables can be dropped from a `DataLakeCatalog` database. +`DROP TABLE` sends a delete request to the remote catalog, which removes +the table entry from the catalog. + +```sql +DROP TABLE catalog_db.`namespace.table_name` +``` + +By default, ClickHouse does not request the catalog to delete the underlying data. In order to do it, use the `data_lake_delete_data_on_drop` setting: + +```sql +DROP TABLE catalog_db.`namespace.table_name` +SETTINGS data_lake_delete_data_on_drop = 1 +``` + +:::note +Whether data files are actually deleted depends on the catalog itself. +The `purgeRequested` flag is sent to the catalog, but the catalog may choose to ignore it. + +Some catalogs support only one of the two modes: + +- Glue: `DROP TABLE` only removes the catalog entry and never deletes the underlying data files. + `DROP TABLE` with `data_lake_delete_data_on_drop = 1` is rejected instead of silently leaving the data behind. +- S3 Tables (`catalog_type = 's3tables'`): a table cannot be dropped without deleting its data. + `DROP TABLE` without `data_lake_delete_data_on_drop = 1` is rejected. With the setting enabled, both + the table and its data are permanently deleted. +::: + ## Examples {#examples} See below sections for examples of using the `DataLakeCatalog` engine: diff --git a/src/Databases/DataLake/DatabaseDataLake.h b/src/Databases/DataLake/DatabaseDataLake.h index fc67aad2b133..09349358aa6b 100644 --- a/src/Databases/DataLake/DatabaseDataLake.h +++ b/src/Databases/DataLake/DatabaseDataLake.h @@ -57,16 +57,19 @@ class DatabaseDataLake final : public IDatabase, WithContext std::vector> getTablesForBackup(const FilterByNameFunction &, const ContextPtr &) const override { return {}; } + void validateCreateTableEngine(const ASTStorage & storage) const override; + void createTable( - ContextPtr /*context*/, - const String & /*name*/, + ContextPtr context, + const String & name, const StoragePtr & /*table*/, - const ASTPtr & /*query*/) override {} + const ASTPtr & query) override; void dropTable( /// NOLINT ContextPtr context_, const String & name, - bool /*sync*/) override; + bool /*sync*/, + bool if_exists) override; void applySettingsChanges(const SettingsChanges & settings_changes, ContextPtr query_context) override; @@ -92,6 +95,8 @@ class DatabaseDataLake final : public IDatabase, WithContext void validateSettings(); + void validateCreateTableEngineArguments(const ASTFunction & engine) const; + /// Builds `catalog_impl` based on the configured catalog type. Constructing a catalog can /// validate credentials and perform network I/O (e.g. RestCatalog reads the catalog config), /// so on ATTACH (server startup) it is deferred to the first access via `getCatalog` instead diff --git a/src/Databases/DataLake/DatabaseDataLakeSettings.cpp b/src/Databases/DataLake/DatabaseDataLakeSettings.cpp index b4cf78f25a33..43baa6b43eff 100644 --- a/src/Databases/DataLake/DatabaseDataLakeSettings.cpp +++ b/src/Databases/DataLake/DatabaseDataLakeSettings.cpp @@ -34,6 +34,7 @@ namespace ErrorCodes DECLARE(String, aws_role_session_name, "", "Role session name for AWS connection for Glue catalog", 0) \ DECLARE(String, aws_external_id, "", "External id for the AWS STS AssumeRole trust policy for Glue catalog", 0) \ DECLARE(String, storage_endpoint, "", "Object storage endpoint", 0) \ + DECLARE(String, default_base_location, "", "Base URI under which CREATE TABLE places new tables. Used only when the catalog does not report `default-base-location`", 0) \ DECLARE(S3UriStyle, storage_uri_style, S3UriStyle::AUTO, "URL style used when constructing object storage URLs from catalog-provided table locations. Use 'virtual_hosted' when the object storage server requires the bucket in the hostname (e.g. https://bucket.endpoint.com/path/)", 0) \ DECLARE(String, onelake_tenant_id, "", "Tenant id from azure", 0) \ DECLARE(String, onelake_client_id, "", "Client id from azure", 0) \ diff --git a/src/Databases/DataLake/GlueCatalog.cpp b/src/Databases/DataLake/GlueCatalog.cpp index 41c343bc4842..e761dd1ff506 100644 --- a/src/Databases/DataLake/GlueCatalog.cpp +++ b/src/Databases/DataLake/GlueCatalog.cpp @@ -9,6 +9,7 @@ #include #include #include +#include #include #include #include @@ -49,9 +50,11 @@ #include #include #include +#include #include #include #include +#include #include #include #include @@ -62,6 +65,8 @@ namespace DB::ErrorCodes extern const int DATALAKE_DATABASE_ERROR; extern const int FAULT_INJECTED; extern const int CATALOG_NAMESPACE_DISABLED; + extern const int NOT_IMPLEMENTED; + extern const int S3_ERROR; } namespace DB::FailPoints @@ -463,7 +468,7 @@ bool GlueCatalog::tryGetTableMetadata( auto setup_specific_properties = [&] { const auto & table_params = table_outcome.GetParameters(); - if (table_params.contains("metadata_location")) + if (table_params.contains("metadata_location") && !table_params.at("metadata_location").empty()) { result.setDataLakeSpecificProperties(DataLakeSpecificProperties{.iceberg_metadata_file_location = table_params.at("metadata_location")}); } @@ -547,6 +552,17 @@ bool GlueCatalog::tryGetTableMetadata( } result.setSchema(schema); } + + if (result.requiresPartitionAndSortingKeys() && result.isDefaultReadableTable()) + { + if (!result.getDataLakeSpecificProperties().has_value()) + setup_specific_properties(); + if (result.isDefaultReadableTable()) + { + auto [partition_by, order_by, unsupported_properties] = DB::Iceberg::getPartitionAndSortingKeyASTsFromMetadata(getIcebergMetadataObject(result)); + result.setPartitionAndSortingKeys(std::move(partition_by), std::move(order_by), std::move(unsupported_properties)); + } + } } else { @@ -640,13 +656,17 @@ Poco::JSON::Object::Ptr GlueCatalog::getOrFetchMetadataObject(const String & met } String GlueCatalog::getActualTimestampType(const String & column_name, const TableMetadata & table_metadata, const String & glue_column_type) const +{ + return resolveTimestampTypeFromMetadata(getIcebergMetadataObject(table_metadata), column_name, glue_column_type); +} + +Poco::JSON::Object::Ptr GlueCatalog::getIcebergMetadataObject(const TableMetadata & table_metadata) const { auto table_specific_properties = table_metadata.getDataLakeSpecificProperties(); if (!table_specific_properties.has_value()) throw DB::Exception(DB::ErrorCodes::BAD_ARGUMENTS, "Failed to read table metadata, reason why table is unreadable: {}", table_metadata.getReasonWhyTableIsUnreadable()); - auto metadata_object = getOrFetchMetadataObject(table_specific_properties->iceberg_metadata_file_location, table_metadata); - return resolveTimestampTypeFromMetadata(metadata_object, column_name, glue_column_type); + return getOrFetchMetadataObject(table_specific_properties->iceberg_metadata_file_location, table_metadata); } String GlueCatalog::resolveTimestampTypeFromMetadata( @@ -751,6 +771,21 @@ String GlueCatalog::resolveMetadataPathFromTableLocation(const String & table_lo void GlueCatalog::createNamespaceIfNotExists(const String & namespace_name, const String & /*location*/) const { + Aws::Glue::Model::GetDatabaseRequest get_request; + get_request.SetName(namespace_name); + + auto get_outcome = glue_client->GetDatabase(get_request); + if (get_outcome.IsSuccess()) + return; + + if (get_outcome.GetError().GetErrorType() != Aws::Glue::GlueErrors::ENTITY_NOT_FOUND) + { + throw DB::Exception( + DB::ErrorCodes::DATALAKE_DATABASE_ERROR, + "Exception calling GetDatabase for namespace {}: {}", + namespace_name, get_outcome.GetError().GetMessage()); + } + Aws::Glue::Model::CreateDatabaseRequest create_request; Aws::Glue::Model::DatabaseInput db_input; db_input.SetName(namespace_name); @@ -768,13 +803,110 @@ void GlueCatalog::createNamespaceIfNotExists(const String & namespace_name, cons } } -void GlueCatalog::createTable(const String & namespace_name, const String & table_name, const String & new_metadata_path, Poco::JSON::Object::Ptr metadata_content) const +bool GlueCatalog::createTable( + const String & namespace_name, + const String & table_name, + const String & new_metadata_path, + Poco::JSON::Object::Ptr metadata_content, + DB::CompressionMethod metadata_compression_method, + bool if_not_exists) const { if (!isNamespaceAllowed(namespace_name)) throw DB::Exception(DB::ErrorCodes::CATALOG_NAMESPACE_DISABLED, "Failed to create table {}, namespace {} is filtered by `namespaces` database parameter", table_name, namespace_name); + String effective_metadata_path = new_metadata_path; + + DB::ObjectStoragePtr written_metadata_storage; + String written_metadata_file; + + bool registered = false; + SCOPE_EXIT_SAFE({ + if (registered || !written_metadata_storage) + return; + + Aws::Glue::Model::GetTableRequest get_request; + get_request.SetDatabaseName(namespace_name); + get_request.SetName(table_name); + auto get_outcome = glue_client->GetTable(get_request); + + if (get_outcome.IsSuccess()) + { + const auto & table_parameters = get_outcome.GetResult().GetTable().GetParameters(); + auto it = table_parameters.find("metadata_location"); + if (it != table_parameters.end() && it->second == effective_metadata_path) + { + LOG_INFO( + log, + "Table {}.{} is registered in the Glue catalog and points at {}, keeping that file", + namespace_name, + table_name, + effective_metadata_path); + return; + } + } + else if (get_outcome.GetError().GetErrorType() != Aws::Glue::GlueErrors::ENTITY_NOT_FOUND) + { + LOG_WARNING( + log, + "Cannot tell whether table {}.{} was registered in the Glue catalog: {}. " + "Keeping the staged initial metadata file {}", + namespace_name, + table_name, + get_outcome.GetError().GetMessage(), + written_metadata_file); + return; + } + + LOG_INFO( + log, + "Table {}.{} was not registered in the Glue catalog, removing the staged initial metadata file {}", + namespace_name, + table_name, + written_metadata_file); + written_metadata_storage->removeObjectIfExists(DB::StoredObject(written_metadata_file)); + }); + + if (effective_metadata_path.empty() && metadata_content && metadata_content->has("location")) + { + String table_location = metadata_content->getValue("location"); + while (table_location.ends_with('/')) + table_location = table_location.substr(0, table_location.size() - 1); + + TableMetadata dummy_metadata; + auto [object_storage, bucket_name, table_path] = createObjectStorageForEarlyTableAccess(table_location, dummy_metadata); + + /// Name the file exactly like `IcebergMetadata::createInitial` does. + String compression_suffix = DB::toContentEncodingName(metadata_compression_method); + if (!compression_suffix.empty()) + compression_suffix = "." + compression_suffix; + + String metadata_filename = fmt::format("{}/metadata/v1{}.metadata.json", table_path, compression_suffix); + + std::ostringstream oss; // STYLE_CHECK_ALLOW_STD_STRING_STREAM + Poco::JSON::Stringifier::stringify(metadata_content, oss, 4); + String metadata_str = DB::removeEscapedSlashes(oss.str()); + + try + { + DB::Iceberg::writeMessageToFile(metadata_str, metadata_filename, object_storage, getContext(), "*", "", metadata_compression_method); + } + catch (const DB::Exception & e) + { + /// The write is guarded by `If-None-Match: *`, so S3 answers `PreconditionFailed` once the + /// initial metadata file is there - someone else created this table first. + if (if_not_exists && e.code() == DB::ErrorCodes::S3_ERROR && e.message().contains("PreconditionFailed")) + return false; + throw; + } + + written_metadata_storage = object_storage; + written_metadata_file = metadata_filename; + + effective_metadata_path = "s3://" + bucket_name + "/" + metadata_filename; + } + Aws::Glue::Model::CreateTableRequest request; request.SetDatabaseName(namespace_name); @@ -782,12 +914,15 @@ void GlueCatalog::createTable(const String & namespace_name, const String & tabl table_input.SetName(table_name); Aws::Glue::Model::StorageDescriptor sd; - fs::path original_path = new_metadata_path; + if (!effective_metadata_path.empty()) + { + fs::path original_path = effective_metadata_path; - fs::path parent = original_path.parent_path(); - fs::path grandparent = parent.parent_path(); + fs::path parent = original_path.parent_path(); + fs::path grandparent = parent.parent_path(); - sd.SetLocation(grandparent.c_str()); + sd.SetLocation(grandparent.c_str()); + } if (metadata_content) { @@ -799,7 +934,7 @@ void GlueCatalog::createTable(const String & namespace_name, const String & tabl table_input.SetTableType("ICEBERG"); Aws::Map parameters; - parameters["metadata_location"] = new_metadata_path; + parameters["metadata_location"] = effective_metadata_path; parameters["table_type"] = "ICEBERG"; table_input.SetParameters(parameters); @@ -815,7 +950,15 @@ void GlueCatalog::createTable(const String & namespace_name, const String & tabl } if (!response.IsSuccess()) - throw DB::Exception(DB::ErrorCodes::DATALAKE_DATABASE_ERROR, "Can not create metadata in glue catalog: {}", response.GetError().GetMessage()); + { + if (if_not_exists && response.GetError().GetErrorType() == Aws::Glue::GlueErrors::ALREADY_EXISTS) + return false; + throw DB::Exception( + DB::ErrorCodes::DATALAKE_DATABASE_ERROR, "Can not create metadata in glue catalog: {}", response.GetError().GetMessage()); + } + + registered = true; + return true; } bool GlueCatalog::updateTableInGlue( @@ -888,13 +1031,27 @@ bool GlueCatalog::updateSchema( return updateTableInGlue(namespace_name, table_name, new_metadata_path, columns); } -void GlueCatalog::dropTable(const String & namespace_name, const String & table_name) const +void GlueCatalog::dropTable(const String & namespace_name, const String & table_name, bool purge, bool if_exists) const { if (!isNamespaceAllowed(namespace_name)) throw DB::Exception(DB::ErrorCodes::CATALOG_NAMESPACE_DISABLED, "Failed to drop table {}, namespace {} is filtered by `namespaces` database parameter", table_name, namespace_name); + /// Glue's `DeleteTable` removes only the catalog entry; the client-side purge of the data files is + /// not implemented. + /// TODO: implement the client-side purge so `data_lake_delete_data_on_drop` can be honored for Glue. + if (purge) + { + if (if_exists && !existsTable(namespace_name, table_name)) + return; + + throw DB::Exception( + DB::ErrorCodes::NOT_IMPLEMENTED, + "data_lake_delete_data_on_drop is not supported for the Glue catalog: dropping only removes the Glue " + "catalog entry and does not delete the underlying data files"); + } + Aws::Glue::Model::DeleteTableRequest request; request.SetDatabaseName(namespace_name); request.SetName(table_name); @@ -907,7 +1064,8 @@ void GlueCatalog::dropTable(const String & namespace_name, const String & table_ response = glue_client->DeleteTable(request); } - if (!response.IsSuccess()) + if (!response.IsSuccess() + && !(if_exists && response.GetError().GetErrorType() == Aws::Glue::GlueErrors::ENTITY_NOT_FOUND)) throw DB::Exception( DB::ErrorCodes::DATALAKE_DATABASE_ERROR, "Can not delete table from glue catalog: {}", diff --git a/src/Databases/DataLake/GlueCatalog.h b/src/Databases/DataLake/GlueCatalog.h index 4722bcec7f54..28c7d4f03184 100644 --- a/src/Databases/DataLake/GlueCatalog.h +++ b/src/Databases/DataLake/GlueCatalog.h @@ -67,7 +67,13 @@ class GlueCatalog final : public ICatalog, private DB::WithContext return DB::DatabaseDataLakeCatalogType::GLUE; } - void createTable(const String & namespace_name, const String & table_name, const String & new_metadata_path, Poco::JSON::Object::Ptr metadata_content) const override; + bool createTable( + const String & namespace_name, + const String & table_name, + const String & new_metadata_path, + Poco::JSON::Object::Ptr metadata_content, + DB::CompressionMethod metadata_compression_method, + bool if_not_exists) const override; void createNamespaceIfNotExists(const String & namespace_name, const String & location) const override; @@ -82,7 +88,7 @@ class GlueCatalog final : public ICatalog, private DB::WithContext Int32 new_last_column_id, Poco::JSON::Object::Ptr metadata = nullptr) const override; - void dropTable(const String & namespace_name, const String & table_name) const override; + void dropTable(const String & namespace_name, const String & table_name, bool purge, bool if_exists) const override; /// Returns a callback that re-vends fresh AWS credentials from the configured /// credentials provider chain. Invoked by `ReadBufferFromS3` when an S3 call @@ -118,6 +124,8 @@ class GlueCatalog final : public ICatalog, private DB::WithContext /// `glue_column_type` is the raw Glue type (`"timestamp"` or `"timestamp_nano"`) used as a fallback when the column is not found in Iceberg metadata. String getActualTimestampType(const String & column_name, const TableMetadata & table_metadata, const String & glue_column_type) const; + Poco::JSON::Object::Ptr getIcebergMetadataObject(const TableMetadata & table_metadata) const; + String resolveMetadataPathFromTableLocation(const String & table_location, const TableMetadata & table_metadata) const; struct ObjectStorageWithPath diff --git a/src/Databases/DataLake/ICatalog.cpp b/src/Databases/DataLake/ICatalog.cpp index ddeeaa25afbf..84642fa66e2d 100644 --- a/src/Databases/DataLake/ICatalog.cpp +++ b/src/Databases/DataLake/ICatalog.cpp @@ -87,6 +87,25 @@ StorageType parseStorageTypeFromString(const std::string & type) return *storage_type; } +std::string storageTypeToScheme(StorageType type) +{ + switch (type) + { + case StorageType::S3: + return "s3"; + case StorageType::Azure: + return "abfss"; + case StorageType::Local: + return "file"; + case StorageType::HDFS: + return "hdfs"; + case StorageType::Other: + throw DB::Exception( + DB::ErrorCodes::BAD_ARGUMENTS, + "Cannot determine URI scheme for storage type 'Other'"); + } +} + void TableMetadata::setLocation(const std::string & location_) { if (!with_location) @@ -267,6 +286,32 @@ std::optional TableMetadata::getDataLakeSpecificProp return data_lake_specific_metadata; } +void TableMetadata::setPartitionAndSortingKeys(DB::ASTPtr partition_by_, DB::ASTPtr order_by_, std::string unsupported_properties_) +{ + if (!with_partition_and_sorting_keys) + throw DB::Exception(DB::ErrorCodes::LOGICAL_ERROR, "Partition and sorting keys were not requested"); + + partition_by = std::move(partition_by_); + order_by = std::move(order_by_); + unsupported_properties = std::move(unsupported_properties_); +} + +DB::ASTPtr TableMetadata::getPartitionBy() const +{ + if (!with_partition_and_sorting_keys) + throw DB::Exception(DB::ErrorCodes::LOGICAL_ERROR, "Partition and sorting keys were not requested"); + + return partition_by; +} + +DB::ASTPtr TableMetadata::getOrderBy() const +{ + if (!with_partition_and_sorting_keys) + throw DB::Exception(DB::ErrorCodes::LOGICAL_ERROR, "Partition and sorting keys were not requested"); + + return order_by; +} + StorageType TableMetadata::getStorageType() const { if (!storage_type_str.empty()) @@ -341,9 +386,17 @@ DB::SettingsChanges CatalogSettings::allChanged() const return changes; } -void ICatalog::createTable(const String & /*namespace_name*/, const String & /*table_name*/, const String & /*new_metadata_path*/, Poco::JSON::Object::Ptr /*metadata_content*/) const +bool ICatalog::createTable( + const String & /*namespace_name*/, + const String & /*table_name*/, + const String & /*new_metadata_path*/, + Poco::JSON::Object::Ptr /*metadata_content*/, + DB::CompressionMethod /*metadata_compression_method*/, + bool /*if_not_exists*/) const { - throw DB::Exception(DB::ErrorCodes::NOT_IMPLEMENTED, "createTable is not implemented"); + throw DB::Exception(DB::ErrorCodes::NOT_IMPLEMENTED, + "CREATE TABLE is not supported for this DataLakeCatalog catalog type; " + "it is available only for Iceberg REST (including OneLake, BigLake, Delta Sharing) and Glue catalogs"); } void ICatalog::createNamespaceIfNotExists(const String & /*namespace_name*/, const String & /*location*/) const @@ -368,9 +421,11 @@ bool ICatalog::updateSchema( throw DB::Exception(DB::ErrorCodes::NOT_IMPLEMENTED, "updateSchema is not implemented"); } -void ICatalog::dropTable(const String & /*namespace_name*/, const String & /*table_name*/) const +void ICatalog::dropTable(const String & /*namespace_name*/, const String & /*table_name*/, bool /*purge*/, bool /*if_exists*/) const { - throw DB::Exception(DB::ErrorCodes::NOT_IMPLEMENTED, "dropTable is not implemented"); + throw DB::Exception(DB::ErrorCodes::NOT_IMPLEMENTED, + "DROP TABLE is not supported for this DataLakeCatalog catalog type; " + "it is available only for Iceberg REST (including OneLake, BigLake, Delta Sharing) and Glue catalogs"); } ICatalog::PreparedSettingsChangesPtr ICatalog::prepareSettingsChanges(const DB::SettingsChanges & /*changes*/) diff --git a/src/Databases/DataLake/ICatalog.h b/src/Databases/DataLake/ICatalog.h index 9cb18c15177b..f52d39bf5e76 100644 --- a/src/Databases/DataLake/ICatalog.h +++ b/src/Databases/DataLake/ICatalog.h @@ -5,7 +5,9 @@ #include #include #include +#include #include +#include #include #include #include @@ -29,6 +31,7 @@ namespace DataLake using StorageType = DB::DatabaseDataLakeStorageType; StorageType parseStorageTypeFromLocation(const std::string & location); StorageType parseStorageTypeFromString(const std::string &type); +std::string storageTypeToScheme(StorageType type); /// Registry of `ALTER DATABASE ... MODIFY SETTING` validators. Each catalog that /// supports altering settings registers its own validator; catalog types without @@ -65,6 +68,7 @@ class TableMetadata TableMetadata & withStorageCredentials() { with_storage_credentials = true; return *this; } TableMetadata & withDataLakeSpecificProperties() { with_datalake_specific_metadata = true; return *this; } TableMetadata & withForceAddBucket() { force_add_bucket = true; return *this; } + TableMetadata & withPartitionAndSortingKeys() { with_partition_and_sorting_keys = true; return *this; } bool hasLocation() const; bool hasSchema() const; @@ -87,6 +91,14 @@ class TableMetadata void setDataLakeSpecificProperties(std::optional && metadata); std::optional getDataLakeSpecificProperties() const; + void setPartitionAndSortingKeys(DB::ASTPtr partition_by_, DB::ASTPtr order_by_, std::string unsupported_properties_); + const std::string & getUnsupportedProperties() const + { + return unsupported_properties; + } + DB::ASTPtr getPartitionBy() const; + DB::ASTPtr getOrderBy() const; + void setTableUUID(const std::string & uuid_) { table_uuid = uuid_; } std::optional getTableUUID() const { return table_uuid; } @@ -94,6 +106,7 @@ class TableMetadata bool requiresSchema() const { return with_schema; } bool requiresCredentials() const { return with_storage_credentials; } bool requiresDataLakeSpecificProperties() const { return with_datalake_specific_metadata; } + bool requiresPartitionAndSortingKeys() const { return with_partition_and_sorting_keys; } StorageType getStorageType() const; @@ -138,6 +151,10 @@ class TableMetadata /// Specific settings for iceberg and datalake std::optional data_lake_specific_metadata; + DB::ASTPtr partition_by; + DB::ASTPtr order_by; + std::string unsupported_properties; + std::string reason_why_table_is_not_readable; std::optional table_uuid; @@ -147,6 +164,7 @@ class TableMetadata bool with_schema = false; bool with_storage_credentials = false; bool with_datalake_specific_metadata = false; + bool with_partition_and_sorting_keys = false; std::string constructLocation(const std::string & endpoint_, DB::S3UriStyle uri_style) const; }; @@ -211,14 +229,32 @@ class ICatalog /// E.g. one of S3, Azure, Local, HDFS. virtual std::optional getStorageType() const = 0; + /// Catalog-wide base location for new tables, e.g. `s3://warehouse/data`. Empty if unknown. + virtual String getDefaultBaseLocation() const { return ""; } + /// Creates new table in catalog. Callers must ensure the namespace exists before /// writing any table files to storage: a catalog that shares its storage view with /// the data refuses to create a namespace over a plain directory those files create. - virtual void createTable(const String & namespace_name, const String & table_name, const String & new_metadata_path, Poco::JSON::Object::Ptr metadata_content) const; + /// `metadata_compression_method` applies only to catalogs that write the initial metadata file + /// themselves (`new_metadata_path` is empty): they must name it `v1..metadata.json` and compress + /// it accordingly, like `DB::IcebergMetadata::createInitial` does. + /// Returns `true` if this call created the table, `false` if `if_not_exists` is set and the shared + /// catalog reported that another client had already created it. + virtual bool createTable( + const String & namespace_name, + const String & table_name, + const String & new_metadata_path, + Poco::JSON::Object::Ptr metadata_content, + DB::CompressionMethod metadata_compression_method, + bool if_not_exists) const; /// Creates the namespace unless it already exists. virtual void createNamespaceIfNotExists(const String & namespace_name, const String & location) const; + /// Whether the catalog chooses the storage location of a new table itself, so `CREATE TABLE` must + /// neither propose one nor write the initial metadata file. + virtual bool managesTableLocation() const { return false; } + /// Updates metadata in catalog. virtual bool updateMetadata(const String & namespace_name, const String & table_name, const String & new_metadata_path, Poco::JSON::Object::Ptr new_snapshot) const; @@ -239,7 +275,8 @@ class ICatalog Poco::JSON::Object::Ptr metadata = nullptr) const; /// Drop table from catalog. - virtual void dropTable(const String & namespace_name, const String & table_name) const; + /// If `purge`, the catalog is also asked to delete the underlying data files. + virtual void dropTable(const String & namespace_name, const String & table_name, bool purge, bool if_exists) const; /// Does the catalog support transactions or anything like that? /// For example, the Iceberg REST catalog supports atomic operations "compare if snapshot X is equal to" and "add new snapshot Y". diff --git a/src/Databases/DataLake/RestCatalog.cpp b/src/Databases/DataLake/RestCatalog.cpp index 1247222ce654..16d0d5bcc730 100644 --- a/src/Databases/DataLake/RestCatalog.cpp +++ b/src/Databases/DataLake/RestCatalog.cpp @@ -41,6 +41,7 @@ #include #include +#include #include #include @@ -192,6 +193,22 @@ String encodeNamespaceForURI(const String & namespace_name) return encoded; } +/// Per Iceberg REST spec, `namespace` is a JSON array of segments. Split `ns.a.b` on dots. +Poco::JSON::Array::Ptr namespaceToJSONArray(const String & namespace_name) +{ + Poco::JSON::Array::Ptr segments = new Poco::JSON::Array; + size_t start = 0; + while (start <= namespace_name.size()) + { + size_t dot = namespace_name.find('.', start); + if (dot == String::npos) + dot = namespace_name.size(); + segments->add(namespace_name.substr(start, dot - start)); + start = dot + 1; + } + return segments; +} + std::unordered_set getAllowedBigLakeMetadataServiceHosts( const Poco::Util::AbstractConfiguration & config) { @@ -207,7 +224,6 @@ std::unordered_set getAllowedBigLakeMetadataServiceHosts( return allowed; } - } namespace @@ -307,9 +323,7 @@ Poco::JSON::Object::Ptr buildUpdateSchemaRequestBody( { Poco::JSON::Object::Ptr identifier = new Poco::JSON::Object; identifier->set("name", table_name); - Poco::JSON::Array::Ptr namespaces = new Poco::JSON::Array; - namespaces->add(namespace_name); - identifier->set("namespace", namespaces); + identifier->set("namespace", namespaceToJSONArray(namespace_name)); request_body->set("identifier", identifier); } @@ -388,9 +402,7 @@ Poco::JSON::Object::Ptr buildUpdateMetadataRequestBody( { Poco::JSON::Object::Ptr identifier = new Poco::JSON::Object; identifier->set("name", table_name); - Poco::JSON::Array::Ptr namespaces = new Poco::JSON::Array; - namespaces->add(namespace_name); - identifier->set("namespace", namespaces); + identifier->set("namespace", namespaceToJSONArray(namespace_name)); request_body->set("identifier", identifier); } @@ -1160,6 +1172,11 @@ std::optional RestCatalog::getStorageType() const return parseStorageTypeFromLocation(state_snapshot->config.default_base_location); } +String RestCatalog::getDefaultBaseLocation() const +{ + return state.get()->config.default_base_location; +} + DB::ReadWriteBufferFromHTTPPtr RestCatalog::createReadBuffer( const CatalogState & catalog_state, const std::string & endpoint, @@ -1827,6 +1844,12 @@ bool RestCatalog::getTableMetadataImpl( result.setSchema(*schema); } + if (result.requiresPartitionAndSortingKeys()) + { + auto [partition_by, order_by, unsupported_properties] = DB::Iceberg::getPartitionAndSortingKeyASTsFromMetadata(metadata_object); + result.setPartitionAndSortingKeys(std::move(partition_by), std::move(order_by), std::move(unsupported_properties)); + } + if (want_credentials && result.isDefaultReadableTable()) { if (cached_credentials) @@ -1975,11 +1998,9 @@ void RestCatalog::createNamespaceIfNotExists(const String & namespace_name, cons const std::string endpoint = (base_url / state_snapshot->config.prefix / NAMESPACES_ENDPOINT).generic_string(); Poco::JSON::Object::Ptr request_body = new Poco::JSON::Object; - { - Poco::JSON::Array::Ptr namespaces = new Poco::JSON::Array; - namespaces->add(namespace_name); - request_body->set("namespace", namespaces); - } + request_body->set("namespace", namespaceToJSONArray(namespace_name)); + + if (!location.empty()) { Poco::JSON::Object::Ptr properties = new Poco::JSON::Object; properties->set("location", location); @@ -2000,18 +2021,27 @@ void RestCatalog::createNamespaceIfNotExists(const String & namespace_name, cons } } -void RestCatalog::createTable(const String & namespace_name, const String & table_name, const String & /*new_metadata_path*/, Poco::JSON::Object::Ptr metadata_content) const +bool RestCatalog::createTable( + const String & namespace_name, + const String & table_name, + const String & /*new_metadata_path*/, + Poco::JSON::Object::Ptr metadata_content, + DB::CompressionMethod /*metadata_compression_method*/, + bool if_not_exists) const { if (!allowed_namespaces.isNamespaceAllowed(namespace_name, /*nested*/ false)) throw DB::Exception(DB::ErrorCodes::CATALOG_NAMESPACE_DISABLED, "Failed to create table {}, namespace {} is filtered by `namespaces` database parameter", table_name, namespace_name); + const String location = metadata_content->getValue("location"); + const auto state_snapshot = state.get(); const std::string endpoint = (base_url / state_snapshot->config.prefix / NAMESPACES_ENDPOINT / encodeNamespaceForURI(namespace_name) / "tables").generic_string(); Poco::JSON::Object::Ptr request_body = new Poco::JSON::Object; request_body->set("name", table_name); - request_body->set("location", metadata_content->getValue("location")); + if (!managesTableLocation()) + request_body->set("location", location); { Poco::JSON::Object::Ptr initial_schema = metadata_content->getArray("schemas")->getObject(0); Poco::JSON::Array::Ptr identifier_fields = new Poco::JSON::Array; @@ -2020,13 +2050,21 @@ void RestCatalog::createTable(const String & namespace_name, const String & tabl } request_body->set("partition-spec", metadata_content->getArray("partition-specs")->get(0)); + if (metadata_content->has("sort-orders")) { - Poco::JSON::Object::Ptr write_order = new Poco::JSON::Object; - write_order->set("order-id", 0); - Poco::JSON::Array::Ptr fields = new Poco::JSON::Array; - write_order->set("fields", fields); - request_body->set("write-order", write_order); + if (auto sort_orders = metadata_content->getArray("sort-orders"); sort_orders->size() > 0) + { + auto sort_order = sort_orders->getObject(0); + auto fields = sort_order->getArray("fields"); + if (fields && fields->size() > 0) + { + if (sort_order->getValue("order-id") == 0) + sort_order->set("order-id", 1); + request_body->set("write-order", sort_order); + } + } } + request_body->set("stage-create", false); Poco::JSON::Object::Ptr properties = new Poco::JSON::Object; @@ -2043,8 +2081,12 @@ void RestCatalog::createTable(const String & namespace_name, const String & tabl } catch (const DB::HTTPException & ex) { + if (if_not_exists && ex.getHTTPStatus() == Poco::Net::HTTPResponse::HTTPStatus::HTTP_CONFLICT) + return false; throw DB::Exception(DB::ErrorCodes::DATALAKE_DATABASE_ERROR, "Failed to create table {}", ex.displayText()); } + + return true; } @@ -2121,7 +2163,7 @@ bool RestCatalog::updateSchema( return true; } -void RestCatalog::dropTable(const String & namespace_name, const String & table_name) const +void RestCatalog::dropTable(const String & namespace_name, const String & table_name, bool purge, bool if_exists) const { if (!allowed_namespaces.isNamespaceAllowed(namespace_name, /*nested*/ false)) throw DB::Exception(DB::ErrorCodes::CATALOG_NAMESPACE_DISABLED, @@ -2129,9 +2171,9 @@ void RestCatalog::dropTable(const String & namespace_name, const String & table_ table_name, namespace_name); const auto state_snapshot = state.get(); - const std::string endpoint - = (base_url / state_snapshot->config.prefix / NAMESPACES_ENDPOINT / encodeNamespaceForURI(namespace_name) / "tables" / table_name).generic_string() - + "?purgeRequested=False"; + const std::string base_endpoint = + (base_url / state_snapshot->config.prefix / NAMESPACES_ENDPOINT / encodeNamespaceForURI(namespace_name) / "tables" / table_name).generic_string(); + const std::string endpoint = fmt::format("{}?purgeRequested={}", base_endpoint, purge ? "true" : "false"); Poco::JSON::Object::Ptr request_body = nullptr; try @@ -2142,6 +2184,8 @@ void RestCatalog::dropTable(const String & namespace_name, const String & table_ } catch (const DB::HTTPException & ex) { + if (if_exists && ex.getHTTPStatus() == Poco::Net::HTTPResponse::HTTPStatus::HTTP_NOT_FOUND) + return; throw DB::Exception(DB::ErrorCodes::DATALAKE_DATABASE_ERROR, "Failed to drop table {}", ex.displayText()); } } diff --git a/src/Databases/DataLake/RestCatalog.h b/src/Databases/DataLake/RestCatalog.h index 0c54e43280c0..c042b255bfc3 100644 --- a/src/Databases/DataLake/RestCatalog.h +++ b/src/Databases/DataLake/RestCatalog.h @@ -83,12 +83,20 @@ class RestCatalog : public ICatalog, public DB::WithContext std::optional getStorageType() const override; + String getDefaultBaseLocation() const override; + DB::DatabaseDataLakeCatalogType getCatalogType() const override { return DB::DatabaseDataLakeCatalogType::ICEBERG_REST; } - void createTable(const String & namespace_name, const String & table_name, const String & new_metadata_path, Poco::JSON::Object::Ptr metadata_content) const override; + bool createTable( + const String & namespace_name, + const String & table_name, + const String & new_metadata_path, + Poco::JSON::Object::Ptr metadata_content, + DB::CompressionMethod metadata_compression_method, + bool if_not_exists) const override; bool updateMetadata(const String & namespace_name, const String & table_name, const String & new_metadata_path, Poco::JSON::Object::Ptr new_snapshot) const override; @@ -103,7 +111,7 @@ class RestCatalog : public ICatalog, public DB::WithContext bool isTransactional() const override { return true; } - void dropTable(const String & namespace_name, const String & table_name) const override; + void dropTable(const String & namespace_name, const String & table_name, bool purge, bool if_exists) const override; ICatalog::CredentialsRefreshCallback getCredentialsConfigurationCallback(const DB::StorageID & storage_id) override; @@ -317,6 +325,9 @@ class OneLakeCatalog : public RestCatalog return DB::DatabaseDataLakeCatalogType::ICEBERG_ONELAKE; } + /// OneLake keeps data in Azure Data Lake Storage, whatever the catalog reports. + std::optional getStorageType() const override { return StorageType::Azure; } + DB::HTTPHeaderEntries getAuthHeaders( const CatalogState & catalog_state, bool update_token, @@ -361,6 +372,9 @@ class BigLakeCatalog : public RestCatalog return DB::DatabaseDataLakeCatalogType::ICEBERG_BIGLAKE; } + /// BigLake keeps data in Google Cloud Storage, which is accessed through the S3 API. + std::optional getStorageType() const override { return StorageType::S3; } + DB::HTTPHeaderEntries getAuthHeaders( const CatalogState & catalog_state, bool update_token, diff --git a/src/Databases/DataLake/S3TablesCatalog.cpp b/src/Databases/DataLake/S3TablesCatalog.cpp index 961125be9819..a6b86bea1821 100644 --- a/src/Databases/DataLake/S3TablesCatalog.cpp +++ b/src/Databases/DataLake/S3TablesCatalog.cpp @@ -33,6 +33,7 @@ namespace DB::ErrorCodes { extern const int BAD_ARGUMENTS; extern const int DATALAKE_DATABASE_ERROR; + extern const int SUPPORT_IS_DISABLED; } namespace DB::Setting @@ -217,8 +218,21 @@ ICatalog::CredentialsRefreshCallback S3TablesCatalog::getCredentialsConfiguratio }; } -void S3TablesCatalog::dropTable(const String & namespace_name, const String & table_name) const +void S3TablesCatalog::dropTable(const String & namespace_name, const String & table_name, bool delete_data, bool if_exists) const { + /// https://docs.aws.amazon.com/AmazonS3/latest/userguide/s3-tables-delete.html + if (!delete_data) + { + if (if_exists && !existsTable(namespace_name, table_name)) + return; + + throw DB::Exception( + DB::ErrorCodes::SUPPORT_IS_DISABLED, + "S3 Tables cannot drop table {}.{} without deleting its data, and `data_lake_delete_data_on_drop` is disabled. " + "Enable `data_lake_delete_data_on_drop` to drop the table together with its data", + namespace_name, table_name); + } + const auto state_snapshot = state.get(); const std::string endpoint = (base_url / state_snapshot->config.prefix / "namespaces" / namespace_name / "tables" / table_name).string() @@ -233,10 +247,13 @@ void S3TablesCatalog::dropTable(const String & namespace_name, const String & ta } catch (const DB::HTTPException & ex) { - if (ex.getHTTPStatus() == Poco::Net::HTTPResponse::HTTP_NOT_FOUND) + /// `404` is returned by the API when the table does not exist - someone else dropped it first. + if (if_exists && ex.getHTTPStatus() == Poco::Net::HTTPResponse::HTTP_NOT_FOUND) + { LOG_DEBUG(log, "S3 Tables: table {}.{} already does not exist (404 on purge-delete)", namespace_name, table_name); - else - throw DB::Exception(DB::ErrorCodes::DATALAKE_DATABASE_ERROR, "Failed to drop table {}", ex.displayText()); + return; + } + throw DB::Exception(DB::ErrorCodes::DATALAKE_DATABASE_ERROR, "Failed to drop table {}", ex.displayText()); } } diff --git a/src/Databases/DataLake/S3TablesCatalog.h b/src/Databases/DataLake/S3TablesCatalog.h index a878bb17d924..1cff75ecc4ac 100644 --- a/src/Databases/DataLake/S3TablesCatalog.h +++ b/src/Databases/DataLake/S3TablesCatalog.h @@ -35,13 +35,17 @@ class S3TablesCatalog final : public RestCatalog DB::Names getTables() const override; + std::optional getStorageType() const override { return StorageType::S3; } + bool tryGetTableMetadata( const std::string & namespace_name, const std::string & table_name, DB::ContextPtr context_, TableMetadata & result) const override; - void dropTable(const String & namespace_name, const String & table_name) const override; + bool managesTableLocation() const override { return true; } + + void dropTable(const String & namespace_name, const String & table_name, bool delete_data, bool if_exists) const override; ICatalog::CredentialsRefreshCallback getCredentialsConfigurationCallback(const DB::StorageID & storage_id) override; diff --git a/src/Databases/DataLake/tests/gtest_construct_table_location.cpp b/src/Databases/DataLake/tests/gtest_construct_table_location.cpp new file mode 100644 index 000000000000..fa585d2a78a1 --- /dev/null +++ b/src/Databases/DataLake/tests/gtest_construct_table_location.cpp @@ -0,0 +1,43 @@ +#include +#include + +#include + +namespace DataLake::Test +{ + +TEST(ConstructTableLocation, StorageSchemes) +{ + EXPECT_EQ( + constructTableLocation("s3", "http://minio:9000/warehouse/data/", "ns", "tbl"), + "s3://warehouse/data/ns/tbl"); + EXPECT_EQ( + constructTableLocation("abfss", "https://account.dfs.core.windows.net/container/", "ns", "tbl"), + "abfss://container@account.dfs.core.windows.net/ns/tbl"); + EXPECT_EQ( + constructTableLocation("abfss", "https://account.dfs.core.windows.net/container/data", "ns", "tbl"), + "abfss://container@account.dfs.core.windows.net/data/ns/tbl"); + EXPECT_EQ( + constructTableLocation("abfss", "abfss://container@account.dfs.core.windows.net/data", "ns", "tbl"), + "abfss://container@account.dfs.core.windows.net/data/ns/tbl"); + EXPECT_EQ( + constructTableLocation("hdfs", "hdfs://namenode:9000/warehouse", "ns", "tbl"), + "hdfs://namenode:9000/warehouse/ns/tbl"); + EXPECT_EQ( + constructTableLocation("hdfs", "hdfs://namenode:9000", "ns", "tbl"), + "hdfs://namenode:9000/ns/tbl"); + EXPECT_EQ( + constructTableLocation("file", "file:///var/iceberg/warehouse", "ns", "tbl"), + "file:///var/iceberg/warehouse/ns/tbl"); +} + +TEST(ConstructTableLocation, InvalidEndpoints) +{ + EXPECT_THROW(constructTableLocation("s3", "http://minio:9000/", "ns", "tbl"), DB::Exception); + EXPECT_THROW( + constructTableLocation("s3", "https://bucket.s3.amazonaws.com", "ns", "tbl", DB::S3UriStyle::VIRTUAL_HOSTED), + DB::Exception); + EXPECT_THROW(constructTableLocation("abfss", "https://account.dfs.core.windows.net/", "ns", "tbl"), DB::Exception); +} + +} diff --git a/src/Databases/DatabaseAtomic.cpp b/src/Databases/DatabaseAtomic.cpp index 1e21692c2b03..9bf815f58bfa 100644 --- a/src/Databases/DatabaseAtomic.cpp +++ b/src/Databases/DatabaseAtomic.cpp @@ -189,7 +189,7 @@ StoragePtr DatabaseAtomic::detachTable(ContextPtr /* context */, const String & return detached_table; } -void DatabaseAtomic::dropTable(ContextPtr local_context, const String & table_name, bool sync) +void DatabaseAtomic::dropTable(ContextPtr local_context, const String & table_name, bool sync, bool /*if_exists*/) { auto component_guard = Coordination::setCurrentComponent("DatabaseAtomic::dropTable"); waitDatabaseStarted(); diff --git a/src/Databases/DatabaseAtomic.h b/src/Databases/DatabaseAtomic.h index f1626a66087f..a6d4374249d7 100644 --- a/src/Databases/DatabaseAtomic.h +++ b/src/Databases/DatabaseAtomic.h @@ -48,7 +48,7 @@ class DatabaseAtomic : public DatabaseOrdinary bool exchange, bool dictionary) override; - void dropTable(ContextPtr context, const String & table_name, bool sync) override; + void dropTable(ContextPtr context, const String & table_name, bool sync, bool if_exists) override; void dropTableImpl(ContextPtr context, const String & table_name, bool sync); void attachTable(ContextPtr context, const String & name, const StoragePtr & table, const String & relative_table_path) override; diff --git a/src/Databases/DatabaseBackup.cpp b/src/Databases/DatabaseBackup.cpp index a60fb82f166a..77585b6586d7 100644 --- a/src/Databases/DatabaseBackup.cpp +++ b/src/Databases/DatabaseBackup.cpp @@ -187,6 +187,7 @@ void DatabaseBackup::detachTablePermanently(ContextPtr, const String &) void DatabaseBackup::dropTable( ContextPtr, const String &, + bool, bool) { throw Exception(ErrorCodes::UNSUPPORTED_METHOD, "DROP TABLE is not supported for Backup database"); diff --git a/src/Databases/DatabaseBackup.h b/src/Databases/DatabaseBackup.h index 180de68f4037..5abf0aa306e7 100644 --- a/src/Databases/DatabaseBackup.h +++ b/src/Databases/DatabaseBackup.h @@ -45,7 +45,8 @@ class DatabaseBackup final : public DatabaseOrdinary void dropTable( ContextPtr context, const String & table_name, - bool sync) override; + bool sync, + bool if_exists) override; void renameTable( ContextPtr context, diff --git a/src/Databases/DatabaseMemory.cpp b/src/Databases/DatabaseMemory.cpp index 8dd9741c9aeb..bd10a02604a8 100644 --- a/src/Databases/DatabaseMemory.cpp +++ b/src/Databases/DatabaseMemory.cpp @@ -59,7 +59,8 @@ void DatabaseMemory::createTable( void DatabaseMemory::dropTable( ContextPtr /*context*/, const String & table_name, - bool /*sync*/) + bool /*sync*/, + bool /*if_exists*/) { StoragePtr table; { diff --git a/src/Databases/DatabaseMemory.h b/src/Databases/DatabaseMemory.h index 2e7d313c864b..219cf1bd09f3 100644 --- a/src/Databases/DatabaseMemory.h +++ b/src/Databases/DatabaseMemory.h @@ -32,7 +32,8 @@ class DatabaseMemory final : public DatabaseWithOwnTablesBase void dropTable( ContextPtr context, const String & table_name, - bool sync) override; + bool sync, + bool if_exists) override; ASTPtr getCreateTableQueryImpl(const String & name, ContextPtr context, bool throw_on_error) const override; diff --git a/src/Databases/DatabaseOnDisk.cpp b/src/Databases/DatabaseOnDisk.cpp index 89480f6bbece..731a4971a380 100644 --- a/src/Databases/DatabaseOnDisk.cpp +++ b/src/Databases/DatabaseOnDisk.cpp @@ -353,7 +353,7 @@ void DatabaseOnDisk::detachTablePermanently(ContextPtr query_context, const Stri } } -void DatabaseOnDisk::dropTable(ContextPtr local_context, const String & table_name, bool /*sync*/) +void DatabaseOnDisk::dropTable(ContextPtr local_context, const String & table_name, bool /*sync*/, bool /*if_exists*/) { auto component_guard = Coordination::setCurrentComponent("DatabaseOnDisk::dropTable"); waitDatabaseStarted(); diff --git a/src/Databases/DatabaseOnDisk.h b/src/Databases/DatabaseOnDisk.h index 4260c0d82730..bd4070a816a7 100644 --- a/src/Databases/DatabaseOnDisk.h +++ b/src/Databases/DatabaseOnDisk.h @@ -48,7 +48,8 @@ class DatabaseOnDisk : public DatabaseWithOwnTablesBase void dropTable( ContextPtr context, const String & table_name, - bool sync) override; + bool sync, + bool if_exists) override; void renameTable( ContextPtr context, diff --git a/src/Databases/DatabaseOverlay.cpp b/src/Databases/DatabaseOverlay.cpp index b8d861a079c1..cb515b5481cd 100644 --- a/src/Databases/DatabaseOverlay.cpp +++ b/src/Databases/DatabaseOverlay.cpp @@ -72,13 +72,13 @@ void DatabaseOverlay::createTable(ContextPtr context_, const String & table_name getEngineName()); } -void DatabaseOverlay::dropTable(ContextPtr context_, const String & table_name, bool sync) +void DatabaseOverlay::dropTable(ContextPtr context_, const String & table_name, bool sync, bool if_exists) { for (auto & db : databases) { if (db->isTableExist(table_name, context_)) { - db->dropTable(context_, table_name, sync); + db->dropTable(context_, table_name, sync, if_exists); return; } } diff --git a/src/Databases/DatabaseOverlay.h b/src/Databases/DatabaseOverlay.h index 72cb8d71b322..74a710ab1d14 100644 --- a/src/Databases/DatabaseOverlay.h +++ b/src/Databases/DatabaseOverlay.h @@ -29,7 +29,7 @@ class DatabaseOverlay : public IDatabase, protected WithContext void createTable(ContextPtr context, const String & table_name, const StoragePtr & table, const ASTPtr & query) override; - void dropTable(ContextPtr context, const String & table_name, bool sync) override; + void dropTable(ContextPtr context, const String & table_name, bool sync, bool if_exists) override; void attachTable(ContextPtr context, const String & table_name, const StoragePtr & table, const String & relative_table_path) override; diff --git a/src/Databases/DatabaseReplicated.cpp b/src/Databases/DatabaseReplicated.cpp index 27af8b4126be..94d4f393e9fa 100644 --- a/src/Databases/DatabaseReplicated.cpp +++ b/src/Databases/DatabaseReplicated.cpp @@ -2198,7 +2198,7 @@ void DatabaseReplicated::shutdown() DatabaseAtomic::shutdown(); } -void DatabaseReplicated::dropTable(ContextPtr local_context, const String & table_name, bool sync) +void DatabaseReplicated::dropTable(ContextPtr local_context, const String & table_name, bool sync, bool /*if_exists*/) { auto component_guard = Coordination::setCurrentComponent("DatabaseReplicated::dropTable"); waitDatabaseStarted(); diff --git a/src/Databases/DatabaseReplicated.h b/src/Databases/DatabaseReplicated.h index a2934f9a6586..7a5b30b4ed9a 100644 --- a/src/Databases/DatabaseReplicated.h +++ b/src/Databases/DatabaseReplicated.h @@ -79,7 +79,7 @@ class DatabaseReplicated : public DatabaseAtomic String getEngineName() const override { return "Replicated"; } /// If current query is initial, then the following methods add metadata updating ZooKeeper operations to current ZooKeeperMetadataTransaction. - void dropTable(ContextPtr, const String & table_name, bool sync) override; + void dropTable(ContextPtr, const String & table_name, bool sync, bool if_exists) override; void renameTable(ContextPtr context, const String & table_name, IDatabase & to_database, const String & to_table_name, bool exchange, bool dictionary) override; void detachTablePermanently(ContextPtr context, const String & table_name) override; diff --git a/src/Databases/IDatabase.cpp b/src/Databases/IDatabase.cpp index 2e62449b6176..7ae3437df500 100644 --- a/src/Databases/IDatabase.cpp +++ b/src/Databases/IDatabase.cpp @@ -144,7 +144,8 @@ void IDatabase::createTable( void IDatabase::dropTable( /// NOLINT ContextPtr /*context*/, const String & /*name*/, - [[maybe_unused]] bool sync) + [[maybe_unused]] bool sync, + [[maybe_unused]] bool if_exists) { throw Exception(ErrorCodes::NOT_IMPLEMENTED, "There is no DROP TABLE query for Database{}", getEngineName()); } diff --git a/src/Databases/IDatabase.h b/src/Databases/IDatabase.h index b92e15ac35d3..bb4903c9f895 100644 --- a/src/Databases/IDatabase.h +++ b/src/Databases/IDatabase.h @@ -30,6 +30,7 @@ struct IndicesDescription; struct StorageInMemoryMetadata; struct StorageID; class ASTCreateQuery; +class ASTStorage; struct AlterCommand; class AlterCommands; class SettingsChanges; @@ -189,6 +190,8 @@ class IDatabase : public std::enable_shared_from_this virtual bool isDatalakeCatalog() const { return false; } + virtual void validateCreateTableEngine(const ASTStorage & /*storage*/) const {} + /// True for databases such as `MySQL`/`PostgreSQL` whose table list lives on a remote service. /// This is distinct from `isExternal`, which classifies whether the engine supports ClickHouse internal table types. virtual bool isRemoteDatabase() const { return false; } @@ -323,7 +326,8 @@ class IDatabase : public std::enable_shared_from_this virtual void dropTable( /// NOLINT ContextPtr /*context*/, const String & /*name*/, - [[maybe_unused]] bool sync = false); + [[maybe_unused]] bool sync = false, + [[maybe_unused]] bool if_exists = false); /// Add a table to the database, but do not add it to the metadata. The database may not support this method. /// diff --git a/src/Databases/MySQL/DatabaseMySQL.cpp b/src/Databases/MySQL/DatabaseMySQL.cpp index a9494fa2abd3..67ff8eb04f15 100644 --- a/src/Databases/MySQL/DatabaseMySQL.cpp +++ b/src/Databases/MySQL/DatabaseMySQL.cpp @@ -533,7 +533,7 @@ void DatabaseMySQL::detachTablePermanently(ContextPtr, const String & table_name table_iter->second.second->is_detached = true; } -void DatabaseMySQL::dropTable(ContextPtr local_context, const String & table_name, bool /*sync*/) +void DatabaseMySQL::dropTable(ContextPtr local_context, const String & table_name, bool /*sync*/, bool /*if_exists*/) { if (!persistent) throw Exception(ErrorCodes::NOT_IMPLEMENTED, "DROP TABLE is not supported for non-persistent MySQL database"); diff --git a/src/Databases/MySQL/DatabaseMySQL.h b/src/Databases/MySQL/DatabaseMySQL.h index 5ac2e10474d4..7fe80687b58a 100644 --- a/src/Databases/MySQL/DatabaseMySQL.h +++ b/src/Databases/MySQL/DatabaseMySQL.h @@ -79,7 +79,7 @@ class DatabaseMySQL final : public DatabaseWithAltersOnDiskBase, WithContext void detachTablePermanently(ContextPtr context, const String & table_name) override; - void dropTable(ContextPtr context, const String & table_name, bool sync) override; + void dropTable(ContextPtr context, const String & table_name, bool sync, bool if_exists) override; void attachTable(ContextPtr context, const String & table_name, const StoragePtr & storage, const String & relative_table_path) override; diff --git a/src/Databases/PostgreSQL/DatabaseMaterializedPostgreSQL.cpp b/src/Databases/PostgreSQL/DatabaseMaterializedPostgreSQL.cpp index b17234d69570..89061ead199a 100644 --- a/src/Databases/PostgreSQL/DatabaseMaterializedPostgreSQL.cpp +++ b/src/Databases/PostgreSQL/DatabaseMaterializedPostgreSQL.cpp @@ -391,7 +391,7 @@ void DatabaseMaterializedPostgreSQL::attachTable(ContextPtr context_, const Stri catch (...) { /// This is a failed attach table. Remove already created nested table. - DatabaseAtomic::dropTable(current_context, table_name, true); + DatabaseAtomic::dropTable(current_context, table_name, true, /* if_exists */ false); throw; } } @@ -439,7 +439,7 @@ void DatabaseMaterializedPostgreSQL::detachTablePermanently(ContextPtr, const St { auto current_context = Context::createCopy(getContext()->getGlobalContext()); current_context->makeQueryContext(); - DatabaseAtomic::dropTable(current_context, table_name, true); + DatabaseAtomic::dropTable(current_context, table_name, true, /* if_exists */ false); } catch (Exception & e) { @@ -480,10 +480,10 @@ void DatabaseMaterializedPostgreSQL::stopReplication() } -void DatabaseMaterializedPostgreSQL::dropTable(ContextPtr local_context, const String & table_name, bool sync) +void DatabaseMaterializedPostgreSQL::dropTable(ContextPtr local_context, const String & table_name, bool sync, bool if_exists) { /// Modify context into nested_context and pass query to Atomic database. - DatabaseAtomic::dropTable(StorageMaterializedPostgreSQL::makeNestedTableContext(local_context), table_name, sync); + DatabaseAtomic::dropTable(StorageMaterializedPostgreSQL::makeNestedTableContext(local_context), table_name, sync, if_exists); } diff --git a/src/Databases/PostgreSQL/DatabaseMaterializedPostgreSQL.h b/src/Databases/PostgreSQL/DatabaseMaterializedPostgreSQL.h index d1963b2ec8a6..239914dc584c 100644 --- a/src/Databases/PostgreSQL/DatabaseMaterializedPostgreSQL.h +++ b/src/Databases/PostgreSQL/DatabaseMaterializedPostgreSQL.h @@ -59,7 +59,7 @@ class DatabaseMaterializedPostgreSQL : public DatabaseAtomic StoragePtr detachTable(ContextPtr context, const String & table_name) override; - void dropTable(ContextPtr local_context, const String & name, bool sync) override; + void dropTable(ContextPtr local_context, const String & name, bool sync, bool if_exists) override; void drop(ContextPtr local_context) override; diff --git a/src/Databases/PostgreSQL/DatabasePostgreSQL.cpp b/src/Databases/PostgreSQL/DatabasePostgreSQL.cpp index e30586e34f5e..6d85ee6f7475 100644 --- a/src/Databases/PostgreSQL/DatabasePostgreSQL.cpp +++ b/src/Databases/PostgreSQL/DatabasePostgreSQL.cpp @@ -296,7 +296,7 @@ void DatabasePostgreSQL::createTable(ContextPtr local_context, const String & ta } -void DatabasePostgreSQL::dropTable(ContextPtr, const String & table_name, bool /* sync */) +void DatabasePostgreSQL::dropTable(ContextPtr, const String & table_name, bool /* sync */, bool /* if_exists */) { if (!persistent) throw Exception(ErrorCodes::NOT_IMPLEMENTED, "DROP TABLE is not supported for non-persistent MySQL database"); diff --git a/src/Databases/PostgreSQL/DatabasePostgreSQL.h b/src/Databases/PostgreSQL/DatabasePostgreSQL.h index 59d49e37940a..fb218017278f 100644 --- a/src/Databases/PostgreSQL/DatabasePostgreSQL.h +++ b/src/Databases/PostgreSQL/DatabasePostgreSQL.h @@ -54,7 +54,7 @@ class DatabasePostgreSQL final : public DatabaseWithAltersOnDiskBase, WithContex StoragePtr tryGetTable(const String & name, ContextPtr context) const override; void createTable(ContextPtr, const String & table_name, const StoragePtr & storage, const ASTPtr & create_query) override; - void dropTable(ContextPtr, const String & table_name, bool sync) override; + void dropTable(ContextPtr, const String & table_name, bool sync, bool if_exists) override; void attachTable(ContextPtr context, const String & table_name, const StoragePtr & storage, const String & relative_table_path) override; StoragePtr detachTable(ContextPtr context, const String & table_name) override; diff --git a/src/Interpreters/DDLWorker.cpp b/src/Interpreters/DDLWorker.cpp index 81355067db4a..f5ca0071d204 100644 --- a/src/Interpreters/DDLWorker.cpp +++ b/src/Interpreters/DDLWorker.cpp @@ -14,6 +14,7 @@ #include #include #include +#include #include #include #include @@ -572,6 +573,8 @@ bool DDLWorker::tryExecuteQuery(DDLTaskBase & task, const ZooKeeperPtr & zookeep try { + checkQueryDatabasesSupportOnClusterDDL(task.query, context); + auto query_context = task.makeQueryContext(context, zookeeper); chassert(!query_context->getCurrentTransaction()); diff --git a/src/Interpreters/InterpreterAlterQuery.cpp b/src/Interpreters/InterpreterAlterQuery.cpp index 8a26f772239d..0898a7285e19 100644 --- a/src/Interpreters/InterpreterAlterQuery.cpp +++ b/src/Interpreters/InterpreterAlterQuery.cpp @@ -422,6 +422,9 @@ BlockIO InterpreterAlterQuery::executeToTable(const ASTAlterQuery & alter) if (table && table->as()) throw Exception(ErrorCodes::BAD_ARGUMENTS, "Mutations with ON CLUSTER are not allowed for KeeperMap tables"); + if (table_id) + checkDatabaseSupportsOnClusterDDL(DatabaseCatalog::instance().tryGetDatabase(table_id.database_name)); + DDLQueryOnClusterParams params; params.access_to_check = getRequiredAccess(); return executeDDLQueryOnCluster(query_ptr, getContext(), params); diff --git a/src/Interpreters/InterpreterCreateQuery.cpp b/src/Interpreters/InterpreterCreateQuery.cpp index dcd3f3ea2674..35d721bf93b2 100644 --- a/src/Interpreters/InterpreterCreateQuery.cpp +++ b/src/Interpreters/InterpreterCreateQuery.cpp @@ -36,6 +36,7 @@ #include #include #include +#include #include #include #include @@ -50,6 +51,8 @@ #include #include #include +#include +#include #include #include #include @@ -135,6 +138,7 @@ namespace Setting extern const SettingsUInt64 database_replicated_allow_explicit_uuid; extern const SettingsBool database_replicated_allow_heavy_create; extern const SettingsBool database_replicated_allow_only_replicated_engine; + extern const SettingsBool datalake_ignore_unsupported_table_properties; extern const SettingsBool data_type_default_nullable; extern const SettingsSQLSecurityType default_materialized_view_sql_security; extern const SettingsSQLSecurityType default_normal_view_sql_security; @@ -818,9 +822,10 @@ InterpreterCreateQuery::TableProperties InterpreterCreateQuery::getTableProperti if (!create.comment && !as_storage_metadata->comment.empty()) create.set(create.comment, make_intrusive(as_storage_metadata->comment)); - /// Secondary indices and projections make sense only for MergeTree family of storage engines. - /// We should not copy them for other storages. - if (create.storage && endsWith(create.storage->engine->name, "MergeTree")) + /// Retain source properties for `DataLakeCatalog` until its validation either rejects or explicitly omits them. + const auto target_database = DatabaseCatalog::instance().tryGetDatabase(getContext()->resolveDatabase(create.getDatabase())); + const bool is_datalake_catalog = target_database && target_database->isDatalakeCatalog(); + if (is_datalake_catalog || (create.storage && create.storage->engine && endsWith(create.storage->engine->name, "MergeTree"))) { /// Copy secondary indexes but only the ones which were not implicitly created. These will be re-generated later again and need /// not be copied. @@ -834,7 +839,7 @@ InterpreterCreateQuery::TableProperties InterpreterCreateQuery::getTableProperti /// CREATE TABLE AS should copy PRIMARY KEY, ORDER BY, and similar clauses. /// Note: only supports the source table engine is using the new syntax. - if (const auto * merge_tree_data = dynamic_cast(as_storage.get())) + if (const auto * merge_tree_data = dynamic_cast(as_storage.get()); merge_tree_data && !is_datalake_catalog) { if (merge_tree_data->format_version >= MERGE_TREE_DATA_MIN_FORMAT_VERSION_WITH_CUSTOM_PARTITIONING) { @@ -1413,11 +1418,24 @@ void InterpreterCreateQuery::setEngine(ASTCreateQuery & create) const create.set(create.as_table_function, as_create.as_table_function->ptr()); return; } - else if (as_create.storage) + else if (as_create.storage && as_create.storage->engine) { storage_def = boost::static_pointer_cast(as_create.storage->ptr()); create.is_time_series_table = as_create.is_time_series_table; } + else if (as_create.storage) + { + /// A `DataLakeCatalog` that assigns table locations itself shows its tables without `ENGINE`. + /// Only a `DataLakeCatalog` target can be created from such a table: it builds its own storage + /// and copies the keys from the source metadata. + const auto target_database = DatabaseCatalog::instance().tryGetDatabase(getContext()->resolveDatabase(create.getDatabase())); + if (!target_database || !target_database->isDatalakeCatalog()) + throw Exception( + ErrorCodes::INCORRECT_QUERY, + "Cannot CREATE a table AS {}, it has no table engine. Specify ENGINE explicitly", + qualified_name); + return; + } else { throw Exception(ErrorCodes::LOGICAL_ERROR, "Cannot set engine, it's a bug."); @@ -1549,6 +1567,21 @@ bool isReplicated(const ASTStorage & storage) return storage_name.starts_with("Replicated") || storage_name.starts_with("Shared"); } +const char * findUnsupportedDatalakeStorageClause(const ASTStorage & storage, bool allow_settings) +{ + if (storage.primary_key) + return "PRIMARY KEY"; + if (storage.sample_by) + return "SAMPLE BY"; + if (storage.ttl_table) + return "TTL"; + if (storage.unique_key) + return "UNIQUE KEY"; + if (!allow_settings && storage.settings) + return "engine SETTINGS"; + return nullptr; +} + } BlockIO InterpreterCreateQuery::createTable(ASTCreateQuery & create) @@ -1766,6 +1799,12 @@ BlockIO InterpreterCreateQuery::createTable(ASTCreateQuery & create) if (!UserDefinedSQLFunctionFactory::instance().empty()) UserDefinedSQLFunctionVisitor::visit(query_ptr, getContext()); + const bool engine_user_specified = create.storage && create.storage->engine; + + const char * datalake_unsupported_storage_clause = nullptr; + if (create.storage) + datalake_unsupported_storage_clause = findUnsupportedDatalakeStorageClause(*create.storage, engine_user_specified); + /// Set and retrieve list of columns, indices and constraints. Set table engine if needed. Rewrite query in canonical way. TableProperties properties = getTablePropertiesAndNormalizeCreateQuery(create, mode); @@ -1838,6 +1877,7 @@ BlockIO InterpreterCreateQuery::createTable(ASTCreateQuery & create) if (!create.cluster.empty()) { + checkDatabaseSupportsOnClusterDDL(database); chassert(!ddl_guard); return executeQueryOnCluster(create); } @@ -1845,6 +1885,137 @@ BlockIO InterpreterCreateQuery::createTable(ASTCreateQuery & create) if (need_add_to_database && !database) throw Exception(ErrorCodes::UNKNOWN_DATABASE, "Database {} does not exist", backQuoteIfNeed(database_name)); + if (database && database->isDatalakeCatalog()) + { + const bool ignore_unsupported_properties + = getContext()->getSettingsRef()[Setting::datalake_ignore_unsupported_table_properties]; + + if (create.is_ordinary_view || create.is_materialized_view + || create.is_dictionary || create.attach || create.is_clone_as + || create.replace_table || create.replace_view) + throw Exception(ErrorCodes::NOT_IMPLEMENTED, + "DataLakeCatalog supports only plain CREATE TABLE; " + "views, dictionaries, ATTACH, CLONE AS, and REPLACE TABLE are not allowed"); + + if (engine_user_specified) + database->validateCreateTableEngine(*create.storage); + + if (datalake_unsupported_storage_clause && !ignore_unsupported_properties) + { + if (engine_user_specified) + throw Exception(ErrorCodes::BAD_ARGUMENTS, + "DataLakeCatalog CREATE TABLE with an explicit table engine supports only " + "PARTITION BY, ORDER BY, and engine SETTINGS; " + "PRIMARY KEY, SAMPLE BY, TTL, and UNIQUE KEY are not supported " + "(got {})", datalake_unsupported_storage_clause); + + throw Exception(ErrorCodes::BAD_ARGUMENTS, + "DataLakeCatalog CREATE TABLE supports only PARTITION BY and ORDER BY; " + "PRIMARY KEY, SAMPLE BY, TTL, UNIQUE KEY, and engine SETTINGS are not supported " + "(got {})", datalake_unsupported_storage_clause); + } + + if (!ignore_unsupported_properties) + { + for (const auto & column : properties.columns) + { + if (column.default_desc.expression || column.default_desc.kind != ColumnDefaultKind::Default) + throw Exception(ErrorCodes::BAD_ARGUMENTS, + "Column '{}': {} is not yet supported by DataLakeCatalog table creation", + column.name, toString(column.default_desc.kind)); + + if (!column.comment.empty() || column.codec || column.ttl + || !column.settings.empty() || column.statistics.hasExplicitStatistics()) + throw Exception(ErrorCodes::BAD_ARGUMENTS, + "Column '{}': COMMENT, CODEC, TTL, STATISTICS, SETTINGS, and PRIMARY KEY " + "are not supported by DataLakeCatalog table creation", + column.name); + } + + if (!properties.indices.empty() || !properties.constraints.empty() || !properties.projections.empty() + || (create.columns_list && (create.columns_list->primary_key || create.columns_list->primary_key_from_columns))) + throw Exception(ErrorCodes::BAD_ARGUMENTS, + "DataLakeCatalog CREATE TABLE does not support PRIMARY KEY, indices, constraints, or projections"); + } + + if (create.comment && !ignore_unsupported_properties) + throw Exception(ErrorCodes::BAD_ARGUMENTS, + "Table COMMENT is not supported by DataLakeCatalog table creation " + "(note: CREATE TABLE ... AS inherits the comment from the source table)"); + + if (!ignore_unsupported_properties && !as_table_saved.empty()) + { + const ASTStorage * source_storage = create.storage; + ASTPtr source_create_ptr; + if (engine_user_specified) + { + const String as_database_name = getContext()->resolveDatabase(as_database_saved); + source_create_ptr = DatabaseCatalog::instance().getDatabase(as_database_name)->getCreateTableQuery(as_table_saved, getContext()); + const auto & source_create = source_create_ptr->as(); + source_storage = source_create.is_materialized_view + ? source_create.getTargetInnerEngine(ViewTarget::To) + : source_create.storage; + } + + if (source_storage) + { + if (const char * inherited_clause = findUnsupportedDatalakeStorageClause(*source_storage, /*allow_settings=*/ true)) + throw Exception(ErrorCodes::BAD_ARGUMENTS, + "Source table {}.{} has {}, which a DataLakeCatalog table cannot represent; " + "CREATE TABLE ... AS supports only columns, PARTITION BY, and ORDER BY", + backQuoteIfNeed(getContext()->resolveDatabase(as_database_saved)), + backQuoteIfNeed(as_table_saved), inherited_clause); + } + } + + if (ignore_unsupported_properties) + { + /// Validate expressions before omitting properties that the destination cannot store. + KeyDescription primary_key; + if (create.storage) + { + auto * key = create.storage->primary_key ? create.storage->primary_key : create.storage->order_by; + if (key) + primary_key = KeyDescription::getKeyFromAST(key->ptr(), properties.columns, {}, getContext()); + if (create.storage->sample_by) + KeyDescription::getKeyFromAST(create.storage->sample_by->ptr(), properties.columns, {}, getContext()); + if (create.storage->unique_key) + KeyDescription::getKeyFromAST(create.storage->unique_key->ptr(), properties.columns, {}, getContext()); + if (create.storage->ttl_table) + TTLTableDescription::getTTLForTableFromAST( + create.storage->ttl_table->ptr(), properties.columns, getContext(), primary_key, /*is_attach=*/ false); + } + for (const auto & column : properties.columns) + if (column.ttl) + TTLDescription::getTTLFromAST(column.ttl, properties.columns, getContext(), primary_key, /*is_attach=*/ false); + properties.constraints.getExpressions(getContext(), properties.columns.getAllPhysical()); + + ColumnsDescription plain_columns; + for (const auto & column : properties.columns) + plain_columns.add(ColumnDescription(column.name, column.type)); + + properties.columns = std::move(plain_columns); + properties.indices = {}; + properties.constraints = {}; + properties.projections = {}; + + auto columns_list = make_intrusive(); + columns_list->set(columns_list->columns, formatColumns(properties.columns)); + create.set(create.columns_list, columns_list); + create.reset(create.comment); + + if (create.storage) + { + create.storage->reset(create.storage->primary_key); + create.storage->reset(create.storage->sample_by); + create.storage->reset(create.storage->ttl_table); + create.storage->reset(create.storage->unique_key); + if (!engine_user_specified) + create.storage->reset(create.storage->settings); + } + } + } + if (create.isTemporary() && create.replace_table) { chassert(!ddl_guard); @@ -1859,7 +2030,24 @@ BlockIO InterpreterCreateQuery::createTable(ASTCreateQuery & create) } /// Actually creates table - bool created = doCreateTable(create, properties, ddl_guard, mode); + bool created = false; + try + { + created = doCreateTable(create, properties, ddl_guard, mode, engine_user_specified); + } + catch (const Exception & e) + { + if (!create.if_not_exists || e.code() != ErrorCodes::TABLE_ALREADY_EXISTS + || !database || !database->isDatalakeCatalog()) + throw; + LOG_INFO( + getLogger("InterpreterCreateQuery"), + "CREATE TABLE IF NOT EXISTS {}.{} created nothing: {}", + backQuoteIfNeed(create.getDatabase()), + backQuoteIfNeed(create.getTable()), + e.message()); + created = false; + } ddl_guard.reset(); if (!created) /// Table already exists @@ -1937,7 +2125,7 @@ catch (...) bool InterpreterCreateQuery::doCreateTable(ASTCreateQuery & create, const InterpreterCreateQuery::TableProperties & properties, - DDLGuardPtr & ddl_guard, LoadingStrictnessLevel mode) + DDLGuardPtr & ddl_guard, LoadingStrictnessLevel mode, bool engine_user_specified) { if (create.isTemporary()) { @@ -2034,6 +2222,85 @@ bool InterpreterCreateQuery::doCreateTable(ASTCreateQuery & create, database->checkTableNameLength(create.getTable()); } + auto & create_query = query_ptr->as(); + if (database->isDatalakeCatalog() && !as_table_saved.empty()) + { + String as_database_name = getContext()->resolveDatabase(as_database_saved); + auto source_database = DatabaseCatalog::instance().getDatabase(as_database_name); + ASTPtr source_partition_by; + ASTPtr source_order_by; + if (source_database->isDatalakeCatalog()) + { + /// Use the same validated definition as `SHOW CREATE TABLE`, including any explicitly permitted omissions. + auto source_query = source_database->getCreateTableQuery(as_table_saved, getContext()); + const auto & source_create = source_query->as(); + if (source_create.storage) + { + source_partition_by = source_create.storage->partition_by; + source_order_by = source_create.storage->order_by; + } + } + else + { + StoragePtr as_storage = DatabaseCatalog::instance().getTable({as_database_name, as_table_saved}, getContext()); + /// A materialized view keeps its keys in its target table. + if (const auto * materialized_view = as_storage->as()) + as_storage = materialized_view->getTargetTable(); + auto as_storage_metadata = as_storage->getInMemoryMetadataPtr(getContext(), false); + if (as_storage_metadata->isPartitionKeyDefined() && as_storage_metadata->hasPartitionKey()) + source_partition_by = as_storage_metadata->getPartitionKeyAST(); + if (as_storage_metadata->isSortingKeyDefined() && as_storage_metadata->hasSortingKey()) + source_order_by = as_storage_metadata->getSortingKeyAST(); + } + + if (engine_user_specified) + { + if (!create_query.storage->partition_by && source_partition_by) + create_query.storage->set(create_query.storage->partition_by, source_partition_by->clone()); + if (!create_query.storage->order_by && source_order_by) + create_query.storage->set(create_query.storage->order_by, source_order_by->clone()); + } + else + { + ASTPtr partition_by; + ASTPtr order_by; + if (create_query.storage) + { + if (create_query.storage->partition_by) + partition_by = create_query.storage->partition_by->clone(); + if (create_query.storage->order_by) + order_by = create_query.storage->order_by->clone(); + } + + if (!partition_by && source_partition_by) + partition_by = source_partition_by->clone(); + if (!order_by && source_order_by) + order_by = source_order_by->clone(); + + auto storage_ast = make_intrusive(); + create_query.set(create_query.storage, storage_ast); + if (partition_by) + create_query.storage->set(create_query.storage->partition_by, partition_by); + if (order_by) + create_query.storage->set(create_query.storage->order_by, order_by); + } + } + + if (database->isDatalakeCatalog() && !engine_user_specified) + { + if (!create_query.columns_list + || !create_query.columns_list->columns + || create_query.columns_list->columns->children.empty()) + { + auto columns_declare_list = make_intrusive(); + columns_declare_list->set(columns_declare_list->columns, formatColumns(properties.columns)); + create_query.set(create_query.columns_list, columns_declare_list); + } + + database->createTable(getContext(), create.getTable(), nullptr, query_ptr); + return true; + } + data_path = database->getTableDataPath(create); // When creating a table, when checking if the data path exists, it should use the local disk to check, not the database disk. Because the database disk stores metadata files only. auto full_data_path = fs::path{getContext()->getPath()} / data_path; @@ -2146,7 +2413,6 @@ bool InterpreterCreateQuery::doCreateTable(ASTCreateQuery & create, is_restore_from_backup); /// If schema was inferred while storage creation, add columns description to create query. - auto & create_query = query_ptr->as(); addColumnsDescriptionToCreateQueryIfNecessary(create_query, res); /// Add any inferred engine args if needed. For example, data format for engines File/S3/URL/etc if (auto * engine_args = getEngineArgsFromCreateQuery(create_query)) @@ -2269,6 +2535,8 @@ BlockIO InterpreterCreateQuery::doCreateOrReplaceTable(ASTCreateQuery & create, : (create.isView() ? "CREATE OR REPLACE VIEW" : "CREATE OR REPLACE TABLE")) : "REPLACE TABLE"); + chassert(!database->isDatalakeCatalog()); + if (mode <= LoadingStrictnessLevel::CREATE) database->checkTableNameLength(table_to_replace_name); } @@ -2327,7 +2595,7 @@ BlockIO InterpreterCreateQuery::doCreateOrReplaceTable(ASTCreateQuery & create, { /// Create temporary table (random name will be generated) DDLGuardPtr ddl_guard; - [[maybe_unused]] bool done = InterpreterCreateQuery(query_ptr, create_context).doCreateTable(create, properties, ddl_guard, mode); + [[maybe_unused]] bool done = InterpreterCreateQuery(query_ptr, create_context).doCreateTable(create, properties, ddl_guard, mode, /*engine_user_specified=*/false); ddl_guard.reset(); chassert(done); created = true; diff --git a/src/Interpreters/InterpreterCreateQuery.h b/src/Interpreters/InterpreterCreateQuery.h index cb50a1cd9aeb..bed1e44de5b1 100644 --- a/src/Interpreters/InterpreterCreateQuery.h +++ b/src/Interpreters/InterpreterCreateQuery.h @@ -108,7 +108,7 @@ class InterpreterCreateQuery : public IInterpreter, WithMutableContext AccessRightsElements getRequiredAccess() const; /// Create IStorage and add it to database. If table already exists and IF NOT EXISTS specified, do nothing and return false. - bool doCreateTable(ASTCreateQuery & create, const TableProperties & properties, DDLGuardPtr & ddl_guard, LoadingStrictnessLevel mode); + bool doCreateTable(ASTCreateQuery & create, const TableProperties & properties, DDLGuardPtr & ddl_guard, LoadingStrictnessLevel mode, bool engine_user_specified); BlockIO doCreateOrReplaceTable(ASTCreateQuery & create, const InterpreterCreateQuery::TableProperties & properties, LoadingStrictnessLevel mode); BlockIO doCreateOrReplaceTemporaryTable(ASTCreateQuery & create, const InterpreterCreateQuery::TableProperties & properties, LoadingStrictnessLevel mode); #if CLICKHOUSE_CLOUD diff --git a/src/Interpreters/InterpreterDropQuery.cpp b/src/Interpreters/InterpreterDropQuery.cpp index e52f836b04d2..ef1ddbed884a 100644 --- a/src/Interpreters/InterpreterDropQuery.cpp +++ b/src/Interpreters/InterpreterDropQuery.cpp @@ -98,6 +98,9 @@ BlockIO InterpreterDropQuery::execute() BlockIO InterpreterDropQuery::executeSingleDropQuery(const ASTPtr & drop_query_ptr) { auto & drop = drop_query_ptr->as(); + if (!drop.cluster.empty() && (drop.table || drop.database)) + checkDatabaseSupportsOnClusterDDL( + DatabaseCatalog::instance().tryGetDatabase(getContext()->resolveDatabase(drop.getDatabase()))); if (!drop.cluster.empty() && drop.table && !drop.if_empty && !maybeRemoveOnCluster(current_query_ptr, getContext())) { DDLQueryOnClusterParams params; @@ -348,6 +351,8 @@ BlockIO InterpreterDropQuery::executeToTableImpl(const ContextPtr & context_, AS bool check_loading_deps = !check_ref_deps && getContext()->getSettingsRef()[Setting::check_table_dependencies]; DatabaseCatalog::instance().checkTableCanBeRemovedOrRenamed(table_id, check_ref_deps, check_loading_deps, is_drop_or_detach_database); + table->prepareForDrop(context_); + table->flushAndShutdown(true); TableExclusiveLockHolder table_lock; @@ -356,7 +361,7 @@ BlockIO InterpreterDropQuery::executeToTableImpl(const ContextPtr & context_, AS DatabaseCatalog::instance().removeDependencies(table_id, check_ref_deps, check_loading_deps, is_drop_or_detach_database); NamedCollectionFactory::instance().removeDependencies(table_id); - database->dropTable(context_, table_id.table_name, query.sync); + database->dropTable(context_, table_id.table_name, query.sync, query.if_exists); /// We have to clear mmapio cache when dropping table from Ordinary database /// to avoid reading old data if new table with the same name is created diff --git a/src/Interpreters/executeDDLQueryOnCluster.cpp b/src/Interpreters/executeDDLQueryOnCluster.cpp index a1f76f89cfdc..1817371d4a1f 100644 --- a/src/Interpreters/executeDDLQueryOnCluster.cpp +++ b/src/Interpreters/executeDDLQueryOnCluster.cpp @@ -6,6 +6,7 @@ #include #include #include +#include #include #include #include @@ -16,6 +17,7 @@ #include #include #include +#include #include #include #include @@ -49,6 +51,35 @@ extern const int LOGICAL_ERROR; } +void checkQueryDatabasesSupportOnClusterDDL(const ASTPtr & query_ptr, ContextPtr context) +{ + /// `RENAME` / `EXCHANGE TABLE` is `ASTRenameQuery`, which does not derive from `ASTQueryWithTableAndOutput`; + /// queries with no target table (`CREATE DATABASE`, `SYSTEM`, ...) contribute none. + std::vector target_databases; + const auto add_target_database = [&](String name) + { + target_databases.push_back(name.empty() ? context->getCurrentDatabase() : std::move(name)); + }; + + if (const auto * with_table = dynamic_cast(query_ptr.get()); + with_table && with_table->table) + { + add_target_database(with_table->getDatabase()); + } + else if (const auto * rename = dynamic_cast(query_ptr.get()); rename && !rename->database) + { + for (const auto & elem : rename->getElements()) + { + add_target_database(elem.from.getDatabase()); + add_target_database(elem.to.getDatabase()); + } + } + + for (const auto & database_name : target_databases) + checkDatabaseSupportsOnClusterDDL(DatabaseCatalog::instance().tryGetDatabase(database_name)); +} + + bool isSupportedAlterTypeForOnClusterDDLQuery(int type) { chassert(type != ASTAlterCommand::NO_TYPE); @@ -85,6 +116,9 @@ BlockIO executeDDLQueryOnCluster(const ASTPtr & query_ptr_, ContextPtr context, throw Exception(ErrorCodes::NOT_IMPLEMENTED, "Distributed execution is not supported for such DDL queries"); } + /// Initiator-side guard; workers re-check in `DDLWorker::tryExecuteQuery`. + checkQueryDatabasesSupportOnClusterDDL(query_ptr, context); + if (!context->getSettingsRef()[Setting::allow_distributed_ddl]) throw Exception(ErrorCodes::QUERY_IS_PROHIBITED, "Distributed DDL queries are prohibited for the user"); @@ -224,6 +258,14 @@ BlockIO getDDLOnClusterStatus(const String & node_path, const String & replicas_ return io; } +void checkDatabaseSupportsOnClusterDDL(const DatabasePtr & database) +{ + if (database && database->isDatalakeCatalog()) + throw Exception(ErrorCodes::NOT_IMPLEMENTED, + "ON CLUSTER is not supported for DataLakeCatalog databases: " + "the catalog is shared, run the query without ON CLUSTER"); +} + bool maybeRemoveOnCluster(const ASTPtr & query_ptr, ContextPtr context) { const auto * query = dynamic_cast(query_ptr.get()); diff --git a/src/Interpreters/executeDDLQueryOnCluster.h b/src/Interpreters/executeDDLQueryOnCluster.h index 69e0c38834e6..822cdd84c29d 100644 --- a/src/Interpreters/executeDDLQueryOnCluster.h +++ b/src/Interpreters/executeDDLQueryOnCluster.h @@ -19,10 +19,18 @@ namespace DB struct DDLLogEntry; class Cluster; using ClusterPtr = std::shared_ptr; +class IDatabase; +using DatabasePtr = std::shared_ptr; /// Returns true if provided ALTER type can be executed ON CLUSTER bool isSupportedAlterTypeForOnClusterDDLQuery(int type); +/// Throws if DDL against this database's tables does not support `ON CLUSTER`. +void checkDatabaseSupportsOnClusterDDL(const DatabasePtr & database); + +/// Same, for every database the query targets. +void checkQueryDatabasesSupportOnClusterDDL(const ASTPtr & query_ptr, ContextPtr context); + struct DDLQueryOnClusterParams { /// A cluster to execute a distributed query. diff --git a/src/Storages/IStorage.h b/src/Storages/IStorage.h index 851fe4a87c7b..4dc93dae16a0 100644 --- a/src/Storages/IStorage.h +++ b/src/Storages/IStorage.h @@ -556,6 +556,12 @@ It is currently only implemented in StorageObjectStorage. */ virtual void drop() {} + /** Called by `DROP TABLE` while the query is still running. `drop` itself can run much later in a + * background thread, where only the global context is available, so a storage that needs + * query-level settings while dropping (for example `data_lake_delete_data_on_drop`) captures them here. + */ + virtual void prepareForDrop(ContextPtr /* query_context */) {} + virtual void dropInnerTableIfAny(bool /* sync */, ContextPtr /* context */) {} /// Return true if the storage supports TRUNCATE operation. diff --git a/src/Storages/ObjectStorage/DataLakes/DataLakeConfiguration.h b/src/Storages/ObjectStorage/DataLakes/DataLakeConfiguration.h index 92c8bf4cf60a..6095b4dc4ba5 100644 --- a/src/Storages/ObjectStorage/DataLakes/DataLakeConfiguration.h +++ b/src/Storages/ObjectStorage/DataLakes/DataLakeConfiguration.h @@ -343,10 +343,10 @@ class DataLakeConfiguration : public BaseStorageConfiguration, public std::enabl return current_metadata->getColumnMapperForCurrentSchema(storage_metadata_snapshot, context); } - void drop(ContextPtr local_context) override + void drop(bool delete_data) override { if (current_metadata) - current_metadata->drop(local_context); + current_metadata->drop(delete_data); } SinkToStoragePtr write( @@ -853,7 +853,7 @@ class StorageIcebergConfiguration : public StorageObjectStorageConfiguration, pu bool supportsPrewhere() const override { return getImpl().supportsPrewhere(); } - void drop(ContextPtr context) override { getImpl().drop(context); } + void drop(bool delete_data) override { getImpl().drop(delete_data); } protected: void createDynamicConfiguration(ASTs & args, ContextPtr context) diff --git a/src/Storages/ObjectStorage/DataLakes/IDataLakeMetadata.h b/src/Storages/ObjectStorage/DataLakes/IDataLakeMetadata.h index 8784705e85b2..43ee6e5c4c53 100644 --- a/src/Storages/ObjectStorage/DataLakes/IDataLakeMetadata.h +++ b/src/Storages/ObjectStorage/DataLakes/IDataLakeMetadata.h @@ -287,7 +287,8 @@ class IDataLakeMetadata : boost::noncopyable throwNotImplemented("truncate"); } - virtual void drop(ContextPtr) { } + /// `delete_data` is what `StorageObjectStorage::drop` resolved from `data_lake_delete_data_on_drop`. + virtual void drop(bool /* delete_data */) { } virtual ObjectStorageType getObjectStorageType() const { return ObjectStorageType::None; } diff --git a/src/Storages/ObjectStorage/DataLakes/Iceberg/Compaction.cpp b/src/Storages/ObjectStorage/DataLakes/Iceberg/Compaction.cpp index 5855504b22ac..e3f046866040 100644 --- a/src/Storages/ObjectStorage/DataLakes/Iceberg/Compaction.cpp +++ b/src/Storages/ObjectStorage/DataLakes/Iceberg/Compaction.cpp @@ -1023,7 +1023,7 @@ static void writeMetadataFiles( auto new_snapshot = metadata_generator.generateNextMetadata( plan.generator, - generated_metadata_info.path, + Iceberg::IcebergPathFromMetadata{}, history_record.parent_id, append->added_files, total_records_count, diff --git a/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergMetadata.cpp b/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergMetadata.cpp index 7f7d7211c680..0e0b1a8dacef 100644 --- a/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergMetadata.cpp +++ b/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergMetadata.cpp @@ -120,6 +120,7 @@ extern const int LOGICAL_ERROR; extern const int NOT_IMPLEMENTED; extern const int ICEBERG_SPECIFICATION_VIOLATION; extern const int S3_ERROR; +extern const int FILE_ALREADY_EXISTS; extern const int TABLE_ALREADY_EXISTS; extern const int SUPPORT_IS_DISABLED; extern const int METADATA_MISMATCH; @@ -145,7 +146,6 @@ extern const SettingsBool allow_experimental_iceberg_compaction; extern const SettingsBool allow_experimental_geo_types_in_iceberg; extern const SettingsBool allow_iceberg_remove_orphan_files; extern const SettingsBool allow_experimental_expire_snapshots; -extern const SettingsBool iceberg_delete_data_on_drop; } static constexpr size_t MAX_TRANSACTION_RETRIES = 100; @@ -917,22 +917,53 @@ void IcebergMetadata::createInitial( if (!configuration_ptr) throw Exception(ErrorCodes::LOGICAL_ERROR, "Trying to create Iceberg table, but storage configuration is expired"); - std::vector metadata_files; - try + const bool catalog_manages_location = catalog && catalog->managesTableLocation(); + const bool catalog_writes_metadata_file = catalog && catalog->isTransactional(); + + String namespace_name; + String table_name; + if (catalog) + std::tie(namespace_name, table_name) = DataLake::parseTableName(table_id_.getTableName()); + + auto throw_leftover_metadata = [&] { - metadata_files = listFiles(*object_storage, configuration_ptr->getPathForRead().path, "metadata", ".metadata.json"); - } - catch (const Exception & ex) + throw Exception(ErrorCodes::TABLE_ALREADY_EXISTS, + "The catalog has no table {}.{} registered, but Iceberg metadata files are already present at {}, " + "so creating the table there would clash with them. This is usually left behind by a previous " + "`DROP TABLE` without `data_lake_delete_data_on_drop`, which keeps the data and metadata in " + "place: remove the leftover files, or create the table at a different location", + namespace_name, table_name, configuration_ptr->getPathForRead().path); + }; + + if (catalog_manages_location) { - throw Exception(ErrorCodes::BAD_ARGUMENTS, "NoSuchBucket: {}", ex.what()); + DataLake::TableMetadata existing_table; + if (catalog->tryGetTableMetadata(namespace_name, table_name, local_context, existing_table)) + throw Exception(ErrorCodes::TABLE_ALREADY_EXISTS, + "Table {}.{} already exists in the catalog", namespace_name, table_name); } - if (!metadata_files.empty()) + else { - if (if_not_exists) - return; - else - throw Exception( - ErrorCodes::TABLE_ALREADY_EXISTS, "Iceberg table with path {} already exists", configuration_ptr->getPathForRead().path); + std::vector metadata_files; + try + { + metadata_files = listFiles(*object_storage, configuration_ptr->getPathForRead().path, "metadata", ".metadata.json"); + } + catch (const Exception & ex) + { + throw Exception(ErrorCodes::BAD_ARGUMENTS, "NoSuchBucket: {}", ex.what()); + } + if (!metadata_files.empty()) + { + if (!catalog) + { + if (if_not_exists) + return; + throw Exception( + ErrorCodes::TABLE_ALREADY_EXISTS, "Iceberg table with path {} already exists", configuration_ptr->getPathForRead().path); + } + throw_leftover_metadata(); + } } String location_path = configuration_ptr->getRawPath().path; @@ -943,7 +974,8 @@ void IcebergMetadata::createInitial( = configuration_ptr->getTypeName() + "://" + configuration_ptr->getNamespace() + "/" + configuration_ptr->getRawPath().path; auto [metadata_content_object, metadata_content] = createEmptyMetadataFile( - location_path, *columns, partition_by, order_by, local_context, configuration_ptr->getDataLakeSettings()[DataLakeStorageSetting::iceberg_format_version]); + location_path, *columns, partition_by, order_by, local_context, + configuration_ptr->getDataLakeSettings()[DataLakeStorageSetting::iceberg_format_version], /*is_catalog_table=*/ bool(catalog)); auto compression_method_str = local_context->getSettingsRef()[Setting::iceberg_metadata_compression_method].value; auto compression_method = chooseCompressionMethod(compression_method_str, compression_method_str); @@ -951,44 +983,104 @@ void IcebergMetadata::createInitial( if (!compression_suffix.empty()) compression_suffix = "." + compression_suffix; - auto filename = fmt::format("{}metadata/v1{}.metadata.json", configuration_ptr->getRawPath().path, compression_suffix); + auto table_uuid = metadata_content_object->getValue(Iceberg::f_table_uuid); + auto metadata_file_name = catalog_writes_metadata_file + ? fmt::format("v1-{}{}.metadata.json", table_uuid, compression_suffix) + : fmt::format("v1{}.metadata.json", compression_suffix); + auto filename = fmt::format("{}metadata/{}", configuration_ptr->getRawPath().path, metadata_file_name); if (catalog) { + /// The namespace default location is the namespace base, not this table's directory. + String namespace_location = location_path; + while (namespace_location.ends_with('/')) + namespace_location.pop_back(); + + String namespace_path = namespace_name; + std::replace(namespace_path.begin(), namespace_path.end(), '.', '/'); + if (namespace_location.ends_with("/" + namespace_path + "/" + table_name)) + namespace_location.resize(namespace_location.size() - table_name.size() - 1); + else + namespace_location.clear(); + /// Register the namespace before any files are written (but after all local /// validation, so a rejected CREATE leaves no trace in the catalog): a catalog /// that shares its storage view with the data (e.g. SeaweedFS) refuses to create /// a namespace over the plain directory those files would leave behind. - catalog->createNamespaceIfNotExists(DataLake::parseTableName(table_id_.getTableName()).first, location_path); + catalog->createNamespaceIfNotExists(namespace_name, namespace_location); } - try + if (!catalog_writes_metadata_file) { - writeMessageToFile(metadata_content, filename, object_storage, local_context, "*", "", compression_method); + try + { + writeMessageToFile(metadata_content, filename, object_storage, local_context, "*", "", compression_method); + } + catch (const Exception & e) + { + /// The write uses `If-None-Match: *`, so S3 answers `PreconditionFailed` when the metadata + /// file is already there: leftovers from an earlier drop, or a concurrent creation. + const bool precondition_failed + = (e.code() == ErrorCodes::S3_ERROR && e.message().contains("PreconditionFailed")) + || e.code() == ErrorCodes::FILE_ALREADY_EXISTS; + if (if_not_exists && precondition_failed) + { + if (!catalog) + return; + throw_leftover_metadata(); + } + throw; + } } - catch (const Exception & e) + + String filename_version_hint; + try { - /// The write uses `If-None-Match: *`, so S3 returns PreconditionFailed when the metadata file - /// already exists (e.g. leftover data after `DROP TABLE` with `iceberg_delete_data_on_drop` off, - /// or a concurrent creation). When `IF NOT EXISTS` was specified, this is expected. - if (if_not_exists && e.code() == ErrorCodes::S3_ERROR - && e.message().find("PreconditionFailed") != String::npos) - return; - throw; + if (!catalog_writes_metadata_file + && configuration_ptr->getDataLakeSettings()[DataLakeStorageSetting::iceberg_use_version_hint].value) + { + auto version_hint_path = configuration_ptr->getRawPath().path + "metadata/version-hint.text"; + writeMessageToFile("1", version_hint_path, object_storage, local_context, "*", ""); + filename_version_hint = version_hint_path; + } } - - if (configuration_ptr->getDataLakeSettings()[DataLakeStorageSetting::iceberg_use_version_hint].value) + catch (...) { - auto filename_version_hint = configuration_ptr->getRawPath().path + "metadata/version-hint.text"; - writeMessageToFile("1", filename_version_hint, object_storage, local_context, "*", ""); + if (!catalog_writes_metadata_file) + { + tryLogCurrentException(__PRETTY_FUNCTION__, "Removing the files of the Iceberg table that failed to be created"); + object_storage->removeObjectIfExists(StoredObject(filename)); + if (!filename_version_hint.empty()) + object_storage->removeObjectIfExists(StoredObject(filename_version_hint)); + } + throw; } if (catalog) { - auto catalog_filename = configuration_ptr->getTypeName() + "://" + configuration_ptr->getNamespace() + "/" - + configuration_ptr->getRawPath().path + "metadata/v1.metadata.json"; - const auto & [namespace_name, table_name] = DataLake::parseTableName(table_id_.getTableName()); - catalog->createTable(namespace_name, table_name, catalog_filename, metadata_content_object); + auto catalog_filename + = configuration_ptr->getTypeName() + "://" + configuration_ptr->getNamespace() + "/" + filename; + + /// Always ask for a conflict to be reported as `false`, so the files staged above are removed + /// before `TABLE_ALREADY_EXISTS` is thrown, with or without `IF NOT EXISTS`. + if (!catalog->createTable(namespace_name, table_name, catalog_filename, metadata_content_object, compression_method, /* if_not_exists */ true)) + { + if (!catalog_writes_metadata_file) + { + LOG_INFO( + getLogger("IcebergMetadata"), + "Table {}.{} was registered in the catalog by another client, removing the initial metadata file {} " + "written by this `CREATE`", + namespace_name, + table_name, + filename); + object_storage->removeObjectIfExists(StoredObject(filename)); + if (!filename_version_hint.empty()) + object_storage->removeObjectIfExists(StoredObject(filename_version_hint)); + } + throw Exception(ErrorCodes::TABLE_ALREADY_EXISTS, + "Table {}.{} already exists in the catalog", namespace_name, table_name); + } } } @@ -1585,16 +1677,16 @@ SinkToStoragePtr IcebergMetadata::write( } } -void IcebergMetadata::drop(ContextPtr context) +void IcebergMetadata::drop(bool delete_data) { - if (!context->getSettingsRef()[Setting::iceberg_delete_data_on_drop].value) + if (!delete_data) return; /// Files outside `table_path` (secondary storage, or base storage elsewhere in the bucket) are only /// discoverable through the metadata graph the base wipe below removes, so enumerate them first. Let /// a failure propagate rather than wiping the metadata re-enumeration on retry depends on (fail closed). auto external_files = Iceberg::collectReachableFiles( - object_storage, persistent_components, data_lake_settings, context, log, *secondary_storages).external_files; + object_storage, persistent_components, data_lake_settings, Context::getGlobalContextInstance(), log, *secondary_storages).external_files; /// Delete these files leaf-first (reverse of the traversal's append order) so an interrupted drop /// can re-enumerate the rest on retry; batch per storage. Shared files are deleted too, as with `PURGE`. diff --git a/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergMetadata.h b/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergMetadata.h index 35845ae4415b..755559aea900 100644 --- a/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergMetadata.h +++ b/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergMetadata.h @@ -223,7 +223,7 @@ class IcebergMetadata : public IDataLakeMetadata StorageMetadataPtr storage_metadata, ContextPtr local_context) const override; - void drop(ContextPtr context) override; + void drop(bool delete_data) override; Poco::JSON::Object::Ptr getMetadataJSON(ContextPtr local_context) const; diff --git a/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergPath.h b/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergPath.h index 26cb3a564d0f..9d8878f394fc 100644 --- a/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergPath.h +++ b/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergPath.h @@ -89,6 +89,8 @@ class IcebergPathResolver IcebergPathFromMetadata reverseResolve(const String & storage_path) const { + if (!table_location.empty() && storage_path.starts_with(table_location)) + return IcebergPathFromMetadata::deserialize(storage_path); if (storage_path.size() > table_root.size() && storage_path.starts_with(table_root)) return IcebergPathFromMetadata::deserialize(table_location + storage_path.substr(table_root.size())); diff --git a/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergWrites.cpp b/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergWrites.cpp index 0adbf4f2c0e5..3122c1cac600 100644 --- a/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergWrites.cpp +++ b/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergWrites.cpp @@ -1235,7 +1235,7 @@ void generateManifestList( } /// Copy entries from the parent snapshot's manifest list: `use_previous_snapshots` copies all, `carry_forward_manifest_paths` copies only the listed manifests. - if (use_previous_snapshots || !carry_forward_manifest_paths.empty()) + if ((use_previous_snapshots || !carry_forward_manifest_paths.empty()) && new_snapshot->has(Iceberg::f_parent_snapshot_id)) { auto parent_snapshot_id = new_snapshot->getValue(Iceberg::f_parent_snapshot_id); auto snapshots = metadata->getArray(Iceberg::f_snapshots); @@ -1358,6 +1358,7 @@ IcebergStorageSink::IcebergStorageSink( compression_method, persistent_table_components.table_uuid); metadata_compression_method = compression_method; + previous_metadata_file_path = metadata_path; filename_generator = FileNamesGenerator( persistent_table_components.path_resolver.getTableLocation(), (catalog != nullptr && catalog->isTransactional()), metadata_compression_method, write_format); @@ -1583,7 +1584,7 @@ bool IcebergStorageSink::initializeMetadata() total_data_files += static_cast(writer.getDataFiles().size()); auto [new_snapshot, manifest_list_path] = MetadataGenerator(metadata).generateNextMetadata( filename_generator, - metadata_info.path, + previous_metadata_file_path.empty() ? Iceberg::IcebergPathFromMetadata{} : resolver.reverseResolve(previous_metadata_file_path), parent_snapshot, total_data_files, total_rows, @@ -1632,6 +1633,7 @@ bool IcebergStorageSink::initializeMetadata() LOG_DEBUG(log, "Rereading metadata file {} with version {}", metadata_path, last_version); metadata_compression_method = compression_method; + previous_metadata_file_path = metadata_path; filename_generator.setVersion(last_version + 1); metadata = getMetadataJSONObject( diff --git a/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergWrites.h b/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergWrites.h index a7c6c6a322c5..84e139e40562 100644 --- a/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergWrites.h +++ b/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergWrites.h @@ -205,6 +205,7 @@ class IcebergStorageSink final : public SinkToStorage bool initializeMetadata(); FileNamesGenerator filename_generator; + String previous_metadata_file_path; std::optional partitioner; Poco::JSON::Object::Ptr partititon_spec; Int64 partition_spec_id; diff --git a/src/Storages/ObjectStorage/DataLakes/Iceberg/MetadataGenerator.cpp b/src/Storages/ObjectStorage/DataLakes/Iceberg/MetadataGenerator.cpp index 7c232a26b8fe..9dd25697b964 100644 --- a/src/Storages/ObjectStorage/DataLakes/Iceberg/MetadataGenerator.cpp +++ b/src/Storages/ObjectStorage/DataLakes/Iceberg/MetadataGenerator.cpp @@ -397,7 +397,7 @@ Poco::JSON::Object::Ptr MetadataGenerator::getParentSnapshot(Int64 parent_snapsh MetadataGenerator::NextMetadataResult MetadataGenerator::generateNextMetadata( FileNamesGenerator & generator, - const Iceberg::IcebergPathFromMetadata & metadata_file_path, + const Iceberg::IcebergPathFromMetadata & previous_metadata_file_path, Int64 parent_snapshot_id, Int64 added_files, Int64 added_records, @@ -421,7 +421,8 @@ MetadataGenerator::NextMetadataResult MetadataGenerator::generateNextMetadata( auto manifest_list_path = generator.generateManifestListName(snapshot_id, format_version); new_snapshot->set(Iceberg::f_metadata_snapshot_id, snapshot_id); - new_snapshot->set(Iceberg::f_parent_snapshot_id, parent_snapshot_id); + if (parent_snapshot_id != -1) + new_snapshot->set(Iceberg::f_parent_snapshot_id, parent_snapshot_id); auto now = std::chrono::system_clock::now(); auto ms = duration_cast(now.time_since_epoch()); @@ -511,9 +512,10 @@ MetadataGenerator::NextMetadataResult MetadataGenerator::generateNextMetadata( else metadata_object->getObject(Iceberg::f_refs)->getObject(Iceberg::f_main)->set(Iceberg::f_metadata_snapshot_id, snapshot_id); + if (!previous_metadata_file_path.empty()) { Poco::JSON::Object::Ptr new_metadata_item = new Poco::JSON::Object; - new_metadata_item->set(Iceberg::f_metadata_file, metadata_file_path.serialize()); + new_metadata_item->set(Iceberg::f_metadata_file, previous_metadata_file_path.serialize()); new_metadata_item->set(Iceberg::f_timestamp_ms, timestamp); getOrCreateArray(metadata_object, Iceberg::f_metadata_log)->add(new_metadata_item); } diff --git a/src/Storages/ObjectStorage/DataLakes/Iceberg/MetadataGenerator.h b/src/Storages/ObjectStorage/DataLakes/Iceberg/MetadataGenerator.h index 33ff405a65d7..c56d97701971 100644 --- a/src/Storages/ObjectStorage/DataLakes/Iceberg/MetadataGenerator.h +++ b/src/Storages/ObjectStorage/DataLakes/Iceberg/MetadataGenerator.h @@ -29,7 +29,7 @@ class MetadataGenerator NextMetadataResult generateNextMetadata( FileNamesGenerator & generator, - const Iceberg::IcebergPathFromMetadata & metadata_file_path, + const Iceberg::IcebergPathFromMetadata & previous_metadata_file_path, Int64 parent_snapshot_id, Int64 added_files, Int64 added_records, diff --git a/src/Storages/ObjectStorage/DataLakes/Iceberg/Mutations.cpp b/src/Storages/ObjectStorage/DataLakes/Iceberg/Mutations.cpp index 2933f50f9296..5e80c8df477a 100644 --- a/src/Storages/ObjectStorage/DataLakes/Iceberg/Mutations.cpp +++ b/src/Storages/ObjectStorage/DataLakes/Iceberg/Mutations.cpp @@ -437,12 +437,13 @@ static bool writeMetadataFiles( std::optional & chunk_partitioner, Iceberg::FileContentType content_type, SharedHeader sample_block, - bool write_metadata_json_file) + bool write_metadata_json_file, + const Iceberg::IcebergPathFromMetadata & previous_metadata_file_path) { auto metadata_info = filename_generator.generateMetadataPathWithInfo(); auto storage_metadata_name = path_resolver.resolve(metadata_info.path); Int64 parent_snapshot = -1; - if (metadata->has(Iceberg::f_current_snapshot_id)) + if (metadata->has(Iceberg::f_current_snapshot_id) && !metadata->isNull(Iceberg::f_current_snapshot_id)) parent_snapshot = metadata->getValue(Iceberg::f_current_snapshot_id); Int64 total_rows = 0; @@ -461,7 +462,7 @@ static bool writeMetadataFiles( { auto result = MetadataGenerator(metadata).generateNextMetadata( filename_generator, - metadata_info.path, + previous_metadata_file_path, parent_snapshot, /* added_files */ 0, /* added_records */ 0, @@ -476,7 +477,7 @@ static bool writeMetadataFiles( { auto result = MetadataGenerator(metadata).generateNextMetadata( filename_generator, - metadata_info.path, + previous_metadata_file_path, parent_snapshot, /* added_files */ total_files, /* added_records */ total_rows, @@ -672,6 +673,7 @@ void mutate( filename_generator.setCompressionMethod(compression_method); auto metadata = getMetadataJSONObject(metadata_path, object_storage, persistent_table_components.metadata_cache, context, log, compression_method, persistent_table_components.table_uuid); + auto previous_metadata_path = persistent_table_components.path_resolver.reverseResolve(metadata_path); /// Iceberg v3 writers must not add new position-delete files; row-level deletes require /// deletion vectors. Fail closed before any object writes until ClickHouse can write DVs. @@ -715,7 +717,7 @@ void mutate( current_iceberg_snapshot.metadata_file_path = metadata_path; current_iceberg_snapshot.metadata_version = last_version; current_iceberg_snapshot.schema_id = static_cast(current_schema_id); - if (metadata->has(Iceberg::f_current_snapshot_id)) + if (metadata->has(Iceberg::f_current_snapshot_id) && !metadata->isNull(Iceberg::f_current_snapshot_id)) { Int64 snapshot_id_val = metadata->getValue(Iceberg::f_current_snapshot_id); if (snapshot_id_val >= 0) @@ -761,7 +763,8 @@ void mutate( chunk_partitioner, Iceberg::FileContentType::POSITION_DELETE, std::make_shared(getPositionDeleteFileSampleBlock()), - !mutation_files->data_file); + !mutation_files->data_file, + previous_metadata_path); if (!result_delete_files_metadata) continue; @@ -784,7 +787,8 @@ void mutate( chunk_partitioner, Iceberg::FileContentType::DATA, sample_block, - true); + true, + previous_metadata_path); if (!result_data_files_metadata) { continue; diff --git a/src/Storages/ObjectStorage/DataLakes/Iceberg/Utils.cpp b/src/Storages/ObjectStorage/DataLakes/Iceberg/Utils.cpp index d378e21f03a5..0a75343a22b3 100644 --- a/src/Storages/ObjectStorage/DataLakes/Iceberg/Utils.cpp +++ b/src/Storages/ObjectStorage/DataLakes/Iceberg/Utils.cpp @@ -22,6 +22,10 @@ #include #include #include +#include +#include +#include +#include #include #include #include @@ -88,6 +92,7 @@ namespace DB::DataLakeStorageSetting namespace DB::Setting { extern const SettingsString iceberg_metadata_compression_method; +extern const SettingsBool datalake_ignore_unsupported_table_properties; } namespace ProfileEvents @@ -378,7 +383,17 @@ bool writeMetadataFileAndVersionHint( } else { - break; + /// Remove the metadata file written above, otherwise version-hint resolution could later + /// pick this uncommitted file as the latest version. + LOG_INFO( + getLogger("IcebergMetadataFileWriter"), + "Removing the uncommitted Iceberg metadata file {}: the version hint is already at version {}, " + "at or past version {} this commit tried to write, so the write did not commit", + storage_metadata_path, + old_version, + metadata_file_info.version); + object_storage->removeObjectIfExists(StoredObject(storage_metadata_path)); + return false; } ++i; } @@ -843,6 +858,8 @@ static Poco::JSON::Object::Ptr getPartitionField( { if (!param.has_value()) throw Exception(ErrorCodes::BAD_ARGUMENTS, "TRUNCATE function for iceberg partitioning requires one integer parameter"); + if (*param <= 0) + throw Exception(ErrorCodes::BAD_ARGUMENTS, "TRUNCATE function for iceberg partitioning requires a positive width, got {}", *param); result->set(Iceberg::f_transform, fmt::format("truncate[{}]", *param)); return result; } @@ -850,6 +867,8 @@ static Poco::JSON::Object::Ptr getPartitionField( { if (!param.has_value()) throw Exception(ErrorCodes::BAD_ARGUMENTS, "BUCKET function for iceberg partitioning requires one integer parameter"); + if (*param <= 0) + throw Exception(ErrorCodes::BAD_ARGUMENTS, "BUCKET function for iceberg partitioning requires a positive number of buckets, got {}", *param); result->set(Iceberg::f_transform, fmt::format("bucket[{}]", *param)); return result; } @@ -904,18 +923,19 @@ static String parseColumnArgument(const ASTPtr & arg_ast, const String & clickho return identifier->name(); } -static std::pair parseFunction(const ASTPtr & func_object) +static const std::unordered_map clickhouse_name_to_iceberg { - const static std::unordered_map clickhouse_name_to_iceberg = { - {"identity", "identity"}, - {"icebergBucket", "bucket"}, - {"icebergTruncate", "truncate"}, - {"toYearNumSinceEpoch", "year"}, - {"toMonthNumSinceEpoch", "month"}, - {"toRelativeDayNum", "day"}, - {"toRelativeHourNum", "hour"} - }; + {"identity", "identity"}, + {"icebergBucket", "bucket"}, + {"icebergTruncate", "truncate"}, + {"toYearNumSinceEpoch", "year"}, + {"toMonthNumSinceEpoch", "month"}, + {"toRelativeDayNum", "day"}, + {"toRelativeHourNum", "hour"} +}; +static std::pair parseFunction(const ASTPtr & func_object) +{ const auto * func = func_object ? func_object->as() : nullptr; if (!func) throw Exception(ErrorCodes::BAD_ARGUMENTS, "Invalid iceberg sort order expression, expected a function"); @@ -1050,13 +1070,52 @@ static std::vector> parseTransformAndColumnPairs(ASTPt return result; } +/// SQL expressions are validated separately; this only checks whether an Iceberg transform can encode the key. +static bool isRepresentableKey(ASTPtr key, const std::unordered_map & column_name_to_source_id) +{ + key = unwrapOrderByElement(key); + if (const auto * identifier = key->as()) + return column_name_to_source_id.contains(identifier->name()); + + const auto * function = key->as(); + const auto * expressions = key->as(); + if (function && function->name == "tuple") + expressions = function->arguments->as(); + if (expressions) + { + bool representable = true; + for (const auto & child : expressions->children) + representable = isRepresentableKey(child, column_name_to_source_id) && representable; + return representable; + } + + if (!function || !clickhouse_name_to_iceberg.contains(function->name)) + return false; + + const auto & arguments = function->arguments->children; + const bool has_parameter = function->name == "icebergBucket" || function->name == "icebergTruncate"; + if (arguments.size() != (has_parameter ? 2 : 1)) + throw Exception(ErrorCodes::BAD_ARGUMENTS, "Invalid arguments for Iceberg transform {}", function->name); + if (has_parameter) + { + const auto * parameter = arguments.front()->as(); + if (!parameter) + return false; + if (parameter->value.safeGet() <= 0) + throw Exception(ErrorCodes::BAD_ARGUMENTS, "Iceberg transform {} requires a positive parameter", function->name); + } + const auto * column = arguments.back()->as(); + return column && column_name_to_source_id.contains(column->name()); +} + std::pair createEmptyMetadataFile( String path_location, const ColumnsDescription & columns, ASTPtr partition_by, ASTPtr order_by, ContextPtr context, - UInt64 format_version) + UInt64 format_version, + bool is_catalog_table) { std::unordered_map column_name_to_source_id; static Poco::UUIDGenerator uuid_generator; @@ -1097,6 +1156,29 @@ std::pair createEmptyMetadataFile( schema_array->add(schema_representation); new_metadata_file_content->set(Iceberg::f_schemas, schema_array); + if (is_catalog_table) + { + const bool ignore_unsupported_properties = context->getSettingsRef()[Setting::datalake_ignore_unsupported_table_properties]; + auto validate_key = [&](ASTPtr & key, const char * clause) + { + if (!key) + return; + /// Validate even when omissions are allowed: invalid SQL must not silently disappear. + KeyDescription::getKeyFromAST(key, columns, {}, context); + if (!isRepresentableKey(key, column_name_to_source_id)) + { + if (!ignore_unsupported_properties) + throw Exception(ErrorCodes::BAD_ARGUMENTS, + "Cannot represent {} expression {} in Iceberg. " + "Set datalake_ignore_unsupported_table_properties = 1 to omit unsupported properties", + clause, key->formatForLogging()); + key.reset(); + } + }; + validate_key(partition_by, "PARTITION BY"); + validate_key(order_by, "ORDER BY"); + } + new_metadata_file_content->set(Iceberg::f_default_spec_id, 0); Poco::JSON::Object::Ptr partition_spec = new Poco::JSON::Object; partition_spec->set(Iceberg::f_spec_id, 0); @@ -1629,6 +1711,146 @@ std::optional getSortingKeyDisplayStringFromMetadata(Poco::JSON::Object: return std::nullopt; } +/// Returns `std::nullopt` when the field cannot be represented in `CREATE TABLE`, and an empty string for a `void` field. +static std::optional formatIcebergTransformExpression( + Poco::JSON::Object::Ptr field, const std::unordered_map & source_id_to_column_name) +{ + const auto transform = parseTransformAndArgument(field->getValue(f_transform), /*time_zone=*/ ""); + if (!transform) + return std::nullopt; + + if (transform->transform_name == "tuple") + return String{}; + + const auto it = source_id_to_column_name.find(field->getValue(f_source_id)); + if (it == source_id_to_column_name.end()) + return std::nullopt; + + auto column_name = backQuoteIfNeed(it->second); + if (transform->transform_name == "identity") + return column_name; + if (transform->argument) + return fmt::format("{}({}, {})", transform->transform_name, *transform->argument, column_name); + return fmt::format("{}({})", transform->transform_name, column_name); +} + +std::tuple getPartitionAndSortingKeyASTsFromMetadata(const Poco::JSON::Object::Ptr & metadata_object) +{ + const auto schema = metadata_object->has(f_schemas) && metadata_object->has(f_current_schema_id) + ? parseTableSchemaV2Method(metadata_object).first + : parseTableSchemaV1Method(metadata_object).first; + + std::unordered_map source_id_to_column_name; + const auto schema_fields = schema->getArray(f_fields); + for (UInt32 i = 0; i < schema_fields->size(); ++i) + { + const auto field = schema_fields->getObject(i); + source_id_to_column_name[field->getValue(f_id)] = field->getValue(f_name); + } + + Poco::JSON::Array::Ptr partition_fields; + if (metadata_object->has(f_partition_specs)) + { + const auto default_spec_id = metadata_object->getValue(f_default_spec_id); + const auto partition_specs = metadata_object->getArray(f_partition_specs); + for (UInt32 i = 0; i < partition_specs->size(); ++i) + { + const auto partition_spec = partition_specs->getObject(i); + if (partition_spec->getValue(f_spec_id) == default_spec_id) + { + partition_fields = partition_spec->getArray(f_fields); + break; + } + } + } + else if (metadata_object->has(f_partition_spec)) + { + partition_fields = metadata_object->getArray(f_partition_spec); + } + + String unsupported_properties; + ASTPtr partition_by; + if (partition_fields) + { + /// `PARTITION BY` is omitted entirely if any field cannot be represented. + bool representable = true; + std::vector expressions; + for (UInt32 i = 0; representable && i < partition_fields->size(); ++i) + { + auto expression = formatIcebergTransformExpression(partition_fields->getObject(i), source_id_to_column_name); + if (!expression) + representable = false; + else if (!expression->empty()) + expressions.push_back(std::move(*expression)); + } + + if (!representable) + unsupported_properties = "PARTITION BY"; + + if (representable && !expressions.empty()) + { + String partition_by_str = fmt::format("{}", fmt::join(expressions, ", ")); + if (expressions.size() > 1) + partition_by_str = "(" + partition_by_str + ")"; + ParserExpression parser; + partition_by = parseQuery(parser, partition_by_str, 0, DBMS_DEFAULT_MAX_PARSER_DEPTH, DBMS_DEFAULT_MAX_PARSER_BACKTRACKS); + } + } + + ASTPtr order_by; + if (metadata_object->has(f_sort_orders)) + { + const auto default_sort_order_id = metadata_object->getValue(f_default_sort_order_id); + const auto sort_orders = metadata_object->getArray(f_sort_orders); + for (UInt32 i = 0; i < sort_orders->size(); ++i) + { + const auto sort_order = sort_orders->getObject(i); + if (sort_order->getValue(f_order_id) != default_sort_order_id) + continue; + + /// `ORDER BY` is omitted entirely if any field cannot be represented. Table `ORDER BY` has no + /// `NULLS FIRST/LAST`, and `CREATE TABLE` writes `nulls-first`, so any other null order is unrepresentable. + bool representable = true; + std::vector expressions; + const auto sort_fields = sort_order->getArray(f_fields); + for (UInt32 field_index = 0; representable && field_index < sort_fields->size(); ++field_index) + { + const auto sort_field = sort_fields->getObject(field_index); + auto expression = formatIcebergTransformExpression(sort_field, source_id_to_column_name); + if (!expression || Poco::toLower(sort_field->getValue("null-order")) != "nulls-first") + { + representable = false; + continue; + } + if (expression->empty()) + continue; + if (Poco::toLower(sort_field->getValue(f_direction)) == "desc") + *expression += " DESC"; + expressions.push_back(std::move(*expression)); + } + + if (!representable) + { + if (!unsupported_properties.empty()) + unsupported_properties += ", "; + unsupported_properties += "ORDER BY"; + } + + if (representable && !expressions.empty()) + { + String order_by_str = fmt::format("{}", fmt::join(expressions, ", ")); + if (expressions.size() > 1) + order_by_str = "(" + order_by_str + ")"; + ParserStorageOrderByClause parser(/*allow_order_=*/ true); + order_by = parseQuery(parser, order_by_str, 0, DBMS_DEFAULT_MAX_PARSER_DEPTH, DBMS_DEFAULT_MAX_PARSER_BACKTRACKS); + } + break; + } + } + + return {partition_by, order_by, unsupported_properties}; +} + DataTypePtr getFunctionResultType(const String & iceberg_transform_name, DataTypePtr source_type) { if (iceberg_transform_name.starts_with("identity") || iceberg_transform_name.starts_with("truncate")) diff --git a/src/Storages/ObjectStorage/DataLakes/Iceberg/Utils.h b/src/Storages/ObjectStorage/DataLakes/Iceberg/Utils.h index 4d403a785dc1..fe4770d0e9ff 100644 --- a/src/Storages/ObjectStorage/DataLakes/Iceberg/Utils.h +++ b/src/Storages/ObjectStorage/DataLakes/Iceberg/Utils.h @@ -4,6 +4,7 @@ #include "config.h" #include #include +#include #include #include @@ -111,7 +112,8 @@ std::pair createEmptyMetadataFile( ASTPtr partition_by, ASTPtr order_by, ContextPtr context, - UInt64 format_version = 2); + UInt64 format_version = 2, + bool is_catalog_table = false); MetadataFileWithInfo getLatestOrExplicitMetadataFileAndVersion( const ObjectStoragePtr & object_storage, @@ -151,6 +153,8 @@ std::optional getSortingKeyDisplayStringFromMetadata( Poco::JSON::Object::Ptr metadata_object, const NamesAndTypesList & ch_schema); std::optional getPartitionKeyStringFromMetadata( Poco::JSON::Object::Ptr metadata_object, const NamesAndTypesList & ch_schema, ContextPtr local_context); + +std::tuple getPartitionAndSortingKeyASTsFromMetadata(const Poco::JSON::Object::Ptr & metadata_object); void sortBlockByKeyDescription(Block & block, const KeyDescription & sort_description, ContextPtr context); void forEachAvroEntry( diff --git a/src/Storages/ObjectStorage/StorageObjectStorage.cpp b/src/Storages/ObjectStorage/StorageObjectStorage.cpp index e7913b4ec0fc..0f79f5edf2d5 100644 --- a/src/Storages/ObjectStorage/StorageObjectStorage.cpp +++ b/src/Storages/ObjectStorage/StorageObjectStorage.cpp @@ -59,6 +59,7 @@ namespace Setting extern const SettingsInt64 delta_lake_snapshot_start_version; extern const SettingsInt64 delta_lake_snapshot_end_version; extern const SettingsUInt64 max_streams_for_files_processing_in_cluster_functions; + extern const SettingsBool data_lake_delete_data_on_drop; } namespace ErrorCodes @@ -1013,15 +1014,42 @@ void StorageObjectStorage::truncate( object_storage->removeObjectsIfExist(objects); } +void StorageObjectStorage::prepareForDrop(ContextPtr query_context) +{ + delete_data_on_drop = query_context->getSettingsRef()[Setting::data_lake_delete_data_on_drop]; +} + void StorageObjectStorage::drop() { + dropImpl(delete_data_on_drop.load(), catalog, configuration, storage_id, log); +} + +void StorageObjectStorage::dropImpl( + const std::optional & delete_data_on_drop, + const std::shared_ptr & catalog, + const StorageObjectStorageConfigurationPtr & configuration, + const StorageID & storage_id, + const LoggerPtr & log) +{ + const bool delete_data = delete_data_on_drop.value_or(false); + + if (!delete_data_on_drop + && Context::getGlobalContextInstance()->getSettingsRef()[Setting::data_lake_delete_data_on_drop]) + { + LOG_WARNING( + log, + "Keeping the data of table {} although `data_lake_delete_data_on_drop` is enabled server-wide: the value for this drop " + "could not be captured, which happens when the table was never loaded, and data is never deleted on a fallback path. " + "Access the table before dropping it, so that the settings of the `DROP TABLE` query reach the table.", + storage_id.getNameForLogs()); + } + if (catalog) { const auto [namespace_name, table_name] = DataLake::parseTableName(storage_id.getTableName()); - catalog->dropTable(namespace_name, table_name); + catalog->dropTable(namespace_name, table_name, delete_data, /* if_exists */ false); } - /// We cannot use query context here, because drop is executed in the background. - configuration->drop(Context::getGlobalContextInstance()); + configuration->drop(delete_data); } std::unique_ptr StorageObjectStorage::createReadBufferIterator( diff --git a/src/Storages/ObjectStorage/StorageObjectStorage.h b/src/Storages/ObjectStorage/StorageObjectStorage.h index 474e337e17ae..07c6c197c610 100644 --- a/src/Storages/ObjectStorage/StorageObjectStorage.h +++ b/src/Storages/ObjectStorage/StorageObjectStorage.h @@ -18,8 +18,10 @@ #include #include +#include #include #include +#include #include namespace DB @@ -103,6 +105,15 @@ class StorageObjectStorage : public IStorage, public IBackgroundOperation const IcebergCommitExportPartitionArguments & iceberg_commit_export_partition_arguments, ContextPtr local_context) override; + /// Shared `drop` implementation: removes the table from `catalog` (if any) and drops `configuration`, + /// deleting the data only if `delete_data_on_drop` was captured as `true` by `prepareForDrop`. + static void dropImpl( + const std::optional & delete_data_on_drop, + const std::shared_ptr & catalog, + const StorageObjectStorageConfigurationPtr & configuration, + const StorageID & storage_id, + const LoggerPtr & log); + void truncate( const ASTPtr & query, const StorageMetadataPtr & metadata_snapshot, @@ -111,6 +122,8 @@ class StorageObjectStorage : public IStorage, public IBackgroundOperation void drop() override; + void prepareForDrop(ContextPtr query_context) override; + bool supportsPartitionBy() const override { return true; } bool supportsSubcolumns() const override { return true; } @@ -272,6 +285,8 @@ class StorageObjectStorage : public IStorage, public IBackgroundOperation std::shared_ptr catalog; StorageID storage_id; BackgroundJobsAssignee background_operations_assignee; + + std::atomic> delete_data_on_drop; }; } diff --git a/src/Storages/ObjectStorage/StorageObjectStorageCluster.cpp b/src/Storages/ObjectStorage/StorageObjectStorageCluster.cpp index aff028efef01..a8a69596f9c2 100644 --- a/src/Storages/ObjectStorage/StorageObjectStorageCluster.cpp +++ b/src/Storages/ObjectStorage/StorageObjectStorageCluster.cpp @@ -793,6 +793,16 @@ void StorageObjectStorageCluster::drop() IStorageCluster::drop(); } +void StorageObjectStorageCluster::prepareForDrop(ContextPtr query_context) +{ + if (pure_storage) + { + pure_storage->prepareForDrop(query_context); + return; + } + IStorageCluster::prepareForDrop(query_context); +} + void StorageObjectStorageCluster::dropInnerTableIfAny(bool sync, ContextPtr context) { if (getClusterName(context).empty()) diff --git a/src/Storages/ObjectStorage/StorageObjectStorageCluster.h b/src/Storages/ObjectStorage/StorageObjectStorageCluster.h index 28cdaeb29643..7b76e2edbab3 100644 --- a/src/Storages/ObjectStorage/StorageObjectStorageCluster.h +++ b/src/Storages/ObjectStorage/StorageObjectStorageCluster.h @@ -74,6 +74,8 @@ class StorageObjectStorageCluster : public IStorageCluster void drop() override; + void prepareForDrop(ContextPtr query_context) override; + void dropInnerTableIfAny(bool sync, ContextPtr context) override; void truncate( diff --git a/src/Storages/ObjectStorage/StorageObjectStorageConfiguration.h b/src/Storages/ObjectStorage/StorageObjectStorageConfiguration.h index c1af64950396..8ef4f6288b44 100644 --- a/src/Storages/ObjectStorage/StorageObjectStorageConfiguration.h +++ b/src/Storages/ObjectStorage/StorageObjectStorageConfiguration.h @@ -338,7 +338,8 @@ class StorageObjectStorageConfiguration return true; } - virtual void drop(ContextPtr) {} + /// `delete_data` is what `StorageObjectStorage::drop` resolved from `data_lake_delete_data_on_drop`. + virtual void drop(bool /* delete_data */) {} virtual bool isBackgroundExecutable() const { diff --git a/src/Storages/StorageTableFunction.h b/src/Storages/StorageTableFunction.h index 884908cf0ccf..ef7044ff9af9 100644 --- a/src/Storages/StorageTableFunction.h +++ b/src/Storages/StorageTableFunction.h @@ -85,6 +85,13 @@ class StorageTableFunctionProxy final : public StorageProxy nested->drop(); } + void prepareForDrop(ContextPtr query_context) override + { + std::lock_guard lock{nested_mutex}; + if (nested) + nested->prepareForDrop(query_context); + } + void read( QueryPlan & query_plan, const Names & column_names, diff --git a/src/Storages/StorageTableProxy.h b/src/Storages/StorageTableProxy.h index e935de70e221..17ad3f1fcfea 100644 --- a/src/Storages/StorageTableProxy.h +++ b/src/Storages/StorageTableProxy.h @@ -79,6 +79,13 @@ class StorageTableProxy final : public StorageProxy nested->flushAndPrepareForShutdown(); } + void prepareForDrop(ContextPtr query_context) override + { + std::lock_guard lock{nested_mutex}; + if (nested) + nested->prepareForDrop(query_context); + } + void drop() override { std::lock_guard lock{nested_mutex}; diff --git a/tests/integration/test_database_glue/configs/disks.xml b/tests/integration/test_database_glue/configs/disks.xml new file mode 100644 index 000000000000..2e67563a8883 --- /dev/null +++ b/tests/integration/test_database_glue/configs/disks.xml @@ -0,0 +1,12 @@ + + + + + object_storage + local_blob_storage + plain + /var/lib/clickhouse/disks/local_blob_disk/ + + + + diff --git a/tests/integration/test_database_glue/test.py b/tests/integration/test_database_glue/test.py index d366f0f02570..0cfbc8e85ec9 100644 --- a/tests/integration/test_database_glue/test.py +++ b/tests/integration/test_database_glue/test.py @@ -91,7 +91,7 @@ def generate_decimal(precision=9, scale=2): ), ) -DEFAULT_CREATE_TABLE = "CREATE TABLE {}.`{}.{}`\\n(\\n `datetime` Nullable(DateTime64(6)),\\n `symbol` Nullable(String),\\n `bid` Nullable(Float64),\\n `ask` Nullable(Float64),\\n `details` Tuple(created_by Nullable(String)),\\n `map_string_decimal` Map(String, Nullable(Decimal(9, 2)))\\n)\\nENGINE = Iceberg(\\'http://minio1:9001/warehouse-glue/data/\\', \\'minio\\', \\'[HIDDEN]\\')\n" +DEFAULT_CREATE_TABLE = "CREATE TABLE {}.`{}.{}`\\n(\\n `datetime` Nullable(DateTime64(6)),\\n `symbol` Nullable(String),\\n `bid` Nullable(Float64),\\n `ask` Nullable(Float64),\\n `details` Tuple(created_by Nullable(String)),\\n `map_string_decimal` Map(String, Nullable(Decimal(9, 2)))\\n)\\nENGINE = Iceberg(\\'http://minio1:9001/warehouse-glue/data/\\', \\'minio\\', \\'[HIDDEN]\\')\\nPARTITION BY toRelativeDayNum(datetime)\\nORDER BY symbol\n" DEFAULT_PARTITION_SPEC = PartitionSpec( PartitionField( @@ -262,6 +262,7 @@ def started_cluster(): cluster.add_instance( "node1", main_configs=[ + "configs/disks.xml", "configs/query_log.xml", "configs/text_log.xml", ], @@ -689,7 +690,7 @@ def test_timestamps(started_cluster): df = pa.Table.from_pylist(data) table.append(df) - assert node.query(f"SHOW CREATE TABLE {CATALOG_NAME}.`{root_namespace}.{table_name}`") == f"CREATE TABLE {CATALOG_NAME}.`{root_namespace}.{table_name}`\\n(\\n `timestamp` Nullable(DateTime64(6)),\\n `timestamptz` Nullable(DateTime64(6, \\'UTC\\'))\\n)\\nENGINE = Iceberg(\\'http://minio1:9001/warehouse-glue/data/\\', \\'minio\\', \\'[HIDDEN]\\')\n" + assert node.query(f"SHOW CREATE TABLE {CATALOG_NAME}.`{root_namespace}.{table_name}`") == f"CREATE TABLE {CATALOG_NAME}.`{root_namespace}.{table_name}`\\n(\\n `timestamp` Nullable(DateTime64(6)),\\n `timestamptz` Nullable(DateTime64(6, \\'UTC\\'))\\n)\\nENGINE = Iceberg(\\'http://minio1:9001/warehouse-glue/data/\\', \\'minio\\', \\'[HIDDEN]\\')\\nPARTITION BY toRelativeDayNum(timestamp)\\nORDER BY timestamptz\n" assert node.query(f"SELECT * FROM {CATALOG_NAME}.`{root_namespace}.{table_name}`") == "2024-01-01 12:00:00.000000\t2024-01-01 12:00:00.000000\n" @@ -773,6 +774,68 @@ def test_schema_evolution_show_create_and_drop(started_cluster): assert result == "Alice\t\\N\t\\N\nBob\t42\thello\n" +def test_native_create_gzip_metadata(started_cluster): + node = started_cluster.instances["node1"] + + test_ref = f"test_native_create_gzip_{uuid.uuid4()}" + table_name = f"{test_ref}_table" + root_namespace = f"{test_ref}_namespace" + + create_clickhouse_glue_database(started_cluster, node, CATALOG_NAME) + + node.query( + f"CREATE TABLE {CATALOG_NAME}.`{root_namespace}.{table_name}` (x String)", + settings={ + "allow_experimental_database_glue_catalog": 1, + "allow_database_glue_catalog": 1, + "write_full_path_in_iceberg_metadata": 1, + "iceberg_metadata_compression_method": "gzip", + }, + ) + + glue_client = boto3.client( + "glue", region_name="us-east-1", endpoint_url=get_glue_local_url(started_cluster) + ) + table_info = glue_client.get_table(DatabaseName=root_namespace, Name=table_name)["Table"] + metadata_location = table_info["Parameters"]["metadata_location"] + assert metadata_location.endswith(".gzip.metadata.json"), metadata_location + + assert metadata_location.startswith("s3://"), metadata_location + bucket, _, key = metadata_location[len("s3://") :].partition("/") + metadata_bytes = started_cluster.minio_client.get_object(bucket, key).read() + assert metadata_bytes[:2] == b"\x1f\x8b", metadata_bytes[:16] + + create_clickhouse_glue_database(started_cluster, node, CATALOG_NAME) + assert node.query(f"SELECT count() FROM {CATALOG_NAME}.`{root_namespace}.{table_name}`") == "0\n" + + +def test_create_table_engine_backend_mismatch_rejected(started_cluster): + node = started_cluster.instances["node1"] + + test_ref = f"test_engine_backend_mismatch_{uuid.uuid4()}" + root_namespace = f"{test_ref}_namespace" + + create_clickhouse_glue_database(started_cluster, node, CATALOG_NAME) + + for engine in [ + "IcebergAzure('http://acc.blob.core.windows.net/cont/tbl/', 'acc', 'key')", + "IcebergLocal('/var/lib/clickhouse/user_files/tbl/')", + "IcebergHDFS('hdfs://namenode:9000/tbl/')", + "Iceberg('/var/lib/clickhouse/user_files/tbl/') SETTINGS disk = 'local_blob_disk'", + ]: + error = node.query_and_get_error( + f"CREATE TABLE {CATALOG_NAME}.`{root_namespace}.mismatch` (x String) ENGINE = {engine}", + settings={ + "allow_experimental_database_glue_catalog": 1, + "allow_database_glue_catalog": 1, + }, + ) + assert ( + "would be reopened with the catalog's storage backend and become unreadable" in error + ), error + assert "stores tables on S3" in error, error + + def test_schema_evolution(started_cluster): node = started_cluster.instances["node1"] @@ -856,6 +919,12 @@ def test_drop_table(started_cluster): assert len(catalog.list_tables(root_namespace)) == 1 assert node.query(f"SELECT * FROM {CATALOG_NAME}.`{root_namespace}.{table_name}`") == "" + error = node.query_and_get_error( + f"DROP TABLE {CATALOG_NAME}.`{root_namespace}.{table_name}` SETTINGS data_lake_delete_data_on_drop = 1" + ) + assert "not supported for the Glue catalog" in error + assert len(catalog.list_tables(root_namespace)) == 1 + drop_clickhouse_glue_table(node, root_namespace, table_name) assert len(catalog.list_tables(root_namespace)) == 0 diff --git a/tests/integration/test_database_iceberg/test.py b/tests/integration/test_database_iceberg/test.py index 7459389e5604..16059c2f8b62 100644 --- a/tests/integration/test_database_iceberg/test.py +++ b/tests/integration/test_database_iceberg/test.py @@ -11,10 +11,11 @@ import pytest import requests import pytz +from minio import Minio from pyiceberg.catalog import load_catalog from pyiceberg.partitioning import PartitionField, PartitionSpec, UNPARTITIONED_PARTITION_SPEC from pyiceberg.schema import Schema -from pyiceberg.table.sorting import SortField, SortOrder +from pyiceberg.table.sorting import NullOrder, SortField, SortOrder from pyiceberg.transforms import DayTransform, IdentityTransform from pyiceberg.types import ( DoubleType, @@ -60,7 +61,7 @@ ), ) -DEFAULT_CREATE_TABLE = "CREATE TABLE {}.`{}.{}`\\n(\\n `datetime` Nullable(DateTime64(6)),\\n `symbol` Nullable(String),\\n `bid` Nullable(Float64),\\n `ask` Nullable(Float64),\\n `details` Tuple(created_by Nullable(String))\\n)\\nENGINE = Iceberg(\\'http://minio1:9001/warehouse-rest/data/\\', \\'minio\\', \\'[HIDDEN]\\')\n" +DEFAULT_CREATE_TABLE = "CREATE TABLE {}.`{}.{}`\\n(\\n `datetime` Nullable(DateTime64(6)),\\n `symbol` Nullable(String),\\n `bid` Nullable(Float64),\\n `ask` Nullable(Float64),\\n `details` Tuple(created_by Nullable(String))\\n)\\nENGINE = Iceberg(\\'http://minio1:9001/warehouse-rest/data/\\', \\'minio\\', \\'[HIDDEN]\\')\\nPARTITION BY toRelativeDayNum(datetime)\\nORDER BY symbol\n" DEFAULT_PARTITION_SPEC = PartitionSpec( PartitionField( @@ -196,6 +197,7 @@ def started_cluster(): user_configs=[], stay_alive=True, with_iceberg_catalog=True, + with_zookeeper=True, ) cluster.add_instance( @@ -747,7 +749,7 @@ def test_timestamps(started_cluster): df = pa.Table.from_pylist(data) table.append(df) - assert node.query(f"SHOW CREATE TABLE {CATALOG_NAME}.`{root_namespace}.{table_name}`") == f"CREATE TABLE {CATALOG_NAME}.`{root_namespace}.{table_name}`\\n(\\n `timestamp` Nullable(DateTime64(6)),\\n `timestamptz` Nullable(DateTime64(6, \\'UTC\\'))\\n)\\nENGINE = Iceberg(\\'http://minio1:9001/warehouse-rest/data/\\', \\'minio\\', \\'[HIDDEN]\\')\n" + assert node.query(f"SHOW CREATE TABLE {CATALOG_NAME}.`{root_namespace}.{table_name}`") == f"CREATE TABLE {CATALOG_NAME}.`{root_namespace}.{table_name}`\\n(\\n `timestamp` Nullable(DateTime64(6)),\\n `timestamptz` Nullable(DateTime64(6, \\'UTC\\'))\\n)\\nENGINE = Iceberg(\\'http://minio1:9001/warehouse-rest/data/\\', \\'minio\\', \\'[HIDDEN]\\')\\nPARTITION BY toRelativeDayNum(timestamp)\\nORDER BY timestamptz\n" assert node.query(f"SELECT * FROM {CATALOG_NAME}.`{root_namespace}.{table_name}`") == "2024-01-01 12:00:00.000000\t2024-01-01 12:00:00.000000\n" # Berlin - UTC+1 at winter @@ -789,8 +791,8 @@ def test_timestamps(started_cluster): SETTINGS iceberg_timezone_for_timestamptz='Foo/Bar' """) - assert node.query(f"SHOW CREATE TABLE {CATALOG_NAME}.`{root_namespace}.{table_name}` SETTINGS iceberg_timezone_for_timestamptz='UTC'") == f"CREATE TABLE {CATALOG_NAME}.`{root_namespace}.{table_name}`\\n(\\n `timestamp` Nullable(DateTime64(6)),\\n `timestamptz` Nullable(DateTime64(6, \\'UTC\\'))\\n)\\nENGINE = Iceberg(\\'http://minio1:9001/warehouse-rest/data/\\', \\'minio\\', \\'[HIDDEN]\\')\n" - assert node.query(f"SHOW CREATE TABLE {CATALOG_NAME}.`{root_namespace}.{table_name}` SETTINGS iceberg_timezone_for_timestamptz='Europe/Berlin'") == f"CREATE TABLE {CATALOG_NAME}.`{root_namespace}.{table_name}`\\n(\\n `timestamp` Nullable(DateTime64(6)),\\n `timestamptz` Nullable(DateTime64(6, \\'Europe/Berlin\\'))\\n)\\nENGINE = Iceberg(\\'http://minio1:9001/warehouse-rest/data/\\', \\'minio\\', \\'[HIDDEN]\\')\n" + assert node.query(f"SHOW CREATE TABLE {CATALOG_NAME}.`{root_namespace}.{table_name}` SETTINGS iceberg_timezone_for_timestamptz='UTC'") == f"CREATE TABLE {CATALOG_NAME}.`{root_namespace}.{table_name}`\\n(\\n `timestamp` Nullable(DateTime64(6)),\\n `timestamptz` Nullable(DateTime64(6, \\'UTC\\'))\\n)\\nENGINE = Iceberg(\\'http://minio1:9001/warehouse-rest/data/\\', \\'minio\\', \\'[HIDDEN]\\')\\nPARTITION BY toRelativeDayNum(timestamp)\\nORDER BY timestamptz\n" + assert node.query(f"SHOW CREATE TABLE {CATALOG_NAME}.`{root_namespace}.{table_name}` SETTINGS iceberg_timezone_for_timestamptz='Europe/Berlin'") == f"CREATE TABLE {CATALOG_NAME}.`{root_namespace}.{table_name}`\\n(\\n `timestamp` Nullable(DateTime64(6)),\\n `timestamptz` Nullable(DateTime64(6, \\'Europe/Berlin\\'))\\n)\\nENGINE = Iceberg(\\'http://minio1:9001/warehouse-rest/data/\\', \\'minio\\', \\'[HIDDEN]\\')\\nPARTITION BY toRelativeDayNum(timestamp)\\nORDER BY timestamptz\n" assert node.query(f"SELECT timezoneOf(timestamptz) FROM {CATALOG_NAME}.`{root_namespace}.{table_name}` LIMIT 1") == "UTC\n" assert node.query(f"SELECT timezoneOf(timestamptz) FROM {CATALOG_NAME}.`{root_namespace}.{table_name}` LIMIT 1 SETTINGS iceberg_timezone_for_timestamptz='UTC'") == "UTC\n" @@ -1064,6 +1066,57 @@ def test_cluster_select(started_cluster): assert node2.query(f"SELECT * FROM {CATALOG_NAME}.`{root_namespace}.{table_name}`", settings={"parallel_replicas_for_cluster_engines": 1, "enable_parallel_replicas": 2, "cluster_for_parallel_replicas": "cluster_simple"}) == 'pablo\n' +def test_on_cluster_ddl_rejected_for_datalake_catalog(started_cluster): + node1 = started_cluster.instances["node1"] + node2 = started_cluster.instances["node2"] + + test_ref = f"test_on_cluster_ban_{uuid.uuid4().hex}" + table_name = f"{test_ref}_table" + namespace = f"{test_ref}_namespace" + qualified = f"{CATALOG_NAME}.`{namespace}.{table_name}`" + engine = ( + f"ENGINE = IcebergS3('http://minio1:9001/warehouse-rest/{table_name}/', " + f"'{minio_access_key}', '{minio_secret_key}')" + ) + ddl_settings = { + "allow_experimental_database_iceberg": 1, + "allow_database_iceberg": 1, + "write_full_path_in_iceberg_metadata": 1, + "distributed_ddl_output_mode": "throw", + } + + create_clickhouse_iceberg_database(started_cluster, node1, CATALOG_NAME) + create_clickhouse_iceberg_database(started_cluster, node2, CATALOG_NAME) + err = node1.query_and_get_error( + f"CREATE TABLE {qualified} ON CLUSTER cluster_simple (x String) {engine}", + settings=ddl_settings, + ) + assert "ON CLUSTER is not supported for DataLakeCatalog" in err, err + + err = node1.query_and_get_error( + f"DROP DATABASE {CATALOG_NAME} ON CLUSTER cluster_simple", + settings={"distributed_ddl_output_mode": "throw"}, + ) + assert "ON CLUSTER is not supported for DataLakeCatalog" in err, err + assert node1.query(f"EXISTS DATABASE {CATALOG_NAME}") == "1\n" + assert node2.query(f"EXISTS DATABASE {CATALOG_NAME}") == "1\n" + + node2.query(f"DROP DATABASE IF EXISTS {CATALOG_NAME}") + try: + node2.query_and_get_error( + f"CREATE TABLE {qualified} ON CLUSTER cluster_simple (x String) {engine}", + settings=ddl_settings, + ) + + catalog = load_catalog_impl(started_cluster) + existing_namespaces = {".".join(ns) for ns in catalog.list_namespaces()} + if namespace in existing_namespaces: + tables = {ident[-1] for ident in catalog.list_tables(namespace)} + assert table_name not in tables, f"table must not have been created in the shared catalog: {tables}" + finally: + create_clickhouse_iceberg_database(started_cluster, node2, CATALOG_NAME) + + def test_used_storages_in_query_log(started_cluster): node1 = started_cluster.instances["node1"] node2 = started_cluster.instances["node2"] @@ -1227,6 +1280,801 @@ def test_system_tables_with_nullptr_table(started_cluster): node.query(f"DROP DATABASE IF EXISTS {CATALOG_NAME}") + +def test_create_table_as(started_cluster): + node = started_cluster.instances["node1"] + namespace = f"ctas_{uuid.uuid4().hex}" + source = f"default.src_{namespace}" + catalog = load_catalog_impl(started_cluster) + settings = {"allow_database_iceberg": 1, "write_full_path_in_iceberg_metadata": 1} + create_clickhouse_iceberg_database( + started_cluster, + node, + CATALOG_NAME, + additional_settings={"default_base_location": "s3://warehouse-rest/data"}, + ) + node.query( + f"CREATE TABLE {source} (id Int64, name String, dt Date) " + "ENGINE = MergeTree PARTITION BY toYearNumSinceEpoch(dt) ORDER BY (id, name)" + ) + node.query( + f"CREATE MATERIALIZED VIEW {source}_mv ENGINE = MergeTree " + f"PARTITION BY toYearNumSinceEpoch(dt) ORDER BY (id, name) AS SELECT * FROM {source}" + ) + + for name, source_table, explicit_engine, keys in [ + ("inherited", source, False, ""), + ("explicit_engine", source, True, ""), + ("from_mv", f"{source}_mv", True, ""), + ("override", source, False, "PARTITION BY id ORDER BY name"), + ]: + target = f"{CATALOG_NAME}.`{namespace}.{name}`" + engine = ( + f"ENGINE = IcebergS3('http://minio1:9001/warehouse-rest/{namespace}/{name}/', " + f"'{minio_access_key}', '{minio_secret_key}')" + if explicit_engine + else "" + ) + node.query( + f"CREATE TABLE {target} AS {source_table} {engine} {keys}", + settings={ + **settings, + "datalake_ignore_unsupported_table_properties": int( + not explicit_engine + ), + }, + ) + table = catalog.load_table(f"{namespace}.{name}") + assert [field.name for field in table.schema().fields] == ["id", "name", "dt"] + assert [ + (field.source_id, str(field.transform)) for field in table.spec().fields + ] == ([(1, "identity")] if keys else [(3, "year")]) + assert [ + (field.source_id, str(field.transform)) + for field in table.sort_order().fields + ] == ([(2, "identity")] if keys else [(1, "identity"), (2, "identity")]) + node.query(f"DROP TABLE {target}") + + node.query(f"DROP TABLE {source}_mv") + node.query(f"DROP TABLE {source}") + + +@pytest.mark.parametrize("column_definition,explicit_engine", [ + ("name String CODEC(ZSTD)", False), + ("dt Date TTL dt + INTERVAL 1 DAY", True), +]) +def test_create_table_as_rejects_column_modifiers(started_cluster, column_definition, explicit_engine): + node = started_cluster.instances["node1"] + namespace = f"ctas_colmod_{uuid.uuid4().hex}" + src_table = "src_colmod" + target = f"{CATALOG_NAME}.`{namespace}.copied`" + create_clickhouse_iceberg_database( + started_cluster, + node, + CATALOG_NAME, + additional_settings={"default_base_location": "s3://warehouse-rest/data"}, + ) + engine = ( + f"ENGINE = IcebergS3('http://minio1:9001/warehouse-rest/{namespace}/copied/', " + f"'{minio_access_key}', '{minio_secret_key}')" if explicit_engine else "" + ) + ddl = f"CREATE TABLE {target} AS default.{src_table} {engine}" + settings = {"allow_database_iceberg": 1, "write_full_path_in_iceberg_metadata": 1} + node.query(f"DROP TABLE IF EXISTS default.{src_table}") + node.query( + f"CREATE TABLE default.{src_table} (id Int64, {column_definition}) " + "ENGINE = MergeTree ORDER BY id" + ) + error = node.query_and_get_error( + ddl, settings=settings, + ) + assert ( + "COMMENT, CODEC, TTL, STATISTICS, SETTINGS, and PRIMARY KEY are not supported" + in error + ) + + node.query( + ddl, + settings={ + **settings, + "datalake_ignore_unsupported_table_properties": 1, + }, + ) + table = load_catalog_impl(started_cluster).load_table(f"{namespace}.copied") + assert [field.name for field in table.schema().fields] == ["id", column_definition.split()[0]] + node.query(f"DROP TABLE {target}") + node.query(f"DROP TABLE default.{src_table}") + + +@pytest.mark.parametrize("source_property,explicit_engine", [("primary_key", False), ("index", True)]) +def test_create_table_as_rejects_source_storage_clauses( + started_cluster, source_property, explicit_engine +): + node = started_cluster.instances["node1"] + namespace = f"ctas_storage_{uuid.uuid4().hex}" + src_table = "src_storage" + create_clickhouse_iceberg_database( + started_cluster, + node, + CATALOG_NAME, + additional_settings={"default_base_location": "s3://warehouse-rest/data"}, + ) + node.query(f"DROP TABLE IF EXISTS default.{src_table}") + index = ( + ", INDEX idx name TYPE bloom_filter GRANULARITY 1" + if source_property == "index" + else "" + ) + primary_key = "PRIMARY KEY id" if source_property == "primary_key" else "" + node.query( + f"CREATE TABLE default.{src_table} (id UInt64, name String{index}) " + f"ENGINE = MergeTree {primary_key} ORDER BY (id, name)" + ) + engine = ( + f"ENGINE = IcebergS3('http://minio1:9001/warehouse-rest/{namespace}/copied/', " + f"'{minio_access_key}', '{minio_secret_key}')" + if explicit_engine + else "" + ) + target = f"{CATALOG_NAME}.`{namespace}.copied`" + ddl = f"CREATE TABLE {target} AS default.{src_table} {engine}" + settings = { + "allow_database_iceberg": 1, + "write_full_path_in_iceberg_metadata": 1, + } + error = node.query_and_get_error(ddl, settings=settings) + expected_error = ( + "has PRIMARY KEY, which a DataLakeCatalog table cannot represent" + if source_property == "primary_key" + else "does not support PRIMARY KEY, indices" + ) + assert expected_error in error + + node.query( + ddl, + settings={ + **settings, + "datalake_ignore_unsupported_table_properties": 1, + }, + ) + table = load_catalog_impl(started_cluster).load_table(f"{namespace}.copied") + assert [field.name for field in table.schema().fields] == ["id", "name"] + node.query(f"DROP TABLE {target}") + node.query(f"DROP TABLE default.{src_table}") + + +def test_create_table_explicit_columns(started_cluster): + node = started_cluster.instances["node1"] + + namespace = f"test_ctex_{uuid.uuid4().hex}.a.b" + catalog = load_catalog_impl(started_cluster) + + create_clickhouse_iceberg_database( + started_cluster, + node, + CATALOG_NAME, + additional_settings={"default_base_location": "s3://warehouse-rest/data"}, + ) + + node.query( + f""" + CREATE TABLE {CATALOG_NAME}.`{namespace}.explicit` + ( + id Int64, + name String, + value Float64 + ) + PARTITION BY id + ORDER BY name + SETTINGS allow_database_iceberg=1; + """ + ) + + tbl = catalog.load_table(f"{namespace}.explicit") + namespace_location = catalog.load_namespace_properties(namespace)["location"].rstrip("/") + assert tbl.location().rstrip("/") == f"{namespace_location}/explicit" + col_names = [f.name for f in tbl.schema().fields] + assert col_names == ["id", "name", "value"] + + iceberg_types = {f.name: str(f.field_type) for f in tbl.schema().fields} + assert iceberg_types["id"] == "long" + assert iceberg_types["name"] == "string" + assert iceberg_types["value"] == "double" + + node.query( + f"INSERT INTO {CATALOG_NAME}.`{namespace}.explicit` VALUES (1, 'a', 1.5);", + settings={ + "allow_insert_into_iceberg": 1, + "write_full_path_in_iceberg_metadata": 1, + }, + ) + assert ( + node.query( + f"SELECT id, name, value FROM {CATALOG_NAME}.`{namespace}.explicit`" + ) + == "1\ta\t1.5\n" + ) + + node.query( + f"DROP TABLE {CATALOG_NAME}.`{namespace}.explicit` SETTINGS allow_database_iceberg=1" + ) + + +def test_create_non_table_rejected(started_cluster): + node = started_cluster.instances["node1"] + + create_clickhouse_iceberg_database(started_cluster, node, CATALOG_NAME) + + for ddl in [ + f"CREATE VIEW {CATALOG_NAME}.`ns.v` AS SELECT 1", + f"CREATE MATERIALIZED VIEW {CATALOG_NAME}.`ns.mv` ENGINE = Memory AS SELECT 1", + f"ATTACH TABLE {CATALOG_NAME}.`ns.attached` (x Int32) ENGINE = Memory", + f"CREATE OR REPLACE TABLE {CATALOG_NAME}.`ns.replaced` (x Int32)", + ]: + err = node.query_and_get_error(ddl, settings={"allow_database_iceberg": 1}) + assert "supports only plain CREATE TABLE" in err + + node.query("DROP TABLE IF EXISTS default.src_clone") + node.query( + "CREATE TABLE default.src_clone (x Int32) ENGINE = MergeTree ORDER BY x" + ) + err = node.query_and_get_error( + f"CREATE TABLE {CATALOG_NAME}.`ns.cloned` CLONE AS default.src_clone", + settings={"allow_database_iceberg": 1}, + ) + assert "supports only plain CREATE TABLE" in err + node.query("DROP TABLE default.src_clone") + + err = node.query_and_get_error( + f"CREATE TABLE {CATALOG_NAME}.`ns.mem` (x Int32) ENGINE = Memory", + settings={"allow_database_iceberg": 1}, + ) + assert "stores Iceberg-family tables" in err + + +def test_create_table_unsupported_clauses(started_cluster): + node = started_cluster.instances["node1"] + + create_clickhouse_iceberg_database(started_cluster, node, CATALOG_NAME) + + base_ddl = f"CREATE TABLE {CATALOG_NAME}.`ns.unsupp` (id Int64, name String)" + for clause in [ + "PRIMARY KEY id ORDER BY id", + "ORDER BY id SAMPLE BY id", + "ORDER BY id TTL toDate('2099-01-01')", + "ORDER BY id SETTINGS index_granularity = 8192", + "ORDER BY id UNIQUE KEY id", + ]: + err = node.query_and_get_error( + f"{base_ddl} {clause}", + settings={"allow_database_iceberg": 1, "allow_experimental_unique_key": 1}, + ) + assert "supports only PARTITION BY and ORDER BY" in err + + err = node.query_and_get_error( + f"{base_ddl} ORDER BY id COMMENT 'tbl comment'", + settings={"allow_database_iceberg": 1}, + ) + assert "Table COMMENT is not supported" in err + + for table_element in [ + "INDEX idx_name name TYPE bloom_filter GRANULARITY 1", + "PROJECTION p (SELECT id ORDER BY name)", + "CONSTRAINT c CHECK id > 0", + ]: + err = node.query_and_get_error( + f"CREATE TABLE {CATALOG_NAME}.`ns.unsupp_elem` (id Int64, name String, {table_element}) ORDER BY id", + settings={"allow_database_iceberg": 1}, + ) + assert "does not support PRIMARY KEY, indices" in err + + # Column-level PRIMARY KEY is normalized into the storage-level clause, covered above. + for col_clause in [ + "(id Int64 COMMENT 'pk', name String)", + "(id Int64, name String CODEC(ZSTD))", + "(id Int64, dt Date TTL dt + INTERVAL 1 DAY)", + "(id Int64, name String SETTINGS (max_compress_block_size = 1))", + ]: + err = node.query_and_get_error( + f"CREATE TABLE {CATALOG_NAME}.`ns.unsupp_col` {col_clause} ORDER BY id", + settings={"allow_database_iceberg": 1}, + ) + assert "COMMENT, CODEC, TTL, STATISTICS, SETTINGS, and PRIMARY KEY are not supported" in err + + for col_clause in [ + "(id Int64, d Int64 DEFAULT 1)", + "(id Int64, d Int64 MATERIALIZED id + 1)", + ]: + err = node.query_and_get_error( + f"CREATE TABLE {CATALOG_NAME}.`ns.unsupp_def` {col_clause} ORDER BY id", + settings={"allow_database_iceberg": 1}, + ) + assert "is not yet supported" in err + + +def test_create_table_with_engine_unsupported_clauses(started_cluster): + node = started_cluster.instances["node1"] + + create_clickhouse_iceberg_database(started_cluster, node, CATALOG_NAME) + + err = node.query_and_get_error( + f"CREATE TABLE {CATALOG_NAME}.`ns.engine_unsupp` (id Int64) " + f"ENGINE = IcebergS3('http://minio1:9001/warehouse-rest/engine_unsupp/', " + f"'{minio_access_key}', '{minio_secret_key}') PRIMARY KEY id", + settings={"allow_database_iceberg": 1}, + ) + assert "PRIMARY KEY, SAMPLE BY, TTL, and UNIQUE KEY are not supported" in err + + +@pytest.mark.parametrize("explicit_engine", [False, True]) +def test_create_table_ignore_unsupported_properties(started_cluster, explicit_engine): + node = started_cluster.instances["node1"] + namespace = f"ignore_properties_{uuid.uuid4().hex}" + target = f"{CATALOG_NAME}.`{namespace}.created`" + create_clickhouse_iceberg_database( + started_cluster, node, CATALOG_NAME, + additional_settings={"default_base_location": "s3://warehouse-rest/data"}, + ) + engine = ( + f"ENGINE = IcebergS3('http://minio1:9001/warehouse-rest/{namespace}/created/', " + f"'{minio_access_key}', '{minio_secret_key}')" if explicit_engine else "" + ) + ddl = ( + f"CREATE TABLE {target} (id Int64, name String CODEC(ZSTD), dt Date, " + "INDEX idx name TYPE bloom_filter GRANULARITY 1, CONSTRAINT positive CHECK id > 0) " + f"{engine} PARTITION BY id ORDER BY name PRIMARY KEY name TTL dt + INTERVAL 1 DAY " + "COMMENT 'ignored comment'" + ) + settings = {"allow_database_iceberg": 1, "write_full_path_in_iceberg_metadata": 1} + assert "PRIMARY KEY" in node.query_and_get_error(ddl, settings=settings) + settings["datalake_ignore_unsupported_table_properties"] = 1 + node.query(ddl, settings=settings) + table = load_catalog_impl(started_cluster).load_table(f"{namespace}.created") + assert [field.name for field in table.schema().fields] == ["id", "name", "dt"] + assert [(field.source_id, str(field.transform)) for field in table.spec().fields] == [(1, "identity")] + assert [(field.source_id, str(field.transform)) for field in table.sort_order().fields] == [(2, "identity")] + node.query(f"DROP TABLE {target}") + + +def test_create_table_unsupported_key_expressions(started_cluster): + node = started_cluster.instances["node1"] + namespace = f"ignore_keys_{uuid.uuid4().hex}" + create_clickhouse_iceberg_database( + started_cluster, node, CATALOG_NAME, + additional_settings={"default_base_location": "s3://warehouse-rest/data"}, + ) + for suffix, keys, partition_count, sort_count in [ + ("partition", "PARTITION BY (id, id + 1) ORDER BY id", 0, 1), + ("sort", "PARTITION BY id ORDER BY (id, id + 1)", 1, 0), + ]: + target = f"{CATALOG_NAME}.`{namespace}.{suffix}`" + ddl = f"CREATE TABLE {target} (id Int64) {keys}" + settings = {"allow_database_iceberg": 1} + assert "Cannot represent" in node.query_and_get_error(ddl, settings=settings) + settings["datalake_ignore_unsupported_table_properties"] = 1 + node.query(ddl, settings=settings) + table = load_catalog_impl(started_cluster).load_table(f"{namespace}.{suffix}") + assert len(table.spec().fields) == partition_count + assert len(table.sort_order().fields) == sort_count + node.query(f"DROP TABLE {target}") + + for key in [ + "PARTITION BY missing + 1", + "ORDER BY nonexistent_function(id)", + "PRIMARY KEY missing", + "TTL missing + INTERVAL 1 DAY", + ]: + error = node.query_and_get_error( + f"CREATE TABLE {CATALOG_NAME}.`{namespace}.invalid` (id Int64) {key}", + settings={"allow_database_iceberg": 1, "datalake_ignore_unsupported_table_properties": 1}, + ) + assert "UNKNOWN_IDENTIFIER" in error or "UNKNOWN_FUNCTION" in error + + +def test_create_table_invalid_partition_transforms(started_cluster): + node = started_cluster.instances["node1"] + + namespace = f"test_invalid_part_{uuid.uuid4().hex}" + catalog = load_catalog_impl(started_cluster) + catalog.create_namespace(namespace) + + create_clickhouse_iceberg_database( + started_cluster, + node, + CATALOG_NAME, + additional_settings={"default_base_location": "s3://warehouse-rest/data"}, + ) + + for i, transform in enumerate( + ["icebergBucket(0, id)", "icebergTruncate(0, id)"] + ): + tbl = f"bad_{i}" + err = node.query_and_get_error( + f"CREATE TABLE {CATALOG_NAME}.`{namespace}.{tbl}` (id Int64) " + f"PARTITION BY {transform}", + settings={"allow_database_iceberg": 1, "datalake_ignore_unsupported_table_properties": 1}, + ) + assert "positive" in err, err + assert tbl not in [t[1] for t in catalog.list_tables(namespace)] + + + +def test_create_table_with_engine_namespace_location(started_cluster): + node = started_cluster.instances["node1"] + + namespace = f"test_ns_engine_location_{uuid.uuid4().hex[:8]}" + table_dir = f"engine_ns_location_{uuid.uuid4().hex[:8]}" + catalog = load_catalog_impl(started_cluster) + + create_clickhouse_iceberg_database(started_cluster, node, CATALOG_NAME) + + node.query( + f"CREATE TABLE {CATALOG_NAME}.`{namespace}.first` (id Int64) " + f"ENGINE = IcebergS3('http://minio1:9001/warehouse-rest/{table_dir}/', " + f"'{minio_access_key}', '{minio_secret_key}')", + settings={ + "allow_database_iceberg": 1, + "write_full_path_in_iceberg_metadata": 1, + }, + ) + + table_location = catalog.load_table(f"{namespace}.first").location().rstrip("/") + assert table_location.endswith(table_dir), table_location + + # The engine path here is `//`, which does not follow `//
`, + # so there is no namespace base to derive. The table's own directory must not become the namespace + # default location, or later tables in this namespace would land inside it. + ns_location = catalog.load_namespace_properties(namespace).get("location") + if ns_location is not None: + ns_location = ns_location.rstrip("/") + assert ns_location != table_location, (ns_location, table_location) + assert not ns_location.startswith(table_location + "/"), (ns_location, table_location) + + node.query( + f"DROP TABLE {CATALOG_NAME}.`{namespace}.first`", + settings={"allow_database_iceberg": 1}, + ) + + # When the engine path does follow `//
`, the namespace base is unambiguous and + # is registered as the namespace default location, so later tables land next to this one, not inside it. + base_dir = f"engine_ns_base_{uuid.uuid4().hex[:8]}" + nested_namespace = f"test_ns_engine_derived_{uuid.uuid4().hex[:8]}" + node.query( + f"CREATE TABLE {CATALOG_NAME}.`{nested_namespace}.second` (id Int64) " + f"ENGINE = IcebergS3('http://minio1:9001/warehouse-rest/{base_dir}/{nested_namespace}/second/', " + f"'{minio_access_key}', '{minio_secret_key}')", + settings={ + "allow_database_iceberg": 1, + "write_full_path_in_iceberg_metadata": 1, + }, + ) + + ns_location = catalog.load_namespace_properties(nested_namespace).get("location") + assert ns_location is not None, "namespace is missing its location property" + assert ns_location.rstrip("/") == f"s3://warehouse-rest/{base_dir}/{nested_namespace}", ns_location + + node.query( + f"DROP TABLE {CATALOG_NAME}.`{nested_namespace}.second`", + settings={"allow_database_iceberg": 1}, + ) + + +def test_create_table_with_engine_arguments_not_representable(started_cluster): + # Per-table endpoints and credentials cannot survive reopening through the catalog. + node = started_cluster.instances["node1"] + namespace = f"engine_args_{uuid.uuid4().hex}" + create_clickhouse_iceberg_database(started_cluster, node, CATALOG_NAME) + + for arguments, expected in [ + ( + "'http://minio1:9001/warehouse-rest/engine_args/', 'other_key', 'other_secret'", + "per-table storage credentials cannot be preserved", + ), + ( + f"'http://other-minio:9000/other-bucket/engine_args/', '{minio_access_key}', '{minio_secret_key}'", + "is outside the `storage_endpoint`", + ), + ( + f"'http://minio1:9001/warehouse-rest/engine_args/', '{minio_access_key}', '{minio_secret_key}', 'Parquet'", + "per-table storage credentials cannot be preserved", + ), + ]: + error = node.query_and_get_error( + f"CREATE TABLE {CATALOG_NAME}.`{namespace}.rejected` (id Int64) ENGINE = IcebergS3({arguments})", + settings={ + "allow_database_iceberg": 1, + "datalake_ignore_unsupported_table_properties": 1, + "write_full_path_in_iceberg_metadata": 1, + }, + ) + assert expected in error + + catalog = load_catalog_impl(started_cluster) + assert namespace not in {".".join(ns) for ns in catalog.list_namespaces()} + + +def test_show_create_table_round_trip_partition_and_sort_order(started_cluster): + node = started_cluster.instances["node1"] + + suffix = uuid.uuid4().hex[:8] + namespace = f"test_show_create_keys_{suffix}" + source_table_name = f"source_{suffix}" + target_table_name = f"target_{suffix}" + settings = { + "allow_database_iceberg": 1, + "write_full_path_in_iceberg_metadata": 1, + } + + create_clickhouse_iceberg_database( + started_cluster, + node, + CATALOG_NAME, + additional_settings={"default_base_location": "s3://warehouse-rest/data"}, + ) + + source_table = f"{CATALOG_NAME}.`{namespace}.{source_table_name}`" + target_table = f"{CATALOG_NAME}.`{namespace}.{target_table_name}`" + node.query( + f""" + CREATE TABLE {source_table} (id UInt64, dt Date, name String) + PARTITION BY (toYearNumSinceEpoch(dt), icebergBucket(8, id)) + ORDER BY (id, name DESC) + """, + settings=settings, + ) + + create_query = node.query( + f"SHOW CREATE TABLE {source_table} FORMAT TSVRaw", + settings={"format_display_secrets_in_show_and_select": 1}, + ) + assert "ENGINE = Iceberg(" in create_query + assert "PARTITION BY (toYearNumSinceEpoch(dt), icebergBucket(8, id))" in create_query + assert "ORDER BY tuple(id, name DESC)" in create_query + + create_query = create_query.replace( + f"`{namespace}.{source_table_name}`", f"`{namespace}.{target_table_name}`" + ).replace(f"/{source_table_name}/", f"/{target_table_name}/") + node.query(create_query, settings=settings) + + catalog = load_catalog_impl(started_cluster) + source = catalog.load_table(f"{namespace}.{source_table_name}") + target = catalog.load_table(f"{namespace}.{target_table_name}") + + def partition_fields(tbl): + return [(f.source_id, str(f.transform)) for f in tbl.spec().fields] + + def sort_fields(tbl): + return [ + (f.source_id, str(f.transform), str(f.direction)) + for f in tbl.sort_order().fields + ] + + assert partition_fields(target) == partition_fields(source) + assert sort_fields(target) == sort_fields(source) + + node.query(f"DROP TABLE {source_table}", settings=settings) + node.query(f"DROP TABLE {target_table}", settings=settings) + + +@pytest.mark.parametrize("unsupported_key", ["partition", "sort"]) +def test_unsupported_catalog_keys(started_cluster, unsupported_key): + node = started_cluster.instances["node1"] + + suffix = uuid.uuid4().hex[:8] + namespace = f"test_show_create_unrepresentable_keys_{suffix}" + table_name = f"table_{suffix}" + + catalog = load_catalog_impl(started_cluster) + catalog.create_namespace(namespace) + schema = Schema( + NestedField(field_id=1, name="symbol", field_type=StringType(), required=False), + NestedField( + field_id=2, + name="details", + field_type=StructType( + NestedField( + field_id=3, + name="created_by", + field_type=StringType(), + required=False, + ), + ), + required=False, + ), + ) + create_table( + catalog, + namespace, + table_name, + schema=schema, + partition_spec=PartitionSpec( + PartitionField( + source_id=3 if unsupported_key == "partition" else 1, + field_id=1000, + transform=IdentityTransform(), + name="created_by" if unsupported_key == "partition" else "symbol", + ) + ), + sort_order=SortOrder( + SortField( + source_id=1, + transform=IdentityTransform(), + null_order=( + NullOrder.NULLS_LAST + if unsupported_key == "sort" + else NullOrder.NULLS_FIRST + ), + ) + ), + ) + + create_clickhouse_iceberg_database( + started_cluster, + node, + CATALOG_NAME, + additional_settings={"default_base_location": "s3://warehouse-rest/data"}, + ) + source = f"{CATALOG_NAME}.`{namespace}.{table_name}`" + unsupported_clause = ( + "PARTITION BY" if unsupported_key == "partition" else "ORDER BY" + ) + supported_clause = "ORDER BY" if unsupported_key == "partition" else "PARTITION BY" + settings = {"allow_database_iceberg": 1, "write_full_path_in_iceberg_metadata": 1} + + # Reading existing tables does not require a representable `CREATE TABLE` definition. + assert node.query(f"SELECT * FROM {source}") == "" + error = node.query_and_get_error(f"SHOW CREATE TABLE {source}", settings=settings) + assert f"Cannot represent {unsupported_clause}" in error + create_query = node.query( + f"SHOW CREATE TABLE {source} FORMAT TSVRaw", + settings={**settings, "datalake_ignore_unsupported_table_properties": 1}, + ) + assert unsupported_clause not in create_query + assert supported_clause in create_query + + engine = ( + f"ENGINE = IcebergS3('http://minio1:9001/warehouse-rest/{namespace}/copied/', " + f"'{minio_access_key}', '{minio_secret_key}')" + if unsupported_key == "sort" + else "" + ) + target = f"{CATALOG_NAME}.`{namespace}.copied`" + ddl = f"CREATE TABLE {target} AS {source} {engine}" + error = node.query_and_get_error(ddl, settings=settings) + assert f"Cannot represent {unsupported_clause}" in error + node.query( + ddl, settings={**settings, "datalake_ignore_unsupported_table_properties": 1} + ) + copied = catalog.load_table(f"{namespace}.copied") + assert [field.name for field in copied.schema().fields] == ["symbol", "details"] + assert len(copied.spec().fields) == (0 if unsupported_key == "partition" else 1) + assert len(copied.sort_order().fields) == (0 if unsupported_key == "sort" else 1) + node.query(f"DROP TABLE {target}") + node.query(f"DROP TABLE {source}") + + +def test_create_table_with_engine_virtual_hosted_location_rejected(started_cluster): + node = started_cluster.instances["node1"] + + namespace = f"test_ns_engine_vhost_{uuid.uuid4().hex[:8]}" + create_clickhouse_iceberg_database( + started_cluster, + node, + CATALOG_NAME, + { + "storage_endpoint": "https://s3.company.test/root", + "storage_uri_style": "virtual_hosted", + }, + ) + + error = node.query_and_get_error( + f"CREATE TABLE {CATALOG_NAME}.`{namespace}.other_endpoint` (id Int64) " + f"ENGINE = IcebergS3('http://other-minio:9000/other-bucket/p/', " + f"'{minio_access_key}', '{minio_secret_key}')", + settings={ + "allow_database_iceberg": 1, + "write_full_path_in_iceberg_metadata": 1, + }, + ) + assert "explicit table `ENGINE` is not supported with `storage_uri_style = 'virtual_hosted'`" in error, error + + catalog = load_catalog_impl(started_cluster) + assert namespace not in {".".join(ns) for ns in catalog.list_namespaces()} + + +def test_drop_table_purge(started_cluster): + node = started_cluster.instances["node1"] + + namespace = "test_drop_purge_ns" + catalog = load_catalog_impl(started_cluster) + minio_client = started_cluster.minio_client + + create_clickhouse_iceberg_database( + started_cluster, + node, + CATALOG_NAME, + additional_settings={"default_base_location": "s3://warehouse-rest/data"}, + ) + + for table in ["to_keep", "to_purge"]: + node.query( + f"DROP TABLE IF EXISTS {CATALOG_NAME}.`{namespace}.{table}` SETTINGS allow_database_iceberg=1" + ) + node.query( + f""" + CREATE TABLE {CATALOG_NAME}.`{namespace}.{table}` + ( + id Int64 + ) + SETTINGS allow_database_iceberg=1; + """ + ) + + node.query(f"DROP TABLE IF EXISTS {CATALOG_NAME}.`{namespace}.missing`") + + def table_prefix(table): + location = catalog.load_table(f"{namespace}.{table}").location() + assert location.startswith("s3://warehouse-rest/") + prefix = location[len("s3://warehouse-rest/"):].rstrip("/") + "/" + assert list( + minio_client.list_objects("warehouse-rest", prefix=prefix, recursive=True) + ), f"Expected metadata under {prefix} before drop" + return prefix + + keep_prefix = table_prefix("to_keep") + purge_prefix = table_prefix("to_purge") + + node.query( + f"DROP TABLE {CATALOG_NAME}.`{namespace}.to_keep` SETTINGS allow_database_iceberg=1" + ) + node.query( + f"DROP TABLE {CATALOG_NAME}.`{namespace}.to_purge` SETTINGS allow_database_iceberg=1, data_lake_delete_data_on_drop=1" + ) + + table_names = [t[1] for t in catalog.list_tables(namespace)] + assert "to_keep" not in table_names + assert "to_purge" not in table_names + + assert list( + minio_client.list_objects("warehouse-rest", prefix=keep_prefix, recursive=True) + ), f"Expected objects under {keep_prefix} to survive a drop without purge" + + remaining = [ + o.object_name + for o in minio_client.list_objects("warehouse-rest", prefix=purge_prefix, recursive=True) + ] + assert not remaining, f"Expected purge to remove objects under {purge_prefix}, found: {remaining}" + + +def test_create_if_not_exists_with_engine_over_leftover_metadata(started_cluster): + node = started_cluster.instances["node1"] + namespace = f"test_ns_leftover_{uuid.uuid4().hex[:8]}" + engine = ( + f"IcebergS3('http://minio1:9001/warehouse-rest/{namespace}/t/', " + f"'{minio_access_key}', '{minio_secret_key}')" + ) + settings = { + "allow_database_iceberg": 1, + "allow_insert_into_iceberg": 1, + "write_full_path_in_iceberg_metadata": 1, + } + + create_clickhouse_iceberg_database(started_cluster, node, CATALOG_NAME) + node.query( + f"CREATE TABLE {CATALOG_NAME}.`{namespace}.t` (id Int64) ENGINE = {engine}", + settings=settings, + ) + node.query(f"DROP TABLE {CATALOG_NAME}.`{namespace}.t`", settings=settings) + + # Leftover metadata makes `IF NOT EXISTS` a no-op, so `AS SELECT` must not insert. + node.query( + f"CREATE TABLE IF NOT EXISTS {CATALOG_NAME}.`{namespace}.t` (id Int64) " + f"ENGINE = {engine} AS SELECT 1 AS id", + settings=settings, + ) + assert node.query(f"EXISTS TABLE {CATALOG_NAME}.`{namespace}.t`", settings=settings) == "0\n" + + def test_delete_on_lazy_initialized_table(started_cluster): """ Regression test for https://github.com/ClickHouse/ClickHouse/issues/96806. diff --git a/tests/integration/test_storage_iceberg_no_spark/test_drop_delete_data_on_drop.py b/tests/integration/test_storage_iceberg_no_spark/test_drop_delete_data_on_drop.py new file mode 100644 index 000000000000..0c176c4fb5a0 --- /dev/null +++ b/tests/integration/test_storage_iceberg_no_spark/test_drop_delete_data_on_drop.py @@ -0,0 +1,62 @@ +import pytest + +from helpers.iceberg_utils import create_iceberg_table, get_uuid_str + + +def _table_path(table_name): + return f"var/lib/clickhouse/user_files/iceberg_data/default/{table_name}" + + +def _files_left(cluster, storage_type, table_name): + if storage_type == "local": + return int( + cluster.instances["node1"] + .exec_in_container( + [ + "bash", + "-c", + f"find /{_table_path(table_name)} -type f 2>/dev/null | wc -l", + ] + ) + .strip() + ) + + return len( + list( + cluster.minio_client.list_objects( + cluster.minio_bucket, f"{_table_path(table_name)}/", recursive=True + ) + ) + ) + + +@pytest.mark.parametrize("delete_data_on_drop", [0, 1]) +@pytest.mark.parametrize("storage_type", ["local", "s3"]) +def test_drop_honours_query_level_delete_data_on_drop( + started_cluster_iceberg_no_spark, storage_type, delete_data_on_drop +): + instance = started_cluster_iceberg_no_spark.instances["node1"] + table_name = f"test_delete_data_on_drop_{storage_type}_{delete_data_on_drop}_{get_uuid_str()}" + + create_iceberg_table( + storage_type, + instance, + table_name, + started_cluster_iceberg_no_spark, + "(x String, y Int64)", + ) + instance.query(f"INSERT INTO {table_name} VALUES ('123', 1)") + assert instance.query(f"SELECT * FROM {table_name} ORDER BY ALL") == "123\t1\n" + assert _files_left(started_cluster_iceberg_no_spark, storage_type, table_name) > 0 + + instance.query( + f"DROP TABLE {table_name} SYNC", + settings={"data_lake_delete_data_on_drop": delete_data_on_drop}, + ) + + if delete_data_on_drop: + assert ( + _files_left(started_cluster_iceberg_no_spark, storage_type, table_name) == 0 + ) + else: + assert _files_left(started_cluster_iceberg_no_spark, storage_type, table_name) > 0