Skip to content

Commit 24ffc24

Browse files
Move replace integration tests to tests/integration/test_catalog.py
That file is the cross-catalog integration analog to Java's CatalogTests.java; every other mutation (create/rename/drop/etc.) already has a slot there parametrized over six catalog fixtures. The two replace tests have no REST-specific code and don't belong in test_rest_catalog.py. Adopts the file's existing conventions: test_catalog parameter name, database_name + table_name fixtures, no manual cleanup guards (clean_up runs between tests).
1 parent 8d6bc1c commit 24ffc24

2 files changed

Lines changed: 62 additions & 72 deletions

File tree

‎tests/integration/test_catalog.py‎

Lines changed: 62 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -20,6 +20,7 @@
2020
from collections.abc import Generator
2121
from pathlib import Path, PosixPath
2222

23+
import pyarrow as pa
2324
import pytest
2425
from pytest_lazy_fixtures import lf
2526

@@ -43,7 +44,7 @@
4344
from pyiceberg.table.metadata import INITIAL_SPEC_ID
4445
from pyiceberg.table.sorting import INITIAL_SORT_ORDER_ID, SortField, SortOrder
4546
from pyiceberg.transforms import BucketTransform, DayTransform, IdentityTransform
46-
from pyiceberg.types import IntegerType, LongType, NestedField, TimestampType, UUIDType
47+
from pyiceberg.types import BooleanType, IntegerType, LongType, NestedField, StringType, TimestampType, UUIDType
4748
from tests.conftest import (
4849
clean_up,
4950
does_support_atomic_concurrent_updates,
@@ -866,3 +867,63 @@ def test_load_missing_table(test_catalog: Catalog, database_name: str, table_nam
866867

867868
with pytest.raises(NoSuchTableError):
868869
test_catalog.load_table(identifier)
870+
871+
872+
@pytest.mark.integration
873+
@pytest.mark.parametrize("test_catalog", CATALOGS)
874+
def test_replace_table(test_catalog: Catalog, database_name: str, table_name: str) -> None:
875+
test_catalog.create_namespace(database_name)
876+
identifier = (database_name, table_name)
877+
878+
original_schema = Schema(
879+
NestedField(field_id=1, name="id", field_type=LongType(), required=False),
880+
NestedField(field_id=2, name="data", field_type=StringType(), required=False),
881+
)
882+
original = test_catalog.create_table(identifier, schema=original_schema)
883+
original.append(
884+
pa.Table.from_pydict(
885+
{"id": [1, 2, 3], "data": ["a", "b", "c"]},
886+
schema=pa.schema([pa.field("id", pa.int64()), pa.field("data", pa.large_string())]),
887+
)
888+
)
889+
original.refresh()
890+
original_snapshot_id = original.current_snapshot().snapshot_id # type: ignore[union-attr]
891+
892+
new_schema = Schema(
893+
NestedField(field_id=1, name="id", field_type=LongType(), required=False),
894+
NestedField(field_id=2, name="name", field_type=StringType(), required=False),
895+
NestedField(field_id=3, name="active", field_type=BooleanType(), required=False),
896+
)
897+
replaced = test_catalog.replace_table(identifier, schema=new_schema)
898+
899+
assert replaced.metadata.table_uuid == original.metadata.table_uuid
900+
assert replaced.current_snapshot() is None
901+
assert any(s.snapshot_id == original_snapshot_id for s in replaced.metadata.snapshots)
902+
903+
904+
@pytest.mark.integration
905+
@pytest.mark.parametrize("test_catalog", CATALOGS)
906+
def test_replace_table_transaction(test_catalog: Catalog, database_name: str, table_name: str) -> None:
907+
test_catalog.create_namespace(database_name)
908+
identifier = (database_name, table_name)
909+
910+
old_data = pa.Table.from_pydict(
911+
{"id": [1], "data": ["old"]},
912+
schema=pa.schema([pa.field("id", pa.int64()), pa.field("data", pa.large_string())]),
913+
)
914+
original = test_catalog.create_table(identifier, schema=old_data.schema)
915+
original.append(old_data)
916+
old_snapshot_id = test_catalog.load_table(identifier).current_snapshot().snapshot_id # type: ignore[union-attr]
917+
918+
new_data = pa.Table.from_pydict(
919+
{"id": [10, 20], "name": ["alice", "bob"]},
920+
schema=pa.schema([pa.field("id", pa.int64()), pa.field("name", pa.large_string())]),
921+
)
922+
with test_catalog.replace_table_transaction(identifier, schema=new_data.schema) as txn:
923+
txn.append(new_data)
924+
925+
replaced = test_catalog.load_table(identifier)
926+
assert replaced.current_snapshot() is not None
927+
assert replaced.current_snapshot().snapshot_id != old_snapshot_id # type: ignore[union-attr]
928+
assert any(s.snapshot_id == old_snapshot_id for s in replaced.metadata.snapshots)
929+
assert replaced.scan().to_arrow().num_rows == 2

‎tests/integration/test_rest_catalog.py‎

Lines changed: 0 additions & 71 deletions
Original file line numberDiff line numberDiff line change
@@ -18,15 +18,12 @@
1818

1919
import time
2020

21-
import pyarrow as pa
2221
import pytest
2322
from pytest_lazy_fixtures import lf
2423

25-
from pyiceberg.catalog import Catalog
2624
from pyiceberg.catalog.rest import RestCatalog
2725
from pyiceberg.exceptions import NoSuchViewError
2826
from pyiceberg.schema import Schema
29-
from pyiceberg.types import BooleanType, LongType, NestedField, StringType
3027
from pyiceberg.view.metadata import SQLViewRepresentation, ViewVersion
3128

3229
TEST_NAMESPACE_IDENTIFIER = "TEST NS"
@@ -72,74 +69,6 @@ def test_create_namespace_if_already_existing(catalog: RestCatalog) -> None:
7269
assert catalog.namespace_exists(TEST_NAMESPACE_IDENTIFIER)
7370

7471

75-
@pytest.mark.integration
76-
@pytest.mark.parametrize("catalog", [lf("session_catalog")])
77-
def test_replace_table(catalog: Catalog) -> None:
78-
identifier = f"default.test_replace_table_e2e_{catalog.name}"
79-
if not catalog.namespace_exists("default"):
80-
catalog.create_namespace("default")
81-
if catalog.table_exists(identifier):
82-
catalog.drop_table(identifier)
83-
84-
original_schema = Schema(
85-
NestedField(field_id=1, name="id", field_type=LongType(), required=False),
86-
NestedField(field_id=2, name="data", field_type=StringType(), required=False),
87-
)
88-
original = catalog.create_table(identifier, schema=original_schema)
89-
original.append(
90-
pa.Table.from_pydict(
91-
{"id": [1, 2, 3], "data": ["a", "b", "c"]},
92-
schema=pa.schema([pa.field("id", pa.int64()), pa.field("data", pa.large_string())]),
93-
)
94-
)
95-
original.refresh()
96-
original_snapshot_id = original.current_snapshot().snapshot_id # type: ignore[union-attr]
97-
98-
new_schema = Schema(
99-
NestedField(field_id=1, name="id", field_type=LongType(), required=False),
100-
NestedField(field_id=2, name="name", field_type=StringType(), required=False),
101-
NestedField(field_id=3, name="active", field_type=BooleanType(), required=False),
102-
)
103-
replaced = catalog.replace_table(identifier, schema=new_schema)
104-
105-
assert replaced.metadata.table_uuid == original.metadata.table_uuid
106-
assert replaced.current_snapshot() is None
107-
assert any(s.snapshot_id == original_snapshot_id for s in replaced.metadata.snapshots)
108-
catalog.drop_table(identifier)
109-
110-
111-
@pytest.mark.integration
112-
@pytest.mark.parametrize("catalog", [lf("session_catalog")])
113-
def test_replace_table_transaction(catalog: Catalog) -> None:
114-
identifier = f"default.test_replace_rtas_{catalog.name}"
115-
if not catalog.namespace_exists("default"):
116-
catalog.create_namespace("default")
117-
if catalog.table_exists(identifier):
118-
catalog.drop_table(identifier)
119-
120-
old_data = pa.Table.from_pydict(
121-
{"id": [1], "data": ["old"]},
122-
schema=pa.schema([pa.field("id", pa.int64()), pa.field("data", pa.large_string())]),
123-
)
124-
original = catalog.create_table(identifier, schema=old_data.schema)
125-
original.append(old_data)
126-
old_snapshot_id = catalog.load_table(identifier).current_snapshot().snapshot_id # type: ignore[union-attr]
127-
128-
new_data = pa.Table.from_pydict(
129-
{"id": [10, 20], "name": ["alice", "bob"]},
130-
schema=pa.schema([pa.field("id", pa.int64()), pa.field("name", pa.large_string())]),
131-
)
132-
with catalog.replace_table_transaction(identifier, schema=new_data.schema) as txn:
133-
txn.append(new_data)
134-
135-
replaced = catalog.load_table(identifier)
136-
assert replaced.current_snapshot() is not None
137-
assert replaced.current_snapshot().snapshot_id != old_snapshot_id # type: ignore[union-attr]
138-
assert any(s.snapshot_id == old_snapshot_id for s in replaced.metadata.snapshots)
139-
assert replaced.scan().to_arrow().num_rows == 2
140-
catalog.drop_table(identifier)
141-
142-
14372
@pytest.mark.integration
14473
@pytest.mark.parametrize("catalog", [lf("session_catalog")])
14574
def test_load_view(catalog: RestCatalog, table_schema_nested: Schema, database_name: str, view_name: str) -> None:

0 commit comments

Comments
 (0)