Skip to content

Commit 3017c0d

Browse files
Fix NaN upper-bound check in strict metrics NotIn
A NaN upper bound emptied the literal set through the `upper >= val` filter (False for every val), producing a false ROWS_MUST_MATCH. This can drop data files whose rows only partially match the delete filter in Table.delete(). Mirrors the existing NaN lower-bound guard.
1 parent 9299bdb commit 3017c0d

2 files changed

Lines changed: 30 additions & 0 deletions

File tree

‎pyiceberg/expressions/visitors.py‎

Lines changed: 5 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -1784,6 +1784,11 @@ def visit_not_in(self, term: BoundTerm, literals: set[L]) -> bool:
17841784
if upper_bytes is not None:
17851785
upper = _from_byte_buffer(field.field_type, upper_bytes)
17861786

1787+
if self._is_nan(upper):
1788+
# NaN indicates unreliable bounds.
1789+
# See the StrictMetricsEvaluator docs for more.
1790+
return ROWS_MIGHT_NOT_MATCH
1791+
17871792
literals = {val for val in literals if upper >= val}
17881793

17891794
if len(literals) == 0:

‎tests/expressions/test_evaluator.py‎

Lines changed: 25 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -1358,6 +1358,31 @@ def test_strict_not_equal_and_not_in_with_mixed_nans_and_matching_bounds(field_t
13581358
assert should_read == ROWS_MIGHT_NOT_MATCH, "Should not match: bounds prove the non-NaN value is 5.0"
13591359

13601360

1361+
@pytest.mark.parametrize("field_type", [FloatType(), DoubleType()])
1362+
def test_strict_not_in_with_nan_upper_bound(field_type: PrimitiveType) -> None:
1363+
schema = Schema(NestedField(1, "x", field_type, required=False))
1364+
# Column contains {1.0, NaN}: min is 1.0, max is NaN (NaN sorts greatest).
1365+
# No NaN stats are present, but the row 1.0 is in the literal set, so the
1366+
# file cannot be proven to fully match NotIn.
1367+
data_file = DataFile.from_args(
1368+
file_path="file.parquet",
1369+
file_format=FileFormat.PARQUET,
1370+
partition={},
1371+
record_count=2,
1372+
file_size_in_bytes=1,
1373+
value_counts={1: 2},
1374+
null_value_counts={1: 0},
1375+
nan_value_counts={1: 1},
1376+
lower_bounds={1: to_bytes(field_type, 1.0)},
1377+
upper_bounds={1: to_bytes(field_type, float("nan"))},
1378+
)
1379+
1380+
should_read = _StrictMetricsEvaluator(schema, NotIn("x", {1.0, 2.0})).eval(data_file)
1381+
assert should_read == ROWS_MIGHT_NOT_MATCH, (
1382+
"NaN upper bound makes the bounds unusable: the non-NaN row 1.0 is in the literal set"
1383+
)
1384+
1385+
13611386
@pytest.mark.parametrize("field_type", [FloatType(), DoubleType()])
13621387
def test_strict_not_equal_and_not_in_with_all_nans(field_type: PrimitiveType) -> None:
13631388
schema = Schema(NestedField(1, "x", field_type, required=False))

0 commit comments

Comments
 (0)