Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
15 changes: 11 additions & 4 deletions pyiceberg/expressions/literals.py
Original file line number Diff line number Diff line change
Expand Up @@ -75,6 +75,13 @@ def _parse_numeric_string(value: str) -> Decimal:
return number


def _to_integral(value: Decimal, type_var: IcebergType) -> int:
"""Convert a Decimal to an int, rejecting values with a fractional part instead of rounding them."""
if value != value.to_integral_value():
raise ValueError(f"Could not convert {value} into a {type_var}, value has a fractional part")
return int(value)


class Literal(IcebergRootModel[L], Generic[L], ABC): # type: ignore
"""Literal which has a value and can be converted between types."""

Expand Down Expand Up @@ -527,24 +534,24 @@ def _(self, type_var: DecimalType) -> Literal[Decimal]:
raise ValueError(f"Could not convert {self.value} into a {type_var}")

@to.register(IntegerType)
def _(self, _: IntegerType) -> Literal[int]:
def _(self, type_var: IntegerType) -> Literal[int]:
value_int = int(self.value.to_integral_value())
if value_int > IntegerType.max:
return IntAboveMax()
elif value_int < IntegerType.min:
return IntBelowMin()
else:
return LongLiteral(value_int)
return LongLiteral(_to_integral(self.value, type_var))

@to.register(LongType)
def _(self, _: LongType) -> Literal[int]:
def _(self, type_var: LongType) -> Literal[int]:
value_int = int(self.value.to_integral_value())
if value_int > LongType.max:
return LongAboveMax()
elif value_int < LongType.min:
return LongBelowMin()
else:
return LongLiteral(value_int)
return LongLiteral(_to_integral(self.value, type_var))

@to.register(FloatType)
def _(self, _: FloatType) -> Literal[float]:
Expand Down
14 changes: 14 additions & 0 deletions tests/catalog/test_catalog_behaviors.py
Original file line number Diff line number Diff line change
Expand Up @@ -1345,6 +1345,20 @@ def test_append_nan_to_identity_partitioned_table(catalog: Catalog) -> None:
assert sorted(tbl.scan(row_filter="value is nan").to_arrow()["id"].to_pylist()) == [2, 4]


def test_scan_integer_column_with_decimal_literal(catalog: Catalog) -> None:

Copy link
Copy Markdown
Collaborator

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Could you also add a test for 2.0? That should also pass

Copy link
Copy Markdown
Contributor Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Added in c4489d7. 2.0 (and 2, 2.00, -2.0) has no fractional part, so it still converts and filters exactly:

  • tests/expressions/test_literals.py::test_integral_decimal_to_integral_type is now parametrized over "2", "2.0", "2.00", "-2.0" for both IntegerType and LongType, and checks each one becomes LongLiteral(int(value)).
  • tests/catalog/test_catalog_behaviors.py::test_scan_integer_column_with_decimal_literal checks x > 2.0 returns [3, 4] and now also x = 2.0 returns [2], on the memory, sql and sql_without_rowcount catalogs.

The 2.0 cases pass on both main and this branch, as you expected. The fractional cases (2.5, 2.6, -2.5, 0.1, and the x > 2.6 scan) fail on main and pass here: 11 failed / 8 passed with main's literals.py, and 19 passed with the fix. pytest tests/expressions tests/catalog/test_catalog_behaviors.py: 1307 passed. ruff check and ruff format --check are clean.

catalog.create_namespace("default")
identifier = f"default.scan_integer_decimal_literal_{catalog.name}"
tbl = catalog.create_table(identifier=identifier, schema=pa.schema([pa.field("x", pa.int32())]))
tbl.append(pa.table({"x": pa.array([1, 2, 3, 4], pa.int32())}))

# A literal with no fractional part, such as 2.0, still filters exactly
assert sorted(tbl.scan(row_filter="x > 2.0").to_arrow()["x"].to_pylist()) == [3, 4]
assert tbl.scan(row_filter="x = 2.0").to_arrow()["x"].to_pylist() == [2]
# Rounding 2.6 to 3 would turn x > 2.6 into x > 3 and silently drop x = 3
with pytest.raises(ValueError, match="Could not convert 2.6 into a int"):
tbl.scan(row_filter="x > 2.6").to_arrow()


def test_record_batch_reader_consumed_exactly_once(catalog: Catalog) -> None:
"""The streaming path must consume the underlying generator exactly once.
A regression that drained the reader twice (e.g. an extra .schema access
Expand Down
15 changes: 15 additions & 0 deletions tests/expressions/test_literals.py
Original file line number Diff line number Diff line change
Expand Up @@ -919,6 +919,21 @@ def test_decimal_to_long_below_min() -> None:
assert isinstance(DecimalLiteral(Decimal(LongType.min - 1)).to(LongType()), LongBelowMin)


@pytest.mark.parametrize("value", ["2.5", "2.6", "-2.5", "0.1"])
@pytest.mark.parametrize("target_type", [IntegerType(), LongType()])
def test_fractional_decimal_to_integral_type_raises(value: str, target_type: PrimitiveType) -> None:
# Rounding would change the predicate, e.g. x > 2.6 would become x > 3 and drop x = 3
with pytest.raises(ValueError, match=f"Could not convert {value} into a {target_type}"):
_ = DecimalLiteral(Decimal(value)).to(target_type)


@pytest.mark.parametrize("value", ["2", "2.0", "2.00", "-2.0"])
@pytest.mark.parametrize("target_type", [IntegerType(), LongType()])
def test_integral_decimal_to_integral_type(value: str, target_type: PrimitiveType) -> None:
# A decimal literal without a fractional part, such as 2.0, still converts exactly
assert DecimalLiteral(Decimal(value)).to(target_type) == LongLiteral(int(Decimal(value)))


def test_string_to_integer_type_invalid_value() -> None:
with pytest.raises(ValueError) as e:
_ = literal("abc").to(IntegerType())
Expand Down
Loading