Akash3121 commented on code in PR #10119:
URL: https://github.com/apache/paimon/pull/10119#discussion_r4084113698
##########
paimon-python/pypaimon/common/where_parser.py:
##########
@@ -114,21 +116,55 @@ def _build_field_type_map(fields: List[DataField]) ->
Dict[str, Optional[str]]:
return result
+def _cast_decimal(value_str: str, type_name: str) -> decimal.Decimal:
+ """Cast a DECIMAL literal, rescaled to the column's scale.
+
+ A DECIMAL column reads back as decimal.Decimal, and the pushed-down Arrow
+ filter binds the literal's own scale rather than the column's, so a literal
+ written at a different scale (e.g. ``50`` against a ``DECIMAL(10, 2)``
+ ``50.00``) would silently match nothing. Rescale to the column scale when
the
+ value is exact there; otherwise keep it as written (it then correctly
matches
+ no row). A malformed literal is re-raised as ValueError, which
+ parse_where_clause documents and the CLI relies on.
+ """
+ try:
+ value = decimal.Decimal(value_str)
+ except decimal.InvalidOperation:
+ raise ValueError(f"Invalid DECIMAL literal: {value_str!r}") from None
+
+ match = re.search(r'\(\s*\d+\s*,\s*(\d+)\s*\)', type_name)
+ scale = int(match.group(1)) if match else 0
+ try:
+ rescaled = value.quantize(decimal.Decimal(1).scaleb(-scale))
Review Comment:
Python’s default decimal context has precision 28, while Paimon supports
DECIMAL precision up to 38. For a ` DECIMAL(38,2) ` value such as
`123456789012345678901234567890123456.00` , querying with the exact
integer-form literal causes quantize to raise `InvalidOperation` ; this
catch returns the scale-0 value, and Arrow dataset filtering silently returns
zero rows. Please quantize inside a local context whose precision is at least
the declared column precision, and add a `DECIMAL(38,2)` regression where the
integer-form literal must match the stored scale-2 value.
background/reproduced: the integer-form literal returned 0 rows, while the
same literal ending in ` .00 `returned 1.
##########
paimon-python/pypaimon/common/where_parser.py:
##########
@@ -114,21 +116,55 @@ def _build_field_type_map(fields: List[DataField]) ->
Dict[str, Optional[str]]:
return result
+def _cast_decimal(value_str: str, type_name: str) -> decimal.Decimal:
+ """Cast a DECIMAL literal, rescaled to the column's scale.
+
+ A DECIMAL column reads back as decimal.Decimal, and the pushed-down Arrow
+ filter binds the literal's own scale rather than the column's, so a literal
+ written at a different scale (e.g. ``50`` against a ``DECIMAL(10, 2)``
+ ``50.00``) would silently match nothing. Rescale to the column scale when
the
+ value is exact there; otherwise keep it as written (it then correctly
matches
+ no row). A malformed literal is re-raised as ValueError, which
+ parse_where_clause documents and the CLI relies on.
+ """
+ try:
+ value = decimal.Decimal(value_str)
+ except decimal.InvalidOperation:
+ raise ValueError(f"Invalid DECIMAL literal: {value_str!r}") from None
+
+ match = re.search(r'\(\s*\d+\s*,\s*(\d+)\s*\)', type_name)
+ scale = int(match.group(1)) if match else 0
+ try:
+ rescaled = value.quantize(decimal.Decimal(1).scaleb(-scale))
+ except decimal.InvalidOperation:
+ return value
+ return rescaled if rescaled == value else value
+
+
def _cast_literal(value_str: str, type_name: str) -> Any:
"""Cast a literal string to the appropriate Python type based on the field
type."""
integer_types = {'TINYINT', 'SMALLINT', 'INT', 'INTEGER', 'BIGINT'}
float_types = {'FLOAT', 'DOUBLE'}
+ decimal_types = {'DECIMAL', 'NUMERIC', 'DEC'}
base_type = type_name.split('(')[0].strip()
if base_type in integer_types:
return int(value_str)
if base_type in float_types:
return float(value_str)
- if base_type.startswith('DECIMAL') or base_type in ('DECIMAL', 'NUMERIC',
'DEC'):
- return float(value_str)
+ if base_type in decimal_types:
+ return _cast_decimal(value_str, type_name)
if base_type == 'BOOLEAN':
return value_str.lower() in ('true', '1', 'yes')
+ if base_type == 'DATE':
+ # DATE/TIME columns read back as datetime.date / datetime.time;
leaving the
+ # literal a string makes the arrow comparison kernel raise instead of
+ # filtering. TIMESTAMP is intentionally left out: its LOCAL TIME ZONE
form
+ # reads back tz-aware and needs dedicated normalization.
+ return datetime.date.fromisoformat(value_str)
+ if base_type == 'TIME':
Review Comment:
`date.fromisoformat ` and `time.fromisoformat` were introduced in Python
3.7, but this package still declares `python_requires >=3.6 ` and runs a
Python 3.6 CI job. On 3.6, every DATE or TIME WHERE literal now raises
`AttributeError` . The green 3.6 job does not catch this because it runs only
the curated py36 subset and excludes `where_parser_test.py` . Could these
literals be parsed through a 3.6-compatible helper and covered by the 3.6
subset?
##########
paimon-python/pypaimon/common/where_parser.py:
##########
@@ -114,21 +116,55 @@ def _build_field_type_map(fields: List[DataField]) ->
Dict[str, Optional[str]]:
return result
+def _cast_decimal(value_str: str, type_name: str) -> decimal.Decimal:
+ """Cast a DECIMAL literal, rescaled to the column's scale.
+
+ A DECIMAL column reads back as decimal.Decimal, and the pushed-down Arrow
+ filter binds the literal's own scale rather than the column's, so a literal
+ written at a different scale (e.g. ``50`` against a ``DECIMAL(10, 2)``
+ ``50.00``) would silently match nothing. Rescale to the column scale when
the
+ value is exact there; otherwise keep it as written (it then correctly
matches
+ no row). A malformed literal is re-raised as ValueError, which
+ parse_where_clause documents and the CLI relies on.
+ """
+ try:
+ value = decimal.Decimal(value_str)
+ except decimal.InvalidOperation:
+ raise ValueError(f"Invalid DECIMAL literal: {value_str!r}") from None
+
+ match = re.search(r'\(\s*\d+\s*,\s*(\d+)\s*\)', type_name)
+ scale = int(match.group(1)) if match else 0
+ try:
+ rescaled = value.quantize(decimal.Decimal(1).scaleb(-scale))
+ except decimal.InvalidOperation:
+ return value
+ return rescaled if rescaled == value else value
+
+
def _cast_literal(value_str: str, type_name: str) -> Any:
"""Cast a literal string to the appropriate Python type based on the field
type."""
integer_types = {'TINYINT', 'SMALLINT', 'INT', 'INTEGER', 'BIGINT'}
float_types = {'FLOAT', 'DOUBLE'}
+ decimal_types = {'DECIMAL', 'NUMERIC', 'DEC'}
base_type = type_name.split('(')[0].strip()
if base_type in integer_types:
return int(value_str)
if base_type in float_types:
return float(value_str)
- if base_type.startswith('DECIMAL') or base_type in ('DECIMAL', 'NUMERIC',
'DEC'):
- return float(value_str)
+ if base_type in decimal_types:
+ return _cast_decimal(value_str, type_name)
if base_type == 'BOOLEAN':
return value_str.lower() in ('true', '1', 'yes')
+ if base_type == 'DATE':
+ # DATE/TIME columns read back as datetime.date / datetime.time;
leaving the
+ # literal a string makes the arrow comparison kernel raise instead of
+ # filtering. TIMESTAMP is intentionally left out: its LOCAL TIME ZONE
form
+ # reads back tz-aware and needs dedicated normalization.
+ return datetime.date.fromisoformat(value_str)
+ if base_type == 'TIME':
+ return datetime.time.fromisoformat(value_str)
Review Comment:
same as above; `time.fromisoformat` accepts values such as
`12:30:00+01:00` , but Paimon ` TIME` has no timezone. Arrow then compares
only the wall-clock value, so this literal currently matches a stored `
12:30:00` while silently discarding `+01:00` . Please reject parsed values
with non-null ` tzinfo` (or otherwise define explicit normalization) and add a
test proving invalid offset-bearing TIME input fails with ` ValueError `
--
This is an automated message from the Apache Git Service.
To respond to the message, please log on to GitHub and use the
URL above to go to the specific comment.
To unsubscribe, e-mail: [email protected]
For queries about this service, please contact Infrastructure at:
[email protected]