diff --git a/CHANGELOG.md b/CHANGELOG.md index bce9e16e4..ebed79a53 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -1,5 +1,8 @@ # Release History +# Unreleased +- Fix: with pandas enabled (the default), a floating-point NaN in a FLOAT or DOUBLE column is now returned as `float('nan')` instead of `None`, so it can be told apart from SQL NULL. This matches what `_disable_pandas=True` already returned. + # 4.6.0 (2026-09-24) - Upgrade Databricks SQL Kernel to 1.1.0; the kernel dependency is now stable and no longer experimental. - Transparently auto-recover Thrift connections to Reyden / Real-Time warehouses: when a warehouse rejects the default Thrift protocol (SQLSTATE `KP001`), the session is re-opened on the kernel backend and the warehouse is remembered so later connections skip Thrift. Applies only when no backend was chosen explicitly. diff --git a/src/databricks/sql/result_set.py b/src/databricks/sql/result_set.py index f24a07505..57f23954e 100644 --- a/src/databricks/sql/result_set.py +++ b/src/databricks/sql/result_set.py @@ -8,6 +8,7 @@ try: import pyarrow + import pyarrow.compute except ImportError: pyarrow = None @@ -130,6 +131,14 @@ def _convert_arrow_table(self, table): ) res = df.to_numpy(na_value=None, dtype="object") + + # The nullable float dtypes turn IEEE NaN into NA, which would make it + # indistinguishable from SQL NULL, so restore NaN values from Arrow. + for i, field in enumerate(table_renamed.schema): + if pyarrow.types.is_floating(field.type): + is_nan = pyarrow.compute.is_nan(table_renamed.column(i)) + res[is_nan.fill_null(False).to_numpy(), i] = float("nan") + return [ResultRow(*v) for v in res] @property diff --git a/tests/unit/test_pandas_compatibility.py b/tests/unit/test_pandas_compatibility.py index 5434559b2..585864a1b 100644 --- a/tests/unit/test_pandas_compatibility.py +++ b/tests/unit/test_pandas_compatibility.py @@ -7,6 +7,7 @@ """ import datetime +import math import unittest from decimal import Decimal from unittest.mock import Mock @@ -122,6 +123,50 @@ def test_float_types(self): self.assertIsNone(rows[1].float32_col) self.assertAlmostEqual(rows[2].float64_col, 4.5) + def _nan_and_null_table(self): + values = [[float("nan"), None], [float("inf"), float("-inf"), 1.5]] + table = pa.table( + { + "id": pa.chunked_array([[1, 2], [3, 4, 5]], type=pa.int64()), + "float32_col": pa.chunked_array(values, type=pa.float32()), + "float64_col": pa.chunked_array(values, type=pa.float64()), + } + ) + description = [ + ("id", "bigint", None, None, None, None, None), + ("float32_col", "float", None, None, None, None, None), + ("float64_col", "double", None, None, None, None, None), + ] + return table, description + + def test_float_nan_is_not_converted_to_null(self): + table, description = self._nan_and_null_table() + + rows = _make_result_set(description)._convert_arrow_table(table) + + self.assertEqual([row.id for row in rows], [1, 2, 3, 4, 5]) + for column in ("float32_col", "float64_col"): + values = [getattr(row, column) for row in rows] + self.assertIsInstance(values[0], float) + self.assertTrue(math.isnan(values[0])) + self.assertEqual(values[1:], [None, float("inf"), float("-inf"), 1.5]) + + def test_float_nan_and_null_match_disable_pandas_path(self): + table, description = self._nan_and_null_table() + + def normalized(rows): + return [ + ["NaN" if isinstance(v, float) and math.isnan(v) else v for v in row] + for row in rows + ] + + with_pandas = _make_result_set(description)._convert_arrow_table(table) + without_pandas = _make_result_set( + description, disable_pandas=True + )._convert_arrow_table(table) + + self.assertEqual(normalized(with_pandas), normalized(without_pandas)) + def test_boolean_type(self): table = pa.table( {