From 7f4ccd1b98f96661944d8f100f31795632de5c31 Mon Sep 17 00:00:00 2001 From: Amin Ghadersohi Date: Sat, 26 Sep 2026 09:56:35 +0000 Subject: [PATCH 1/2] Convert nested result columns without numpy float conversion With pandas enabled, ARRAY/MAP/STRUCT values went through numpy: an ARRAY containing a NULL came back as float64, so values beyond 2**53 lost precision and NULL became NaN. Take nested columns directly from Arrow (to_pylist), as the disable_pandas path already does. Signed-off-by: Amin Ghadersohi --- src/databricks/sql/result_set.py | 22 +++++++++++++++++++++- tests/unit/test_pandas_compatibility.py | 18 ++++++++++++++++++ 2 files changed, 39 insertions(+), 1 deletion(-) diff --git a/src/databricks/sql/result_set.py b/src/databricks/sql/result_set.py index f24a07505..b07ef2f4f 100644 --- a/src/databricks/sql/result_set.py +++ b/src/databricks/sql/result_set.py @@ -130,7 +130,27 @@ def _convert_arrow_table(self, table): ) res = df.to_numpy(na_value=None, dtype="object") - return [ResultRow(*v) for v in res] + + # pandas converts nested (list/map/struct) values through numpy, which + # turns e.g. ARRAY with a NULL element into float64 (precision + # loss beyond 2**53, NULL as NaN). Take nested columns straight from + # Arrow, as the disable_pandas path does. + nested = { + index: table.column(index).to_pylist() + for index, field in enumerate(table.schema) + if pyarrow.types.is_nested(field.type) + } + if not nested: + return [ResultRow(*v) for v in res] + return [ + ResultRow( + *[ + nested[index][position] if index in nested else value + for index, value in enumerate(row) + ] + ) + for position, row in enumerate(res) + ] @property def rownumber(self): diff --git a/tests/unit/test_pandas_compatibility.py b/tests/unit/test_pandas_compatibility.py index 5434559b2..f7959ee68 100644 --- a/tests/unit/test_pandas_compatibility.py +++ b/tests/unit/test_pandas_compatibility.py @@ -294,6 +294,24 @@ def test_list_type(self): self.assertIsNone(rows[1].list_col) self.assertEqual(list(rows[2].list_col), [4, 5]) + def test_list_of_bigint_with_null_is_exact(self): + """Nested integers must not go through numpy float64.""" + big = 9007199254740993 # 2**53 + 1: not representable as float64 + table = pa.table( + { + "list_col": pa.array( + [[big, -9223372036854775808, None]], type=pa.list_(pa.int64()) + ), + } + ) + description = [("list_col", "array", None, None, None, None, None)] + + rows = _make_result_set(description)._convert_arrow_table(table) + + values = list(rows[0].list_col) + self.assertEqual(values, [big, -9223372036854775808, None]) + self.assertTrue(all(type(v) is int for v in values[:2])) + def test_struct_type(self): table = pa.table( { From 5ee15c3405cd4f8d41b311adab7c0fd6a2da6e1f Mon Sep 17 00:00:00 2001 From: Amin Ghadersohi Date: Sat, 26 Sep 2026 23:32:46 +0000 Subject: [PATCH 2/2] Keep ARRAY results as numpy.ndarray and skip pandas for nested columns Nested columns are converted from Arrow directly, so they are no longer converted through pandas and then replaced. ARRAY values stay numpy.ndarray, the documented type: native dtype when the elements are float, or integer/boolean without NULLs; object dtype with exact int and None when integer elements contain NULL (previously float64 with NaN). MAP stays a list of tuples and STRUCT a dict, with the same conversion applied to their nested values. Signed-off-by: Amin Ghadersohi --- CHANGELOG.md | 3 + src/databricks/sql/result_set.py | 127 +++++++++++++++++++----- tests/unit/test_pandas_compatibility.py | 89 +++++++++++++++++ 3 files changed, 196 insertions(+), 23 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index bce9e16e4..cf16662c8 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -1,5 +1,8 @@ # Release History +# Unreleased +- Fix: with pandas enabled (the default), complex-type values no longer lose precision: an ARRAY of integers containing a NULL (at any nesting level, including inside MAP and STRUCT values) is returned as an object `numpy.ndarray` of exact `int`/`None` instead of float64 with NaN, and nested columns are converted from Arrow directly instead of through pandas. Return types are unchanged: ARRAY is a `numpy.ndarray`, MAP a list of tuples, STRUCT a dict. + # 4.6.0 (2026-09-24) - Upgrade Databricks SQL Kernel to 1.1.0; the kernel dependency is now stable and no longer experimental. - Transparently auto-recover Thrift connections to Reyden / Real-Time warehouses: when a warehouse rejects the default Thrift protocol (SQLSTATE `KP001`), the session is re-opened on the kernel backend and the warehouse is remembered so later connections skip Thrift. Applies only when no backend was chosen explicitly. diff --git a/src/databricks/sql/result_set.py b/src/databricks/sql/result_set.py index b07ef2f4f..2dcca3c44 100644 --- a/src/databricks/sql/result_set.py +++ b/src/databricks/sql/result_set.py @@ -4,6 +4,7 @@ from typing import List, Optional, TYPE_CHECKING, Tuple import logging +import numpy import pandas try: @@ -121,36 +122,46 @@ def _convert_arrow_table(self, table): pyarrow.string(): pandas.StringDtype(), } - # Need to rename columns, as the to_pandas function cannot handle duplicate column names - table_renamed = table.rename_columns([str(c) for c in range(table.num_columns)]) - df = table_renamed.to_pandas( - types_mapper=dtype_mapping.get, - date_as_object=True, - timestamp_as_object=True, - ) - - res = df.to_numpy(na_value=None, dtype="object") - - # pandas converts nested (list/map/struct) values through numpy, which - # turns e.g. ARRAY with a NULL element into float64 (precision - # loss beyond 2**53, NULL as NaN). Take nested columns straight from - # Arrow, as the disable_pandas path does. + # Nested (ARRAY/MAP/STRUCT) columns are converted from Arrow directly + # and never pass through pandas: pandas turns an integer array + # containing a NULL into float64 (precision loss beyond 2**53, NULL as + # NaN). See _nested_column_to_python for the returned shapes. nested = { - index: table.column(index).to_pylist() + index: _nested_column_to_python(table.column(index)) for index, field in enumerate(table.schema) if pyarrow.types.is_nested(field.type) } + scalar_indices = [i for i in range(table.num_columns) if i not in nested] + + res = None + if scalar_indices: + scalar_table = table.select(scalar_indices) if nested else table + # Need to rename columns, as the to_pandas function cannot handle duplicate column names + scalar_table = scalar_table.rename_columns( + [str(c) for c in range(scalar_table.num_columns)] + ) + df = scalar_table.to_pandas( + types_mapper=dtype_mapping.get, + date_as_object=True, + timestamp_as_object=True, + ) + res = df.to_numpy(na_value=None, dtype="object") + if not nested: return [ResultRow(*v) for v in res] - return [ - ResultRow( - *[ - nested[index][position] if index in nested else value - for index, value in enumerate(row) - ] + + rows = [] + for position in range(table.num_rows): + scalars = iter(res[position]) if res is not None else iter(()) + rows.append( + ResultRow( + *[ + nested[index][position] if index in nested else next(scalars) + for index in range(table.num_columns) + ] + ) ) - for position, row in enumerate(res) - ] + return rows @property def rownumber(self): @@ -491,3 +502,73 @@ def map_col_type(type_): (column.name, map_col_type(column.datatype), None, None, None, None, None) for column in table_schema_message.columns ] + + +def _object_array(values: list) -> "numpy.ndarray": + """A 1-D object ndarray holding ``values`` as-is (no numpy broadcasting).""" + out = numpy.empty(len(values), dtype=object) + for i, value in enumerate(values): + out[i] = value + return out + + +def _arrow_array_to_python(array) -> list: + """Convert one Arrow array to a list of Python values, one per slot. + + Shapes match what the pandas conversion has always returned for complex + types (see ``_use_arrow_native_complex_types``): ARRAY is a + ``numpy.ndarray``, MAP is a list of ``(key, value)`` tuples, STRUCT is a + dict, NULL is None. The difference is that integer/boolean array elements + are exact: an ARRAY without NULLs is an int64 ndarray as before, + and one with NULLs is an object ndarray of ``int``/``None`` instead of a + float64 ndarray with NaN. + """ + t = array.type + if pyarrow.types.is_map(t): + result: list = [] + for scalar in array: + if not scalar.is_valid: + result.append(None) + continue + keys, items = scalar.values.flatten() + result.append( + list(zip(_arrow_array_to_python(keys), _arrow_array_to_python(items))) + ) + return result + if ( + pyarrow.types.is_list(t) + or pyarrow.types.is_large_list(t) + or pyarrow.types.is_fixed_size_list(t) + ): + return [ + _arrow_list_values_to_ndarray(scalar.values) if scalar.is_valid else None + for scalar in array + ] + if pyarrow.types.is_struct(t): + names = [t.field(i).name for i in range(t.num_fields)] + children = [_arrow_array_to_python(child) for child in array.flatten()] + return [ + dict(zip(names, (child[i] for child in children))) if valid else None + for i, valid in enumerate(array.is_valid().to_pylist()) + ] + return array.to_pylist() + + +def _arrow_list_values_to_ndarray(values) -> "numpy.ndarray": + """The elements of one ARRAY value as a numpy.ndarray.""" + t = values.type + if pyarrow.types.is_floating(t) or ( + (pyarrow.types.is_integer(t) or pyarrow.types.is_boolean(t)) + and values.null_count == 0 + ): + # Native dtype, as pandas produced; floats keep NaN for NULL. + return values.to_numpy(zero_copy_only=False, writable=True) + return _object_array(_arrow_array_to_python(values)) + + +def _nested_column_to_python(column) -> list: + """Convert a (chunked) nested Arrow column to one Python value per row.""" + result: list = [] + for chunk in column.chunks: + result.extend(_arrow_array_to_python(chunk)) + return result diff --git a/tests/unit/test_pandas_compatibility.py b/tests/unit/test_pandas_compatibility.py index f7959ee68..f9f5eb9d0 100644 --- a/tests/unit/test_pandas_compatibility.py +++ b/tests/unit/test_pandas_compatibility.py @@ -11,6 +11,7 @@ from decimal import Decimal from unittest.mock import Mock +import numpy import pandas import pytest @@ -308,10 +309,98 @@ def test_list_of_bigint_with_null_is_exact(self): rows = _make_result_set(description)._convert_arrow_table(table) + # Still a numpy.ndarray (the documented ARRAY type), but object dtype + # holding exact ints and None rather than float64 with NaN. + self.assertIsInstance(rows[0].list_col, numpy.ndarray) values = list(rows[0].list_col) self.assertEqual(values, [big, -9223372036854775808, None]) self.assertTrue(all(type(v) is int for v in values[:2])) + def test_list_without_null_keeps_native_dtype(self): + table = pa.table( + { + "ints": pa.array([[1, 2], []], type=pa.list_(pa.int64())), + "floats": pa.array([[1.5, None], [2.0]], type=pa.list_(pa.float64())), + "strings": pa.array([["a", "b"], ["c"]], type=pa.list_(pa.string())), + } + ) + description = [ + ("ints", "array", None, None, None, None, None), + ("floats", "array", None, None, None, None, None), + ("strings", "array", None, None, None, None, None), + ] + + rows = _make_result_set(description)._convert_arrow_table(table) + + self.assertIsInstance(rows[0].ints, numpy.ndarray) + self.assertEqual(rows[0].ints.dtype, numpy.int64) + self.assertEqual(rows[1].ints.tolist(), []) + self.assertEqual(rows[0].floats.dtype, numpy.float64) + self.assertTrue(numpy.isnan(rows[0].floats[1])) + self.assertIsInstance(rows[1].strings, numpy.ndarray) + self.assertEqual(rows[1].strings.dtype, object) + self.assertEqual(rows[0].strings.tolist(), ["a", "b"]) + + def test_nested_complex_values_are_exact(self): + big = 9007199254740993 + table = pa.table( + { + "id": pa.array([1, 2], type=pa.int64()), + "array_array": pa.array( + [[[big, None], [1]], None], type=pa.list_(pa.list_(pa.int64())) + ), + "map_array": pa.array( + [[("a", [big, None])], [("b", None)]], + type=pa.map_(pa.string(), pa.list_(pa.int64())), + ), + "struct_col": pa.array( + [{"x": big, "y": [None, 2]}, {"x": None, "y": None}], + type=pa.struct([("x", pa.int64()), ("y", pa.list_(pa.int64()))]), + ), + "name": pa.array(["a", None], type=pa.string()), + } + ) + description = [ + (name, "t", None, None, None, None, None) for name in table.column_names + ] + + rows = _make_result_set(description)._convert_arrow_table(table) + + self.assertEqual([r.id for r in rows], [1, 2]) + self.assertEqual([r.name for r in rows], ["a", None]) + + outer = rows[0].array_array + self.assertIsInstance(outer, numpy.ndarray) + self.assertIsInstance(outer[0], numpy.ndarray) + self.assertEqual(outer[0].tolist(), [big, None]) + self.assertEqual(outer[1].dtype, numpy.int64) + self.assertIsNone(rows[1].array_array) + + ((key, value),) = rows[0].map_array + self.assertEqual(key, "a") + self.assertIsInstance(value, numpy.ndarray) + self.assertEqual(value.tolist(), [big, None]) + self.assertEqual(rows[1].map_array, [("b", None)]) + + self.assertEqual(rows[0].struct_col["x"], big) + self.assertEqual(rows[0].struct_col["y"].tolist(), [None, 2]) + self.assertEqual(rows[1].struct_col, {"x": None, "y": None}) + + def test_only_nested_columns_across_chunks(self): + list_type = pa.list_(pa.int64()) + column = pa.chunked_array( + [pa.array([[1]], list_type), pa.array([None, [2, None]], list_type)] + ) + table = pa.table({"list_col": column}) + description = [("list_col", "array", None, None, None, None, None)] + + rows = _make_result_set(description)._convert_arrow_table(table) + + self.assertEqual(len(rows), 3) + self.assertEqual(rows[0].list_col.tolist(), [1]) + self.assertIsNone(rows[1].list_col) + self.assertEqual(rows[2].list_col.tolist(), [2, None]) + def test_struct_type(self): table = pa.table( {