Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
3 changes: 3 additions & 0 deletions CHANGELOG.md
Original file line number Diff line number Diff line change
@@ -1,5 +1,8 @@
# Release History

# Unreleased
- Fix: with pandas enabled (the default), complex-type values no longer lose precision: an ARRAY of integers containing a NULL (at any nesting level, including inside MAP and STRUCT values) is returned as an object `numpy.ndarray` of exact `int`/`None` instead of float64 with NaN, and nested columns are converted from Arrow directly instead of through pandas. Return types are unchanged: ARRAY is a `numpy.ndarray`, MAP a list of tuples, STRUCT a dict.

# 4.6.0 (2026-09-24)
- Upgrade Databricks SQL Kernel to 1.1.0; the kernel dependency is now stable and no longer experimental.
- Transparently auto-recover Thrift connections to Reyden / Real-Time warehouses: when a warehouse rejects the default Thrift protocol (SQLSTATE `KP001`), the session is re-opened on the kernel backend and the warehouse is remembered so later connections skip Thrift. Applies only when no backend was chosen explicitly.
Expand Down
121 changes: 111 additions & 10 deletions src/databricks/sql/result_set.py
Original file line number Diff line number Diff line change
Expand Up @@ -4,6 +4,7 @@
from typing import List, Optional, TYPE_CHECKING, Tuple

import logging
import numpy
import pandas

try:
Expand Down Expand Up @@ -121,16 +122,46 @@ def _convert_arrow_table(self, table):
pyarrow.string(): pandas.StringDtype(),
}

# Need to rename columns, as the to_pandas function cannot handle duplicate column names
table_renamed = table.rename_columns([str(c) for c in range(table.num_columns)])
df = table_renamed.to_pandas(
types_mapper=dtype_mapping.get,
date_as_object=True,
timestamp_as_object=True,
)

res = df.to_numpy(na_value=None, dtype="object")
return [ResultRow(*v) for v in res]
# Nested (ARRAY/MAP/STRUCT) columns are converted from Arrow directly
# and never pass through pandas: pandas turns an integer array
# containing a NULL into float64 (precision loss beyond 2**53, NULL as
# NaN). See _nested_column_to_python for the returned shapes.
nested = {
index: _nested_column_to_python(table.column(index))
for index, field in enumerate(table.schema)
if pyarrow.types.is_nested(field.type)
}
Comment thread
aminghadersohi marked this conversation as resolved.
scalar_indices = [i for i in range(table.num_columns) if i not in nested]

res = None
if scalar_indices:
scalar_table = table.select(scalar_indices) if nested else table
# Need to rename columns, as the to_pandas function cannot handle duplicate column names
scalar_table = scalar_table.rename_columns(
[str(c) for c in range(scalar_table.num_columns)]
)
df = scalar_table.to_pandas(
types_mapper=dtype_mapping.get,
date_as_object=True,
timestamp_as_object=True,
)
res = df.to_numpy(na_value=None, dtype="object")

if not nested:
return [ResultRow(*v) for v in res]

rows = []
for position in range(table.num_rows):
scalars = iter(res[position]) if res is not None else iter(())
rows.append(
ResultRow(
*[
nested[index][position] if index in nested else next(scalars)
for index in range(table.num_columns)
]
)
)
return rows

@property
def rownumber(self):
Expand Down Expand Up @@ -471,3 +502,73 @@ def map_col_type(type_):
(column.name, map_col_type(column.datatype), None, None, None, None, None)
for column in table_schema_message.columns
]


def _object_array(values: list) -> "numpy.ndarray":
"""A 1-D object ndarray holding ``values`` as-is (no numpy broadcasting)."""
out = numpy.empty(len(values), dtype=object)
for i, value in enumerate(values):
out[i] = value
return out


def _arrow_array_to_python(array) -> list:
"""Convert one Arrow array to a list of Python values, one per slot.

Shapes match what the pandas conversion has always returned for complex
types (see ``_use_arrow_native_complex_types``): ARRAY is a
``numpy.ndarray``, MAP is a list of ``(key, value)`` tuples, STRUCT is a
dict, NULL is None. The difference is that integer/boolean array elements
are exact: an ARRAY<BIGINT> without NULLs is an int64 ndarray as before,
and one with NULLs is an object ndarray of ``int``/``None`` instead of a
float64 ndarray with NaN.
"""
t = array.type
if pyarrow.types.is_map(t):
result: list = []
for scalar in array:
if not scalar.is_valid:
result.append(None)
continue
keys, items = scalar.values.flatten()
result.append(
list(zip(_arrow_array_to_python(keys), _arrow_array_to_python(items)))
)
return result
if (
pyarrow.types.is_list(t)
or pyarrow.types.is_large_list(t)
or pyarrow.types.is_fixed_size_list(t)
):
return [
_arrow_list_values_to_ndarray(scalar.values) if scalar.is_valid else None
for scalar in array
]
if pyarrow.types.is_struct(t):
names = [t.field(i).name for i in range(t.num_fields)]
children = [_arrow_array_to_python(child) for child in array.flatten()]
return [
dict(zip(names, (child[i] for child in children))) if valid else None
for i, valid in enumerate(array.is_valid().to_pylist())
]
return array.to_pylist()


def _arrow_list_values_to_ndarray(values) -> "numpy.ndarray":
"""The elements of one ARRAY value as a numpy.ndarray."""
t = values.type
if pyarrow.types.is_floating(t) or (
(pyarrow.types.is_integer(t) or pyarrow.types.is_boolean(t))
and values.null_count == 0
):
# Native dtype, as pandas produced; floats keep NaN for NULL.
return values.to_numpy(zero_copy_only=False, writable=True)
return _object_array(_arrow_array_to_python(values))


def _nested_column_to_python(column) -> list:
"""Convert a (chunked) nested Arrow column to one Python value per row."""
result: list = []
for chunk in column.chunks:
result.extend(_arrow_array_to_python(chunk))
return result
107 changes: 107 additions & 0 deletions tests/unit/test_pandas_compatibility.py
Original file line number Diff line number Diff line change
Expand Up @@ -11,6 +11,7 @@
from decimal import Decimal
from unittest.mock import Mock

import numpy
import pandas
import pytest

Expand Down Expand Up @@ -294,6 +295,112 @@ def test_list_type(self):
self.assertIsNone(rows[1].list_col)
self.assertEqual(list(rows[2].list_col), [4, 5])

def test_list_of_bigint_with_null_is_exact(self):
"""Nested integers must not go through numpy float64."""
big = 9007199254740993 # 2**53 + 1: not representable as float64
table = pa.table(
{
"list_col": pa.array(
[[big, -9223372036854775808, None]], type=pa.list_(pa.int64())
),
}
)
description = [("list_col", "array", None, None, None, None, None)]

rows = _make_result_set(description)._convert_arrow_table(table)

# Still a numpy.ndarray (the documented ARRAY type), but object dtype
# holding exact ints and None rather than float64 with NaN.
self.assertIsInstance(rows[0].list_col, numpy.ndarray)
values = list(rows[0].list_col)
self.assertEqual(values, [big, -9223372036854775808, None])
self.assertTrue(all(type(v) is int for v in values[:2]))

def test_list_without_null_keeps_native_dtype(self):
table = pa.table(
{
"ints": pa.array([[1, 2], []], type=pa.list_(pa.int64())),
"floats": pa.array([[1.5, None], [2.0]], type=pa.list_(pa.float64())),
"strings": pa.array([["a", "b"], ["c"]], type=pa.list_(pa.string())),
}
)
description = [
("ints", "array", None, None, None, None, None),
("floats", "array", None, None, None, None, None),
("strings", "array", None, None, None, None, None),
]

rows = _make_result_set(description)._convert_arrow_table(table)

self.assertIsInstance(rows[0].ints, numpy.ndarray)
self.assertEqual(rows[0].ints.dtype, numpy.int64)
self.assertEqual(rows[1].ints.tolist(), [])
self.assertEqual(rows[0].floats.dtype, numpy.float64)
self.assertTrue(numpy.isnan(rows[0].floats[1]))
self.assertIsInstance(rows[1].strings, numpy.ndarray)
self.assertEqual(rows[1].strings.dtype, object)
self.assertEqual(rows[0].strings.tolist(), ["a", "b"])

def test_nested_complex_values_are_exact(self):
big = 9007199254740993
table = pa.table(
{
"id": pa.array([1, 2], type=pa.int64()),
"array_array": pa.array(
[[[big, None], [1]], None], type=pa.list_(pa.list_(pa.int64()))
),
"map_array": pa.array(
[[("a", [big, None])], [("b", None)]],
type=pa.map_(pa.string(), pa.list_(pa.int64())),
),
"struct_col": pa.array(
[{"x": big, "y": [None, 2]}, {"x": None, "y": None}],
type=pa.struct([("x", pa.int64()), ("y", pa.list_(pa.int64()))]),
),
"name": pa.array(["a", None], type=pa.string()),
}
)
description = [
(name, "t", None, None, None, None, None) for name in table.column_names
]

rows = _make_result_set(description)._convert_arrow_table(table)

self.assertEqual([r.id for r in rows], [1, 2])
self.assertEqual([r.name for r in rows], ["a", None])

outer = rows[0].array_array
self.assertIsInstance(outer, numpy.ndarray)
self.assertIsInstance(outer[0], numpy.ndarray)
self.assertEqual(outer[0].tolist(), [big, None])
self.assertEqual(outer[1].dtype, numpy.int64)
self.assertIsNone(rows[1].array_array)

((key, value),) = rows[0].map_array
self.assertEqual(key, "a")
self.assertIsInstance(value, numpy.ndarray)
self.assertEqual(value.tolist(), [big, None])
self.assertEqual(rows[1].map_array, [("b", None)])

self.assertEqual(rows[0].struct_col["x"], big)
self.assertEqual(rows[0].struct_col["y"].tolist(), [None, 2])
self.assertEqual(rows[1].struct_col, {"x": None, "y": None})

def test_only_nested_columns_across_chunks(self):
list_type = pa.list_(pa.int64())
column = pa.chunked_array(
[pa.array([[1]], list_type), pa.array([None, [2, None]], list_type)]
)
table = pa.table({"list_col": column})
description = [("list_col", "array", None, None, None, None, None)]

rows = _make_result_set(description)._convert_arrow_table(table)

self.assertEqual(len(rows), 3)
self.assertEqual(rows[0].list_col.tolist(), [1])
self.assertIsNone(rows[1].list_col)
self.assertEqual(rows[2].list_col.tolist(), [2, None])

def test_struct_type(self):
table = pa.table(
{
Expand Down