diff --git a/CHANGELOG.md b/CHANGELOG.md index bce9e16e4..cf16662c8 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -1,5 +1,8 @@ # Release History +# Unreleased +- Fix: with pandas enabled (the default), complex-type values no longer lose precision: an ARRAY of integers containing a NULL (at any nesting level, including inside MAP and STRUCT values) is returned as an object `numpy.ndarray` of exact `int`/`None` instead of float64 with NaN, and nested columns are converted from Arrow directly instead of through pandas. Return types are unchanged: ARRAY is a `numpy.ndarray`, MAP a list of tuples, STRUCT a dict. + # 4.6.0 (2026-09-24) - Upgrade Databricks SQL Kernel to 1.1.0; the kernel dependency is now stable and no longer experimental. - Transparently auto-recover Thrift connections to Reyden / Real-Time warehouses: when a warehouse rejects the default Thrift protocol (SQLSTATE `KP001`), the session is re-opened on the kernel backend and the warehouse is remembered so later connections skip Thrift. Applies only when no backend was chosen explicitly. diff --git a/src/databricks/sql/result_set.py b/src/databricks/sql/result_set.py index f24a07505..2dcca3c44 100644 --- a/src/databricks/sql/result_set.py +++ b/src/databricks/sql/result_set.py @@ -4,6 +4,7 @@ from typing import List, Optional, TYPE_CHECKING, Tuple import logging +import numpy import pandas try: @@ -121,16 +122,46 @@ def _convert_arrow_table(self, table): pyarrow.string(): pandas.StringDtype(), } - # Need to rename columns, as the to_pandas function cannot handle duplicate column names - table_renamed = table.rename_columns([str(c) for c in range(table.num_columns)]) - df = table_renamed.to_pandas( - types_mapper=dtype_mapping.get, - date_as_object=True, - timestamp_as_object=True, - ) - - res = df.to_numpy(na_value=None, dtype="object") - return [ResultRow(*v) for v in res] + # Nested (ARRAY/MAP/STRUCT) columns are converted from Arrow directly + # and never pass through pandas: pandas turns an integer array + # containing a NULL into float64 (precision loss beyond 2**53, NULL as + # NaN). See _nested_column_to_python for the returned shapes. + nested = { + index: _nested_column_to_python(table.column(index)) + for index, field in enumerate(table.schema) + if pyarrow.types.is_nested(field.type) + } + scalar_indices = [i for i in range(table.num_columns) if i not in nested] + + res = None + if scalar_indices: + scalar_table = table.select(scalar_indices) if nested else table + # Need to rename columns, as the to_pandas function cannot handle duplicate column names + scalar_table = scalar_table.rename_columns( + [str(c) for c in range(scalar_table.num_columns)] + ) + df = scalar_table.to_pandas( + types_mapper=dtype_mapping.get, + date_as_object=True, + timestamp_as_object=True, + ) + res = df.to_numpy(na_value=None, dtype="object") + + if not nested: + return [ResultRow(*v) for v in res] + + rows = [] + for position in range(table.num_rows): + scalars = iter(res[position]) if res is not None else iter(()) + rows.append( + ResultRow( + *[ + nested[index][position] if index in nested else next(scalars) + for index in range(table.num_columns) + ] + ) + ) + return rows @property def rownumber(self): @@ -471,3 +502,73 @@ def map_col_type(type_): (column.name, map_col_type(column.datatype), None, None, None, None, None) for column in table_schema_message.columns ] + + +def _object_array(values: list) -> "numpy.ndarray": + """A 1-D object ndarray holding ``values`` as-is (no numpy broadcasting).""" + out = numpy.empty(len(values), dtype=object) + for i, value in enumerate(values): + out[i] = value + return out + + +def _arrow_array_to_python(array) -> list: + """Convert one Arrow array to a list of Python values, one per slot. + + Shapes match what the pandas conversion has always returned for complex + types (see ``_use_arrow_native_complex_types``): ARRAY is a + ``numpy.ndarray``, MAP is a list of ``(key, value)`` tuples, STRUCT is a + dict, NULL is None. The difference is that integer/boolean array elements + are exact: an ARRAY without NULLs is an int64 ndarray as before, + and one with NULLs is an object ndarray of ``int``/``None`` instead of a + float64 ndarray with NaN. + """ + t = array.type + if pyarrow.types.is_map(t): + result: list = [] + for scalar in array: + if not scalar.is_valid: + result.append(None) + continue + keys, items = scalar.values.flatten() + result.append( + list(zip(_arrow_array_to_python(keys), _arrow_array_to_python(items))) + ) + return result + if ( + pyarrow.types.is_list(t) + or pyarrow.types.is_large_list(t) + or pyarrow.types.is_fixed_size_list(t) + ): + return [ + _arrow_list_values_to_ndarray(scalar.values) if scalar.is_valid else None + for scalar in array + ] + if pyarrow.types.is_struct(t): + names = [t.field(i).name for i in range(t.num_fields)] + children = [_arrow_array_to_python(child) for child in array.flatten()] + return [ + dict(zip(names, (child[i] for child in children))) if valid else None + for i, valid in enumerate(array.is_valid().to_pylist()) + ] + return array.to_pylist() + + +def _arrow_list_values_to_ndarray(values) -> "numpy.ndarray": + """The elements of one ARRAY value as a numpy.ndarray.""" + t = values.type + if pyarrow.types.is_floating(t) or ( + (pyarrow.types.is_integer(t) or pyarrow.types.is_boolean(t)) + and values.null_count == 0 + ): + # Native dtype, as pandas produced; floats keep NaN for NULL. + return values.to_numpy(zero_copy_only=False, writable=True) + return _object_array(_arrow_array_to_python(values)) + + +def _nested_column_to_python(column) -> list: + """Convert a (chunked) nested Arrow column to one Python value per row.""" + result: list = [] + for chunk in column.chunks: + result.extend(_arrow_array_to_python(chunk)) + return result diff --git a/tests/unit/test_pandas_compatibility.py b/tests/unit/test_pandas_compatibility.py index 5434559b2..f9f5eb9d0 100644 --- a/tests/unit/test_pandas_compatibility.py +++ b/tests/unit/test_pandas_compatibility.py @@ -11,6 +11,7 @@ from decimal import Decimal from unittest.mock import Mock +import numpy import pandas import pytest @@ -294,6 +295,112 @@ def test_list_type(self): self.assertIsNone(rows[1].list_col) self.assertEqual(list(rows[2].list_col), [4, 5]) + def test_list_of_bigint_with_null_is_exact(self): + """Nested integers must not go through numpy float64.""" + big = 9007199254740993 # 2**53 + 1: not representable as float64 + table = pa.table( + { + "list_col": pa.array( + [[big, -9223372036854775808, None]], type=pa.list_(pa.int64()) + ), + } + ) + description = [("list_col", "array", None, None, None, None, None)] + + rows = _make_result_set(description)._convert_arrow_table(table) + + # Still a numpy.ndarray (the documented ARRAY type), but object dtype + # holding exact ints and None rather than float64 with NaN. + self.assertIsInstance(rows[0].list_col, numpy.ndarray) + values = list(rows[0].list_col) + self.assertEqual(values, [big, -9223372036854775808, None]) + self.assertTrue(all(type(v) is int for v in values[:2])) + + def test_list_without_null_keeps_native_dtype(self): + table = pa.table( + { + "ints": pa.array([[1, 2], []], type=pa.list_(pa.int64())), + "floats": pa.array([[1.5, None], [2.0]], type=pa.list_(pa.float64())), + "strings": pa.array([["a", "b"], ["c"]], type=pa.list_(pa.string())), + } + ) + description = [ + ("ints", "array", None, None, None, None, None), + ("floats", "array", None, None, None, None, None), + ("strings", "array", None, None, None, None, None), + ] + + rows = _make_result_set(description)._convert_arrow_table(table) + + self.assertIsInstance(rows[0].ints, numpy.ndarray) + self.assertEqual(rows[0].ints.dtype, numpy.int64) + self.assertEqual(rows[1].ints.tolist(), []) + self.assertEqual(rows[0].floats.dtype, numpy.float64) + self.assertTrue(numpy.isnan(rows[0].floats[1])) + self.assertIsInstance(rows[1].strings, numpy.ndarray) + self.assertEqual(rows[1].strings.dtype, object) + self.assertEqual(rows[0].strings.tolist(), ["a", "b"]) + + def test_nested_complex_values_are_exact(self): + big = 9007199254740993 + table = pa.table( + { + "id": pa.array([1, 2], type=pa.int64()), + "array_array": pa.array( + [[[big, None], [1]], None], type=pa.list_(pa.list_(pa.int64())) + ), + "map_array": pa.array( + [[("a", [big, None])], [("b", None)]], + type=pa.map_(pa.string(), pa.list_(pa.int64())), + ), + "struct_col": pa.array( + [{"x": big, "y": [None, 2]}, {"x": None, "y": None}], + type=pa.struct([("x", pa.int64()), ("y", pa.list_(pa.int64()))]), + ), + "name": pa.array(["a", None], type=pa.string()), + } + ) + description = [ + (name, "t", None, None, None, None, None) for name in table.column_names + ] + + rows = _make_result_set(description)._convert_arrow_table(table) + + self.assertEqual([r.id for r in rows], [1, 2]) + self.assertEqual([r.name for r in rows], ["a", None]) + + outer = rows[0].array_array + self.assertIsInstance(outer, numpy.ndarray) + self.assertIsInstance(outer[0], numpy.ndarray) + self.assertEqual(outer[0].tolist(), [big, None]) + self.assertEqual(outer[1].dtype, numpy.int64) + self.assertIsNone(rows[1].array_array) + + ((key, value),) = rows[0].map_array + self.assertEqual(key, "a") + self.assertIsInstance(value, numpy.ndarray) + self.assertEqual(value.tolist(), [big, None]) + self.assertEqual(rows[1].map_array, [("b", None)]) + + self.assertEqual(rows[0].struct_col["x"], big) + self.assertEqual(rows[0].struct_col["y"].tolist(), [None, 2]) + self.assertEqual(rows[1].struct_col, {"x": None, "y": None}) + + def test_only_nested_columns_across_chunks(self): + list_type = pa.list_(pa.int64()) + column = pa.chunked_array( + [pa.array([[1]], list_type), pa.array([None, [2, None]], list_type)] + ) + table = pa.table({"list_col": column}) + description = [("list_col", "array", None, None, None, None, None)] + + rows = _make_result_set(description)._convert_arrow_table(table) + + self.assertEqual(len(rows), 3) + self.assertEqual(rows[0].list_col.tolist(), [1]) + self.assertIsNone(rows[1].list_col) + self.assertEqual(rows[2].list_col.tolist(), [2, None]) + def test_struct_type(self): table = pa.table( {