From 0c3a7c7b4d0e6d8ad83c0bcc7f255ed714817f62 Mon Sep 17 00:00:00 2001 From: Ray Walker Date: Sat, 3 Oct 2026 04:50:52 +1000 Subject: [PATCH 1/3] fix(arrow): reject dict values that pyarrow stores as the wrong column (LAB-7490) pa.table iterates every dict value as a column, so a str value became a column of its characters, bytes or bytearray a column of byte values, and a dict a column of its keys with its values dropped, all without an error. Through @cache the next call then returned that wrong DataFrame as a hit. _to_table now raises TypeError for str, bytes, bytearray and Mapping values before calling pa.table. Both serialize() and serialize_to_sink() go through _to_table, so both paths reject them, and the streaming path writes nothing to the sink. Dicts of lists, NumPy arrays, pandas Series and pyarrow arrays are unaffected. The test that asserted a nested dict serializes is replaced by tests for each rejected and accepted shape on both paths, plus a decorator test on a File backend. docs/serializers/arrow.md lists these values as unsupported and its type-checking example quotes the message the serializer raises. --- docs/serializers/arrow.md | 5 +- src/cachekit/serializers/arrow_serializer.py | 6 ++ tests/unit/test_arrow_serializer.py | 88 ++++++++++++++++++-- 3 files changed, 89 insertions(+), 10 deletions(-) diff --git a/docs/serializers/arrow.md b/docs/serializers/arrow.md index 271df747..9465312c 100644 --- a/docs/serializers/arrow.md +++ b/docs/serializers/arrow.md @@ -99,8 +99,9 @@ ArrowSerializer supports: - Scalar values (int, str, float) - Lists of objects - Dicts with a number or bool value (`{"id": 1}`) +- Dicts with a string, bytes or dict value (`{"name": "Alice"}`, `{"user": {"name": "x"}}`) -Every dict value must be a list or an array. A string or dict value is not rejected, but it is stored wrong: a string becomes a column of its characters, and a nested dict a column of its keys, with its values lost. Flatten nested dicts into columns first, or use [AutoSerializer](./auto.md). +Every dict value must be a list or an array. Flatten nested dicts into columns first, or use [AutoSerializer](./auto.md). **Type checking example:** ```python @@ -117,7 +118,7 @@ try: serializer.serialize({"key": "value"}) except TypeError as e: print(e) - # "ArrowSerializer only supports DataFrames. Use StandardSerializer for dict types." + # "... Got a dict that is not convertible to an Arrow table: value for 'key' is str, not a list or array. ..." ``` ## Performance Benchmarks diff --git a/src/cachekit/serializers/arrow_serializer.py b/src/cachekit/serializers/arrow_serializer.py index 69b60e45..67c96f8d 100644 --- a/src/cachekit/serializers/arrow_serializer.py +++ b/src/cachekit/serializers/arrow_serializer.py @@ -22,6 +22,7 @@ from __future__ import annotations +from collections.abc import Mapping from typing import TYPE_CHECKING, Any, BinaryIO, ClassVar from .base import SerializationError, SerializationFormat, SerializationMetadata @@ -263,6 +264,11 @@ def _to_table(obj: Any) -> pa.Table: # type: ignore[name-defined] # (e.g. dict-of-scalars -> "'int' object is not iterable") into the # documented TypeError so callers get a consistent, actionable message. try: + for name, column in obj.items(): + # pa.table iterates these instead of rejecting them, storing a wrong column: + # a str as its characters, bytes as byte values, a dict as its keys only. + if isinstance(column, (str, bytes, bytearray, Mapping)): + raise TypeError(f"value for {name!r} is {type(column).__name__}, not a list or array") return pa.table(obj) except (pa.ArrowInvalid, pa.ArrowTypeError, TypeError, ValueError) as e: raise TypeError( diff --git a/tests/unit/test_arrow_serializer.py b/tests/unit/test_arrow_serializer.py index be67a07a..42ec679c 100644 --- a/tests/unit/test_arrow_serializer.py +++ b/tests/unit/test_arrow_serializer.py @@ -5,17 +5,29 @@ from __future__ import annotations +import io + import pytest # ArrowSerializer requires the [data] extra — absent e.g. in the free-threaded # CI lane until pandas/pyarrow ship free-threaded wheels (LAB-511). pd = pytest.importorskip("pandas") pa = pytest.importorskip("pyarrow") +np = pytest.importorskip("numpy") from cachekit.serializers.arrow_serializer import ArrowSerializer # noqa: E402 from cachekit.serializers.base import SerializationError, SerializationFormat, SerializationMetadata # noqa: E402 +def _serialize_via(path: str, serializer: ArrowSerializer, obj: object) -> tuple[bytes, SerializationMetadata]: + """Serialize through the buffered (``serialize``) or streaming (``serialize_to_sink``) path.""" + if path == "buffered": + return serializer.serialize(obj) + sink = io.BytesIO() + metadata = serializer.serialize_to_sink(obj, sink) + return sink.getvalue(), metadata + + class TestArrowSerializerBasics: """Test basic ArrowSerializer functionality.""" @@ -205,6 +217,21 @@ def test_dict_of_arrays_with_arrow_return_format(self): assert isinstance(result, pa.Table) assert result.column_names == ["a", "b"] + @pytest.mark.parametrize("path", ["buffered", "streaming"]) + @pytest.mark.parametrize( + "make_column", + [list, np.array, pd.Series, pa.array], + ids=["list", "numpy", "series", "pyarrow"], + ) + def test_dict_of_columns_round_trips_on_both_paths(self, path, make_column): + """Every accepted column container round-trips, buffered and streamed.""" + serializer = ArrowSerializer() + data_dict = {"n": make_column([1, 2, 3]), "s": make_column(["x", "y", "z"])} + + data, metadata = _serialize_via(path, serializer, data_dict) + + assert serializer.deserialize(data, metadata).to_dict("list") == {"n": [1, 2, 3], "s": ["x", "y", "z"]} + class TestErrorHandling: """Test error handling for unsupported types and corrupted data.""" @@ -221,16 +248,61 @@ def test_scalar_value_raises_type_error(self): assert "Got: int" in error_msg assert "For scalar values or nested dicts, use AutoSerializer" in error_msg - def test_non_columnar_dict_successfully_serialized(self): - """Arrow can handle certain dict structures (converts to struct/list types).""" + @pytest.mark.parametrize("path", ["buffered", "streaming"]) + @pytest.mark.parametrize( + ("obj", "detail"), + [ + ({"name": "Alice"}, "value for 'name' is str"), + ({"b": b"ab"}, "value for 'b' is bytes"), + ({"b": bytearray(b"ab")}, "value for 'b' is bytearray"), + ({"user": {"name": "x"}}, "value for 'user' is dict"), + ({"ok": [1, 2], "name": "Alice"}, "value for 'name' is str"), + ], + ids=["str", "bytes", "bytearray", "dict", "one-bad-column"], + ) + def test_non_column_dict_value_raises_type_error(self, path, obj, detail): + """pyarrow would iterate these values into a wrong column (a str into its characters, + bytes into their byte values, a dict into its keys), so they are rejected instead.""" serializer = ArrowSerializer() - nested = {"key": {"nested": "value"}} - # Arrow will convert this successfully (struct/list types) - # This is actually valid - Arrow has flexible schema support - data, metadata = serializer.serialize(nested) - assert isinstance(data, bytes) - assert isinstance(metadata, SerializationMetadata) + with pytest.raises(TypeError) as exc_info: + _serialize_via(path, serializer, obj) + + error_msg = str(exc_info.value) + assert "Got a dict that is not convertible to an Arrow table" in error_msg + assert f"{detail}, not a list or array" in error_msg + + def test_rejected_dict_writes_nothing_to_sink(self): + """The streaming path rejects the value before writing a byte.""" + sink = io.BytesIO() + + with pytest.raises(TypeError): + ArrowSerializer().serialize_to_sink({"name": "Alice"}, sink) + + assert sink.getvalue() == b"" + + def test_decorator_reruns_instead_of_returning_a_wrong_hit(self, tmp_path): + """Through @cache with a File backend, a rejected dict is never cached, so the + second call runs the function and returns the original dict, not a DataFrame.""" + from cachekit import cache + from cachekit.backends.file import FileBackend + from cachekit.backends.file.config import FileBackendConfig + + backend = FileBackend(FileBackendConfig(cache_dir=str(tmp_path), max_size_mb=10, max_value_mb=5)) + calls = 0 + + @cache(serializer="arrow", backend=backend, ttl=60) + def load_user() -> dict: + nonlocal calls + calls += 1 + return {"name": "Alice"} + + assert load_user() == {"name": "Alice"} + second = load_user() + + assert calls == 2 + assert isinstance(second, dict) + assert second == {"name": "Alice"} def test_string_raises_type_error(self): """String value raises TypeError.""" From 44b559d5cb2e78e6700accb157bc3247b3e48fa7 Mon Sep 17 00:00:00 2001 From: Ray Walker Date: Sat, 3 Oct 2026 05:58:33 +1000 Subject: [PATCH 2/3] fix(arrow): keep accepting a Mapping that is an Arrow column adapter (LAB-7490) pa.table converts a value through __arrow_array__ or __arrow_c_array__ before it tries to iterate it, so a Mapping that implements either one is stored as the column it describes, not as its keys. The new value guard rejected it anyway, which broke a shape that round-tripped before. Skip the str/bytes/bytearray/Mapping rejection when the value exposes an Arrow array protocol. A test round-trips a Mapping adapter for each protocol on both the buffered and streaming paths; its keys differ from its column, so a wrong conversion would fail the test. --- src/cachekit/serializers/arrow_serializer.py | 5 ++- tests/unit/test_arrow_serializer.py | 39 ++++++++++++++++++++ 2 files changed, 43 insertions(+), 1 deletion(-) diff --git a/src/cachekit/serializers/arrow_serializer.py b/src/cachekit/serializers/arrow_serializer.py index 67c96f8d..043a76b9 100644 --- a/src/cachekit/serializers/arrow_serializer.py +++ b/src/cachekit/serializers/arrow_serializer.py @@ -267,7 +267,10 @@ def _to_table(obj: Any) -> pa.Table: # type: ignore[name-defined] for name, column in obj.items(): # pa.table iterates these instead of rejecting them, storing a wrong column: # a str as its characters, bytes as byte values, a dict as its keys only. - if isinstance(column, (str, bytes, bytearray, Mapping)): + # A value with an Arrow array protocol is converted through it, not iterated. + if isinstance(column, (str, bytes, bytearray, Mapping)) and not ( + hasattr(column, "__arrow_array__") or hasattr(column, "__arrow_c_array__") + ): raise TypeError(f"value for {name!r} is {type(column).__name__}, not a list or array") return pa.table(obj) except (pa.ArrowInvalid, pa.ArrowTypeError, TypeError, ValueError) as e: diff --git a/tests/unit/test_arrow_serializer.py b/tests/unit/test_arrow_serializer.py index 42ec679c..1b2bbbd2 100644 --- a/tests/unit/test_arrow_serializer.py +++ b/tests/unit/test_arrow_serializer.py @@ -6,6 +6,7 @@ from __future__ import annotations import io +from collections.abc import Mapping import pytest @@ -28,6 +29,33 @@ def _serialize_via(path: str, serializer: ArrowSerializer, obj: object) -> tuple return sink.getvalue(), metadata +class _KeysMapping(Mapping): + """A Mapping whose iteration yields its keys, so pa.table alone would store ["k1", "k2"].""" + + def __getitem__(self, key): + raise KeyError(key) + + def __iter__(self): + return iter(["k1", "k2"]) + + def __len__(self): + return 2 + + +class _ArrowArrayMapping(_KeysMapping): + """Also an Arrow column adapter: pyarrow converts it through ``__arrow_array__`` to [10, 20].""" + + def __arrow_array__(self, type=None): + return pa.array([10, 20]) + + +class _ArrowCArrayMapping(_KeysMapping): + """Same adapter through the Arrow PyCapsule protocol (``__arrow_c_array__``).""" + + def __arrow_c_array__(self, requested_schema=None): + return pa.array([10, 20]).__arrow_c_array__(requested_schema) + + class TestArrowSerializerBasics: """Test basic ArrowSerializer functionality.""" @@ -272,6 +300,17 @@ def test_non_column_dict_value_raises_type_error(self, path, obj, detail): assert "Got a dict that is not convertible to an Arrow table" in error_msg assert f"{detail}, not a list or array" in error_msg + @pytest.mark.parametrize("path", ["buffered", "streaming"]) + @pytest.mark.parametrize("adapter", [_ArrowArrayMapping, _ArrowCArrayMapping], ids=["arrow_array", "arrow_c_array"]) + def test_mapping_with_arrow_array_protocol_round_trips(self, path, adapter): + """pyarrow converts a value through its Arrow array protocol rather than iterating it, + so a Mapping that implements one is a correct column and is not rejected.""" + serializer = ArrowSerializer() + + data, metadata = _serialize_via(path, serializer, {"c": adapter()}) + + assert serializer.deserialize(data, metadata).to_dict("list") == {"c": [10, 20]} + def test_rejected_dict_writes_nothing_to_sink(self): """The streaming path rejects the value before writing a byte.""" sink = io.BytesIO() From 9c7685ca660c2e1246390540939d59a435d22ba6 Mon Sep 17 00:00:00 2001 From: Ray Walker Date: Sat, 3 Oct 2026 09:40:39 +1000 Subject: [PATCH 3/3] docs(arrow): describe accepted dict columns as Arrow-convertible (LAB-7490) The sentence said every dict value must be a list or an array, which excluded inputs the serializer accepts and round-trips: tuples, typed memoryviews, and Mapping adapters that implement the Arrow array protocol. --- docs/serializers/arrow.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/docs/serializers/arrow.md b/docs/serializers/arrow.md index 9465312c..8f952891 100644 --- a/docs/serializers/arrow.md +++ b/docs/serializers/arrow.md @@ -101,7 +101,7 @@ ArrowSerializer supports: - Dicts with a number or bool value (`{"id": 1}`) - Dicts with a string, bytes or dict value (`{"name": "Alice"}`, `{"user": {"name": "x"}}`) -Every dict value must be a list or an array. Flatten nested dicts into columns first, or use [AutoSerializer](./auto.md). +Every dict value must be a column Arrow can convert: a list, tuple, NumPy array, pandas Series, pyarrow array or typed `memoryview`, or any object that implements the Arrow array protocol (`__arrow_array__` or `__arrow_c_array__`), a `Mapping` included. A `memoryview` is read as its element type, so a memoryview of `bytes` stores byte values. Flatten nested dicts into columns first, or use [AutoSerializer](./auto.md). **Type checking example:** ```python