From 1c1606c6bd05b46409cde523640f7549fca8edc0 Mon Sep 17 00:00:00 2001 From: Ajayrama Kumaraswamy Date: Mon, 27 Jul 2026 18:26:24 +0200 Subject: [PATCH 1/8] chore: added prek to dev deps --- pyproject.toml | 1 + 1 file changed, 1 insertion(+) diff --git a/pyproject.toml b/pyproject.toml index 03181eadb..14e8040ca 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -64,6 +64,7 @@ extra = [ [dependency-groups] dev = [ "bump2version", + "prek>=0.4.11", ] test = [ "pytest", From 2672c013bba9485028055580519eb3bd4dbebc1c Mon Sep 17 00:00:00 2001 From: Ajayrama Kumaraswamy Date: Mon, 27 Jul 2026 18:28:03 +0200 Subject: [PATCH 2/8] feat: writing table to zarr using AnnData.write_zarr for table version 2 --- src/spatialdata/_io/exceptions.py | 16 +++++++++++++ src/spatialdata/_io/io_table.py | 40 +++++++++++++++++++++++++------ 2 files changed, 49 insertions(+), 7 deletions(-) create mode 100644 src/spatialdata/_io/exceptions.py diff --git a/src/spatialdata/_io/exceptions.py b/src/spatialdata/_io/exceptions.py new file mode 100644 index 000000000..d12414451 --- /dev/null +++ b/src/spatialdata/_io/exceptions.py @@ -0,0 +1,16 @@ +from __future__ import annotations + +from ome_zarr.format import Format + + +class FormatVersionUnknownError(ValueError): + """Exception raised when an unknown element format is encountered.""" + + def __init__(self, element_type: str, version_encountered: Format): + self.element_type = element_type + self.version_encountered = version_encountered + self.message = ( + f"Encountered unknown element format version " + f"`{self.version_encountered}` for element of type `{self.element_type}`" + ) + super().__init__(self.message) diff --git a/src/spatialdata/_io/io_table.py b/src/spatialdata/_io/io_table.py index 3eb4b0927..d1a0aedd4 100644 --- a/src/spatialdata/_io/io_table.py +++ b/src/spatialdata/_io/io_table.py @@ -9,6 +9,7 @@ from anndata._io.specs import write_elem as write_adata from ome_zarr.format import Format +from spatialdata._io.exceptions import FormatVersionUnknownError from spatialdata._io.format import ( CurrentTablesFormat, TablesFormats, @@ -62,10 +63,35 @@ def write_table( else: region, region_key, instance_key = (None, None, None) - write_adata(group, name, table) - tables_group = group[name] - tables_group.attrs["spatialdata-encoding-type"] = group_type - tables_group.attrs["region"] = region - tables_group.attrs["region_key"] = region_key - tables_group.attrs["instance_key"] = instance_key - tables_group.attrs["version"] = element_format.spatialdata_format_version + # Ensure the table group exists + table_group = group.require_group(name=name) + + assert element_format in TablesFormats.values(), FormatVersionUnknownError( + element_type="table", version_encountered=element_format + ) + + if element_format == TablesFormatV02(): + # solution of passing path directly roughly based on: + # https://github.com/scverse/anndata/issues/1548#issuecomment-2199801855 + + # Write the table to the path of the table group + table.write_zarr(store=str(table_group.store_path), consolidate_metadata=False) + # anndata writes to zarr v3 by default, no way to specify, breaks our support for zarr v2 + # hence the workaround with if-else ladder + group = zarr.open_group(group.store_path, mode="a", use_consolidated=False) + table_group = group[name] + elif element_format == TablesFormatV01(): + write_adata(group, name, table) + table_group = group[name] + else: + raise NotImplementedError( + "This should be unreachable, please raise an issue on Github with this error message " + "and a minimum example that works standalone" + ) + # should be unreachable + + table_group.attrs["spatialdata-encoding-type"] = group_type + table_group.attrs["region"] = region + table_group.attrs["region_key"] = region_key + table_group.attrs["instance_key"] = instance_key + table_group.attrs["version"] = element_format.spatialdata_format_version From a88accbdf8662986d9c992dc8d88f90259e28fcf Mon Sep 17 00:00:00 2001 From: Ajayrama Kumaraswamy Date: Tue, 28 Jul 2026 14:22:19 +0200 Subject: [PATCH 3/8] chore: added ruff to dev deps --- pyproject.toml | 1 + 1 file changed, 1 insertion(+) diff --git a/pyproject.toml b/pyproject.toml index 14e8040ca..cc2b573c4 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -65,6 +65,7 @@ extra = [ dev = [ "bump2version", "prek>=0.4.11", + "ruff>=0.16.0", ] test = [ "pytest", From 26a928fa54f8b354aefcbfd41416838a6ce21e10 Mon Sep 17 00:00:00 2001 From: Ajayrama Kumaraswamy Date: Tue, 28 Jul 2026 15:14:37 +0200 Subject: [PATCH 4/8] fix: using internal resolve store function + categorical when writing tables to zarr v2 --- src/spatialdata/_io/io_table.py | 9 +++++++-- 1 file changed, 7 insertions(+), 2 deletions(-) diff --git a/src/spatialdata/_io/io_table.py b/src/spatialdata/_io/io_table.py index d1a0aedd4..8384a4704 100644 --- a/src/spatialdata/_io/io_table.py +++ b/src/spatialdata/_io/io_table.py @@ -9,6 +9,7 @@ from anndata._io.specs import write_elem as write_adata from ome_zarr.format import Format +from spatialdata._io._utils import _resolve_zarr_store from spatialdata._io.exceptions import FormatVersionUnknownError from spatialdata._io.format import ( CurrentTablesFormat, @@ -74,13 +75,17 @@ def write_table( # solution of passing path directly roughly based on: # https://github.com/scverse/anndata/issues/1548#issuecomment-2199801855 + # resolve the store from the group + # needed by `AnnData.write_zarr` below to directly write into the path of the group + resolved_store = _resolve_zarr_store(table_group) + # Write the table to the path of the table group - table.write_zarr(store=str(table_group.store_path), consolidate_metadata=False) + table.write_zarr(store=resolved_store, consolidate_metadata=False) # anndata writes to zarr v3 by default, no way to specify, breaks our support for zarr v2 # hence the workaround with if-else ladder - group = zarr.open_group(group.store_path, mode="a", use_consolidated=False) table_group = group[name] elif element_format == TablesFormatV01(): + table.strings_to_categoricals() write_adata(group, name, table) table_group = group[name] else: From 247cdc699bb173267f4adcf444cdf83f11d05888 Mon Sep 17 00:00:00 2001 From: Ajayrama Kumaraswamy Date: Tue, 28 Jul 2026 19:32:56 +0200 Subject: [PATCH 5/8] fix: test no longer expectes 'nan' after table round trip instead of pd.NA/np.nan --- tests/io/test_readwrite.py | 9 ++------- 1 file changed, 2 insertions(+), 7 deletions(-) diff --git a/tests/io/test_readwrite.py b/tests/io/test_readwrite.py index 034c01d37..7a8d403ee 100644 --- a/tests/io/test_readwrite.py +++ b/tests/io/test_readwrite.py @@ -16,7 +16,6 @@ import zarr from anndata import AnnData from numpy.random import default_rng -from packaging.version import Version from shapely import MultiPolygon, Polygon from upath import UPath from xarray import DataArray @@ -1294,8 +1293,7 @@ def test_sdata_with_nan_in_obs(tmp_path: Path) -> None: Regression test for https://github.com/scverse/spatialdata/issues/399 Previously this raised TypeError: expected unicode string, found nan. - Now the write succeeds, though NaN values in object-dtype columns are - converted to the string "nan" after round-trip. + Now the write succeeds, and NaN values are preserved round trip """ from spatialdata.models import TableModel @@ -1329,8 +1327,5 @@ def test_sdata_with_nan_in_obs(tmp_path: Path) -> None: assert r1.iloc[0] == "string" assert r2.iloc[1] == 3 - if Version(pd.__version__) >= Version("3"): - assert pd.isna(r1.iloc[1]) - else: # After round-trip, NaN in object-dtype column becomes string "nan" on pandas 2 - assert r1.iloc[1] == "nan" + assert pd.isna(r1.iloc[1]) assert np.isnan(r2.iloc[0]) From 75dcba925de716fada9f2c83093a4821ef52c05f Mon Sep 17 00:00:00 2001 From: Ajayrama Kumaraswamy Date: Tue, 28 Jul 2026 19:35:00 +0200 Subject: [PATCH 6/8] feat: added hatch env configs for testing version combinations of anndata/pandas --- pyproject.toml | 21 +++++++++++++++++++++ 1 file changed, 21 insertions(+) diff --git a/pyproject.toml b/pyproject.toml index cc2b573c4..5104a9fba 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -246,3 +246,24 @@ memray-flame = "memray flamegraph --temporal" [tool.pixi.environments] profiling = { features = ["profiling"], solve-group = "default" } + +[tool.hatch.envs.test] +dependency-groups = ["test"] + +[tool.hatch.envs.test-anndata-pandas] +template = "test" +extra-dependencies = ["zarr>=3"] +scripts.test-readwrite = ["pip list|grep anndata && pip list|grep pandas && pytest tests/io/test_readwrite.py"] +scripts.test-all = ["pip list|grep anndata && pip list|grep pandas && pytest ."] + +[[tool.hatch.envs.test-anndata-pandas.matrix]] +anndata-pandas = ["0.13-2", "0.13-3"] + +[tool.hatch.envs.test-anndata-pandas.overrides] +matrix.anndata-pandas.extra-dependencies = [ + # every option when if is True gets included + {value="anndata~=0.13", if = ["0.13-2"]}, + {value="pandas>=2.3,<3", if = ["0.13-2"]}, + {value="anndata~=0.13", if = ["0.13-3"]}, + {value="pandas~=3.0", if = ["0.13-3"]}, +] From 129bf9e315390998aa5a7269a30f62da3fc0989a Mon Sep 17 00:00:00 2001 From: Ajayrama Kumaraswamy Date: Fri, 31 Jul 2026 16:49:30 +0200 Subject: [PATCH 7/8] feat: update hatch config for testing pandas/anndata versions --- pyproject.toml | 7 +++++-- 1 file changed, 5 insertions(+), 2 deletions(-) diff --git a/pyproject.toml b/pyproject.toml index 5104a9fba..91139c400 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -253,17 +253,20 @@ dependency-groups = ["test"] [tool.hatch.envs.test-anndata-pandas] template = "test" extra-dependencies = ["zarr>=3"] +scripts.test = ["pip list|grep anndata && pip list|grep pandas && pytest {args}"] scripts.test-readwrite = ["pip list|grep anndata && pip list|grep pandas && pytest tests/io/test_readwrite.py"] scripts.test-all = ["pip list|grep anndata && pip list|grep pandas && pytest ."] [[tool.hatch.envs.test-anndata-pandas.matrix]] -anndata-pandas = ["0.13-2", "0.13-3"] +anndata-pandas = ["0.13-2", "0.13-3", "0.12-2"] [tool.hatch.envs.test-anndata-pandas.overrides] matrix.anndata-pandas.extra-dependencies = [ - # every option when if is True gets included + # every option where the if-condition is True gets included {value="anndata~=0.13", if = ["0.13-2"]}, {value="pandas>=2.3,<3", if = ["0.13-2"]}, {value="anndata~=0.13", if = ["0.13-3"]}, {value="pandas~=3.0", if = ["0.13-3"]}, + {value="anndata>=0.12,<0.13", if = ["0.12-2"]}, + {value="pandas>=2.3,<3", if = ["0.12-2"]}, ] From 5f4afca3fd813e342ec8ac50e87b38ae88092207 Mon Sep 17 00:00:00 2001 From: Ajayrama Kumaraswamy Date: Fri, 31 Jul 2026 16:52:19 +0200 Subject: [PATCH 8/8] feat: added DeprecationWarnings when writing to zarr v2 --- src/spatialdata/_io/exceptions.py | 10 ++++++++++ src/spatialdata/_io/io_points.py | 6 ++++++ src/spatialdata/_io/io_raster.py | 12 ++++++++++++ src/spatialdata/_io/io_shapes.py | 7 +++++++ src/spatialdata/_io/io_table.py | 26 +++++++++++++------------- 5 files changed, 48 insertions(+), 13 deletions(-) diff --git a/src/spatialdata/_io/exceptions.py b/src/spatialdata/_io/exceptions.py index d12414451..66f5802b7 100644 --- a/src/spatialdata/_io/exceptions.py +++ b/src/spatialdata/_io/exceptions.py @@ -14,3 +14,13 @@ def __init__(self, element_type: str, version_encountered: Format): f"`{self.version_encountered}` for element of type `{self.element_type}`" ) super().__init__(self.message) + + +class WritingToZarrV2DeprecationWarning(DeprecationWarning): + """Warning raised when writing to zarr v2 format.""" + + message = ( + "Writing to zarr v2 format is currently deprecated in spatialdata " + "and will be removed in a future version. " + "Please consider writing to zarr v3." + ) diff --git a/src/spatialdata/_io/io_points.py b/src/spatialdata/_io/io_points.py index 03ef33389..bb203cad2 100644 --- a/src/spatialdata/_io/io_points.py +++ b/src/spatialdata/_io/io_points.py @@ -1,5 +1,6 @@ from __future__ import annotations +import warnings from pathlib import Path import zarr @@ -12,6 +13,7 @@ _write_metadata, overwrite_coordinate_transformations_non_raster, ) +from spatialdata._io.exceptions import WritingToZarrV2DeprecationWarning from spatialdata._io.format import CurrentPointsFormat, PointsFormats, _parse_version from spatialdata.models import get_axes_names from spatialdata.transformations._utils import ( @@ -65,6 +67,10 @@ def write_points( element_format The format of the points element used to store it. """ + if element_format.zarr_format == 2: + warnings.warn( + message=WritingToZarrV2DeprecationWarning.message, category=WritingToZarrV2DeprecationWarning, stacklevel=2 + ) axes = get_axes_names(points) transformations = _get_transformations(points) assert transformations is not None # mypy: validate_element() in _write_element guarantees this diff --git a/src/spatialdata/_io/io_raster.py b/src/spatialdata/_io/io_raster.py index 276f016bd..b9a2964f0 100644 --- a/src/spatialdata/_io/io_raster.py +++ b/src/spatialdata/_io/io_raster.py @@ -1,5 +1,6 @@ from __future__ import annotations +import warnings from collections.abc import Sequence from pathlib import Path from typing import Any, Literal, TypeGuard, cast @@ -23,6 +24,7 @@ overwrite_channel_names, overwrite_coordinate_transformations_raster, ) +from spatialdata._io.exceptions import WritingToZarrV2DeprecationWarning from spatialdata._io.format import ( CurrentRasterFormat, RasterFormatType, @@ -581,6 +583,11 @@ def write_image( raster_compressor: dict[Literal["lz4", "zstd"], int] | None = None, **metadata: str | JSONDict | list[JSONDict], ) -> None: + if element_format.zarr_format == 2: + warnings.warn( + message=WritingToZarrV2DeprecationWarning.message, category=WritingToZarrV2DeprecationWarning, stacklevel=2 + ) + _write_raster( raster_type="image", raster_data=image, @@ -603,6 +610,11 @@ def write_labels( raster_compressor: dict[Literal["lz4", "zstd"], int] | None = None, **metadata: JSONDict, ) -> None: + if element_format.zarr_format == 2: + warnings.warn( + message=WritingToZarrV2DeprecationWarning.message, category=WritingToZarrV2DeprecationWarning, stacklevel=2 + ) + _write_raster( raster_type="labels", raster_data=labels, diff --git a/src/spatialdata/_io/io_shapes.py b/src/spatialdata/_io/io_shapes.py index 3b6e18e39..f8528868d 100644 --- a/src/spatialdata/_io/io_shapes.py +++ b/src/spatialdata/_io/io_shapes.py @@ -1,5 +1,6 @@ from __future__ import annotations +import warnings from pathlib import Path from typing import Any, Literal @@ -15,6 +16,7 @@ _write_metadata, overwrite_coordinate_transformations_non_raster, ) +from spatialdata._io.exceptions import WritingToZarrV2DeprecationWarning from spatialdata._io.format import ( CurrentShapesFormat, ShapesFormats, @@ -93,6 +95,11 @@ def write_shapes( Whether to use the WKB or geoarrow encoding for GeoParquet. See :meth:`geopandas.GeoDataFrame.to_parquet` for details. If None, uses the value from :attr:`spatialdata.settings.shapes_geometry_encoding`. """ + if element_format.zarr_format == 2: + warnings.warn( + message=WritingToZarrV2DeprecationWarning.message, category=WritingToZarrV2DeprecationWarning, stacklevel=2 + ) + from spatialdata.config import settings if geometry_encoding is None: diff --git a/src/spatialdata/_io/io_table.py b/src/spatialdata/_io/io_table.py index 8384a4704..da6ef9b5c 100644 --- a/src/spatialdata/_io/io_table.py +++ b/src/spatialdata/_io/io_table.py @@ -1,5 +1,7 @@ from __future__ import annotations +import warnings +from importlib.metadata import version from pathlib import Path import numpy as np @@ -10,7 +12,7 @@ from ome_zarr.format import Format from spatialdata._io._utils import _resolve_zarr_store -from spatialdata._io.exceptions import FormatVersionUnknownError +from spatialdata._io.exceptions import FormatVersionUnknownError, WritingToZarrV2DeprecationWarning from spatialdata._io.format import ( CurrentTablesFormat, TablesFormats, @@ -58,6 +60,11 @@ def write_table( group_type: str = "ngff:regions_table", element_format: Format = CurrentTablesFormat(), ) -> None: + if element_format.zarr_format == 2: + warnings.warn( + message=WritingToZarrV2DeprecationWarning.message, category=WritingToZarrV2DeprecationWarning, stacklevel=2 + ) + if TableModel.ATTRS_KEY in table.uns: region, region_key, instance_key = get_table_keys(table) TableModel.validate(table) @@ -71,29 +78,22 @@ def write_table( element_type="table", version_encountered=element_format ) - if element_format == TablesFormatV02(): - # solution of passing path directly roughly based on: + if element_format.zarr_format == 3 and version("anndata") >= "0.13": + # `write_zarr` in anndata v0.13 and above can only write to zarr v3 + # solution of passing resolved store directly roughly based on: # https://github.com/scverse/anndata/issues/1548#issuecomment-2199801855 # resolve the store from the group - # needed by `AnnData.write_zarr` below to directly write into the path of the group resolved_store = _resolve_zarr_store(table_group) # Write the table to the path of the table group table.write_zarr(store=resolved_store, consolidate_metadata=False) - # anndata writes to zarr v3 by default, no way to specify, breaks our support for zarr v2 - # hence the workaround with if-else ladder + table_group = group[name] - elif element_format == TablesFormatV01(): + else: table.strings_to_categoricals() write_adata(group, name, table) table_group = group[name] - else: - raise NotImplementedError( - "This should be unreachable, please raise an issue on Github with this error message " - "and a minimum example that works standalone" - ) - # should be unreachable table_group.attrs["spatialdata-encoding-type"] = group_type table_group.attrs["region"] = region