From b1b5f22ae92e9f05804606d92fb42494159d2cb2 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Mon, 28 Sep 2026 14:01:15 +0200 Subject: [PATCH 01/94] fix(zarr-metadata): extension points are written as objects, which every reader takes `to_json`, `to_key_value` and `canonicalize` wrote a field with nothing to configure by its bare name: `"crc32c"`, `"default"`. The spec allows that since v3.1, but a Zarr v3.0 reader takes no short-hand name in `codecs` (core L585-L592), and no zarr-python release reads one there or in `chunk_key_encoding`, so documents the package wrote could not be opened by them. Every extension point but a data type is now written as an object, `{"name": ...}`; a data type with nothing to configure keeps its bare name, as core data types have been written since v3.0. Re-writing zarr-python's own documents through the model, zarr-python now opens all of them, where 14 to 22 of 80 failed. Assisted-by: ClaudeCode:claude-opus-5-5 --- packages/zarr-metadata/changes/370.bugfix.md | 13 +++++ .../src/zarr_metadata/model/_array.py | 54 ++++++++++++------- .../src/zarr_metadata/v3/_definition.py | 20 ++++--- .../src/zarr_metadata/v3/definition.py | 8 +-- .../zarr-metadata/tests/model/test_array.py | 27 ++++++++-- .../tests/v3/test_every_definition.py | 24 ++++++--- 6 files changed, 108 insertions(+), 38 deletions(-) create mode 100644 packages/zarr-metadata/changes/370.bugfix.md diff --git a/packages/zarr-metadata/changes/370.bugfix.md b/packages/zarr-metadata/changes/370.bugfix.md new file mode 100644 index 0000000000..f78c3ca9d7 --- /dev/null +++ b/packages/zarr-metadata/changes/370.bugfix.md @@ -0,0 +1,13 @@ +`to_json`, `to_key_value` and `canonicalize` write every extension point +but a data type as an object -- `{"name": "crc32c"}` -- where they wrote +the bare name `"crc32c"` for a field with nothing to configure. A Zarr +v3.0 reader takes no short-hand name in `codecs`, and no zarr-python +release reads one there or in `chunk_key_encoding`, so a document the +package wrote could not be opened by them. A data type with nothing to +configure is still written by its bare name, as core data types are. +A field alone -- `ZarrV3NamedConfig.to_json`, and the pydantic +`ZarrV3MetadataField` -- keeps its shortest spelling, since which one its +readers take depends on the extension point it fills. `to_json` and +`to_key_value` write a configuration as it was read, so a field one holds +-- a shard's inner codecs -- keeps the spelling the document gave it; +`canonicalize` writes those as objects too. diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_array.py b/packages/zarr-metadata/src/zarr_metadata/model/_array.py index 6d3979dba3..bd09044532 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_array.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/_array.py @@ -63,19 +63,20 @@ class ZarrV3NamedConfig: must_understand: bool = True def to_json(self) -> ZarrV3MetadataFieldJSON: + """The field in its shortest spelling: its bare name when it has nothing to configure and must be understood, else an object. + + Which spelling a reader takes depends on the extension point the + field fills, which a field alone does not know: zarr-python reads a + core data type only by its bare name, and a Zarr v3.0 reader takes + no bare name in `codecs` + (https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v3/core/index.rst#L585-L592). + `ZarrV3ArrayMetadata.to_json` knows, and writes each extension point + as its readers take it. The configuration is written as it was + read, a field it holds too. + """ if not self.configuration and self.must_understand: return self.name - # `configuration` is ReadOnly, so it is set in the literal rather than - # assigned afterwards. to_json output shares no mutable state with the - # model. - out: ZarrV3NamedConfigJSON = ( - {"name": self.name, "configuration": copy.deepcopy(self.configuration)} - if self.configuration - else {"name": self.name} - ) - if not self.must_understand: - out["must_understand"] = False - return out + return _object_json(self) @classmethod def from_json(cls, data: object) -> ZarrV3NamedConfig: @@ -91,6 +92,20 @@ def from_json(cls, data: object) -> ZarrV3NamedConfig: ) +def _object_json(field: ZarrV3NamedConfig) -> ZarrV3NamedConfigJSON: + """A field as an object: its name, its configuration if it has one, and `must_understand` only when `false`.""" + # `configuration` is ReadOnly, so it is set in the literal rather than + # assigned afterwards. The output shares no mutable state with the model. + out: ZarrV3NamedConfigJSON = ( + {"name": field.name, "configuration": copy.deepcopy(field.configuration)} + if field.configuration + else {"name": field.name} + ) + if not field.must_understand: + out["must_understand"] = False + return out + + ZarrV3MetadataField: TypeAlias = ZarrV3NamedConfig """The in-memory model of one field of a v3 metadata document. @@ -164,9 +179,12 @@ class ZarrV3ArrayMetadata: `from_key_value` read each through the definition that claims its name in a scope -- `CORE_AND_EXTENSIONS` unless a `context` is passed -- and the model holds what they read as written; `fill_value` is held - verbatim in its JSON form. Equivalent extension - spellings normalize to shorthand strings when configuration is empty and - understanding is required. + verbatim in its JSON form. `to_json` writes each extension point as + its readers take it: the data type in its shortest spelling -- a core + data type by its bare name, as they have been written since Zarr v3.0 + -- and every other as an object, `{"name": ...}`, since a Zarr v3.0 + reader takes no bare name in `codecs`. A configuration is written as + it was read, a field it holds too. """ zarr_format: Literal[3] = field(default=3, init=False) @@ -269,9 +287,9 @@ def to_json(self) -> ZarrV3ArrayMetadataJSON: "shape": self.shape, "fill_value": copy.deepcopy(self.fill_value), "data_type": self.data_type.to_json(), - "chunk_grid": self.chunk_grid.to_json(), - "codecs": tuple(codec.to_json() for codec in self.codecs), - "chunk_key_encoding": self.chunk_key_encoding.to_json(), + "chunk_grid": _object_json(self.chunk_grid), + "codecs": tuple(_object_json(codec) for codec in self.codecs), + "chunk_key_encoding": _object_json(self.chunk_key_encoding), } if self.dimension_names is not UNSET: out["dimension_names"] = self.dimension_names @@ -279,7 +297,7 @@ def to_json(self) -> ZarrV3ArrayMetadataJSON: out["attributes"] = copy.deepcopy(self.attributes) if len(self.storage_transformers) > 0: out["storage_transformers"] = tuple( - transformer.to_json() for transformer in self.storage_transformers + _object_json(transformer) for transformer in self.storage_transformers ) # Extra fields are the TypedDict's `extra_items` (PEP 728). Assign them # by key rather than `out.update(**...)`: type checkers understand the diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py b/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py index c18a0bd260..0d7ab15abc 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py @@ -139,7 +139,7 @@ def unknown_storage(*_: object) -> None: class EmptyConfiguration(TypedDict, closed=True): - """The configuration of a definition with nothing to configure, written as its bare name.""" + """The configuration of a definition with nothing to configure: its field is written with its name alone.""" @dataclass(frozen=True, kw_only=True, slots=True) @@ -1113,10 +1113,14 @@ def canonicalize( `canonical` has the rest -- judged again, so a `canonical` that gives a configuration that does not hold is a `ValueError`, a fault in the definition rather than the field. The envelope takes the fewest - words: the bare name when nothing is configured, and no - `must_understand`, since `true` is what absence means. A name nothing - in scope claims comes back as written, since what it simplifies to is - its own definition's call. + words every reader takes: a data type with nothing to configure is its + bare name, as core data types have been written since Zarr v3.0; any + other field is an object, `{"name": ...}`, since a Zarr v3.0 reader + takes no short-hand name in `codecs` + (https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v3/core/index.rst#L585-L592); + and there is no `must_understand`, since `true` is what absence means. + A name nothing in scope claims comes back as written, since what it + simplifies to is its own definition's call. """ resolved, problems = resolve(data, kind, context, loc) if len(problems) != 0: @@ -1151,9 +1155,9 @@ def _canonical_field(definition: Definition[Any], resolved: Resolved[Any]) -> JS carrying = _carrying_name(definition, simplified) if carrying is not None: return carrying - if len(simplified) == 0: - return name - return {"name": name, "configuration": simplified} + if len(simplified) != 0: + return {"name": name, "configuration": simplified} + return name if isinstance(definition, DataTypeDefinition) else {"name": name} def _replaced(value: JSONValue, path: Loc, new: JSONValue) -> JSONValue: diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/definition.py b/packages/zarr-metadata/src/zarr_metadata/v3/definition.py index bdd28ba803..cd833aa87c 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/definition.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/definition.py @@ -143,7 +143,7 @@ def acme_lz4_rules( unjudged, its size unknown with the rest of it. `ZarrV3MetadataFieldJSON` is the same JSON, but checks as JSON and nothing more, so a definition refuses a member typed with it. An extension with nothing to configure -takes `EmptyConfiguration`, and is written as its bare name. +takes `EmptyConfiguration`, and is written with its name alone. A data type also says what its fill value is: `fill_value`, the JSON shape of one as an annotation the checker reads -- `Int8FillValue` -- and @@ -207,8 +207,10 @@ def acme_lz4_rules( field in its own simplest spelling, then the definition's `canonical` -- blosc drops a `typesize` that `noshuffle` ignores, a rectilinear grid run-length encodes its chunk shapes -- and the envelope in the fewest -words; raw bits write their size back into the name, in decimal, so -`r008` is `r8`. A field with any problem, an unknown key included, has +words every reader takes: a data type with nothing to configure is its +bare name, any other field an object, `{"name": ...}`, as a Zarr v3.0 +reader takes no short-hand name in `codecs`; raw bits write their size +back into the name, in decimal, so `r008` is `r8`. A field with any problem, an unknown key included, has none: a simpler spelling of it would erase what its author wrote. What `canonical` gives is judged again: one that does not hold is a `ValueError`, a fault in the definition. diff --git a/packages/zarr-metadata/tests/model/test_array.py b/packages/zarr-metadata/tests/model/test_array.py index 768a9c0d0c..c3ecc7b2a2 100644 --- a/packages/zarr-metadata/tests/model/test_array.py +++ b/packages/zarr-metadata/tests/model/test_array.py @@ -270,11 +270,32 @@ def test_v3_to_json_emits_canonical_document() -> None: "fill_value": 0, "data_type": "int32", "chunk_grid": {"name": "regular", "configuration": {"chunk_shape": (10,)}}, - "codecs": ("bytes",), - "chunk_key_encoding": "default", + "codecs": ({"name": "bytes"},), + "chunk_key_encoding": {"name": "default"}, } +@pytest.mark.parametrize( + "data_type", + ["int32", {"name": "int32"}, {"name": "int32", "configuration": {}}], + ids=["bare-name", "object", "empty-configuration"], +) +def test_v3_a_data_type_with_nothing_to_configure_is_written_by_its_bare_name( + data_type: object, +) -> None: + # As core data types have been written since Zarr v3.0, which is how + # zarr-python reads them; every other extension point is an object. + document = { + **ZarrV3ArrayMetadata.create_default(shape=(10,)).to_json(), + "data_type": data_type, + "codecs": [{"name": "bytes", "configuration": {"endian": "little"}}], + } + model = ZarrV3ArrayMetadata.from_json(document) + # The field alone is spelled as the document spells it, so a reader + # handed either takes it. + assert model.to_json()["data_type"] == model.data_type.to_json() == "int32" + + def test_v3_dimension_names_included_when_present() -> None: """V3 to_json includes dimension_names when they are set.""" out: dict[str, object] = dict( @@ -320,7 +341,7 @@ def test_v3_single_storage_transformer_included() -> None: out: dict[str, object] = dict( ZarrV3ArrayMetadata.create_default(storage_transformers=(st,)).to_json() ) - assert out["storage_transformers"] == ("some_transformer",) + assert out["storage_transformers"] == ({"name": "some_transformer"},) def test_v3_no_storage_transformers_omitted() -> None: diff --git a/packages/zarr-metadata/tests/v3/test_every_definition.py b/packages/zarr-metadata/tests/v3/test_every_definition.py index 9df43313f0..c4214b1cf9 100644 --- a/packages/zarr-metadata/tests/v3/test_every_definition.py +++ b/packages/zarr-metadata/tests/v3/test_every_definition.py @@ -203,8 +203,12 @@ def test_every_example_reads_and_its_simplest_spelling_is_stable(key: str, field }, }, ), - # Nothing configured: the bare name, and no `must_understand: true`. - ({"name": "crc32c", "configuration": {}, "must_understand": True}, "crc32c"), + # Nothing configured: the name alone, as an object, which every reader + # takes, and no `must_understand: true`. + ({"name": "crc32c", "configuration": {}, "must_understand": True}, {"name": "crc32c"}), + # But a data type with nothing configured is its bare name, as core + # data types are written. + ({"name": "int8", "configuration": {}, "must_understand": True}, "int8"), # Nested fields each in their own simplest spelling. ( { @@ -222,10 +226,10 @@ def test_every_example_reads_and_its_simplest_spelling_is_stable(key: str, field "name": "sharding_indexed", "configuration": { "chunk_shape": (2,), - "codecs": ("bytes",), + "codecs": ({"name": "bytes"},), "index_codecs": ( {"name": "bytes", "configuration": {"endian": "little"}}, - "crc32c", + {"name": "crc32c"}, ), }, }, @@ -249,10 +253,18 @@ def test_every_example_reads_and_its_simplest_spelling_is_stable(key: str, field }, ), ], - ids=["blosc-noshuffle", "bare-name", "sharding-nested", "cast-value-target", "rectilinear-rle"], + ids=[ + "blosc-noshuffle", + "name-alone", + "data-type-bare-name", + "sharding-nested", + "cast-value-target", + "rectilinear-rle", + ], ) def test_the_simplest_spelling(field: dict[str, Any], simplest: object) -> None: - kind = ChunkGridDefinition if field["name"] == "rectilinear" else CodecDefinition + kinds = {"rectilinear": ChunkGridDefinition, "int8": DataTypeDefinition} + kind = kinds.get(field["name"], CodecDefinition) assert canonicalize(field, kind, CORE_AND_EXTENSIONS) == (simplest, ()) From f21439cd3386ff27e1d27be202b0a24b05f7bd3f Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Mon, 28 Sep 2026 14:28:48 +0200 Subject: [PATCH 02/94] fix(zarr-metadata): problems show the JSON they found, and the boundary says what the package decided Four users of the new APIs found the same rough edges. A problem's message showed the Python value -- `got None`, `got (1,)` -- and named shapes in Python's words, "a sequence", "a mapping", a one-member Literal as a one-element tuple, a union of two objects as "an object or an object". Values now show as the JSON the document holds, shapes in JSON's words, and a value of another JSON type than a closed set's is `invalid_type`, so the kind tells a wrong type from a wrong value. `help` of a validator printed its default scope in full, some 14,000 characters; a scope's repr now says how many definitions it holds. The README's validation boundary says what the package decides where the specs leave it open: non-finite numbers in attributes, written back as bare tokens; `must_understand: false` refused at every extension point; a chunk length of 0 along a dimension of length 0. The models' I/O methods have docstrings, and no public docstring names a private function. Assisted-by: ClaudeCode:claude-opus-5-5 --- packages/zarr-metadata/README.md | 24 ++++- packages/zarr-metadata/changes/371.bugfix.md | 9 ++ packages/zarr-metadata/changes/371.doc.md | 7 ++ packages/zarr-metadata/changes/371.misc.md | 3 + packages/zarr-metadata/docs/index.md | 24 ++++- .../zarr-metadata/src/zarr_metadata/_json.py | 44 +++++++- .../src/zarr_metadata/_typed_json.py | 100 ++++++++++++------ .../src/zarr_metadata/model/_array.py | 53 +++++++++- .../src/zarr_metadata/model/_group.py | 74 ++++++++++++- .../src/zarr_metadata/model/_validation.py | 51 ++++----- .../src/zarr_metadata/v3/_common.py | 10 +- .../src/zarr_metadata/v3/_definition.py | 31 ++++-- .../src/zarr_metadata/v3/_registry.py | 5 + .../src/zarr_metadata/v3/codec/bytes.py | 4 +- .../src/zarr_metadata/v3/codec/cast_value.py | 8 +- .../zarr_metadata/v3/codec/scale_offset.py | 4 +- .../src/zarr_metadata/v3/codec/transpose.py | 4 +- .../src/zarr_metadata/v3/data_type/_float.py | 6 +- .../zarr_metadata/v3/data_type/_numpy_time.py | 2 +- .../src/zarr_metadata/v3/data_type/bytes.py | 4 +- .../src/zarr_metadata/v3/data_type/struct.py | 4 +- .../src/zarr_metadata/v3/definition.py | 6 +- .../zarr-metadata/tests/model/test_array.py | 33 +++++- .../zarr-metadata/tests/model/test_group.py | 16 ++- .../tests/model/test_refine_json.py | 21 +++- .../zarr-metadata/tests/test_typed_json.py | 39 ++++++- .../tests/v3/test_definitions.py | 11 ++ 27 files changed, 485 insertions(+), 112 deletions(-) create mode 100644 packages/zarr-metadata/changes/371.bugfix.md create mode 100644 packages/zarr-metadata/changes/371.doc.md create mode 100644 packages/zarr-metadata/changes/371.misc.md diff --git a/packages/zarr-metadata/README.md b/packages/zarr-metadata/README.md index 1febfe80c3..8b01bbbff8 100644 --- a/packages/zarr-metadata/README.md +++ b/packages/zarr-metadata/README.md @@ -56,8 +56,8 @@ members that the strict model parser rejects. The model validators enforce the declared document structure and a small set of context-free consistency rules, including fixed format literals, finite -JSON numbers, non-negative dimensions, and one `dimension_names` entry per -array dimension. In a v3 document they also read +JSON numbers outside `attributes`, non-negative dimensions, and one +`dimension_names` entry per array dimension. In a v3 document they also read each extension point -- the data type, chunk grid, chunk key encoding, each codec and each storage transformer -- through the definition that claims its name in a scope, `CORE_AND_EXTENSIONS` unless a `context` is passed: a @@ -71,6 +71,26 @@ against the chunk it is handed, a shard's inner and index codecs too. The validators do no arithmetic on values: whether a fill value survives a `cast_value` round trip is not judged. +Three choices the specs' words leave open, or settle two ways: + +- **`attributes` may hold `NaN`, `Infinity` and `-Infinity`.** The spec + interprets no attribute, and zarr-python and xarray write those numbers + there (a CF `_FillValue`, say). The models read them, and `to_key_value` + writes them back as those bare tokens, which a strict JSON parser + refuses. `check`, from `zarr_metadata.typed_json`, refuses a non-finite + number wherever it is, attributes included. +- **`must_understand: false` is refused at every extension point**, codecs + and storage transformers too, though the core spec names only the data + type, chunk grid and chunk key encoding: a reader that skips a codec + reads wrong bytes as surely as one that skips a data type reads wrong + values. It keeps its meaning on an unknown top-level member, which a + reader can skip. +- **A chunk length of 0 is allowed along a dimension of length 0.** The + core spec asks for non-zero chunk lengths only "when the corresponding + dimensions of the arrays have non-zero length"; the regular grid spec + says chunk sizes are greater than zero. The package follows the core + spec, which zarr-python 3.0 and 3.1 wrote for an empty dimension. + The Pydantic integration's generated JSON Schemas express independently checkable document structure and field constraints, but they are not a replacement for runtime model validation. Standard JSON Schema treats a diff --git a/packages/zarr-metadata/changes/371.bugfix.md b/packages/zarr-metadata/changes/371.bugfix.md new file mode 100644 index 0000000000..b1f1381086 --- /dev/null +++ b/packages/zarr-metadata/changes/371.bugfix.md @@ -0,0 +1,9 @@ +A problem's message shows what the document holds as JSON -- `null`, +`[1, 2]`, `"C"` -- where it showed the Python value, `None`, `(1, 2)`, +`'C'`, and names what was expected in JSON's words: an array, not a +sequence; an object, not a mapping; a lone value alone, not a +one-element tuple; a union of objects once, not "an object or an +object". A value of another JSON type than each of a closed set's -- +`"zarr_format": "3"`, or `"version": 0.5` where a TypedDict says +`"0.5"` -- is `invalid_type`, where it was `invalid_value`, so the kind +tells a wrong type from a wrong value. diff --git a/packages/zarr-metadata/changes/371.doc.md b/packages/zarr-metadata/changes/371.doc.md new file mode 100644 index 0000000000..d0017875f5 --- /dev/null +++ b/packages/zarr-metadata/changes/371.doc.md @@ -0,0 +1,7 @@ +The validation boundary says what the package decides where the specs +leave it open or settle it two ways: `attributes` may hold `NaN`, +`Infinity` and `-Infinity`, which `to_key_value` writes as bare tokens; +`must_understand: false` is refused at every extension point, codecs +too; a chunk length of 0 is allowed along a dimension of length 0. The +models' `from_json`, `to_json`, `from_key_value` and `to_key_value` +have docstrings, and no public docstring names a private function. diff --git a/packages/zarr-metadata/changes/371.misc.md b/packages/zarr-metadata/changes/371.misc.md new file mode 100644 index 0000000000..a10dc222ac --- /dev/null +++ b/packages/zarr-metadata/changes/371.misc.md @@ -0,0 +1,3 @@ +A scope's repr says how many definitions it holds, `Context(<33 +definitions>)`, so `help` of a validator no longer prints every +definition in its default scope. diff --git a/packages/zarr-metadata/docs/index.md b/packages/zarr-metadata/docs/index.md index d0fe4a7fec..ed7f68f2b2 100644 --- a/packages/zarr-metadata/docs/index.md +++ b/packages/zarr-metadata/docs/index.md @@ -71,8 +71,8 @@ members that the strict model parser rejects. The model validators enforce the declared document structure and a small set of context-free consistency rules, including fixed format literals, finite -JSON numbers, non-negative dimensions, and one `dimension_names` entry per -array dimension. In a v3 document they also read +JSON numbers outside `attributes`, non-negative dimensions, and one +`dimension_names` entry per array dimension. In a v3 document they also read each extension point -- the data type, chunk grid, chunk key encoding, each codec and each storage transformer -- through the definition that claims its name in a scope, `CORE_AND_EXTENSIONS` unless a `context` is passed: a @@ -86,6 +86,26 @@ against the chunk it is handed, a shard's inner and index codecs too. The validators do no arithmetic on values: whether a fill value survives a `cast_value` round trip is not judged. +Three choices the specs' words leave open, or settle two ways: + +- **`attributes` may hold `NaN`, `Infinity` and `-Infinity`.** The spec + interprets no attribute, and zarr-python and xarray write those numbers + there (a CF `_FillValue`, say). The models read them, and `to_key_value` + writes them back as those bare tokens, which a strict JSON parser + refuses. `check`, from `zarr_metadata.typed_json`, refuses a non-finite + number wherever it is, attributes included. +- **`must_understand: false` is refused at every extension point**, codecs + and storage transformers too, though the core spec names only the data + type, chunk grid and chunk key encoding: a reader that skips a codec + reads wrong bytes as surely as one that skips a data type reads wrong + values. It keeps its meaning on an unknown top-level member, which a + reader can skip. +- **A chunk length of 0 is allowed along a dimension of length 0.** The + core spec asks for non-zero chunk lengths only "when the corresponding + dimensions of the arrays have non-zero length"; the regular grid spec + says chunk sizes are greater than zero. The package follows the core + spec, which zarr-python 3.0 and 3.1 wrote for an empty dimension. + ## Scope At minimum, this library supports what Zarr-Python needs: the complete diff --git a/packages/zarr-metadata/src/zarr_metadata/_json.py b/packages/zarr-metadata/src/zarr_metadata/_json.py index a5c0a9b6af..30b95dbad3 100644 --- a/packages/zarr-metadata/src/zarr_metadata/_json.py +++ b/packages/zarr-metadata/src/zarr_metadata/_json.py @@ -9,6 +9,7 @@ from __future__ import annotations +import json import math from collections.abc import Mapping, Sequence from dataclasses import dataclass @@ -118,7 +119,7 @@ def prefixed( def validate_json(value: object) -> tuple[ValidationProblem, ...]: - """Return every reason `value` is not JSON-serializable (recursively): `refine_json`'s problems.""" + """Return every reason `value` is not JSON, each where it sits: a float that is not finite, a key that is not a string, a value of no JSON type.""" return refine_json(value)[1] @@ -195,6 +196,43 @@ def _refine(value: object, loc: tuple[str | int, ...], *, finite: bool) -> _Refi ) +def shown(value: object) -> str: + """`value` as a problem's message shows it: as the JSON a document writes, `null` and `[1, 2]`, or by its repr when it is not JSON.""" + refined, problems = _refine(value, (), finite=False) + if len(problems) != 0: + return repr(value) + return json.dumps(refined, ensure_ascii=False) + + +def choices(allowed: Sequence[object]) -> str: + """A closed set of values as a message names it: `"C"` alone, or `one of ["C", "F"]`.""" + values = sorted(dict.fromkeys(shown(value) for value in allowed)) + return values[0] if len(values) == 1 else f"one of [{', '.join(values)}]" + + +def json_type(value: object) -> str: + """The JSON type of `value`, as a message names it: "a string", "a number", "null".""" + if value is None: + return "null" + if isinstance(value, bool): + return "a boolean" + if isinstance(value, (int, float)): + return "a number" + if isinstance(value, str): + return "a string" + if isinstance(value, Mapping): + return "an object" + if isinstance(value, (list, tuple)): + return "an array" + return "a value" + + +def refused_kind(value: object, allowed: Sequence[object]) -> ProblemKind: + """What is wrong with `value`, outside a closed set: its type, when none of the set is of its JSON type, else its value.""" + same_type = json_type(value) in {json_type(entry) for entry in allowed} + return "invalid_value" if same_type else "invalid_type" + + def is_canonical_json(value: object, *, finite: bool = True) -> TypeGuard[JSONValue]: """Whether `value` already uses the concrete containers in `JSONValue`. @@ -260,11 +298,15 @@ def arrays_to_tuples(obj: object) -> object: "ProblemKind", "ValidationProblem", "arrays_to_tuples", + "choices", "is_canonical_json", "is_json", + "json_type", "parse_json", "prefixed", "refine_json", "refine_user_data", + "refused_kind", + "shown", "validate_json", ] diff --git a/packages/zarr-metadata/src/zarr_metadata/_typed_json.py b/packages/zarr-metadata/src/zarr_metadata/_typed_json.py index b8dbc1cd80..c4bf7c6fa0 100644 --- a/packages/zarr-metadata/src/zarr_metadata/_typed_json.py +++ b/packages/zarr-metadata/src/zarr_metadata/_typed_json.py @@ -57,7 +57,14 @@ from typing_extensions import NoExtraItems, TypeIs, is_typeddict from zarr_metadata._common import JSONValue -from zarr_metadata._json import ValidationProblem, is_json, refine_json +from zarr_metadata._json import ( + ValidationProblem, + choices, + is_json, + refine_json, + refused_kind, + shown, +) if TYPE_CHECKING: from zarr_metadata._json import ProblemKind @@ -375,43 +382,66 @@ def _typeddict_bases(typeddict: type) -> tuple[type, ...]: def describe(annotation: object, seen: frozenset[object] = frozenset()) -> str: """The annotation as a message would name it: "an integer", "an object".""" + return _named(annotation, seen)[0] + + +def _named(annotation: object, seen: frozenset[object]) -> tuple[str, str]: + """The annotation as a message names one of it and many of it: "an integer", "integers".""" inner = strip_annotation(annotation)[0] if inner is int: - return "an integer" + return "an integer", "integers" if inner is float: - return "a number" + return "a number", "numbers" if inner is bool: - return "a boolean" + return "a boolean", "booleans" if inner is str: - return "a string" + return "a string", "strings" if inner is None or inner is types.NoneType: - return "null" + return "null", "nulls" if inner is JSONValue: - return "a JSON value" + return "a JSON value", "JSON values" origin = get_origin(inner) if origin is Literal: - return f"one of {tuple(sorted(get_args(inner), key=repr))!r}" + values = get_args(inner) + listed = ", ".join(sorted(dict.fromkeys(shown(value) for value in values))) + return choices(values), f"values in [{listed}]" if is_union(inner): - return " or ".join(describe(branch, seen) for branch in get_args(inner)) + # Each shape once -- two TypedDicts are both "an object" -- and a + # broader one takes in a narrower: a number an integer, a string + # the strings a `Literal` names. + branches = get_args(inner) + named = dict.fromkeys(_named(branch, seen) for branch in branches) + if ("a number", "numbers") in named: + named.pop(("an integer", "integers"), None) + if ("a string", "strings") in named: + for branch in filter(_strings_only, branches): + named.pop(_named(branch, seen), None) + return " or ".join(one for one, _ in named), " or ".join(many for _, many in named) if origin is tuple: arguments = get_args(inner) if len(arguments) == 2 and arguments[1] is Ellipsis: - return f"an array of {describe(arguments[0], seen)} elements" + each = _named(arguments[0], seen)[1] + return f"an array of {each}", f"arrays of {each}" if len(arguments) == 2: - return f"a [{describe(arguments[0], seen)}, {describe(arguments[1], seen)}] pair" - return f"an array of {len(arguments)} elements" - if origin in (Mapping, dict): - return "an object" + pair = f"[{describe(arguments[0], seen)}, {describe(arguments[1], seen)}] pair" + return f"a {pair}", f"{pair}s" + return f"an array of {len(arguments)} elements", f"arrays of {len(arguments)} elements" + if origin in (Mapping, dict) or (isinstance(inner, type) and is_typeddict(inner)): + return "an object", "objects" if isinstance(inner, NewType): - return describe(inner.__supertype__, seen) - if isinstance(inner, type) and is_typeddict(inner): - return "an object" + return _named(inner.__supertype__, seen) if is_alias(inner): alias = cast("typing_extensions.TypeAliasType", inner) if alias in seen or _holds(alias_value(alias), alias, frozenset()): - return f"a {alias.__name__}" - return describe(alias_value(alias), seen | {alias}) - return "a value" + return f"a {alias.__name__}", f"{alias.__name__} values" + return _named(alias_value(alias), seen | {alias}) + return "a value", "values" + + +def _strings_only(annotation: object) -> bool: + """Whether `annotation` is a `Literal` of strings, which "a string" takes in.""" + inner = strip_annotation(annotation)[0] + return get_origin(inner) is Literal and all(isinstance(value, str) for value in get_args(inner)) def _holds(annotation: object, alias: object, seen: frozenset[object]) -> bool: @@ -491,7 +521,7 @@ def _scalar(description: str, admits: Callable[[object], bool]) -> Parser: def parse(value: object, loc: Loc) -> Parsed: if admits(value): return value, () - return value, problem(loc, f"expected {description}, got {value!r}") + return value, problem(loc, f"expected {description}, got {shown(value)}") return parse @@ -510,14 +540,15 @@ def one_of(allowed: tuple[object, ...]) -> Parser: """A member whose type is a closed set of values. Equal and of the same type: JSON `true` is not the integer 1, though - Python says `True == 1`. + Python says `True == 1`. A value of a JSON type none of them has -- + a number, where each is a string -- is of the wrong type; one of the + right type, the wrong value. """ def parse(value: object, loc: Loc) -> Parsed: if not any(value == entry and type(value) is type(entry) for entry in allowed): - return value, problem( - loc, f"expected one of {allowed!r}, got {value!r}", "invalid_value" - ) + message = f"expected {choices(allowed)}, got {shown(value)}" + return value, problem(loc, message, refused_kind(value, allowed)) return value, () return parse @@ -528,7 +559,7 @@ def sequence_of(element: Parser) -> Parser: def parse(value: object, loc: Loc) -> Parsed: if not isinstance(value, (list, tuple)): - return value, problem(loc, f"expected a sequence, got {value!r}") + return value, problem(loc, f"expected an array, got {shown(value)}") entries = cast("list[object] | tuple[object, ...]", value) parsed: list[object] = [] found: list[ValidationProblem] = [] @@ -546,10 +577,10 @@ def fixed_tuple(elements: Sequence[Parser], description: str) -> Parser: def parse(value: object, loc: Loc) -> Parsed: if not isinstance(value, (list, tuple)): - return value, problem(loc, f"expected {description}, got {value!r}") + return value, problem(loc, f"expected {description}, got {shown(value)}") entries = tuple(cast("list[object] | tuple[object, ...]", value)) if len(entries) != len(elements): - return entries, problem(loc, f"expected {description}, got {entries!r}") + return entries, problem(loc, f"expected {description}, got {shown(entries)}") parsed: list[object] = [] found: list[ValidationProblem] = [] for position, (element, entry) in enumerate(zip(elements, entries, strict=True)): @@ -603,7 +634,7 @@ def parse(value: object, loc: Loc) -> Parsed: return min(clean, key=lambda found: found[:2])[2] if len(failed) != 0: return min(failed, key=lambda found: found[:2])[2] - return value, problem(loc, f"expected {description}, got {value!r}") + return value, problem(loc, f"expected {description}, got {shown(value)}") return parse @@ -616,10 +647,9 @@ def _by_tag(branches: Sequence[Branch], tag: Tag, value: Mapping[str, object], l said = value[key] index = picks.get((type(said), said)) if _hashable(said) else None if index is None: - allowed = tuple(sorted((entry for _, entry in picks), key=repr)) - return value, problem( - (*loc, key), f"expected one of {allowed!r}, got {said!r}", "invalid_value" - ) + allowed = tuple(entry for _, entry in picks) + message = f"expected {choices(allowed)}, got {shown(said)}" + return value, problem((*loc, key), message, refused_kind(said, allowed)) return branches[index][1](value, loc) @@ -646,7 +676,7 @@ def object_of(members: Mapping[str, tuple[Parser, bool]], extra: Parser | None) def parse(value: object, loc: Loc) -> Parsed: if not isinstance(value, Mapping): - return value, problem(loc, f"expected an object, got {value!r}") + return value, problem(loc, f"expected an object, got {shown(value)}") entries = cast("Mapping[str, object]", value) parsed: dict[str, object] = {} found: list[ValidationProblem] = [] @@ -678,7 +708,7 @@ def mapping_of(value: Parser) -> Parser: def parse(candidate: object, loc: Loc) -> Parsed: if not isinstance(candidate, Mapping): - return candidate, problem(loc, f"expected an object, got {candidate!r}") + return candidate, problem(loc, f"expected an object, got {shown(candidate)}") entries = cast("Mapping[str, object]", candidate) parsed: dict[str, object] = {} found: list[ValidationProblem] = [] diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_array.py b/packages/zarr-metadata/src/zarr_metadata/model/_array.py index bd09044532..c66d565f74 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_array.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/_array.py @@ -80,6 +80,13 @@ def to_json(self) -> ZarrV3MetadataFieldJSON: @classmethod def from_json(cls, data: object) -> ZarrV3NamedConfig: + """A field read from `data`, its JSON: a bare name, or an object of a `name` and, if it says them, a `configuration` and a `must_understand`. + + `MetadataValidationError` for anything else. It is read in no scope, + so nothing judges its configuration, which is held as written, and a + `must_understand` of `false` is held too, though a document's + validators refuse one at every extension point. + """ field = parse_metadata_field_v3(data) if isinstance(field, str): return cls(name=field, configuration={}, must_understand=True) @@ -210,7 +217,8 @@ def create_default(cls, **overrides: Unpack[ZarrV3ArrayMetadataPartial]) -> Zarr (the same fields accepted by `update`). Overriding `shape` without `chunk_grid` derives a consistent default grid: one regular chunk covering the array (`chunk_shape` equal to `shape`, with a length of - 1 for a dimension of length 0, since a chunk length is at least 1: + 1 for a dimension of length 0, which every reader takes: the core + spec allows 0 there and the regular grid spec does not, https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v3/chunk-grids/regular-grid/index.rst#L40). The derivation is deliberately one-way. A user-supplied `chunk_grid` @@ -279,6 +287,14 @@ def __post_init__(self) -> None: ) def to_json(self) -> ZarrV3ArrayMetadataJSON: + """The document as JSON, arrays as tuples, sharing no mutable state with the model. + + Each extension point as its readers take it: the data type as + `ZarrV3NamedConfig.to_json` writes it, and every other as an object; + `dimension_names` when set, and `attributes` and `storage_transformers` + when not empty. Not validated: `to_key_value` is the writer that + refuses a model that is not valid. + """ # to_json output shares no mutable state with the model: every value # that can hold a mutable container is deep-copied. out: ZarrV3ArrayMetadataJSON = { @@ -311,6 +327,12 @@ def to_json(self) -> ZarrV3ArrayMetadataJSON: def from_json( cls, data: object, *, context: Context = CORE_AND_EXTENSIONS ) -> ZarrV3ArrayMetadata: + """The model of `data`, a v3 array document read in `context`. + + `MetadataValidationError` with every problem `validate_array_metadata_v3` + finds. A member the spec does not define is held in `extra_fields`. The + model shares no mutable state with `data`. + """ # A read model shares no mutable state with what it read. parsed = copy.deepcopy(parse_array_metadata_v3(data, context=context)) extra_fields: dict[str, ZarrV3ExtensionField] = { @@ -347,6 +369,11 @@ def must_understand_fields(self) -> dict[str, ZarrV3ExtensionField]: def from_key_value( cls, mapping: Mapping[StoreKey, bytes], *, context: Context = CORE_AND_EXTENSIONS ) -> ZarrV3ArrayMetadata: + """The model of the array document at `zarr.json` in `mapping`, read in `context`. + + `MetadataValidationError` when the key is missing, its bytes are not + JSON, or the document is not valid. + """ return cls.from_json( load_store_json(mapping, ZARR_V3_ARRAY_METADATA_STORE_KEY), context=context ) @@ -354,6 +381,14 @@ def from_key_value( def to_key_value( self, *, indent: int | str | None = None, context: Context = CORE_AND_EXTENSIONS ) -> Mapping[ZarrV3ArrayMetadataStoreKey, bytes]: + """The document as a store holds it: JSON bytes at `zarr.json`, indented by `indent`. + + Validated first, in `context`: a model that is not valid raises + `MetadataValidationError` with every problem, and nothing is written. + `NaN`, `Infinity` and `-Infinity` in `attributes` are written as those + bare tokens, as zarr-python writes them, which a strict JSON parser + refuses. + """ # A model built by hand is not validated: its document is written only # if it reads as `from_json` reads one in `context`, and every problem # is raised. @@ -484,6 +519,12 @@ def to_json(self) -> ZarrV2ArrayMetadataJSON: @classmethod def from_json(cls, data: object) -> ZarrV2ArrayMetadata: + """The model of `data`, a v2 array document with its attributes under `attributes`. + + `MetadataValidationError` with every problem `validate_array_metadata_v2` + finds. A missing `dimension_separator` is read as `"."`, which is + written back. The model shares no mutable state with `data`. + """ # A read model shares no mutable state with what it read. parsed = copy.deepcopy(parse_array_metadata_v2(data)) return cls( @@ -500,6 +541,11 @@ def from_json(cls, data: object) -> ZarrV2ArrayMetadata: @classmethod def from_key_value(cls, mapping: Mapping[StoreKey, bytes]) -> ZarrV2ArrayMetadata: + """The model of the array at `.zarray` in `mapping`, with the attributes at `.zattrs` when there is one. + + `MetadataValidationError` when `.zarray` is missing, bytes are not + JSON, `.zarray` holds `attributes`, or the document is not valid. + """ zarray_raw = load_store_json(mapping, ZARR_V2_ARRAY_METADATA_STORE_KEY) if not isinstance(zarray_raw, Mapping): return cls.from_json(zarray_raw) @@ -522,6 +568,11 @@ def from_key_value(cls, mapping: Mapping[StoreKey, bytes]) -> ZarrV2ArrayMetadat def to_key_value( self, *, indent: int | str | None = None ) -> Mapping[ZarrV2ArrayMetadataStoreKey | ZarrV2AttributesStoreKey, bytes]: + """The document as a store holds it: `.zarray` without the attributes, and `.zattrs` with them when they are set, even empty. + + Validated first: a model that is not valid raises + `MetadataValidationError` with every problem, and nothing is written. + """ # Attributes live only in the sibling `.zattrs` file; the `.zarray` # document must exclude them. The `.zattrs` key is present exactly # when attributes are set (even empty) — UNSET emits no file. A model diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_group.py b/packages/zarr-metadata/src/zarr_metadata/model/_group.py index fa56f633aa..eb9d57b84a 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_group.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/_group.py @@ -16,6 +16,8 @@ arrays_to_tuples, refine_json, refine_user_data, + refused_kind, + shown, ) from zarr_metadata.model._array import ( ZarrV3ArrayMetadata, @@ -122,6 +124,12 @@ def update(self, **kwargs: Unpack[ZarrV3GroupMetadataPartial]) -> ZarrV3GroupMet return dataclasses.replace(self, **kwargs) def to_json(self) -> ZarrV3GroupMetadataJSON: + """The document as JSON, sharing no mutable state with the model. + + `attributes` when not empty, `consolidated_metadata` when set, and + each extra field as held. Not validated: `to_key_value` is the writer + that refuses a model that is not valid. + """ # to_json output shares no mutable state with the model: every value # that can hold a mutable container is deep-copied. out: ZarrV3GroupMetadataJSON = { @@ -141,6 +149,13 @@ def to_json(self) -> ZarrV3GroupMetadataJSON: def from_json( cls, data: object, *, context: Context = CORE_AND_EXTENSIONS ) -> ZarrV3GroupMetadata: + """The model of `data`, a v3 group document read in `context`, with each document its consolidated metadata holds. + + `MetadataValidationError` with every problem `validate_group_metadata_v3` + finds. A `consolidated_metadata` of `null` is read as none, and not + written back. A member the spec does not define is held in + `extra_fields`. + """ # A read model shares no mutable state with what it read. parsed = copy.deepcopy(parse_group_metadata_v3(data, context=context)) consolidated_raw: object = parsed.get(ZARR_V3_CONSOLIDATED_METADATA_KEY, UNSET) @@ -179,6 +194,11 @@ def must_understand_fields(self) -> dict[str, ZarrV3ExtensionField]: def from_key_value( cls, mapping: Mapping[StoreKey, bytes], *, context: Context = CORE_AND_EXTENSIONS ) -> ZarrV3GroupMetadata: + """The model of the group document at `zarr.json` in `mapping`, read in `context`. + + `MetadataValidationError` when the key is missing, its bytes are not + JSON, or the document is not valid. + """ return cls.from_json( load_store_json(mapping, ZARR_V3_GROUP_METADATA_STORE_KEY), context=context ) @@ -186,6 +206,14 @@ def from_key_value( def to_key_value( self, *, indent: int | str | None = None, context: Context = CORE_AND_EXTENSIONS ) -> Mapping[ZarrV3GroupMetadataStoreKey, bytes]: + """The document as a store holds it: JSON bytes at `zarr.json`, indented by `indent`. + + Validated first, in `context`: a model that is not valid raises + `MetadataValidationError` with every problem, and nothing is written. + `NaN`, `Infinity` and `-Infinity` in `attributes` are written as those + bare tokens, as zarr-python writes them, which a strict JSON parser + refuses. + """ # A model built by hand is not validated: its document is written only # if it reads as `from_json` reads one in `context`, and every problem # is raised. @@ -222,6 +250,7 @@ def __post_init__(self) -> None: ) def to_json(self) -> ZarrV3ConsolidatedMetadataJSON: + """The `consolidated_metadata` member as JSON: its `kind`, `must_understand: false`, and each document by its path.""" # `must_understand` is emitted as the literal False: the field is typed # permissively as `bool`, but `__post_init__` guarantees the value. return { @@ -234,6 +263,10 @@ def to_json(self) -> ZarrV3ConsolidatedMetadataJSON: def from_json( cls, data: object, *, context: Context = CORE_AND_EXTENSIONS ) -> ZarrV3ConsolidatedMetadata: + """The model of `data`, a group's `consolidated_metadata` member, each document read in `context` as the array or group its `node_type` says. + + `MetadataValidationError` with every problem found. + """ normalized = arrays_to_tuples(data) problems = validate_consolidated_metadata_v3(normalized, context=context) if len(problems) != 0: @@ -320,12 +353,22 @@ def to_json(self) -> ZarrV2GroupMetadataJSON: @classmethod def from_json(cls, data: object) -> ZarrV2GroupMetadata: + """The model of `data`, a v2 group document with its attributes under `attributes`. + + `MetadataValidationError` with every problem `validate_group_metadata_v2` + finds. The model shares no mutable state with `data`. + """ # A read model shares no mutable state with what it read. parsed = copy.deepcopy(parse_group_metadata_v2(data)) return cls(attributes=(dict(parsed["attributes"]) if "attributes" in parsed else UNSET)) @classmethod def from_key_value(cls, mapping: Mapping[StoreKey, bytes]) -> ZarrV2GroupMetadata: + """The model of the group at `.zgroup` in `mapping`, with the attributes at `.zattrs` when there is one. + + `MetadataValidationError` when `.zgroup` is missing, bytes are not + JSON, `.zgroup` holds `attributes`, or the document is not valid. + """ zgroup_raw = load_store_json(mapping, ZARR_V2_GROUP_METADATA_STORE_KEY) if not isinstance(zgroup_raw, Mapping): return cls.from_json(zgroup_raw) @@ -348,6 +391,11 @@ def from_key_value(cls, mapping: Mapping[StoreKey, bytes]) -> ZarrV2GroupMetadat def to_key_value( self, *, indent: int | str | None = None ) -> Mapping[ZarrV2GroupMetadataStoreKey | ZarrV2AttributesStoreKey, bytes]: + """The document as a store holds it: `.zgroup` without the attributes, and `.zattrs` with them when they are set, even empty. + + Validated first: a model that is not valid raises + `MetadataValidationError` with every problem, and nothing is written. + """ # Attributes live only in the sibling `.zattrs` file; the `.zgroup` # document must exclude them. The `.zattrs` key is present exactly # when attributes are set (even empty) — UNSET emits no file. A model @@ -380,6 +428,7 @@ class ZarrV2ConsolidatedMetadata: metadata: dict[str, JSONValue] def to_json(self) -> dict[str, JSONValue]: + """The `.zmetadata` document as JSON, sharing no mutable state with the model.""" # to_json output shares no mutable state with the model. return { "zarr_consolidated_format": self.zarr_consolidated_format, @@ -388,10 +437,17 @@ def to_json(self) -> dict[str, JSONValue]: @classmethod def from_json(cls, data: object) -> ZarrV2ConsolidatedMetadata: + """The model of `data`, a `.zmetadata` document, its entries held as written. + + `MetadataValidationError` with every problem: a member missing or + unexpected, a format other than 1, an entry that is not JSON. A + `.zattrs` entry is user data, and may hold `NaN`, `Infinity` and + `-Infinity`. + """ normalized = arrays_to_tuples(data) if not isinstance(normalized, Mapping): raise MetadataValidationError( - [ValidationProblem((), "expected a mapping", "invalid_type")] + [ValidationProblem((), "expected an object", "invalid_type")] ) doc = cast("Mapping[object, object]", normalized) problems: list[ValidationProblem] = [ @@ -414,8 +470,8 @@ def from_json(cls, data: object) -> ZarrV2ConsolidatedMetadata: problems.append( ValidationProblem( ("zarr_consolidated_format",), - f"expected 1, got {doc['zarr_consolidated_format']!r}", - "invalid_value", + f"expected 1, got {shown(doc['zarr_consolidated_format'])}", + refused_kind(doc["zarr_consolidated_format"], (1,)), ) ) refined: dict[str, JSONValue] = {} @@ -426,7 +482,7 @@ def from_json(cls, data: object) -> ZarrV2ConsolidatedMetadata: ): problems.append( ValidationProblem( - ("metadata",), "expected a mapping with string keys", "invalid_type" + ("metadata",), "expected an object with string keys", "invalid_type" ) ) else: @@ -447,11 +503,21 @@ def from_json(cls, data: object) -> ZarrV2ConsolidatedMetadata: @classmethod def from_key_value(cls, mapping: Mapping[StoreKey, bytes]) -> ZarrV2ConsolidatedMetadata: + """The model of the document at `.zmetadata` in `mapping`. + + `MetadataValidationError` when the key is missing, its bytes are not + JSON, or the document is not valid. + """ return cls.from_json(load_store_json(mapping, ZARR_V2_CONSOLIDATED_METADATA_STORE_KEY)) def to_key_value( self, *, indent: int | str | None = None ) -> Mapping[ZarrV2ConsolidatedMetadataStoreKey, bytes]: + """The document as a store holds it: JSON bytes at `.zmetadata`, indented by `indent`. + + Validated first: a model that is not valid raises + `MetadataValidationError` with every problem, and nothing is written. + """ # A model built by hand is not validated: it is written only as # `from_json` reads it, and every problem is raised. document = ZarrV2ConsolidatedMetadata.from_json(self.to_json()).to_json() diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_validation.py b/packages/zarr-metadata/src/zarr_metadata/model/_validation.py index d1ae10b83a..3d1b137dd2 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_validation.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/_validation.py @@ -32,6 +32,8 @@ ValidationProblem, arrays_to_tuples, refine_user_data, + refused_kind, + shown, validate_json, ) from zarr_metadata._json import is_canonical_json as _is_canonical_json @@ -133,11 +135,10 @@ def _unexpected_keys( def _check_literal( doc: Mapping[object, object], key: str, expected: object ) -> tuple[ValidationProblem, ...]: - """One `invalid_value` problem if `doc[key]` is present but not `expected`.""" + """One problem if `doc[key]` is present but not `expected`: of its type, when it is not of `expected`'s JSON type, else of its value.""" if key in doc and (type(doc[key]) is not type(expected) or doc[key] != expected): - return ( - ValidationProblem((key,), f"expected {expected!r}, got {doc[key]!r}", "invalid_value"), - ) + message = f"expected {shown(expected)}, got {shown(doc[key])}" + return (ValidationProblem((key,), message, refused_kind(doc[key], (expected,))),) return () @@ -200,7 +201,7 @@ def _dimension_lengths( return None, () value = doc[key] if not _is_int_sequence(value): - return None, (ValidationProblem((key,), "expected a sequence of int", "invalid_type"),) + return None, (ValidationProblem((key,), "expected an array of integers", "invalid_type"),) if any(item < 0 for item in value): return None, (ValidationProblem((key,), "expected non-negative integers", "invalid_value"),) return tuple(value), () @@ -326,7 +327,7 @@ def _validate_attributes(value: object) -> tuple[ValidationProblem, ...]: ): return ( ValidationProblem( - ("attributes",), "expected a mapping with string keys", "invalid_type" + ("attributes",), "expected an object with string keys", "invalid_type" ), ) problems: list[ValidationProblem] = [] @@ -372,7 +373,7 @@ def validate_array_metadata_v3( reports as `must_understand_fields`. """ if not isinstance(value, Mapping): - return (ValidationProblem((), "expected a mapping", "invalid_type"),) + return (ValidationProblem((), "expected an object", "invalid_type"),) doc = cast("Mapping[object, object]", value) problems: list[ValidationProblem] = list(_missing_keys(ARRAY_METADATA_REQUIRED_KEYS_V3, doc)) problems.extend(_validate_other_members(doc, ARRAY_METADATA_STANDARD_KEYS_V3)) @@ -417,7 +418,7 @@ def validate_array_metadata_v3( if key in doc: entries = doc[key] if not _is_array(entries): - problems.append(ValidationProblem((key,), "expected a sequence", "invalid_type")) + problems.append(ValidationProblem((key,), "expected an array", "invalid_type")) else: listed[key] = [] for index, entry in enumerate(entries): @@ -439,12 +440,12 @@ def validate_array_metadata_v3( names = doc["dimension_names"] if not _is_array(names): problems.append( - ValidationProblem(("dimension_names",), "expected a sequence", "invalid_type") + ValidationProblem(("dimension_names",), "expected an array", "invalid_type") ) elif not all(item is None or isinstance(item, str) for item in names): problems.append( ValidationProblem( - ("dimension_names",), "expected items of str or None", "invalid_type" + ("dimension_names",), "expected an array of strings or null", "invalid_type" ) ) elif shape is not None and len(names) != len(shape): @@ -489,7 +490,7 @@ def validate_array_metadata_v2(value: object) -> tuple[ValidationProblem, ...]: codec configurations (mappings with a string `id`). """ if not isinstance(value, Mapping): - return (ValidationProblem((), "expected a mapping", "invalid_type"),) + return (ValidationProblem((), "expected an object", "invalid_type"),) doc = cast("Mapping[object, object]", value) # Unlike the group document ("Other keys MUST NOT be present", # https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v2/v2.0.rst#L313), the v2 array document is open: other keys "SHOULD NOT be @@ -516,15 +517,14 @@ def validate_array_metadata_v2(value: object) -> tuple[ValidationProblem, ...]: problems.append( ValidationProblem( ("dtype",), - "expected a v2 dtype string or a sequence of field records", + "expected a v2 dtype string or an array of field records", "invalid_type", ) ) if "order" in doc and doc["order"] not in ("C", "F"): + message = f'expected "C" or "F", got {shown(doc["order"])}' problems.append( - ValidationProblem( - ("order",), f"expected 'C' or 'F', got {doc['order']!r}", "invalid_value" - ) + ValidationProblem(("order",), message, refused_kind(doc["order"], ("C", "F"))) ) if "compressor" in doc: compressor = doc["compressor"] @@ -538,7 +538,7 @@ def validate_array_metadata_v2(value: object) -> tuple[ValidationProblem, ...]: problems.append( ValidationProblem( ("filters",), - "expected null or a sequence of codec configurations with string 'id's", + "expected null or an array of codec configurations, each with a string 'id'", "invalid_type", ) ) @@ -548,11 +548,12 @@ def validate_array_metadata_v2(value: object) -> tuple[ValidationProblem, ...]: for index, item in enumerate(filters): problems.extend(_prefix("filters", _prefix(index, validate_json(item)))) if "dimension_separator" in doc and doc["dimension_separator"] not in (".", "/"): + separator = doc["dimension_separator"] problems.append( ValidationProblem( ("dimension_separator",), - f"expected '.' or '/', got {doc['dimension_separator']!r}", - "invalid_value", + f'expected "." or "/", got {shown(separator)}', + refused_kind(separator, (".", "/")), ) ) if "fill_value" in doc: @@ -591,7 +592,7 @@ def validate_consolidated_metadata_v3( `ZarrV3ConsolidatedMetadata.from_json` accepts in the same scope. """ if not isinstance(value, Mapping): - return (ValidationProblem((), "expected a mapping", "invalid_type"),) + return (ValidationProblem((), "expected an object", "invalid_type"),) env = cast("Mapping[object, object]", value) problems: list[ValidationProblem] = [ ValidationProblem((key,), "missing required key", "missing_key") @@ -605,7 +606,7 @@ def validate_consolidated_metadata_v3( if "metadata" in env: entries = env["metadata"] if not isinstance(entries, Mapping): - problems.append(ValidationProblem(("metadata",), "expected a mapping", "invalid_type")) + problems.append(ValidationProblem(("metadata",), "expected an object", "invalid_type")) else: for key, entry in cast("Mapping[object, object]", entries).items(): if not isinstance(key, str): @@ -650,12 +651,12 @@ def validate_group_metadata_v3( Unknown top-level keys are allowed (they map to `extra_fields`); a reader must understand each one that does not say `must_understand: false`, which the model reports as `must_understand_fields`. A - `consolidated_metadata` key, if present, is deep-validated (envelope - and entries) via `validate_consolidated_metadata_v3`, each array in it - read as `validate_array_metadata_v3` reads one, in `context`. + `consolidated_metadata` member, if present, is validated too: its + envelope, and each document it holds by its path, each array read as + `validate_array_metadata_v3` reads one, in `context`. """ if not isinstance(value, Mapping): - return (ValidationProblem((), "expected a mapping", "invalid_type"),) + return (ValidationProblem((), "expected an object", "invalid_type"),) doc = cast("Mapping[object, object]", value) problems: list[ValidationProblem] = list(_missing_keys(GROUP_METADATA_REQUIRED_KEYS_V3, doc)) problems.extend( @@ -709,7 +710,7 @@ def validate_group_metadata_v2(value: object) -> tuple[ValidationProblem, ...]: optional `attributes` mapping folded in from `.zattrs`. """ if not isinstance(value, Mapping): - return (ValidationProblem((), "expected a mapping", "invalid_type"),) + return (ValidationProblem((), "expected an object", "invalid_type"),) doc = cast("Mapping[object, object]", value) problems: list[ValidationProblem] = list(_missing_keys(GROUP_METADATA_REQUIRED_KEYS_V2, doc)) problems.extend(_unexpected_keys(GROUP_METADATA_STANDARD_KEYS_V2, doc)) diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/_common.py b/packages/zarr-metadata/src/zarr_metadata/v3/_common.py index e6d76d8bea..bd412783fd 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/_common.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/_common.py @@ -35,10 +35,10 @@ def validate_metadata_field_v3( ) -> tuple[ValidationProblem, ...]: """Return every reason `value` is not a v3 metadata field. - A metadata field is a bare name string or a mapping containing `name` and - optional `configuration` and `must_understand` members: an envelope, as - `envelope_problems` judges it, around a configuration whose members are - JSON. + A metadata field is a bare name, or an envelope around a configuration + whose members are JSON: an object of a string `name`, a `configuration` + that is an object of string keys, a boolean `must_understand`, and + nothing else. """ envelope = envelope_problems(value, allow_must_understand_false=allow_must_understand_false) return (*envelope, *_configuration_json_problems(value)) @@ -82,7 +82,7 @@ def envelope_problems( configuration = field["configuration"] if not isinstance(configuration, Mapping): problems.append( - ValidationProblem(("configuration",), "expected a mapping", "invalid_type") + ValidationProblem(("configuration",), "expected an object", "invalid_type") ) elif not all(isinstance(k, str) for k in cast("Mapping[object, object]", configuration)): problems.append( diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py b/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py index 0d7ab15abc..5a9a9926fe 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py @@ -51,7 +51,7 @@ from typing_extensions import TypeAliasType, TypedDict, TypeVar, is_typeddict from zarr_metadata._common import JSONValue, ZarrV3NamedConfigJSON -from zarr_metadata._json import ValidationProblem, refine_json +from zarr_metadata._json import ValidationProblem, refine_json, shown from zarr_metadata._typed_json import ( Loc, Parsed, @@ -192,6 +192,11 @@ def __post_init__(self) -> None: msg = f"{self.name!r}: {error}" raise TypeError(msg) from error + def __repr__(self) -> str: + # Short, as a reading that holds definitions shows them: in full, a + # definition's repr is each function it holds, at its address. + return f"{type(self).__name__}(name={self.name!r})" + def _refusal(self) -> str | None: """What is wrong with the members a kind adds; None when nothing is, or it adds none.""" return None @@ -299,7 +304,7 @@ def _carrying_name( return f"r{configuration['bits']}" -@dataclass(frozen=True, kw_only=True, slots=True) +@dataclass(frozen=True, kw_only=True, slots=True, repr=False) class DataTypeDefinition(Definition[C]): """A data type, and the fill value an array of it takes. @@ -349,7 +354,7 @@ def _fill_value_parser(annotation: object) -> Parser: return parser(annotation, no_leaf) -@dataclass(frozen=True, kw_only=True, slots=True) +@dataclass(frozen=True, kw_only=True, slots=True, repr=False) class ChunkGridDefinition(Definition[C]): """A chunk grid, and the arrays it fits. @@ -374,7 +379,7 @@ class ChunkGridDefinition(Definition[C]): """The lengths its chunks take along each axis of an array of a shape it fits, None where unknown.""" -@dataclass(frozen=True, kw_only=True, slots=True) +@dataclass(frozen=True, kw_only=True, slots=True, repr=False) class ChunkKeyEncodingDefinition(Definition[C]): """A chunk key encoding.""" @@ -397,7 +402,7 @@ class ChunkKeyEncodingDefinition(Definition[C]): """The functions no codec of a kind is asked: a bytes -> bytes codec is handed bytes, and only an array -> array codec hands on a chunk.""" -@dataclass(frozen=True, kw_only=True, slots=True) +@dataclass(frozen=True, kw_only=True, slots=True, repr=False) class CodecDefinition(Definition[C]): """A codec: what it does to what it is handed, and whether the size of what it gives out is static. @@ -452,7 +457,7 @@ def _refusal(self) -> str | None: return None -@dataclass(frozen=True, kw_only=True, slots=True) +@dataclass(frozen=True, kw_only=True, slots=True, repr=False) class StorageTransformerDefinition(Definition[C]): """A storage transformer.""" @@ -551,7 +556,7 @@ def _field(annotation: object) -> Parser | None: def parse(value: object, loc: Loc) -> Parsed: if not isinstance(value, (str, Mapping)): - return value, problem(loc, f"expected a metadata field, got {value!r}") + return value, problem(loc, f"expected a metadata field, got {shown(value)}") # Refined JSON, which the checker only knows as `object`. return _NestedField(loc, kind, cast("JSONValue", value), static), () @@ -740,7 +745,11 @@ def named_configuration( return name, None, () configuration = entry["configuration"] if not isinstance(configuration, Mapping): - return name, None, problem(("configuration",), f"expected an object, got {configuration!r}") + return ( + name, + None, + problem(("configuration",), f"expected an object, got {shown(configuration)}"), + ) return name, cast("Mapping[str, object]", configuration), () @@ -982,8 +991,10 @@ def resolve( sits, as with a document's fields. A name nothing claims is `out_of_scope`: an unmodelled extension, left unjudged, which is what keeps the format open. `loc` - prefixes every problem. `kind` is one of `KINDS`, with or without - type arguments; anything else is a `TypeError`. + prefixes every problem. `kind` is one of the five kinds -- + `CodecDefinition`, `DataTypeDefinition`, `ChunkGridDefinition`, + `ChunkKeyEncodingDefinition`, `StorageTransformerDefinition` -- with + or without type arguments; anything else is a `TypeError`. """ asked = as_kind(kind) refined, problems = refine_json(data, loc) diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/_registry.py b/packages/zarr-metadata/src/zarr_metadata/v3/_registry.py index e5beff8c64..547523d7f9 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/_registry.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/_registry.py @@ -113,6 +113,11 @@ def definitions(self) -> tuple[Definition[Any], ...]: """Every definition in scope, kind by kind.""" return tuple(entry for table in self.tables.values() for entry in table.values()) + def __repr__(self) -> str: + # Short, as a default argument shows it: in full, a scope's repr is + # every definition's, and `help` of a validator runs to pages. + return f"Context(<{len(self.definitions())} definitions>)" + def claimant(self, kind: type[D], name: str) -> D | None: """The definition of `kind` in scope that reads `name`, a name a document writes; None if none does. diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/codec/bytes.py b/packages/zarr-metadata/src/zarr_metadata/v3/codec/bytes.py index c98f1d712d..89b72df98e 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/codec/bytes.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/codec/bytes.py @@ -9,7 +9,7 @@ from typing_extensions import TypedDict -from zarr_metadata._json import ValidationProblem +from zarr_metadata._json import ValidationProblem, shown from zarr_metadata.v3._definition import ( Chunk, CodecDefinition, @@ -101,7 +101,7 @@ def _chunk_rules( elif storage == "variable_length": yield ValidationProblem( (), - f"expected a data type of fixed size, got {written!r}, whose values vary in size", + f"expected a data type of fixed size, got {shown(written)}, whose values vary in size", "invalid_value", ) diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/codec/cast_value.py b/packages/zarr-metadata/src/zarr_metadata/v3/codec/cast_value.py index d5a018a1e6..14be64d638 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/codec/cast_value.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/codec/cast_value.py @@ -10,7 +10,7 @@ from typing_extensions import TypedDict from zarr_metadata._common import JSONValue -from zarr_metadata._json import ValidationProblem +from zarr_metadata._json import ValidationProblem, shown from zarr_metadata.v3._definition import ( Chunk, CodecDefinition, @@ -160,14 +160,14 @@ def _rules( if name in _NO_REAL_NUMBERS: yield ValidationProblem( ("data_type",), - f"expected a data type that models real numbers, got {written!r}", + f"expected a data type that models real numbers, got {shown(written)}", "invalid_value", ) return if configuration.get("out_of_range") == "wrap" and name in FLOATING_POINT: yield ValidationProblem( ("out_of_range",), - f"expected an integral data_type to wrap to, got {written!r}", + f"expected an integral data_type to wrap to, got {shown(written)}", "invalid_value", ) for at, scalar in _scalars(configuration, "target"): @@ -192,7 +192,7 @@ def _chunk_rules( written, _, _ = named_configuration(source.json) yield ValidationProblem( (), - f"expected a chunk of a data type that models real numbers, got {written!r}", + f"expected a chunk of a data type that models real numbers, got {shown(written)}", "invalid_value", ) return diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/codec/scale_offset.py b/packages/zarr-metadata/src/zarr_metadata/v3/codec/scale_offset.py index e4807fe6fe..d6bcb854b8 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/codec/scale_offset.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/codec/scale_offset.py @@ -10,7 +10,7 @@ from typing_extensions import TypedDict from zarr_metadata._common import JSONValue -from zarr_metadata._json import ValidationProblem +from zarr_metadata._json import ValidationProblem, shown from zarr_metadata.v3._definition import ( Chunk, CodecDefinition, @@ -102,7 +102,7 @@ def _chunk_rules( written, _, _ = named_configuration(source.json) yield ValidationProblem( (), - f"expected a chunk of a data type with arithmetic, got {written!r}", + f"expected a chunk of a data type with arithmetic, got {shown(written)}", "invalid_value", ) return diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/codec/transpose.py b/packages/zarr-metadata/src/zarr_metadata/v3/codec/transpose.py index 2fb88338f9..83805c4c1a 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/codec/transpose.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/codec/transpose.py @@ -9,7 +9,7 @@ from typing_extensions import TypedDict -from zarr_metadata._json import ValidationProblem +from zarr_metadata._json import ValidationProblem, shown from zarr_metadata.v3._definition import Chunk, CodecDefinition, Nested TRANSPOSE_CODEC_NAME: Final = "transpose" @@ -60,7 +60,7 @@ def _rules( if sorted(order) != list(range(len(order))): yield ValidationProblem( ("order",), - f"expected a permutation of 0..{len(order) - 1}, got {order!r}", + f"expected a permutation of 0..{len(order) - 1}, got {shown(order)}", "invalid_value", ) diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/_float.py b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/_float.py index 3b81bd7a7c..444d13a422 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/_float.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/_float.py @@ -13,7 +13,7 @@ from collections.abc import Callable, Iterable, Iterator from typing import Any, Literal, get_args -from zarr_metadata._json import ValidationProblem +from zarr_metadata._json import ValidationProblem, choices, shown from zarr_metadata.v3._definition import EmptyConfiguration, Nested FloatSpecialFillValue = Literal["NaN", "Infinity", "-Infinity"] @@ -48,8 +48,8 @@ def _float_fill_value( except ValueError: yield ValidationProblem( (), - f"expected a number, one of {get_args(FloatSpecialFillValue)!r}, or a {name} " - f"hex string, got {value!r}", + f"expected a number, {choices(get_args(FloatSpecialFillValue))}, or a {name} " + f"hex string, got {shown(value)}", "invalid_value", ) diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/_numpy_time.py b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/_numpy_time.py index 7023495b93..26beaccdeb 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/_numpy_time.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/_numpy_time.py @@ -58,7 +58,7 @@ def numpy_time_fill_value_rules( """ if isinstance(value, int) and not -(2**63) <= value <= 2**63 - 1: yield ValidationProblem( - (), f"expected a signed 64-bit integer or 'NaT', got {value}", "invalid_value" + (), f'expected a signed 64-bit integer or "NaT", got {value}', "invalid_value" ) diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/bytes.py b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/bytes.py index 530b88a5c2..c815ac50e8 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/bytes.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/bytes.py @@ -8,7 +8,7 @@ from collections.abc import Iterator from typing import Final, Literal, NewType -from zarr_metadata._json import ValidationProblem +from zarr_metadata._json import ValidationProblem, shown from zarr_metadata.v3._definition import ( DataTypeDefinition, EmptyConfiguration, @@ -60,7 +60,7 @@ def _fill_value_rules( base64_bytes(value) except ValueError: yield ValidationProblem( - (), f"expected standard-alphabet base64, got {value!r}", "invalid_value" + (), f"expected standard-alphabet base64, got {shown(value)}", "invalid_value" ) diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/struct.py b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/struct.py index 3f3d284204..a8dc844c33 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/struct.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/struct.py @@ -10,7 +10,7 @@ from typing_extensions import ReadOnly, TypedDict from zarr_metadata._common import JSONValue -from zarr_metadata._json import ValidationProblem +from zarr_metadata._json import ValidationProblem, shown from zarr_metadata.v3._definition import ( DataTypeDefinition, DataTypeField, @@ -99,7 +99,7 @@ def _rules(configuration: StructConfiguration, nested: Nested) -> Iterator[Valid written, _, _ = named_configuration(field_type.json) yield ValidationProblem( ("fields", index, "data_type"), - f"expected a data type of fixed size, got {written!r}, whose values vary in size", + f"expected a data type of fixed size, got {shown(written)}, whose values vary in size", "invalid_value", ) diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/definition.py b/packages/zarr-metadata/src/zarr_metadata/v3/definition.py index cd833aa87c..1e3aa4ba90 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/definition.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/definition.py @@ -79,7 +79,8 @@ as the scope read them: a field that is read keeps what it read inside it as `Resolved.nested`, a `Nested` mapping by where each sits, so a struct's rules reach its field types. `judge`, which reads in no scope, -hands them none. +hands them none. A rule's message shows a value as the package's own +messages do, as JSON, with `shown`: `null`, `[1, 2]`, `"C"`. from collections.abc import Iterator @@ -232,7 +233,7 @@ class creation. """ from zarr_metadata._common import JSONValue -from zarr_metadata._json import MetadataValidationError, ProblemKind, ValidationProblem +from zarr_metadata._json import MetadataValidationError, ProblemKind, ValidationProblem, shown from zarr_metadata._typed_json import Loc, check from zarr_metadata.v3._common import ZarrV3MetadataFieldJSON from zarr_metadata.v3._definition import ( @@ -308,5 +309,6 @@ class creation. "fill_value_problems", "read_pipeline", "resolve", + "shown", "storage_of", ] diff --git a/packages/zarr-metadata/tests/model/test_array.py b/packages/zarr-metadata/tests/model/test_array.py index c3ecc7b2a2..161c909469 100644 --- a/packages/zarr-metadata/tests/model/test_array.py +++ b/packages/zarr-metadata/tests/model/test_array.py @@ -1974,7 +1974,7 @@ def test_v2_null_dimension_separator_rejected() -> None: document grammar has no null spelling for this field.""" doc = dict(ZarrV2ArrayMetadata.create_default().to_json()) | {"dimension_separator": None} problems = validate_array_metadata_v2(doc) - assert [(p.loc, p.kind) for p in problems] == [(("dimension_separator",), "invalid_value")] + assert [(p.loc, p.kind) for p in problems] == [(("dimension_separator",), "invalid_type")] def test_dimension_names_absent_and_all_null_are_distinct() -> None: @@ -1995,3 +1995,34 @@ def test_dimension_names_absent_and_all_null_are_distinct() -> None: assert "dimension_names" not in absent.to_json() assert absent.to_json() == absent_doc assert explicit.to_json() == explicit_doc + + +@pytest.mark.parametrize( + ("validate", "document", "member", "value", "kind"), + [ + (validate_array_metadata_v3, ZarrV3ArrayMetadata, "zarr_format", "3", "invalid_type"), + (validate_array_metadata_v3, ZarrV3ArrayMetadata, "zarr_format", 2, "invalid_value"), + (validate_array_metadata_v3, ZarrV3ArrayMetadata, "node_type", 5, "invalid_type"), + (validate_array_metadata_v3, ZarrV3ArrayMetadata, "node_type", "group", "invalid_value"), + (validate_array_metadata_v2, ZarrV2ArrayMetadata, "order", 1, "invalid_type"), + (validate_array_metadata_v2, ZarrV2ArrayMetadata, "order", "Q", "invalid_value"), + ( + validate_array_metadata_v2, + ZarrV2ArrayMetadata, + "dimension_separator", + ":", + "invalid_value", + ), + ], +) +def test_error_a_member_outside_the_values_it_takes( + validate: Callable[[object], tuple[ValidationProblem, ...]], + document: type[ZarrV3ArrayMetadata | ZarrV2ArrayMetadata], + member: str, + value: object, + kind: str, +) -> None: + # Of the wrong type when none of the values it takes is of its JSON + # type -- a string where a number belongs -- else of the wrong value. + written = {**document.create_default().to_json(), member: value} + assert [(p.loc, p.kind) for p in validate(written)] == [((member,), kind)] diff --git a/packages/zarr-metadata/tests/model/test_group.py b/packages/zarr-metadata/tests/model/test_group.py index 845e303fd4..a13a37a5e6 100644 --- a/packages/zarr-metadata/tests/model/test_group.py +++ b/packages/zarr-metadata/tests/model/test_group.py @@ -265,7 +265,7 @@ def test_group_v2_omits_empty_attributes() -> None: def test_group_v2_not_a_mapping() -> None: """parse_group_metadata_v2 rejects a non-mapping document.""" - with pytest.raises(MetadataValidationError, match="expected a mapping"): + with pytest.raises(MetadataValidationError, match="expected an object"): parse_group_metadata_v2([1, 2, 3]) @@ -339,7 +339,7 @@ def test_consolidated_v3_entry_without_node_type_rejected() -> None: def test_consolidated_v3_not_a_mapping() -> None: """from_json rejects a non-mapping consolidated document.""" - with pytest.raises(MetadataValidationError, match="expected a mapping"): + with pytest.raises(MetadataValidationError, match="expected an object"): ZarrV3ConsolidatedMetadata.from_json(5) @@ -369,6 +369,16 @@ def test_consolidated_v2_verbatim_roundtrip() -> None: assert model.to_json() == doc +@pytest.mark.parametrize(("value", "kind"), [("1", "invalid_type"), (2, "invalid_value")]) +def test_error_consolidated_v2_format_other_than_1(value: object, kind: str) -> None: + document = {"zarr_consolidated_format": value, "metadata": {}} + with pytest.raises(MetadataValidationError) as raised: + ZarrV2ConsolidatedMetadata.from_json(document) + assert [(p.loc, p.kind) for p in raised.value.problems] == [ + (("zarr_consolidated_format",), kind) + ] + + def test_consolidated_v2_key_value_roundtrip() -> None: """from_key_value(to_key_value()) is the identity for .zmetadata documents.""" model = ZarrV2ConsolidatedMetadata.from_json( @@ -395,7 +405,7 @@ def test_consolidated_v2_envelope_validation() -> None: def test_consolidated_v2_not_a_mapping() -> None: """from_json rejects a non-mapping .zmetadata document.""" - with pytest.raises(MetadataValidationError, match="expected a mapping"): + with pytest.raises(MetadataValidationError, match="expected an object"): ZarrV2ConsolidatedMetadata.from_json([1]) diff --git a/packages/zarr-metadata/tests/model/test_refine_json.py b/packages/zarr-metadata/tests/model/test_refine_json.py index 5ce759d40a..7f2d514c96 100644 --- a/packages/zarr-metadata/tests/model/test_refine_json.py +++ b/packages/zarr-metadata/tests/model/test_refine_json.py @@ -8,7 +8,7 @@ import pytest -from zarr_metadata._json import refine_json, refine_user_data +from zarr_metadata._json import refine_json, refine_user_data, shown if TYPE_CHECKING: from collections.abc import Callable @@ -101,3 +101,22 @@ def test_a_value_nested_hundreds_deep_is_read() -> None: refined, problems = refine_json(deep) assert problems == () assert refined is not None + + +@pytest.mark.parametrize( + ("value", "text"), + [ + (None, "null"), + (True, "true"), + ("C", '"C"'), + ((1, (2,)), "[1, [2]]"), + ({"a": None}, '{"a": null}'), + (float("nan"), "NaN"), + ({1: 2}, "{1: 2}"), + ], + ids=["null", "true", "string", "array", "object", "non-finite", "not-json"], +) +def test_a_value_is_shown_as_the_json_a_document_writes(value: object, text: str) -> None: + # What a problem's message says a document holds: its JSON, and the + # value's repr only when it is not JSON. + assert shown(value) == text diff --git a/packages/zarr-metadata/tests/test_typed_json.py b/packages/zarr-metadata/tests/test_typed_json.py index f225ba9eae..cedb2aec90 100644 --- a/packages/zarr-metadata/tests/test_typed_json.py +++ b/packages/zarr-metadata/tests/test_typed_json.py @@ -38,6 +38,7 @@ from zarr_metadata.typed_json import check from zarr_metadata.v2.array import ZarrV2ArrayMetadataJSON, ZarrV2DataTypeMetadata from zarr_metadata.v3.array import ZarrV3ArrayMetadataJSON +from zarr_metadata.v3.data_type.float32 import Float32FillValue if TYPE_CHECKING: from _pytest.mark import ParameterSet @@ -171,7 +172,8 @@ def test_a_value_of_the_shape_reads_as_itself( (str, 1, (), "invalid_type"), (None, 0, (), "invalid_type"), (Literal["a"], "b", (), "invalid_value"), - (Literal[1], True, (), "invalid_value"), + (Literal[1], True, (), "invalid_type"), + (Literal["0.5"], 0.5, (), "invalid_type"), (tuple[int, ...], (1, "x"), (1,), "invalid_type"), (tuple[int, ...], 5, (), "invalid_type"), (int | str, 2.5, (), "invalid_type"), @@ -184,6 +186,7 @@ def test_a_value_of_the_shape_reads_as_itself( "invalid_type", ), (Blosc | Gzip, {"name": "zstd", "configuration": {}}, ("name",), "invalid_value"), + (Blosc | Gzip, {"name": 5, "configuration": {}}, ("name",), "invalid_type"), (Blosc | Gzip, {"configuration": {}}, ("name",), "missing_key"), (Triple | Pair, {"a": 1, "b": "x"}, ("b",), "invalid_type"), ], @@ -195,6 +198,7 @@ def test_a_value_of_the_shape_reads_as_itself( "zero-is-not-null", "literal-other", "literal-bool-is-not-int", + "literal-of-another-type", "element", "not-a-sequence", "no-branch", @@ -202,6 +206,7 @@ def test_a_value_of_the_shape_reads_as_itself( "not-json", "a-tag-picks-the-branch-that-reports", "a-tag-no-branch-has", + "a-tag-of-another-type", "a-tag-missing", "untagged-the-closest-branch-reports", ], @@ -871,9 +876,39 @@ def test_a_shape_is_what_a_union_dispatches_on(annotation: object, shape: str | assert shape_of(annotation) == shape +@pytest.mark.parametrize( + ("annotation", "value", "message"), + [ + (Blosc | Gzip, 3, "expected an object, got 3"), + (Literal["0.5"], 0.5, 'expected "0.5", got 0.5'), + (Literal["C", "F"], "Q", 'expected one of ["C", "F"], got "Q"'), + (tuple[str, ...], "nuclei", 'expected an array, got "nuclei"'), + (str, None, "expected a string, got null"), + ], + ids=["a-union-of-objects", "one-value", "values", "array", "null"], +) +def test_a_message_names_what_was_expected_and_shows_the_json_it_got( + annotation: object, value: object, message: str +) -> None: + assert [problem.message for problem in _read(annotation, value)[1]] == [message] + + def test_a_shape_is_described_as_a_message_would_name_it() -> None: - assert describe(Literal[0, "auto"]) == "one of ('auto', 0)" + assert describe(Literal[0, "auto"]) == 'one of ["auto", 0]' + assert describe(Literal["0.5"]) == '"0.5"' assert describe(int | None) == "an integer or null" + assert describe(Blosc | Gzip) == "an object" + assert describe(float | int | None) == "a number or null" + # An array's elements are named as many. + assert describe(tuple[int, ...]) == "an array of integers" + assert describe(tuple[int | None, ...]) == "an array of integers or nulls" + assert describe(tuple[tuple[str, ...], ...]) == "an array of arrays of strings" + assert describe(tuple[Literal["C", "F"], ...]) == 'an array of values in ["C", "F"]' + assert describe(int | tuple[int, ...]) == "an integer or an array of integers" + # A broader shape takes in a narrower one: the hex string a float32 + # is spelled as, the strings of its non-finite values. + assert describe(Float32FillValue) == "a number or a string" + assert describe(Literal["C", "F"] | None) == 'one of ["C", "F"] or null' assert describe(Width) == "an integer" assert describe(ZarrV2DataTypeMetadata) == "a ZarrV2DataTypeMetadata" diff --git a/packages/zarr-metadata/tests/v3/test_definitions.py b/packages/zarr-metadata/tests/v3/test_definitions.py index 87626bcab9..26192788f3 100644 --- a/packages/zarr-metadata/tests/v3/test_definitions.py +++ b/packages/zarr-metadata/tests/v3/test_definitions.py @@ -681,6 +681,17 @@ def test_a_scope_takes_a_name_over() -> None: assert scope.claimant(ChunkGridDefinition, "regular") is REGULAR_CHUNK_GRID +def test_a_scope_is_shown_by_how_many_definitions_it_holds() -> None: + # Short, as a validator's default argument is shown by `help`. + assert repr(CORE) == f"Context(<{len(CORE.definitions())} definitions>)" + + +def test_a_definition_is_shown_by_its_kind_and_name() -> None: + # Short, as a reading that holds it shows it. + assert repr(GZIP_CODEC) == "CodecDefinition(name='gzip')" + assert repr(INT8_DATA_TYPE) == "DataTypeDefinition(name='int8')" + + class Unreadable(TypedDict, closed=True): members: set[int] From f3bfd68a61f8b56498f4f566477bc05fe13b7d8d Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Mon, 28 Sep 2026 14:13:46 +0200 Subject: [PATCH 03/94] feat(zarr-metadata): read_array_metadata_v3 returns what it read beside the problems The v3 array validator read every extension point, the chunks the codecs are handed and each codec's stage, then kept only the problems, so a caller wanting any of it -- a core-only policy, each codec's stage, what was left unjudged -- resolved every field again. The reading is now returned as a ZarrV3ArrayMetadataReading, and validate_array_metadata_v3 is its problems. fields() gives each field with where it sits in the document, and after it the fields it holds, as the new fields_of gives them for any field resolve read. Resolved gains the kind a field was read as, which a field nothing in scope claims keeps too: before, a nested field nothing claimed told nothing of its kind, and a pipelines hook naming data types nothing claims read them as codecs. Built by hand, a Resolved is checked as resolve builds one: its kind one of the five, and its definition of that kind. A field its rules refuse keeps the fields it read inside it, whose problems it already reported, so fields() leaves none of them out. Assisted-by: ClaudeCode:claude-opus-5-5 --- packages/zarr-metadata/README.md | 22 ++ packages/zarr-metadata/changes/372.feature.md | 12 + packages/zarr-metadata/docs/index.md | 22 ++ .../src/zarr_metadata/model/__init__.py | 23 +- .../src/zarr_metadata/model/_validation.py | 143 +++++++--- .../src/zarr_metadata/v3/_definition.py | 108 ++++++-- .../src/zarr_metadata/v3/_pipeline.py | 13 +- .../v3/codec/sharding_indexed.py | 5 +- .../src/zarr_metadata/v3/definition.py | 15 +- .../tests/model/test_read_array_metadata.py | 261 ++++++++++++++++++ .../tests/v3/test_definitions.py | 93 ++++++- .../tests/v3/test_fill_values.py | 2 +- .../zarr-metadata/tests/v3/test_pipelines.py | 29 +- 13 files changed, 664 insertions(+), 84 deletions(-) create mode 100644 packages/zarr-metadata/changes/372.feature.md create mode 100644 packages/zarr-metadata/tests/model/test_read_array_metadata.py diff --git a/packages/zarr-metadata/README.md b/packages/zarr-metadata/README.md index 8b01bbbff8..75dfdbf5c8 100644 --- a/packages/zarr-metadata/README.md +++ b/packages/zarr-metadata/README.md @@ -91,6 +91,28 @@ Three choices the specs' words leave open, or settle two ways: says chunk sizes are greater than zero. The package follows the core spec, which zarr-python 3.0 and 3.1 wrote for an empty dimension. +`read_array_metadata_v3` returns what the validator read, beside the +problems: each field as the scope read it, with where it sits and the +kind it was read as, and each codec with the chunk it is handed. A +consumer's own policy is a walk over the fields, with nothing read +twice. Which fields go beyond the core spec, say -- a field that names +nothing is a problem already: + +```python +from zarr_metadata.model import read_array_metadata_v3 +from zarr_metadata.v3.definition import CORE + +reading, problems = read_array_metadata_v3(raw) +beyond_core = [ + loc + for loc, field in reading.fields() + if field.name is not None and CORE.claimant(field.read_as, field.name) is None +] +``` + +A member the spec does not define is not a field; the model's +`must_understand_fields` names those a reader must understand. + The Pydantic integration's generated JSON Schemas express independently checkable document structure and field constraints, but they are not a replacement for runtime model validation. Standard JSON Schema treats a diff --git a/packages/zarr-metadata/changes/372.feature.md b/packages/zarr-metadata/changes/372.feature.md new file mode 100644 index 0000000000..e56d18a1b6 --- /dev/null +++ b/packages/zarr-metadata/changes/372.feature.md @@ -0,0 +1,12 @@ +`read_array_metadata_v3` returns what the v3 array validator read, +beside the problems it found, as a `ZarrV3ArrayMetadataReading`: each +extension point as the scope read it, the chunks the codecs are handed, +and the codecs as a pipeline, each with the chunk it is handed. Its +`fields()` gives each field with where it sits in the document, and +after it the fields it holds -- a shard's codecs, a struct's field types +-- as `fields_of` gives them for any field `resolve` read, so a policy +over a document's fields, the core spec's alone, say, reads nothing +twice. `Resolved` says the kind a field was read as, `read_as`, which a +field nothing in scope claims keeps too, and the `name` it was written +with, and a field its rules refuse keeps the fields it holds in +`nested`, as they were read. diff --git a/packages/zarr-metadata/docs/index.md b/packages/zarr-metadata/docs/index.md index ed7f68f2b2..f3a8009ef0 100644 --- a/packages/zarr-metadata/docs/index.md +++ b/packages/zarr-metadata/docs/index.md @@ -106,6 +106,28 @@ Three choices the specs' words leave open, or settle two ways: says chunk sizes are greater than zero. The package follows the core spec, which zarr-python 3.0 and 3.1 wrote for an empty dimension. +`read_array_metadata_v3` returns what the validator read, beside the +problems: each field as the scope read it, with where it sits and the +kind it was read as, and each codec with the chunk it is handed. A +consumer's own policy is a walk over the fields, with nothing read +twice. Which fields go beyond the core spec, say -- a field that names +nothing is a problem already: + +```python +from zarr_metadata.model import read_array_metadata_v3 +from zarr_metadata.v3.definition import CORE + +reading, problems = read_array_metadata_v3(raw) +beyond_core = [ + loc + for loc, field in reading.fields() + if field.name is not None and CORE.claimant(field.read_as, field.name) is None +] +``` + +A member the spec does not define is not a field; the model's +`must_understand_fields` names those a reader must understand. + ## Scope At minimum, this library supports what Zarr-Python needs: the complete diff --git a/packages/zarr-metadata/src/zarr_metadata/model/__init__.py b/packages/zarr-metadata/src/zarr_metadata/model/__init__.py index a148392584..9236221b88 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/__init__.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/__init__.py @@ -5,15 +5,16 @@ structure and, in a v3 document, read each extension point (codecs, chunk grids, data types, ...) through the definition that claims its name in a scope, `CORE_AND_EXTENSIONS` unless a `context` is passed, and judge the -fill value against the data type it names, the chunk grid against -the shape, and the codecs as a pipeline, each against the chunk it is -handed. Each document concept gets a -`validate_*` function returning every problem found (a tuple of -`ValidationProblem`, each with a machine-readable `kind`), an `is_*` type -guard, and a `parse_*` function that narrows or raises -`MetadataValidationError`. Model `from_json` / `from_key_value` constructors -raise `MetadataValidationError` for every ingestion failure, including -missing store keys and undecodable bytes, and the v3 ones take the same +fill value against the data type it names, the chunk grid against the +shape, and the codecs as a pipeline, each against the chunk it is +handed. Each document concept gets a `validate_*` function returning +every problem found (a tuple of `ValidationProblem`, each with a +machine-readable `kind`), an `is_*` type guard, and a `parse_*` function +that narrows or raises `MetadataValidationError`; a v3 array document +also gets `read_array_metadata_v3`, which returns what it read beside +them. Model `from_json` / `from_key_value` constructors raise +`MetadataValidationError` for every ingestion failure, including missing +store keys and undecodable bytes, and the v3 ones take the same `context`. """ @@ -51,6 +52,7 @@ GROUP_METADATA_REQUIRED_KEYS_V2, GROUP_METADATA_REQUIRED_KEYS_V3, GROUP_METADATA_STANDARD_KEYS_V3, + ZarrV3ArrayMetadataReading, is_array_metadata_v2, is_array_metadata_v3, is_group_metadata_v2, @@ -59,6 +61,7 @@ parse_array_metadata_v3, parse_group_metadata_v2, parse_group_metadata_v3, + read_array_metadata_v3, validate_array_metadata_v2, validate_array_metadata_v3, validate_group_metadata_v2, @@ -130,6 +133,7 @@ "ZarrV2GroupMetadataStoreKey", "ZarrV3ArrayMetadata", "ZarrV3ArrayMetadataPartial", + "ZarrV3ArrayMetadataReading", "ZarrV3ArrayMetadataStoreKey", "ZarrV3ConsolidatedMetadata", "ZarrV3GroupMetadata", @@ -149,6 +153,7 @@ "parse_group_metadata_v3", "parse_json", "parse_metadata_field_v3", + "read_array_metadata_v3", "validate_array_metadata_v2", "validate_array_metadata_v3", "validate_group_metadata_v2", diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_validation.py b/packages/zarr-metadata/src/zarr_metadata/model/_validation.py index 3d1b137dd2..5256222674 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_validation.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/_validation.py @@ -7,13 +7,13 @@ name nothing in the scope claims is left unjudged. A v3 fill value is judged against the data type it names, the chunk grid against the shape, and the codecs as a pipeline, each against the chunk it is -handed. Each concept -gets a `validate_*` function returning every problem found, an `is_*` -type guard, and a `parse_*` function that narrows or raises -`MetadataValidationError`. The guards are `TypeGuard`s, -not `TypeIs`: True narrows a value to its document type, and False says -nothing about its type, since a value can be well typed and still not a -valid document. +handed. Each concept gets a `validate_*` function returning every +problem found, an `is_*` type guard, and a `parse_*` function that +narrows or raises `MetadataValidationError`; a v3 array document also +gets `read_array_metadata_v3`, which returns what it read beside them. +The guards are `TypeGuard`s, not `TypeIs`: True narrows a value to its +document type, and False says nothing about its type, since a value can +be well typed and still not a valid document. Every `ValidationProblem` carries a machine-readable `kind` alongside its human-readable `message`, so consumers can dispatch on the failure mode @@ -23,9 +23,11 @@ from __future__ import annotations +import dataclasses import json from collections.abc import Mapping, Sequence -from typing import Any, Final, TypeGuard, TypeVar, cast +from dataclasses import dataclass +from typing import TYPE_CHECKING, Any, Final, TypeGuard, TypeVar, cast from zarr_metadata._json import ( MetadataValidationError, @@ -51,14 +53,20 @@ Resolved, StorageTransformerDefinition, chunk_grid_lengths, + fields_of, fill_value_problems, resolve, ) -from zarr_metadata.v3._pipeline import read_pipeline +from zarr_metadata.v3._pipeline import Stage, read_pipeline from zarr_metadata.v3._registry import CORE_AND_EXTENSIONS, Context from zarr_metadata.v3.array import ZarrV3ArrayMetadataJSON from zarr_metadata.v3.group import ZarrV3GroupMetadataJSON +if TYPE_CHECKING: + from collections.abc import Iterator + + from zarr_metadata._typed_json import Loc + # The standard top-level keys of a v3 array metadata document. Anything outside # this set is an extension field. Built from the TypedDict's required/optional # key sets (which resolve inherited keys, unlike `__annotations__`). @@ -350,30 +358,63 @@ def _validate_attributes(value: object) -> tuple[ValidationProblem, ...]: """Its lists of extension points, and the kind each entry is read as.""" -def validate_array_metadata_v3( - value: object, *, context: Context = CORE_AND_EXTENSIONS -) -> tuple[ValidationProblem, ...]: - """Return every reason `value` is not a valid v3 array document. +@dataclass(frozen=True, slots=True) +class ZarrV3ArrayMetadataReading: + """A v3 array document as a scope read it: each extension point, and its codecs as a pipeline. - Its structure, and each extension point read through the definition - that claims its name in `context`: a gzip `level` out of range, a key a - codec's configuration does not declare. The fill value is judged - against the data type as `context` read it -- an `int8` fill value of - 300 -- and the chunk grid against the shape: a regular grid with a - chunk length for each of two dimensions, over an array of three. The - codecs are read as a pipeline: in order, each judged against the chunk - it is handed -- a `transpose` whose `order` has another number of - axes, a shard its inner chunks do not divide -- and a shard's inner - and index codecs too. - A name nothing in `context` claims is left unjudged, with any fill - value of it, and a codec of that name leaves the codec after it - handed a chunk nothing is known of. Unknown top-level keys are - allowed (they map to `extra_fields`); a reader must understand each - one that does not say `must_understand: false`, which the model - reports as `must_understand_fields`. + A field the document does not hold is None, and a list of them it does + not hold as a list is empty. + """ + + data_type: Resolved[DataTypeDefinition[Any]] | None = None + """The data type, as the scope read it.""" + chunk_grid: Resolved[ChunkGridDefinition[Any]] | None = None + """The chunk grid, as the scope read it.""" + chunk_key_encoding: Resolved[ChunkKeyEncodingDefinition[Any]] | None = None + """The chunk key encoding, as the scope read it.""" + chunk: Chunk = dataclasses.field(default_factory=Chunk) + """The chunks the codecs are handed: the lengths the grid's chunks take along each axis of the shape, of the data type.""" + pipeline: tuple[Stage, ...] = () + """The codecs, read as a pipeline: each as the scope read it, with the chunk it is handed.""" + storage_transformers: tuple[Resolved[StorageTransformerDefinition[Any]], ...] = () + """The storage transformers, each as the scope read it.""" + + def fields(self) -> Iterator[tuple[Loc, Resolved[Any]]]: + """Each field the document holds, as the scope read it, with where it sits in the document. + + The extension points, then each codec and storage transformer at its + index, each followed by the fields it holds, as `fields_of` gives + them: a shard's codecs, a struct's field types. + """ + for key, field in ( + ("data_type", self.data_type), + ("chunk_grid", self.chunk_grid), + ("chunk_key_encoding", self.chunk_key_encoding), + ): + if field is not None: + yield from fields_of(field, (key,)) + for index, stage in enumerate(self.pipeline): + yield from fields_of(stage.codec, ("codecs", index)) + for index, transformer in enumerate(self.storage_transformers): + yield from fields_of(transformer, ("storage_transformers", index)) + + +def read_array_metadata_v3( + value: object, *, context: Context = CORE_AND_EXTENSIONS +) -> tuple[ZarrV3ArrayMetadataReading, tuple[ValidationProblem, ...]]: + """`value`, a v3 array document, as `context` read it, and every reason it is not a valid one. + + The problems are `validate_array_metadata_v3`'s; the reading is what + was read to find them, so nothing need read the document again: each + extension point as `context` read it -- what claims it, or that + nothing in scope does -- the chunks the codecs are handed, and each + codec with the chunk it is handed. A policy over the fields, the core + spec's alone, say, is a walk over its `fields()`. A value that is not + a mapping holds no field, and reads as nothing. """ if not isinstance(value, Mapping): - return (ValidationProblem((), "expected an object", "invalid_type"),) + nothing = ZarrV3ArrayMetadataReading() + return nothing, (ValidationProblem((), "expected an object", "invalid_type"),) doc = cast("Mapping[object, object]", value) problems: list[ValidationProblem] = list(_missing_keys(ARRAY_METADATA_REQUIRED_KEYS_V3, doc)) problems.extend(_validate_other_members(doc, ARRAY_METADATA_STANDARD_KEYS_V3)) @@ -428,9 +469,11 @@ def validate_array_metadata_v3( # The codecs are read as a pipeline, the first handed the grid's chunks # of the array's data type: in order, each judged against the chunk it # is handed. That holds one array -> bytes codec, so it is not empty. + chunk = Chunk(lengths, read.get("data_type")) + pipeline: tuple[Stage, ...] = () if "codecs" in listed: - chunk = Chunk(lengths, read.get("data_type")) - problems.extend(read_pipeline(listed["codecs"], chunk, ("codecs",))[1]) + pipeline, found = read_pipeline(listed["codecs"], chunk, ("codecs",)) + problems.extend(found) if "attributes" in doc: problems.extend(_validate_attributes(doc["attributes"])) if "dimension_names" in doc: @@ -456,7 +499,41 @@ def validate_array_metadata_v3( "invalid_value", ) ) - return tuple(problems) + reading = ZarrV3ArrayMetadataReading( + data_type=read.get("data_type"), + chunk_grid=read.get("chunk_grid"), + chunk_key_encoding=read.get("chunk_key_encoding"), + chunk=chunk, + pipeline=pipeline, + storage_transformers=tuple(listed.get("storage_transformers", ())), + ) + return reading, tuple(problems) + + +def validate_array_metadata_v3( + value: object, *, context: Context = CORE_AND_EXTENSIONS +) -> tuple[ValidationProblem, ...]: + """Return every reason `value` is not a valid v3 array document. + + Its structure, and each extension point read through the definition + that claims its name in `context`: a gzip `level` out of range, a key a + codec's configuration does not declare. The fill value is judged + against the data type as `context` read it -- an `int8` fill value of + 300 -- and the chunk grid against the shape: a regular grid with a + chunk length for each of two dimensions, over an array of three. The + codecs are read as a pipeline: in order, each judged against the chunk + it is handed -- a `transpose` whose `order` has another number of + axes, a shard its inner chunks do not divide -- and a shard's inner + and index codecs too. + A name nothing in `context` claims is left unjudged, with any fill + value of it, and a codec of that name leaves the codec after it + handed a chunk nothing is known of. Unknown top-level keys are + allowed (they map to `extra_fields`); a reader must understand each + one that does not say `must_understand: false`, which the model + reports as `must_understand_fields`. `read_array_metadata_v3` returns + what was read to find them, beside them. + """ + return read_array_metadata_v3(value, context=context)[1] def is_array_metadata_v3( diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py b/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py index 5a9a9926fe..5b8c2a910b 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py @@ -775,6 +775,11 @@ class Resolved(Generic[D]): `must_understand` of `false` -- is reported with the field, and leaves the resolution as it is; so is a problem of a field the configuration holds, which is that field's own, with its own resolution in `nested`. + Built by hand, a reading is checked as `resolve` builds one: `read_as` + is one of the five kinds, type arguments dropped; a definition is of that + kind and filed under the name the field is written with; a field read + has a definition and a configuration, one unread no configuration, and + one nothing claims no definition. Anything else is a `TypeError`. """ json: JSONValue @@ -789,8 +794,66 @@ class Resolved(Generic[D]): A struct's field types at `("fields", 0, "data_type")`, a shard's codecs at `("codecs", 0)`: what a definition's functions consult about - the fields inside its own. Empty unless the field was read. + the fields inside its own. Each is read whatever became of the field + holding it, so a field its rules refuse keeps them too; empty when + its configuration was not checked against its TypedDict -- nothing + claims its name, it is not an object, a configuration it requires is + missing, or its name carries it, as raw bits' does. """ + read_as: type[Definition[Any]] = dataclasses.field(kw_only=True) + """The kind of metadata the field was read as -- `CodecDefinition`, `DataTypeDefinition` -- whether or not anything in scope claims it: the `kind` `resolve` was asked for. + + Named apart from a codec's `kind` -- array -> array and so on, which + its definition says -- and a problem's. + """ + + def __post_init__(self) -> None: + # The runtime half of the annotations, as `Chunk` checks its own: a + # reading built by hand, as an extension's may be, fails here rather + # than where a function trusts its kind. + kind = as_kind(self.read_as) + object.__setattr__(self, "read_as", kind) + definition = cast("object", self.definition) + if definition is not None and not isinstance(definition, kind): + msg = ( + f"a field read as a {kind.__name__} is read by one, " + f"got a {type(definition).__name__}" + ) + raise TypeError(msg) + refusal = _contradiction(self) + if refusal is not None: + raise TypeError(refusal) + + @property + def name(self) -> str | None: + """The name the field was written with -- `"r16"`, though its definition is filed under `r*` -- or None when it names none. + + What a message names it by. Which data type or codec it is, its + definition says, and its configuration: `r16` is `r*` of 16 bits. + None too for a field that is not JSON -- a `NaN` anywhere in it -- + which is held as None, and which its problems say. + """ + return named_configuration(self.json)[0] + + +def _contradiction(resolved: Resolved[Any]) -> str | None: + """What a reading says of itself that cannot hold together; None when it holds.""" + definition, configuration = resolved.definition, resolved.configuration + if (resolved.resolution == "read") != (configuration is not None): + return f"a field's configuration is held when it was read, got one {resolved.resolution!r}" + if resolved.resolution == "read" and definition is None: + return "a field read is read by a definition, got none" + if resolved.resolution == "out_of_scope": + if definition is not None: + return f"a field nothing in scope claims has no definition, got {definition.name!r}" + if resolved.name is None or len(resolved.nested) != 0: + return "a field nothing in scope claims is named, and holds no field read inside it" + name = resolved.name + if definition is not None and ( + name is None or spelled(resolved.read_as, name)[0] != definition.name + ): + return f"a field named {name!r} is read by the definition filed under it, got {definition.name!r}" + return None Nested: TypeAlias = Mapping[Loc, Resolved[Any]] @@ -835,11 +898,10 @@ def rank(self) -> int | None: def _is_data_type_field(value: object) -> bool: - """Whether `value` is a data type field a scope read: one read as a data type, or by nothing.""" - if not isinstance(value, Resolved): - return False - definition = cast("Resolved[Any]", value).definition - return definition is None or isinstance(definition, DataTypeDefinition) + """Whether `value` is a field a scope read as a data type.""" + return ( + isinstance(value, Resolved) and cast("Resolved[Any]", value).read_as is DataTypeDefinition + ) def _is_lengths(value: object) -> TypeGuard[Lengths]: @@ -857,6 +919,19 @@ def _is_lengths(value: object) -> TypeGuard[Lengths]: return True +def fields_of(resolved: Resolved[Any], loc: Loc = ()) -> Iterator[tuple[Loc, Resolved[Any]]]: + """`resolved`, a field a scope read, where it sits, then each field it holds and theirs in turn, each where it sits. + + `loc` is where `resolved` sits; a field it holds sits in its + configuration, at `(*loc, "configuration", *place)`, as `resolve` + locates its problems. What each holds is its `nested`: a field whose + configuration was not checked holds none. + """ + yield loc, resolved + for place, inner in resolved.nested.items(): + yield from fields_of(inner, (*loc, "configuration", *place)) + + def configuration_of(resolved: Resolved[Any], definition: Definition[C]) -> C | None: """The configuration `resolved` holds, typed as `definition` declares it, if `definition` read it. @@ -999,7 +1074,7 @@ def resolve( asked = as_kind(kind) refined, problems = refine_json(data, loc) if len(problems) != 0: - return Resolved(None, "invalid", None, None), problems + return Resolved(None, "invalid", None, None, read_as=asked), problems resolved, found = _resolve_field(refined, asked, context, loc) return cast("Resolved[D]", resolved), found @@ -1023,21 +1098,21 @@ def _read( ) -> tuple[Resolved[Definition[Any]], Problems]: name, given, malformed = named_configuration(data) if name is None: - return Resolved(data, "invalid", None, None), () + return Resolved(data, "invalid", None, None, read_as=kind), () definition = context.claimant(kind, name) if len(malformed) != 0: # A configuration that is not an object, which the envelope's # problems say; the name still says what claims the field. - return Resolved(data, "invalid", definition, None), () + return Resolved(data, "invalid", definition, None, read_as=kind), () if definition is None: - return Resolved(data, "out_of_scope", None, None), () + return Resolved(data, "out_of_scope", None, None, read_as=kind), () _, carried = spelled(kind, name) if carried is not None: - return _read_carried(data, definition, given, carried, loc) + return _read_carried(data, kind, definition, given, carried, loc) at = (*loc, "configuration") if given is None and definition.requires_configuration: missing = problem(at, f"{name!r} requires a configuration", "missing_key") - return Resolved(data, "invalid", definition, None), missing + return Resolved(data, "invalid", definition, None, read_as=kind), missing typed, found, nested = _checked(definition.configuration, {} if given is None else given, at) # The rules may read a field the configuration holds by its name, so # they are asked only when each one is named; any other problem with @@ -1058,8 +1133,8 @@ def _read( if configuration is not None: own.extend(ruled(definition, lambda: definition.rules(configuration, within), at)) if configuration is None or not _usable(own): - return Resolved(data, "invalid", definition, None), (*own, *inside) - return Resolved(data, "read", definition, configuration, within), (*own, *inside) + return Resolved(data, "invalid", definition, None, within, read_as=kind), (*own, *inside) + return Resolved(data, "read", definition, configuration, within, read_as=kind), (*own, *inside) def _named(field: _NestedField) -> bool: @@ -1070,6 +1145,7 @@ def _named(field: _NestedField) -> bool: def _read_carried( data: JSONValue, + kind: type[Definition[Any]], definition: Definition[Any], given: Mapping[str, object] | None, carried: Mapping[str, JSONValue], @@ -1088,8 +1164,8 @@ def _read_carried( configuration, judged = definition.judge(carried) problems = (*beside, *(ValidationProblem(loc, found.message, found.kind) for found in judged)) if configuration is None or not _usable(problems): - return Resolved(data, "invalid", definition, None), problems - return Resolved(data, "read", definition, configuration), problems + return Resolved(data, "invalid", definition, None, read_as=kind), problems + return Resolved(data, "read", definition, configuration, read_as=kind), problems def _sized(field: _NestedField, definition: Definition[Any] | None) -> Problems: diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/_pipeline.py b/packages/zarr-metadata/src/zarr_metadata/v3/_pipeline.py index a36017dc73..30772b3186 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/_pipeline.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/_pipeline.py @@ -224,10 +224,8 @@ def _held( ) -> tuple[Resolved[CodecDefinition[Any]], ...]: """The codecs `member` of the configuration holds, as the scope read them. - A member holding anything but a list of fields, or fields a definition - of another kind read, is a `TypeError`: a fault in the definition that - names it. Fields nothing in scope claims tell nothing of their kind, - and are read as codecs nothing claims. + A member holding anything but a list of fields read as codecs is a + `TypeError`: a fault in the definition that names it. """ entries: object = configuration.get(member) places = ( @@ -236,12 +234,7 @@ def _held( else None ) if places is None or not all( - place in nested - and ( - nested[place].definition is None - or isinstance(nested[place].definition, CodecDefinition) - ) - for place in places + place in nested and nested[place].read_as is CodecDefinition for place in places ): msg = f"{definition.name!r}: its pipelines name {member!r}, which holds no list of codecs" raise TypeError(msg) diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/codec/sharding_indexed.py b/packages/zarr-metadata/src/zarr_metadata/v3/codec/sharding_indexed.py index 98dd1560cf..8b856e7729 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/codec/sharding_indexed.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/codec/sharding_indexed.py @@ -14,6 +14,7 @@ Chunk, CodecDefinition, CodecField, + DataTypeDefinition, Lengths, Nested, Resolved, @@ -125,7 +126,9 @@ def _chunk_rules( ) -_UINT64: Final = Resolved(UINT64_DATA_TYPE_NAME, "read", UINT64_DATA_TYPE, {}) +_UINT64: Final = Resolved( + UINT64_DATA_TYPE_NAME, "read", UINT64_DATA_TYPE, {}, read_as=DataTypeDefinition +) """The data type of a shard index, which the spec fixes whatever the scope holds.""" diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/definition.py b/packages/zarr-metadata/src/zarr_metadata/v3/definition.py index 1e3aa4ba90..33b273e482 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/definition.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/definition.py @@ -38,14 +38,19 @@ 3. `resolve(field, CodecDefinition, CORE_AND_EXTENSIONS)` reads a whole field in a scope: its envelope judged, its name related to a definition, its configuration judged, and each nested field read the - same way. It returns `Resolved` -- the field's JSON, its - `resolution`, the definition and the checked configuration -- and - every problem. A name nothing in scope claims is `out_of_scope`: - left unjudged, which is what keeps the format open. A field is read + same way. It returns `Resolved` -- the field's JSON and the `name` + it was written with, its `resolution`, the definition, the checked + configuration and the kind it was read as, `read_as` -- and every + problem. A name nothing in scope + claims is `out_of_scope`: left unjudged, which is what keeps the + format open, and still read as the kind it was asked for. A field is read as one of the five kinds, with or without type arguments; `resolve(field, Definition, scope)` is a `TypeError`, since nothing is filed under it. `configuration_of(resolved, GZIP_CODEC)` is the configuration typed as that definition's TypedDict, when it read it. + `fields_of(resolved)` gives the field and each field it holds, with + where each sits. A whole v3 array document is read by + `read_array_metadata_v3`, in `zarr_metadata.model`. from zarr_metadata.v3.codec.gzip import GZIP_CODEC from zarr_metadata.v3.definition import ( @@ -262,6 +267,7 @@ class creation. canonicalize, chunk_grid_lengths, configuration_of, + fields_of, fill_value_problems, resolve, storage_of, @@ -306,6 +312,7 @@ class creation. "check", "chunk_grid_lengths", "configuration_of", + "fields_of", "fill_value_problems", "read_pipeline", "resolve", diff --git a/packages/zarr-metadata/tests/model/test_read_array_metadata.py b/packages/zarr-metadata/tests/model/test_read_array_metadata.py new file mode 100644 index 0000000000..91c7646377 --- /dev/null +++ b/packages/zarr-metadata/tests/model/test_read_array_metadata.py @@ -0,0 +1,261 @@ +"""A v3 array document, read: each field as a scope read it, and its codecs as a pipeline. + +`read_array_metadata_v3` returns what the validator read to find what is +wrong with a document, beside the problems it found: each field with +where it sits and the kind it was read as, the chunks the codecs are +handed, and each codec with the chunk it is handed. +""" + +from __future__ import annotations + +from typing import TYPE_CHECKING, Any, cast + +import pytest + +from zarr_metadata._json import arrays_to_tuples +from zarr_metadata.model import ( + ValidationProblem, + ZarrV3ArrayMetadata, + ZarrV3ArrayMetadataReading, + read_array_metadata_v3, + validate_array_metadata_v3, +) +from zarr_metadata.v3.definition import ( + CORE, + CORE_AND_EXTENSIONS, + ChunkGridDefinition, + ChunkKeyEncodingDefinition, + CodecDefinition, + Context, + DataTypeDefinition, + Definition, + StorageTransformerDefinition, +) + +if TYPE_CHECKING: + from zarr_metadata.v3.definition import Lengths, Loc + +LITTLE = {"name": "bytes", "configuration": {"endian": "little"}} +ZSTD = {"name": "zstd", "configuration": {"level": 1}} + +POINTS: list[tuple[Loc, type[Definition[Any]], str]] = [ + (("data_type",), DataTypeDefinition, "read"), + (("chunk_grid",), ChunkGridDefinition, "read"), + (("chunk_key_encoding",), ChunkKeyEncodingDefinition, "read"), +] +"""A default document's single extension points, each read where it sits.""" + + +def _document(shape: tuple[int, ...] = (4,), **fields: object) -> dict[str, Any]: + """A default v3 array document of `shape` with `fields` in it, arrays as tuples as a reader's are.""" + document = {**ZarrV3ArrayMetadata.create_default(shape=shape).to_json(), **fields} + return cast("dict[str, Any]", arrays_to_tuples(document)) + + +def _codec(loc: Loc, resolution: str = "read") -> tuple[Loc, type[Definition[Any]], str]: + return (loc, CodecDefinition, resolution) + + +def _shard(chunk_shape: list[int], codecs: list[object] | None = None, **more: object) -> object: + configuration = { + "chunk_shape": chunk_shape, + "codecs": [LITTLE] if codecs is None else codecs, + "index_codecs": [LITTLE], + **more, + } + return {"name": "sharding_indexed", "configuration": configuration} + + +@pytest.mark.parametrize( + ("document", "context", "fields", "handed"), + [ + (_document(), CORE_AND_EXTENSIONS, [*POINTS, _codec(("codecs", 0))], [(frozenset({4}),)]), + # Each codec is handed the chunk the one before it hands on. + ( + _document( + (4, 6), codecs=[{"name": "transpose", "configuration": {"order": [1, 0]}}, "bytes"] + ), + CORE_AND_EXTENSIONS, + [*POINTS, _codec(("codecs", 0)), _codec(("codecs", 1))], + [(frozenset({4}), frozenset({6})), (frozenset({6}), frozenset({4}))], + ), + # The fields a shard holds come after it, where they sit in its + # configuration; read in the core spec's scope, an extension inside + # it is out of scope where it sits. + ( + _document( + (16, 16), + chunk_grid={"name": "regular", "configuration": {"chunk_shape": [8, 8]}}, + codecs=[_shard([4, 4], codecs=[LITTLE, ZSTD], index_codecs=[LITTLE, "crc32c"])], + ), + CORE, + [ + *POINTS, + _codec(("codecs", 0)), + _codec(("codecs", 0, "configuration", "codecs", 0)), + _codec(("codecs", 0, "configuration", "codecs", 1), "out_of_scope"), + _codec(("codecs", 0, "configuration", "index_codecs", 0)), + _codec(("codecs", 0, "configuration", "index_codecs", 1)), + ], + [(frozenset({8}), frozenset({8}))], + ), + # A shard within a shard: each field, then the fields it holds. + ( + _document( + (16, 16), + chunk_grid={"name": "regular", "configuration": {"chunk_shape": [8, 8]}}, + codecs=[_shard([4, 4], codecs=[_shard([2, 2])])], + ), + CORE_AND_EXTENSIONS, + [ + *POINTS, + _codec(("codecs", 0)), + _codec(("codecs", 0, "configuration", "codecs", 0)), + _codec(("codecs", 0, "configuration", "codecs", 0, "configuration", "codecs", 0)), + _codec( + ("codecs", 0, "configuration", "codecs", 0, "configuration", "index_codecs", 0) + ), + _codec(("codecs", 0, "configuration", "index_codecs", 0)), + ], + [(frozenset({8}), frozenset({8}))], + ), + # A shard its check refuses keeps the fields it read inside it. + ( + _document( + (16, 16), + chunk_grid={"name": "regular", "configuration": {"chunk_shape": [8, 8]}}, + codecs=[_shard([4, 4], index_location="middle", codecs=[LITTLE, ZSTD])], + ), + CORE, + [ + *POINTS, + _codec(("codecs", 0), "invalid"), + _codec(("codecs", 0, "configuration", "codecs", 0)), + _codec(("codecs", 0, "configuration", "codecs", 1), "out_of_scope"), + _codec(("codecs", 0, "configuration", "index_codecs", 0)), + ], + [(frozenset({8}), frozenset({8}))], + ), + # A struct's field types are data types, whether or not anything + # in scope claims them. + ( + _document( + data_type={ + "name": "struct", + "configuration": { + "fields": [ + {"name": "a", "data_type": "int8"}, + {"name": "b", "data_type": "acme.t"}, + ] + }, + }, + fill_value={"a": 0, "b": "anything"}, + codecs=[LITTLE], + ), + CORE_AND_EXTENSIONS, + [ + POINTS[0], + ( + ("data_type", "configuration", "fields", 0, "data_type"), + DataTypeDefinition, + "read", + ), + ( + ("data_type", "configuration", "fields", 1, "data_type"), + DataTypeDefinition, + "out_of_scope", + ), + *POINTS[1:], + _codec(("codecs", 0)), + ], + [(frozenset({4}),)], + ), + # A field its definition refuses, or nothing claims, is read as + # the kind it sits in; a grid nothing claims says no lengths, and + # a codec after the array -> bytes codec is handed bytes. + ( + _document( + chunk_grid={"name": "acme.grid", "configuration": {}}, + codecs=["bytes", {"name": "gzip", "configuration": {"level": 99}}], + storage_transformers=[{"name": "acme.transformer"}], + ), + CORE_AND_EXTENSIONS, + [ + POINTS[0], + (("chunk_grid",), ChunkGridDefinition, "out_of_scope"), + POINTS[2], + _codec(("codecs", 0)), + _codec(("codecs", 1), "invalid"), + (("storage_transformers", 0), StorageTransformerDefinition, "out_of_scope"), + ], + [(None,), None], + ), + # A data type that is not JSON is read as nothing. + ( + _document(data_type=float("nan")), + CORE_AND_EXTENSIONS, + [(("data_type",), DataTypeDefinition, "invalid"), *POINTS[1:], _codec(("codecs", 0))], + [(frozenset({4}),)], + ), + ], + ids=[ + "default", + "transposed", + "a-shard-holding-an-extension-in-the-core-scope", + "a-shard-within-a-shard", + "a-shard-its-check-refuses", + "a-struct-s-field-types", + "fields-read-as-nothing", + "a-data-type-that-is-not-json", + ], +) +def test_a_document_reads_as_each_field_where_it_sits_and_its_codecs_as_a_pipeline( + document: dict[str, Any], + context: Context, + fields: list[tuple[Loc, type[Definition[Any]], str]], + handed: list[Lengths | None], +) -> None: + reading, problems = read_array_metadata_v3(document, context=context) + assert problems == validate_array_metadata_v3(document, context=context) + assert [(loc, field.read_as, field.resolution) for loc, field in reading.fields()] == fields + assert reading.chunk.data_type is reading.data_type + assert reading.pipeline[0].incoming == reading.chunk + assert [None if s.incoming is None else s.incoming.lengths for s in reading.pipeline] == handed + + +def test_error_a_value_that_is_not_a_mapping_reads_as_nothing() -> None: + reading, problems = read_array_metadata_v3(["not", "a", "document"]) + assert reading == ZarrV3ArrayMetadataReading() + assert list(reading.fields()) == [] + assert problems == (ValidationProblem((), "expected an object", "invalid_type"),) + + +@pytest.mark.parametrize( + ("member", "attribute"), + [("codecs", "pipeline"), ("storage_transformers", "storage_transformers")], +) +def test_error_a_list_of_fields_that_is_not_a_list_reads_as_empty( + member: str, attribute: str +) -> None: + document = _document() + document[member] = "bytes" + reading, problems = read_array_metadata_v3(document) + assert getattr(reading, attribute) == () + assert [(p.loc, p.kind) for p in problems] == [((member,), "invalid_type")] + + +@pytest.mark.parametrize("member", ["data_type", "chunk_grid", "chunk_key_encoding"]) +def test_error_a_document_without_a_field_reads_it_as_none(member: str) -> None: + document = _document() + del document[member] + reading, problems = read_array_metadata_v3(document) + assert getattr(reading, member) is None + assert (member,) not in [loc for loc, _ in reading.fields()] + assert ValidationProblem((member,), "missing required key", "missing_key") in problems + + +def test_error_a_document_without_a_shape_hands_its_codecs_chunks_of_no_known_rank() -> None: + document = _document() + del document["shape"] + reading, _ = read_array_metadata_v3(document) + assert reading.chunk.lengths is None diff --git a/packages/zarr-metadata/tests/v3/test_definitions.py b/packages/zarr-metadata/tests/v3/test_definitions.py index 26192788f3..4776b43362 100644 --- a/packages/zarr-metadata/tests/v3/test_definitions.py +++ b/packages/zarr-metadata/tests/v3/test_definitions.py @@ -19,6 +19,7 @@ from zarr_metadata.v3.codec.crc32c import Empty from zarr_metadata.v3.codec.gzip import GZIP_CODEC, GzipCodecConfiguration from zarr_metadata.v3.data_type.int8 import INT8_DATA_TYPE +from zarr_metadata.v3.data_type.raw import RAW_BYTES_DATA_TYPE from zarr_metadata.v3.definition import ( CORE, CORE_AND_EXTENSIONS, @@ -31,6 +32,8 @@ Definition, JSONValue, Nested, + Resolution, + Resolved, StorageTransformerDefinition, ValidationProblem, ZarrV3MetadataFieldJSON, @@ -221,10 +224,94 @@ def test_a_read_field_keeps_the_fields_it_read_inside() -> None: cast = {"name": "cast_value", "configuration": {"data_type": "int8"}} resolved, _ = resolve(cast, CodecDefinition, CORE_AND_EXTENSIONS) assert resolved.nested[("data_type",)].definition is INT8_DATA_TYPE - # A field holding none, or one that was not read, has nothing inside. + # A field holding none, or whose configuration was not checked, has + # nothing inside; one its check or rules refuse keeps what it read. assert resolve("int8", DataTypeDefinition, CORE)[0].nested == {} - unread = {"name": "cast_value", "configuration": {"data_type": "int8", "rounding": 1}} - assert resolve(unread, CodecDefinition, CORE_AND_EXTENSIONS)[0].nested == {} + unclaimed = {"name": "acme.cast", "configuration": {"data_type": "int8"}} + assert resolve(unclaimed, CodecDefinition, CORE_AND_EXTENSIONS)[0].nested == {} + refused = {"name": "cast_value", "configuration": {"data_type": "int8", "rounding": 1}} + resolved, _ = resolve(refused, CodecDefinition, CORE_AND_EXTENSIONS) + assert resolved.resolution == "invalid" + assert resolved.nested[("data_type",)].definition is INT8_DATA_TYPE + + +@pytest.mark.parametrize( + ("field", "kind", "name"), + [ + ("int8", DataTypeDefinition, "int8"), + ({"name": "gzip", "configuration": {"level": 1}}, CodecDefinition, "gzip"), + # Raw bits: the name written, not the one its definition is filed under. + ("r16", DataTypeDefinition, "r16"), + ({"name": "acme.codec"}, CodecDefinition, "acme.codec"), + ({"configuration": {}}, CodecDefinition, None), + (5, CodecDefinition, None), + ], + ids=["bare-name", "object", "raw-bits", "out-of-scope", "no-name", "not-a-field"], +) +def test_a_reading_says_the_name_the_field_was_written_with( + field: JSONValue, kind: type[Definition[Any]], name: str | None +) -> None: + assert resolve(field, kind, CORE_AND_EXTENSIONS)[0].name == name + + +INT8 = resolve("int8", DataTypeDefinition, CORE)[0] +"""An `int8` field as `CORE` reads it.""" + + +def test_a_reading_built_by_hand_is_of_the_kind_it_says() -> None: + # Type arguments dropped, as `resolve` drops them. + built = Resolved("int8", "read", INT8_DATA_TYPE, {}, read_as=DataTypeDefinition[Any]) + assert built.read_as is DataTypeDefinition + # A name read by the definition filed under another, as raw bits are. + raw = Resolved("r16", "read", RAW_BYTES_DATA_TYPE, {"bits": 16}, read_as=DataTypeDefinition) + assert (raw.name, raw.definition) == ("r16", RAW_BYTES_DATA_TYPE) + + +@pytest.mark.parametrize( + ("json", "resolution", "definition", "configuration", "nested"), + [ + ("int8", "read", None, {}, {}), + ("int8", "read", INT8_DATA_TYPE, None, {}), + ("int8", "invalid", INT8_DATA_TYPE, {}, {}), + ("int8", "out_of_scope", INT8_DATA_TYPE, None, {}), + (5, "out_of_scope", None, None, {}), + ("acme.t", "out_of_scope", None, None, {("a",): INT8}), + ], + ids=[ + "read-by-nothing", + "read-without-configuration", + "unread-with-one", + "claimed-out-of-scope", + "out-of-scope-without-a-name", + "out-of-scope-holding-a-field-read", + ], +) +def test_error_a_reading_built_by_hand_that_its_resolution_contradicts( + json: JSONValue, + resolution: Resolution, + definition: DataTypeDefinition[Any] | None, + configuration: dict[str, JSONValue] | None, + nested: Nested, +) -> None: + with pytest.raises(TypeError, match="a field"): + Resolved(json, resolution, definition, configuration, nested, read_as=DataTypeDefinition) + + +def test_error_a_reading_built_by_hand_by_a_definition_filed_under_another_name() -> None: + with pytest.raises( + TypeError, match="a field named 'int16' is read by the definition filed under it" + ): + Resolved("int16", "read", INT8_DATA_TYPE, {}, read_as=DataTypeDefinition) + + +def test_error_a_reading_built_by_hand_of_no_kind() -> None: + with pytest.raises(TypeError, match="is not a kind of metadata"): + Resolved("int8", "read", None, None, read_as=Definition) + + +def test_error_a_reading_built_by_hand_by_a_definition_of_another_kind() -> None: + with pytest.raises(TypeError, match="read as a DataTypeDefinition is read by one, got a Codec"): + Resolved("gzip", "read", GZIP_CODEC, {"level": 1}, read_as=DataTypeDefinition) @pytest.mark.parametrize( diff --git a/packages/zarr-metadata/tests/v3/test_fill_values.py b/packages/zarr-metadata/tests/v3/test_fill_values.py index 10be671772..c7b719338b 100644 --- a/packages/zarr-metadata/tests/v3/test_fill_values.py +++ b/packages/zarr-metadata/tests/v3/test_fill_values.py @@ -285,5 +285,5 @@ def deep(levels: int) -> dict[str, object]: def test_a_struct_read_without_its_field_types_leaves_its_fields_unjudged() -> None: # A reading built by hand, holding no field type's reading. configuration = {"fields": ({"name": "a", "data_type": "int8"},)} - struct = Resolved(STRUCT, "read", STRUCT_DATA_TYPE, configuration) + struct = Resolved(STRUCT, "read", STRUCT_DATA_TYPE, configuration, read_as=DataTypeDefinition) assert fill_value_problems(struct, {"a": 300}) == () diff --git a/packages/zarr-metadata/tests/v3/test_pipelines.py b/packages/zarr-metadata/tests/v3/test_pipelines.py index acdf015a2c..ae19e4eb50 100644 --- a/packages/zarr-metadata/tests/v3/test_pipelines.py +++ b/packages/zarr-metadata/tests/v3/test_pipelines.py @@ -367,7 +367,12 @@ def transition(configuration: Empty, nested: Nested, chunk: Chunk) -> Chunk: @pytest.mark.parametrize( ("lengths", "data_type", "match"), - [((4, 2), None, "a chunk's lengths are"), (None, "float32", "a chunk's data type is")], + [ + ((4, 2), None, "a chunk's lengths are"), + (None, "float32", "a chunk's data type is"), + # A field read as a codec, though nothing claims it. + (None, resolve("acme.t", CodecDefinition, SCOPE)[0], "a chunk's data type is"), + ], ) def test_error_a_transition_that_builds_a_chunk_of_something_else( lengths: object, data_type: object, match: str @@ -446,8 +451,8 @@ class AcmeHolderConfiguration(TypedDict, closed=True): types: tuple[DataTypeField, ...] -def _holder(pipelines: object) -> Resolved[CodecDefinition[Any]]: - """A codec holding a pipeline of codecs and a list of data types, whose pipelines are `pipelines`.""" +def _holder(pipelines: object, types: JSONValue = ("uint8",)) -> Resolved[CodecDefinition[Any]]: + """A codec holding a pipeline of codecs and a list of data types, `types`, whose pipelines are `pipelines`.""" holder = CodecDefinition( name="acme.holder", configuration=AcmeHolderConfiguration, @@ -455,7 +460,7 @@ def _holder(pipelines: object) -> Resolved[CodecDefinition[Any]]: size="dynamic", pipelines=pipelines, # pyright: ignore[reportArgumentType] ) - field = {"name": "acme.holder", "configuration": {"codecs": [LITTLE], "types": ["uint8"]}} + field = {"name": "acme.holder", "configuration": {"codecs": [LITTLE], "types": types}} return resolve(field, CodecDefinition, SCOPE.extended_with(holder))[0] @@ -465,9 +470,19 @@ def test_error_pipelines_that_give_something_else(given: object) -> None: read_pipeline([_holder(lambda configuration, nested, chunk: given)], CHUNK) -@pytest.mark.parametrize("member", ["nowhere", "types"]) -def test_error_pipelines_that_name_a_member_holding_no_codecs(member: str) -> None: - holder = _holder(lambda configuration, nested, chunk: {member: Chunk()}) +@pytest.mark.parametrize( + ("member", "types"), + [ + ("nowhere", ("uint8",)), + ("types", ("uint8",)), + # Data types nothing in scope claims are still read as data types. + ("types", ("acme.t",)), + ], +) +def test_error_pipelines_that_name_a_member_holding_no_codecs( + member: str, types: JSONValue +) -> None: + holder = _holder(lambda configuration, nested, chunk: {member: Chunk()}, types) with pytest.raises(TypeError, match=f"its pipelines name {member!r}, which holds no list"): read_pipeline([holder], CHUNK) From 4ba6beeacd827c51bd64ecb01841a3bc0f986d63 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Mon, 28 Sep 2026 19:37:08 +0200 Subject: [PATCH 04/94] refactor(zarr-metadata)!: a field reads as Read, Unclaimed or Refused, and one read builds a document's reading and model - `resolve` gives one of three frozen dataclasses, built with keywords: `Read`, by the definition in scope that claims the field's name, with its definition and configuration; `Unclaimed`, when nothing in scope claims it; or `Refused`, with the definition that refused it, if any. `Resolved` is their union, for `match`. Each answers `name`, `read_as`, `definition` and `nested`. They compare by what they read, not how they were spelled: a configuration holds each field in it as `to_json` writes it. - Models hold those fields. `ZarrV3NamedConfig`, `ZarrV3MetadataField` (the model's and the pydantic one), `ZarrV3ArrayMetadataPartial` and `ZarrV3GroupMetadataPartial` are gone. - `read_array_metadata_v3` returns one reading, holding every problem and, when there is none, the model built by the same read; `read_group_metadata_v3` does the same for a group, reading each document its consolidated metadata holds once and holding the model of each that has no problem. `from_json` is the model, or the problems raised; `validate_*` are the problems, and build no models. - A model holds no scope, as a pydantic model holds no validation context. `update(context=..., **members)` reads only the members it is given, in that scope; the fields the model holds pass through as they were read; `UNSET` leaves a member out; a group keeps the documents its consolidated metadata holds unless it is given others. The pydantic field types read in the scope their validation context holds. - `to_key_value` reads the document it writes by the model's own fields, and refuses one with a problem, so a model changed by hand is never written invalid. - A scope pickles as its definitions and copies as itself; the core definitions compare equal once pickled; `configuration_of` takes an equal definition. A read copies no document first, reads each member once, and reads a field a configuration holds a frame deeper than it. - The changelog describes what ships: #370's and #372's fragments fold into #373's, and the statements of 4434, 4436 and 4443 that this replaces are dropped from theirs. Assisted-by: ClaudeCode:claude-opus-5-5 --- packages/zarr-metadata/README.md | 21 +- packages/zarr-metadata/changes/372.feature.md | 12 - .../changes/{370.bugfix.md => 373.bugfix.md} | 12 +- .../zarr-metadata/changes/373.feature.1.md | 15 + .../zarr-metadata/changes/373.feature.2.md | 12 + packages/zarr-metadata/changes/373.feature.md | 14 + packages/zarr-metadata/changes/373.removal.md | 11 + .../zarr-metadata/changes/4434.feature.1.md | 3 +- .../zarr-metadata/changes/4434.feature.md | 8 +- .../zarr-metadata/changes/4436.feature.md | 6 +- .../zarr-metadata/changes/4443.feature.2.md | 2 +- packages/zarr-metadata/docs/index.md | 21 +- .../src/zarr_metadata/__init__.py | 12 +- .../zarr-metadata/src/zarr_metadata/_json.py | 22 + .../src/zarr_metadata/model/__init__.py | 31 +- .../src/zarr_metadata/model/_array.py | 473 ++++++++-------- .../src/zarr_metadata/model/_group.py | 516 ++++++++++++++---- .../src/zarr_metadata/model/_validation.py | 310 +++++------ .../src/zarr_metadata/pydantic.py | 78 ++- .../src/zarr_metadata/v3/_definition.py | 418 +++++++++----- .../src/zarr_metadata/v3/_pipeline.py | 23 +- .../src/zarr_metadata/v3/_registry.py | 16 +- .../src/zarr_metadata/v3/codec/_arithmetic.py | 13 +- .../src/zarr_metadata/v3/codec/bytes.py | 3 +- .../src/zarr_metadata/v3/codec/cast_value.py | 22 +- .../zarr_metadata/v3/codec/scale_offset.py | 12 +- .../v3/codec/sharding_indexed.py | 10 +- .../src/zarr_metadata/v3/data_type/_float.py | 76 +-- .../zarr_metadata/v3/data_type/_integer.py | 31 +- .../src/zarr_metadata/v3/data_type/struct.py | 5 +- .../src/zarr_metadata/v3/definition.py | 43 +- .../zarr-metadata/tests/model/test_array.py | 500 +++++++++++------ .../tests/model/test_extension_points.py | 21 +- .../zarr-metadata/tests/model/test_group.py | 253 ++++++++- .../tests/model/test_pydantic.py | 4 +- .../tests/model/test_pydantic_module.py | 59 +- .../tests/model/test_read_array_metadata.py | 85 +-- .../tests/model/test_sentinel.py | 10 +- .../tests/model/test_store_json.py | 26 +- .../zarr-metadata/tests/test_public_api.py | 15 +- .../tests/v3/test_definitions.py | 447 +++++++++++---- .../tests/v3/test_every_definition.py | 53 +- .../tests/v3/test_fill_values.py | 6 +- 43 files changed, 2422 insertions(+), 1308 deletions(-) delete mode 100644 packages/zarr-metadata/changes/372.feature.md rename packages/zarr-metadata/changes/{370.bugfix.md => 373.bugfix.md} (55%) create mode 100644 packages/zarr-metadata/changes/373.feature.1.md create mode 100644 packages/zarr-metadata/changes/373.feature.2.md create mode 100644 packages/zarr-metadata/changes/373.feature.md create mode 100644 packages/zarr-metadata/changes/373.removal.md diff --git a/packages/zarr-metadata/README.md b/packages/zarr-metadata/README.md index 75dfdbf5c8..bb677345ea 100644 --- a/packages/zarr-metadata/README.md +++ b/packages/zarr-metadata/README.md @@ -91,25 +91,32 @@ Three choices the specs' words leave open, or settle two ways: says chunk sizes are greater than zero. The package follows the core spec, which zarr-python 3.0 and 3.1 wrote for an empty dimension. -`read_array_metadata_v3` returns what the validator read, beside the -problems: each field as the scope read it, with where it sits and the -kind it was read as, and each codec with the chunk it is handed. A -consumer's own policy is a walk over the fields, with nothing read -twice. Which fields go beyond the core spec, say -- a field that names -nothing is a problem already: +`read_array_metadata_v3` reads a document once and returns everything +the read found: each field as the scope read it -- `Read` by the +definition that claims its name, `Unclaimed` when none does, or +`Refused` -- with where it sits and the kind it was read as, each codec +with the chunk it is handed, every problem, and the model when there is +none; `from_json` is that model, or the problems raised. A consumer's +own policy is a walk over the fields, with nothing read twice. Which +fields go beyond the core spec, say -- a field that names nothing is a +problem already: ```python from zarr_metadata.model import read_array_metadata_v3 from zarr_metadata.v3.definition import CORE -reading, problems = read_array_metadata_v3(raw) +reading = read_array_metadata_v3(raw) beyond_core = [ loc for loc, field in reading.fields() if field.name is not None and CORE.claimant(field.read_as, field.name) is None ] +metadata = reading.metadata # None when reading.problems is not empty ``` +`read_group_metadata_v3` reads a group the same way, and each document +its consolidated metadata holds once. + A member the spec does not define is not a field; the model's `must_understand_fields` names those a reader must understand. diff --git a/packages/zarr-metadata/changes/372.feature.md b/packages/zarr-metadata/changes/372.feature.md deleted file mode 100644 index e56d18a1b6..0000000000 --- a/packages/zarr-metadata/changes/372.feature.md +++ /dev/null @@ -1,12 +0,0 @@ -`read_array_metadata_v3` returns what the v3 array validator read, -beside the problems it found, as a `ZarrV3ArrayMetadataReading`: each -extension point as the scope read it, the chunks the codecs are handed, -and the codecs as a pipeline, each with the chunk it is handed. Its -`fields()` gives each field with where it sits in the document, and -after it the fields it holds -- a shard's codecs, a struct's field types --- as `fields_of` gives them for any field `resolve` read, so a policy -over a document's fields, the core spec's alone, say, reads nothing -twice. `Resolved` says the kind a field was read as, `read_as`, which a -field nothing in scope claims keeps too, and the `name` it was written -with, and a field its rules refuse keeps the fields it holds in -`nested`, as they were read. diff --git a/packages/zarr-metadata/changes/370.bugfix.md b/packages/zarr-metadata/changes/373.bugfix.md similarity index 55% rename from packages/zarr-metadata/changes/370.bugfix.md rename to packages/zarr-metadata/changes/373.bugfix.md index f78c3ca9d7..ed59a34206 100644 --- a/packages/zarr-metadata/changes/370.bugfix.md +++ b/packages/zarr-metadata/changes/373.bugfix.md @@ -4,10 +4,8 @@ the bare name `"crc32c"` for a field with nothing to configure. A Zarr v3.0 reader takes no short-hand name in `codecs`, and no zarr-python release reads one there or in `chunk_key_encoding`, so a document the package wrote could not be opened by them. A data type with nothing to -configure is still written by its bare name, as core data types are. -A field alone -- `ZarrV3NamedConfig.to_json`, and the pydantic -`ZarrV3MetadataField` -- keeps its shortest spelling, since which one its -readers take depends on the extension point it fills. `to_json` and -`to_key_value` write a configuration as it was read, so a field one holds --- a shard's inner codecs -- keeps the spelling the document gave it; -`canonicalize` writes those as objects too. +configure is still written by its bare name, as core data types are. A +field's own `to_json` writes it the same way, and each field its +configuration holds -- a shard's inner codecs -- is written so too, +however the document spelled it; `canonicalize` spells a field nothing +in scope claims by its kind as well. diff --git a/packages/zarr-metadata/changes/373.feature.1.md b/packages/zarr-metadata/changes/373.feature.1.md new file mode 100644 index 0000000000..1bedd89a36 --- /dev/null +++ b/packages/zarr-metadata/changes/373.feature.1.md @@ -0,0 +1,15 @@ +A field a scope reads is one of three frozen dataclasses, built with +keywords: `Read`, by the definition in scope that claims its name, +holding that definition and the configuration it allowed; `Unclaimed`, +when nothing in scope claims it, which is left unjudged; or `Refused`, +holding the definition that claims its name and refused it, if any. +`Resolved` is their union, for `match`. Each answers the kind it was +read as, `read_as`; the `name` it was written with -- `"r16"`, whose +definition is filed under `r*`; its `definition`; and the fields it +holds as the scope read them, `nested`, which a refused field keeps too. +Two fields are equal when they read the same, however each was spelled: +a configuration holds each field in it as a document writes it, and that +is what rules are handed. A codec that is not JSON takes its kind from +the definition that claims its name, as one refused for a configuration +that is JSON does, so the pipeline's order is judged with it. A scope +pickles, and the core definitions compare equal once pickled. diff --git a/packages/zarr-metadata/changes/373.feature.2.md b/packages/zarr-metadata/changes/373.feature.2.md new file mode 100644 index 0000000000..efb6e69fba --- /dev/null +++ b/packages/zarr-metadata/changes/373.feature.2.md @@ -0,0 +1,12 @@ +A v3 model holds each extension point as a scope read it, and holds no +scope, as a pydantic model holds no validation context: +`update(context=..., **members)` reads only the members it is given, as +JSON, in the scope it is given, keeps each field the model holds as it +was read, and leaves out a member given as `UNSET`; a group keeps the +documents its consolidated metadata holds unless it is given others. +`to_key_value` reads the document it writes by the model's own fields, +and refuses one with a problem, so a model changed by hand is never +written invalid. The pydantic field types read in the scope their +validation context holds: the context itself, or its +`zarr_metadata_context` item. `create_default`'s codec is `bytes` with a +little `endian`, so the default of any data type of fixed size is valid. diff --git a/packages/zarr-metadata/changes/373.feature.md b/packages/zarr-metadata/changes/373.feature.md new file mode 100644 index 0000000000..081c563e57 --- /dev/null +++ b/packages/zarr-metadata/changes/373.feature.md @@ -0,0 +1,14 @@ +`read_array_metadata_v3` reads a v3 array document once and returns +everything the read found, as a `ZarrV3ArrayMetadataReading`: each +extension point as the scope read it, the chunks the codecs are handed, +the codecs as a pipeline, each with the chunk it is handed, every +problem, and, when there is none, the document's model, built by the +same read. Its `fields()` gives each field with where it sits in the +document, and after it the fields it holds -- a shard's codecs, a +struct's field types -- as `fields_of` gives them for any field +`resolve` read, so a policy over a document's fields, the core spec's +alone, say, reads nothing twice. `read_group_metadata_v3` reads a v3 +group the same way, and each document its consolidated metadata holds +once, holding the model of each that has no problem. `from_json` is the +model, or the problems raised; `validate_*` are the problems, and build +no model. diff --git a/packages/zarr-metadata/changes/373.removal.md b/packages/zarr-metadata/changes/373.removal.md new file mode 100644 index 0000000000..3eab927630 --- /dev/null +++ b/packages/zarr-metadata/changes/373.removal.md @@ -0,0 +1,11 @@ +`ZarrV3NamedConfig`, `ZarrV3MetadataField` -- the model's and the +pydantic one -- `ZarrV3ArrayMetadataPartial` and +`ZarrV3GroupMetadataPartial` are gone: a model holds each extension point +as a scope read it, a `Read` or an `Unclaimed`, and `update` takes +members of the document as JSON, as `ZarrV3ArrayMetadataUpdate` and +`ZarrV3GroupMetadataUpdate` type them, and as `create_default` takes +them. `Resolved` is no longer a class with a `resolution`: match on its +variant, each built with keywords. `update` takes the scope it reads in, +`context`, which has no default. `to_key_value` no longer takes a +`context`: it reads the document it writes by the model's own fields, +and refuses one with a problem. diff --git a/packages/zarr-metadata/changes/4434.feature.1.md b/packages/zarr-metadata/changes/4434.feature.1.md index 3b7eec2471..d2a4316247 100644 --- a/packages/zarr-metadata/changes/4434.feature.1.md +++ b/packages/zarr-metadata/changes/4434.feature.1.md @@ -5,7 +5,6 @@ gives wrong bytes as surely as ignoring a data type gives wrong values, and the spec naming only three points is read as an oversight. A document that declares a codec or a storage transformer ignorable now has a problem at `codecs.N.must_understand` or `storage_transformers.N.must_understand`, -so the readers refuse it, and `to_key_value` refuses to write a model that -holds one. The JSON schema `zarr_metadata.pydantic` generates for an array +so the readers refuse it, and no model writes one. The JSON schema `zarr_metadata.pydantic` generates for an array document says the same. A metadata field read on its own still takes `must_understand: false`. diff --git a/packages/zarr-metadata/changes/4434.feature.md b/packages/zarr-metadata/changes/4434.feature.md index 6f1e4c89da..5768a29019 100644 --- a/packages/zarr-metadata/changes/4434.feature.md +++ b/packages/zarr-metadata/changes/4434.feature.md @@ -22,8 +22,8 @@ field's envelope judged, and then the rules; `resolve(field, CodecDefinition, scope)` reads a whole field in a scope -- its envelope judged, its name related to a definition, its configuration judged, each nested field read the same way -- and returns what the scope made of it, -`Resolved`, with every problem located. A name nothing in scope claims is -`out_of_scope` and left unjudged. A shard's pipelines, a cast's target +with every problem located. A name nothing in scope claims is left +unjudged. A shard's pipelines, a cast's target type and a struct's field types are nested fields, annotated `CodecField` and `DataTypeField`, and read in the scope their field is read in. `configuration_of(resolved, GZIP_CODEC)` is a read configuration typed as @@ -44,8 +44,8 @@ nothing of the keys it does not declare, and a member typed `canonicalize(field, kind, scope)` gives a field without problems in its simplest equivalent spelling: each nested field in its own simplest spelling, then its definition's own `canonical`, and the envelope in the -fewest words -- the bare name when nothing is configured, and no -`must_understand`. A field with any problem, an unknown key included, has +fewest words every reader takes, with no `must_understand`. A field with +any problem, an unknown key included, has none, since a simpler spelling would erase what its author wrote. What `canonical` gives is judged again, and one that does not hold is a `ValueError`, a fault in the definition. diff --git a/packages/zarr-metadata/changes/4436.feature.md b/packages/zarr-metadata/changes/4436.feature.md index 05defe18e6..3aeb6d5b0a 100644 --- a/packages/zarr-metadata/changes/4436.feature.md +++ b/packages/zarr-metadata/changes/4436.feature.md @@ -7,7 +7,5 @@ against the definition that claims its name in a scope, configuration, and a document holding one, accepted before, is refused. So do the `is_*` and `parse_*` beside it and the v3 group validators, with each array an inline `consolidated_metadata` holds, and the v3 -model classes' `from_json`, `from_key_value` and `to_key_value` take the -same `context`, so a model read in a scope is written in it. The -pydantic types read in `CORE_AND_EXTENSIONS`. A name nothing in scope -claims is left unjudged. +model classes' `from_json` and `from_key_value` take the same `context`. +A name nothing in scope claims is left unjudged. diff --git a/packages/zarr-metadata/changes/4443.feature.2.md b/packages/zarr-metadata/changes/4443.feature.2.md index d3eef59b49..004658702b 100644 --- a/packages/zarr-metadata/changes/4443.feature.2.md +++ b/packages/zarr-metadata/changes/4443.feature.2.md @@ -6,5 +6,5 @@ struct fill value missing a field are each a problem at `fill_value`, where the package accepted them before. `fill_value_problems(data_type, value)` judges a fill value against a data type field a scope read, and a field that is read keeps the fields it read inside as -`Resolved.nested`, a `Nested` mapping by location. A data type nothing +`nested`, a `Nested` mapping by location. A data type nothing in scope claims leaves its fill value unjudged. diff --git a/packages/zarr-metadata/docs/index.md b/packages/zarr-metadata/docs/index.md index f3a8009ef0..881459bd68 100644 --- a/packages/zarr-metadata/docs/index.md +++ b/packages/zarr-metadata/docs/index.md @@ -106,25 +106,32 @@ Three choices the specs' words leave open, or settle two ways: says chunk sizes are greater than zero. The package follows the core spec, which zarr-python 3.0 and 3.1 wrote for an empty dimension. -`read_array_metadata_v3` returns what the validator read, beside the -problems: each field as the scope read it, with where it sits and the -kind it was read as, and each codec with the chunk it is handed. A -consumer's own policy is a walk over the fields, with nothing read -twice. Which fields go beyond the core spec, say -- a field that names -nothing is a problem already: +`read_array_metadata_v3` reads a document once and returns everything +the read found: each field as the scope read it -- `Read` by the +definition that claims its name, `Unclaimed` when none does, or +`Refused` -- with where it sits and the kind it was read as, each codec +with the chunk it is handed, every problem, and the model when there is +none; `from_json` is that model, or the problems raised. A consumer's +own policy is a walk over the fields, with nothing read twice. Which +fields go beyond the core spec, say -- a field that names nothing is a +problem already: ```python from zarr_metadata.model import read_array_metadata_v3 from zarr_metadata.v3.definition import CORE -reading, problems = read_array_metadata_v3(raw) +reading = read_array_metadata_v3(raw) beyond_core = [ loc for loc, field in reading.fields() if field.name is not None and CORE.claimant(field.read_as, field.name) is None ] +metadata = reading.metadata # None when reading.problems is not empty ``` +`read_group_metadata_v3` reads a group the same way, and each document +its consolidated metadata holds once. + A member the spec does not define is not a field; the model's `must_understand_fields` names those a reader must understand. diff --git a/packages/zarr-metadata/src/zarr_metadata/__init__.py b/packages/zarr-metadata/src/zarr_metadata/__init__.py index 1a6b39f04d..5b7e154449 100644 --- a/packages/zarr-metadata/src/zarr_metadata/__init__.py +++ b/packages/zarr-metadata/src/zarr_metadata/__init__.py @@ -23,14 +23,12 @@ ZarrV2GroupMetadataPartial, ZarrV2GroupMetadataStoreKey, ZarrV3ArrayMetadata, - ZarrV3ArrayMetadataPartial, ZarrV3ArrayMetadataStoreKey, + ZarrV3ArrayMetadataUpdate, ZarrV3ConsolidatedMetadata, ZarrV3GroupMetadata, - ZarrV3GroupMetadataPartial, ZarrV3GroupMetadataStoreKey, - ZarrV3MetadataField, - ZarrV3NamedConfig, + ZarrV3GroupMetadataUpdate, ) from zarr_metadata.v2.array import ( ZARR_V2_ARRAY_DIMENSION_SEPARATOR, @@ -382,19 +380,17 @@ "ZarrV3ArrayMetadata", "ZarrV3ArrayMetadataJSON", "ZarrV3ArrayMetadataJSONPartial", - "ZarrV3ArrayMetadataPartial", "ZarrV3ArrayMetadataStoreKey", + "ZarrV3ArrayMetadataUpdate", "ZarrV3ConsolidatedMetadata", "ZarrV3ConsolidatedMetadataJSON", "ZarrV3ExtensionField", "ZarrV3GroupMetadata", "ZarrV3GroupMetadataJSON", "ZarrV3GroupMetadataJSONPartial", - "ZarrV3GroupMetadataPartial", "ZarrV3GroupMetadataStoreKey", - "ZarrV3MetadataField", + "ZarrV3GroupMetadataUpdate", "ZarrV3MetadataFieldJSON", - "ZarrV3NamedConfig", "ZarrV3NamedConfigJSON", "ZstdCodecMetadata", "ZstdCodecName", diff --git a/packages/zarr-metadata/src/zarr_metadata/_json.py b/packages/zarr-metadata/src/zarr_metadata/_json.py index 30b95dbad3..82b20d4f9f 100644 --- a/packages/zarr-metadata/src/zarr_metadata/_json.py +++ b/packages/zarr-metadata/src/zarr_metadata/_json.py @@ -271,6 +271,27 @@ def parse_json(value: object) -> JSONValue: return refined +def copied(value: JSONValue) -> JSONValue: + """`value` in containers of its own, sharing nothing with it: each object a new `dict`, each array a new one of its type. + + One frame for each level of nesting, as `refine_json` reads, so a + value refined is copied however deep it is. + """ + if isinstance(value, Mapping): + members: dict[str, JSONValue] = {} + for key, item in value.items(): + members[key] = copied(item) + return members + if isinstance(value, (tuple, list)): + # A loop, not a comprehension, which is a frame of its own before + # Python 3.12: one frame for each level. + entries: list[JSONValue] = [] + for item in value: + entries.append(copied(item)) # noqa: PERF401 + return entries if isinstance(value, list) else tuple(entries) + return value + + def arrays_to_tuples(obj: object) -> object: """Recursively materialize mappings and convert array-like values to tuples.""" if isinstance(obj, Sequence) and not isinstance(obj, (str, bytes, bytearray)): @@ -299,6 +320,7 @@ def arrays_to_tuples(obj: object) -> object: "ValidationProblem", "arrays_to_tuples", "choices", + "copied", "is_canonical_json", "is_json", "json_type", diff --git a/packages/zarr-metadata/src/zarr_metadata/model/__init__.py b/packages/zarr-metadata/src/zarr_metadata/model/__init__.py index 9236221b88..f387d56352 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/__init__.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/__init__.py @@ -10,9 +10,10 @@ handed. Each document concept gets a `validate_*` function returning every problem found (a tuple of `ValidationProblem`, each with a machine-readable `kind`), an `is_*` type guard, and a `parse_*` function -that narrows or raises `MetadataValidationError`; a v3 array document -also gets `read_array_metadata_v3`, which returns what it read beside -them. Model `from_json` / `from_key_value` constructors raise +that narrows or raises `MetadataValidationError`; a v3 array or group +document also gets `read_array_metadata_v3` or `read_group_metadata_v3`, +one read that returns what it read, the problems, and the model when +there are none. Model `from_json` / `from_key_value` constructors raise `MetadataValidationError` for every ingestion failure, including missing store keys and undecodable bytes, and the v3 ones take the same `context`. @@ -30,9 +31,8 @@ ZarrV2ArrayMetadata, ZarrV2ArrayMetadataPartial, ZarrV3ArrayMetadata, - ZarrV3ArrayMetadataPartial, - ZarrV3MetadataField, - ZarrV3NamedConfig, + ZarrV3ArrayMetadataUpdate, + read_array_metadata_v3, ) from zarr_metadata.model._group import ( ZarrV2ConsolidatedMetadata, @@ -40,7 +40,12 @@ ZarrV2GroupMetadataPartial, ZarrV3ConsolidatedMetadata, ZarrV3GroupMetadata, - ZarrV3GroupMetadataPartial, + ZarrV3GroupMetadataReading, + ZarrV3GroupMetadataUpdate, + is_group_metadata_v3, + parse_group_metadata_v3, + read_group_metadata_v3, + validate_group_metadata_v3, ) from zarr_metadata.model._sentinel import UNSET from zarr_metadata.model._validation import ( @@ -56,16 +61,12 @@ is_array_metadata_v2, is_array_metadata_v3, is_group_metadata_v2, - is_group_metadata_v3, parse_array_metadata_v2, parse_array_metadata_v3, parse_group_metadata_v2, - parse_group_metadata_v3, - read_array_metadata_v3, validate_array_metadata_v2, validate_array_metadata_v3, validate_group_metadata_v2, - validate_group_metadata_v3, ) # Store keys are facts about the on-disk specs, so they are defined in the @@ -132,15 +133,14 @@ "ZarrV2GroupMetadataPartial", "ZarrV2GroupMetadataStoreKey", "ZarrV3ArrayMetadata", - "ZarrV3ArrayMetadataPartial", "ZarrV3ArrayMetadataReading", "ZarrV3ArrayMetadataStoreKey", + "ZarrV3ArrayMetadataUpdate", "ZarrV3ConsolidatedMetadata", "ZarrV3GroupMetadata", - "ZarrV3GroupMetadataPartial", + "ZarrV3GroupMetadataReading", "ZarrV3GroupMetadataStoreKey", - "ZarrV3MetadataField", - "ZarrV3NamedConfig", + "ZarrV3GroupMetadataUpdate", "is_array_metadata_v2", "is_array_metadata_v3", "is_group_metadata_v2", @@ -154,6 +154,7 @@ "parse_json", "parse_metadata_field_v3", "read_array_metadata_v3", + "read_group_metadata_v3", "validate_array_metadata_v2", "validate_array_metadata_v3", "validate_group_metadata_v2", diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_array.py b/packages/zarr-metadata/src/zarr_metadata/model/_array.py index c66d565f74..f8c28b3c36 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_array.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/_array.py @@ -4,33 +4,47 @@ import copy import dataclasses -from collections.abc import Mapping +from collections.abc import Callable, Mapping from dataclasses import dataclass, field -from typing import TYPE_CHECKING, Literal, TypeAlias, cast +from typing import TYPE_CHECKING, Any, Literal, cast from typing_extensions import TypedDict, Unpack from zarr_metadata._json import ( MetadataValidationError, ValidationProblem, + copied, ) from zarr_metadata.model._sentinel import UNSET from zarr_metadata.model._validation import ( ARRAY_METADATA_STANDARD_KEYS_V3, + NO_SCOPE, + ArrayMembersV3, StoreKey, + ZarrV3ArrayMetadataReading, + dimension_lengths, dump_store_json, load_store_json, parse_array_metadata_v2, - parse_array_metadata_v3, + read_array_v3, ) from zarr_metadata.v2.array import ZARR_V2_ARRAY_METADATA_STORE_KEY from zarr_metadata.v2.attributes import ZARR_V2_ATTRIBUTES_STORE_KEY -from zarr_metadata.v3._common import parse_metadata_field_v3 -from zarr_metadata.v3._registry import CORE_AND_EXTENSIONS -from zarr_metadata.v3.array import ZARR_V3_ARRAY_METADATA_STORE_KEY +from zarr_metadata.v3._definition import ( + ChunkGridDefinition, + ChunkKeyEncodingDefinition, + CodecDefinition, + DataTypeDefinition, + Read, + StorageTransformerDefinition, + Unclaimed, + document_json, +) +from zarr_metadata.v3._registry import CORE_AND_EXTENSIONS, Context +from zarr_metadata.v3.array import ZARR_V3_ARRAY_METADATA_STORE_KEY, ZarrV3ExtensionField if TYPE_CHECKING: - from zarr_metadata._common import JSONValue, ZarrV3NamedConfigJSON + from zarr_metadata._common import JSONValue from zarr_metadata.v2.array import ( ZarrV2ArrayDimensionSeparator, ZarrV2ArrayMetadataJSON, @@ -41,91 +55,11 @@ from zarr_metadata.v2.attributes import ZarrV2AttributesStoreKey from zarr_metadata.v2.codec import ZarrV2CodecMetadata from zarr_metadata.v3._common import ZarrV3MetadataFieldJSON - from zarr_metadata.v3._registry import Context from zarr_metadata.v3.array import ( ZarrV3ArrayMetadataJSON, + ZarrV3ArrayMetadataJSONPartial, ZarrV3ArrayMetadataStoreKey, - ZarrV3ExtensionField, - ) - - -@dataclass(frozen=True, slots=True, kw_only=True) -class ZarrV3NamedConfig: - """A normalized v3 metadata field with its reader obligation. - - Bare names and missing configurations normalize to an empty configuration. - Bare names and missing `must_understand` members normalize to the spec's - implicit `True` value (https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v3/core/index.rst#L1571-L1573). - """ - - name: str - configuration: dict[str, JSONValue] - must_understand: bool = True - - def to_json(self) -> ZarrV3MetadataFieldJSON: - """The field in its shortest spelling: its bare name when it has nothing to configure and must be understood, else an object. - - Which spelling a reader takes depends on the extension point the - field fills, which a field alone does not know: zarr-python reads a - core data type only by its bare name, and a Zarr v3.0 reader takes - no bare name in `codecs` - (https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v3/core/index.rst#L585-L592). - `ZarrV3ArrayMetadata.to_json` knows, and writes each extension point - as its readers take it. The configuration is written as it was - read, a field it holds too. - """ - if not self.configuration and self.must_understand: - return self.name - return _object_json(self) - - @classmethod - def from_json(cls, data: object) -> ZarrV3NamedConfig: - """A field read from `data`, its JSON: a bare name, or an object of a `name` and, if it says them, a `configuration` and a `must_understand`. - - `MetadataValidationError` for anything else. It is read in no scope, - so nothing judges its configuration, which is held as written, and a - `must_understand` of `false` is held too, though a document's - validators refuse one at every extension point. - """ - field = parse_metadata_field_v3(data) - if isinstance(field, str): - return cls(name=field, configuration={}, must_understand=True) - # A read model shares no mutable state with what it read. - configuration = copy.deepcopy(dict(field.get("configuration", {}))) - return cls( - name=field["name"], - configuration=configuration, - must_understand=field.get("must_understand", True), - ) - - -def _object_json(field: ZarrV3NamedConfig) -> ZarrV3NamedConfigJSON: - """A field as an object: its name, its configuration if it has one, and `must_understand` only when `false`.""" - # `configuration` is ReadOnly, so it is set in the literal rather than - # assigned afterwards. The output shares no mutable state with the model. - out: ZarrV3NamedConfigJSON = ( - {"name": field.name, "configuration": copy.deepcopy(field.configuration)} - if field.configuration - else {"name": field.name} ) - if not field.must_understand: - out["must_understand"] = False - return out - - -ZarrV3MetadataField: TypeAlias = ZarrV3NamedConfig -"""The in-memory model of one field of a v3 metadata document. - -This is the role-named alias for annotation positions: model fields and -consumer signatures should say `ZarrV3MetadataField` (the logical meaning) -rather than `ZarrV3NamedConfig` (the serialized form the field currently -takes). Today every metadata field normalizes to a named configuration plus -its reader obligation, so the alias is exactly `ZarrV3NamedConfig`; if a future -spec revision adds a field form that cannot be normalized to those values, -this alias widens to a union and annotation sites do not change. Mirrors the -raw-layer split between `ZarrV3NamedConfigJSON` (shape) and -`ZarrV3MetadataFieldJSON` (field union). -""" def must_understand_subset( @@ -148,30 +82,22 @@ def must_understand_subset( } -class ZarrV3ArrayMetadataPartial(TypedDict, total=False): - """ - Partial form of the constructor-settable fields of `ZarrV3ArrayMetadata`. +class ZarrV3ArrayMetadataUpdate(TypedDict, total=False, extra_items=ZarrV3ExtensionField | UNSET): + """The members `ZarrV3ArrayMetadata.update` puts in place: each as a document writes it, or `UNSET` to leave out one a document may leave out. - Every key is optional and typed with the model's own (not serialized) - value types, so it describes valid keyword arguments to - `ZarrV3ArrayMetadata.update`. The `init=False` fields `zarr_format` and - `node_type` are intentionally excluded, since they cannot be passed to - `dataclasses.replace`. - - Drift between this type and the model's settable fields is prevented by - `tests/model/test_array.py::test_partial_keys_match_settable_model_fields`. + Those are `dimension_names`, `attributes`, `storage_transformers`, and + a member the spec does not define. """ shape: tuple[int, ...] + data_type: ZarrV3MetadataFieldJSON + chunk_grid: ZarrV3MetadataFieldJSON + chunk_key_encoding: ZarrV3MetadataFieldJSON fill_value: JSONValue - data_type: ZarrV3MetadataField - chunk_grid: ZarrV3MetadataField - codecs: tuple[ZarrV3MetadataField, ...] - chunk_key_encoding: ZarrV3MetadataField + codecs: tuple[ZarrV3MetadataFieldJSON, ...] + attributes: Mapping[str, JSONValue] | UNSET + storage_transformers: tuple[ZarrV3MetadataFieldJSON, ...] | UNSET dimension_names: tuple[str | None, ...] | UNSET - attributes: dict[str, JSONValue] - storage_transformers: tuple[ZarrV3MetadataField, ...] - extra_fields: dict[str, ZarrV3ExtensionField] @dataclass(frozen=True, slots=True, kw_only=True) @@ -179,99 +105,90 @@ class ZarrV3ArrayMetadata: """In-memory model of a v3 array metadata document. A canonical, semantically lossless representation of the `zarr.json` - content for an array. Extension points (`data_type`, `chunk_grid`, - `chunk_key_encoding`, `codecs`, `storage_transformers`) are held as - `ZarrV3MetadataField` values (currently `ZarrV3NamedConfig` name, - configuration, and obligation records). `from_json` and - `from_key_value` read each through the definition that claims its name - in a scope -- `CORE_AND_EXTENSIONS` unless a `context` is passed -- and - the model holds what they read as written; `fill_value` is held - verbatim in its JSON form. `to_json` writes each extension point as - its readers take it: the data type in its shortest spelling -- a core - data type by its bare name, as they have been written since Zarr v3.0 - -- and every other as an object, `{"name": ...}`, since a Zarr v3.0 - reader takes no bare name in `codecs`. A configuration is written as - it was read, a field it holds too. + content for an array. Each extension point -- `data_type`, + `chunk_grid`, `chunk_key_encoding`, each codec and storage transformer + -- is held as a scope read it: `Read`, holding the definition that + read it, or `Unclaimed`, an extension that scope left unjudged. + `fill_value` is held verbatim in its JSON form. + + A model holds no scope: each field keeps the definition that read it, + and a scope is asked only to read new JSON -- by `from_json`, + `create_default`, and `update`, which each take one. `to_key_value` + reads the document it writes by the model's own fields, and refuses + one with a problem, so a model changed by hand, as + `dataclasses.replace` changes one, is never written invalid. `to_json` + writes each extension point as its readers take it, as `Read.to_json` + says. A model pickles when the definitions its fields hold do: ones + whose functions are defined at a module's top level. """ zarr_format: Literal[3] = field(default=3, init=False) node_type: Literal["array"] = field(default="array", init=False) shape: tuple[int, ...] fill_value: JSONValue - data_type: ZarrV3MetadataField - chunk_grid: ZarrV3MetadataField - codecs: tuple[ZarrV3MetadataField, ...] - chunk_key_encoding: ZarrV3MetadataField + data_type: Read[DataTypeDefinition[Any]] | Unclaimed + chunk_grid: Read[ChunkGridDefinition[Any]] | Unclaimed + codecs: tuple[Read[CodecDefinition[Any]] | Unclaimed, ...] + chunk_key_encoding: Read[ChunkKeyEncodingDefinition[Any]] | Unclaimed dimension_names: tuple[str | None, ...] | UNSET attributes: dict[str, JSONValue] - storage_transformers: tuple[ZarrV3MetadataField, ...] + storage_transformers: tuple[Read[StorageTransformerDefinition[Any]] | Unclaimed, ...] extra_fields: dict[str, ZarrV3ExtensionField] @classmethod - def create_default(cls, **overrides: Unpack[ZarrV3ArrayMetadataPartial]) -> ZarrV3ArrayMetadata: - """ - Create a default (empty) v3 array metadata model, with optional overrides. - - The default is a structurally-valid scalar `uint8` array — the array - analog of `list()` returning `[]`. Any field can be overridden by keyword - (the same fields accepted by `update`). Overriding `shape` without - `chunk_grid` derives a consistent default grid: one regular chunk - covering the array (`chunk_shape` equal to `shape`, with a length of - 1 for a dimension of length 0, which every reader takes: the core - spec allows 0 there and the regular grid spec does not, + def create_default( + cls, + *, + context: Context = CORE_AND_EXTENSIONS, + **overrides: Unpack[ZarrV3ArrayMetadataJSONPartial], + ) -> ZarrV3ArrayMetadata: + """A scalar `uint8` array, or the one `overrides`, members of its document, make of it, read in `context`. + + `MetadataValidationError` when the document they make has a + problem, so members that go together are passed together: a data + type with a fill value of it, a grid with the shape it fits. The + default codec is `bytes` with a little `endian`, which takes a data + type of any fixed size. Overriding `shape` without `chunk_grid` + derives a consistent default grid: one regular chunk covering the + array (`chunk_shape` equal to `shape`, with a length of 1 for a + dimension of length 0, which every reader takes: the core spec + allows 0 there and the regular grid spec does not, https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v3/chunk-grids/regular-grid/index.rst#L40). - - The derivation is deliberately one-way. A user-supplied `chunk_grid` - is an extension point and is taken verbatim — deriving `shape` from - it would require interpreting the grid's configuration, which this - layer never does (and cannot do for unrecognized grid names). So - overriding `chunk_grid` without `shape` keeps the scalar default - `shape=()`, which a grid of another rank does not fit: consistency - between the two is the caller's responsibility, so pass them - together. So is a fill value for an overridden `data_type`: - the default `fill_value` is `0`, which a data type whose fill value - is not an integer -- `bool`, `string`, a complex or struct type -- - refuses, so pass the two together; and so are its codecs: the - default `bytes` codec has no `endian`, which a data type whose - values take several bytes needs. """ - if "shape" in overrides and "chunk_grid" not in overrides: - chunk_shape = tuple(max(length, 1) for length in overrides["shape"]) - overrides["chunk_grid"] = ZarrV3NamedConfig( - name="regular", configuration={"chunk_shape": chunk_shape} - ) - default = cls( - shape=(), - fill_value=0, - data_type=ZarrV3NamedConfig(name="uint8", configuration={}), - chunk_grid=ZarrV3NamedConfig(name="regular", configuration={"chunk_shape": ()}), - codecs=(ZarrV3NamedConfig(name="bytes", configuration={}),), - chunk_key_encoding=ZarrV3NamedConfig(name="default", configuration={}), - dimension_names=UNSET, - attributes={}, - storage_transformers=(), - extra_fields={}, - ) - return default.update(**overrides) - - def update(self, **kwargs: Unpack[ZarrV3ArrayMetadataPartial]) -> ZarrV3ArrayMetadata: - """ - Return a new `ZarrV3ArrayMetadata` with the given fields updated. - - Only the constructor-settable fields listed in - `ZarrV3ArrayMetadataPartial` can be updated; any attempt to update - other fields (including the fixed `zarr_format` / `node_type`) is - rejected at the type level. Each given field fully replaces its - previous value, including `extra_fields`. + # The grid derives from a shape the read takes; one it refuses is + # reported by the read, and derives nothing. + lengths, _ = dimension_lengths(cast("Mapping[object, object]", overrides), "shape") + document: dict[str, object] = { + "zarr_format": 3, + "node_type": "array", + "shape": (), + "fill_value": 0, + "data_type": "uint8", + "chunk_grid": { + "name": "regular", + "configuration": {"chunk_shape": tuple(max(length, 1) for length in lengths or ())}, + }, + "codecs": ({"name": "bytes", "configuration": {"endian": "little"}},), + "chunk_key_encoding": {"name": "default"}, + } + return cls.from_json({**document, **overrides}, context=context) - This is useful for test fixtures that want to override a few fields of a - base template without having to re-specify the entire document. + def update( + self, *, context: Context, **members: Unpack[ZarrV3ArrayMetadataUpdate] + ) -> ZarrV3ArrayMetadata: + """This model with `members` in their place, each read in `context`; `UNSET` leaves an optional member out. - No re-validation is performed (`update` is `dataclasses.replace`), so - a repair or edit can produce an invalid document; validity is checked - on `from_json`, not on field replacement. + Only the members given are read in `context`: each field the model + holds is kept as it was read, whatever scope read it. The document + they make is then read as a whole, so `MetadataValidationError` + when it has a problem, and members that go together are passed + together: a `shape` with a grid that fits it. """ - return dataclasses.replace(self, **kwargs) + document = {**held_document(self), **members} + for key, value in members.items(): + if value is UNSET: + del document[key] + return type(self).from_json(document, context=context) def __post_init__(self) -> None: overlap = set(self.extra_fields.keys()).intersection(ARRAY_METADATA_STANDARD_KEYS_V3) @@ -285,43 +202,28 @@ def __post_init__(self) -> None: ) ] ) + # The runtime half of the annotations: each extension point a field + # read as its kind, read or unclaimed, as a read gives it. + for key, kind, nodes in ( + ("data_type", DataTypeDefinition, (self.data_type,)), + ("chunk_grid", ChunkGridDefinition, (self.chunk_grid,)), + ("chunk_key_encoding", ChunkKeyEncodingDefinition, (self.chunk_key_encoding,)), + ("codecs", CodecDefinition, self.codecs), + ("storage_transformers", StorageTransformerDefinition, self.storage_transformers), + ): + for node in cast("tuple[object, ...]", nodes): + if not isinstance(node, (Read, Unclaimed)) or node.read_as is not kind: + msg = f"{key}: expected a field read as a {kind.__name__}, got {node!r}" + raise TypeError(msg) def to_json(self) -> ZarrV3ArrayMetadataJSON: """The document as JSON, arrays as tuples, sharing no mutable state with the model. - Each extension point as its readers take it: the data type as - `ZarrV3NamedConfig.to_json` writes it, and every other as an object; - `dimension_names` when set, and `attributes` and `storage_transformers` - when not empty. Not validated: `to_key_value` is the writer that - refuses a model that is not valid. + Each extension point as its readers take it, as `Read.to_json` + writes one; `dimension_names` when set, and `attributes` and + `storage_transformers` when not empty. """ - # to_json output shares no mutable state with the model: every value - # that can hold a mutable container is deep-copied. - out: ZarrV3ArrayMetadataJSON = { - "zarr_format": self.zarr_format, - "node_type": self.node_type, - "shape": self.shape, - "fill_value": copy.deepcopy(self.fill_value), - "data_type": self.data_type.to_json(), - "chunk_grid": _object_json(self.chunk_grid), - "codecs": tuple(_object_json(codec) for codec in self.codecs), - "chunk_key_encoding": _object_json(self.chunk_key_encoding), - } - if self.dimension_names is not UNSET: - out["dimension_names"] = self.dimension_names - if len(self.attributes) > 0: - out["attributes"] = copy.deepcopy(self.attributes) - if len(self.storage_transformers) > 0: - out["storage_transformers"] = tuple( - _object_json(transformer) for transformer in self.storage_transformers - ) - # Extra fields are the TypedDict's `extra_items` (PEP 728). Assign them - # by key rather than `out.update(**...)`: type checkers understand the - # indexed-write path against `extra_items`, but not the `update(**...)` - # overload. - for key, value in self.extra_fields.items(): - out[key] = copy.deepcopy(value) - return out + return cast("ZarrV3ArrayMetadataJSON", copied(cast("JSONValue", array_json(self)))) @classmethod def from_json( @@ -329,29 +231,14 @@ def from_json( ) -> ZarrV3ArrayMetadata: """The model of `data`, a v3 array document read in `context`. - `MetadataValidationError` with every problem `validate_array_metadata_v3` - finds. A member the spec does not define is held in `extra_fields`. The - model shares no mutable state with `data`. + `MetadataValidationError` with every problem the read finds. + `read_array_metadata_v3` gives the reading this model is built + from, and the problems of a document with some. """ - # A read model shares no mutable state with what it read. - parsed = copy.deepcopy(parse_array_metadata_v3(data, context=context)) - extra_fields: dict[str, ZarrV3ExtensionField] = { - k: v for k, v in parsed.items() if k not in ARRAY_METADATA_STANDARD_KEYS_V3 - } - return cls( - shape=parsed["shape"], - fill_value=parsed["fill_value"], - data_type=ZarrV3NamedConfig.from_json(parsed["data_type"]), - chunk_grid=ZarrV3NamedConfig.from_json(parsed["chunk_grid"]), - codecs=tuple(ZarrV3NamedConfig.from_json(c) for c in parsed["codecs"]), - chunk_key_encoding=ZarrV3NamedConfig.from_json(parsed["chunk_key_encoding"]), - dimension_names=parsed.get("dimension_names", UNSET), - attributes=dict(parsed.get("attributes", {})), - storage_transformers=tuple( - ZarrV3NamedConfig.from_json(t) for t in parsed.get("storage_transformers", ()) - ), - extra_fields=extra_fields, - ) + reading = read_array_metadata_v3(data, context=context) + if reading.metadata is None: + raise MetadataValidationError(reading.problems) + return reading.metadata @property def must_understand_fields(self) -> dict[str, ZarrV3ExtensionField]: @@ -379,21 +266,111 @@ def from_key_value( ) def to_key_value( - self, *, indent: int | str | None = None, context: Context = CORE_AND_EXTENSIONS + self, *, indent: int | str | None = None ) -> Mapping[ZarrV3ArrayMetadataStoreKey, bytes]: """The document as a store holds it: JSON bytes at `zarr.json`, indented by `indent`. - Validated first, in `context`: a model that is not valid raises - `MetadataValidationError` with every problem, and nothing is written. - `NaN`, `Infinity` and `-Infinity` in `attributes` are written as those - bare tokens, as zarr-python writes them, which a strict JSON parser - refuses. + `MetadataValidationError` when the document has a problem, as the + model's own fields read it, so no invalid document is written, even + of a model changed by hand. `NaN`, `Infinity` and `-Infinity` in + `attributes` are written as those bare tokens, as zarr-python + writes them, which a strict JSON parser refuses. """ - # A model built by hand is not validated: its document is written only - # if it reads as `from_json` reads one in `context`, and every problem - # is raised. - document = parse_array_metadata_v3(self.to_json(), context=context) - return {ZARR_V3_ARRAY_METADATA_STORE_KEY: dump_store_json(document, indent=indent)} + problems = array_problems(self) + if len(problems) != 0: + raise MetadataValidationError(problems) + return {ZARR_V3_ARRAY_METADATA_STORE_KEY: dump_store_json(array_json(self), indent=indent)} + + +def array_problems(model: ZarrV3ArrayMetadata) -> tuple[ValidationProblem, ...]: + """What is wrong with `model`'s document, as the model's own fields read it: nothing, for a model a read built.""" + return read_array_v3(held_document(model), NO_SCOPE)[0].problems + + +def array_json(model: ZarrV3ArrayMetadata) -> ZarrV3ArrayMetadataJSON: + """`model`'s document as JSON, holding the model's own values: what `to_key_value` serializes, which changes nothing, and `to_json` copies.""" + return cast("ZarrV3ArrayMetadataJSON", _document(model, document_json)) + + +def held_document(model: ZarrV3ArrayMetadata) -> dict[str, object]: + """`model`'s document with each field as it was read, which a read takes as it is: what `update` and `to_key_value` read, reading no field again.""" + return _document(model, _as_read) + + +def _as_read(field: Read[Any] | Unclaimed) -> object: + return field + + +def _document( + model: ZarrV3ArrayMetadata, write: Callable[[Read[Any] | Unclaimed], object] +) -> dict[str, object]: + """`model`'s document, each field as `write` gives it, the rest as the model holds it.""" + out: dict[str, object] = { + "zarr_format": model.zarr_format, + "node_type": model.node_type, + "shape": model.shape, + "fill_value": model.fill_value, + "data_type": write(model.data_type), + "chunk_grid": write(model.chunk_grid), + "codecs": tuple(write(codec) for codec in model.codecs), + "chunk_key_encoding": write(model.chunk_key_encoding), + } + if model.dimension_names is not UNSET: + out["dimension_names"] = model.dimension_names + if len(model.attributes) > 0: + out["attributes"] = model.attributes + if len(model.storage_transformers) > 0: + out["storage_transformers"] = tuple( + write(transformer) for transformer in model.storage_transformers + ) + out.update(model.extra_fields) + return out + + +def read_array_metadata_v3( + value: object, *, context: Context = CORE_AND_EXTENSIONS +) -> ZarrV3ArrayMetadataReading: + """`value`, a v3 array document, as `context` read it, whatever it holds. + + Everything a read finds, in one: each extension point as `context` + read it -- `Read` by the definition that claims its name, `Unclaimed`, + or `Refused` -- the chunks the codecs are handed, each codec with the + chunk it is handed, every problem `validate_array_metadata_v3` finds, + and, when there is none, the document's model, holding the same + fields. A policy over the fields, the core spec's alone, say, is a + walk over its `fields()`. A value that is not an object holds no + field. + """ + reading, members = read_array_v3(value, context) + if members is None: + return reading + return dataclasses.replace(reading, metadata=array_model(reading, members)) + + +def array_model( + reading: ZarrV3ArrayMetadataReading, members: ArrayMembersV3 +) -> ZarrV3ArrayMetadata: + """The model of a document its reading found nothing wrong with: its fields as read, and its other members as the read refined them.""" + return ZarrV3ArrayMetadata( + shape=members.shape, + fill_value=members.fill_value, + data_type=cast("Read[DataTypeDefinition[Any]] | Unclaimed", reading.data_type), + chunk_grid=cast("Read[ChunkGridDefinition[Any]] | Unclaimed", reading.chunk_grid), + codecs=tuple( + cast("Read[CodecDefinition[Any]] | Unclaimed", stage.codec) + for stage in reading.pipeline + ), + chunk_key_encoding=cast( + "Read[ChunkKeyEncodingDefinition[Any]] | Unclaimed", reading.chunk_key_encoding + ), + dimension_names=members.dimension_names, + attributes=members.attributes, + storage_transformers=cast( + "tuple[Read[StorageTransformerDefinition[Any]] | Unclaimed, ...]", + reading.storage_transformers, + ), + extra_fields=members.extra_fields, + ) class ZarrV2ArrayMetadataPartial(TypedDict, total=False): diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_group.py b/packages/zarr-metadata/src/zarr_metadata/model/_group.py index eb9d57b84a..3b14ed9d6f 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_group.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/_group.py @@ -4,9 +4,9 @@ import copy import dataclasses -from collections.abc import Mapping +from collections.abc import Callable, Mapping from dataclasses import dataclass, field -from typing import TYPE_CHECKING, Literal, cast +from typing import TYPE_CHECKING, Any, Final, Literal, TypeGuard, cast from typing_extensions import TypedDict, Unpack @@ -14,60 +14,69 @@ MetadataValidationError, ValidationProblem, arrays_to_tuples, + copied, + is_canonical_json, refine_json, refine_user_data, refused_kind, shown, ) +from zarr_metadata._json import prefixed as _prefix from zarr_metadata.model._array import ( ZarrV3ArrayMetadata, + array_json, + array_model, + held_document, must_understand_subset, ) from zarr_metadata.model._sentinel import UNSET from zarr_metadata.model._validation import ( + GROUP_METADATA_REQUIRED_KEYS_V3, GROUP_METADATA_STANDARD_KEYS_V3, + NO_SCOPE, + ArrayMembersV3, StoreKey, + ZarrV3ArrayMetadataReading, + attributes_of, + check_literal, dump_store_json, load_store_json, + missing_keys, + other_members, parse_group_metadata_v2, - parse_group_metadata_v3, - validate_consolidated_metadata_v3, + read_array_v3, + unexpected_keys, ) from zarr_metadata.v2.attributes import ZARR_V2_ATTRIBUTES_STORE_KEY from zarr_metadata.v2.consolidated import ZARR_V2_CONSOLIDATED_METADATA_STORE_KEY from zarr_metadata.v2.group import ZARR_V2_GROUP_METADATA_STORE_KEY -from zarr_metadata.v3._registry import CORE_AND_EXTENSIONS +from zarr_metadata.v3._registry import CORE_AND_EXTENSIONS, Context +from zarr_metadata.v3.array import ZarrV3ExtensionField from zarr_metadata.v3.consolidated import ZARR_V3_CONSOLIDATED_METADATA_KEY -from zarr_metadata.v3.group import ZARR_V3_GROUP_METADATA_STORE_KEY +from zarr_metadata.v3.group import ZARR_V3_GROUP_METADATA_STORE_KEY, ZarrV3GroupMetadataJSON if TYPE_CHECKING: + from collections.abc import Iterator + from zarr_metadata._common import JSONValue + from zarr_metadata._typed_json import Loc from zarr_metadata.v2.attributes import ZarrV2AttributesStoreKey from zarr_metadata.v2.consolidated import ZarrV2ConsolidatedMetadataStoreKey from zarr_metadata.v2.group import ZarrV2GroupMetadataJSON, ZarrV2GroupMetadataStoreKey - from zarr_metadata.v3._registry import Context - from zarr_metadata.v3.array import ZarrV3ExtensionField + from zarr_metadata.v3._definition import Resolved from zarr_metadata.v3.consolidated import ZarrV3ConsolidatedMetadataJSON - from zarr_metadata.v3.group import ZarrV3GroupMetadataJSON, ZarrV3GroupMetadataStoreKey - + from zarr_metadata.v3.group import ZarrV3GroupMetadataJSONPartial, ZarrV3GroupMetadataStoreKey -class ZarrV3GroupMetadataPartial(TypedDict, total=False): - """ - Partial form of the constructor-settable fields of `ZarrV3GroupMetadata`. - Every key is optional and typed with the model's own value types, so it - describes valid keyword arguments to `ZarrV3GroupMetadata.update` and - `create_default`. The `init=False` fields `zarr_format` and `node_type` - are intentionally excluded, since they cannot be passed to - `dataclasses.replace`. +class ZarrV3GroupMetadataUpdate(TypedDict, total=False, extra_items=ZarrV3ExtensionField | UNSET): + """The members `ZarrV3GroupMetadata.update` puts in place: each as a document writes it, or `UNSET` to leave it out. - Drift between this type and the model's settable fields is prevented by - `tests/model/test_group.py::test_group_partial_keys_match_settable_model_fields`. + `consolidated_metadata` is a member the spec does not define, so it is + one of the extra items: given, its documents are read; left out, the + group keeps the models it holds. """ - attributes: dict[str, JSONValue] - consolidated_metadata: ZarrV3ConsolidatedMetadata | UNSET - extra_fields: dict[str, ZarrV3ExtensionField] + attributes: Mapping[str, JSONValue] | UNSET @dataclass(frozen=True, slots=True, kw_only=True) @@ -76,9 +85,13 @@ class ZarrV3GroupMetadata: A canonical, semantically lossless representation of the `zarr.json` content for a group. The `consolidated_metadata` reference-implementation - convention is modeled as a typed field holding thin child models, each - array read in the scope the group is read in; every other unknown - top-level key lands in `extra_fields` verbatim. + convention is modeled as a typed field holding the model of each + document it holds; every other unknown top-level key lands in + `extra_fields` verbatim. A model holds no scope, as + `ZarrV3ArrayMetadata` holds none: `from_json`, `create_default` and + `update` each take the one they read new JSON in. `to_key_value` reads + the document it writes, and each document its consolidated metadata + holds, by the models' own fields, and refuses one with a problem. """ zarr_format: Literal[3] = field(default=3, init=False) @@ -99,51 +112,49 @@ def __post_init__(self) -> None: ) ] ) + # The runtime half of the annotations. + consolidated = cast("object", self.consolidated_metadata) + if consolidated is not UNSET and not isinstance(consolidated, ZarrV3ConsolidatedMetadata): + msg = f"consolidated_metadata: expected ZarrV3ConsolidatedMetadata or UNSET, got {consolidated!r}" + raise TypeError(msg) @classmethod - def create_default(cls, **overrides: Unpack[ZarrV3GroupMetadataPartial]) -> ZarrV3GroupMetadata: - """ - Create a default (empty) v3 group metadata model, with optional overrides. - - The default is a structurally-valid group with no attributes — the group - analog of `list()` returning `[]`. Any field can be overridden by keyword - (the same fields accepted by `update`). - """ - default = cls(attributes={}, consolidated_metadata=UNSET, extra_fields={}) - return default.update(**overrides) + def create_default( + cls, + *, + context: Context = CORE_AND_EXTENSIONS, + **members: Unpack[ZarrV3GroupMetadataJSONPartial], + ) -> ZarrV3GroupMetadata: + """A group with no attributes, or the one `members` of its document make of it, read in `context`; `MetadataValidationError` when its document has a problem.""" + return cls.from_json({"zarr_format": 3, "node_type": "group", **members}, context=context) - def update(self, **kwargs: Unpack[ZarrV3GroupMetadataPartial]) -> ZarrV3GroupMetadata: - """ - Return a new `ZarrV3GroupMetadata` with the given fields updated. + def update( + self, *, context: Context, **members: Unpack[ZarrV3GroupMetadataUpdate] + ) -> ZarrV3GroupMetadata: + """This model with `members` in their place, each read in `context`; `UNSET` leaves one out. - Only the constructor-settable fields listed in - `ZarrV3GroupMetadataPartial` can be updated; the fixed `zarr_format` / - `node_type` are rejected at the type level. Each given field fully - replaces its previous value, including `extra_fields`. + Only the members given are read. A `consolidated_metadata` given + has each document it holds read in `context`; left out, the group + keeps the models it holds, however each was read, and reads none + of them again. `MetadataValidationError` when the document the + members make has a problem. """ - return dataclasses.replace(self, **kwargs) + document = {**_own_document(self), **members} + for key, value in members.items(): + if value is UNSET: + del document[key] + updated = type(self).from_json(document, context=context) + if ZARR_V3_CONSOLIDATED_METADATA_KEY in members: + return updated + return dataclasses.replace(updated, consolidated_metadata=self.consolidated_metadata) def to_json(self) -> ZarrV3GroupMetadataJSON: """The document as JSON, sharing no mutable state with the model. `attributes` when not empty, `consolidated_metadata` when set, and - each extra field as held. Not validated: `to_key_value` is the writer - that refuses a model that is not valid. + each extra field as held. """ - # to_json output shares no mutable state with the model: every value - # that can hold a mutable container is deep-copied. - out: ZarrV3GroupMetadataJSON = { - "zarr_format": self.zarr_format, - "node_type": self.node_type, - } - if len(self.attributes) > 0: - out["attributes"] = copy.deepcopy(self.attributes) - if self.consolidated_metadata is not UNSET: - # Consolidated metadata is a known non-core top-level JSON field. - out[ZARR_V3_CONSOLIDATED_METADATA_KEY] = self.consolidated_metadata.to_json() - for key, value in self.extra_fields.items(): - out[key] = copy.deepcopy(value) - return out + return cast("ZarrV3GroupMetadataJSON", copied(cast("JSONValue", group_json(self)))) @classmethod def from_json( @@ -151,32 +162,14 @@ def from_json( ) -> ZarrV3GroupMetadata: """The model of `data`, a v3 group document read in `context`, with each document its consolidated metadata holds. - `MetadataValidationError` with every problem `validate_group_metadata_v3` - finds. A `consolidated_metadata` of `null` is read as none, and not - written back. A member the spec does not define is held in - `extra_fields`. + `MetadataValidationError` with every problem the read finds. A + `consolidated_metadata` of `null` is read as none, and not written + back. A member the spec does not define is held in `extra_fields`. """ - # A read model shares no mutable state with what it read. - parsed = copy.deepcopy(parse_group_metadata_v3(data, context=context)) - consolidated_raw: object = parsed.get(ZARR_V3_CONSOLIDATED_METADATA_KEY, UNSET) - consolidated: ZarrV3ConsolidatedMetadata | UNSET - if consolidated_raw is UNSET or consolidated_raw is None: - # consolidated_metadata: null was written by a historical - # zarr-python bug; it gets no model representation. It is read as - # absence and never written back — repaired, not preserved. - consolidated = UNSET - else: - consolidated = ZarrV3ConsolidatedMetadata.from_json(consolidated_raw, context=context) - extra_fields: dict[str, ZarrV3ExtensionField] = { - k: v - for k, v in parsed.items() - if k not in GROUP_METADATA_STANDARD_KEYS_V3 and k != ZARR_V3_CONSOLIDATED_METADATA_KEY - } - return cls( - attributes=dict(parsed.get("attributes", {})), - consolidated_metadata=consolidated, - extra_fields=extra_fields, - ) + reading = read_group_metadata_v3(data, context=context) + if reading.metadata is None: + raise MetadataValidationError(reading.problems) + return reading.metadata @property def must_understand_fields(self) -> dict[str, ZarrV3ExtensionField]: @@ -204,21 +197,20 @@ def from_key_value( ) def to_key_value( - self, *, indent: int | str | None = None, context: Context = CORE_AND_EXTENSIONS + self, *, indent: int | str | None = None ) -> Mapping[ZarrV3GroupMetadataStoreKey, bytes]: """The document as a store holds it: JSON bytes at `zarr.json`, indented by `indent`. - Validated first, in `context`: a model that is not valid raises - `MetadataValidationError` with every problem, and nothing is written. - `NaN`, `Infinity` and `-Infinity` in `attributes` are written as those - bare tokens, as zarr-python writes them, which a strict JSON parser - refuses. + `MetadataValidationError` when the document, or one its + consolidated metadata holds, has a problem, as the models' own + fields read it. `NaN`, `Infinity` and `-Infinity` in `attributes` + are written as those bare tokens, as zarr-python writes them, + which a strict JSON parser refuses. """ - # A model built by hand is not validated: its document is written only - # if it reads as `from_json` reads one in `context`, and every problem - # is raised. - document = parse_group_metadata_v3(self.to_json(), context=context) - return {ZARR_V3_GROUP_METADATA_STORE_KEY: dump_store_json(document, indent=indent)} + problems = read_group_v3(_held_group(self), NO_SCOPE)[0].problems + if len(problems) != 0: + raise MetadataValidationError(problems) + return {ZARR_V3_GROUP_METADATA_STORE_KEY: dump_store_json(group_json(self), indent=indent)} @dataclass(frozen=True, slots=True, kw_only=True) @@ -227,9 +219,9 @@ class ZarrV3ConsolidatedMetadata: Models the reference-implementation convention where consolidated metadata is embedded as an extension field on a group's `zarr.json`. Each entry in - `metadata` is a complete child document, held as a thin array or group - model. `must_understand` is typed permissively as `bool` to mirror the - document shape, but only `False` is valid; this is enforced at runtime. + `metadata` is the model of a complete child document, array or group. + `must_understand` is typed permissively as `bool` to mirror the document + shape, but only `False` is valid; this is enforced at runtime. """ kind: Literal["inline"] = field(default="inline", init=False) @@ -248,38 +240,328 @@ def __post_init__(self) -> None: ) ] ) + # The runtime half of the annotations. + for path, node in cast("dict[object, object]", self.metadata).items(): + if not isinstance(node, (ZarrV3ArrayMetadata, ZarrV3GroupMetadata)): + msg = f"metadata[{path!r}]: expected a v3 array or group model, got {node!r}" + raise TypeError(msg) def to_json(self) -> ZarrV3ConsolidatedMetadataJSON: - """The `consolidated_metadata` member as JSON: its `kind`, `must_understand: false`, and each document by its path.""" - # `must_understand` is emitted as the literal False: the field is typed - # permissively as `bool`, but `__post_init__` guarantees the value. - return { - "kind": self.kind, - "must_understand": False, - "metadata": {key: node.to_json() for key, node in self.metadata.items()}, - } + """The `consolidated_metadata` member as JSON, sharing no mutable state with the model: its `kind`, `must_understand: false`, and each document by its path.""" + return cast( + "ZarrV3ConsolidatedMetadataJSON", copied(cast("JSONValue", consolidated_json(self))) + ) @classmethod def from_json( cls, data: object, *, context: Context = CORE_AND_EXTENSIONS ) -> ZarrV3ConsolidatedMetadata: - """The model of `data`, a group's `consolidated_metadata` member, each document read in `context` as the array or group its `node_type` says. + """The model of `data`, a group's `consolidated_metadata` member, each document read once in `context`, as the array or group its `node_type` says. `MetadataValidationError` with every problem found. """ - normalized = arrays_to_tuples(data) - problems = validate_consolidated_metadata_v3(normalized, context=context) + readings, members, problems = _read_consolidated_v3(data, context) if len(problems) != 0: raise MetadataValidationError(problems) - env = cast("Mapping[str, object]", normalized) - entries: dict[str, ZarrV3ArrayMetadata | ZarrV3GroupMetadata] = {} - for key, entry in cast("Mapping[str, object]", env["metadata"]).items(): - node_type = cast("Mapping[str, object]", entry).get("node_type") + return cls(metadata=_models(readings, members)[1]) + + +def group_json(model: ZarrV3GroupMetadata) -> ZarrV3GroupMetadataJSON: + """`model`'s document as JSON, holding the model's own values: what `to_key_value` serializes, which changes nothing, and `to_json` copies.""" + return cast("ZarrV3GroupMetadataJSON", _group_document(model, array_json, group_json)) + + +def consolidated_json(model: ZarrV3ConsolidatedMetadata) -> ZarrV3ConsolidatedMetadataJSON: + """`model`, a `consolidated_metadata` member, as JSON, holding the model's own values, as `group_json` holds them.""" + return cast( + "ZarrV3ConsolidatedMetadataJSON", _consolidated_document(model, array_json, group_json) + ) + + +def _held_group(model: ZarrV3GroupMetadata) -> dict[str, object]: + """`model`'s document with each field of each document it holds as it was read, which a read takes as it is, as `held_document` gives an array's.""" + return _group_document(model, held_document, _held_group) + + +def _own_document(model: ZarrV3GroupMetadata) -> dict[str, object]: + """`model`'s document without its consolidated metadata: the members that are the group's own.""" + out: dict[str, object] = {"zarr_format": model.zarr_format, "node_type": model.node_type} + if len(model.attributes) > 0: + out["attributes"] = model.attributes + out.update(model.extra_fields) + return out + + +def _group_document( + model: ZarrV3GroupMetadata, + array: Callable[[ZarrV3ArrayMetadata], object], + group: Callable[[ZarrV3GroupMetadata], object], +) -> dict[str, object]: + """`model`'s document, each document its consolidated metadata holds as `array` or `group` gives it.""" + out = _own_document(model) + if model.consolidated_metadata is not UNSET: + # Consolidated metadata is a known non-core top-level JSON field. + out[ZARR_V3_CONSOLIDATED_METADATA_KEY] = _consolidated_document( + model.consolidated_metadata, array, group + ) + return out + + +def _consolidated_document( + model: ZarrV3ConsolidatedMetadata, + array: Callable[[ZarrV3ArrayMetadata], object], + group: Callable[[ZarrV3GroupMetadata], object], +) -> dict[str, object]: + """The member, each document it holds as `array` or `group` gives it.""" + # `must_understand` is emitted as the literal False: the field is typed + # permissively as `bool`, but `__post_init__` guarantees the value. + return { + "kind": model.kind, + "must_understand": False, + "metadata": { + path: array(node) if isinstance(node, ZarrV3ArrayMetadata) else group(node) + for path, node in model.metadata.items() + }, + } + + +def _no_documents() -> Mapping[str, ZarrV3ArrayMetadataReading | ZarrV3GroupMetadataReading]: + """What a group whose consolidated metadata holds none, or that has none, holds: nothing.""" + return {} + + +@dataclass(frozen=True, slots=True) +class ZarrV3GroupMetadataReading: + """A v3 group document as a scope read it, whatever it holds: each document its consolidated metadata holds, as read, every problem, and the model when there is none.""" + + consolidated: Mapping[str, ZarrV3ArrayMetadataReading | ZarrV3GroupMetadataReading] = ( + dataclasses.field(default_factory=_no_documents) + ) + """Each document its consolidated metadata holds, as read, by its path.""" + problems: tuple[ValidationProblem, ...] = () + """Every reason the document is not a valid one.""" + metadata: ZarrV3GroupMetadata | None = None + """The document's model when there is no problem; None otherwise.""" + + def fields(self) -> Iterator[tuple[Loc, Resolved[Any]]]: + """Each field of each document its consolidated metadata holds, as read, with where it sits in this document.""" + for path, reading in self.consolidated.items(): + for loc, node in reading.fields(): + yield (ZARR_V3_CONSOLIDATED_METADATA_KEY, "metadata", path, *loc), node + + +@dataclass(frozen=True, slots=True) +class GroupMembersV3: + """What a read refined of a v3 group document, as the models hold it: its own members, and those of each document its consolidated metadata holds that a model can be built of.""" + + attributes: dict[str, JSONValue] | None + """Its attributes; None when they have a problem.""" + extra_fields: dict[str, JSONValue] + """Each member the spec does not define that is JSON.""" + consolidated: Mapping[str, ArrayMembersV3 | GroupMembersV3] | UNSET + """Each array its consolidated metadata holds that has no problem, and each group, by path; UNSET when it holds none.""" + + +def read_group_metadata_v3( + value: object, *, context: Context = CORE_AND_EXTENSIONS +) -> ZarrV3GroupMetadataReading: + """`value`, a v3 group document, as `context` read it, whatever it holds. + + Everything a read finds, in one: each document its consolidated + metadata holds, read once, as `read_array_metadata_v3` and this read + one; every problem `validate_group_metadata_v3` finds; and, when there + is none, the group's model, whose consolidated metadata holds the + models of those documents, which their readings hold too. A value that + is not an object holds nothing. + """ + reading, members = read_group_v3(value, context) + if members is None: + return reading + return _with_models(reading, members) + + +def read_group_v3( + value: object, context: Context +) -> tuple[ZarrV3GroupMetadataReading, GroupMembersV3 | None]: + """`value`, a v3 group document, as `context` read it, without models, and its members refined; None when it is not an object. + + A field a scope already read -- a model's own, handed back -- is taken + as it is, as `read_array_v3` takes one. + """ + if not isinstance(value, Mapping): + problems = (ValidationProblem((), "expected an object", "invalid_type"),) + return ZarrV3GroupMetadataReading(problems=problems), None + doc = cast("Mapping[object, object]", value) + found: list[ValidationProblem] = list(missing_keys(GROUP_METADATA_REQUIRED_KEYS_V3, doc)) + extra_fields, others = other_members( + doc, + GROUP_METADATA_STANDARD_KEYS_V3, + additional_reserved_keys=frozenset({ZARR_V3_CONSOLIDATED_METADATA_KEY}), + ) + found.extend(others) + found.extend(check_literal(doc, "zarr_format", 3)) + found.extend(check_literal(doc, "node_type", "group")) + attributes: dict[str, JSONValue] | None = {} + if "attributes" in doc: + attributes, problems = attributes_of(doc["attributes"]) + found.extend(problems) + # consolidated_metadata: null, which a historical zarr-python bug wrote, + # is read as none, so those stores stay readable; the model never + # writes it back. + raw = doc.get(ZARR_V3_CONSOLIDATED_METADATA_KEY) + consolidated: dict[str, ZarrV3ArrayMetadataReading | ZarrV3GroupMetadataReading] = {} + held: Mapping[str, ArrayMembersV3 | GroupMembersV3] | UNSET = UNSET + if raw is not None: + consolidated, held, inside = _read_consolidated_v3(raw, context) + found.extend(_prefix(ZARR_V3_CONSOLIDATED_METADATA_KEY, inside)) + reading = ZarrV3GroupMetadataReading(consolidated, tuple(found)) + return reading, GroupMembersV3(attributes, extra_fields, held) + + +_CONSOLIDATED_MEMBERS: Final = ("kind", "must_understand", "metadata") +"""The members of an inline `consolidated_metadata`, in the order the convention declares them.""" + + +def _read_consolidated_v3( + value: object, context: Context +) -> tuple[ + dict[str, ZarrV3ArrayMetadataReading | ZarrV3GroupMetadataReading], + dict[str, ArrayMembersV3 | GroupMembersV3], + tuple[ValidationProblem, ...], +]: + """An inline `consolidated_metadata` member, as `context` read it: each document it holds, read once, by its path; the members of each a model can be built of; and every problem, located in the member.""" + if not isinstance(value, Mapping): + return {}, {}, (ValidationProblem((), "expected an object", "invalid_type"),) + env = cast("Mapping[object, object]", value) + # Missing members are reported in the order the envelope declares them. + problems: list[ValidationProblem] = [ + ValidationProblem((key,), "missing required key", "missing_key") + for key in _CONSOLIDATED_MEMBERS + if key not in env + ] + problems.extend(unexpected_keys(frozenset(_CONSOLIDATED_MEMBERS), env)) + problems.extend(check_literal(env, "kind", "inline")) + if "must_understand" in env and env["must_understand"] is not False: + problems.append(ValidationProblem(("must_understand",), "expected False", "invalid_value")) + readings: dict[str, ZarrV3ArrayMetadataReading | ZarrV3GroupMetadataReading] = {} + members: dict[str, ArrayMembersV3 | GroupMembersV3] = {} + entries = env.get("metadata") + if "metadata" in env and not isinstance(entries, Mapping): + problems.append(ValidationProblem(("metadata",), "expected an object", "invalid_type")) + elif isinstance(entries, Mapping): + for key, entry in cast("Mapping[object, object]", entries).items(): + if not isinstance(key, str): + problems.append( + ValidationProblem(("metadata",), f"non-string key {key!r}", "invalid_type") + ) + continue + node_type = _node_type(entry) + child: ArrayMembersV3 | GroupMembersV3 | None if node_type == "array": - entries[key] = ZarrV3ArrayMetadata.from_json(entry, context=context) + readings[key], child = read_array_v3(entry, context) + elif node_type == "group": + readings[key], child = read_group_v3(entry, context) else: - entries[key] = ZarrV3GroupMetadata.from_json(entry, context=context) - return cls(metadata=entries) + problems.append( + ValidationProblem( + ("metadata", key, "node_type"), + "expected 'array' or 'group'", + "invalid_value", + ) + ) + continue + if child is not None: + members[key] = child + problems.extend(_prefix("metadata", _prefix(key, readings[key].problems))) + return readings, members, tuple(problems) + + +def _node_type(document: object) -> object: + """The `node_type` a document says it is; None when it is not an object.""" + if not isinstance(document, Mapping): + return None + return cast("Mapping[object, object]", document).get("node_type") + + +def _with_models( + reading: ZarrV3GroupMetadataReading, members: GroupMembersV3 +) -> ZarrV3GroupMetadataReading: + """`reading`, holding the model of each document its consolidated metadata holds that has no problem, and its own when it has none.""" + readings, models = ( + (reading.consolidated, {}) + if members.consolidated is UNSET + else _models(reading.consolidated, members.consolidated) + ) + if len(reading.problems) != 0: + return dataclasses.replace(reading, consolidated=readings) + model = ZarrV3GroupMetadata( + attributes=cast("dict[str, JSONValue]", members.attributes), + consolidated_metadata=( + UNSET if members.consolidated is UNSET else ZarrV3ConsolidatedMetadata(metadata=models) + ), + extra_fields=members.extra_fields, + ) + return dataclasses.replace(reading, consolidated=readings, metadata=model) + + +def _models( + readings: Mapping[str, ZarrV3ArrayMetadataReading | ZarrV3GroupMetadataReading], + members: Mapping[str, ArrayMembersV3 | GroupMembersV3], +) -> tuple[ + dict[str, ZarrV3ArrayMetadataReading | ZarrV3GroupMetadataReading], + dict[str, ZarrV3ArrayMetadata | ZarrV3GroupMetadata], +]: + """Each reading, holding its model when a model can be built of its document, and those models, by path.""" + held: dict[str, ZarrV3ArrayMetadataReading | ZarrV3GroupMetadataReading] = dict(readings) + models: dict[str, ZarrV3ArrayMetadata | ZarrV3GroupMetadata] = {} + for path, child in members.items(): + reading = readings[path] + if isinstance(reading, ZarrV3ArrayMetadataReading): + array = array_model(reading, cast("ArrayMembersV3", child)) + held[path], models[path] = dataclasses.replace(reading, metadata=array), array + else: + group = _with_models(reading, cast("GroupMembersV3", child)) + held[path] = group + if group.metadata is not None: + models[path] = group.metadata + return held, models + + +def validate_group_metadata_v3( + value: object, *, context: Context = CORE_AND_EXTENSIONS +) -> tuple[ValidationProblem, ...]: + """Return every reason `value` is not a valid v3 group document. + + Unknown top-level keys are allowed (they map to `extra_fields`); a + reader must understand each one that does not say `must_understand: + false`, which the model reports as `must_understand_fields`. A + `consolidated_metadata` member, if present, is validated too: its + envelope, and each document it holds by its path, each array read as + `validate_array_metadata_v3` reads one, in `context`. These are the + `problems` of `read_group_metadata_v3`, which holds what was read to + find them. + """ + return read_group_v3(value, context)[0].problems + + +def is_group_metadata_v3( + value: object, *, context: Context = CORE_AND_EXTENSIONS +) -> TypeGuard[ZarrV3GroupMetadataJSON]: + """Whether `value` is a v3 group document `validate_group_metadata_v3` finds nothing wrong with, written with tuples.""" + return is_canonical_json(value, finite=False) and not validate_group_metadata_v3( + value, context=context + ) + + +def parse_group_metadata_v3( + value: object, *, context: Context = CORE_AND_EXTENSIONS +) -> ZarrV3GroupMetadataJSON: + """Return `value` narrowed to `ZarrV3GroupMetadataJSON`, or raise `MetadataValidationError`.""" + normalized = arrays_to_tuples(value) + problems = validate_group_metadata_v3(normalized, context=context) + if len(problems) != 0: + raise MetadataValidationError(problems) + return cast("ZarrV3GroupMetadataJSON", normalized) class ZarrV2GroupMetadataPartial(TypedDict, total=False): diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_validation.py b/packages/zarr-metadata/src/zarr_metadata/model/_validation.py index 5256222674..1e262467ef 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_validation.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/_validation.py @@ -10,7 +10,8 @@ handed. Each concept gets a `validate_*` function returning every problem found, an `is_*` type guard, and a `parse_*` function that narrows or raises `MetadataValidationError`; a v3 array document also -gets `read_array_metadata_v3`, which returns what it read beside them. +gets `read_array_metadata_v3`, one read that returns what it read, the +problems, and the model when there are none. The guards are `TypeGuard`s, not `TypeIs`: True narrows a value to its document type, and False says nothing about its type, since a value can be well typed and still not a valid document. @@ -33,6 +34,7 @@ MetadataValidationError, ValidationProblem, arrays_to_tuples, + refine_json, refine_user_data, refused_kind, shown, @@ -40,6 +42,7 @@ ) from zarr_metadata._json import is_canonical_json as _is_canonical_json from zarr_metadata._json import prefixed as _prefix +from zarr_metadata.model._sentinel import UNSET from zarr_metadata.v2.array import ZarrV2ArrayMetadataJSON from zarr_metadata.v2.group import ZarrV2GroupMetadataJSON from zarr_metadata.v3._definition import ( @@ -50,8 +53,10 @@ DataTypeDefinition, Definition, Lengths, + Read, Resolved, StorageTransformerDefinition, + Unclaimed, chunk_grid_lengths, fields_of, fill_value_problems, @@ -65,7 +70,9 @@ if TYPE_CHECKING: from collections.abc import Iterator + from zarr_metadata._common import JSONValue from zarr_metadata._typed_json import Loc + from zarr_metadata.model._array import ZarrV3ArrayMetadata # The standard top-level keys of a v3 array metadata document. Anything outside # this set is an extension field. Built from the TypedDict's required/optional @@ -113,7 +120,7 @@ ) -def _missing_keys( +def missing_keys( required: frozenset[str], doc: Mapping[object, object] ) -> tuple[ValidationProblem, ...]: """One `missing_key` problem per required key absent from `doc`.""" @@ -123,7 +130,7 @@ def _missing_keys( ) -def _unexpected_keys( +def unexpected_keys( allowed: frozenset[str], doc: Mapping[object, object] ) -> tuple[ValidationProblem, ...]: """One problem per member outside a closed document's declared shape.""" @@ -140,7 +147,7 @@ def _unexpected_keys( return tuple(problems) -def _check_literal( +def check_literal( doc: Mapping[object, object], key: str, expected: object ) -> tuple[ValidationProblem, ...]: """One problem if `doc[key]` is present but not `expected`: of its type, when it is not of `expected`'s JSON type, else of its value.""" @@ -150,7 +157,7 @@ def _check_literal( return () -def _validate_other_members( +def other_members_problems( doc: Mapping[object, object], standard_keys: frozenset[str], *, @@ -161,6 +168,17 @@ def _validate_other_members( For a document open to other members: v3 extension fields, and the members a v2 array's readers ignore. """ + return other_members(doc, standard_keys, additional_reserved_keys=additional_reserved_keys)[1] + + +def other_members( + doc: Mapping[object, object], + standard_keys: frozenset[str], + *, + additional_reserved_keys: frozenset[str] = frozenset(), +) -> tuple[dict[str, JSONValue], tuple[ValidationProblem, ...]]: + """Each member outside `standard_keys` refined to JSON, and every problem, as `other_members_problems` finds them; a member that is not JSON is left out.""" + members: dict[str, JSONValue] = {} problems: list[ValidationProblem] = [] reserved_keys = standard_keys | additional_reserved_keys for key, value in doc.items(): @@ -171,8 +189,11 @@ def _validate_other_members( continue if key in reserved_keys: continue - problems.extend(_prefix(key, validate_json(value))) - return tuple(problems) + refined, found = refine_json(value, (key,)) + problems.extend(found) + if len(found) == 0: + members[key] = refined + return members, tuple(problems) def _is_array(value: object) -> TypeGuard[Sequence[object]]: @@ -197,7 +218,7 @@ def _is_int_sequence(value: object) -> TypeGuard[Sequence[int]]: ) -def _dimension_lengths( +def dimension_lengths( doc: Mapping[object, object], key: str ) -> tuple[tuple[int, ...] | None, tuple[ValidationProblem, ...]]: """The dimension lengths `doc` holds at `key` (`shape`, `chunks`), and every problem with them. @@ -321,7 +342,7 @@ def _validate_codec_v2(value: object) -> tuple[ValidationProblem, ...]: return validate_json(value) -def _validate_attributes(value: object) -> tuple[ValidationProblem, ...]: +def validate_attributes(value: object) -> tuple[ValidationProblem, ...]: """Validate an `attributes` value: a mapping with string keys. Returns a problem at `("attributes",)` if it is not, else `[]`. Shared by the @@ -330,18 +351,28 @@ def _validate_attributes(value: object) -> tuple[ValidationProblem, ...]: already-parent-relative `("attributes",)` loc, since it is only ever called with a document's `attributes` value. """ + return attributes_of(value)[1] + + +def attributes_of( + value: object, +) -> tuple[dict[str, JSONValue] | None, tuple[ValidationProblem, ...]]: + """An `attributes` value refined as user data, and every problem `validate_attributes` finds; None when it is not an object with string keys, or holds a value that is not JSON.""" if not isinstance(value, Mapping) or not all( isinstance(k, str) for k in cast("Mapping[object, object]", value) ): - return ( + return None, ( ValidationProblem( ("attributes",), "expected an object with string keys", "invalid_type" ), ) + attributes: dict[str, JSONValue] = {} problems: list[ValidationProblem] = [] for key, item in cast("Mapping[str, object]", value).items(): - problems.extend(refine_user_data(item, ("attributes", key))[1]) - return tuple(problems) + refined, found = refine_user_data(item, ("attributes", key)) + problems.extend(found) + attributes[key] = refined + return (attributes if len(problems) == 0 else None), tuple(problems) _EXTENSION_POINTS_V3: Final[tuple[tuple[str, type[Definition[Any]]], ...]] = ( @@ -360,7 +391,7 @@ def _validate_attributes(value: object) -> tuple[ValidationProblem, ...]: @dataclass(frozen=True, slots=True) class ZarrV3ArrayMetadataReading: - """A v3 array document as a scope read it: each extension point, and its codecs as a pipeline. + """A v3 array document as a scope read it, whatever it holds: each extension point, its codecs as a pipeline, every problem, and the model when there is none. A field the document does not hold is None, and a list of them it does not hold as a list is empty. @@ -378,6 +409,10 @@ class ZarrV3ArrayMetadataReading: """The codecs, read as a pipeline: each as the scope read it, with the chunk it is handed.""" storage_transformers: tuple[Resolved[StorageTransformerDefinition[Any]], ...] = () """The storage transformers, each as the scope read it.""" + problems: tuple[ValidationProblem, ...] = () + """Every reason the document is not a valid one.""" + metadata: ZarrV3ArrayMetadata | None = None + """The document's model, holding these fields, when there is no problem; None otherwise.""" def fields(self) -> Iterator[tuple[Loc, Resolved[Any]]]: """Each field the document holds, as the scope read it, with where it sits in the document. @@ -399,28 +434,59 @@ def fields(self) -> Iterator[tuple[Loc, Resolved[Any]]]: yield from fields_of(transformer, ("storage_transformers", index)) -def read_array_metadata_v3( - value: object, *, context: Context = CORE_AND_EXTENSIONS -) -> tuple[ZarrV3ArrayMetadataReading, tuple[ValidationProblem, ...]]: - """`value`, a v3 array document, as `context` read it, and every reason it is not a valid one. - - The problems are `validate_array_metadata_v3`'s; the reading is what - was read to find them, so nothing need read the document again: each - extension point as `context` read it -- what claims it, or that - nothing in scope does -- the chunks the codecs are handed, and each - codec with the chunk it is handed. A policy over the fields, the core - spec's alone, say, is a walk over its `fields()`. A value that is not - a mapping holds no field, and reads as nothing. +NO_SCOPE: Final = Context.of() +"""A scope of no definitions, which a model's own document is read in: its fields are read already, and nothing else in it is a field.""" + + +@dataclass(frozen=True, slots=True) +class ArrayMembersV3: + """The members of a v3 array document a read found nothing wrong with, other than its fields, refined as the model holds them.""" + + shape: tuple[int, ...] + fill_value: JSONValue + dimension_names: tuple[str | None, ...] | UNSET + attributes: dict[str, JSONValue] + extra_fields: dict[str, JSONValue] + + +def read_field( + value: object, kind: type[Definition[Any]], context: Context, loc: Loc +) -> tuple[Resolved[Any], tuple[ValidationProblem, ...]]: + """`value`, one of a document's fields, as `context` reads it -- or as it is, when a scope has read it already. + + A model's own fields come back this way, so `update` reads only the + members it is given, and a field read in one scope keeps the + definition that read it, however the scope it is handed on in differs: + a pydantic model instance is taken as it is, too. + """ + if _read_as(value, kind): + return cast("Resolved[Any]", value), () + return resolve(value, kind, context, loc) + + +def _read_as(value: object, kind: type[Definition[Any]]) -> bool: + """Whether `value` is a field a scope already read, as a `kind`, which a model holds.""" + return isinstance(value, (Read, Unclaimed)) and value.read_as is kind + + +def read_array_v3( + value: object, context: Context +) -> tuple[ZarrV3ArrayMetadataReading, ArrayMembersV3 | None]: + """`value`, a v3 array document, as `context` read it, without its model, and its other members refined; None when it has a problem. + + `read_array_metadata_v3` builds the model from the two. A field a + scope already read -- a model's own, handed back -- is taken as it is. """ if not isinstance(value, Mapping): - nothing = ZarrV3ArrayMetadataReading() - return nothing, (ValidationProblem((), "expected an object", "invalid_type"),) + not_an_object = (ValidationProblem((), "expected an object", "invalid_type"),) + return ZarrV3ArrayMetadataReading(problems=not_an_object), None doc = cast("Mapping[object, object]", value) - problems: list[ValidationProblem] = list(_missing_keys(ARRAY_METADATA_REQUIRED_KEYS_V3, doc)) - problems.extend(_validate_other_members(doc, ARRAY_METADATA_STANDARD_KEYS_V3)) - problems.extend(_check_literal(doc, "zarr_format", 3)) - problems.extend(_check_literal(doc, "node_type", "array")) - shape, shape_problems = _dimension_lengths(doc, "shape") + problems: list[ValidationProblem] = list(missing_keys(ARRAY_METADATA_REQUIRED_KEYS_V3, doc)) + extra_fields, found = other_members(doc, ARRAY_METADATA_STANDARD_KEYS_V3) + problems.extend(found) + problems.extend(check_literal(doc, "zarr_format", 3)) + problems.extend(check_literal(doc, "node_type", "array")) + shape, shape_problems = dimension_lengths(doc, "shape") problems.extend(shape_problems) # Each extension point is read by `resolve`, which judges its envelope # -- every extension *point* must be understood, so a `must_understand` @@ -435,18 +501,17 @@ def read_array_metadata_v3( read: dict[str, Resolved[Any]] = {} for key, kind in _EXTENSION_POINTS_V3: if key in doc: - read[key], found = resolve(doc[key], kind, context, (key,)) + read[key], found = read_field(doc[key], kind, context, (key,)) problems.extend(found) # The fill value is JSON, and judged by the data type the scope read, # when there is one: a data type nothing in scope claims leaves it # unjudged. + fill_value: JSONValue = None if "fill_value" in doc: - if "data_type" in read: - problems.extend( - fill_value_problems(read["data_type"], doc["fill_value"], ("fill_value",)) - ) - else: - problems.extend(_prefix("fill_value", validate_json(doc["fill_value"]))) + fill_value, found = refine_json(doc["fill_value"], ("fill_value",)) + problems.extend(found) + if len(found) == 0 and "data_type" in read: + problems.extend(fill_value_problems(read["data_type"], fill_value, ("fill_value",))) # The chunk grid is judged against the shape, once both are read, and # says the lengths of the chunks the first codec is handed: an entry # for each dimension of the shape, None where nothing says it. @@ -463,7 +528,7 @@ def read_array_metadata_v3( else: listed[key] = [] for index, entry in enumerate(entries): - resolved, found = resolve(entry, kind, context, (key, index)) + resolved, found = read_field(entry, kind, context, (key, index)) listed[key].append(resolved) problems.extend(found) # The codecs are read as a pipeline, the first handed the grid's chunks @@ -474,8 +539,10 @@ def read_array_metadata_v3( if "codecs" in listed: pipeline, found = read_pipeline(listed["codecs"], chunk, ("codecs",)) problems.extend(found) + attributes: dict[str, JSONValue] | None = {} if "attributes" in doc: - problems.extend(_validate_attributes(doc["attributes"])) + attributes, found = attributes_of(doc["attributes"]) + problems.extend(found) if "dimension_names" in doc: # Simple typed sequences (dimension_names, shape, chunks) report a single # field-level loc, not per-bad-item locs; per-index locs are reserved for @@ -506,8 +573,19 @@ def read_array_metadata_v3( chunk=chunk, pipeline=pipeline, storage_transformers=tuple(listed.get("storage_transformers", ())), + problems=tuple(problems), + ) + if len(problems) != 0 or shape is None or attributes is None: + return reading, None + names = doc.get("dimension_names", UNSET) + members = ArrayMembersV3( + shape=shape, + fill_value=fill_value, + dimension_names=UNSET if names is UNSET else tuple(cast("Sequence[str | None]", names)), + attributes=attributes, + extra_fields=extra_fields, ) - return reading, tuple(problems) + return reading, members def validate_array_metadata_v3( @@ -530,10 +608,10 @@ def validate_array_metadata_v3( handed a chunk nothing is known of. Unknown top-level keys are allowed (they map to `extra_fields`); a reader must understand each one that does not say `must_understand: false`, which the model - reports as `must_understand_fields`. `read_array_metadata_v3` returns - what was read to find them, beside them. + reports as `must_understand_fields`. These are the `problems` of + `read_array_metadata_v3`, which holds what was read to find them. """ - return read_array_metadata_v3(value, context=context)[1] + return read_array_v3(value, context)[0].problems def is_array_metadata_v3( @@ -575,11 +653,11 @@ def validate_array_metadata_v2(value: object) -> tuple[ValidationProblem, ...]: # implementations" (https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v2/v2.0.rst#L91-L92), so a member outside # ARRAY_METADATA_STANDARD_KEYS_V2 is not a problem for being there. Ignored # is not unchecked: it is JSON, and its key a string, as in v3. - problems: list[ValidationProblem] = list(_missing_keys(ARRAY_METADATA_REQUIRED_KEYS_V2, doc)) - problems.extend(_validate_other_members(doc, ARRAY_METADATA_STANDARD_KEYS_V2)) - problems.extend(_check_literal(doc, "zarr_format", 2)) - shape, shape_problems = _dimension_lengths(doc, "shape") - chunks, chunks_problems = _dimension_lengths(doc, "chunks") + problems: list[ValidationProblem] = list(missing_keys(ARRAY_METADATA_REQUIRED_KEYS_V2, doc)) + problems.extend(other_members_problems(doc, ARRAY_METADATA_STANDARD_KEYS_V2)) + problems.extend(check_literal(doc, "zarr_format", 2)) + shape, shape_problems = dimension_lengths(doc, "shape") + chunks, chunks_problems = dimension_lengths(doc, "chunks") problems.extend(shape_problems) problems.extend(chunks_problems) if shape is not None and chunks is not None and len(shape) != len(chunks): @@ -636,7 +714,7 @@ def validate_array_metadata_v2(value: object) -> tuple[ValidationProblem, ...]: if "fill_value" in doc: problems.extend(_prefix("fill_value", validate_json(doc["fill_value"]))) if "attributes" in doc: - problems.extend(_validate_attributes(doc["attributes"])) + problems.extend(validate_attributes(doc["attributes"])) return tuple(problems) @@ -658,128 +736,6 @@ def parse_array_metadata_v2(value: object) -> ZarrV2ArrayMetadataJSON: return cast("ZarrV2ArrayMetadataJSON", normalized) -def validate_consolidated_metadata_v3( - value: object, *, context: Context = CORE_AND_EXTENSIONS -) -> tuple[ValidationProblem, ...]: - """Return every reason `value` is not a valid inline consolidated envelope. - - Locs are value-relative (the caller prefixes with `consolidated_metadata` - where appropriate). Entries recurse into the array and group document - validators, in `context`, so a validator verdict agrees with what - `ZarrV3ConsolidatedMetadata.from_json` accepts in the same scope. - """ - if not isinstance(value, Mapping): - return (ValidationProblem((), "expected an object", "invalid_type"),) - env = cast("Mapping[object, object]", value) - problems: list[ValidationProblem] = [ - ValidationProblem((key,), "missing required key", "missing_key") - for key in ("kind", "must_understand", "metadata") - if key not in env - ] - problems.extend(_unexpected_keys(frozenset({"kind", "must_understand", "metadata"}), env)) - problems.extend(_check_literal(env, "kind", "inline")) - if "must_understand" in env and env["must_understand"] is not False: - problems.append(ValidationProblem(("must_understand",), "expected False", "invalid_value")) - if "metadata" in env: - entries = env["metadata"] - if not isinstance(entries, Mapping): - problems.append(ValidationProblem(("metadata",), "expected an object", "invalid_type")) - else: - for key, entry in cast("Mapping[object, object]", entries).items(): - if not isinstance(key, str): - problems.append( - ValidationProblem(("metadata",), f"non-string key {key!r}", "invalid_type") - ) - continue - entry_obj: object = entry - node_type: object = None - if isinstance(entry, Mapping): - node_type = cast("Mapping[object, object]", entry).get("node_type") - if node_type == "array": - problems.extend( - _prefix( - "metadata", - _prefix(key, validate_array_metadata_v3(entry_obj, context=context)), - ) - ) - elif node_type == "group": - problems.extend( - _prefix( - "metadata", - _prefix(key, validate_group_metadata_v3(entry_obj, context=context)), - ) - ) - else: - problems.append( - ValidationProblem( - ("metadata", key, "node_type"), - "expected 'array' or 'group'", - "invalid_value", - ) - ) - return tuple(problems) - - -def validate_group_metadata_v3( - value: object, *, context: Context = CORE_AND_EXTENSIONS -) -> tuple[ValidationProblem, ...]: - """Return every reason `value` is not a valid v3 group document. - - Unknown top-level keys are allowed (they map to `extra_fields`); a - reader must understand each one that does not say `must_understand: - false`, which the model reports as `must_understand_fields`. A - `consolidated_metadata` member, if present, is validated too: its - envelope, and each document it holds by its path, each array read as - `validate_array_metadata_v3` reads one, in `context`. - """ - if not isinstance(value, Mapping): - return (ValidationProblem((), "expected an object", "invalid_type"),) - doc = cast("Mapping[object, object]", value) - problems: list[ValidationProblem] = list(_missing_keys(GROUP_METADATA_REQUIRED_KEYS_V3, doc)) - problems.extend( - _validate_other_members( - doc, - GROUP_METADATA_STANDARD_KEYS_V3, - additional_reserved_keys=frozenset({"consolidated_metadata"}), - ) - ) - problems.extend(_check_literal(doc, "zarr_format", 3)) - problems.extend(_check_literal(doc, "node_type", "group")) - if "attributes" in doc: - problems.extend(_validate_attributes(doc["attributes"])) - if "consolidated_metadata" in doc and doc["consolidated_metadata"] is not None: - # consolidated_metadata: null (a historical zarr-python bug) is - # structurally accepted so those stores remain readable, but the model - # repairs it to absence on read and never writes it back. - problems.extend( - _prefix( - "consolidated_metadata", - validate_consolidated_metadata_v3(doc["consolidated_metadata"], context=context), - ) - ) - return tuple(problems) - - -def is_group_metadata_v3( - value: object, *, context: Context = CORE_AND_EXTENSIONS -) -> TypeGuard[ZarrV3GroupMetadataJSON]: - """Whether `value` is a v3 group document `validate_group_metadata_v3` finds nothing wrong with, written with tuples.""" - return _is_canonical_json(value, finite=False) and not validate_group_metadata_v3( - value, context=context - ) - - -def parse_group_metadata_v3( - value: object, *, context: Context = CORE_AND_EXTENSIONS -) -> ZarrV3GroupMetadataJSON: - """Return `value` narrowed to `ZarrV3GroupMetadataJSON`, or raise `MetadataValidationError`.""" - normalized = arrays_to_tuples(value) - problems = validate_group_metadata_v3(normalized, context=context) - if len(problems) != 0: - raise MetadataValidationError(problems) - return cast(ZarrV3GroupMetadataJSON, normalized) - - def validate_group_metadata_v2(value: object) -> tuple[ValidationProblem, ...]: """Return every reason `value` is not a structurally-valid v2 group doc. @@ -789,11 +745,11 @@ def validate_group_metadata_v2(value: object) -> tuple[ValidationProblem, ...]: if not isinstance(value, Mapping): return (ValidationProblem((), "expected an object", "invalid_type"),) doc = cast("Mapping[object, object]", value) - problems: list[ValidationProblem] = list(_missing_keys(GROUP_METADATA_REQUIRED_KEYS_V2, doc)) - problems.extend(_unexpected_keys(GROUP_METADATA_STANDARD_KEYS_V2, doc)) - problems.extend(_check_literal(doc, "zarr_format", 2)) + problems: list[ValidationProblem] = list(missing_keys(GROUP_METADATA_REQUIRED_KEYS_V2, doc)) + problems.extend(unexpected_keys(GROUP_METADATA_STANDARD_KEYS_V2, doc)) + problems.extend(check_literal(doc, "zarr_format", 2)) if "attributes" in doc: - problems.extend(_validate_attributes(doc["attributes"])) + problems.extend(validate_attributes(doc["attributes"])) return tuple(problems) diff --git a/packages/zarr-metadata/src/zarr_metadata/pydantic.py b/packages/zarr-metadata/src/zarr_metadata/pydantic.py index 2cb69f50d2..adf285f5f4 100644 --- a/packages/zarr-metadata/src/zarr_metadata/pydantic.py +++ b/packages/zarr-metadata/src/zarr_metadata/pydantic.py @@ -8,11 +8,17 @@ freely with non-pydantic code (equality, isinstance, nesting). Validation delegates to the library: a raw document routes through `from_json` (the single source of truth for validation and normalization, so pydantic's -field-level coercion can never bypass it), reading v3 extension points in -`CORE_AND_EXTENSIONS`, since a field type holds no scope; a reader with a -scope of its own calls `from_json(..., context=...)` itself. An existing model -instance passes through unchanged, and serialization emits the canonical -document via `to_json`. `MetadataValidationError` subclasses `ValueError`, +field-level coercion can never bypass it). A v3 field type reads extension +points in the scope pydantic's validation context holds, as pydantic hands +any validator its context: the context itself, when it is a `Context`, or +its `"zarr_metadata_context"` item, when it is a mapping; otherwise +`CORE_AND_EXTENSIONS`: + + TypeAdapter(zmp.ZarrV3ArrayMetadata).validate_python(document, context=SCOPE) + ArrayManifest.model_validate(data, context={"zarr_metadata_context": SCOPE}) + +An existing model instance passes through unchanged, as pydantic's does, and +serialization emits the canonical document via `to_json`. `MetadataValidationError` subclasses `ValueError`, so a failed parse surfaces as a pydantic `ValidationError` carrying the loc-annotated problem messages. @@ -37,9 +43,10 @@ class ArrayManifest(BaseModel): from __future__ import annotations -from typing import TYPE_CHECKING, Annotated, TypeVar +from collections.abc import Mapping +from typing import TYPE_CHECKING, Annotated, Final, Protocol, TypeVar, cast -from pydantic import BeforeValidator, InstanceOf, PlainSerializer +from pydantic import BeforeValidator, InstanceOf, PlainSerializer, ValidationInfo from zarr_metadata import model as _model from zarr_metadata._pydantic_schema import ( @@ -60,9 +67,7 @@ class ArrayManifest(BaseModel): from zarr_metadata._pydantic_schema import ( ZarrV3GroupMetadataJSON as _ZarrV3GroupMetadataSchema, ) -from zarr_metadata._pydantic_schema import ( - ZarrV3MetadataFieldJSON as _ZarrV3MetadataFieldSchema, -) +from zarr_metadata.v3._registry import CORE_AND_EXTENSIONS, Context if TYPE_CHECKING: from collections.abc import Callable @@ -81,10 +86,45 @@ def coerce(value: object) -> _M: return coerce +CONTEXT_KEY: Final = "zarr_metadata_context" +"""The item of a mapping pydantic's validation context is that holds the scope a v3 field type reads in.""" + + +_Read_co = TypeVar("_Read_co", covariant=True) + + +class _Reads(Protocol[_Read_co]): + def __call__(self, data: object, /, *, context: Context) -> _Read_co: ... + + +def _read_in_scope(cls: type[_M], read: _Reads[_M]) -> Callable[[object, ValidationInfo], _M]: + """A validator that passes instances of `cls` through and reads anything else in the scope the validation context holds.""" + + def coerce(value: object, info: ValidationInfo) -> _M: + if isinstance(value, cls): + return value + return read(value, context=_scope(info.context)) + + return coerce + + +def _scope(context: object) -> Context: + """The scope a validation context holds: itself, a `Context`; its `CONTEXT_KEY` item; or, holding none, `CORE_AND_EXTENSIONS`.""" + if isinstance(context, Context): + return context + if not isinstance(context, Mapping) or CONTEXT_KEY not in context: + return CORE_AND_EXTENSIONS + scope = cast("Mapping[object, object]", context)[CONTEXT_KEY] + if not isinstance(scope, Context): + msg = f"{CONTEXT_KEY}: the scope to read in is a Context, got {scope!r}" + raise TypeError(msg) + return scope + + ZarrV3ArrayMetadata = Annotated[ InstanceOf[_model.ZarrV3ArrayMetadata], BeforeValidator( - _coerce_to(_model.ZarrV3ArrayMetadata, _model.ZarrV3ArrayMetadata.from_json), + _read_in_scope(_model.ZarrV3ArrayMetadata, _model.ZarrV3ArrayMetadata.from_json), json_schema_input_type=_ZarrV3ArrayMetadataSchema, ), PlainSerializer(_model.ZarrV3ArrayMetadata.to_json, return_type=_ZarrV3ArrayMetadataSchema), @@ -104,7 +144,7 @@ def coerce(value: object) -> _M: ZarrV3GroupMetadata = Annotated[ InstanceOf[_model.ZarrV3GroupMetadata], BeforeValidator( - _coerce_to(_model.ZarrV3GroupMetadata, _model.ZarrV3GroupMetadata.from_json), + _read_in_scope(_model.ZarrV3GroupMetadata, _model.ZarrV3GroupMetadata.from_json), json_schema_input_type=_ZarrV3GroupMetadataSchema, ), PlainSerializer(_model.ZarrV3GroupMetadata.to_json, return_type=_ZarrV3GroupMetadataSchema), @@ -124,7 +164,7 @@ def coerce(value: object) -> _M: ZarrV3ConsolidatedMetadata = Annotated[ InstanceOf[_model.ZarrV3ConsolidatedMetadata], BeforeValidator( - _coerce_to( + _read_in_scope( _model.ZarrV3ConsolidatedMetadata, _model.ZarrV3ConsolidatedMetadata.from_json, ), @@ -153,22 +193,12 @@ def coerce(value: object) -> _M: ] """Field type for a v2 `.zmetadata` document.""" -ZarrV3MetadataField = Annotated[ - InstanceOf[_model.ZarrV3NamedConfig], - BeforeValidator( - _coerce_to(_model.ZarrV3NamedConfig, _model.ZarrV3NamedConfig.from_json), - json_schema_input_type=_ZarrV3MetadataFieldSchema, - ), - PlainSerializer(_model.ZarrV3NamedConfig.to_json, return_type=_ZarrV3MetadataFieldSchema), -] -"""Field type for one normalized v3 metadata extension envelope.""" - __all__ = [ + "CONTEXT_KEY", "ZarrV2ArrayMetadata", "ZarrV2ConsolidatedMetadata", "ZarrV2GroupMetadata", "ZarrV3ArrayMetadata", "ZarrV3ConsolidatedMetadata", "ZarrV3GroupMetadata", - "ZarrV3MetadataField", ] diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py b/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py index 5b8c2a910b..dfa8189f36 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py @@ -51,7 +51,7 @@ from typing_extensions import TypeAliasType, TypedDict, TypeVar, is_typeddict from zarr_metadata._common import JSONValue, ZarrV3NamedConfigJSON -from zarr_metadata._json import ValidationProblem, refine_json, shown +from zarr_metadata._json import ValidationProblem, copied, refine_json, shown from zarr_metadata._typed_json import ( Loc, Parsed, @@ -634,25 +634,50 @@ def leaf(annotation: object) -> Parser | None: def _checked( shape: type, value: object, loc: Loc ) -> tuple[object, Problems, tuple[_NestedField, ...]]: - """`value` checked as `shape`: the typed value, every problem, and the fields nested in it, in order.""" + """`value` checked as `shape`: the typed value, every problem, and the fields nested in it, in order, each put back as it was written.""" + typed, found, nested = _typed(shape, value, loc) + return _put_back(typed, _as_written), found, nested + + +def _typed( + shape: type, value: object, loc: Loc +) -> tuple[object, Problems, tuple[_NestedField, ...]]: + """`value` checked as `shape`, each nested field still where the checker met it: the typed value, every problem, and those fields, in order.""" checker = _checker(shape) typed, found = checker.parse(value, loc) if not checker.nests: return typed, found, () nested: list[_NestedField] = [] - return _put_back(typed, nested), found, tuple(nested) + _collect(typed, nested) + return typed, found, tuple(nested) -def _put_back(value: object, nested: list[_NestedField]) -> object: - """`value` with each nested field the checker handed back put back as its JSON, collected in order.""" +def _collect(value: object, nested: list[_NestedField]) -> None: + """Each nested field the checker handed back in `value`, in order.""" if isinstance(value, _NestedField): nested.append(value) - return value.json + elif isinstance(value, tuple): + for entry in cast("tuple[object, ...]", value): + _collect(entry, nested) + elif isinstance(value, dict): + for entry in cast("dict[str, object]", value).values(): + _collect(entry, nested) + + +def _as_written(field: _NestedField) -> JSONValue: + """A nested field put back as it was written.""" + return field.json + + +def _put_back(value: object, put: Callable[[_NestedField], JSONValue]) -> object: + """`value` with each nested field the checker handed back put back as `put` gives it.""" + if isinstance(value, _NestedField): + return put(value) if isinstance(value, tuple): - return tuple(_put_back(entry, nested) for entry in cast("tuple[object, ...]", value)) + return tuple(_put_back(entry, put) for entry in cast("tuple[object, ...]", value)) if isinstance(value, dict): entries = cast("dict[str, object]", value) - return {key: _put_back(entry, nested) for key, entry in entries.items()} + return {key: _put_back(entry, put) for key, entry in entries.items()} return value @@ -753,109 +778,186 @@ def named_configuration( return name, cast("Mapping[str, object]", configuration), () -Unread = Literal["out_of_scope", "invalid"] -"""A field no definition read: nothing in scope claims its name, or it could not be read.""" - -Resolution = Literal["read"] | Unread -"""What a scope made of a field: read by the definition that claims it, or unread, and why.""" - - def _nothing_nested() -> Nested: - """What a field that holds no field, or was not read, holds inside: nothing.""" + """What a field that holds no field it read holds inside: nothing.""" return {} -@dataclass(frozen=True, slots=True) -class Resolved(Generic[D]): - """One metadata field, as read in a scope: its JSON, and what the scope made of it. +@dataclass(frozen=True, slots=True, kw_only=True) +class Read(Generic[D]): + """A field a definition in scope read: the name it is written with, the definition, and the configuration it allowed. - `resolution` is what became of the configuration: read by the - definition that claims the name, claimed by nothing, or not readable. A problem with the envelope around it -- a stray member, a - `must_understand` of `false` -- is reported with the field, and leaves - the resolution as it is; so is a problem of a field the configuration - holds, which is that field's own, with its own resolution in `nested`. - Built by hand, a reading is checked as `resolve` builds one: `read_as` - is one of the five kinds, type arguments dropped; a definition is of that - kind and filed under the name the field is written with; a field read - has a definition and a configuration, one unread no configuration, and - one nothing claims no definition. Anything else is a `TypeError`. + `must_understand` of `false` -- is reported with the field and leaves + it read; so is a problem of a field its configuration holds, which is + that field's own, as `nested` says. Two fields are equal when they + read the same, however each was spelled: `"bytes"` and + `{"name": "bytes"}` are one field. """ - json: JSONValue - """The field as written, refined: arrays as tuples. `None` for a value that was not JSON, as for `null`.""" - resolution: Resolution - definition: D | None - """The definition that claims the field's name; None when nothing in scope does, or it names none.""" - configuration: Mapping[str, JSONValue] | None - """The configuration, type-checked and allowed by the rules, when the field was read; None otherwise.""" + json: JSONValue = dataclasses.field(compare=False) + """The field as written, refined: arrays as tuples.""" + name: str + """The name it is written with: `"r16"`, though its definition is filed under `r*`.""" + definition: D + """The definition that read it.""" + configuration: Mapping[str, JSONValue] + """The configuration, type-checked and allowed by the rules; for raw bits, what the name carries. + + Each field it holds is written as a document writes it, as that + field's `to_json` writes it, so the configuration says what was read + however it was spelled: a shard's `"crc32c"` and `{"name": "crc32c"}` + are one index codec. + """ nested: Nested = dataclasses.field(default_factory=_nothing_nested) """The fields the configuration holds, each as the scope read it, by where it sits in the configuration. A struct's field types at `("fields", 0, "data_type")`, a shard's codecs at `("codecs", 0)`: what a definition's functions consult about - the fields inside its own. Each is read whatever became of the field - holding it, so a field its rules refuse keeps them too; empty when - its configuration was not checked against its TypedDict -- nothing - claims its name, it is not an object, a configuration it requires is - missing, or its name carries it, as raw bits' does. + the fields inside its own. """ - read_as: type[Definition[Any]] = dataclasses.field(kw_only=True) - """The kind of metadata the field was read as -- `CodecDefinition`, `DataTypeDefinition` -- whether or not anything in scope claims it: the `kind` `resolve` was asked for. + read_as: type[Definition[Any]] = dataclasses.field(init=False, repr=False, compare=False) + """The kind of metadata it was read as: its definition's.""" - Named apart from a codec's `kind` -- array -> array and so on, which - its definition says -- and a problem's. + def __post_init__(self) -> None: + # The runtime half of the annotations: a field read by hand, as an + # extension's may be, fails here rather than where a function trusts it. + definition = cast("object", self.definition) + kind = kind_of(cast("Definition[Any]", definition)) + refusal = ( + f"a field read is read by a definition of a kind, got {definition!r}" + if kind is None + else _misread(definition, kind, self.name) + ) + if refusal is not None: + raise TypeError(refusal) + object.__setattr__(self, "read_as", kind) + + def to_json(self) -> JSONValue: + """The field as a document writes it, for every reader: its configuration as read, sharing nothing with the field. + + The envelope takes the fewest words every reader takes: a data type + with nothing to configure is its bare name, as core data types have + been written since Zarr v3.0; any other field is an object, + `{"name": ...}`, since a Zarr v3.0 reader takes no bare name in + `codecs` + (https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v3/core/index.rst#L585-L592). + A name that carries its configuration, as raw bits' does, is + written alone. + """ + return copied(document_json(self)) + + +@dataclass(frozen=True, slots=True, kw_only=True) +class Unclaimed: + """A field nothing in scope claims: an extension the scope leaves unjudged, which is what keeps the format open. + + Equal to another when it is written with the same name and + configuration, however each was spelled. """ + json: JSONValue = dataclasses.field(compare=False) + """The field as written, refined: arrays as tuples.""" + name: str + """The name nothing in scope claims.""" + read_as: type[Definition[Any]] + """The kind of metadata it was read as: what a definition that claimed it would be.""" + configuration: Mapping[str, JSONValue] = dataclasses.field(init=False) + """The configuration as written, which nothing judged; empty when none is written.""" + + def __post_init__(self) -> None: + # The runtime half of the annotations; `read_as` with its type + # arguments dropped, as `resolve` drops them. + object.__setattr__(self, "read_as", as_kind(self.read_as)) + name = cast("object", self.name) + if not isinstance(name, str): + msg = f"a field nothing in scope claims is named, got {name!r}" + raise TypeError(msg) + _, written, _ = named_configuration(self.json) + configuration: Mapping[str, object] = {} if written is None else written + object.__setattr__(self, "configuration", cast("Mapping[str, JSONValue]", configuration)) + + @property + def definition(self) -> None: + """The definition that read it: none did.""" + return None + + @property + def nested(self) -> Nested: + """The fields its configuration holds as the scope read them: none, since nothing read its configuration.""" + return _nothing_nested() + + def to_json(self) -> JSONValue: + """The field as a document writes it, sharing nothing with the field: its configuration as written, in the envelope every reader takes, as `Read.to_json` writes one.""" + return copied(document_json(self)) + + +@dataclass(frozen=True, slots=True, kw_only=True) +class Refused(Generic[D]): + """A field that could not be read -- not a field at all, not JSON, or refused by the definition that claims its name -- as its problems say.""" + + json: JSONValue | None + """The field as written, refined: arrays as tuples; None when it is not JSON.""" + name: str | None + """The name it is written with; None when it names none.""" + read_as: type[Definition[Any]] + """The kind of metadata it was read as.""" + definition: D | None = None + """The definition that claims its name and refused it; None when nothing in scope claims it, or it names none.""" + nested: Nested = dataclasses.field(default_factory=_nothing_nested) + """The fields its configuration holds, each as the scope read it; empty when its configuration was not checked against its TypedDict.""" + def __post_init__(self) -> None: - # The runtime half of the annotations, as `Chunk` checks its own: a - # reading built by hand, as an extension's may be, fails here rather - # than where a function trusts its kind. + # The runtime half of the annotations; `read_as` with its type + # arguments dropped, as `resolve` drops them. kind = as_kind(self.read_as) object.__setattr__(self, "read_as", kind) definition = cast("object", self.definition) - if definition is not None and not isinstance(definition, kind): - msg = ( - f"a field read as a {kind.__name__} is read by one, " - f"got a {type(definition).__name__}" - ) - raise TypeError(msg) - refusal = _contradiction(self) + refusal = None if definition is None else _misread(definition, kind, self.name) if refusal is not None: raise TypeError(refusal) - @property - def name(self) -> str | None: - """The name the field was written with -- `"r16"`, though its definition is filed under `r*` -- or None when it names none. - What a message names it by. Which data type or codec it is, its - definition says, and its configuration: `r16` is `r*` of 16 bits. - None too for a field that is not JSON -- a `NaN` anywhere in it -- - which is held as None, and which its problems say. - """ - return named_configuration(self.json)[0] - - -def _contradiction(resolved: Resolved[Any]) -> str | None: - """What a reading says of itself that cannot hold together; None when it holds.""" - definition, configuration = resolved.definition, resolved.configuration - if (resolved.resolution == "read") != (configuration is not None): - return f"a field's configuration is held when it was read, got one {resolved.resolution!r}" - if resolved.resolution == "read" and definition is None: - return "a field read is read by a definition, got none" - if resolved.resolution == "out_of_scope": - if definition is not None: - return f"a field nothing in scope claims has no definition, got {definition.name!r}" - if resolved.name is None or len(resolved.nested) != 0: - return "a field nothing in scope claims is named, and holds no field read inside it" - name = resolved.name - if definition is not None and ( - name is None or spelled(resolved.read_as, name)[0] != definition.name - ): - return f"a field named {name!r} is read by the definition filed under it, got {definition.name!r}" +Resolved = TypeAliasType("Resolved", "Read[D] | Unclaimed | Refused[D]", type_params=(D,)) +"""One metadata field as a scope read it: read by the definition that claims its name, claimed by nothing, or refused.""" + +_FIELDS: Final = (Read, Unclaimed, Refused) +"""The three things a scope makes of a field.""" + + +def _misread(definition: object, kind: type[Definition[Any]], name: object) -> str | None: + """What is wrong with `definition` as the one that read a field named `name` as a `kind`; None when nothing is.""" + if not isinstance(definition, kind): + return f"a field read as a {kind.__name__} is read by one, got {definition!r}" + filed = cast("Definition[Any]", definition).name + if not isinstance(name, str) or spelled(kind, name)[0] != filed: + return f"a field named {name!r} is read by the definition filed under it, got {filed!r}" return None +def document_json(field: Resolved[Any]) -> JSONValue: + """A field as a document writes it, holding the field's own values: as `to_json` writes it, or as it was written when it was refused. + + What a writer serializes, which changes nothing, so it copies nothing; + `to_json` is this, copied. + """ + if isinstance(field, Refused): + return field.json + kind = field.read_as + if isinstance(field, Read) and spelled(kind, field.name)[1] is not None: + return _envelope_json(kind, field.name, {}) + return _envelope_json(kind, field.name, field.configuration) + + +def _envelope_json( + kind: type[Definition[Any]], name: str, configuration: Mapping[str, JSONValue] +) -> JSONValue: + """A field's envelope in the fewest words every reader takes: a data type with nothing to configure by its bare name, any other field an object.""" + if len(configuration) != 0: + return {"name": name, "configuration": configuration} + return name if kind is DataTypeDefinition else {"name": name} + + Nested: TypeAlias = Mapping[Loc, Resolved[Any]] """The fields a configuration holds, each as the scope read it, by where it sits in the configuration.""" @@ -899,9 +1001,7 @@ def rank(self) -> int | None: def _is_data_type_field(value: object) -> bool: """Whether `value` is a field a scope read as a data type.""" - return ( - isinstance(value, Resolved) and cast("Resolved[Any]", value).read_as is DataTypeDefinition - ) + return isinstance(value, _FIELDS) and cast("Resolved[Any]", value).read_as is DataTypeDefinition def _is_lengths(value: object) -> TypeGuard[Lengths]: @@ -935,11 +1035,12 @@ def fields_of(resolved: Resolved[Any], loc: Loc = ()) -> Iterator[tuple[Loc, Res def configuration_of(resolved: Resolved[Any], definition: Definition[C]) -> C | None: """The configuration `resolved` holds, typed as `definition` declares it, if `definition` read it. - `Resolved` holds a configuration as the mapping every one is; asked - with the definition that read the field, this is the same mapping, as - its TypedDict. None when another definition read it, or none did. + `Read` holds a configuration as the mapping every one is; asked with + the definition that read the field -- or one equal to it, as a + pickled field's is -- this is the same mapping, as its TypedDict. None + when another definition read it, or none did. """ - if resolved.definition is not definition or resolved.configuration is None: + if not isinstance(resolved, Read) or resolved.definition != definition: return None return cast("C", resolved.configuration) @@ -960,10 +1061,9 @@ def fill_value_problems( value unjudged. `loc` prefixes every problem. """ refined, problems = refine_json(value, loc) - definition = data_type.definition - configuration = data_type.configuration - if len(problems) != 0 or definition is None or configuration is None: + if len(problems) != 0 or not isinstance(data_type, Read): return problems + definition, configuration = data_type.definition, data_type.configuration typed, problems = _fill_value_parser(definition.fill_value)(refined, loc) if not _usable(problems): return problems @@ -983,10 +1083,9 @@ def storage_of(data_type: Resolved[DataTypeDefinition[Any]]) -> StorageClass | N checked to be a storage class, and an error it raises says which data type's storage raised it. """ - definition = data_type.definition - configuration = data_type.configuration - if definition is None or configuration is None: + if not isinstance(data_type, Read): return None + definition, configuration = data_type.definition, data_type.configuration found = asked( definition, "storage", @@ -1020,10 +1119,9 @@ def chunk_grid_lengths( definition, not the field. """ unknown: Lengths = (None,) * len(shape) - definition = chunk_grid.definition - configuration = chunk_grid.configuration - if definition is None or configuration is None: + if not isinstance(chunk_grid, Read): return unknown, () + definition, configuration = chunk_grid.definition, chunk_grid.configuration at = (*loc, "configuration") problems = ruled( definition, lambda: definition.shape_rules(configuration, chunk_grid.nested, shape), at @@ -1063,9 +1161,10 @@ def resolve( configuration is checked against its TypedDict and judged by its rules; each nested field the check met is read the same way, in the same scope, and what is wrong with one is its own, reported where it - sits, as with a document's fields. A name nothing claims is - `out_of_scope`: an unmodelled - extension, left unjudged, which is what keeps the format open. `loc` + sits, as with a document's fields. What comes back is `Read` by the + definition that claims the name; `Unclaimed` when nothing in scope + claims it, an unmodelled extension left unjudged, which is what keeps + the format open; or `Refused`, with the problems that say why. `loc` prefixes every problem. `kind` is one of the five kinds -- `CodecDefinition`, `DataTypeDefinition`, `ChunkGridDefinition`, `ChunkKeyEncodingDefinition`, `StorageTransformerDefinition` -- with @@ -1074,7 +1173,12 @@ def resolve( asked = as_kind(kind) refined, problems = refine_json(data, loc) if len(problems) != 0: - return Resolved(None, "invalid", None, None, read_as=asked), problems + # Not JSON, so not read; its name, if it has one, still says what + # claims it. + name = named_configuration(data)[0] + claimant = None if name is None else context.claimant(asked, name) + refused = Refused(json=None, name=name, read_as=asked, definition=claimant) + return cast("Resolved[D]", refused), problems resolved, found = _resolve_field(refined, asked, context, loc) return cast("Resolved[D]", resolved), found @@ -1084,9 +1188,10 @@ def _resolve_field( ) -> tuple[Resolved[Definition[Any]], Problems]: """A refined field with its envelope judged, then read. - The resolution is what became of the configuration. A stray member or - a `must_understand` of `false` says nothing about it, so it is reported - beside the field that was read, which later layers can still judge. + What the scope made of it is what became of the configuration. A + stray member or a `must_understand` of `false` says nothing about it, + so it is reported beside the field that was read, which later layers + can still judge. """ envelope = _located(loc, envelope_problems(data, allow_must_understand_false=False)) resolved, found = _read(data, kind, context, loc) @@ -1098,43 +1203,56 @@ def _read( ) -> tuple[Resolved[Definition[Any]], Problems]: name, given, malformed = named_configuration(data) if name is None: - return Resolved(data, "invalid", None, None, read_as=kind), () + return Refused(json=data, name=None, read_as=kind), () definition = context.claimant(kind, name) if len(malformed) != 0: # A configuration that is not an object, which the envelope's # problems say; the name still says what claims the field. - return Resolved(data, "invalid", definition, None, read_as=kind), () + return Refused(json=data, name=name, read_as=kind, definition=definition), () if definition is None: - return Resolved(data, "out_of_scope", None, None, read_as=kind), () + return Unclaimed(json=data, name=name, read_as=kind), () _, carried = spelled(kind, name) if carried is not None: - return _read_carried(data, kind, definition, given, carried, loc) + return _read_carried(data, name, definition, given, carried, loc) at = (*loc, "configuration") if given is None and definition.requires_configuration: missing = problem(at, f"{name!r} requires a configuration", "missing_key") - return Resolved(data, "invalid", definition, None, read_as=kind), missing - typed, found, nested = _checked(definition.configuration, {} if given is None else given, at) - # The rules may read a field the configuration holds by its name, so - # they are asked only when each one is named; any other problem with - # one is its own, reported where it sits, as a document's fields are. - sound = _usable(found) and all(_named(field) for field in nested) - configuration = cast("Mapping[str, JSONValue]", typed) if sound else None - # The fields it holds are read first, so the rules see them as the - # scope read them; their problems are reported after the rules'. + return Refused(json=data, name=name, read_as=kind, definition=definition), missing + typed, found, nested = _typed(definition.configuration, {} if given is None else given, at) + # The fields it holds are read first, each a frame deeper than this + # one, so the rules see them as the scope read them; each is put back + # as a document writes it, so the configuration says what was read + # however each was spelled. Their problems are reported after the + # rules'. within: dict[Loc, Resolved[Any]] = {} + written: dict[Loc, JSONValue] = {} inside: list[ValidationProblem] = [] for field in nested: inside.extend(_envelope(field)) inner, found_inside = _read(field.json, field.kind, context, field.loc) within[field.loc[len(at) :]] = inner + written[field.loc] = document_json(inner) inside.extend(found_inside) - inside.extend(_sized(field, inner.definition)) + inside.extend(_sized(field, inner)) + # The rules may read a field the configuration holds by its name, so + # they are asked only when each one is named; any other problem with + # one is its own, reported where it sits, as a document's fields are. + sound = _usable(found) and all(_named(field) for field in nested) + configuration = ( + cast("Mapping[str, JSONValue]", _put_back(typed, lambda field: written[field.loc])) + if sound + else None + ) own = list(found) if configuration is not None: own.extend(ruled(definition, lambda: definition.rules(configuration, within), at)) if configuration is None or not _usable(own): - return Resolved(data, "invalid", definition, None, within, read_as=kind), (*own, *inside) - return Resolved(data, "read", definition, configuration, within, read_as=kind), (*own, *inside) + refused = Refused(json=data, name=name, read_as=kind, definition=definition, nested=within) + return refused, (*own, *inside) + read = Read( + json=data, name=name, definition=definition, configuration=configuration, nested=within + ) + return read, (*own, *inside) def _named(field: _NestedField) -> bool: @@ -1145,7 +1263,7 @@ def _named(field: _NestedField) -> bool: def _read_carried( data: JSONValue, - kind: type[Definition[Any]], + name: str, definition: Definition[Any], given: Mapping[str, object] | None, carried: Mapping[str, JSONValue], @@ -1164,24 +1282,26 @@ def _read_carried( configuration, judged = definition.judge(carried) problems = (*beside, *(ValidationProblem(loc, found.message, found.kind) for found in judged)) if configuration is None or not _usable(problems): - return Resolved(data, "invalid", definition, None, read_as=kind), problems - return Resolved(data, "read", definition, configuration, read_as=kind), problems + refused = Refused(json=data, name=name, read_as=DataTypeDefinition, definition=definition) + return refused, problems + return Read(json=data, name=name, definition=definition, configuration=configuration), problems -def _sized(field: _NestedField, definition: Definition[Any] | None) -> Problems: +def _sized(field: _NestedField, inner: Resolved[Any]) -> Problems: """A codec of dynamic size in a member that takes codecs of static size, as a problem at the field. A name nothing in scope claims is left unjudged, its size unknown, as everything else about it is. """ + definition = inner.definition if not field.static or not isinstance(definition, CodecDefinition): return () if definition.size == "static": return () - name, _, _ = named_configuration(field.json) return problem( field.loc, - f"{name!r} is a codec of dynamic size, and only codecs of static size may be used here", + f"{inner.name!r} is a codec of dynamic size, and only codecs of static size may be used " + "here", "invalid_value", ) @@ -1206,31 +1326,30 @@ def canonicalize( takes no short-hand name in `codecs` (https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v3/core/index.rst#L585-L592); and there is no `must_understand`, since `true` is what absence means. - A name nothing in scope claims comes back as written, since what it - simplifies to is its own definition's call. + A name nothing in scope claims keeps the configuration it was written + with, since what it simplifies to is its own definition's call. """ resolved, problems = resolve(data, kind, context, loc) if len(problems) != 0: return None, problems - if resolved.definition is None: - return resolved.json, () - return _canonical_field(resolved.definition, resolved), () + return _simplest(resolved), () -def _canonical_field(definition: Definition[Any], resolved: Resolved[Any]) -> JSONValue: - """A field that read, in its simplest equivalent spelling: the fields it holds first, then its own members. +def _simplest(field: Resolved[Any]) -> JSONValue: + """A field without problems in its simplest spelling: one nothing in scope claims as a document writes it.""" + if isinstance(field, Read): + return _canonical_field(field) + if isinstance(field, Unclaimed): + return field.to_json() + return field.json - Each field it holds is spelled from what `resolve` read of it, kept in - `nested`; one nothing in scope claims keeps the spelling it was written - in. - """ - name, _, _ = named_configuration(resolved.json) - configuration: JSONValue = dict(resolved.configuration or {}) + +def _canonical_field(resolved: Read[Any]) -> JSONValue: + """A field that read, in its simplest equivalent spelling: the fields it holds first, then its own members.""" + definition, name = resolved.definition, resolved.name + configuration: JSONValue = dict(resolved.configuration) for loc, inner in resolved.nested.items(): - simplest = ( - inner.json if inner.definition is None else _canonical_field(inner.definition, inner) - ) - configuration = _replaced(configuration, loc, simplest) + configuration = _replaced(configuration, loc, _simplest(inner)) simplified = cast("Mapping[str, JSONValue]", definition.canonical(configuration)) _, refused = definition.judge(simplified) if len(refused) != 0: @@ -1242,9 +1361,7 @@ def _canonical_field(definition: Definition[Any], resolved: Resolved[Any]) -> JS carrying = _carrying_name(definition, simplified) if carrying is not None: return carrying - if len(simplified) != 0: - return {"name": name, "configuration": simplified} - return name if isinstance(definition, DataTypeDefinition) else {"name": name} + return _envelope_json(resolved.read_as, name, simplified) def _replaced(value: JSONValue, path: Loc, new: JSONValue) -> JSONValue: @@ -1278,13 +1395,14 @@ def _replaced(value: JSONValue, path: Loc, new: JSONValue) -> JSONValue: "EmptyConfiguration", "Lengths", "Nested", - "Resolution", + "Read", + "Refused", "Resolved", "StaticCodecField", "StorageClass", "StorageTransformerDefinition", "StorageTransformerField", - "Unread", + "Unclaimed", "as_kind", "asked", "canonicalize", diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/_pipeline.py b/packages/zarr-metadata/src/zarr_metadata/v3/_pipeline.py index 30772b3186..9435d71d6e 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/_pipeline.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/_pipeline.py @@ -31,7 +31,15 @@ from typing import TYPE_CHECKING, Any, Final, TypeGuard, cast from zarr_metadata._json import ValidationProblem -from zarr_metadata.v3._definition import Chunk, CodecDefinition, CodecKind, Resolved, asked, ruled +from zarr_metadata.v3._definition import ( + Chunk, + CodecDefinition, + CodecKind, + Read, + Resolved, + asked, + ruled, +) if TYPE_CHECKING: from collections.abc import Iterator, Sequence @@ -95,7 +103,7 @@ def read_pipeline( # array -- past the array -> bytes codec, or a codec of unknown kind. handed: Chunk | None = chunk for index, codec in enumerate(codecs): - definition, configuration = codec.definition, codec.configuration + definition = codec.definition if definition is None: stages.append(Stage(codec, handed)) handed = None @@ -107,17 +115,18 @@ def read_pipeline( incoming = Chunk() if handed is None else handed at = (*loc, index, "configuration") inner = _no_stages() - if configuration is not None: - problems.extend(_chunk_problems(definition, configuration, codec.nested, incoming, at)) - inner, found = _inner_pipelines(definition, configuration, codec.nested, incoming, at) + if isinstance(codec, Read): + configuration, nested = codec.configuration, codec.nested + problems.extend(_chunk_problems(definition, configuration, nested, incoming, at)) + inner, found = _inner_pipelines(definition, configuration, nested, incoming, at) problems.extend(found) stages.append(Stage(codec, incoming, inner)) if definition.kind == "array_bytes": handed = None - elif configuration is None: + elif not isinstance(codec, Read): handed = Chunk() else: - handed = _handed_on(definition, configuration, codec.nested, incoming, at) + handed = _handed_on(definition, codec.configuration, codec.nested, incoming, at) return tuple(stages), tuple(problems) diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/_registry.py b/packages/zarr-metadata/src/zarr_metadata/v3/_registry.py index 547523d7f9..d7475cd1c7 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/_registry.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/_registry.py @@ -11,7 +11,7 @@ uses nothing an implementation could refuse for being optional. `CORE_AND_EXTENSIONS` adds what `zarr-extensions` registers and this package defines. A name in neither is not refused -- that is what keeps -the format open -- it is read as `out_of_scope` and left unjudged. +the format open -- it is read as `Unclaimed` and left unjudged. """ from __future__ import annotations @@ -57,6 +57,8 @@ from zarr_metadata.v3.data_type.uint64 import UINT64_DATA_TYPE if TYPE_CHECKING: + from collections.abc import Callable + from zarr_metadata.v3._definition import D Tables = Mapping[type[Definition[Any]], Mapping[str, Definition[Any]]] @@ -118,6 +120,18 @@ def __repr__(self) -> str: # every definition's, and `help` of a validator runs to pages. return f"Context(<{len(self.definitions())} definitions>)" + def __reduce__(self) -> tuple[Callable[..., Context], tuple[Definition[Any], ...]]: + # A scope is its definitions, so it pickles as them, and goes to + # another process with the documents it is to read there. + return (Context.of, self.definitions()) + + def __copy__(self) -> Context: + return self + + def __deepcopy__(self, memo: dict[int, object]) -> Context: + # A scope never changes, so a copy of it is itself. + return self + def claimant(self, kind: type[D], name: str) -> D | None: """The definition of `kind` in scope that reads `name`, a name a document writes; None if none does. diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/codec/_arithmetic.py b/packages/zarr-metadata/src/zarr_metadata/v3/codec/_arithmetic.py index 47dd9d27f1..4cf96bb904 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/codec/_arithmetic.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/codec/_arithmetic.py @@ -17,9 +17,9 @@ which both codecs do, while neither lists them. """ -from typing import Any, Final +from typing import Final -from zarr_metadata.v3._definition import RAW_BYTES_NAME, Resolved +from zarr_metadata.v3._definition import RAW_BYTES_NAME from zarr_metadata.v3.data_type.bool import BOOL_DATA_TYPE_NAME from zarr_metadata.v3.data_type.bytes import BYTES_DATA_TYPE_NAME from zarr_metadata.v3.data_type.complex64 import COMPLEX64_DATA_TYPE_NAME @@ -67,11 +67,4 @@ """ -def read_name(data_type: Resolved[Any] | None) -> str | None: - """The name the definition that read `data_type` is filed under; None when no definition read it.""" - if data_type is None or data_type.resolution != "read" or data_type.definition is None: - return None - return data_type.definition.name - - -__all__ = ["COMPLEX", "FLOATING_POINT", "NOT_NUMBERS", "read_name"] +__all__ = ["COMPLEX", "FLOATING_POINT", "NOT_NUMBERS"] diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/codec/bytes.py b/packages/zarr-metadata/src/zarr_metadata/v3/codec/bytes.py index 89b72df98e..cbd3e5cdf3 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/codec/bytes.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/codec/bytes.py @@ -14,7 +14,6 @@ Chunk, CodecDefinition, Nested, - named_configuration, storage_of, ) @@ -91,7 +90,7 @@ def _chunk_rules( if chunk.data_type is None: return storage = storage_of(chunk.data_type) - written, _, _ = named_configuration(chunk.data_type.json) + written = chunk.data_type.name if storage == "multi_byte" and "endian" not in configuration: yield ValidationProblem( ("endian",), diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/codec/cast_value.py b/packages/zarr-metadata/src/zarr_metadata/v3/codec/cast_value.py index 14be64d638..1580074b76 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/codec/cast_value.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/codec/cast_value.py @@ -16,15 +16,10 @@ CodecDefinition, DataTypeField, Nested, + Read, fill_value_problems, - named_configuration, -) -from zarr_metadata.v3.codec._arithmetic import ( - COMPLEX, - FLOATING_POINT, - NOT_NUMBERS, - read_name, ) +from zarr_metadata.v3.codec._arithmetic import COMPLEX, FLOATING_POINT, NOT_NUMBERS CAST_VALUE_CODEC_NAME: Final = "cast_value" """The `name` field value of the `cast_value` codec.""" @@ -153,10 +148,9 @@ def _rules( models no real numbers is the one problem reported of it. """ target = nested.get(("data_type",)) - name = read_name(target) - if target is None or name is None: + if not isinstance(target, Read): return - written, _, _ = named_configuration(target.json) + name, written = target.definition.name, target.name if name in _NO_REAL_NUMBERS: yield ValidationProblem( ("data_type",), @@ -185,14 +179,12 @@ def _chunk_rules( so the data type it is handed is held to what the one it casts to is. """ source = chunk.data_type - name = read_name(source) - if source is None or name is None: + if not isinstance(source, Read): return - if name in _NO_REAL_NUMBERS: - written, _, _ = named_configuration(source.json) + if source.definition.name in _NO_REAL_NUMBERS: yield ValidationProblem( (), - f"expected a chunk of a data type that models real numbers, got {shown(written)}", + f"expected a chunk of a data type that models real numbers, got {shown(source.name)}", "invalid_value", ) return diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/codec/scale_offset.py b/packages/zarr-metadata/src/zarr_metadata/v3/codec/scale_offset.py index d6bcb854b8..72e6d50e14 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/codec/scale_offset.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/codec/scale_offset.py @@ -15,10 +15,10 @@ Chunk, CodecDefinition, Nested, + Read, fill_value_problems, - named_configuration, ) -from zarr_metadata.v3.codec._arithmetic import NOT_NUMBERS, read_name +from zarr_metadata.v3.codec._arithmetic import NOT_NUMBERS SCALE_OFFSET_CODEC_NAME: Final = "scale_offset" """The `name` field value of the `scale_offset` codec.""" @@ -95,14 +95,12 @@ def _chunk_rules( A null is the rules' to refuse. """ source = chunk.data_type - name = read_name(source) - if source is None or name is None: + if not isinstance(source, Read): return - if name in NOT_NUMBERS: - written, _, _ = named_configuration(source.json) + if source.definition.name in NOT_NUMBERS: yield ValidationProblem( (), - f"expected a chunk of a data type with arithmetic, got {shown(written)}", + f"expected a chunk of a data type with arithmetic, got {shown(source.name)}", "invalid_value", ) return diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/codec/sharding_indexed.py b/packages/zarr-metadata/src/zarr_metadata/v3/codec/sharding_indexed.py index 8b856e7729..b7d42aa1e1 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/codec/sharding_indexed.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/codec/sharding_indexed.py @@ -14,10 +14,9 @@ Chunk, CodecDefinition, CodecField, - DataTypeDefinition, Lengths, Nested, - Resolved, + Read, StaticCodecField, ) from zarr_metadata.v3.data_type.uint64 import UINT64_DATA_TYPE, UINT64_DATA_TYPE_NAME @@ -126,8 +125,11 @@ def _chunk_rules( ) -_UINT64: Final = Resolved( - UINT64_DATA_TYPE_NAME, "read", UINT64_DATA_TYPE, {}, read_as=DataTypeDefinition +_UINT64: Final = Read( + json=UINT64_DATA_TYPE_NAME, + name=UINT64_DATA_TYPE_NAME, + definition=UINT64_DATA_TYPE, + configuration={}, ) """The data type of a shard index, which the spec fixes whatever the scope holds.""" diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/_float.py b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/_float.py index 444d13a422..3f7b56b5da 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/_float.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/_float.py @@ -9,8 +9,8 @@ (https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v3/data-types/index.rst#L88-L91). """ -import functools from collections.abc import Callable, Iterable, Iterator +from dataclasses import dataclass from typing import Any, Literal, get_args from zarr_metadata._json import ValidationProblem, choices, shown @@ -28,30 +28,32 @@ def float_fill_value_rules( A number takes any value, since a reader rounds it. A string is one of the named values, or a hex string of the type's own width: the type-checked shape takes any string there, and `hex_form` raises - `ValueError` for one that is not. A partial application of a - module-level function, so a definition holding it pickles. + `ValueError` for one that is not. """ - return functools.partial(_float_fill_value, name, hex_form) - - -def _float_fill_value( - name: str, - hex_form: Callable[[str], str], - configuration: EmptyConfiguration, - nested: Nested, - value: float | str, -) -> Iterator[ValidationProblem]: - if not isinstance(value, str) or value in get_args(FloatSpecialFillValue): - return - try: - hex_form(value) - except ValueError: - yield ValidationProblem( - (), - f"expected a number, {choices(get_args(FloatSpecialFillValue))}, or a {name} " - f"hex string, got {shown(value)}", - "invalid_value", - ) + return _FloatFillValue(name, hex_form) + + +@dataclass(frozen=True, slots=True) +class _FloatFillValue: + """A float fill value: a value rather than a closure, so a definition holding it is equal to itself after a pickle or a deep copy.""" + + name: str + hex_form: Callable[[str], str] + + def __call__( + self, configuration: EmptyConfiguration, nested: Nested, value: float | str + ) -> Iterator[ValidationProblem]: + if not isinstance(value, str) or value in get_args(FloatSpecialFillValue): + return + try: + self.hex_form(value) + except ValueError: + yield ValidationProblem( + (), + f"expected a number, {choices(get_args(FloatSpecialFillValue))}, or a " + f"{self.name} hex string, got {shown(value)}", + "invalid_value", + ) def complex_fill_value_rules( @@ -63,18 +65,24 @@ def complex_fill_value_rules( `component` is the fill value rules of the component's own float type. """ - return functools.partial(_complex_fill_value, component) + return _ComplexFillValue(component) -def _complex_fill_value( - component: Callable[[EmptyConfiguration, Nested, Any], Iterable[ValidationProblem]], - configuration: EmptyConfiguration, - nested: Nested, - value: tuple[float | str, float | str], -) -> Iterator[ValidationProblem]: - for index, part in enumerate(value): - for found in component(configuration, nested, part): - yield ValidationProblem((index, *found.loc), found.message, found.kind) +@dataclass(frozen=True, slots=True) +class _ComplexFillValue: + """A complex fill value, each component judged at its index: a value rather than a closure, as `_FloatFillValue` is.""" + + component: Callable[[EmptyConfiguration, Nested, Any], Iterable[ValidationProblem]] + + def __call__( + self, + configuration: EmptyConfiguration, + nested: Nested, + value: tuple[float | str, float | str], + ) -> Iterator[ValidationProblem]: + for index, part in enumerate(value): + for found in self.component(configuration, nested, part): + yield ValidationProblem((index, *found.loc), found.message, found.kind) __all__ = ["FloatSpecialFillValue", "complex_fill_value_rules", "float_fill_value_rules"] diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/_integer.py b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/_integer.py index 4bf1005fd5..0a8320ee19 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/_integer.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/_integer.py @@ -5,8 +5,8 @@ (https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v3/data-types/index.rst#L59-L61). """ -import functools from collections.abc import Callable, Iterator +from dataclasses import dataclass from zarr_metadata._json import ValidationProblem from zarr_metadata.v3._definition import EmptyConfiguration, Nested @@ -15,21 +15,26 @@ def integer_fill_value_rules( low: int, high: int ) -> Callable[[EmptyConfiguration, Nested, int], Iterator[ValidationProblem]]: - """The fill value rules of an integer data type whose range is `[low, high]`. + """The fill value rules of an integer data type whose range is `[low, high]`.""" + return _InRange(low, high) - A partial application of a module-level function, so a definition - holding it pickles. - """ - return functools.partial(_in_range, low, high) +@dataclass(frozen=True, slots=True) +class _InRange: + """An integer in `[low, high]`: a value rather than a closure, so a definition holding it is equal to itself after a pickle or a deep copy.""" -def _in_range( - low: int, high: int, configuration: EmptyConfiguration, nested: Nested, value: int -) -> Iterator[ValidationProblem]: - if not low <= value <= high: - yield ValidationProblem( - (), f"expected an integer in [{low}, {high}], got {value}", "invalid_value" - ) + low: int + high: int + + def __call__( + self, configuration: EmptyConfiguration, nested: Nested, value: int + ) -> Iterator[ValidationProblem]: + if not self.low <= value <= self.high: + yield ValidationProblem( + (), + f"expected an integer in [{self.low}, {self.high}], got {value}", + "invalid_value", + ) def byte_value_problems(values: tuple[int, ...]) -> Iterator[ValidationProblem]: diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/struct.py b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/struct.py index a8dc844c33..140f8ae847 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/struct.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/struct.py @@ -17,7 +17,6 @@ Nested, StorageClass, fill_value_problems, - named_configuration, storage_of, ) @@ -96,10 +95,10 @@ def _rules(configuration: StructConfiguration, nested: Nested) -> Iterator[Valid ) field_type = nested.get(("fields", index, "data_type")) if field_type is not None and storage_of(field_type) == "variable_length": - written, _, _ = named_configuration(field_type.json) yield ValidationProblem( ("fields", index, "data_type"), - f"expected a data type of fixed size, got {shown(written)}, whose values vary in size", + f"expected a data type of fixed size, got {shown(field_type.name)}, whose values " + "vary in size", "invalid_value", ) diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/definition.py b/packages/zarr-metadata/src/zarr_metadata/v3/definition.py index 33b273e482..8ce2f400bc 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/definition.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/definition.py @@ -38,13 +38,20 @@ 3. `resolve(field, CodecDefinition, CORE_AND_EXTENSIONS)` reads a whole field in a scope: its envelope judged, its name related to a definition, its configuration judged, and each nested field read the - same way. It returns `Resolved` -- the field's JSON and the `name` - it was written with, its `resolution`, the definition, the checked - configuration and the kind it was read as, `read_as` -- and every - problem. A name nothing in scope - claims is `out_of_scope`: left unjudged, which is what keeps the - format open, and still read as the kind it was asked for. A field is read - as one of the five kinds, with or without type arguments; + same way. It returns what the scope made of the field, and every + problem: `Read` by the definition that claims its name, with the + configuration it checked and allowed and the fields it holds, each + read the same way; `Unclaimed`, a name nothing in scope claims, left + unjudged, which is what keeps the format open; or `Refused`, whose + problems say why. Each has the field's JSON, the `name` it is + written with, the kind it was read as, `read_as`, the `definition` + that claims its name -- None for `Unclaimed` -- and the fields it + holds as the scope read them, `nested`. The two a model holds, `Read` + and `Unclaimed`, have a `configuration` and `to_json()`, the field as + a document writes it; two of them are equal when they read the same, + however each was spelled. `Resolved` is the three, for `match`. A + field is read as one of the five kinds, with or without type + arguments; `resolve(field, Definition, scope)` is a `TypeError`, since nothing is filed under it. `configuration_of(resolved, GZIP_CODEC)` is the configuration typed as that definition's TypedDict, when it read it. @@ -62,7 +69,7 @@ resolved, problems = resolve({"name": "gzip", "configuration": {"level": 12}}, CodecDefinition, CORE_AND_EXTENSIONS) - resolved.resolution # 'invalid' + resolved # Refused(..., definition=CodecDefinition(name='gzip'), ...) problems[0].loc # ('configuration', 'level') resolved, problems = resolve({"name": "gzip", "configuration": {"level": 5}}, @@ -82,10 +89,14 @@ which `typing.TypedDict` does not take on the versions this package supports. The rules are handed the configuration and the fields it holds as the scope read them: a field that is read keeps what it read inside it -as `Resolved.nested`, a `Nested` mapping by where each sits, so a -struct's rules reach its field types. `judge`, which reads in no scope, +as `Read.nested`, a `Nested` mapping by where each sits, so a struct's +rules reach its field types. `judge`, which reads in no scope, hands them none. A rule's message shows a value as the package's own -messages do, as JSON, with `shown`: `null`, `[1, 2]`, `"C"`. +messages do, as JSON, with `shown`: `null`, `[1, 2]`, `"C"`. Define each +function at a module's top level: a model holds the definitions that +read its fields, so it pickles, and compares equal once loaded, only +when they do -- a lambda or a closure does not pickle, and a +`functools.partial` pickles but compares unequal to itself loaded. from collections.abc import Iterator @@ -257,13 +268,14 @@ class creation. EmptyConfiguration, Lengths, Nested, - Resolution, + Read, + Refused, Resolved, StaticCodecField, StorageClass, StorageTransformerDefinition, StorageTransformerField, - Unread, + Unclaimed, canonicalize, chunk_grid_lengths, configuration_of, @@ -298,14 +310,15 @@ class creation. "MetadataValidationError", "Nested", "ProblemKind", - "Resolution", + "Read", + "Refused", "Resolved", "Stage", "StaticCodecField", "StorageClass", "StorageTransformerDefinition", "StorageTransformerField", - "Unread", + "Unclaimed", "ValidationProblem", "ZarrV3MetadataFieldJSON", "canonicalize", diff --git a/packages/zarr-metadata/tests/model/test_array.py b/packages/zarr-metadata/tests/model/test_array.py index 161c909469..9083fee2d0 100644 --- a/packages/zarr-metadata/tests/model/test_array.py +++ b/packages/zarr-metadata/tests/model/test_array.py @@ -24,9 +24,6 @@ ZarrV2ArrayMetadata, ZarrV2ArrayMetadataPartial, ZarrV3ArrayMetadata, - ZarrV3ArrayMetadataPartial, - ZarrV3MetadataField, - ZarrV3NamedConfig, is_array_metadata_v2, is_array_metadata_v3, is_group_metadata_v2, @@ -42,9 +39,12 @@ validate_json, validate_metadata_field_v3, ) +from zarr_metadata.v3.array import ZarrV3ArrayMetadataJSONPartial +from zarr_metadata.v3.codec.gzip import GZIP_CODEC +from zarr_metadata.v3.definition import CORE, CORE_AND_EXTENSIONS, Read, Unclaimed, configuration_of if TYPE_CHECKING: - from zarr_metadata._common import JSONValue + from zarr_metadata._common import JSONValue, ZarrV3NamedConfigJSON from zarr_metadata.v2 import ZarrV2CodecMetadata # --- public exports -------------------------------------------------------- @@ -188,81 +188,18 @@ def test_string_nan_fill_value_roundtrips() -> None: # The string form round-trips cleanly under default dataclass equality, # unlike a raw float('nan'), which is not JSON. """A float array's string 'NaN' fill_value round-trips cleanly.""" - m = ZarrV3ArrayMetadata.create_default( - fill_value="NaN", - data_type=ZarrV3NamedConfig(name="float32", configuration={}), - codecs=(ZarrV3NamedConfig(name="bytes", configuration={"endian": "little"}),), - ) + m = ZarrV3ArrayMetadata.create_default(fill_value="NaN", data_type="float32") assert ZarrV3ArrayMetadata.from_json(m.to_json()) == m assert ZarrV3ArrayMetadata.from_json(m.to_json()).fill_value == "NaN" -# --- ZarrV3NamedConfig.to_json ------------------------------------------------ - -ZARR_TO_JSON_CASES = [ - Expect( - ZarrV3NamedConfig(name="regular", configuration={"chunk_shape": [1]}), - {"name": "regular", "configuration": {"chunk_shape": [1]}}, - id="with-configuration", - ), - Expect( - ZarrV3NamedConfig(name="bytes", configuration={}), - "bytes", - id="empty-configuration-shorthand", - ), -] - - -@pytest.mark.parametrize("case", ZARR_TO_JSON_CASES, ids=lambda c: c.id) -def test_zarr_metadata_v3_to_json(case: Expect[ZarrV3NamedConfig, object]) -> None: - """ZarrV3NamedConfig.to_json emits the canonical extension form.""" - assert case.input.to_json() == case.output - - -def test_zarr_metadata_v3_to_json_preserves_false_obligation() -> None: - """An empty optional extension stays an object so false is not lost.""" - model = ZarrV3NamedConfig(name="optional", configuration={}, must_understand=False) - assert model.to_json() == {"name": "optional", "must_understand": False} - - -# --- ZarrV3NamedConfig.from_json ----------------------------------------------- - -ZARR_FROM_JSON_CASES = [ - Expect("bytes", ZarrV3NamedConfig(name="bytes", configuration={}), id="bare-string"), - Expect( - {"name": "regular", "configuration": {"chunk_shape": [1]}}, - ZarrV3NamedConfig(name="regular", configuration={"chunk_shape": (1,)}), - id="object-with-config", - ), - Expect( - {"name": "bytes"}, - ZarrV3NamedConfig(name="bytes", configuration={}), - id="object-without-config", - ), -] - - -@pytest.mark.parametrize("case", ZARR_FROM_JSON_CASES, ids=lambda c: c.id) -def test_zarr_metadata_v3_from_json(case: Expect[object, ZarrV3NamedConfig]) -> None: - """ZarrV3NamedConfig.from_json parses both the bare-string and object forms.""" - assert ZarrV3NamedConfig.from_json(case.input) == case.output - - -def test_zarr_metadata_v3_from_json_preserves_false_obligation() -> None: - """Explicit false is represented on the normalized model.""" - model = ZarrV3NamedConfig.from_json({"name": "optional", "must_understand": False}) - assert model.must_understand is False - - # --- V3 baseline ----------------------------------------------------------- def test_v3_to_json_emits_canonical_document() -> None: """V3 to_json emits exactly the expected document (which covers every spec-required key by construction).""" - out = ZarrV3ArrayMetadata.create_default( - shape=(10,), data_type=ZarrV3NamedConfig(name="int32", configuration={}) - ).to_json() + out = ZarrV3ArrayMetadata.create_default(shape=(10,), data_type="int32").to_json() assert out == { "zarr_format": 3, "node_type": "array", @@ -270,7 +207,7 @@ def test_v3_to_json_emits_canonical_document() -> None: "fill_value": 0, "data_type": "int32", "chunk_grid": {"name": "regular", "configuration": {"chunk_shape": (10,)}}, - "codecs": ({"name": "bytes"},), + "codecs": ({"name": "bytes", "configuration": {"endian": "little"}},), "chunk_key_encoding": {"name": "default"}, } @@ -299,15 +236,16 @@ def test_v3_a_data_type_with_nothing_to_configure_is_written_by_its_bare_name( def test_v3_dimension_names_included_when_present() -> None: """V3 to_json includes dimension_names when they are set.""" out: dict[str, object] = dict( - ZarrV3ArrayMetadata.create_default(dimension_names=("x",)).to_json() + ZarrV3ArrayMetadata.create_default(shape=(4,), dimension_names=("x",)).to_json() ) assert out["dimension_names"] == ("x",) def test_v3_dimension_names_omitted_when_none() -> None: """V3 to_json omits dimension_names when they are UNSET.""" - out = ZarrV3ArrayMetadata.create_default(dimension_names=UNSET).to_json() - assert "dimension_names" not in out + model = ZarrV3ArrayMetadata.create_default() + assert model.dimension_names is UNSET + assert "dimension_names" not in model.to_json() # --- BUG 1: attributes gated on dimension_names ---------------------------- @@ -320,11 +258,9 @@ def test_v3_attributes_included_when_dimension_names_is_none() -> None: so non-empty attributes were silently dropped when there were no dimension names. """ - out: dict[str, object] = dict( - ZarrV3ArrayMetadata.create_default( - dimension_names=UNSET, attributes={"foo": "bar"} - ).to_json() - ) + model = ZarrV3ArrayMetadata.create_default(attributes={"foo": "bar"}) + assert model.dimension_names is UNSET + out: dict[str, object] = dict(model.to_json()) assert out["attributes"] == {"foo": "bar"} @@ -337,7 +273,7 @@ def test_v3_single_storage_transformer_included() -> None: Regression: the guard used ``> 1`` instead of ``> 0``, dropping a lone storage transformer. """ - st = ZarrV3NamedConfig(name="some_transformer", configuration={}) + st: ZarrV3NamedConfigJSON = {"name": "some_transformer"} out: dict[str, object] = dict( ZarrV3ArrayMetadata.create_default(storage_transformers=(st,)).to_json() ) @@ -355,16 +291,18 @@ def test_v3_no_storage_transformers_omitted() -> None: def test_v3_extra_fields_merged() -> None: """V3 to_json merges extra_fields into the top-level document.""" - out = ZarrV3ArrayMetadata.create_default( - extra_fields={"my_ext": {"must_understand": False}} - ).to_json() - assert out["my_ext"] == {"must_understand": False} + model = ZarrV3ArrayMetadata.create_default(my_ext={"must_understand": False}) + assert model.extra_fields == {"my_ext": {"must_understand": False}} + assert model.to_json()["my_ext"] == {"must_understand": False} def test_v3_extra_fields_overlapping_standard_field_rejected() -> None: """Constructing a V3 model with an extra field that collides with a standard key is rejected.""" with pytest.raises(ValueError): - ZarrV3ArrayMetadata.create_default(extra_fields={"shape": {"must_understand": False}}) + dataclasses.replace( + ZarrV3ArrayMetadata.create_default(), + extra_fields={"shape": {"must_understand": False}}, + ) # --- V3 key/value ---------------------------------------------------------- @@ -408,7 +346,7 @@ def test_v3_create_default_is_valid_empty_array() -> None: """V3 create_default builds a structurally valid empty array that round-trips.""" m = ZarrV3ArrayMetadata.create_default() assert m.shape == () - assert m.data_type == ZarrV3NamedConfig(name="uint8", configuration={}) + assert m.data_type.to_json() == "uint8" assert m.fill_value == 0 assert m.attributes == {} assert m.extra_fields == {} @@ -423,7 +361,7 @@ def test_v3_create_default_applies_overrides() -> None: assert m.shape == (4, 4) assert m.attributes == {"a": 1} # un-overridden fields keep their defaults - assert m.data_type == ZarrV3NamedConfig(name="uint8", configuration={}) + assert m.data_type.to_json() == "uint8" def test_v2_create_default_is_valid_empty_array() -> None: @@ -463,7 +401,11 @@ def test_update_returns_new_instance( ) -> None: """update returns a new instance with the field replaced, leaving the original unchanged.""" base = model_cls.create_default(shape=(10,)) - updated = base.update(shape=(20,)) + updated = ( + base.update(context=CORE_AND_EXTENSIONS, shape=(20,)) + if isinstance(base, ZarrV3ArrayMetadata) + else base.update(shape=(20,)) + ) assert updated.shape == (20,) assert base.shape == (10,) # original unchanged assert isinstance(updated, model_cls) @@ -481,34 +423,121 @@ def test_update_no_args_returns_equal_model( ) -> None: """update with no arguments returns a model equal to the original.""" base = model_cls.create_default() - assert base.update() == base + updated = ( + base.update(context=CORE_AND_EXTENSIONS) + if isinstance(base, ZarrV3ArrayMetadata) + else base.update() + ) + assert updated == base # V3-only update tests — kept direct (extra_fields is v3-specific) -def test_update_can_replace_extra_fields() -> None: - """update can replace the extra_fields mapping.""" - base = ZarrV3ArrayMetadata.create_default(extra_fields={}) - updated = base.update(extra_fields={"my_ext": {"must_understand": False}}) +def test_update_can_add_an_extension_member() -> None: + """update can add a member the spec does not define, which the model holds in extra_fields.""" + base = ZarrV3ArrayMetadata.create_default() + updated = base.update(context=CORE_AND_EXTENSIONS, my_ext={"must_understand": False}) assert updated.extra_fields == {"my_ext": {"must_understand": False}} -def test_update_replaces_extra_fields_rather_than_merging() -> None: - """update replaces extra_fields wholesale rather than merging.""" - base = ZarrV3ArrayMetadata.create_default(extra_fields={"a": {"must_understand": False}}) - updated = base.update(extra_fields={"b": {"must_understand": True}}) - assert updated.extra_fields == {"b": {"must_understand": True}} +def test_update_reads_only_the_members_it_is_given_in_its_scope() -> None: + """Each field the model holds is kept as it was read; the members given are read in `context`. + `zstd` is an extension, which `CORE` leaves unclaimed and unjudged. + """ + little: ZarrV3NamedConfigJSON = {"name": "bytes", "configuration": {"endian": "little"}} + zstd: ZarrV3NamedConfigJSON = {"name": "zstd", "configuration": {"level": 3, "checksum": False}} + base = ZarrV3ArrayMetadata.create_default(context=CORE, codecs=(little, zstd)) + assert isinstance(base.codecs[1], Unclaimed) + kept = base.update(context=CORE_AND_EXTENSIONS, attributes={"k": 1}) + assert kept.codecs[1] is base.codecs[1] + given = base.update(context=CORE_AND_EXTENSIONS, codecs=(little, zstd)) + assert isinstance(given.codecs[1], Read) + + +def test_update_leaves_out_a_member_given_as_unset() -> None: + """Each member a document may leave out: `dimension_names`, `attributes`, one the spec does not define.""" + base = ZarrV3ArrayMetadata.create_default( + shape=(2,), dimension_names=("x",), attributes={"a": 1}, my_ext={"must_understand": False} + ) + for updated, member in ( + (base.update(context=CORE_AND_EXTENSIONS, dimension_names=UNSET), "dimension_names"), + (base.update(context=CORE_AND_EXTENSIONS, attributes=UNSET), "attributes"), + (base.update(context=CORE_AND_EXTENSIONS, my_ext=UNSET), "my_ext"), + ): + written = updated.to_json() + assert member not in written + assert {key: value for key, value in base.to_json().items() if key != member} == written -def test_partial_keys_match_settable_model_fields() -> None: - """The partial TypedDict must list exactly the constructor-settable fields. - Guards against drift: adding/removing a settable field on the model - without updating ``ZarrV3ArrayMetadataPartial`` fails here. - """ - settable = {f.name for f in dataclasses.fields(ZarrV3ArrayMetadata) if f.init} - assert set(ZarrV3ArrayMetadataPartial.__annotations__) == settable +@pytest.mark.parametrize( + "members", + [ + {"data_type": {"name": "uint8"}}, + {"codecs": ("bytes",)}, + {"codecs": ({"name": "bytes", "configuration": {}, "must_understand": True},)}, + {"chunk_key_encoding": {"name": "default", "configuration": {}}}, + {"codecs": ({"name": "bytes"}, {"name": "acme.codec", "configuration": {}})}, + { + "shape": (4,), + "codecs": ( + { + "name": "sharding_indexed", + "configuration": { + "chunk_shape": (2,), + "codecs": ({"name": "bytes"},), + "index_codecs": ( + {"name": "bytes", "configuration": {"endian": "little"}}, + "crc32c", + ), + }, + }, + ), + }, + ], + ids=[ + "data-type-object", + "codec-bare", + "codec-verbose", + "empty-configuration", + "unclaimed", + "a-field-a-codec-holds-bare", + ], +) +def test_a_model_is_what_its_document_says_not_how_it_is_spelled( + members: ZarrV3ArrayMetadataJSONPartial, +) -> None: + """A model reads back from its own document as itself, and equals the model of any spelling of it.""" + model = ZarrV3ArrayMetadata.create_default(**members) + assert ZarrV3ArrayMetadata.from_json(model.to_json()) == model + written = {**model.to_json(), **members} + assert ZarrV3ArrayMetadata.from_json(written) == model + + +def test_a_model_pickles_and_copies_with_its_definitions() -> None: + model = ZarrV3ArrayMetadata.create_default( + codecs=( + {"name": "bytes", "configuration": {"endian": "little"}}, + {"name": "gzip", "configuration": {"level": 1}}, + ), + context=CORE, + ) + for again in (pickle.loads(pickle.dumps(model)), copy.copy(model), copy.deepcopy(model)): + assert again == model + assert configuration_of(again.codecs[1], GZIP_CODEC) == {"level": 1} + + +def test_update_replaces_a_member_rather_than_merging_into_it() -> None: + """update replaces each member it is given whole, and keeps the others.""" + base = ZarrV3ArrayMetadata.create_default( + a={"must_understand": False, "x": 1}, b={"must_understand": False} + ) + updated = base.update(context=CORE_AND_EXTENSIONS, a={"must_understand": False}) + assert updated.extra_fields == { + "a": {"must_understand": False}, + "b": {"must_understand": False}, + } # --- V2 model -------------------------------------------------------------- @@ -576,22 +605,17 @@ def test_arrays_to_tuples(case: Expect[object, object]) -> None: def test_v3_from_json_reconstructs_required_fields() -> None: """V3 from_json reconstructs the required fields from a document.""" doc = ZarrV3ArrayMetadata.create_default( - shape=(7,), - attributes={"a": 1}, - data_type=ZarrV3NamedConfig(name="int32", configuration={}), - codecs=(ZarrV3NamedConfig(name="bytes", configuration={"endian": "little"}),), + shape=(7,), attributes={"a": 1}, data_type="int32" ).to_json() model = ZarrV3ArrayMetadata.from_json(doc) assert model.shape == (7,) - assert model.data_type == ZarrV3NamedConfig(name="int32", configuration={}) + assert model.data_type.to_json() == "int32" assert model.attributes == {"a": 1} def test_v3_from_json_defaults_for_omitted_optionals() -> None: """V3 from_json supplies defaults for omitted optional fields.""" - doc = ZarrV3ArrayMetadata.create_default( - attributes={}, storage_transformers=(), dimension_names=UNSET - ).to_json() + doc = ZarrV3ArrayMetadata.create_default(attributes={}, storage_transformers=()).to_json() # to_json omits these entirely; from_json must restore defaults model = ZarrV3ArrayMetadata.from_json(doc) assert model.attributes == {} @@ -601,9 +625,7 @@ def test_v3_from_json_defaults_for_omitted_optionals() -> None: def test_v3_from_json_routes_unknown_keys_to_extra_fields() -> None: """V3 from_json routes unknown top-level keys into extra_fields.""" - doc = ZarrV3ArrayMetadata.create_default( - extra_fields={"my_ext": {"must_understand": False}} - ).to_json() + doc = ZarrV3ArrayMetadata.create_default(my_ext={"must_understand": False}).to_json() model = ZarrV3ArrayMetadata.from_json(doc) assert model.extra_fields == {"my_ext": {"must_understand": False}} @@ -669,19 +691,14 @@ def test_from_key_value_missing_key_raises( shape=(10,), attributes={"a": 1}, dimension_names=("x",), - storage_transformers=(ZarrV3NamedConfig(name="t", configuration={}),), - extra_fields={"ext": {"must_understand": False}}, + storage_transformers=({"name": "t"},), + ext={"must_understand": False}, ), id="v3-full", ), pytest.param( ZarrV3ArrayMetadata, - ZarrV3ArrayMetadata.create_default( - attributes={}, - dimension_names=UNSET, - storage_transformers=(), - extra_fields={}, - ), + ZarrV3ArrayMetadata.create_default(attributes={}, storage_transformers=()), id="v3-empty-optionals", ), pytest.param( @@ -738,10 +755,10 @@ def test_v2_roundtrip_json_model_json() -> None: ZarrV3ArrayMetadata.create_default( shape=(2,), attributes={"a": {"b": [1]}}, - # A name nothing in the scope claims, so reading it back judges only + # A name nothing in the scope claims, so reading it judges only # the document, and its configuration can nest. - codecs=(ZarrV3NamedConfig(name="acme.nested", configuration={"opts": {"level": 1}}),), - extra_fields={"ext": {"must_understand": False, "cfg": {"x": [1]}}}, + codecs=({"name": "acme.nested", "configuration": {"opts": {"level": 1}}},), + ext={"must_understand": False, "cfg": {"x": [1]}}, ), id="v3", ), @@ -787,7 +804,7 @@ def test_v3_parser_accepts_bare_string_data_type() -> None: doc["data_type"] = "int32" doc["codecs"] = ({"name": "bytes", "configuration": {"endian": "little"}},) model = ZarrV3ArrayMetadata.from_json(doc) - assert model.data_type == ZarrV3NamedConfig(name="int32", configuration={}) + assert (model.data_type.json, model.data_type.name) == ("int32", "int32") assert model.to_json()["data_type"] == "int32" @@ -1112,7 +1129,7 @@ def test_parse_metadata_field_v3( # invalid cases (a subset check, so accumulation of OTHER problems is allowed). -def _build_v3(**overrides: Unpack[ZarrV3ArrayMetadataPartial]) -> dict[str, object]: +def _build_v3(**overrides: Unpack[ZarrV3ArrayMetadataJSONPartial]) -> dict[str, object]: return dict(ZarrV3ArrayMetadata.create_default(**overrides).to_json()) @@ -1145,7 +1162,7 @@ def _set(key: str, value: object) -> Callable[[dict], object]: id="valid-with-attributes-and-dim-names", ), Expect( - lambda: _build_v3(extra_fields={"my_ext": {"must_understand": False}}), + lambda: _build_v3(my_ext={"must_understand": False}), frozenset(), id="valid-with-extra-fields", ), @@ -1268,17 +1285,12 @@ def test_array_metadata_guards( ExpectFail(lambda: {"zarr_format": 2}, MetadataValidationError, id="x"), id="v2-missing-required", ), - pytest.param( - ZarrV3NamedConfig, - ExpectFail(lambda: 5, MetadataValidationError, id="x"), - id="zarr-metadata-bad-input", - ), ] @pytest.mark.parametrize(("model", "case"), FROM_JSON_REJECT_PARAMS) def test_from_json_rejects_malformed( - model: type[ZarrV3ArrayMetadata | ZarrV2ArrayMetadata | ZarrV3NamedConfig], + model: type[ZarrV3ArrayMetadata | ZarrV2ArrayMetadata], case: ExpectFail[Callable[[], object]], ) -> None: """from_json raises MetadataValidationError on a malformed document.""" @@ -1617,12 +1629,153 @@ def test_from_key_value_rejects_non_standard_json_constant() -> None: ZarrV3ArrayMetadata.from_key_value({"zarr.json": raw.encode()}) -def test_to_key_value_rejects_non_finite_model_value() -> None: - """Strict encoding prevents directly-constructed models from writing invalid JSON.""" - model = ZarrV3ArrayMetadata.create_default(fill_value=float("nan")) +@pytest.mark.parametrize( + ("members", "problems"), + [ + ({"fill_value": math.nan}, [(("fill_value",), "invalid_value")]), + # The default fill value, 0, is not a boolean: a data type goes with + # a fill value of it. + ({"data_type": "bool"}, [(("fill_value",), "invalid_type")]), + # Values of several bytes take an `endian`. + ( + {"data_type": "int16", "codecs": ({"name": "bytes"},)}, + [(("codecs", 0, "configuration", "endian"), "missing_key")], + ), + ({"shape": (2,), "dimension_names": ("x", "y")}, [(("dimension_names",), "invalid_value")]), + # The shape is not derived from a grid, which the scalar default + # shape does not fit: a grid goes with the shape it fits. + ( + {"chunk_grid": {"name": "regular", "configuration": {"chunk_shape": (10, 10)}}}, + [(("chunk_grid", "configuration", "chunk_shape"), "invalid_value")], + ), + ], + ids=[ + "non-finite-fill-value", + "fill-value-of-another-type", + "no-endian", + "names-past-shape", + "grid-of-another-rank", + ], +) +def test_error_create_default_refuses_a_document_with_a_problem( + members: ZarrV3ArrayMetadataJSONPartial, problems: list[tuple[tuple[str | int, ...], str]] +) -> None: + """A model comes from a read, so a document its read refuses makes none, and nothing is written.""" + with pytest.raises(MetadataValidationError) as raised: + ZarrV3ArrayMetadata.create_default(**members) + assert [(p.loc, p.kind) for p in raised.value.problems] == problems + - with pytest.raises(MetadataValidationError, match="fill_value: non-finite float nan"): +@pytest.mark.parametrize( + ("members", "problems"), + [ + ({"fill_value": 300}, [(("fill_value",), "invalid_value")]), + # A shape goes with a grid that fits it. + ({"shape": (4, 4)}, [(("chunk_grid", "configuration", "chunk_shape"), "invalid_value")]), + ], + ids=["fill-value-out-of-range", "shape-without-its-grid"], +) +def test_error_update_refuses_a_document_with_a_problem( + members: ZarrV3ArrayMetadataJSONPartial, problems: list[tuple[tuple[str | int, ...], str]] +) -> None: + base = ZarrV3ArrayMetadata.create_default(shape=(4,)) + with pytest.raises(MetadataValidationError) as raised: + base.update(context=CORE_AND_EXTENSIONS, **members) + assert [(p.loc, p.kind) for p in raised.value.problems] == problems + + +def test_error_update_refuses_to_leave_out_a_member_a_document_holds() -> None: + base = ZarrV3ArrayMetadata.create_default(shape=(4,)) + with pytest.raises(MetadataValidationError) as raised: + base.update(context=CORE_AND_EXTENSIONS, shape=UNSET) # pyright: ignore[reportArgumentType] + assert [(p.loc, p.kind) for p in raised.value.problems] == [(("shape",), "missing_key")] + + +@pytest.mark.parametrize( + ("changes", "problems"), + [ + ({"fill_value": math.nan}, [(("fill_value",), "invalid_value")]), + ({"dimension_names": ("x", "y")}, [(("dimension_names",), "invalid_value")]), + ( + {"shape": (4, 4)}, + [(("chunk_grid", "configuration", "chunk_shape"), "invalid_value")], + ), + ({"attributes": {1: "a"}}, [(("attributes",), "invalid_type")]), + ], + ids=[ + "fill-value-not-json", + "names-for-another-rank", + "shape-without-its-grid", + "attribute-key", + ], +) +def test_error_to_key_value_refuses_a_model_changed_by_hand_into_an_invalid_one( + changes: dict[str, object], problems: list[tuple[tuple[str | int, ...], str]] +) -> None: + """As its own fields read it: no invalid document is written, however the model came to be.""" + model = dataclasses.replace(ZarrV3ArrayMetadata.create_default(shape=(4,)), **changes) + with pytest.raises(MetadataValidationError) as raised: model.to_key_value() + assert [(p.loc, p.kind) for p in raised.value.problems] == problems + + +@pytest.mark.parametrize( + ("shape", "kind"), + [ + (5, "invalid_type"), + (None, "invalid_type"), + (("a",), "invalid_type"), + ((-1,), "invalid_value"), + ], + ids=["a-number", "null", "not-integers", "negative"], +) +def test_error_create_default_reports_a_shape_it_cannot_read(shape: object, kind: str) -> None: + """As the read reports it, rather than failing to derive a grid from it.""" + with pytest.raises(MetadataValidationError) as raised: + ZarrV3ArrayMetadata.create_default(shape=shape) # pyright: ignore[reportArgumentType] + assert [(p.loc, p.kind) for p in raised.value.problems] == [(("shape",), kind)] + + +def test_a_codec_that_is_not_json_is_placed_by_the_definition_that_claims_its_name() -> None: + """Its kind is that definition's, as a codec refused for a configuration that is JSON takes it: `gzip` is a bytes -> bytes codec, so the pipeline still lacks an array -> bytes one.""" + document = { + **ZarrV3ArrayMetadata.create_default().to_json(), + "codecs": [{"name": "gzip", "configuration": {"level": math.nan}}], + } + assert [(p.loc, p.kind) for p in validate_array_metadata_v3(document)] == [ + (("codecs", 0, "configuration", "level"), "invalid_value"), + (("codecs",), "invalid_value"), + ] + + +def test_a_shard_nested_hundreds_deep_is_read() -> None: + """A field it holds is read one frame deeper than it, as deep as the interpreter goes.""" + little = {"name": "bytes", "configuration": {"endian": "little"}} + codecs: list[object] = [little] + for _ in range(200): + codecs = [ + { + "name": "sharding_indexed", + "configuration": {"chunk_shape": [1], "codecs": codecs, "index_codecs": [little]}, + } + ] + document = {**ZarrV3ArrayMetadata.create_default(shape=(2,)).to_json(), "codecs": codecs} + assert validate_array_metadata_v3(document) == () + + +def test_a_model_is_written_as_deep_as_it_is_read() -> None: + """A fill value hundreds deep, of a data type nothing in scope claims, which takes any JSON.""" + fill_value: dict[str, object] = {} + for _ in range(600): + fill_value = {"x": fill_value} + document = { + **ZarrV3ArrayMetadata.create_default().to_json(), + "data_type": "acme.deep", + "fill_value": fill_value, + } + model = ZarrV3ArrayMetadata.from_json(document) + assert model.to_json()["fill_value"] == fill_value + assert ZarrV3ArrayMetadata.from_key_value(model.to_key_value()) == model def test_v3_node_type_literal_enforced() -> None: @@ -1680,22 +1833,13 @@ def test_from_key_value_missing_key_kind() -> None: def test_extra_fields_overlap_raises_metadata_error() -> None: """The extra-fields overlap invariant raises MetadataValidationError (a ValueError).""" with pytest.raises(MetadataValidationError, match="Extra fields") as exc_info: - ZarrV3ArrayMetadata.create_default(extra_fields={"shape": {"must_understand": False}}) + dataclasses.replace( + ZarrV3ArrayMetadata.create_default(), + extra_fields={"shape": {"must_understand": False}}, + ) assert [p.kind for p in exc_info.value.problems] == ["invalid_value"] -def test_extension_point_fields_annotated_with_role_alias() -> None: - """Extension-point fields are annotated with ZarrV3MetadataField (the - logical role), not ZarrV3NamedConfig (the current serialized form), so a - future widening of the field union does not move annotation sites.""" - assert ZarrV3MetadataField is ZarrV3NamedConfig - annotations = ZarrV3ArrayMetadata.__annotations__ - for field_name in ("data_type", "chunk_grid", "chunk_key_encoding"): - assert annotations[field_name] == "ZarrV3MetadataField" - for field_name in ("codecs", "storage_transformers"): - assert annotations[field_name] == "tuple[ZarrV3MetadataField, ...]" - - # --- Adversarial-probe fixes: documents that used to pass validation --------- @@ -1848,12 +1992,10 @@ def test_must_understand_fields_partition() -> None: https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v3/core/index.rst#L1575-L1578 """ model = ZarrV3ArrayMetadata.create_default( - extra_fields={ - "ext_a": {"name": "a", "must_understand": False}, - "ext_b": {"name": "b"}, - "ext_c": {"name": "c", "must_understand": True}, - "ext_d": 123, - } + ext_a={"name": "a", "must_understand": False}, + ext_b={"name": "b"}, + ext_c={"name": "c", "must_understand": True}, + ext_d=123, ) assert set(model.must_understand_fields) == {"ext_b", "ext_c", "ext_d"} recognized = {"ext_b"} @@ -1862,9 +2004,7 @@ def test_must_understand_fields_partition() -> None: def test_must_understand_fields_empty_when_all_waived() -> None: """must_understand_fields is empty when every extra field is explicitly waived.""" - model = ZarrV3ArrayMetadata.create_default( - extra_fields={"ext_a": {"name": "a", "must_understand": False}} - ) + model = ZarrV3ArrayMetadata.create_default(ext_a={"name": "a", "must_understand": False}) assert model.must_understand_fields == {} @@ -1879,10 +2019,10 @@ def test_dimension_names_null_field_rejected() -> None: doc = dict(ZarrV3ArrayMetadata.create_default().to_json()) | {"dimension_names": None} problems = validate_array_metadata_v3(doc) assert [(p.loc, p.kind) for p in problems] == [(("dimension_names",), "invalid_type")] - # and the model's own None spelling correctly maps to key absence - assert ( - "dimension_names" not in ZarrV3ArrayMetadata.create_default(dimension_names=UNSET).to_json() - ) + # and the model's own UNSET spelling correctly maps to key absence + model = ZarrV3ArrayMetadata.create_default() + assert model.dimension_names is UNSET + assert "dimension_names" not in model.to_json() # --- create_default derives the chunk grid from shape ------------------------ @@ -1893,16 +2033,17 @@ def test_v3_create_default_chunk_grid_follows_shape() -> None: one chunk covering the array (chunk_shape == shape), instead of silently keeping the scalar default's 0-d grid.""" model = ZarrV3ArrayMetadata.create_default(shape=(100, 100)) - assert model.chunk_grid == ZarrV3NamedConfig( - name="regular", configuration={"chunk_shape": (100, 100)} - ) + assert model.chunk_grid.to_json() == { + "name": "regular", + "configuration": {"chunk_shape": (100, 100)}, + } def test_v3_create_default_explicit_chunk_grid_respected() -> None: """An explicit chunk_grid override wins over the shape-derived default.""" - grid = ZarrV3NamedConfig(name="regular", configuration={"chunk_shape": (10, 10)}) + grid: ZarrV3NamedConfigJSON = {"name": "regular", "configuration": {"chunk_shape": (10, 10)}} model = ZarrV3ArrayMetadata.create_default(shape=(100, 100), chunk_grid=grid) - assert model.chunk_grid == grid + assert model.chunk_grid.to_json() == grid def test_v2_create_default_chunks_follow_shape() -> None: @@ -1922,19 +2063,18 @@ def test_v3_create_default_zero_length_dimensions() -> None: the regular grid asks for chunk sizes greater than zero, so the written grid is one every reader takes, and it fits the shape it chunks.""" model = ZarrV3ArrayMetadata.create_default(shape=(0, 3)) - assert model.chunk_grid.configuration["chunk_shape"] == (1, 3) + assert model.chunk_grid.to_json() == { + "name": "regular", + "configuration": {"chunk_shape": (1, 3)}, + } assert validate_array_metadata_v3(model.to_json()) == () -def test_create_default_derivation_is_one_way() -> None: - """Overriding the chunk grid (v3) or chunks (v2) without shape leaves the - scalar default shape=() untouched: a user-supplied chunk_grid is an - extension point taken verbatim, and deriving shape from it would require - interpreting grid configurations, which the model layer never does.""" - grid = ZarrV3NamedConfig(name="regular", configuration={"chunk_shape": (10, 10)}) - v3 = ZarrV3ArrayMetadata.create_default(chunk_grid=grid) - assert v3.shape == () - assert v3.chunk_grid == grid +def test_v2_create_default_derivation_is_one_way() -> None: + """Overriding chunks without shape leaves the scalar default shape=() + untouched. The v3 model does not derive a shape from its grid either, and + refuses a grid the default shape does not fit: see + `test_error_create_default_refuses_a_document_with_a_problem`.""" v2 = ZarrV2ArrayMetadata.create_default(chunks=(10, 10)) assert v2.shape == () assert v2.chunks == (10, 10) diff --git a/packages/zarr-metadata/tests/model/test_extension_points.py b/packages/zarr-metadata/tests/model/test_extension_points.py index 1459e4ec48..b67216a309 100644 --- a/packages/zarr-metadata/tests/model/test_extension_points.py +++ b/packages/zarr-metadata/tests/model/test_extension_points.py @@ -78,13 +78,13 @@ def test_a_document_reads_through_the_definitions_in_its_scope( document: dict[str, Any], context: Context ) -> None: # What the scope holds judges; what it does not is left to the reader. - # The model reads and writes in the same scope as the validators. + # The model reads in the same scope as the validators, and what it + # writes reads back in that scope as the same model. assert validate_array_metadata_v3(document, context=context) == () assert is_array_metadata_v3(document, context=context) assert parse_array_metadata_v3(document, context=context) is not None model = ZarrV3ArrayMetadata.from_json(document, context=context) - written = model.to_key_value(context=context) - assert ZarrV3ArrayMetadata.from_key_value(written, context=context) == model + assert ZarrV3ArrayMetadata.from_key_value(model.to_key_value(), context=context) == model @pytest.mark.parametrize( @@ -179,20 +179,7 @@ def test_an_array_in_a_group_s_consolidated_metadata_reads_in_the_group_s_scope( lenient = CORE.extended_with(LENIENT_GZIP) assert validate_group_metadata_v3(group, context=lenient) == () model = ZarrV3GroupMetadata.from_json(group, context=lenient) - written = model.to_key_value(context=lenient) - assert ZarrV3GroupMetadata.from_key_value(written, context=lenient) == model - - -def test_error_a_model_is_not_written_in_a_scope_that_refuses_it() -> None: - # The writer judges in the scope it is given, as the reader does: the - # default one refuses the level the reader's own gzip took. - document = _document(codecs=[BYTES, {"name": "gzip", "configuration": {"level": 99}}]) - model = ZarrV3ArrayMetadata.from_json(document, context=CORE.extended_with(LENIENT_GZIP)) - with pytest.raises(MetadataValidationError) as raised: - model.to_key_value() - assert [(problem.loc, problem.kind) for problem in raised.value.problems] == [ - (("codecs", 1, "configuration", "level"), "invalid_value") - ] + assert ZarrV3GroupMetadata.from_key_value(model.to_key_value(), context=lenient) == model def test_error_an_envelope_is_judged_once() -> None: diff --git a/packages/zarr-metadata/tests/model/test_group.py b/packages/zarr-metadata/tests/model/test_group.py index a13a37a5e6..860c4e9bab 100644 --- a/packages/zarr-metadata/tests/model/test_group.py +++ b/packages/zarr-metadata/tests/model/test_group.py @@ -3,40 +3,57 @@ import copy import dataclasses import json +import math from collections import UserDict -from collections.abc import Callable +from collections.abc import Callable, Iterator +from typing import cast import pytest from tests.model._cases import mutate_nested_containers +from zarr_metadata._common import JSONValue, ZarrV3NamedConfigJSON from zarr_metadata._json import ( MetadataValidationError, ValidationProblem, arrays_to_tuples, ) from zarr_metadata.model import UNSET -from zarr_metadata.model._array import ZarrV3ArrayMetadata +from zarr_metadata.model._array import ZarrV3ArrayMetadata, ZarrV3ArrayMetadataUpdate from zarr_metadata.model._group import ( ZarrV2ConsolidatedMetadata, ZarrV2GroupMetadata, ZarrV2GroupMetadataPartial, ZarrV3ConsolidatedMetadata, ZarrV3GroupMetadata, - ZarrV3GroupMetadataPartial, + ZarrV3GroupMetadataReading, + ZarrV3GroupMetadataUpdate, + is_group_metadata_v3, + parse_group_metadata_v3, + read_group_metadata_v3, + validate_group_metadata_v3, ) from zarr_metadata.model._validation import ( is_group_metadata_v2, - is_group_metadata_v3, parse_group_metadata_v2, - parse_group_metadata_v3, validate_group_metadata_v2, - validate_group_metadata_v3, ) from zarr_metadata.v2.group import ( ZarrV2GroupMetadataJSON, ZarrV2GroupMetadataJSONPartial, ZarrV2ZGroupJSON, ) +from zarr_metadata.v3.array import ZarrV3ArrayMetadataJSONPartial +from zarr_metadata.v3.codec.gzip import GZIP_CODEC, GzipCodecConfiguration +from zarr_metadata.v3.definition import ( + CORE, + CORE_AND_EXTENSIONS, + CodecDefinition, + EmptyConfiguration, + Nested, + Refused, + Unclaimed, +) +from zarr_metadata.v3.group import ZarrV3GroupMetadataJSONPartial # --- ZarrV3GroupMetadata --------------------------------------------------- @@ -206,7 +223,7 @@ def test_group_v3_key_value_roundtrip() -> None: def test_group_v3_update() -> None: """update replaces the given fields and returns a new instance.""" base = ZarrV3GroupMetadata.create_default() - updated = base.update(attributes={"a": 1}) + updated = base.update(context=CORE_AND_EXTENSIONS, attributes={"a": 1}) assert updated.attributes == {"a": 1} assert base.attributes == {} @@ -279,17 +296,23 @@ def test_group_v2_missing_required_key() -> None: def test_group_partial_keys_match_settable_model_fields() -> None: - """Each group partial TypedDict must list exactly the settable model fields. + """The v2 group partial TypedDict lists exactly the settable model fields. - Guards against drift: adding/removing a settable field on a group model + Guards against drift: adding/removing a settable field on the model without updating its `*Partial` TypedDict fails here. """ - for model_cls, partial_cls in ( - (ZarrV3GroupMetadata, ZarrV3GroupMetadataPartial), - (ZarrV2GroupMetadata, ZarrV2GroupMetadataPartial), + settable = {f.name for f in dataclasses.fields(ZarrV2GroupMetadata) if f.init} + assert set(ZarrV2GroupMetadataPartial.__annotations__) == settable + + +def test_update_takes_every_member_of_the_document_it_may_change() -> None: + """Each v3 model's `update` takes each member of its document but `zarr_format` and `node_type`, which it cannot change.""" + for update, partial in ( + (ZarrV3ArrayMetadataUpdate, ZarrV3ArrayMetadataJSONPartial), + (ZarrV3GroupMetadataUpdate, ZarrV3GroupMetadataJSONPartial), ): - settable = {f.name for f in dataclasses.fields(model_cls) if f.init} - assert set(partial_cls.__annotations__) == settable + fixed = {"zarr_format", "node_type"} + assert set(update.__annotations__) == set(partial.__annotations__) - fixed # --- ZarrV3ConsolidatedMetadata -------------------------------------------- @@ -343,6 +366,199 @@ def test_consolidated_v3_not_a_mapping() -> None: ZarrV3ConsolidatedMetadata.from_json(5) +# --- read_group_metadata_v3 ------------------------------------------------ + +LITTLE: ZarrV3NamedConfigJSON = {"name": "bytes", "configuration": {"endian": "little"}} +GZIP_99 = {"name": "gzip", "configuration": {"level": 99}} + +A: tuple[str, ...] = ("consolidated_metadata", "metadata", "a") +"""Where the document at path `a` in a group's consolidated metadata sits in the group's.""" + + +def _array(**members: object) -> dict[str, object]: + return {**ZarrV3ArrayMetadata.create_default(shape=(4,)).to_json(), **members} + + +def _inline(**documents: object) -> dict[str, object]: + return {"kind": "inline", "must_understand": False, "metadata": documents} + + +def _group(**members: object) -> dict[str, object]: + return {"zarr_format": 3, "node_type": "group", **members} + + +def test_error_a_consolidated_envelope_reports_the_members_it_lacks_in_its_order() -> None: + document = _group(consolidated_metadata={"kind": "inline"}) + assert [(p.loc, p.kind) for p in validate_group_metadata_v3(document)] == [ + (("consolidated_metadata", "must_understand"), "missing_key"), + (("consolidated_metadata", "metadata"), "missing_key"), + ] + + +ACME_X = CodecDefinition( + name="acme.x", configuration=EmptyConfiguration, kind="bytes_bytes", size="dynamic" +) + + +def test_group_update_keeps_the_documents_it_holds() -> None: + """Read in no scope again: one read in a scope the call's does not claim keeps each field as it was read.""" + scope = CORE_AND_EXTENSIONS.extended_with(ACME_X) + child = ZarrV3ArrayMetadata.create_default( + context=scope, shape=(4,), codecs=(LITTLE, {"name": "acme.x"}) + ) + group = ZarrV3GroupMetadata( + attributes={}, + consolidated_metadata=ZarrV3ConsolidatedMetadata(metadata={"a": child}), + extra_fields={}, + ) + updated = group.update(context=CORE_AND_EXTENSIONS, attributes={"k": 1}) + assert updated.attributes == {"k": 1} + assert updated.consolidated_metadata is group.consolidated_metadata + + +def test_group_update_reads_the_documents_it_is_given_in_its_scope() -> None: + """And `UNSET` leaves them out. `zstd` is an extension, which `CORE` leaves unclaimed.""" + zstd = {"name": "zstd", "configuration": {"level": 3, "checksum": False}} + member = cast("JSONValue", _inline(a=_array(codecs=[LITTLE, zstd]))) + updated = ZarrV3GroupMetadata.create_default().update( + context=CORE, consolidated_metadata=member + ) + assert updated.consolidated_metadata is not UNSET + child = updated.consolidated_metadata.metadata["a"] + assert isinstance(child, ZarrV3ArrayMetadata) + assert isinstance(child.codecs[1], Unclaimed) + removed = updated.update(context=CORE, consolidated_metadata=UNSET) + assert removed.consolidated_metadata is UNSET + + +def test_error_to_key_value_refuses_a_group_holding_a_document_with_a_problem() -> None: + """A document it holds changed by hand into an invalid one, as its own fields read it.""" + child = dataclasses.replace(ZarrV3ArrayMetadata.create_default(shape=(4,)), fill_value=math.nan) + group = ZarrV3GroupMetadata( + attributes={}, + consolidated_metadata=ZarrV3ConsolidatedMetadata(metadata={"a": child}), + extra_fields={}, + ) + with pytest.raises(MetadataValidationError) as raised: + group.to_key_value() + assert [(p.loc, p.kind) for p in raised.value.problems] == [ + ((*A, "fill_value"), "invalid_value") + ] + + +def _fields_of_an_array(*at: str | int) -> list[tuple[str | int, ...]]: + """Where a default array's fields sit, under `at`.""" + points = ("data_type", "chunk_grid", "chunk_key_encoding") + return [*((*at, point) for point in points), (*at, "codecs", 0)] + + +@pytest.mark.parametrize( + ("document", "paths", "locs"), + [ + (_group(), [], []), + # A null, which a historical zarr-python bug wrote, holds nothing. + (_group(consolidated_metadata=None), [], []), + ( + _group(consolidated_metadata=_inline(a=_array(), g=_group())), + ["a", "g"], + _fields_of_an_array(*A), + ), + # A group's consolidated metadata in a group's: each field located + # from the root of the outer document. + ( + _group( + consolidated_metadata=_inline(g=_group(consolidated_metadata=_inline(b=_array()))) + ), + ["g"], + _fields_of_an_array( + "consolidated_metadata", "metadata", "g", "consolidated_metadata", "metadata", "b" + ), + ), + ], + ids=["no-consolidated-metadata", "null", "an-array-and-a-group", "nested"], +) +def test_a_group_reads_each_document_its_consolidated_metadata_holds( + document: dict[str, object], paths: list[str], locs: list[tuple[str | int, ...]] +) -> None: + reading = read_group_metadata_v3(document) + assert reading.problems == () + assert list(reading.consolidated) == paths + assert [loc for loc, _ in reading.fields()] == locs + model = reading.metadata + assert model is not None + assert model == ZarrV3GroupMetadata.from_json(document) + # It holds the model each document's own reading built. + consolidated = model.consolidated_metadata + held = {} if consolidated is UNSET else consolidated.metadata + assert all(held[path] is reading.consolidated[path].metadata for path in paths) + + +def test_each_document_its_consolidated_metadata_holds_is_read_once() -> None: + reads: list[GzipCodecConfiguration] = [] + + def counted( + configuration: GzipCodecConfiguration, nested: Nested + ) -> Iterator[ValidationProblem]: + reads.append(configuration) + yield from () + + scope = CORE_AND_EXTENSIONS.extended_with(dataclasses.replace(GZIP_CODEC, rules=counted)) + array = _array(codecs=[LITTLE, {"name": "gzip", "configuration": {"level": 5}}]) + group = _group(consolidated_metadata=_inline(a=array)) + assert read_group_metadata_v3(group, context=scope).problems == () + assert reads == [{"level": 5}] + ZarrV3GroupMetadata.from_json(group, context=scope) + assert len(reads) == 2 + + +def test_error_a_group_document_that_is_not_an_object_reads_as_nothing() -> None: + not_an_object = ValidationProblem((), "expected an object", "invalid_type") + assert read_group_metadata_v3([1]) == ZarrV3GroupMetadataReading(problems=(not_an_object,)) + + +@pytest.mark.parametrize( + ("document", "problems", "models", "refused"), + [ + # The group's own member: each document it holds still has its model. + ( + _group(attributes=5, consolidated_metadata=_inline(a=_array())), + [(("attributes",), "invalid_type")], + {"a": True}, + [], + ), + # A document it holds: that one has none, and its sibling has one; + # the field refused is found where it sits, like every field. + ( + _group(consolidated_metadata=_inline(a=_array(codecs=[LITTLE, GZIP_99]), b=_array())), + [((*A, "codecs", 1, "configuration", "level"), "invalid_value")], + {"a": False, "b": True}, + [(*A, "codecs", 1)], + ), + # One of no node type is not read at all. + ( + _group(consolidated_metadata=_inline(a={"zarr_format": 3})), + [((*A, "node_type"), "invalid_value")], + {}, + [], + ), + ], + ids=["group", "consolidated-document", "no-node-type"], +) +def test_error_a_group_document_with_a_problem_reads_as_no_model( + document: dict[str, object], + problems: list[tuple[tuple[str | int, ...], str]], + models: dict[str, bool], + refused: list[tuple[str | int, ...]], +) -> None: + reading = read_group_metadata_v3(document) + assert [(p.loc, p.kind) for p in reading.problems] == problems + assert reading.metadata is None + assert { + path: read.metadata is not None for path, read in reading.consolidated.items() + } == models + assert [loc for loc, field in reading.fields() if isinstance(field, Refused)] == refused + + # --- ZarrV2ConsolidatedMetadata -------------------------------------------- @@ -505,7 +721,7 @@ def test_group_v3_validator_agrees_with_from_json_on_consolidated() -> None: }, ) for doc in bad_docs: - assert validate_group_metadata_v3(doc) != [], doc + assert validate_group_metadata_v3(doc) != (), doc with pytest.raises(MetadataValidationError): ZarrV3GroupMetadata.from_json(doc) @@ -569,10 +785,7 @@ def test_group_must_understand_fields_partition() -> None: https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v3/core/index.rst#L1571-L1573 """ model = ZarrV3GroupMetadata.create_default( - extra_fields={ - "waived": {"name": "w", "must_understand": False}, - "implicit": {"name": "i"}, - } + waived={"name": "w", "must_understand": False}, implicit={"name": "i"} ) assert set(model.must_understand_fields) == {"implicit"} @@ -594,7 +807,7 @@ def test_group_v3_null_consolidated_metadata_repaired_to_absence() -> None: TO_JSON_NO_ALIASING_PARAMS = [ pytest.param( - ZarrV3GroupMetadata.create_default( + ZarrV3GroupMetadata( attributes={"a": {"b": [1]}}, consolidated_metadata=ZarrV3ConsolidatedMetadata( metadata={ diff --git a/packages/zarr-metadata/tests/model/test_pydantic.py b/packages/zarr-metadata/tests/model/test_pydantic.py index 7951cac5c3..2cc8822fe1 100644 --- a/packages/zarr-metadata/tests/model/test_pydantic.py +++ b/packages/zarr-metadata/tests/model/test_pydantic.py @@ -42,7 +42,7 @@ ) from zarr_metadata import JSONValue -from zarr_metadata.model import ZarrV3ArrayMetadata, ZarrV3NamedConfig +from zarr_metadata.model import ZarrV3ArrayMetadata # --- the integration (this is the example) ----------------------------------- @@ -169,8 +169,6 @@ def build_and_use() -> None: "ZarrV3ExtensionField": ZarrV3ExtensionField, "ZarrV3MetadataFieldJSON": ZarrV3MetadataFieldJSON, "ZarrV3ArrayMetadataJSON": ZarrV3ArrayMetadataJSON, - "ZarrV3NamedConfig": ZarrV3NamedConfig, - "ZarrV3MetadataField": ZarrV3NamedConfig, "UNSET": UNSET, }, ) diff --git a/packages/zarr-metadata/tests/model/test_pydantic_module.py b/packages/zarr-metadata/tests/model/test_pydantic_module.py index cb45310759..70a3abc8b4 100644 --- a/packages/zarr-metadata/tests/model/test_pydantic_module.py +++ b/packages/zarr-metadata/tests/model/test_pydantic_module.py @@ -23,8 +23,8 @@ ZarrV3ArrayMetadata, ZarrV3ConsolidatedMetadata, ZarrV3GroupMetadata, - ZarrV3NamedConfig, ) +from zarr_metadata.v3.definition import CORE, Read V3_ARRAY_DOC = dict(ZarrV3ArrayMetadata.create_default(shape=(4,)).to_json()) V2_ARRAY_DOC = dict(ZarrV2ArrayMetadata.create_default(shape=(4,), chunks=(2,)).to_json()) @@ -57,7 +57,6 @@ V2_CONSOLIDATED_DOC, id="consolidated-v2", ), - pytest.param(zmp.ZarrV3MetadataField, ZarrV3NamedConfig, {"name": "bytes"}, id="field-v3"), ] @@ -105,7 +104,6 @@ def test_json_schema_generation() -> None: class Manifest(BaseModel): metadata: zmp.ZarrV3ArrayMetadata - codec: zmp.ZarrV3MetadataField schema = Manifest.model_json_schema() metadata_schema = schema["$defs"]["ZarrV3ArrayMetadataJSON"] @@ -125,10 +123,6 @@ class Manifest(BaseModel): "title": "Zarr Format", "type": "integer", } - assert schema["properties"]["codec"]["anyOf"] == [ - {"type": "string"}, - {"$ref": "#/$defs/ZarrV3NamedConfigJSON"}, - ] def test_json_schema_generation_emits_no_warnings() -> None: @@ -140,7 +134,6 @@ def test_json_schema_generation_emits_no_warnings() -> None: zmp.ZarrV2GroupMetadata, zmp.ZarrV3ConsolidatedMetadata, zmp.ZarrV2ConsolidatedMetadata, - zmp.ZarrV3MetadataField, ) with warnings.catch_warnings(): @@ -211,10 +204,10 @@ def test_v3_array_schema_rejects_false_at_every_extension_point(field: str) -> N def test_metadata_field_schema_rejects_unknown_members() -> None: """Named-configuration envelopes are closed in both runtime and schema validation.""" - _assert_runtime_and_schema_reject( - zmp.ZarrV3MetadataField, - {"name": "example", "unexpected": 1}, - ) + doc = json.loads(json.dumps(V3_ARRAY_DOC)) + doc["codecs"] = [{"name": "bytes", "unexpected": 1}] + + _assert_runtime_and_schema_reject(zmp.ZarrV3ArrayMetadata, doc) @pytest.mark.parametrize( @@ -264,15 +257,6 @@ class Manifest(BaseModel): assert Manifest.model_validate_json(manifest.model_dump_json()) == manifest -def test_metadata_field_serializes_shorthand_and_false_object() -> None: - """The optional integration exposes the core model's canonical extension form.""" - adapter = TypeAdapter(zmp.ZarrV3MetadataField) - assert adapter.dump_python(adapter.validate_python({"name": "bytes"})) == "bytes" - assert adapter.dump_python( - adapter.validate_python({"name": "optional", "must_understand": False}) - ) == {"name": "optional", "must_understand": False} - - def test_core_package_does_not_import_pydantic() -> None: """Importing zarr_metadata (in a fresh interpreter) must not import pydantic: the integration is opt-in via zarr_metadata.pydantic.""" @@ -296,3 +280,36 @@ class Manifest(BaseModel): held = Manifest.model_validate_json(written).metadata.attributes["x"] assert isinstance(held, float) assert math.isnan(held) + + +# --- the scope a v3 field type reads in -------------------------------------- + +_WITH_ZSTD = ZarrV3ArrayMetadata.create_default( + shape=(4,), + codecs=({"name": "bytes"}, {"name": "zstd", "configuration": {"level": 3, "checksum": False}}), +).to_json() + + +@pytest.mark.parametrize( + ("context", "read"), + [ + (None, True), + (CORE, False), + ({zmp.CONTEXT_KEY: CORE}, False), + ({"another validator's": 1}, True), + ], + ids=["none", "a-scope", "a-mapping-holding-one", "a-mapping-holding-none"], +) +def test_a_v3_field_type_reads_in_the_scope_the_validation_context_holds( + context: object, read: bool +) -> None: + """As pydantic hands any validator its context; `zstd` is an extension, which `CORE` leaves unclaimed.""" + model = TypeAdapter(zmp.ZarrV3ArrayMetadata).validate_python(_WITH_ZSTD, context=context) + assert isinstance(model.codecs[1], Read) is read + + +def test_error_a_validation_context_holds_a_scope_that_is_not_one() -> None: + with pytest.raises(TypeError, match=zmp.CONTEXT_KEY): + TypeAdapter(zmp.ZarrV3ArrayMetadata).validate_python( + _WITH_ZSTD, context={zmp.CONTEXT_KEY: "CORE"} + ) diff --git a/packages/zarr-metadata/tests/model/test_read_array_metadata.py b/packages/zarr-metadata/tests/model/test_read_array_metadata.py index 91c7646377..d25455ee6a 100644 --- a/packages/zarr-metadata/tests/model/test_read_array_metadata.py +++ b/packages/zarr-metadata/tests/model/test_read_array_metadata.py @@ -1,9 +1,9 @@ """A v3 array document, read: each field as a scope read it, and its codecs as a pipeline. -`read_array_metadata_v3` returns what the validator read to find what is -wrong with a document, beside the problems it found: each field with -where it sits and the kind it was read as, the chunks the codecs are -handed, and each codec with the chunk it is handed. +`read_array_metadata_v3` returns everything one read of a document finds: +each field with where it sits, the kind it was read as and what the scope +made of it, the chunks the codecs are handed, each codec with the chunk it +is handed, every problem, and the document's model when there is none. """ from __future__ import annotations @@ -29,7 +29,10 @@ Context, DataTypeDefinition, Definition, + Read, + Refused, StorageTransformerDefinition, + Unclaimed, ) if TYPE_CHECKING: @@ -38,10 +41,10 @@ LITTLE = {"name": "bytes", "configuration": {"endian": "little"}} ZSTD = {"name": "zstd", "configuration": {"level": 1}} -POINTS: list[tuple[Loc, type[Definition[Any]], str]] = [ - (("data_type",), DataTypeDefinition, "read"), - (("chunk_grid",), ChunkGridDefinition, "read"), - (("chunk_key_encoding",), ChunkKeyEncodingDefinition, "read"), +POINTS: list[tuple[Loc, type[Definition[Any]], type]] = [ + (("data_type",), DataTypeDefinition, Read), + (("chunk_grid",), ChunkGridDefinition, Read), + (("chunk_key_encoding",), ChunkKeyEncodingDefinition, Read), ] """A default document's single extension points, each read where it sits.""" @@ -52,8 +55,8 @@ def _document(shape: tuple[int, ...] = (4,), **fields: object) -> dict[str, Any] return cast("dict[str, Any]", arrays_to_tuples(document)) -def _codec(loc: Loc, resolution: str = "read") -> tuple[Loc, type[Definition[Any]], str]: - return (loc, CodecDefinition, resolution) +def _codec(loc: Loc, variant: type = Read) -> tuple[Loc, type[Definition[Any]], type]: + return (loc, CodecDefinition, variant) def _shard(chunk_shape: list[int], codecs: list[object] | None = None, **more: object) -> object: @@ -93,7 +96,7 @@ def _shard(chunk_shape: list[int], codecs: list[object] | None = None, **more: o *POINTS, _codec(("codecs", 0)), _codec(("codecs", 0, "configuration", "codecs", 0)), - _codec(("codecs", 0, "configuration", "codecs", 1), "out_of_scope"), + _codec(("codecs", 0, "configuration", "codecs", 1), Unclaimed), _codec(("codecs", 0, "configuration", "index_codecs", 0)), _codec(("codecs", 0, "configuration", "index_codecs", 1)), ], @@ -129,9 +132,9 @@ def _shard(chunk_shape: list[int], codecs: list[object] | None = None, **more: o CORE, [ *POINTS, - _codec(("codecs", 0), "invalid"), + _codec(("codecs", 0), Refused), _codec(("codecs", 0, "configuration", "codecs", 0)), - _codec(("codecs", 0, "configuration", "codecs", 1), "out_of_scope"), + _codec(("codecs", 0, "configuration", "codecs", 1), Unclaimed), _codec(("codecs", 0, "configuration", "index_codecs", 0)), ], [(frozenset({8}), frozenset({8}))], @@ -158,12 +161,12 @@ def _shard(chunk_shape: list[int], codecs: list[object] | None = None, **more: o ( ("data_type", "configuration", "fields", 0, "data_type"), DataTypeDefinition, - "read", + Read, ), ( ("data_type", "configuration", "fields", 1, "data_type"), DataTypeDefinition, - "out_of_scope", + Unclaimed, ), *POINTS[1:], _codec(("codecs", 0)), @@ -182,19 +185,19 @@ def _shard(chunk_shape: list[int], codecs: list[object] | None = None, **more: o CORE_AND_EXTENSIONS, [ POINTS[0], - (("chunk_grid",), ChunkGridDefinition, "out_of_scope"), + (("chunk_grid",), ChunkGridDefinition, Unclaimed), POINTS[2], _codec(("codecs", 0)), - _codec(("codecs", 1), "invalid"), - (("storage_transformers", 0), StorageTransformerDefinition, "out_of_scope"), + _codec(("codecs", 1), Refused), + (("storage_transformers", 0), StorageTransformerDefinition, Unclaimed), ], [(None,), None], ), - # A data type that is not JSON is read as nothing. + # A data type that is not JSON is refused. ( _document(data_type=float("nan")), CORE_AND_EXTENSIONS, - [(("data_type",), DataTypeDefinition, "invalid"), *POINTS[1:], _codec(("codecs", 0))], + [(("data_type",), DataTypeDefinition, Refused), *POINTS[1:], _codec(("codecs", 0))], [(frozenset({4}),)], ), ], @@ -212,22 +215,40 @@ def _shard(chunk_shape: list[int], codecs: list[object] | None = None, **more: o def test_a_document_reads_as_each_field_where_it_sits_and_its_codecs_as_a_pipeline( document: dict[str, Any], context: Context, - fields: list[tuple[Loc, type[Definition[Any]], str]], + fields: list[tuple[Loc, type[Definition[Any]], type]], handed: list[Lengths | None], ) -> None: - reading, problems = read_array_metadata_v3(document, context=context) - assert problems == validate_array_metadata_v3(document, context=context) - assert [(loc, field.read_as, field.resolution) for loc, field in reading.fields()] == fields + reading = read_array_metadata_v3(document, context=context) + assert reading.problems == validate_array_metadata_v3(document, context=context) + # A model only of a document with no problem. + assert (reading.metadata is None) is (reading.problems != ()) + assert [(loc, field.read_as, type(field)) for loc, field in reading.fields()] == fields assert reading.chunk.data_type is reading.data_type assert reading.pipeline[0].incoming == reading.chunk assert [None if s.incoming is None else s.incoming.lengths for s in reading.pipeline] == handed +def test_a_document_with_no_problem_reads_as_its_model_holding_the_fields_read() -> None: + # The fields the read made, not a second reading of them. + document = _document(codecs=[{"name": "transpose", "configuration": {"order": [0]}}, LITTLE]) + reading = read_array_metadata_v3(document) + model = reading.metadata + assert reading.problems == () + assert model is not None + assert model == ZarrV3ArrayMetadata.from_json(document) + assert model.data_type is reading.data_type + assert model.chunk_grid is reading.chunk_grid + assert model.chunk_key_encoding is reading.chunk_key_encoding + assert all( + codec is stage.codec for codec, stage in zip(model.codecs, reading.pipeline, strict=True) + ) + + def test_error_a_value_that_is_not_a_mapping_reads_as_nothing() -> None: - reading, problems = read_array_metadata_v3(["not", "a", "document"]) - assert reading == ZarrV3ArrayMetadataReading() + reading = read_array_metadata_v3(["not", "a", "document"]) + not_an_object = ValidationProblem((), "expected an object", "invalid_type") + assert reading == ZarrV3ArrayMetadataReading(problems=(not_an_object,)) assert list(reading.fields()) == [] - assert problems == (ValidationProblem((), "expected an object", "invalid_type"),) @pytest.mark.parametrize( @@ -239,23 +260,23 @@ def test_error_a_list_of_fields_that_is_not_a_list_reads_as_empty( ) -> None: document = _document() document[member] = "bytes" - reading, problems = read_array_metadata_v3(document) + reading = read_array_metadata_v3(document) assert getattr(reading, attribute) == () - assert [(p.loc, p.kind) for p in problems] == [((member,), "invalid_type")] + assert [(p.loc, p.kind) for p in reading.problems] == [((member,), "invalid_type")] @pytest.mark.parametrize("member", ["data_type", "chunk_grid", "chunk_key_encoding"]) def test_error_a_document_without_a_field_reads_it_as_none(member: str) -> None: document = _document() del document[member] - reading, problems = read_array_metadata_v3(document) + reading = read_array_metadata_v3(document) assert getattr(reading, member) is None assert (member,) not in [loc for loc, _ in reading.fields()] - assert ValidationProblem((member,), "missing required key", "missing_key") in problems + assert ValidationProblem((member,), "missing required key", "missing_key") in reading.problems def test_error_a_document_without_a_shape_hands_its_codecs_chunks_of_no_known_rank() -> None: document = _document() del document["shape"] - reading, _ = read_array_metadata_v3(document) + reading = read_array_metadata_v3(document) assert reading.chunk.lengths is None diff --git a/packages/zarr-metadata/tests/model/test_sentinel.py b/packages/zarr-metadata/tests/model/test_sentinel.py index 252a4cdd30..17385d46c2 100644 --- a/packages/zarr-metadata/tests/model/test_sentinel.py +++ b/packages/zarr-metadata/tests/model/test_sentinel.py @@ -34,21 +34,23 @@ # stay distinct from absent), and UNSET nested inside consolidated metadata. MODEL_CASES = { "array-v3-dimension-names-unset": ZarrV3ArrayMetadata.create_default(shape=(4,)), - "array-v3-dimension-names-set": ZarrV3ArrayMetadata.create_default(shape=(2, 2)).update( - dimension_names=("x", None) + "array-v3-dimension-names-set": ZarrV3ArrayMetadata.create_default( + shape=(2, 2), dimension_names=("x", None) ), "array-v2-attributes-unset": ZarrV2ArrayMetadata.create_default(shape=(4,)), "array-v2-attributes-empty": ZarrV2ArrayMetadata.create_default(shape=(4,), attributes={}), "group-v2-attributes-unset": ZarrV2GroupMetadata.create_default(), "group-v2-attributes-set": ZarrV2GroupMetadata.create_default(attributes={"a": 1}), "group-v3-consolidated-unset": ZarrV3GroupMetadata.create_default(), - "group-v3-consolidated-with-unset-inside": ZarrV3GroupMetadata.create_default( + "group-v3-consolidated-with-unset-inside": ZarrV3GroupMetadata( + attributes={}, consolidated_metadata=ZarrV3ConsolidatedMetadata( metadata={ "child": ZarrV3ArrayMetadata.create_default(shape=(4,)), "subgroup": ZarrV3GroupMetadata.create_default(), } - ) + ), + extra_fields={}, ), } diff --git a/packages/zarr-metadata/tests/model/test_store_json.py b/packages/zarr-metadata/tests/model/test_store_json.py index 76bfd252b4..fd50eaa236 100644 --- a/packages/zarr-metadata/tests/model/test_store_json.py +++ b/packages/zarr-metadata/tests/model/test_store_json.py @@ -4,7 +4,7 @@ so an attribute may hold `NaN`, `Infinity` or `-Infinity`. The model reads such a store, validates it, and writes it back the same way. Wherever the spec interprets a value, a non-finite number is refused when read, and a -document the reader would refuse is not written. +document the reader would refuse makes no model, so it is not written. """ from __future__ import annotations @@ -295,34 +295,36 @@ def test_error_a_non_finite_number_outside_attributes_is_located_when_read( @pytest.mark.parametrize( - ("model", "problems"), + ("build", "problems"), [ ( - ZarrV3ArrayMetadata.create_default(fill_value=math.nan, attributes={"x": math.nan}), + lambda: ZarrV3ArrayMetadata.create_default( + fill_value=math.nan, attributes={"x": math.nan} + ), [(("fill_value",), "invalid_value")], ), ( - ZarrV3GroupMetadata.create_default(attributes={"s": {1, 2}}), # pyright: ignore[reportArgumentType] + lambda: ZarrV3GroupMetadata.create_default(attributes={"s": {1, 2}}), # pyright: ignore[reportArgumentType] [(("attributes", "s"), "invalid_type")], ), ( - ZarrV3GroupMetadata.create_default(attributes={1: "a"}), # pyright: ignore[reportArgumentType] + lambda: ZarrV3GroupMetadata.create_default(attributes={1: "a"}), # pyright: ignore[reportArgumentType] [(("attributes",), "invalid_type")], ), ( - ZarrV3ArrayMetadata.create_default(shape=(2,), dimension_names=("x", "y")), + lambda: ZarrV3ArrayMetadata.create_default(shape=(2,), dimension_names=("x", "y")), [(("dimension_names",), "invalid_value")], ), ], ids=["non-finite-fill-value", "not-json", "non-string-key", "dimension-names-past-shape"], ) def test_error_a_document_the_reader_refuses_is_not_written( - model: _Stored, problems: list[tuple[tuple[str | int, ...], str]] + build: Callable[[], _Stored], problems: list[tuple[tuple[str | int, ...], str]] ) -> None: - # A model built by hand is not validated; the writer validates what it - # writes as the reader does. A non-string key was written as a string, - # a value that is not JSON raised `TypeError`, and the last was written - # and then refused on read. + # A model comes from a read of its document, so one the reader refuses + # makes no model to write. Written, a non-string key became a string, a + # value that is not JSON raised `TypeError`, and the last was refused + # only when read again. with pytest.raises(MetadataValidationError) as raised: - model.to_key_value() + build() assert [(problem.loc, problem.kind) for problem in raised.value.problems] == problems diff --git a/packages/zarr-metadata/tests/test_public_api.py b/packages/zarr-metadata/tests/test_public_api.py index cb88193103..22fdd0c863 100644 --- a/packages/zarr-metadata/tests/test_public_api.py +++ b/packages/zarr-metadata/tests/test_public_api.py @@ -45,15 +45,13 @@ def _group_rank(s: str) -> int: "ZarrV2ArrayMetadata", "ZarrV2ArrayMetadataPartial", "ZarrV3ArrayMetadata", - "ZarrV3ArrayMetadataPartial", + "ZarrV3ArrayMetadataUpdate", "ZarrV2GroupMetadata", "ZarrV2GroupMetadataPartial", "ZarrV3GroupMetadata", - "ZarrV3GroupMetadataPartial", + "ZarrV3GroupMetadataUpdate", "ZarrV2ConsolidatedMetadata", "ZarrV3ConsolidatedMetadata", - "ZarrV3NamedConfig", - "ZarrV3MetadataField", "ValidationProblem", "MetadataValidationError", "ProblemKind", @@ -240,7 +238,7 @@ def test_all_is_grouped_and_unique() -> None: # Core document/model names: the format version comes first (`ZarrV2` / # `ZarrV3`), then the CamelCase entity, then an optional role suffix -# (`JSON`, `JSONPartial`, `Partial`, `StoreKey`) — validated loosely here +# (`JSON`, `JSONPartial`, `Partial`, `Reading`, `StoreKey`) — validated loosely here # because `JSON` decomposes into single-letter words under any strict # word-splitting regex. _CORE_NAME = re.compile(r"^ZarrV[23](?:[A-Z][a-z0-9]*)+$") @@ -293,11 +291,14 @@ def test_all_is_grouped_and_unique() -> None: "Lengths", "Loc", "Nested", - "Resolution", + # What a scope made of a field: `Read` by the definition that claims + # its name, `Unclaimed`, or `Refused`; `Resolved` is the three. + "Read", + "Refused", "Resolved", "Stage", "StorageClass", - "Unread", + "Unclaimed", "CastOutOfRangeMode", "CastRoundingMode", "Endianness", diff --git a/packages/zarr-metadata/tests/v3/test_definitions.py b/packages/zarr-metadata/tests/v3/test_definitions.py index 4776b43362..f17c670529 100644 --- a/packages/zarr-metadata/tests/v3/test_definitions.py +++ b/packages/zarr-metadata/tests/v3/test_definitions.py @@ -2,7 +2,9 @@ from __future__ import annotations +import copy import math +import pickle from collections.abc import ( Mapping, # noqa: TC003 - a TypedDict's annotations are evaluated at run time ) @@ -32,9 +34,10 @@ Definition, JSONValue, Nested, - Resolution, - Resolved, + Read, + Refused, StorageTransformerDefinition, + Unclaimed, ValidationProblem, ZarrV3MetadataFieldJSON, check, @@ -66,11 +69,8 @@ def acme_stack_rules( # And what the scope read of it: a codec of another kind is refused, # one nothing in scope claims left be. inner = nested.get(("codecs", index)) - if ( - inner is not None - and isinstance(inner.definition, CodecDefinition) - and inner.definition.kind != "bytes_bytes" - ): + definition = inner.definition if isinstance(inner, (Read, Refused)) else None + if isinstance(definition, CodecDefinition) and definition.kind != "bytes_bytes": yield ValidationProblem( ("codecs", index), "a stack holds bytes -> bytes codecs", "invalid_value" ) @@ -197,6 +197,10 @@ def acme_paired_rules( ) +INT8: Final = Read(json="int8", name="int8", definition=INT8_DATA_TYPE, configuration={}) +"""An `int8` field as a scope reads it: by its definition, holding nothing inside.""" + + def _locs(problems: tuple[ValidationProblem, ...]) -> list[tuple[tuple[str | int, ...], str]]: return [(found.loc, found.kind) for found in problems] @@ -216,23 +220,26 @@ def test_a_read_field_keeps_the_fields_it_read_inside() -> None: }, } resolved, _ = resolve(shard, CodecDefinition, CORE_AND_EXTENSIONS) + assert isinstance(resolved, Read) assert {loc: inner.json for loc, inner in resolved.nested.items()} == { ("codecs", 0): "bytes", ("index_codecs", 0): "bytes", ("index_codecs", 1): "crc32c", } - cast = {"name": "cast_value", "configuration": {"data_type": "int8"}} - resolved, _ = resolve(cast, CodecDefinition, CORE_AND_EXTENSIONS) - assert resolved.nested[("data_type",)].definition is INT8_DATA_TYPE - # A field holding none, or whose configuration was not checked, has - # nothing inside; one its check or rules refuse keeps what it read. - assert resolve("int8", DataTypeDefinition, CORE)[0].nested == {} + cast_value = {"name": "cast_value", "configuration": {"data_type": "int8"}} + resolved, _ = resolve(cast_value, CodecDefinition, CORE_AND_EXTENSIONS) + assert isinstance(resolved, Read) + assert resolved.nested[("data_type",)] == INT8 + # A field holding none has nothing inside, and one nothing in scope + # claims holds no field at all; one its check or rules refuse keeps + # what it read. + assert resolve("int8", DataTypeDefinition, CORE)[0] == INT8 unclaimed = {"name": "acme.cast", "configuration": {"data_type": "int8"}} - assert resolve(unclaimed, CodecDefinition, CORE_AND_EXTENSIONS)[0].nested == {} + assert isinstance(resolve(unclaimed, CodecDefinition, CORE_AND_EXTENSIONS)[0], Unclaimed) refused = {"name": "cast_value", "configuration": {"data_type": "int8", "rounding": 1}} resolved, _ = resolve(refused, CodecDefinition, CORE_AND_EXTENSIONS) - assert resolved.resolution == "invalid" - assert resolved.nested[("data_type",)].definition is INT8_DATA_TYPE + assert isinstance(resolved, Refused) + assert resolved.nested[("data_type",)] == INT8 @pytest.mark.parametrize( @@ -254,92 +261,229 @@ def test_a_reading_says_the_name_the_field_was_written_with( assert resolve(field, kind, CORE_AND_EXTENSIONS)[0].name == name -INT8 = resolve("int8", DataTypeDefinition, CORE)[0] -"""An `int8` field as `CORE` reads it.""" - - -def test_a_reading_built_by_hand_is_of_the_kind_it_says() -> None: - # Type arguments dropped, as `resolve` drops them. - built = Resolved("int8", "read", INT8_DATA_TYPE, {}, read_as=DataTypeDefinition[Any]) - assert built.read_as is DataTypeDefinition - # A name read by the definition filed under another, as raw bits are. - raw = Resolved("r16", "read", RAW_BYTES_DATA_TYPE, {"bits": 16}, read_as=DataTypeDefinition) - assert (raw.name, raw.definition) == ("r16", RAW_BYTES_DATA_TYPE) +KINDLESS: Final = Definition(name="acme.kindless", configuration=Empty) +"""A definition of no kind, which no scope files.""" @pytest.mark.parametrize( - ("json", "resolution", "definition", "configuration", "nested"), + ("built", "read_as"), [ - ("int8", "read", None, {}, {}), - ("int8", "read", INT8_DATA_TYPE, None, {}), - ("int8", "invalid", INT8_DATA_TYPE, {}, {}), - ("int8", "out_of_scope", INT8_DATA_TYPE, None, {}), - (5, "out_of_scope", None, None, {}), - ("acme.t", "out_of_scope", None, None, {("a",): INT8}), - ], - ids=[ - "read-by-nothing", - "read-without-configuration", - "unread-with-one", - "claimed-out-of-scope", - "out-of-scope-without-a-name", - "out-of-scope-holding-a-field-read", + (INT8, DataTypeDefinition), + # A name read by the definition filed under another, as raw bits are. + ( + Read( + json="r16", name="r16", definition=RAW_BYTES_DATA_TYPE, configuration={"bits": 16} + ), + DataTypeDefinition, + ), + # Type arguments dropped, as `resolve` drops them. + ( + Unclaimed(json="acme.t", name="acme.t", read_as=DataTypeDefinition[Any]), + DataTypeDefinition, + ), + ( + Refused( + json="int8", + name="int8", + read_as=DataTypeDefinition[Any], + definition=INT8_DATA_TYPE, + ), + DataTypeDefinition, + ), + # Not a field, so named nothing and claimed by nothing. + (Refused(json=5, name=None, read_as=CodecDefinition), CodecDefinition), ], + ids=["read", "raw-bits", "unclaimed", "refused", "refused-nameless"], ) -def test_error_a_reading_built_by_hand_that_its_resolution_contradicts( - json: JSONValue, - resolution: Resolution, - definition: DataTypeDefinition[Any] | None, - configuration: dict[str, JSONValue] | None, - nested: Nested, +def test_a_field_built_by_hand_is_of_the_kind_it_says( + built: Read[Any] | Unclaimed | Refused[Any], read_as: type[Definition[Any]] ) -> None: - with pytest.raises(TypeError, match="a field"): - Resolved(json, resolution, definition, configuration, nested, read_as=DataTypeDefinition) + assert built.read_as is read_as -def test_error_a_reading_built_by_hand_by_a_definition_filed_under_another_name() -> None: +@pytest.mark.parametrize("definition", [None, KINDLESS], ids=["none", "kindless"]) +def test_error_a_field_read_by_hand_by_what_is_not_a_definition_of_a_kind( + definition: Definition[Any] | None, +) -> None: + with pytest.raises(TypeError, match="a field read is read by a definition of a kind"): + Read( + json="acme.kindless", + name="acme.kindless", + definition=definition, # pyright: ignore[reportArgumentType] + configuration={}, + ) + + +@pytest.mark.parametrize("name", ["int16", "r16"]) +def test_error_a_field_read_by_hand_by_a_definition_filed_under_another_name(name: str) -> None: + with pytest.raises( + TypeError, match=f"a field named '{name}' is read by the definition filed under it" + ): + Read(json=name, name=name, definition=INT8_DATA_TYPE, configuration={}) + + +def test_error_a_field_refused_by_hand_by_a_definition_of_another_kind() -> None: + with pytest.raises( + TypeError, match="read as a DataTypeDefinition is read by one, got CodecDefinition" + ): + Refused(json="gzip", name="gzip", read_as=DataTypeDefinition, definition=GZIP_CODEC) + + +@pytest.mark.parametrize("name", ["int16", None]) +def test_error_a_field_refused_by_hand_by_a_definition_filed_under_another_name( + name: str | None, +) -> None: with pytest.raises( - TypeError, match="a field named 'int16' is read by the definition filed under it" + TypeError, match=f"a field named {name!r} is read by the definition filed under it" ): - Resolved("int16", "read", INT8_DATA_TYPE, {}, read_as=DataTypeDefinition) + Refused(json=name, name=name, read_as=DataTypeDefinition, definition=INT8_DATA_TYPE) -def test_error_a_reading_built_by_hand_of_no_kind() -> None: +@pytest.mark.parametrize( + "build", + [ + lambda: Unclaimed(json="acme.t", name="acme.t", read_as=Definition), + lambda: Refused(json="acme.t", name="acme.t", read_as=Definition), + ], + ids=["unclaimed", "refused"], +) +def test_error_a_field_built_by_hand_of_no_kind(build: Callable[[], object]) -> None: with pytest.raises(TypeError, match="is not a kind of metadata"): - Resolved("int8", "read", None, None, read_as=Definition) + build() -def test_error_a_reading_built_by_hand_by_a_definition_of_another_kind() -> None: - with pytest.raises(TypeError, match="read as a DataTypeDefinition is read by one, got a Codec"): - Resolved("gzip", "read", GZIP_CODEC, {"level": 1}, read_as=DataTypeDefinition) +def test_error_a_field_nothing_claims_built_by_hand_without_a_name() -> None: + with pytest.raises(TypeError, match="a field nothing in scope claims is named"): + Unclaimed(json=5, name=None, read_as=CodecDefinition) # pyright: ignore[reportArgumentType] + + +@pytest.mark.parametrize( + ("field", "kind", "written"), + [ + # A data type with nothing to configure by its bare name, however + # it was written; every other field an object. + ("int8", DataTypeDefinition, "int8"), + ( + {"name": "int8", "configuration": {}, "must_understand": True}, + DataTypeDefinition, + "int8", + ), + ("crc32c", CodecDefinition, {"name": "crc32c"}), + ("default", ChunkKeyEncodingDefinition, {"name": "default"}), + ( + {"name": "regular", "configuration": {"chunk_shape": [2, 3]}}, + ChunkGridDefinition, + {"name": "regular", "configuration": {"chunk_shape": (2, 3)}}, + ), + # Each field a configuration holds written the same way. + ( + { + "name": "sharding_indexed", + "configuration": { + "chunk_shape": [2], + "codecs": ["bytes"], + "index_codecs": ["bytes", {"name": "crc32c", "configuration": {}}], + }, + }, + CodecDefinition, + { + "name": "sharding_indexed", + "configuration": { + "chunk_shape": (2,), + "codecs": ({"name": "bytes"},), + "index_codecs": ({"name": "bytes"}, {"name": "crc32c"}), + }, + }, + ), + ( + {"name": "cast_value", "configuration": {"data_type": {"name": "int8"}}}, + CodecDefinition, + {"name": "cast_value", "configuration": {"data_type": "int8"}}, + ), + # One nothing in scope claims inside it too; one refused as written. + ( + { + "name": "acme.stack", + "configuration": { + "codecs": [ + "zfpy", + {"name": "gzip", "configuration": {"level": 12}, "must_understand": True}, + ] + }, + }, + CodecDefinition, + { + "name": "acme.stack", + "configuration": { + "codecs": ( + {"name": "zfpy"}, + {"name": "gzip", "configuration": {"level": 12}, "must_understand": True}, + ) + }, + }, + ), + # A name that carries its configuration is written alone. + ("r16", DataTypeDefinition, "r16"), + ({"name": "r16", "configuration": {}}, DataTypeDefinition, "r16"), + # Nothing in scope claims it: its configuration as written. + ("zfpy", CodecDefinition, {"name": "zfpy"}), + ( + {"name": "zfpy", "configuration": {"x": [1]}}, + CodecDefinition, + {"name": "zfpy", "configuration": {"x": (1,)}}, + ), + ({"name": "acme.t", "configuration": {}}, DataTypeDefinition, "acme.t"), + ], + ids=[ + "data-type", + "data-type-spelled-out", + "codec", + "chunk-key-encoding", + "configured", + "nested", + "nested-data-type", + "nested-unclaimed-and-refused", + "raw-bits", + "raw-bits-spelled-out", + "unclaimed-codec", + "unclaimed-configured", + "unclaimed-data-type", + ], +) +def test_a_field_is_written_as_every_reader_takes_it( + field: JSONValue, kind: type[Definition[Any]], written: JSONValue +) -> None: + resolved, _ = resolve(field, kind, SCOPE) + assert isinstance(resolved, (Read, Unclaimed)) + assert resolved.to_json() == written @pytest.mark.parametrize( - ("field", "kind", "resolution", "configuration", "problems"), + ("field", "kind", "variant", "configuration", "problems"), [ ( {"name": "gzip", "configuration": {"level": 5}}, CodecDefinition, - "read", + Read, {"level": 5}, [], ), - ("crc32c", CodecDefinition, "read", {}, []), - ({"name": "crc32c"}, CodecDefinition, "read", {}, []), - ({"name": "crc32c", "configuration": {}}, CodecDefinition, "read", {}, []), - ({"name": "crc32c", "must_understand": True}, CodecDefinition, "read", {}, []), - ("bytes", CodecDefinition, "read", {}, []), + ("crc32c", CodecDefinition, Read, {}, []), + ({"name": "crc32c"}, CodecDefinition, Read, {}, []), + ({"name": "crc32c", "configuration": {}}, CodecDefinition, Read, {}, []), + ({"name": "crc32c", "must_understand": True}, CodecDefinition, Read, {}, []), + ("bytes", CodecDefinition, Read, {}, []), ( {"name": "bytes", "configuration": {"endian": "big"}}, CodecDefinition, - "read", + Read, {"endian": "big"}, [], ), ( {"name": "regular", "configuration": {"chunk_shape": [2, 3]}}, ChunkGridDefinition, - "read", + Read, {"chunk_shape": (2, 3)}, [], ), @@ -348,7 +492,7 @@ def test_error_a_reading_built_by_hand_by_a_definition_of_another_kind() -> None ( {"name": "regular", "configuration": {"chunk_shape": [0, 3]}}, ChunkGridDefinition, - "read", + Read, {"chunk_shape": (0, 3)}, [], ), @@ -357,7 +501,7 @@ def test_error_a_reading_built_by_hand_by_a_definition_of_another_kind() -> None ( {"name": "gzip", "configuration": {"level": 5, "extra": 1}}, CodecDefinition, - "read", + Read, {"level": 5}, [(("configuration", "extra"), "unknown_key")], ), @@ -365,19 +509,19 @@ def test_error_a_reading_built_by_hand_by_a_definition_of_another_kind() -> None ( {"name": "acme.bounded", "configuration": {"level": 5, "windw": 10}}, CodecDefinition, - "read", + Read, {"level": 5}, [(("configuration", "windw"), "unknown_key")], ), # A key that is not required, written as a string under postponed # annotations, and every key of a `total=False` TypedDict. - ("acme.level", CodecDefinition, "read", {}, []), - ("acme.total", CodecDefinition, "read", {}, []), + ("acme.level", CodecDefinition, Read, {}, []), + ("acme.total", CodecDefinition, Read, {}, []), # A kind with type arguments is that kind. ( {"name": "gzip", "configuration": {"level": 5}}, CodecDefinition[Any], - "read", + Read, {"level": 5}, [], ), @@ -387,7 +531,7 @@ def test_error_a_reading_built_by_hand_by_a_definition_of_another_kind() -> None "configuration": {"label": "a", "children": [{"label": "b", "children": []}]}, }, CodecDefinition, - "read", + Read, {"label": "a", "children": ({"label": "b", "children": ()},)}, [], ), @@ -399,26 +543,28 @@ def test_error_a_reading_built_by_hand_by_a_definition_of_another_kind() -> None "configuration": {"fallback": {"codec": {"name": "gzip"}, "note": "x"}}, }, CodecDefinition, - "read", + Read, {"fallback": {"codec": {"name": "gzip"}, "note": "x"}}, [], ), - # `extra_items` of a field alias: every other key holds a codec. + # `extra_items` of a field alias: every other key holds a codec, + # held as a document writes it. ( {"name": "acme.routes", "configuration": {"fast": "crc32c"}}, CodecDefinition, - "read", - {"fast": "crc32c"}, + Read, + {"fast": {"name": "crc32c"}}, [], ), # Nothing in scope claims it: left unjudged, not refused. - ({"name": "zfpy", "configuration": {"x": 1}}, CodecDefinition, "out_of_scope", None, []), - # A nested field is read in the same scope; one out of scope is left be. + ({"name": "zfpy", "configuration": {"x": 1}}, CodecDefinition, Unclaimed, None, []), + # A nested field is read in the same scope, and held as a document + # writes it; one out of scope is left be. ( {"name": "acme.stack", "configuration": {"codecs": ["crc32c", "zfpy"]}}, CodecDefinition, - "read", - {"codecs": ("crc32c", "zfpy")}, + Read, + {"codecs": ({"name": "crc32c"}, {"name": "zfpy"})}, [], ), ], @@ -447,13 +593,13 @@ def test_error_a_reading_built_by_hand_by_a_definition_of_another_kind() -> None def test_a_field_is_read_in_scope( field: object, kind: type[Definition[Any]], - resolution: str, + variant: type, configuration: dict[str, object] | None, problems: list[tuple[tuple[str | int, ...], str]], ) -> None: resolved, found = resolve(field, kind, SCOPE) - assert resolved.resolution == resolution - assert resolved.configuration == configuration + assert type(resolved) is variant + assert (resolved.configuration if isinstance(resolved, Read) else None) == configuration assert _locs(found) == problems @@ -461,8 +607,8 @@ def test_error_a_rule_refuses_a_value() -> None: resolved, found = resolve( {"name": "gzip", "configuration": {"level": 12}}, CodecDefinition, SCOPE ) - assert resolved.resolution == "invalid" - assert resolved.configuration is None + assert isinstance(resolved, Refused) + assert resolved.definition is GZIP_CODEC assert _locs(found) == [(("configuration", "level"), "invalid_value")] @@ -470,20 +616,25 @@ def test_error_a_member_of_the_wrong_type_is_not_asked_of_the_rules() -> None: resolved, found = resolve( {"name": "gzip", "configuration": {"level": "5"}}, CodecDefinition, SCOPE ) - assert resolved.resolution == "invalid" + assert isinstance(resolved, Refused) assert _locs(found) == [(("configuration", "level"), "invalid_type")] def test_error_a_required_configuration_is_missing() -> None: resolved, found = resolve("gzip", CodecDefinition, SCOPE) - assert resolved.resolution == "invalid" + assert isinstance(resolved, Refused) assert _locs(found) == [(("configuration",), "missing_key")] def test_error_the_configuration_is_not_an_object() -> None: - # Unread, and still claimed by the definition its name names. + # Refused, and still claimed by the definition its name names. resolved, found = resolve({"name": "gzip", "configuration": 5}, CodecDefinition, SCOPE) - assert (resolved.resolution, resolved.definition) == ("invalid", GZIP_CODEC) + assert resolved == Refused( + json={"name": "gzip", "configuration": 5}, + name="gzip", + read_as=CodecDefinition, + definition=GZIP_CODEC, + ) assert [found.loc for found in found] == [("configuration",)] @@ -491,7 +642,7 @@ def test_error_must_understand_false_is_refused() -> None: # The envelope's problem, reported with the field; the configuration # was read, so a later layer can still judge the codec. resolved, found = resolve({"name": "crc32c", "must_understand": False}, CodecDefinition, SCOPE) - assert resolved.resolution == "read" + assert isinstance(resolved, Read) assert _locs(found) == [(("must_understand",), "invalid_value")] @@ -579,22 +730,28 @@ def test_error_null_is_not_a_field() -> None: # not a verdict. Read as a field or checked as a configuration, it is # a value of the wrong type. resolved, found = resolve(None, CodecDefinition, SCOPE) - assert (resolved.resolution, _locs(found)) == ("invalid", [((), "invalid_type")]) + assert (resolved, _locs(found)) == ( + Refused(json=None, name=None, read_as=CodecDefinition), + [((), "invalid_type")], + ) configuration, found = GZIP_CODEC.judge(None) assert (configuration, _locs(found)) == (None, [((), "invalid_type")]) def test_error_a_value_that_is_not_json() -> None: + # Held as None, not JSON; its name still says what claims it. resolved, found = resolve( {"name": "gzip", "configuration": {"level": math.nan}}, CodecDefinition, SCOPE ) - assert (resolved.json, resolved.resolution) == (None, "invalid") + assert resolved == Refused( + json=None, name="gzip", read_as=CodecDefinition, definition=GZIP_CODEC + ) assert _locs(found) == [(("configuration", "level"), "invalid_value")] def test_error_a_value_that_is_not_a_field() -> None: resolved, found = resolve(5, CodecDefinition, SCOPE) - assert resolved.resolution == "invalid" + assert resolved == Refused(json=5, name=None, read_as=CodecDefinition) assert len(found) == 1 @@ -602,7 +759,7 @@ def test_error_a_regular_grid_extent_is_negative() -> None: resolved, found = resolve( {"name": "regular", "configuration": {"chunk_shape": [2, -1]}}, ChunkGridDefinition, SCOPE ) - assert resolved.resolution == "invalid" + assert isinstance(resolved, Refused) assert _locs(found) == [(("configuration", "chunk_shape", 1), "invalid_value")] @@ -614,8 +771,8 @@ def test_error_a_nested_field_is_judged_where_it_sits() -> None: "configuration": {"codecs": ["crc32c", {"name": "gzip", "configuration": {"level": 12}}]}, } resolved, found = resolve(field, CodecDefinition, SCOPE) - assert resolved.resolution == "read" - assert resolved.nested[("codecs", 1)].resolution == "invalid" + assert isinstance(resolved, Read) + assert isinstance(resolved.nested[("codecs", 1)], Refused) assert _locs(found) == [ (("configuration", "codecs", 1, "configuration", "level"), "invalid_value") ] @@ -627,8 +784,8 @@ def test_error_a_nested_field_in_extra_items_is_judged_where_it_sits() -> None: "configuration": {"slow": {"name": "gzip", "configuration": {"level": 12}}}, } resolved, found = resolve(field, CodecDefinition, SCOPE) - assert resolved.resolution == "read" - assert resolved.nested[("slow",)].resolution == "invalid" + assert isinstance(resolved, Read) + assert isinstance(resolved.nested[("slow",)], Refused) assert _locs(found) == [(("configuration", "slow", "configuration", "level"), "invalid_value")] @@ -639,7 +796,7 @@ def test_error_a_nested_envelope_s_problem_is_its_own_and_the_rules_are_asked() resolved, found = resolve( {"name": "acme.stack", "configuration": {"codecs": [unread]}}, CodecDefinition, SCOPE ) - assert resolved.resolution == "read" + assert isinstance(resolved, Read) assert _locs(found) == [(("configuration", "codecs", 0, "must_understand"), "invalid_value")] # Its rules are asked all the same: one refuses the array -> bytes # codec it holds, which is the stack's own problem. @@ -648,7 +805,7 @@ def test_error_a_nested_envelope_s_problem_is_its_own_and_the_rules_are_asked() CodecDefinition, SCOPE, ) - assert resolved.resolution == "invalid" + assert isinstance(resolved, Refused) assert _locs(found) == [ (("configuration", "codecs", 1), "invalid_value"), (("configuration", "codecs", 0, "must_understand"), "invalid_value"), @@ -660,7 +817,7 @@ def test_error_a_container_rule_is_not_asked_of_a_malformed_nested_field() -> No # reported where it sits, and the rule is not asked, as `judge` would not. field = {"name": "acme.stack", "configuration": {"codecs": [{"configuration": {}}]}} resolved, found = resolve(field, CodecDefinition, SCOPE) - assert resolved.resolution == "invalid" + assert isinstance(resolved, Refused) assert _locs(found) == [(("configuration", "codecs", 0, "name"), "invalid_type")] assert ACME_STACK.judge(field["configuration"])[0] is None @@ -682,7 +839,7 @@ def test_error_a_rule_reads_the_fields_the_configuration_holds_as_the_scope_read # nothing read. field = {"name": "acme.stack", "configuration": {"codecs": ["crc32c", "bytes"]}} resolved, found = resolve(field, CodecDefinition, SCOPE) - assert resolved.resolution == "invalid" + assert isinstance(resolved, Refused) assert _locs(found) == [(("configuration", "codecs", 1), "invalid_value")] assert ACME_STACK.judge(field["configuration"])[1] == () @@ -691,7 +848,7 @@ def test_error_a_nested_member_is_not_a_field() -> None: resolved, found = resolve( {"name": "acme.stack", "configuration": {"codecs": [5]}}, CodecDefinition, SCOPE ) - assert resolved.resolution == "invalid" + assert isinstance(resolved, Refused) assert _locs(found) == [(("configuration", "codecs", 0), "invalid_type")] @@ -773,6 +930,86 @@ def test_a_scope_is_shown_by_how_many_definitions_it_holds() -> None: assert repr(CORE) == f"Context(<{len(CORE.definitions())} definitions>)" +def test_a_scope_pickles_as_its_definitions_and_copies_as_itself() -> None: + # A model holds the scope it was read in, and goes to another process with it. + again = pickle.loads(pickle.dumps(CORE_AND_EXTENSIONS)) + assert again.definitions() == CORE_AND_EXTENSIONS.definitions() + assert copy.copy(CORE) is CORE + assert copy.deepcopy(CORE) is CORE + + +LE: Final = {"name": "bytes", "configuration": {"endian": "little"}} +SHARD: Final = {"chunk_shape": [2], "codecs": [LE], "index_location": "end"} + + +@pytest.mark.parametrize( + ("one", "other", "kind", "equal"), + [ + ("uint8", {"name": "uint8"}, DataTypeDefinition, True), + ({"name": "bytes"}, {"name": "bytes", "configuration": {}}, CodecDefinition, True), + ( + {"name": "gzip", "configuration": {"level": 1}}, + {"name": "gzip", "configuration": {"level": 1}, "must_understand": True}, + CodecDefinition, + True, + ), + ("acme.codec", {"name": "acme.codec", "configuration": {}}, CodecDefinition, True), + ( + { + "name": "sharding_indexed", + "configuration": {**SHARD, "index_codecs": [LE, "crc32c"]}, + }, + { + "name": "sharding_indexed", + "configuration": {**SHARD, "index_codecs": [LE, {"name": "crc32c"}]}, + }, + CodecDefinition, + True, + ), + ( + {"name": "gzip", "configuration": {"level": 1}}, + {"name": "gzip", "configuration": {"level": 2}}, + CodecDefinition, + False, + ), + ( + {"name": "acme.codec", "configuration": {"a": 1}}, + {"name": "acme.codec", "configuration": {"a": 2}}, + CodecDefinition, + False, + ), + ("int8", "uint8", DataTypeDefinition, False), + ], + ids=[ + "bare-or-object", + "no-or-empty-configuration", + "must-understand-true-or-absent", + "unclaimed-bare-or-object", + "a-field-it-holds-bare-or-object", + "another-configuration", + "unclaimed-another-configuration", + "another-name", + ], +) +def test_two_fields_are_equal_when_they_read_the_same( + one: JSONValue, other: JSONValue, kind: type[Definition[Any]], equal: bool +) -> None: + """However each was spelled; each is written as it reads, and equal fields are written the same.""" + first, _ = resolve(one, kind, CORE_AND_EXTENSIONS) + second, _ = resolve(other, kind, CORE_AND_EXTENSIONS) + assert isinstance(first, (Read, Unclaimed)) + assert isinstance(second, (Read, Unclaimed)) + assert (first == second) is equal + assert (first.to_json() == second.to_json()) is equal + + +def test_a_field_copied_or_pickled_is_read_by_a_definition_equal_to_its_own() -> None: + field, _ = resolve({"name": "gzip", "configuration": {"level": 1}}, CodecDefinition, CORE) + for again in (pickle.loads(pickle.dumps(field)), copy.deepcopy(field)): + assert again == field + assert configuration_of(again, GZIP_CODEC) == {"level": 1} + + def test_a_definition_is_shown_by_its_kind_and_name() -> None: # Short, as a reading that holds it shows it. assert repr(GZIP_CODEC) == "CodecDefinition(name='gzip')" diff --git a/packages/zarr-metadata/tests/v3/test_every_definition.py b/packages/zarr-metadata/tests/v3/test_every_definition.py index c4214b1cf9..279521aa98 100644 --- a/packages/zarr-metadata/tests/v3/test_every_definition.py +++ b/packages/zarr-metadata/tests/v3/test_every_definition.py @@ -23,6 +23,9 @@ Context, DataTypeDefinition, Definition, + Read, + Refused, + Unclaimed, ValidationProblem, canonicalize, configuration_of, @@ -139,9 +142,10 @@ CASES = [(key, field) for key, fields in EXAMPLES.items() for field in fields] -def _read(key: str, field: object) -> tuple[str, list[tuple[tuple[str | int, ...], str]]]: +def _read(key: str, field: object) -> tuple[type, list[tuple[tuple[str | int, ...], str]]]: + """What the scope made of `field` -- `Read`, `Unclaimed` or `Refused` -- and where each problem is.""" resolved, problems = resolve(field, KINDS[key.split(":")[0]], CORE_AND_EXTENSIONS) - return resolved.resolution, [(found.loc, found.kind) for found in problems] + return type(resolved), [(found.loc, found.kind) for found in problems] def _problems(key: str, field: object) -> list[tuple[tuple[str | int, ...], str]]: @@ -170,7 +174,7 @@ def test_every_definition_in_scope_has_an_example() -> None: def test_every_example_reads_and_its_simplest_spelling_is_stable(key: str, field: object) -> None: # Read, with nothing wrong; and its simplest spelling reads the same, # and is its own simplest spelling. - assert _read(key, field) == ("read", []) + assert _read(key, field) == (Read, []) kind = KINDS[key.split(":")[0]] simplest, problems = canonicalize(field, kind, CORE_AND_EXTENSIONS) assert problems == () @@ -283,7 +287,8 @@ def test_raw_bits_read_as_r_star_with_the_size_their_name_carries( field: object, bits: int, simplest: str ) -> None: resolved, problems = resolve(field, DataTypeDefinition, CORE_AND_EXTENSIONS) - assert (resolved.resolution, problems) == ("read", ()) + assert problems == () + assert isinstance(resolved, Read) assert resolved.definition is RAW_BYTES_DATA_TYPE assert resolved.json == field assert configuration_of(resolved, RAW_BYTES_DATA_TYPE) == {"bits": bits} @@ -296,10 +301,13 @@ def test_a_reader_reads_raw_bits_its_own_way_by_defining_r_star() -> None: mine = DataTypeDefinition(name="r*", configuration=RawBytesConfiguration) scope = CORE_AND_EXTENSIONS.extended_with(mine) resolved, problems = resolve("r12", DataTypeDefinition, scope) + assert problems == () + assert isinstance(resolved, Read) assert resolved.definition is mine - assert (resolved.resolution, problems) == ("read", ()) - again = scope.extended_with(RAW_BYTES_DATA_TYPE) - assert resolve("r12", DataTypeDefinition, again)[0].definition is RAW_BYTES_DATA_TYPE + # The package's own takes no 12 bits, and refuses them. + again, _ = resolve("r12", DataTypeDefinition, scope.extended_with(RAW_BYTES_DATA_TYPE)) + assert isinstance(again, Refused) + assert again.definition is RAW_BYTES_DATA_TYPE @pytest.mark.parametrize("field", ["r*", {"name": "r*", "configuration": {"bits": 16}}]) @@ -307,11 +315,34 @@ def test_r_star_is_notation_that_names_nothing(field: object) -> None: # How the specification's table writes raw bits, and no document's name # for them: read as any name nothing in scope claims. resolved, problems = resolve(field, DataTypeDefinition, CORE_AND_EXTENSIONS) - assert (resolved.resolution, resolved.definition, problems) == ("out_of_scope", None, ()) + assert (type(resolved), problems) == (Unclaimed, ()) -def test_an_unclaimed_field_keeps_its_own_spelling() -> None: - assert canonicalize({"name": "zfpy"}, CodecDefinition, CORE) == ({"name": "zfpy"}, ()) +@pytest.mark.parametrize( + ("field", "kind", "simplest"), + [ + ({"name": "zfpy"}, CodecDefinition, {"name": "zfpy"}), + # What it simplifies to is its own definition's call, so its + # configuration is kept as written; its envelope is its kind's. + ( + {"name": "zfpy", "configuration": {"level": [1]}}, + CodecDefinition, + {"name": "zfpy", "configuration": {"level": (1,)}}, + ), + ("zfpy", CodecDefinition, {"name": "zfpy"}), + ( + {"name": "zfpy", "configuration": {}, "must_understand": True}, + CodecDefinition, + {"name": "zfpy"}, + ), + ({"name": "acme.decimal"}, DataTypeDefinition, "acme.decimal"), + ], + ids=["object", "configured", "bare-codec", "spelled-out", "data-type"], +) +def test_an_unclaimed_field_keeps_its_configuration_in_its_kind_s_envelope( + field: object, kind: type[Definition[Any]], simplest: object +) -> None: + assert canonicalize(field, kind, CORE) == (simplest, ()) @pytest.mark.parametrize( @@ -647,7 +678,7 @@ def test_error_a_configuration_written_beside_a_raw_bits_name() -> None: # The name carries the configuration, so one written beside it holds # nothing: each member is a key nothing declares. assert _read("data_type:r*", {"name": "r16", "configuration": {"bits": 16}}) == ( - "read", + Read, [(("configuration", "bits"), "unknown_key")], ) diff --git a/packages/zarr-metadata/tests/v3/test_fill_values.py b/packages/zarr-metadata/tests/v3/test_fill_values.py index c7b719338b..8757d02e62 100644 --- a/packages/zarr-metadata/tests/v3/test_fill_values.py +++ b/packages/zarr-metadata/tests/v3/test_fill_values.py @@ -23,7 +23,7 @@ EmptyConfiguration, JSONValue, Nested, - Resolved, + Read, ValidationProblem, fill_value_problems, resolve, @@ -285,5 +285,7 @@ def deep(levels: int) -> dict[str, object]: def test_a_struct_read_without_its_field_types_leaves_its_fields_unjudged() -> None: # A reading built by hand, holding no field type's reading. configuration = {"fields": ({"name": "a", "data_type": "int8"},)} - struct = Resolved(STRUCT, "read", STRUCT_DATA_TYPE, configuration, read_as=DataTypeDefinition) + struct = Read( + json=STRUCT, name="struct", definition=STRUCT_DATA_TYPE, configuration=configuration + ) assert fill_value_problems(struct, {"a": 300}) == () From 94271efedd1400da2652ca95e7b9d3209d63f91b Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Mon, 28 Sep 2026 22:15:24 +0200 Subject: [PATCH 05/94] feat(zarr-metadata): a zarr.json is read as the node its node_type says, as a discriminated union reads its tag - `read_node_metadata_v3(value, *, context)` reads a v3 `zarr.json` of either kind: as `read_array_metadata_v3` or `read_group_metadata_v3` reads it, as its `node_type` says -- the tag of a union, as pydantic's discriminator and zod's discriminated union read one -- or, when it says neither or is not an object, as `ZarrV3UnknownNodeReading`, with the problem, reading nothing else of it. `ZarrV3NodeMetadataReading` is the three; `validate_node_metadata_v3` gives the problems. - `node_metadata_from_json_v3` and `node_metadata_from_key_value_v3` build the model of either kind, `ZarrV3NodeMetadata`, as the models' own `from_json` and `from_key_value` build one, and as pydantic's `TypeAdapter` validates a discriminated union from Python and JSON. - A group's consolidated metadata reads each document it holds as a node, so one of no node type is in `reading.consolidated`, as unknown, rather than missing. Its problem is located as before, and reads as a missing key when the entry has no `node_type`, and at the entry when it is not an object. Assisted-by: ClaudeCode:claude-opus-5-5 --- packages/zarr-metadata/README.md | 8 +- packages/zarr-metadata/changes/374.feature.md | 16 ++ packages/zarr-metadata/docs/index.md | 8 +- .../src/zarr_metadata/model/__init__.py | 14 ++ .../src/zarr_metadata/model/_group.py | 167 ++++++++++++++---- .../zarr-metadata/tests/model/test_group.py | 109 +++++++++++- 6 files changed, 282 insertions(+), 40 deletions(-) create mode 100644 packages/zarr-metadata/changes/374.feature.md diff --git a/packages/zarr-metadata/README.md b/packages/zarr-metadata/README.md index bb677345ea..5179986f96 100644 --- a/packages/zarr-metadata/README.md +++ b/packages/zarr-metadata/README.md @@ -115,7 +115,13 @@ metadata = reading.metadata # None when reading.problems is not empty ``` `read_group_metadata_v3` reads a group the same way, and each document -its consolidated metadata holds once. +its consolidated metadata holds once. `read_node_metadata_v3` reads a +`zarr.json` of either kind as the node its `node_type` says it is, as +a discriminated union reads its tag: a document that says neither reads +as `ZarrV3UnknownNodeReading`, with the problem, and nothing else of it +is read. `node_metadata_from_json_v3` and `node_metadata_from_key_value_v3` +build the model of either kind, as the models' own `from_json` and +`from_key_value` build one. A member the spec does not define is not a field; the model's `must_understand_fields` names those a reader must understand. diff --git a/packages/zarr-metadata/changes/374.feature.md b/packages/zarr-metadata/changes/374.feature.md new file mode 100644 index 0000000000..2ba397eb40 --- /dev/null +++ b/packages/zarr-metadata/changes/374.feature.md @@ -0,0 +1,16 @@ +`read_node_metadata_v3` reads a v3 `zarr.json` of either kind as the node +its `node_type` says it is -- the tag of a union, as pydantic's +discriminator and zod's discriminated union read one: an array as +`read_array_metadata_v3` reads it, a group as `read_group_metadata_v3` +does, and a document whose `node_type` says neither, or that is not an +object, as `ZarrV3UnknownNodeReading`, with the problem, nothing else of +it read. `ZarrV3NodeMetadataReading` is the three, and +`validate_node_metadata_v3` gives the problems. +`node_metadata_from_json_v3` and `node_metadata_from_key_value_v3` build +the model of either kind, `ZarrV3NodeMetadata`, as the models' own +`from_json` and `from_key_value` build one, so a node is read from a +store's bytes without its `node_type` read first. A group's consolidated +metadata reads each document it holds so: one of no node type is in +`reading.consolidated`, where it was missing, an entry with no +`node_type` has a `missing_key` problem, and an entry that is not an +object has an `invalid_type` problem at the entry. diff --git a/packages/zarr-metadata/docs/index.md b/packages/zarr-metadata/docs/index.md index 881459bd68..c6fe748954 100644 --- a/packages/zarr-metadata/docs/index.md +++ b/packages/zarr-metadata/docs/index.md @@ -130,7 +130,13 @@ metadata = reading.metadata # None when reading.problems is not empty ``` `read_group_metadata_v3` reads a group the same way, and each document -its consolidated metadata holds once. +its consolidated metadata holds once. `read_node_metadata_v3` reads a +`zarr.json` of either kind as the node its `node_type` says it is, as +a discriminated union reads its tag: a document that says neither reads +as `ZarrV3UnknownNodeReading`, with the problem, and nothing else of it +is read. `node_metadata_from_json_v3` and `node_metadata_from_key_value_v3` +build the model of either kind, as the models' own `from_json` and +`from_key_value` build one. A member the spec does not define is not a field; the model's `must_understand_fields` names those a reader must understand. diff --git a/packages/zarr-metadata/src/zarr_metadata/model/__init__.py b/packages/zarr-metadata/src/zarr_metadata/model/__init__.py index f387d56352..0d9195ff9c 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/__init__.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/__init__.py @@ -42,10 +42,17 @@ ZarrV3GroupMetadata, ZarrV3GroupMetadataReading, ZarrV3GroupMetadataUpdate, + ZarrV3NodeMetadata, + ZarrV3NodeMetadataReading, + ZarrV3UnknownNodeReading, is_group_metadata_v3, + node_metadata_from_json_v3, + node_metadata_from_key_value_v3, parse_group_metadata_v3, read_group_metadata_v3, + read_node_metadata_v3, validate_group_metadata_v3, + validate_node_metadata_v3, ) from zarr_metadata.model._sentinel import UNSET from zarr_metadata.model._validation import ( @@ -141,12 +148,17 @@ "ZarrV3GroupMetadataReading", "ZarrV3GroupMetadataStoreKey", "ZarrV3GroupMetadataUpdate", + "ZarrV3NodeMetadata", + "ZarrV3NodeMetadataReading", + "ZarrV3UnknownNodeReading", "is_array_metadata_v2", "is_array_metadata_v3", "is_group_metadata_v2", "is_group_metadata_v3", "is_json", "is_metadata_field_v3", + "node_metadata_from_json_v3", + "node_metadata_from_key_value_v3", "parse_array_metadata_v2", "parse_array_metadata_v3", "parse_group_metadata_v2", @@ -155,10 +167,12 @@ "parse_metadata_field_v3", "read_array_metadata_v3", "read_group_metadata_v3", + "read_node_metadata_v3", "validate_array_metadata_v2", "validate_array_metadata_v3", "validate_group_metadata_v2", "validate_group_metadata_v3", "validate_json", "validate_metadata_field_v3", + "validate_node_metadata_v3", ] diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_group.py b/packages/zarr-metadata/src/zarr_metadata/model/_group.py index 3b14ed9d6f..446d8ec900 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_group.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/_group.py @@ -8,12 +8,13 @@ from dataclasses import dataclass, field from typing import TYPE_CHECKING, Any, Final, Literal, TypeGuard, cast -from typing_extensions import TypedDict, Unpack +from typing_extensions import TypeAliasType, TypedDict, Unpack from zarr_metadata._json import ( MetadataValidationError, ValidationProblem, arrays_to_tuples, + choices, copied, is_canonical_json, refine_json, @@ -28,6 +29,7 @@ array_model, held_document, must_understand_subset, + read_array_metadata_v3, ) from zarr_metadata.model._sentinel import UNSET from zarr_metadata.model._validation import ( @@ -325,7 +327,7 @@ def _consolidated_document( } -def _no_documents() -> Mapping[str, ZarrV3ArrayMetadataReading | ZarrV3GroupMetadataReading]: +def _no_documents() -> Mapping[str, ZarrV3NodeMetadataReading]: """What a group whose consolidated metadata holds none, or that has none, holds: nothing.""" return {} @@ -334,10 +336,10 @@ def _no_documents() -> Mapping[str, ZarrV3ArrayMetadataReading | ZarrV3GroupMeta class ZarrV3GroupMetadataReading: """A v3 group document as a scope read it, whatever it holds: each document its consolidated metadata holds, as read, every problem, and the model when there is none.""" - consolidated: Mapping[str, ZarrV3ArrayMetadataReading | ZarrV3GroupMetadataReading] = ( - dataclasses.field(default_factory=_no_documents) + consolidated: Mapping[str, ZarrV3NodeMetadataReading] = dataclasses.field( + default_factory=_no_documents ) - """Each document its consolidated metadata holds, as read, by its path.""" + """Each document its consolidated metadata holds, as `read_node_metadata_v3` reads one, by its path.""" problems: tuple[ValidationProblem, ...] = () """Every reason the document is not a valid one.""" metadata: ZarrV3GroupMetadata | None = None @@ -350,6 +352,122 @@ def fields(self) -> Iterator[tuple[Loc, Resolved[Any]]]: yield (ZARR_V3_CONSOLIDATED_METADATA_KEY, "metadata", path, *loc), node +@dataclass(frozen=True, slots=True) +class ZarrV3UnknownNodeReading: + """A v3 document of no node type the spec defines -- its `node_type` missing, or neither `"array"` nor `"group"` -- or not an object at all: nothing else of it is read, as its problem says.""" + + problems: tuple[ValidationProblem, ...] + """Why it is no node.""" + + @property + def metadata(self) -> None: + """Its model: none, since no node type says which model it is.""" + return None + + def fields(self) -> Iterator[tuple[Loc, Resolved[Any]]]: + """Its fields as read: none, since none of them is read.""" + return iter(()) + + +ZarrV3NodeMetadataReading = TypeAliasType( + "ZarrV3NodeMetadataReading", + "ZarrV3ArrayMetadataReading | ZarrV3GroupMetadataReading | ZarrV3UnknownNodeReading", +) +"""A v3 `zarr.json` as `read_node_metadata_v3` reads it: as the array or group its `node_type` says, or as neither.""" + + +def read_node_metadata_v3( + value: object, *, context: Context = CORE_AND_EXTENSIONS +) -> ZarrV3NodeMetadataReading: + """`value`, a v3 `zarr.json`, read in `context` as the node its `node_type` says it is. + + The node type is the tag of a union, as pydantic's discriminator and + zod's discriminated union read one: an array is read as + `read_array_metadata_v3` reads it, a group as `read_group_metadata_v3` + does, and a document that says neither, or is not an object, is + `ZarrV3UnknownNodeReading`, with the problem. So no caller reads + `node_type` from JSON it has not read. + """ + node_type, problems = _node_type(value) + if node_type == "array": + return read_array_metadata_v3(value, context=context) + if node_type == "group": + return read_group_metadata_v3(value, context=context) + return ZarrV3UnknownNodeReading(problems) + + +ZarrV3NodeMetadata = TypeAliasType( + "ZarrV3NodeMetadata", "ZarrV3ArrayMetadata | ZarrV3GroupMetadata" +) +"""The model of a v3 `zarr.json`: an array's or a group's, as its `node_type` says.""" + + +def node_metadata_from_json_v3( + data: object, *, context: Context = CORE_AND_EXTENSIONS +) -> ZarrV3NodeMetadata: + """The model of `data`, a v3 `zarr.json` read in `context`, as the node its `node_type` says. + + What `ZarrV3ArrayMetadata.from_json` or `ZarrV3GroupMetadata.from_json` + gives, as pydantic's `TypeAdapter` validates a discriminated union. + `MetadataValidationError` with every problem `read_node_metadata_v3` + finds, a `node_type` that says neither among them. + """ + reading = read_node_metadata_v3(data, context=context) + if reading.metadata is None: + raise MetadataValidationError(reading.problems) + return reading.metadata + + +def node_metadata_from_key_value_v3( + mapping: Mapping[StoreKey, bytes], *, context: Context = CORE_AND_EXTENSIONS +) -> ZarrV3NodeMetadata: + """The model of the document at `zarr.json` in `mapping`, read in `context` as the node its `node_type` says, as `node_metadata_from_json_v3` reads one. + + `MetadataValidationError` when the key is missing, its bytes are not + JSON, or the document is not a valid array or group. + """ + # An array's document and a group's are both at `zarr.json`. + document = load_store_json(mapping, ZARR_V3_GROUP_METADATA_STORE_KEY) + return node_metadata_from_json_v3(document, context=context) + + +def validate_node_metadata_v3( + value: object, *, context: Context = CORE_AND_EXTENSIONS +) -> tuple[ValidationProblem, ...]: + """Every reason `value` is not a valid v3 `zarr.json`: those `validate_array_metadata_v3` or `validate_group_metadata_v3` finds in the node its `node_type` says it is, or why it says neither.""" + return _read_node_v3(value, context)[0].problems + + +def _read_node_v3( + value: object, context: Context +) -> tuple[ZarrV3NodeMetadataReading, ArrayMembersV3 | GroupMembersV3 | None]: + """`value` read as `read_node_metadata_v3` reads it, without models, and its members refined.""" + node_type, problems = _node_type(value) + if node_type == "array": + return read_array_v3(value, context) + if node_type == "group": + return read_group_v3(value, context) + return ZarrV3UnknownNodeReading(problems), None + + +_NODE_TYPES: Final = ("array", "group") +"""The node types the spec defines.""" + + +def _node_type(value: object) -> tuple[str | None, tuple[ValidationProblem, ...]]: + """The node type `value` says it is, one of `_NODE_TYPES`; None, with the problem, when it says none of them, or is not an object.""" + if not isinstance(value, Mapping): + return None, (ValidationProblem((), "expected an object", "invalid_type"),) + document = cast("Mapping[object, object]", value) + if "node_type" not in document: + return None, (ValidationProblem(("node_type",), "missing required key", "missing_key"),) + node_type = document["node_type"] + if isinstance(node_type, str) and node_type in _NODE_TYPES: + return node_type, () + message = f"expected {choices(_NODE_TYPES)}, got {shown(node_type)}" + return None, (ValidationProblem(("node_type",), message, refused_kind(node_type, _NODE_TYPES)),) + + @dataclass(frozen=True, slots=True) class GroupMembersV3: """What a read refined of a v3 group document, as the models hold it: its own members, and those of each document its consolidated metadata holds that a model can be built of.""" @@ -409,7 +527,7 @@ def read_group_v3( # is read as none, so those stores stay readable; the model never # writes it back. raw = doc.get(ZARR_V3_CONSOLIDATED_METADATA_KEY) - consolidated: dict[str, ZarrV3ArrayMetadataReading | ZarrV3GroupMetadataReading] = {} + consolidated: dict[str, ZarrV3NodeMetadataReading] = {} held: Mapping[str, ArrayMembersV3 | GroupMembersV3] | UNSET = UNSET if raw is not None: consolidated, held, inside = _read_consolidated_v3(raw, context) @@ -425,11 +543,11 @@ def read_group_v3( def _read_consolidated_v3( value: object, context: Context ) -> tuple[ - dict[str, ZarrV3ArrayMetadataReading | ZarrV3GroupMetadataReading], + dict[str, ZarrV3NodeMetadataReading], dict[str, ArrayMembersV3 | GroupMembersV3], tuple[ValidationProblem, ...], ]: - """An inline `consolidated_metadata` member, as `context` read it: each document it holds, read once, by its path; the members of each a model can be built of; and every problem, located in the member.""" + """An inline `consolidated_metadata` member, as `context` read it: each document it holds, read once, as `read_node_metadata_v3` reads one, by its path; the members of each a model can be built of; and every problem, located in the member.""" if not isinstance(value, Mapping): return {}, {}, (ValidationProblem((), "expected an object", "invalid_type"),) env = cast("Mapping[object, object]", value) @@ -443,7 +561,7 @@ def _read_consolidated_v3( problems.extend(check_literal(env, "kind", "inline")) if "must_understand" in env and env["must_understand"] is not False: problems.append(ValidationProblem(("must_understand",), "expected False", "invalid_value")) - readings: dict[str, ZarrV3ArrayMetadataReading | ZarrV3GroupMetadataReading] = {} + readings: dict[str, ZarrV3NodeMetadataReading] = {} members: dict[str, ArrayMembersV3 | GroupMembersV3] = {} entries = env.get("metadata") if "metadata" in env and not isinstance(entries, Mapping): @@ -455,34 +573,13 @@ def _read_consolidated_v3( ValidationProblem(("metadata",), f"non-string key {key!r}", "invalid_type") ) continue - node_type = _node_type(entry) - child: ArrayMembersV3 | GroupMembersV3 | None - if node_type == "array": - readings[key], child = read_array_v3(entry, context) - elif node_type == "group": - readings[key], child = read_group_v3(entry, context) - else: - problems.append( - ValidationProblem( - ("metadata", key, "node_type"), - "expected 'array' or 'group'", - "invalid_value", - ) - ) - continue + readings[key], child = _read_node_v3(entry, context) if child is not None: members[key] = child problems.extend(_prefix("metadata", _prefix(key, readings[key].problems))) return readings, members, tuple(problems) -def _node_type(document: object) -> object: - """The `node_type` a document says it is; None when it is not an object.""" - if not isinstance(document, Mapping): - return None - return cast("Mapping[object, object]", document).get("node_type") - - def _with_models( reading: ZarrV3GroupMetadataReading, members: GroupMembersV3 ) -> ZarrV3GroupMetadataReading: @@ -505,21 +602,21 @@ def _with_models( def _models( - readings: Mapping[str, ZarrV3ArrayMetadataReading | ZarrV3GroupMetadataReading], + readings: Mapping[str, ZarrV3NodeMetadataReading], members: Mapping[str, ArrayMembersV3 | GroupMembersV3], ) -> tuple[ - dict[str, ZarrV3ArrayMetadataReading | ZarrV3GroupMetadataReading], + dict[str, ZarrV3NodeMetadataReading], dict[str, ZarrV3ArrayMetadata | ZarrV3GroupMetadata], ]: """Each reading, holding its model when a model can be built of its document, and those models, by path.""" - held: dict[str, ZarrV3ArrayMetadataReading | ZarrV3GroupMetadataReading] = dict(readings) + held: dict[str, ZarrV3NodeMetadataReading] = dict(readings) models: dict[str, ZarrV3ArrayMetadata | ZarrV3GroupMetadata] = {} for path, child in members.items(): reading = readings[path] if isinstance(reading, ZarrV3ArrayMetadataReading): array = array_model(reading, cast("ArrayMembersV3", child)) held[path], models[path] = dataclasses.replace(reading, metadata=array), array - else: + elif isinstance(reading, ZarrV3GroupMetadataReading): group = _with_models(reading, cast("GroupMembersV3", child)) held[path] = group if group.metadata is not None: diff --git a/packages/zarr-metadata/tests/model/test_group.py b/packages/zarr-metadata/tests/model/test_group.py index 860c4e9bab..59c17c6d3d 100644 --- a/packages/zarr-metadata/tests/model/test_group.py +++ b/packages/zarr-metadata/tests/model/test_group.py @@ -27,12 +27,18 @@ ZarrV3GroupMetadata, ZarrV3GroupMetadataReading, ZarrV3GroupMetadataUpdate, + ZarrV3UnknownNodeReading, is_group_metadata_v3, + node_metadata_from_json_v3, + node_metadata_from_key_value_v3, parse_group_metadata_v3, read_group_metadata_v3, + read_node_metadata_v3, validate_group_metadata_v3, + validate_node_metadata_v3, ) from zarr_metadata.model._validation import ( + ZarrV3ArrayMetadataReading, is_group_metadata_v2, parse_group_metadata_v2, validate_group_metadata_v2, @@ -534,11 +540,11 @@ def test_error_a_group_document_that_is_not_an_object_reads_as_nothing() -> None {"a": False, "b": True}, [(*A, "codecs", 1)], ), - # One of no node type is not read at all. + # One of no node type is read as none, and nothing else of it is. ( _group(consolidated_metadata=_inline(a={"zarr_format": 3})), - [((*A, "node_type"), "invalid_value")], - {}, + [((*A, "node_type"), "missing_key")], + {"a": False}, [], ), ], @@ -559,6 +565,103 @@ def test_error_a_group_document_with_a_problem_reads_as_no_model( assert [loc for loc, field in reading.fields() if isinstance(field, Refused)] == refused +# --- read_node_metadata_v3 ------------------------------------------------- + + +@pytest.mark.parametrize( + ("document", "reading", "problems"), + [ + (_array(), ZarrV3ArrayMetadataReading, []), + (_group(attributes={"a": 1}), ZarrV3GroupMetadataReading, []), + (_array(fill_value=300), ZarrV3ArrayMetadataReading, [(("fill_value",), "invalid_value")]), + (_group(attributes=5), ZarrV3GroupMetadataReading, [(("attributes",), "invalid_type")]), + ], + ids=["array", "group", "array-with-a-problem", "group-with-a-problem"], +) +def test_a_node_is_read_as_the_node_its_node_type_says( + document: dict[str, object], + reading: type[ZarrV3ArrayMetadataReading | ZarrV3GroupMetadataReading], + problems: list[tuple[tuple[str | int, ...], str]], +) -> None: + """As that node's own read reads it, and its model only when it has no problem.""" + read = read_node_metadata_v3(document) + assert type(read) is reading + assert [(p.loc, p.kind) for p in read.problems] == problems + assert [(p.loc, p.kind) for p in validate_node_metadata_v3(document)] == problems + assert (read.metadata is not None) is (len(problems) == 0) + + +@pytest.mark.parametrize( + ("node_type", "kind"), + [("dataset", "invalid_value"), (5, "invalid_type"), (None, "invalid_type")], + ids=["another-kind", "a-number", "null"], +) +def test_error_a_node_type_the_spec_does_not_define(node_type: object, kind: str) -> None: + """Nothing else of the document is read, so nothing else is judged.""" + read = read_node_metadata_v3({**_array(), "node_type": node_type, "shape": "not a shape"}) + assert isinstance(read, ZarrV3UnknownNodeReading) + assert [(p.loc, p.kind) for p in read.problems] == [(("node_type",), kind)] + assert read.metadata is None + assert list(read.fields()) == [] + + +def test_error_a_document_without_a_node_type() -> None: + document = {key: value for key, value in _array().items() if key != "node_type"} + read = read_node_metadata_v3(document) + assert isinstance(read, ZarrV3UnknownNodeReading) + assert [(p.loc, p.kind) for p in read.problems] == [(("node_type",), "missing_key")] + + +def test_error_a_node_that_is_not_an_object() -> None: + read = read_node_metadata_v3([_array()]) + assert isinstance(read, ZarrV3UnknownNodeReading) + assert [(p.loc, p.kind) for p in read.problems] == [((), "invalid_type")] + + +@pytest.mark.parametrize( + "model", + [ + ZarrV3ArrayMetadata.create_default(shape=(4,)), + ZarrV3GroupMetadata.create_default(attributes={"a": 1}), + ], + ids=["array", "group"], +) +def test_a_node_is_built_as_the_model_its_node_type_says( + model: ZarrV3ArrayMetadata | ZarrV3GroupMetadata, +) -> None: + """From its JSON and from a store's bytes alike, as the model's own class builds it.""" + built = node_metadata_from_key_value_v3(model.to_key_value()) + assert type(built) is type(model) + assert built == model + assert node_metadata_from_json_v3(model.to_json()) == model + + +def test_error_a_node_built_from_a_store_without_its_document() -> None: + with pytest.raises(MetadataValidationError) as raised: + node_metadata_from_key_value_v3({}) + assert [(p.loc, p.kind) for p in raised.value.problems] == [(("zarr.json",), "missing_key")] + + +def test_error_a_node_built_from_bytes_that_are_not_json() -> None: + with pytest.raises(MetadataValidationError) as raised: + node_metadata_from_key_value_v3({"zarr.json": b"{"}) + assert [(p.loc, p.kind) for p in raised.value.problems] == [(("zarr.json",), "invalid_json")] + + +def test_error_a_node_built_from_a_document_of_no_node_type() -> None: + document = json.dumps({**_array(), "node_type": "dataset"}).encode() + with pytest.raises(MetadataValidationError) as raised: + node_metadata_from_key_value_v3({"zarr.json": document}) + assert [(p.loc, p.kind) for p in raised.value.problems] == [(("node_type",), "invalid_value")] + + +def test_error_a_node_built_from_a_document_with_a_problem() -> None: + """The problems of the node its `node_type` says it is.""" + with pytest.raises(MetadataValidationError) as raised: + node_metadata_from_json_v3(_array(fill_value=300)) + assert [(p.loc, p.kind) for p in raised.value.problems] == [(("fill_value",), "invalid_value")] + + # --- ZarrV2ConsolidatedMetadata -------------------------------------------- From 2f7672b9882e231c8c17a42fc0d8816e11737094 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Mon, 28 Sep 2026 23:54:53 +0200 Subject: [PATCH 06/94] feat(zarr-metadata)!: a bound is its member's type, and a problem carries what was found and what was expected Two borrowings from pydantic and zod. Bounds on the type. A number's type carries its bounds, in annotated-types' vocabulary as pydantic reads it: a gzip `level` is `Annotated[int, Interval(ge=0, le=9)]`. The checker holds a value to `Gt`, `Ge`, `Lt`, `Le` and `Interval`, one bound from each side, at any depth: one problem per value, whose message says what the type admits and whose `ctx` holds the bounds. A bound is held as the number it equals, so Python's `Annotated` cache, which takes `Ge(True)` for `Ge(1)`, cannot make a verdict or a message depend on import order. `Annotated` may also carry a note, a string or a `Doc`. Any other metadata is a `TypeError` when the type is compiled, so the checker and pydantic never read one type two ways. `typeddict_keys` keeps a member's metadata. The bounds that were rules move onto the types: gzip, zstd, blosc's `clevel` and `blocksize`, the regular and rectilinear grids, the shard's inner chunk shape, the eight integer fill values, byte values, and the numpy time types' `scale_factor` and ticks. `_InRange`, `byte_value_problems` and the numpy time rules are gone. Problems carry data. `ValidationProblem` gains `input`, the JSON the value handed in holds at `loc` (`UNSET` where nothing is, or where what is there is not JSON, so an error always pickles), and `ctx`, what was expected: a type's bounds by pydantic's names, or `expected` for a closed set. Neither takes part in equality or the repr, and `ctx` is read-only. Every reader fills `input` from the value its caller handed it, so a rule says only where a problem is. `UNSET` moves below the problem record. BREAKING CHANGE: `typeddict_keys(...).members` keeps `Annotated` metadata, and `check` and `Definition.check` hold values to the bounds a type carries. A definition's rules are asked only of a configuration within its bounds, as pydantic's after-validators are, so a value out of bounds hides a rule's problem until it is fixed: blosc's `typesize` behind a bad `clevel`, an `r16` fill value's count behind a bad byte. A definition that reuses a configuration TypedDict takes its bounds with it. A v2 `order` or `dimension_separator`, and a consolidated envelope's `must_understand`, are reported as values outside a closed set, `expected one of ["C", "F"], got "Q"`, and a `must_understand` that is not a boolean as `invalid_type`. A numpy time fill value out of range is reported as out of its integer range, without naming "NaT". Assisted-by: ClaudeCode:claude-opus-5-5 --- packages/zarr-metadata/changes/376.feature.md | 18 + packages/zarr-metadata/changes/376.removal.md | 9 + packages/zarr-metadata/pyproject.toml | 4 + .../zarr-metadata/src/zarr_metadata/_json.py | 177 ++++++++- .../zarr_metadata/{model => }/_sentinel.py | 4 +- .../src/zarr_metadata/_typed_json.py | 210 +++++++++-- .../src/zarr_metadata/model/__init__.py | 2 +- .../src/zarr_metadata/model/_array.py | 14 +- .../src/zarr_metadata/model/_group.py | 56 +-- .../src/zarr_metadata/model/_validation.py | 55 ++- .../src/zarr_metadata/typed_json.py | 21 +- .../src/zarr_metadata/v3/_common.py | 8 +- .../src/zarr_metadata/v3/_definition.py | 59 +-- .../src/zarr_metadata/v3/_pipeline.py | 6 +- .../v3/chunk_grid/rectilinear.py | 34 +- .../zarr_metadata/v3/chunk_grid/regular.py | 34 +- .../src/zarr_metadata/v3/codec/blosc.py | 26 +- .../src/zarr_metadata/v3/codec/gzip.py | 19 +- .../v3/codec/sharding_indexed.py | 17 +- .../src/zarr_metadata/v3/codec/zstd.py | 34 +- .../src/zarr_metadata/v3/data_type/_byte.py | 14 + .../zarr_metadata/v3/data_type/_integer.py | 49 --- .../zarr_metadata/v3/data_type/_numpy_time.py | 64 +--- .../src/zarr_metadata/v3/data_type/bytes.py | 7 +- .../src/zarr_metadata/v3/data_type/int16.py | 8 +- .../src/zarr_metadata/v3/data_type/int32.py | 8 +- .../src/zarr_metadata/v3/data_type/int64.py | 8 +- .../src/zarr_metadata/v3/data_type/int8.py | 8 +- .../v3/data_type/numpy_datetime64.py | 10 +- .../v3/data_type/numpy_timedelta64.py | 10 +- .../src/zarr_metadata/v3/data_type/raw.py | 7 +- .../src/zarr_metadata/v3/data_type/uint16.py | 8 +- .../src/zarr_metadata/v3/data_type/uint32.py | 8 +- .../src/zarr_metadata/v3/data_type/uint64.py | 8 +- .../src/zarr_metadata/v3/data_type/uint8.py | 8 +- .../src/zarr_metadata/v3/definition.py | 79 ++-- .../tests/model/test_extension_points.py | 11 +- .../zarr-metadata/tests/test_problem_data.py | 351 ++++++++++++++++++ .../zarr-metadata/tests/test_typed_json.py | 198 +++++++++- .../tests/v3/test_definitions.py | 29 +- .../tests/v3/test_every_definition.py | 106 ++++++ .../tests/v3/test_fill_values.py | 26 ++ packages/zarr-metadata/uv.lock | 6 +- 43 files changed, 1396 insertions(+), 442 deletions(-) create mode 100644 packages/zarr-metadata/changes/376.feature.md create mode 100644 packages/zarr-metadata/changes/376.removal.md rename packages/zarr-metadata/src/zarr_metadata/{model => }/_sentinel.py (90%) create mode 100644 packages/zarr-metadata/src/zarr_metadata/v3/data_type/_byte.py delete mode 100644 packages/zarr-metadata/src/zarr_metadata/v3/data_type/_integer.py create mode 100644 packages/zarr-metadata/tests/test_problem_data.py diff --git a/packages/zarr-metadata/changes/376.feature.md b/packages/zarr-metadata/changes/376.feature.md new file mode 100644 index 0000000000..ed04787613 --- /dev/null +++ b/packages/zarr-metadata/changes/376.feature.md @@ -0,0 +1,18 @@ +A number's type carries its bounds, as pydantic reads them: +annotated-types' `Gt`, `Ge`, `Lt`, `Le` and `Interval`, one bound from +each side, at any depth, so a gzip `level` is +`Annotated[int, Interval(ge=0, le=9)]`. `check`, a definition's `judge` +and every reader hold a value to them: one problem, `invalid_value`, +saying what the type admits. `Annotated` may also carry a note, a string +or a `Doc`; any other metadata is a `TypeError` when the type is +compiled, so the checker and pydantic never read one type two ways. The +package's own bounds are declared so -- the gzip and zstd levels, +blosc's `clevel` and `blocksize`, the regular and rectilinear grids, the +shard's inner chunk shape, the integer fill values, byte values, and the +numpy time types' `scale_factor` and ticks -- where they were rules. A +`ValidationProblem` carries its message's data, as pydantic's errors and +zod's issues do: `input`, the JSON the value handed in holds at its +`loc`, `UNSET` where nothing is, and `ctx`, what was expected -- the +bounds by pydantic's names, `{"ge": 0, "le": 9}`, or the values of a +closed set, `{"expected": ["array", "group"]}`. Every reader fills +`input`, so a definition's rules say only where a problem is. diff --git a/packages/zarr-metadata/changes/376.removal.md b/packages/zarr-metadata/changes/376.removal.md new file mode 100644 index 0000000000..699fe0718a --- /dev/null +++ b/packages/zarr-metadata/changes/376.removal.md @@ -0,0 +1,9 @@ +`typeddict_keys(...).members` keeps the `Annotated` metadata of a key's +type, which carries its bounds. A definition's rules are asked only of a +configuration within its bounds, as pydantic's after-validators are, so +a value out of bounds is the one problem reported until it is fixed. A +definition that reuses one of the package's configuration TypedDicts +takes its bounds with it. A v2 `order` or `dimension_separator`, and +the `must_understand` of a consolidated envelope, are reported as values +outside a closed set -- `expected one of ["C", "F"], got "Q"` -- and a +`must_understand` that is not a boolean as `invalid_type`. diff --git a/packages/zarr-metadata/pyproject.toml b/packages/zarr-metadata/pyproject.toml index 31d548bf2a..8d37031881 100644 --- a/packages/zarr-metadata/pyproject.toml +++ b/packages/zarr-metadata/pyproject.toml @@ -32,6 +32,10 @@ classifiers = [ ] keywords = ["zarr"] dependencies = [ + # The constraints a TypedDict's members carry -- `Ge`, `Interval`, + # `MinLen` -- which the checker enforces, in the vocabulary pydantic + # and Hypothesis read as well. + "annotated-types>=0.6", # >=4.16: first release where `Sentinel` pickles by reference # (`__reduce__` returns the sentinel's name), so `UNSET` — and any model # holding it — can cross process boundaries and be deep-copied with its diff --git a/packages/zarr-metadata/src/zarr_metadata/_json.py b/packages/zarr-metadata/src/zarr_metadata/_json.py index 82b20d4f9f..99c66e5b5c 100644 --- a/packages/zarr-metadata/src/zarr_metadata/_json.py +++ b/packages/zarr-metadata/src/zarr_metadata/_json.py @@ -9,13 +9,16 @@ from __future__ import annotations +import dataclasses import json import math -from collections.abc import Mapping, Sequence +from collections.abc import Callable, Mapping, Sequence from dataclasses import dataclass -from typing import Literal, TypeGuard, cast, get_args +from types import MappingProxyType +from typing import Final, Literal, TypeGuard, cast, get_args from zarr_metadata._common import JSONValue +from zarr_metadata._sentinel import UNSET ProblemKind = Literal["missing_key", "invalid_type", "invalid_value", "invalid_json", "unknown_key"] """Machine-readable classification of a `ValidationProblem`. @@ -36,20 +39,57 @@ """ +_NO_CTX: Final[Mapping[str, JSONValue]] = MappingProxyType({}) + + +def _no_ctx() -> Mapping[str, JSONValue]: + """What a problem that says nothing more than its type was expected holds as its `ctx`: nothing.""" + return _NO_CTX + + @dataclass(frozen=True, slots=True) class ValidationProblem: - """A single problem found in a value: where it is, what is wrong, and what kind of wrong. + """A single problem found in a value: where it is, what is wrong, what kind of wrong, and the data the message is made of. `loc` is the path from the root of what was judged to the offending value, e.g. `("codecs", 0, "name")` in a document, and an empty `loc` refers to that root. `kind` classifies the failure mode for programmatic dispatch; `message` is the human-readable description. + + `input` and `ctx` are what the message says, as data, as pydantic's + `ErrorDetails` and zod's issues carry theirs. `input` is the JSON found + at `loc` -- `12`, for a gzip `level` of 12 -- and `UNSET` where nothing + is there, as zod has it for a key that is missing (pydantic gives the + object missing it), or where what is there is not JSON, which the + message shows. It is the object the caller handed in, as pydantic's + is, not a copy: a caller that changes its document afterwards changes + what `input` shows. `ctx` is what was expected, where that is more + than a type: + + - `gt`, `ge`, `lt` and `le`: the bounds the value's type carries, as + pydantic names them -- `{"ge": 0, "le": 9}` for a gzip `level`, + whose type is `Annotated[int, Interval(ge=0, le=9)]` -- or a rule + says. + - `expected`: the values of a closed set, as zod's `values` holds + them, in the order the message lists them -- a `Literal`'s, + `node_type`'s, `zarr_format`'s. + + Neither takes part in equality or the repr: a problem is the same + problem when it is found at the same place and says the same thing. + Every function that returns or raises problems fills `input` from the + value its caller handed it, so a rule says only where a problem is. """ loc: tuple[str | int, ...] message: str kind: ProblemKind + input: JSONValue | UNSET = dataclasses.field( + default=UNSET, kw_only=True, compare=False, repr=False + ) + ctx: Mapping[str, JSONValue] = dataclasses.field( + default_factory=_no_ctx, kw_only=True, compare=False, repr=False + ) def __post_init__(self) -> None: # The runtime half of the annotations: a rule written without a type @@ -70,11 +110,45 @@ def __post_init__(self) -> None: if not isinstance(kind, str) or kind not in get_args(ProblemKind): msg = f"a ValidationProblem's kind is one of {get_args(ProblemKind)!r}, got {kind!r}" raise TypeError(msg) + ctx = cast("object", self.ctx) + if ctx is _NO_CTX: + return + if ( + not isinstance(ctx, Mapping) + or not all(isinstance(key, str) for key in cast("Mapping[object, object]", ctx)) + or not is_json(dict(cast("Mapping[str, object]", ctx))) + ): + msg = f"a ValidationProblem's ctx is an object of JSON values, got {ctx!r}" + raise TypeError(msg) + # Held as a view of a copy of its own, arrays as tuples, so a raised + # error, a finished report, cannot be edited through it. + held = cast( + "dict[str, JSONValue]", arrays_to_tuples(dict(cast("Mapping[str, object]", ctx))) + ) + object.__setattr__(self, "ctx", MappingProxyType(held)) def __str__(self) -> str: location = ".".join(str(part) for part in self.loc) if self.loc else "" return f"{location}: {self.message}" + def __reduce__( + self, + ) -> tuple[Callable[..., ValidationProblem], tuple[object, ...]]: + # Pickled and copied as its constructor called again: the view + # `ctx` is held as does not pickle, and the dict it views does. + return (_problem, (self.loc, self.message, self.kind, self.input, dict(self.ctx))) + + +def _problem( + loc: tuple[str | int, ...], + message: str, + kind: ProblemKind, + found: JSONValue | UNSET, + ctx: Mapping[str, JSONValue], +) -> ValidationProblem: + """A problem built again from what `ValidationProblem.__reduce__` gives.""" + return ValidationProblem(loc, message, kind, input=found, ctx=ctx) + class MetadataValidationError(ValueError): """Raised when a value fails validation, by the entry points that raise rather than report. @@ -115,12 +189,76 @@ def prefixed( loc_head: str | int, problems: Sequence[ValidationProblem] ) -> tuple[ValidationProblem, ...]: """Prepend `loc_head` to the `loc` of every problem (for nested validators).""" - return tuple(ValidationProblem((loc_head, *p.loc), p.message, p.kind) for p in problems) + return tuple(dataclasses.replace(p, loc=(loc_head, *p.loc)) for p in problems) + + +def value_at(value: object, loc: tuple[str | int, ...]) -> object: + """What `value` holds at `loc`, each key naming a member of an object and each index an element of an array; `UNSET` where it holds nothing.""" + for part in loc: + if isinstance(part, str) and isinstance(value, Mapping): + members = cast("Mapping[object, object]", value) + if part not in members: + return UNSET + value = members[part] + elif ( + isinstance(part, int) + and isinstance(value, Sequence) + and not isinstance(value, (str, bytes, bytearray)) + and 0 <= part < len(cast("Sequence[object]", value)) + ): + value = cast("Sequence[object]", value)[part] + else: + return UNSET + return value + + +def with_input( + problems: Sequence[ValidationProblem], value: object, loc: tuple[str | int, ...] = () +) -> tuple[ValidationProblem, ...]: + """`problems`, found in `value`, which sits at `loc`: each given, as its `input`, the JSON `value` holds at its own `loc`. + + What every function that judges a value does with what it found, so a + rule says only where a problem is, and the problem carries what is + there. The function a caller called is the last to do it, so a + problem's `input` is what the value the caller handed in holds, not a + copy some reader inside it made. A problem whose `loc` names nothing + in `value` -- a key that is missing -- or names what is not JSON + keeps what it holds, `UNSET` unless a reader inside found JSON there: + no problem holds what might not pickle, or copy. + """ + if len(problems) == 0: + return () + filled: list[ValidationProblem] = [] + for found in problems: + if found.loc[: len(loc)] == loc: + there = value_at(value, found.loc[len(loc) :]) + if there is not found.input and _holdable(there): + found = dataclasses.replace(found, input=cast("JSONValue", there)) + filled.append(found) + return tuple(filled) + + +def _holdable(value: object) -> bool: + """Whether a problem can hold `value` as its input: JSON, and shallow enough to walk. + + What a validator did not walk -- a member it only reports -- may be + deeper than the interpreter walks; such a value would not pickle + either, and is held as nothing. + """ + try: + return is_canonical_json(value, finite=False) + except RecursionError: + return False + + +def not_an_object(value: object) -> tuple[ValidationProblem, ...]: + """What is wrong with a document that is not an object: the one problem, at its root.""" + return with_input((ValidationProblem((), "expected an object", "invalid_type"),), value) def validate_json(value: object) -> tuple[ValidationProblem, ...]: """Return every reason `value` is not JSON, each where it sits: a float that is not finite, a key that is not a string, a value of no JSON type.""" - return refine_json(value)[1] + return with_input(refine_json(value)[1], value) def refine_json( @@ -206,10 +344,30 @@ def shown(value: object) -> str: def choices(allowed: Sequence[object]) -> str: """A closed set of values as a message names it: `"C"` alone, or `one of ["C", "F"]`.""" - values = sorted(dict.fromkeys(shown(value) for value in allowed)) + values = [shown(value) for value in listed(allowed)] return values[0] if len(values) == 1 else f"one of [{', '.join(values)}]" +def listed(allowed: Sequence[object]) -> tuple[object, ...]: + """A closed set of values, each once, in the order a message lists them: by the JSON each is written as.""" + written = {shown(value): value for value in allowed} + return tuple(written[json_text] for json_text in sorted(written)) + + +def outside_of( + loc: tuple[str | int, ...], value: object, allowed: Sequence[object] +) -> ValidationProblem: + """The problem with `value`, found at `loc`, outside the closed set `allowed`. + + Its message lists the set, its kind says whether the value is of the + wrong JSON type or of the right one with the wrong value, and its + `ctx` holds the set, as `expected`. + """ + message = f"expected {choices(allowed)}, got {shown(value)}" + expected = cast("tuple[JSONValue, ...]", listed(allowed)) + return ValidationProblem(loc, message, refused_kind(value, allowed), ctx={"expected": expected}) + + def json_type(value: object) -> str: """The JSON type of `value`, as a message names it: "a string", "a number", "null".""" if value is None: @@ -267,7 +425,7 @@ def parse_json(value: object) -> JSONValue: """Return a canonical `JSONValue`, or raise `MetadataValidationError`.""" refined, problems = refine_json(value) if len(problems) != 0: - raise MetadataValidationError(problems) + raise MetadataValidationError(with_input(problems, value)) return refined @@ -324,6 +482,9 @@ def arrays_to_tuples(obj: object) -> object: "is_canonical_json", "is_json", "json_type", + "listed", + "not_an_object", + "outside_of", "parse_json", "prefixed", "refine_json", @@ -331,4 +492,6 @@ def arrays_to_tuples(obj: object) -> object: "refused_kind", "shown", "validate_json", + "value_at", + "with_input", ] diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_sentinel.py b/packages/zarr-metadata/src/zarr_metadata/_sentinel.py similarity index 90% rename from packages/zarr-metadata/src/zarr_metadata/model/_sentinel.py rename to packages/zarr-metadata/src/zarr_metadata/_sentinel.py index ad71e216fa..7b159688b2 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_sentinel.py +++ b/packages/zarr-metadata/src/zarr_metadata/_sentinel.py @@ -4,7 +4,9 @@ JSON `null` in the document (a v2 `compressor`/`filters` value, an unnamed dimension inside `dimension_names`), and `UNSET` always means the document key is absent. The two are never interchangeable, so a model value can never -leak into a document as a spelling the writer did not intend. +leak into a document as a spelling the writer did not intend. A problem's +`input` keeps the same invariant: it is `UNSET` where nothing was found at +the problem's `loc`, and `None` where a `null` was. Check with identity: `if model.dimension_names is UNSET: ...`. diff --git a/packages/zarr-metadata/src/zarr_metadata/_typed_json.py b/packages/zarr-metadata/src/zarr_metadata/_typed_json.py index c4bf7c6fa0..a5c38f2476 100644 --- a/packages/zarr-metadata/src/zarr_metadata/_typed_json.py +++ b/packages/zarr-metadata/src/zarr_metadata/_typed_json.py @@ -8,7 +8,10 @@ `tuple[T1, T2]`, a union of those, an object described by a TypedDict, `Mapping[str, V]`, a `NewType` as the type it names, and a type alias as the type it stands for -- which is what keeps it small. An annotation -outside these implies no parser. +outside these implies no parser. A number's type may carry bounds, as +annotated-types spells them and pydantic reads them -- +`Annotated[int, Interval(ge=0, le=9)]` -- and a value out of them is a +problem; `constraints_of` says which the checker reads. A TypedDict is read as the typing spec defines it, which is not always what its runtime attributes say: `typeddict_keys` reads which keys it @@ -30,10 +33,12 @@ from __future__ import annotations import functools +import math +import operator import sys import types import typing -from collections.abc import Callable, Mapping, Sequence +from collections.abc import Callable, Iterator, Mapping, Sequence from collections.abc import Set as AbstractSet from dataclasses import dataclass from typing import ( @@ -53,6 +58,7 @@ get_type_hints, ) +import annotated_types import typing_extensions from typing_extensions import NoExtraItems, TypeIs, is_typeddict @@ -61,9 +67,10 @@ ValidationProblem, choices, is_json, + outside_of, refine_json, - refused_kind, shown, + with_input, ) if TYPE_CHECKING: @@ -138,7 +145,9 @@ def strip_annotation(annotation: object) -> tuple[object, tuple[object, ...]]: origin = get_origin(annotation) if origin is Annotated: inner, *extras = get_args(annotation) - metadata.extend(extras) + # An inner layer's metadata first, as `Annotated` flattens + # nested layers. + metadata[:0] = extras annotation = inner elif origin in _QUALIFIERS: (annotation,) = get_args(annotation) @@ -146,6 +155,14 @@ def strip_annotation(annotation: object) -> tuple[object, tuple[object, ...]]: return annotation, tuple(metadata) +def unqualified(annotation: object) -> object: + """A TypedDict key's annotation with its qualifiers peeled and its `Annotated` metadata kept, which says more of the value's type.""" + inner, metadata = strip_annotation(annotation) + if len(metadata) == 0: + return inner + return Annotated[(inner, *metadata)] + + Qualifier: TypeAlias = Literal["Required", "NotRequired", "ReadOnly"] """A qualifier on a TypedDict key, by name: `typing` and `typing_extensions` may each spell one.""" @@ -209,10 +226,11 @@ class TypedDictKeys: """What a TypedDict says of an object's keys, read as the typing spec defines it. `members` holds every key it declares, its bases' included, with the - type of the key's value -- qualifiers and `Annotated` metadata peeled - -- and whether the key is required. `extra_items` is what any other - key may hold: `Never` when the TypedDict is closed, `object` when it - is open, and the `extra_items` type otherwise. `declared` is whether + type of the key's value -- its qualifiers peeled, and any `Annotated` + metadata kept, since a constraint there is part of the type -- and + whether the key is required. `extra_items` is what any other key may + hold: `Never` when the TypedDict is closed, `object` when it is open, + and the `extra_items` type otherwise. `declared` is whether that was said, by the TypedDict or a base, rather than defaulted: a TypedDict that says nothing is open. """ @@ -229,12 +247,13 @@ def required(self) -> frozenset[str]: @property def closed(self) -> bool: """Whether a key it does not declare is not a key of the type: `closed=True`, or `extra_items=Never`.""" - return self.extra_items is Never or self.extra_items is NoReturn + extra_items = strip_annotation(self.extra_items)[0] + return extra_items is Never or extra_items is NoReturn @property def open(self) -> bool: """Whether a key it does not declare may hold anything: the default, or `closed=False`.""" - return self.extra_items is object + return strip_annotation(self.extra_items)[0] is object @functools.cache @@ -276,7 +295,7 @@ def typeddict_keys(typeddict: type) -> TypedDictKeys: required = False else: required = key in required_at_runtime - members[key] = (strip_annotation(hint)[0], required) + members[key] = (unqualified(hint), required) extra_items, declared = _openness(typeddict) return TypedDictKeys(types.MappingProxyType(members), extra_items, declared) @@ -339,7 +358,7 @@ def _openness(typeddict: type) -> tuple[object, bool]: if closed is not None: return (Never if closed else object), True inherited = [_openness(base) for base in _typeddict_bases(typeddict)] - restricted = {extra for extra, _ in inherited if extra is not object} + restricted = {extra for extra, _ in inherited if strip_annotation(extra)[0] is not object} if len(restricted) > 1: msg = ( f"{typeddict.__name__}: its bases disagree on what a key they do not declare may " @@ -352,7 +371,7 @@ def _openness(typeddict: type) -> tuple[object, bool]: def _extra_items_type(typeddict: type, extra_items: object) -> object: - """The `extra_items` type, evaluated where `typeddict` was defined, `ReadOnly` peeled.""" + """The `extra_items` type, evaluated where `typeddict` was defined, `ReadOnly` peeled and any `Annotated` metadata kept.""" if isinstance(extra_items, (str, ForwardRef)): holder = type( "_ExtraItems", @@ -364,7 +383,7 @@ def _extra_items_type(typeddict: type, extra_items: object) -> object: except NameError as error: msg = f"{typeddict.__name__}: {error}; its extra_items must resolve in its module" raise TypeError(msg) from error - return strip_annotation(extra_items)[0] + return unqualified(extra_items) def _typeddict_bases(typeddict: type) -> tuple[type, ...]: @@ -547,8 +566,7 @@ def one_of(allowed: tuple[object, ...]) -> Parser: def parse(value: object, loc: Loc) -> Parsed: if not any(value == entry and type(value) is type(entry) for entry in allowed): - message = f"expected {choices(allowed)}, got {shown(value)}" - return value, problem(loc, message, refused_kind(value, allowed)) + return value, (outside_of(loc, value, allowed),) return value, () return parse @@ -647,9 +665,7 @@ def _by_tag(branches: Sequence[Branch], tag: Tag, value: Mapping[str, object], l said = value[key] index = picks.get((type(said), said)) if _hashable(said) else None if index is None: - allowed = tuple(entry for _, entry in picks) - message = f"expected {choices(allowed)}, got {shown(said)}" - return value, problem((*loc, key), message, refused_kind(said, allowed)) + return value, (outside_of((*loc, key), said, tuple(entry for _, entry in picks)),) return branches[index][1](value, loc) @@ -721,6 +737,137 @@ def parse(candidate: object, loc: Loc) -> Parsed: return parse +# --- constraints --------------------------------------------------------- + +Constraints: TypeAlias = Mapping[str, int | float] +"""The bounds a type carries, by the names pydantic gives them: `{"ge": 0, "le": 9}`.""" + +_BOUNDS: Final[tuple[tuple[type, str], ...]] = ( + (annotated_types.Gt, "gt"), + (annotated_types.Ge, "ge"), + (annotated_types.Lt, "lt"), + (annotated_types.Le, "le"), +) +"""The annotated-types constraints the checker reads, each with its name.""" + +_FROM_BELOW: Final = frozenset({"gt", "ge"}) +"""The bounds a value must be above.""" + +_FROM_ABOVE: Final = frozenset({"lt", "le"}) +"""The bounds a value must be below.""" + +_EXCLUSIVE: Final = frozenset({"gt", "lt"}) +"""The bounds a value may not equal.""" + +_HOLDS: Final[Mapping[str, Callable[[float, float], bool]]] = { + "gt": operator.gt, + "ge": operator.ge, + "lt": operator.lt, + "le": operator.le, +} +"""Whether a number keeps within a bound of each name.""" + + +def constraints_of(metadata: Sequence[object]) -> dict[str, int | float]: + """The bounds among an `Annotated` type's metadata, by name: at most one from below, and one from above. + + annotated-types' `Gt`, `Ge`, `Lt` and `Le`, and an `Interval`, which + unpacks to them, as pydantic reads them. A note -- a string, a `Doc` + -- says nothing of what a value is, and is passed over. Anything else + is a `TypeError`: a constraint the checker does not read -- a + `MinLen`, a `Predicate`, pydantic's `Field` -- since a type that says + what its values are, and a checker that does not hold them to it, + would disagree; a second bound from one side, which pydantic reads as + the last one said; and a bound that is not a finite number. A bound + is held as the number it equals, an integer when it is one: + `Ge(True)`, `Ge(1.0)` and `Ge(1)` are one bound, as Python's + `Annotated` cache, which takes equal metadata for the same, may hand + back any of them for another. + """ + said: dict[str, int | float] = {} + for item in _unpacked(metadata): + if isinstance(item, (str, typing_extensions.Doc)): + continue + name = next((name for kind, name in _BOUNDS if isinstance(item, kind)), None) + if name is None: + msg = f"{item!r} is not a constraint the checker reads, which are Gt, Ge, Lt and Le" + raise TypeError(msg) + side = _FROM_BELOW if name in _FROM_BELOW else _FROM_ABOVE + if not side.isdisjoint(said): + where = "below" if side is _FROM_BELOW else "above" + msg = f"{item!r} is a second bound from {where}; a type takes one from each side" + raise TypeError(msg) + said[name] = _bound(item, name) + return said + + +def _unpacked(metadata: Sequence[object]) -> Iterator[object]: + """`metadata`, each group of constraints in it -- an `Interval` -- unpacked.""" + for item in metadata: + if isinstance(item, annotated_types.GroupedMetadata): + yield from _unpacked(tuple(cast("Sequence[object]", item))) + else: + yield item + + +def _bound(item: object, name: str) -> int | float: + """The number `item`, a bound of `name`, holds a value to, an integer when it is one; `TypeError` for what is not a finite number.""" + bound = cast("object", getattr(item, name)) + if isinstance(bound, int): + return int(bound) # a `bool` as the integer it equals, as the `Annotated` cache has it + if isinstance(bound, float) and math.isfinite(bound): + return int(bound) if bound.is_integer() else bound + msg = f"{item!r}: a bound is a finite number" + raise TypeError(msg) + + +def _constrainable(inner: object, said: Constraints) -> None: + """Refuse bounds on what is not a number: a string, an array, a value of any type.""" + if len(said) != 0 and shape_of(inner) not in ("int", "number"): + msg = f"{', '.join(said)}: a bound is on a number, and {describe(inner)} is not one" + raise TypeError(msg) + + +def constrained(parse: Parser, description: str, said: Constraints) -> Parser: + """What `parse` reads, held to the bounds its type carries: one problem, `invalid_value`, when it breaks any. + + The message says what the type admits -- "expected an integer in [0, + 9], got 12" -- and the problem's `ctx` holds the bounds. Only a + value its type reads is held to them, as rules are asked only of a + value of their type. + """ + expected = f"expected {description} {_admitted(said)}" + ctx: dict[str, JSONValue] = dict(said) + holds = tuple((_HOLDS[name], bound) for name, bound in said.items()) + + def parse_constrained(value: object, loc: Loc) -> Parsed: + typed, found = parse(value, loc) + if len(found) != 0: + return typed, found + number = cast("float", typed) + for keeps, bound in holds: + if not keeps(number, bound): + message = f"{expected}, got {shown(value)}" + return typed, (ValidationProblem(loc, message, "invalid_value", ctx=ctx),) + return typed, () + + return parse_constrained + + +def _admitted(said: Constraints) -> str: + """What the bounds admit, as a message says it after the type: "in [0, 9]", "in (0, 1)", ">= 1".""" + low = next(((name, bound) for name, bound in said.items() if name in _FROM_BELOW), None) + high = next(((name, bound) for name, bound in said.items() if name in _FROM_ABOVE), None) + if low is not None and high is not None: + opening = "(" if low[0] in _EXCLUSIVE else "[" + closing = ")" if high[0] in _EXCLUSIVE else "]" + return f"in {opening}{shown(low[1])}, {shown(high[1])}{closing}" + if low is not None: + return f"{'>' if low[0] in _EXCLUSIVE else '>='} {shown(low[1])}" + name, bound = cast("tuple[str, int | float]", high) + return f"{'<' if name in _EXCLUSIVE else '<='} {shown(bound)}" + + # --- the compiler -------------------------------------------------------- _Building: TypeAlias = dict[object, Parser] @@ -827,6 +974,9 @@ def compile_() -> Parser | None: if keys.closed: return object_of(members, None) if keys.open: + # Any JSON value, which a note does not change and a bound + # cannot: vetted all the same. + _constrainable(object, constraints_of(strip_annotation(keys.extra_items)[1])) return object_of(members, _JSON) extra = _compile(keys.extra_items, leaf, building) return None if extra is None else object_of(members, extra) @@ -854,7 +1004,19 @@ def _literal(inner: object) -> Parser | None: def _compile(annotation: object, leaf: Leaf, building: _Building) -> Parser | None: - inner = strip_annotation(annotation)[0] + inner, metadata = strip_annotation(annotation) + parse = _compile_type(inner, leaf, building) + if parse is None or len(metadata) == 0: + return parse + said = constraints_of(metadata) + if len(said) == 0: + return parse + _constrainable(inner, said) + return constrained(parse, describe(inner), said) + + +def _compile_type(inner: object, leaf: Leaf, building: _Building) -> Parser | None: + """The parser of a type, `Annotated` peeled from it.""" found = leaf(inner) if found is not None: return found @@ -967,14 +1129,15 @@ def check( raise TypeError(msg) refined, problems = refine_json(value, loc) if len(problems) != 0: - return None, problems + return None, with_input(problems, value, loc) typed, found = _checker(shape)(refined, loc) readable = all(problem.kind == "unknown_key" for problem in found) - return (cast("T", typed) if readable else None), found + return (cast("T", typed) if readable else None), with_input(found, value, loc) __all__ = [ "Branch", + "Constraints", "Leaf", "Loc", "Parsed", @@ -985,6 +1148,8 @@ def check( "alias_value", "any_of", "check", + "constrained", + "constraints_of", "describe", "fixed_tuple", "has_shape", @@ -1003,5 +1168,6 @@ def check( "shape_of", "strip_annotation", "typeddict_keys", + "unqualified", "unread_in", ] diff --git a/packages/zarr-metadata/src/zarr_metadata/model/__init__.py b/packages/zarr-metadata/src/zarr_metadata/model/__init__.py index 0d9195ff9c..c5f45073fc 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/__init__.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/__init__.py @@ -27,6 +27,7 @@ parse_json, validate_json, ) +from zarr_metadata._sentinel import UNSET from zarr_metadata.model._array import ( ZarrV2ArrayMetadata, ZarrV2ArrayMetadataPartial, @@ -54,7 +55,6 @@ validate_group_metadata_v3, validate_node_metadata_v3, ) -from zarr_metadata.model._sentinel import UNSET from zarr_metadata.model._validation import ( ARRAY_METADATA_OPTIONAL_KEYS_V3, ARRAY_METADATA_REQUIRED_KEYS_V2, diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_array.py b/packages/zarr-metadata/src/zarr_metadata/model/_array.py index f8c28b3c36..13d31ded89 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_array.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/_array.py @@ -14,8 +14,9 @@ MetadataValidationError, ValidationProblem, copied, + with_input, ) -from zarr_metadata.model._sentinel import UNSET +from zarr_metadata._sentinel import UNSET from zarr_metadata.model._validation import ( ARRAY_METADATA_STANDARD_KEYS_V3, NO_SCOPE, @@ -528,15 +529,10 @@ def from_key_value(cls, mapping: Mapping[StoreKey, bytes]) -> ZarrV2ArrayMetadat return cls.from_json(zarray_raw) zarray = cast("Mapping[str, object]", zarray_raw) if "attributes" in zarray: - raise MetadataValidationError( - [ - ValidationProblem( - ("attributes",), - "unexpected document member", - "invalid_value", - ) - ] + refused = ValidationProblem( + ("attributes",), "unexpected document member", "invalid_value" ) + raise MetadataValidationError(with_input((refused,), zarray)) if ZARR_V2_ATTRIBUTES_STORE_KEY in mapping: zattrs = load_store_json(mapping, ZARR_V2_ATTRIBUTES_STORE_KEY) return cls.from_json({**zarray, "attributes": zattrs}) diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_group.py b/packages/zarr-metadata/src/zarr_metadata/model/_group.py index 446d8ec900..e1b3542f1a 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_group.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/_group.py @@ -14,15 +14,16 @@ MetadataValidationError, ValidationProblem, arrays_to_tuples, - choices, copied, is_canonical_json, + not_an_object, + outside_of, refine_json, refine_user_data, - refused_kind, - shown, + with_input, ) from zarr_metadata._json import prefixed as _prefix +from zarr_metadata._sentinel import UNSET from zarr_metadata.model._array import ( ZarrV3ArrayMetadata, array_json, @@ -31,7 +32,6 @@ must_understand_subset, read_array_metadata_v3, ) -from zarr_metadata.model._sentinel import UNSET from zarr_metadata.model._validation import ( GROUP_METADATA_REQUIRED_KEYS_V3, GROUP_METADATA_STANDARD_KEYS_V3, @@ -457,15 +457,14 @@ def _read_node_v3( def _node_type(value: object) -> tuple[str | None, tuple[ValidationProblem, ...]]: """The node type `value` says it is, one of `_NODE_TYPES`; None, with the problem, when it says none of them, or is not an object.""" if not isinstance(value, Mapping): - return None, (ValidationProblem((), "expected an object", "invalid_type"),) + return None, not_an_object(value) document = cast("Mapping[object, object]", value) if "node_type" not in document: return None, (ValidationProblem(("node_type",), "missing required key", "missing_key"),) node_type = document["node_type"] if isinstance(node_type, str) and node_type in _NODE_TYPES: return node_type, () - message = f"expected {choices(_NODE_TYPES)}, got {shown(node_type)}" - return None, (ValidationProblem(("node_type",), message, refused_kind(node_type, _NODE_TYPES)),) + return None, with_input((outside_of(("node_type",), node_type, _NODE_TYPES),), document) @dataclass(frozen=True, slots=True) @@ -507,8 +506,7 @@ def read_group_v3( as it is, as `read_array_v3` takes one. """ if not isinstance(value, Mapping): - problems = (ValidationProblem((), "expected an object", "invalid_type"),) - return ZarrV3GroupMetadataReading(problems=problems), None + return ZarrV3GroupMetadataReading(problems=not_an_object(value)), None doc = cast("Mapping[object, object]", value) found: list[ValidationProblem] = list(missing_keys(GROUP_METADATA_REQUIRED_KEYS_V3, doc)) extra_fields, others = other_members( @@ -532,7 +530,7 @@ def read_group_v3( if raw is not None: consolidated, held, inside = _read_consolidated_v3(raw, context) found.extend(_prefix(ZARR_V3_CONSOLIDATED_METADATA_KEY, inside)) - reading = ZarrV3GroupMetadataReading(consolidated, tuple(found)) + reading = ZarrV3GroupMetadataReading(consolidated, with_input(found, doc)) return reading, GroupMembersV3(attributes, extra_fields, held) @@ -559,8 +557,7 @@ def _read_consolidated_v3( ] problems.extend(unexpected_keys(frozenset(_CONSOLIDATED_MEMBERS), env)) problems.extend(check_literal(env, "kind", "inline")) - if "must_understand" in env and env["must_understand"] is not False: - problems.append(ValidationProblem(("must_understand",), "expected False", "invalid_value")) + problems.extend(check_literal(env, "must_understand", False)) readings: dict[str, ZarrV3NodeMetadataReading] = {} members: dict[str, ArrayMembersV3 | GroupMembersV3] = {} entries = env.get("metadata") @@ -654,11 +651,10 @@ def parse_group_metadata_v3( value: object, *, context: Context = CORE_AND_EXTENSIONS ) -> ZarrV3GroupMetadataJSON: """Return `value` narrowed to `ZarrV3GroupMetadataJSON`, or raise `MetadataValidationError`.""" - normalized = arrays_to_tuples(value) - problems = validate_group_metadata_v3(normalized, context=context) + problems = validate_group_metadata_v3(value, context=context) if len(problems) != 0: raise MetadataValidationError(problems) - return cast("ZarrV3GroupMetadataJSON", normalized) + return cast("ZarrV3GroupMetadataJSON", arrays_to_tuples(value)) class ZarrV2GroupMetadataPartial(TypedDict, total=False): @@ -753,15 +749,10 @@ def from_key_value(cls, mapping: Mapping[StoreKey, bytes]) -> ZarrV2GroupMetadat return cls.from_json(zgroup_raw) zgroup = cast("Mapping[str, object]", zgroup_raw) if "attributes" in zgroup: - raise MetadataValidationError( - [ - ValidationProblem( - ("attributes",), - "unexpected document member", - "invalid_value", - ) - ] + refused = ValidationProblem( + ("attributes",), "unexpected document member", "invalid_value" ) + raise MetadataValidationError(with_input((refused,), zgroup)) if ZARR_V2_ATTRIBUTES_STORE_KEY in mapping: zattrs = load_store_json(mapping, ZARR_V2_ATTRIBUTES_STORE_KEY) return cls.from_json({**zgroup, "attributes": zattrs}) @@ -825,9 +816,7 @@ def from_json(cls, data: object) -> ZarrV2ConsolidatedMetadata: """ normalized = arrays_to_tuples(data) if not isinstance(normalized, Mapping): - raise MetadataValidationError( - [ValidationProblem((), "expected an object", "invalid_type")] - ) + raise MetadataValidationError(not_an_object(data)) doc = cast("Mapping[object, object]", normalized) problems: list[ValidationProblem] = [ ValidationProblem((key,), "missing required key", "missing_key") @@ -841,18 +830,7 @@ def from_json(cls, data: object) -> ZarrV2ConsolidatedMetadata: for key in doc if key not in {"zarr_consolidated_format", "metadata"} ) - if "zarr_consolidated_format" in doc and ( - not isinstance(doc["zarr_consolidated_format"], int) - or isinstance(doc["zarr_consolidated_format"], bool) - or doc["zarr_consolidated_format"] != 1 - ): - problems.append( - ValidationProblem( - ("zarr_consolidated_format",), - f"expected 1, got {shown(doc['zarr_consolidated_format'])}", - refused_kind(doc["zarr_consolidated_format"], (1,)), - ) - ) + problems.extend(check_literal(doc, "zarr_consolidated_format", 1)) refined: dict[str, JSONValue] = {} if "metadata" in doc: entries = doc["metadata"] @@ -877,7 +855,7 @@ def from_json(cls, data: object) -> ZarrV2ConsolidatedMetadata: problems.extend(found) refined[key] = entry if len(problems) != 0: - raise MetadataValidationError(problems) + raise MetadataValidationError(with_input(problems, data)) return cls(metadata=refined) @classmethod diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_validation.py b/packages/zarr-metadata/src/zarr_metadata/model/_validation.py index 1e262467ef..65e00161c4 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_validation.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/_validation.py @@ -34,15 +34,16 @@ MetadataValidationError, ValidationProblem, arrays_to_tuples, + not_an_object, + outside_of, refine_json, refine_user_data, - refused_kind, - shown, validate_json, + with_input, ) from zarr_metadata._json import is_canonical_json as _is_canonical_json from zarr_metadata._json import prefixed as _prefix -from zarr_metadata.model._sentinel import UNSET +from zarr_metadata._sentinel import UNSET from zarr_metadata.v2.array import ZarrV2ArrayMetadataJSON from zarr_metadata.v2.group import ZarrV2GroupMetadataJSON from zarr_metadata.v3._definition import ( @@ -152,8 +153,7 @@ def check_literal( ) -> tuple[ValidationProblem, ...]: """One problem if `doc[key]` is present but not `expected`: of its type, when it is not of `expected`'s JSON type, else of its value.""" if key in doc and (type(doc[key]) is not type(expected) or doc[key] != expected): - message = f"expected {shown(expected)}, got {shown(doc[key])}" - return (ValidationProblem((key,), message, refused_kind(doc[key], (expected,))),) + return (outside_of((key,), doc[key], (expected,)),) return () @@ -478,8 +478,7 @@ def read_array_v3( scope already read -- a model's own, handed back -- is taken as it is. """ if not isinstance(value, Mapping): - not_an_object = (ValidationProblem((), "expected an object", "invalid_type"),) - return ZarrV3ArrayMetadataReading(problems=not_an_object), None + return ZarrV3ArrayMetadataReading(problems=not_an_object(value)), None doc = cast("Mapping[object, object]", value) problems: list[ValidationProblem] = list(missing_keys(ARRAY_METADATA_REQUIRED_KEYS_V3, doc)) extra_fields, found = other_members(doc, ARRAY_METADATA_STANDARD_KEYS_V3) @@ -573,7 +572,7 @@ def read_array_v3( chunk=chunk, pipeline=pipeline, storage_transformers=tuple(listed.get("storage_transformers", ())), - problems=tuple(problems), + problems=with_input(problems, doc), ) if len(problems) != 0 or shape is None or attributes is None: return reading, None @@ -629,11 +628,10 @@ def parse_array_metadata_v3( value: object, *, context: Context = CORE_AND_EXTENSIONS ) -> ZarrV3ArrayMetadataJSON: """Return `value` as `ZarrV3ArrayMetadataJSON`, or raise `MetadataValidationError`.""" - normalized = arrays_to_tuples(value) - problems = validate_array_metadata_v3(normalized, context=context) + problems = validate_array_metadata_v3(value, context=context) if len(problems) != 0: raise MetadataValidationError(problems) - return cast("ZarrV3ArrayMetadataJSON", normalized) + return cast("ZarrV3ArrayMetadataJSON", arrays_to_tuples(value)) def validate_array_metadata_v2(value: object) -> tuple[ValidationProblem, ...]: @@ -645,7 +643,7 @@ def validate_array_metadata_v2(value: object) -> tuple[ValidationProblem, ...]: codec configurations (mappings with a string `id`). """ if not isinstance(value, Mapping): - return (ValidationProblem((), "expected an object", "invalid_type"),) + return not_an_object(value) doc = cast("Mapping[object, object]", value) # Unlike the group document ("Other keys MUST NOT be present", # https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v2/v2.0.rst#L313), the v2 array document is open: other keys "SHOULD NOT be @@ -677,10 +675,7 @@ def validate_array_metadata_v2(value: object) -> tuple[ValidationProblem, ...]: ) ) if "order" in doc and doc["order"] not in ("C", "F"): - message = f'expected "C" or "F", got {shown(doc["order"])}' - problems.append( - ValidationProblem(("order",), message, refused_kind(doc["order"], ("C", "F"))) - ) + problems.append(outside_of(("order",), doc["order"], ("C", "F"))) if "compressor" in doc: compressor = doc["compressor"] if compressor is not None: @@ -703,19 +698,14 @@ def validate_array_metadata_v2(value: object) -> tuple[ValidationProblem, ...]: for index, item in enumerate(filters): problems.extend(_prefix("filters", _prefix(index, validate_json(item)))) if "dimension_separator" in doc and doc["dimension_separator"] not in (".", "/"): - separator = doc["dimension_separator"] problems.append( - ValidationProblem( - ("dimension_separator",), - f'expected "." or "/", got {shown(separator)}', - refused_kind(separator, (".", "/")), - ) + outside_of(("dimension_separator",), doc["dimension_separator"], (".", "/")) ) if "fill_value" in doc: problems.extend(_prefix("fill_value", validate_json(doc["fill_value"]))) if "attributes" in doc: problems.extend(validate_attributes(doc["attributes"])) - return tuple(problems) + return with_input(problems, doc) def is_array_metadata_v2(value: object) -> TypeGuard[ZarrV2ArrayMetadataJSON]: @@ -729,11 +719,10 @@ def is_array_metadata_v2(value: object) -> TypeGuard[ZarrV2ArrayMetadataJSON]: def parse_array_metadata_v2(value: object) -> ZarrV2ArrayMetadataJSON: """Return `value` as `ZarrV2ArrayMetadataJSON`, or raise `MetadataValidationError`.""" - normalized = arrays_to_tuples(value) - problems = validate_array_metadata_v2(normalized) + problems = validate_array_metadata_v2(value) if len(problems) != 0: raise MetadataValidationError(problems) - return cast("ZarrV2ArrayMetadataJSON", normalized) + return cast("ZarrV2ArrayMetadataJSON", arrays_to_tuples(value)) def validate_group_metadata_v2(value: object) -> tuple[ValidationProblem, ...]: @@ -743,14 +732,14 @@ def validate_group_metadata_v2(value: object) -> tuple[ValidationProblem, ...]: optional `attributes` mapping folded in from `.zattrs`. """ if not isinstance(value, Mapping): - return (ValidationProblem((), "expected an object", "invalid_type"),) + return not_an_object(value) doc = cast("Mapping[object, object]", value) problems: list[ValidationProblem] = list(missing_keys(GROUP_METADATA_REQUIRED_KEYS_V2, doc)) problems.extend(unexpected_keys(GROUP_METADATA_STANDARD_KEYS_V2, doc)) problems.extend(check_literal(doc, "zarr_format", 2)) if "attributes" in doc: problems.extend(validate_attributes(doc["attributes"])) - return tuple(problems) + return with_input(problems, doc) def is_group_metadata_v2(value: object) -> TypeGuard[ZarrV2GroupMetadataJSON]: @@ -760,11 +749,10 @@ def is_group_metadata_v2(value: object) -> TypeGuard[ZarrV2GroupMetadataJSON]: def parse_group_metadata_v2(value: object) -> ZarrV2GroupMetadataJSON: """Return `value` narrowed to `ZarrV2GroupMetadataJSON`, or raise `MetadataValidationError`.""" - normalized = arrays_to_tuples(value) - problems = validate_group_metadata_v2(normalized) + problems = validate_group_metadata_v2(value) if len(problems) != 0: raise MetadataValidationError(problems) - return cast(ZarrV2GroupMetadataJSON, normalized) + return cast(ZarrV2GroupMetadataJSON, arrays_to_tuples(value)) StoreKey = TypeVar("StoreKey", bound=str) @@ -798,9 +786,10 @@ def load_store_json(mapping: Mapping[StoreKey, bytes], key: str) -> object: # raises `TypeError` on most else. raw = cast("object", stored[key]) if not isinstance(raw, bytes): - raise MetadataValidationError( - [ValidationProblem((key,), f"expected bytes, got {type(raw).__name__}", "invalid_type")] + refused = ValidationProblem( + (key,), f"expected bytes, got {type(raw).__name__}", "invalid_type" ) + raise MetadataValidationError(with_input((refused,), stored)) try: return json.loads(raw) except (UnicodeDecodeError, ValueError) as exc: diff --git a/packages/zarr-metadata/src/zarr_metadata/typed_json.py b/packages/zarr-metadata/src/zarr_metadata/typed_json.py index 6fb5c14db4..097411a898 100644 --- a/packages/zarr-metadata/src/zarr_metadata/typed_json.py +++ b/packages/zarr-metadata/src/zarr_metadata/typed_json.py @@ -31,10 +31,18 @@ - Each annotation is evaluated in the module of the class that wrote it, as the spec has it, where `get_type_hints` would read what a subclass inherited in the subclass's module. -- `ReadOnly` and `Annotated` are peeled wherever they are written; a - `NewType` reads as the type it names, a type alias as the type it - stands for, and a TypedDict or alias that holds itself as deep as the - value goes. +- `Required`, `NotRequired` and `ReadOnly` are peeled wherever they are + written; a `NewType` reads as the type it names, a type alias as the + type it stands for, and a TypedDict or alias that holds itself as deep + as the value goes. +- A number's type may carry bounds, in annotated-types' vocabulary as + pydantic reads it: `Gt`, `Ge`, `Lt`, `Le` and `Interval`, at any depth, + so `tuple[Annotated[int, Ge(1)], ...]` bounds each element. A value out + of them is a problem, `invalid_value`, whose message says what the type + admits and whose `ctx` holds the bounds. `Annotated` may also carry a + note, a string or a `Doc`; any other metadata is a `TypeError`, since + a constraint `check` does not read would be one it does not hold a + value to. - A union of TypedDicts that each require a key as a `Literal` of values of their own is read by the branch that key names, and its problems are that branch's. Otherwise a value is read by the first branch it @@ -47,7 +55,10 @@ exceptions: `ValidationProblem(loc, message, kind)`, with `kind` one of `missing_key`, `invalid_type`, `invalid_value`, `unknown_key` and `invalid_json`, so a caller that tolerates a key the TypedDict does not -declare can tell it from a wrong value. +declare can tell it from a wrong value. Each carries its message's data, +as pydantic's errors and zod's issues do: `input`, the JSON the value +holds at `loc`, and `ctx`, what was expected there -- a type's bounds, or +the values of a `Literal`. `check` reads the shapes JSON takes and no others -- `int`, `float` for any number, `bool`, `str`, `None`, `JSONValue`, a `Literal`, diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/_common.py b/packages/zarr-metadata/src/zarr_metadata/v3/_common.py index bd412783fd..04a0874bf6 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/_common.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/_common.py @@ -17,6 +17,7 @@ is_canonical_json, prefixed, validate_json, + with_input, ) ZarrV3MetadataFieldJSON = str | ZarrV3NamedConfigJSON @@ -41,7 +42,7 @@ def validate_metadata_field_v3( nothing else. """ envelope = envelope_problems(value, allow_must_understand_false=allow_must_understand_false) - return (*envelope, *_configuration_json_problems(value)) + return with_input((*envelope, *_configuration_json_problems(value)), value) def envelope_problems( @@ -134,11 +135,10 @@ def is_metadata_field_v3(value: object) -> TypeGuard[ZarrV3MetadataFieldJSON]: def parse_metadata_field_v3(value: object) -> ZarrV3MetadataFieldJSON: """Return `value` narrowed to `ZarrV3MetadataFieldJSON`, or raise `MetadataValidationError`.""" - normalized = arrays_to_tuples(value) - problems = validate_metadata_field_v3(normalized) + problems = validate_metadata_field_v3(value) if len(problems) != 0: raise MetadataValidationError(problems) - return cast(ZarrV3MetadataFieldJSON, normalized) + return cast(ZarrV3MetadataFieldJSON, arrays_to_tuples(value)) __all__ = [ diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py b/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py index dfa8189f36..fcf4dc9a6d 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py @@ -51,7 +51,8 @@ from typing_extensions import TypeAliasType, TypedDict, TypeVar, is_typeddict from zarr_metadata._common import JSONValue, ZarrV3NamedConfigJSON -from zarr_metadata._json import ValidationProblem, copied, refine_json, shown +from zarr_metadata._json import ValidationProblem, copied, refine_json, shown, with_input +from zarr_metadata._sentinel import UNSET from zarr_metadata._typed_json import ( Loc, Parsed, @@ -156,9 +157,10 @@ class Definition(Generic[C]): type; with `closed=False`, anything. `rules` yields what the spec disallows in a configuration of that type, as it finds each; it is handed only a configuration that has passed the check, holding what - the TypedDict admits and nothing else, and the fields it holds as the - scope read them -- a struct's field types -- which is nothing when no - scope read it. `judge` is the two, for a caller holding JSON. + the TypedDict admits and nothing else, each member within the bounds + its type carries, and the fields it holds as the scope read them -- a + struct's field types -- which is nothing when no scope read it. + `judge` is the two, for a caller holding JSON. `canonical` is where two spellings of the configuration that mean the same thing are made one. @@ -212,16 +214,17 @@ def check(self, value: object, loc: Loc = ()) -> tuple[C | None, Problems]: `zarr_metadata.typed_json.check` is the type check alone; this also judges the envelope of each metadata field a member holds. """ - return _configuration_checked(value, self.configuration, loc) + configuration, problems = _configuration_checked(value, self.configuration, loc) + return configuration, with_input(problems, value, loc) def judge(self, value: object, loc: Loc = ()) -> tuple[C | None, Problems]: """`value` type-checked, then judged by the rules: the configuration if it holds, and every problem. - The rules are asked only of a configuration that type-checked and - whose nested fields are well formed, holding what its TypedDict - admits and nothing else, so a caller holding JSON never reaches a - rule with a member of the wrong type, or one the type says cannot - be there. No scope reads the fields it holds, so the rules see + The rules are asked only of a configuration that type-checked, + its bounds kept, and whose nested fields are well formed, holding + what its TypedDict admits and nothing else, so a caller holding + JSON never reaches a rule with a member of the wrong type, out of + its bounds, or one the type says cannot be there. No scope reads the fields it holds, so the rules see none of them read, and a rule about one -- a struct's field of a type whose values vary in size -- finds nothing to judge: `resolve` reads the field in a scope, and asks every rule. @@ -230,7 +233,10 @@ def judge(self, value: object, loc: Loc = ()) -> tuple[C | None, Problems]: if configuration is None: return None, problems refused = ruled(self, lambda: self.rules(configuration, _nothing_nested()), loc) - return (configuration if len(refused) == 0 else None), (*problems, *refused) + return (configuration if len(refused) == 0 else None), ( + *problems, + *with_input(refused, value, loc), + ) def _malformed(definition: Definition[Any]) -> str | None: @@ -309,9 +315,10 @@ class DataTypeDefinition(Definition[C]): """A data type, and the fill value an array of it takes. `fill_value` is the JSON shape of a fill value -- `Int8FillValue`, an - annotation the checker reads as it reads a configuration's members -- - and `fill_value_rules` is what the spec disallows in a fill value of - that shape: an integer out of range, a hex string of another width. + annotation the checker reads as it reads a configuration's members, + its range among it -- and `fill_value_rules` is what the spec + disallows in a fill value of that shape that the type cannot say: a + hex string of another width. The rules are handed the configuration, the fields it holds as the scope read them (a struct's field types), and the typed fill value. A data type that says nothing of its fill value takes any JSON. @@ -719,9 +726,7 @@ def ruled( def _located(prefix: Loc, problems: Iterable[ValidationProblem]) -> Problems: - return tuple( - ValidationProblem((*prefix, *found.loc), found.message, found.kind) for found in problems - ) + return tuple(dataclasses.replace(found, loc=(*prefix, *found.loc)) for found in problems) def _envelope(field: _NestedField) -> Problems: @@ -1062,17 +1067,17 @@ def fill_value_problems( """ refined, problems = refine_json(value, loc) if len(problems) != 0 or not isinstance(data_type, Read): - return problems + return with_input(problems, value, loc) definition, configuration = data_type.definition, data_type.configuration typed, problems = _fill_value_parser(definition.fill_value)(refined, loc) if not _usable(problems): - return problems + return with_input(problems, value, loc) refused = ruled( definition, lambda: definition.fill_value_rules(configuration, data_type.nested, typed), loc, ) - return (*problems, *refused) + return with_input((*problems, *refused), value, loc) def storage_of(data_type: Resolved[DataTypeDefinition[Any]]) -> StorageClass | None: @@ -1127,7 +1132,7 @@ def chunk_grid_lengths( definition, lambda: definition.shape_rules(configuration, chunk_grid.nested, shape), at ) if len(problems) != 0: - return unknown, problems + return unknown, with_input(problems, chunk_grid.json, loc) lengths = asked( definition, "chunk lengths", @@ -1178,9 +1183,9 @@ def resolve( name = named_configuration(data)[0] claimant = None if name is None else context.claimant(asked, name) refused = Refused(json=None, name=name, read_as=asked, definition=claimant) - return cast("Resolved[D]", refused), problems + return cast("Resolved[D]", refused), with_input(problems, data, loc) resolved, found = _resolve_field(refined, asked, context, loc) - return cast("Resolved[D]", resolved), found + return cast("Resolved[D]", resolved), with_input(found, data, loc) def _resolve_field( @@ -1280,7 +1285,13 @@ def _read_carried( EmptyConfiguration, {} if given is None else given, (*loc, "configuration") ) configuration, judged = definition.judge(carried) - problems = (*beside, *(ValidationProblem(loc, found.message, found.kind) for found in judged)) + # What is wrong with what the name carries is the field's: found at + # the field, where the name is what is there, and what was expected + # of a member of the configuration is not expected of it. + problems = ( + *beside, + *(dataclasses.replace(found, loc=loc, input=UNSET, ctx={}) for found in judged), + ) if configuration is None or not _usable(problems): refused = Refused(json=data, name=name, read_as=DataTypeDefinition, definition=definition) return refused, problems diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/_pipeline.py b/packages/zarr-metadata/src/zarr_metadata/v3/_pipeline.py index 9435d71d6e..6a65295c01 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/_pipeline.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/_pipeline.py @@ -30,7 +30,7 @@ from dataclasses import dataclass from typing import TYPE_CHECKING, Any, Final, TypeGuard, cast -from zarr_metadata._json import ValidationProblem +from zarr_metadata._json import ValidationProblem, with_input from zarr_metadata.v3._definition import ( Chunk, CodecDefinition, @@ -127,7 +127,9 @@ def read_pipeline( handed = Chunk() else: handed = _handed_on(definition, codec.configuration, codec.nested, incoming, at) - return tuple(stages), tuple(problems) + if len(problems) == 0: + return tuple(stages), () + return tuple(stages), with_input(problems, tuple(codec.json for codec in codecs), loc) def _order_problems( diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/chunk_grid/rectilinear.py b/packages/zarr-metadata/src/zarr_metadata/v3/chunk_grid/rectilinear.py index a8191727a8..157b682e42 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/chunk_grid/rectilinear.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/chunk_grid/rectilinear.py @@ -5,12 +5,12 @@ """ from collections.abc import Iterator -from typing import Final, Literal, NotRequired +from typing import Annotated, Final, Literal, NotRequired +from annotated_types import Ge from typing_extensions import TypedDict from zarr_metadata._json import ValidationProblem -from zarr_metadata._typed_json import Loc from zarr_metadata.v3._definition import ChunkGridDefinition, Lengths, Nested RECTILINEAR_CHUNK_GRID_NAME: Final = "rectilinear" @@ -19,12 +19,14 @@ RectilinearChunkGridName = Literal["rectilinear"] """Literal type of the `name` field of the rectilinear chunk grid.""" -RectilinearDimSpec = int | tuple[int | tuple[int, int], ...] +_Positive = Annotated[int, Ge(1)] + +RectilinearDimSpec = _Positive | tuple[_Positive | tuple[_Positive, _Positive], ...] """JSON shape for one dimension's rectilinear spec. Either a bare integer (uniform shorthand for a regular dimension within a rectilinear grid), or a tuple of integers and/or `[value, count]` RLE -pairs. +pairs. Every extent, and every run's length and count, is at least 1. """ @@ -53,29 +55,6 @@ class RectilinearChunkGridObject(TypedDict, closed=True): """ -def _not_positive(loc: Loc, value: int) -> ValidationProblem: - return ValidationProblem(loc, f"expected an integer >= 1, got {value}", "invalid_value") - - -def _rules( - configuration: RectilinearChunkGridConfiguration, nested: Nested -) -> Iterator[ValidationProblem]: - """Every extent, and every run's length and count, is at least 1.""" - for axis, spec in enumerate(configuration["chunk_shapes"]): - if isinstance(spec, int): - if spec < 1: - yield _not_positive(("chunk_shapes", axis), spec) - continue - for index, entry in enumerate(spec): - if isinstance(entry, int): - if entry < 1: - yield _not_positive(("chunk_shapes", axis, index), entry) - continue - for position, value in enumerate(entry): - if value < 1: - yield _not_positive(("chunk_shapes", axis, index, position), value) - - def canonical_dim_spec(spec: RectilinearDimSpec) -> RectilinearDimSpec: """One dimension's chunk sizes in their simplest equivalent form. @@ -178,7 +157,6 @@ def _chunk_lengths( RECTILINEAR_CHUNK_GRID: Final = ChunkGridDefinition( name=RECTILINEAR_CHUNK_GRID_NAME, configuration=RectilinearChunkGridConfiguration, - rules=_rules, canonical=_canonical, shape_rules=_shape_rules, chunk_lengths=_chunk_lengths, diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/chunk_grid/regular.py b/packages/zarr-metadata/src/zarr_metadata/v3/chunk_grid/regular.py index 444ec11f4a..179fdcc5cc 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/chunk_grid/regular.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/chunk_grid/regular.py @@ -5,8 +5,9 @@ """ from collections.abc import Iterator -from typing import Final, Literal, NotRequired +from typing import Annotated, Final, Literal, NotRequired +from annotated_types import Ge from typing_extensions import TypedDict from zarr_metadata._json import ValidationProblem @@ -20,9 +21,16 @@ class RegularChunkGridConfiguration(TypedDict, closed=True): - """Configuration for the regular chunk grid.""" + """Configuration for the regular chunk grid. - chunk_shape: tuple[int, ...] + No chunk extent is negative. "The chunk shape elements are non-zero + when the corresponding dimensions of the arrays have non-zero length": + an extent of 0 is right for a dimension of length 0, which zarr-python + 3.0 and 3.1 wrote, and which a grid alone cannot tell from one that is + not; the shape rules can, beside the array's shape. + """ + + chunk_shape: tuple[Annotated[int, Ge(0)], ...] class RegularChunkGridObject(TypedDict, closed=True): @@ -43,24 +51,6 @@ class RegularChunkGridObject(TypedDict, closed=True): """ -def _rules( - configuration: RegularChunkGridConfiguration, nested: Nested -) -> Iterator[ValidationProblem]: - """No chunk extent is negative. - - "The chunk shape elements are non-zero when the corresponding - dimensions of the arrays have non-zero length": an extent of 0 is - right for a dimension of length 0, which zarr-python 3.0 and 3.1 - wrote, and which a grid alone cannot tell from one that is not; the - shape rules can, beside the array's shape. - """ - for index, extent in enumerate(configuration["chunk_shape"]): - if extent < 0: - yield ValidationProblem( - ("chunk_shape", index), f"expected an integer >= 0, got {extent}", "invalid_value" - ) - - def _shape_rules( configuration: RegularChunkGridConfiguration, nested: Nested, shape: tuple[int, ...] ) -> Iterator[ValidationProblem]: @@ -87,6 +77,7 @@ def _shape_rules( ("chunk_shape", axis), f"expected a chunk length >= 1 for a dimension of length {extent}, got 0", "invalid_value", + ctx={"ge": 1}, ) @@ -107,7 +98,6 @@ def _chunk_lengths( REGULAR_CHUNK_GRID: Final = ChunkGridDefinition( name=REGULAR_CHUNK_GRID_NAME, configuration=RegularChunkGridConfiguration, - rules=_rules, shape_rules=_shape_rules, chunk_lengths=_chunk_lengths, ) diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/codec/blosc.py b/packages/zarr-metadata/src/zarr_metadata/v3/codec/blosc.py index fb162c537d..c5aed10ced 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/codec/blosc.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/codec/blosc.py @@ -5,8 +5,9 @@ """ from collections.abc import Iterator -from typing import Final, Literal, NotRequired, cast +from typing import Annotated, Final, Literal, NotRequired, cast +from annotated_types import Ge, Interval from typing_extensions import TypedDict from zarr_metadata._json import ValidationProblem @@ -35,9 +36,9 @@ class BloscCodecConfiguration(TypedDict, closed=True): """Configuration for the Zarr v3 `blosc` codec.""" cname: BloscCName - clevel: int + clevel: Annotated[int, Interval(ge=0, le=9)] shuffle: BloscShuffle - blocksize: int + blocksize: Annotated[int, Ge(0)] typesize: NotRequired[int] @@ -69,21 +70,13 @@ class BloscCodecObject(TypedDict, closed=True): def _rules(configuration: BloscCodecConfiguration, nested: Nested) -> Iterator[ValidationProblem]: - """Bounds on `clevel` and `blocksize`; `typesize` against `shuffle`. + """`typesize` against `shuffle`. Under `noshuffle` the spec says of `typesize` that "the value is ignored", and the canonical form drops it; under either shuffle it is - required, and positive. + required, and positive. Whether it is required is not the type's to + say, so neither is its bound. """ - clevel, blocksize = configuration["clevel"], configuration["blocksize"] - if not 0 <= clevel <= 9: - yield ValidationProblem( - ("clevel",), f"expected an integer in [0, 9], got {clevel}", "invalid_value" - ) - if blocksize < 0: - yield ValidationProblem( - ("blocksize",), f"expected an integer >= 0, got {blocksize}", "invalid_value" - ) shuffle = configuration["shuffle"] if shuffle != BLOSC_NO_SHUFFLE: typesize = configuration.get("typesize") @@ -93,7 +86,10 @@ def _rules(configuration: BloscCodecConfiguration, nested: Nested) -> Iterator[V ) elif typesize < 1: yield ValidationProblem( - ("typesize",), f"expected a positive integer, got {typesize}", "invalid_value" + ("typesize",), + f"expected a positive integer, got {typesize}", + "invalid_value", + ctx={"ge": 1}, ) diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/codec/gzip.py b/packages/zarr-metadata/src/zarr_metadata/v3/codec/gzip.py index aa07d18421..66e535a165 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/codec/gzip.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/codec/gzip.py @@ -4,13 +4,12 @@ See https://zarr-specs.readthedocs.io/en/latest/v3/codecs/gzip/index.html """ -from collections.abc import Iterator -from typing import Final, Literal, NotRequired +from typing import Annotated, Final, Literal, NotRequired +from annotated_types import Interval from typing_extensions import TypedDict -from zarr_metadata._json import ValidationProblem -from zarr_metadata.v3._definition import CodecDefinition, Nested +from zarr_metadata.v3._definition import CodecDefinition GZIP_CODEC_NAME: Final = "gzip" """The `name` field value of the `gzip` codec.""" @@ -33,7 +32,7 @@ class GzipCodecConfiguration(TypedDict, closed=True): https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v3/codecs/gzip/index.rst#L57-L66 """ - level: int + level: Annotated[int, Interval(ge=0, le=9)] class GzipCodecObject(TypedDict, closed=True): @@ -53,21 +52,11 @@ class GzipCodecObject(TypedDict, closed=True): """ -def _rules(configuration: GzipCodecConfiguration, nested: Nested) -> Iterator[ValidationProblem]: - """`level` is an integer from 0 to 9.""" - level = configuration["level"] - if not 0 <= level <= 9: - yield ValidationProblem( - ("level",), f"expected an integer in [0, 9], got {level}", "invalid_value" - ) - - GZIP_CODEC: Final = CodecDefinition( name=GZIP_CODEC_NAME, configuration=GzipCodecConfiguration, kind="bytes_bytes", size="dynamic", - rules=_rules, ) """The `gzip` codec.""" diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/codec/sharding_indexed.py b/packages/zarr-metadata/src/zarr_metadata/v3/codec/sharding_indexed.py index b7d42aa1e1..2bf5bd8020 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/codec/sharding_indexed.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/codec/sharding_indexed.py @@ -5,8 +5,9 @@ """ from collections.abc import Iterator, Mapping -from typing import Final, Literal, NotRequired +from typing import Annotated, Final, Literal, NotRequired +from annotated_types import Ge from typing_extensions import TypedDict from zarr_metadata._json import ValidationProblem @@ -53,7 +54,7 @@ class ShardingIndexedCodecConfiguration(TypedDict, closed=True): https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v3/codecs/sharding-indexed/index.rst#L157-L161 """ - chunk_shape: tuple[int, ...] + chunk_shape: tuple[Annotated[int, Ge(1)], ...] codecs: tuple[CodecField, ...] index_codecs: tuple[StaticCodecField, ...] index_location: NotRequired[ShardingIndexLocation] @@ -78,17 +79,6 @@ class ShardingIndexedCodecObject(TypedDict, closed=True): """ -def _rules( - configuration: ShardingIndexedCodecConfiguration, nested: Nested -) -> Iterator[ValidationProblem]: - """Every inner chunk extent is at least 1.""" - for index, extent in enumerate(configuration["chunk_shape"]): - if extent < 1: - yield ValidationProblem( - ("chunk_shape", index), f"expected an integer >= 1, got {extent}", "invalid_value" - ) - - def _chunk_rules( configuration: ShardingIndexedCodecConfiguration, nested: Nested, chunk: Chunk ) -> Iterator[ValidationProblem]: @@ -180,7 +170,6 @@ def _per_shard(chunk_shape: tuple[int, ...], lengths: Lengths | None) -> Lengths configuration=ShardingIndexedCodecConfiguration, kind="array_bytes", size="dynamic", - rules=_rules, chunk_rules=_chunk_rules, pipelines=_pipelines, ) diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/codec/zstd.py b/packages/zarr-metadata/src/zarr_metadata/v3/codec/zstd.py index efeefb7edb..61a0695b61 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/codec/zstd.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/codec/zstd.py @@ -6,13 +6,12 @@ proposed the codec, was never merged). """ -from collections.abc import Iterator -from typing import Final, Literal, NotRequired +from typing import Annotated, Final, Literal, NotRequired +from annotated_types import Interval from typing_extensions import TypedDict -from zarr_metadata._json import ValidationProblem -from zarr_metadata.v3._definition import CodecDefinition, Nested +from zarr_metadata.v3._definition import CodecDefinition ZSTD_CODEC_NAME: Final = "zstd" """The `name` field value of the `zstd` codec.""" @@ -20,6 +19,12 @@ ZstdCodecName = Literal["zstd"] """Literal type of the `name` field of the `zstd` codec.""" +ZSTD_MIN_LEVEL: Final = -131072 +"""The lowest `level` zstd accepts: ZSTD_minCLevel(), -(1 << 17).""" + +ZSTD_MAX_LEVEL: Final = 22 +"""The highest `level` zstd accepts: ZSTD_maxCLevel().""" + class ZstdCodecConfiguration(TypedDict, closed=True): """ @@ -30,7 +35,7 @@ class ZstdCodecConfiguration(TypedDict, closed=True): https://github.com/zarr-developers/zarr-extensions/blob/4da7b37a84f76e660902f6d3de3eaef0e0febae6/codecs/zstd/README.md#L9-L19 """ - level: int + level: Annotated[int, Interval(ge=ZSTD_MIN_LEVEL, le=ZSTD_MAX_LEVEL)] checksum: NotRequired[bool] @@ -51,30 +56,11 @@ class ZstdCodecObject(TypedDict, closed=True): https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v3/core/index.rst#L1562-L1564 (short-hand names only "if no configuration metadata is required") """ -ZSTD_MIN_LEVEL: Final = -131072 -"""The lowest `level` zstd accepts: ZSTD_minCLevel(), -(1 << 17).""" - -ZSTD_MAX_LEVEL: Final = 22 -"""The highest `level` zstd accepts: ZSTD_maxCLevel().""" - - -def _rules(configuration: ZstdCodecConfiguration, nested: Nested) -> Iterator[ValidationProblem]: - """`level` is one zstd accepts.""" - level = configuration["level"] - if not ZSTD_MIN_LEVEL <= level <= ZSTD_MAX_LEVEL: - yield ValidationProblem( - ("level",), - f"expected an integer in [{ZSTD_MIN_LEVEL}, {ZSTD_MAX_LEVEL}], got {level}", - "invalid_value", - ) - - ZSTD_CODEC: Final = CodecDefinition( name=ZSTD_CODEC_NAME, configuration=ZstdCodecConfiguration, kind="bytes_bytes", size="dynamic", - rules=_rules, ) """The `zstd` codec.""" diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/_byte.py b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/_byte.py new file mode 100644 index 0000000000..0d4ed5c98e --- /dev/null +++ b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/_byte.py @@ -0,0 +1,14 @@ +"""A byte value, as the fill values of raw bits and of `bytes` hold them. + +Each is an integer in `[0, 255]` +(https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v3/data-types/index.rst#L59-L61). +""" + +from typing import Annotated + +from annotated_types import Interval + +ByteValue = Annotated[int, Interval(ge=0, le=255)] +"""One byte of a fill value: a JSON integer in `[0, 255]`.""" + +__all__ = ["ByteValue"] diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/_integer.py b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/_integer.py deleted file mode 100644 index 0a8320ee19..0000000000 --- a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/_integer.py +++ /dev/null @@ -1,49 +0,0 @@ -"""The fill value rules integers share: a JSON integer within a range. - -The integer data types take one within their own range, and the byte -values of `r*` and `bytes` fill values are integers in `[0, 255]` -(https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v3/data-types/index.rst#L59-L61). -""" - -from collections.abc import Callable, Iterator -from dataclasses import dataclass - -from zarr_metadata._json import ValidationProblem -from zarr_metadata.v3._definition import EmptyConfiguration, Nested - - -def integer_fill_value_rules( - low: int, high: int -) -> Callable[[EmptyConfiguration, Nested, int], Iterator[ValidationProblem]]: - """The fill value rules of an integer data type whose range is `[low, high]`.""" - return _InRange(low, high) - - -@dataclass(frozen=True, slots=True) -class _InRange: - """An integer in `[low, high]`: a value rather than a closure, so a definition holding it is equal to itself after a pickle or a deep copy.""" - - low: int - high: int - - def __call__( - self, configuration: EmptyConfiguration, nested: Nested, value: int - ) -> Iterator[ValidationProblem]: - if not self.low <= value <= self.high: - yield ValidationProblem( - (), - f"expected an integer in [{self.low}, {self.high}], got {value}", - "invalid_value", - ) - - -def byte_value_problems(values: tuple[int, ...]) -> Iterator[ValidationProblem]: - """Each of `values` that is not a byte, an integer in `[0, 255]`, located at its index.""" - for index, value in enumerate(values): - if not 0 <= value <= 255: - yield ValidationProblem( - (index,), f"expected an integer in [0, 255], got {value}", "invalid_value" - ) - - -__all__ = ["byte_value_problems", "integer_fill_value_rules"] diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/_numpy_time.py b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/_numpy_time.py index 26beaccdeb..ffc0ce2aab 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/_numpy_time.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/_numpy_time.py @@ -1,21 +1,29 @@ -"""What the two numpy time types share: a unit, and how many of it one tick is. +"""What the two numpy time types share: how many of a unit one tick is, a count of ticks, and how their values are stored. -Both types' configurations have these two members and the one rule on -them, so the rule is written here once and neither sibling imports it -from the other. So is how their values are stored. +Both types' configurations hold a `scale_factor` of one type, and both +types' fill values count ticks of one width, so each is written here once +and neither sibling imports it from the other. So is how their values are +stored. """ -from collections.abc import Iterator -from typing import Final +from typing import Annotated, Final -from typing_extensions import ReadOnly, TypedDict +from annotated_types import Interval -from zarr_metadata._json import ValidationProblem -from zarr_metadata.v3._definition import Nested, multi_byte +from zarr_metadata.v3._definition import multi_byte NUMPY_TIME_MAX_SCALE_FACTOR: Final = 2**31 - 1 """The largest `scale_factor` numpy stores: the field is a signed int32.""" +NumpyTimeScaleFactor = Annotated[int, Interval(ge=1, le=NUMPY_TIME_MAX_SCALE_FACTOR)] +"""How many of the unit one tick is: a positive int32.""" + +NumpyTimeTicks = Annotated[int, Interval(ge=-(2**63), le=2**63 - 1)] +"""A count of ticks, as a fill value writes one: a signed 64-bit integer. + +https://github.com/zarr-developers/zarr-extensions/blob/4da7b37a84f76e660902f6d3de3eaef0e0febae6/data-types/numpy.datetime64/README.md?plain=1#L109-L112 +""" + numpy_time_storage: Final = multi_byte """Signed 64-bit integers, in the byte order the codecs say. @@ -29,43 +37,9 @@ """ -class NumpyTimeConfiguration(TypedDict): - """The members both numpy time types' configurations have, read-only so either type fits.""" - - unit: ReadOnly[str] - scale_factor: ReadOnly[int] - - -def numpy_time_rules( - configuration: NumpyTimeConfiguration, nested: Nested -) -> Iterator[ValidationProblem]: - """`scale_factor` is a positive int32.""" - scale_factor = configuration["scale_factor"] - if not 1 <= scale_factor <= NUMPY_TIME_MAX_SCALE_FACTOR: - yield ValidationProblem( - ("scale_factor",), - f"expected an integer in [1, {NUMPY_TIME_MAX_SCALE_FACTOR}], got {scale_factor}", - "invalid_value", - ) - - -def numpy_time_fill_value_rules( - configuration: NumpyTimeConfiguration, nested: Nested, value: int | str -) -> Iterator[ValidationProblem]: - """An integer fill value is a signed 64-bit one; `"NaT"` is the other form, which the shape admits. - - https://github.com/zarr-developers/zarr-extensions/blob/4da7b37a84f76e660902f6d3de3eaef0e0febae6/data-types/numpy.datetime64/README.md?plain=1#L109-L112 - """ - if isinstance(value, int) and not -(2**63) <= value <= 2**63 - 1: - yield ValidationProblem( - (), f'expected a signed 64-bit integer or "NaT", got {value}', "invalid_value" - ) - - __all__ = [ "NUMPY_TIME_MAX_SCALE_FACTOR", - "NumpyTimeConfiguration", - "numpy_time_fill_value_rules", - "numpy_time_rules", + "NumpyTimeScaleFactor", + "NumpyTimeTicks", "numpy_time_storage", ] diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/bytes.py b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/bytes.py index c815ac50e8..62a266ffc9 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/bytes.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/bytes.py @@ -15,7 +15,7 @@ Nested, variable_length, ) -from zarr_metadata.v3.data_type._integer import byte_value_problems +from zarr_metadata.v3.data_type._byte import ByteValue BYTES_DATA_TYPE_NAME: Final = "bytes" """The `data_type` value for the variable-length `bytes` type.""" @@ -41,7 +41,7 @@ def base64_bytes(value: str) -> Base64Bytes: return Base64Bytes(value) -BytesFillValue = tuple[int, ...] | Base64Bytes +BytesFillValue = tuple[ByteValue, ...] | Base64Bytes """Permitted JSON shape of the `fill_value` field for `bytes`. Either a JSON array of integers in `[0, 255]` (one per byte), or a @@ -52,9 +52,8 @@ def base64_bytes(value: str) -> Base64Bytes: def _fill_value_rules( configuration: EmptyConfiguration, nested: Nested, value: BytesFillValue ) -> Iterator[ValidationProblem]: - """Integers in `[0, 255]`, or a string of standard-alphabet base64.""" + """A string of standard-alphabet base64, when it is not byte values.""" if not isinstance(value, str): - yield from byte_value_problems(value) return try: base64_bytes(value) diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/int16.py b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/int16.py index 22ff1fd8a4..adf03467b2 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/int16.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/int16.py @@ -4,10 +4,11 @@ See https://zarr-specs.readthedocs.io/en/latest/v3/data-types/index.html """ -from typing import Final, Literal +from typing import Annotated, Final, Literal + +from annotated_types import Interval from zarr_metadata.v3._definition import DataTypeDefinition, EmptyConfiguration, multi_byte -from zarr_metadata.v3.data_type._integer import integer_fill_value_rules INT16_DATA_TYPE_NAME: Final = "int16" """The `data_type` value for the `int16` type.""" @@ -15,7 +16,7 @@ Int16DataTypeName = Literal["int16"] """Literal type of the `data_type` field for `int16`.""" -Int16FillValue = int +Int16FillValue = Annotated[int, Interval(ge=-(2**15), le=2**15 - 1)] """Permitted JSON shape of the `fill_value` field for `int16`: a JSON integer in [-32768, 32767].""" @@ -23,7 +24,6 @@ name=INT16_DATA_TYPE_NAME, configuration=EmptyConfiguration, fill_value=Int16FillValue, - fill_value_rules=integer_fill_value_rules(-(2**15), 2**15 - 1), storage=multi_byte, ) """The `int16` data type: a bare name, with nothing to configure; its fill value an integer in its range.""" diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/int32.py b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/int32.py index 98a039a5d9..6a87f44476 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/int32.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/int32.py @@ -4,10 +4,11 @@ See https://zarr-specs.readthedocs.io/en/latest/v3/data-types/index.html """ -from typing import Final, Literal +from typing import Annotated, Final, Literal + +from annotated_types import Interval from zarr_metadata.v3._definition import DataTypeDefinition, EmptyConfiguration, multi_byte -from zarr_metadata.v3.data_type._integer import integer_fill_value_rules INT32_DATA_TYPE_NAME: Final = "int32" """The `data_type` value for the `int32` type.""" @@ -15,7 +16,7 @@ Int32DataTypeName = Literal["int32"] """Literal type of the `data_type` field for `int32`.""" -Int32FillValue = int +Int32FillValue = Annotated[int, Interval(ge=-(2**31), le=2**31 - 1)] """Permitted JSON shape of the `fill_value` field for `int32`: a JSON integer in [-2**31, 2**31 - 1].""" @@ -23,7 +24,6 @@ name=INT32_DATA_TYPE_NAME, configuration=EmptyConfiguration, fill_value=Int32FillValue, - fill_value_rules=integer_fill_value_rules(-(2**31), 2**31 - 1), storage=multi_byte, ) """The `int32` data type: a bare name, with nothing to configure; its fill value an integer in its range.""" diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/int64.py b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/int64.py index b49d5021b7..707bf88a90 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/int64.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/int64.py @@ -4,10 +4,11 @@ See https://zarr-specs.readthedocs.io/en/latest/v3/data-types/index.html """ -from typing import Final, Literal +from typing import Annotated, Final, Literal + +from annotated_types import Interval from zarr_metadata.v3._definition import DataTypeDefinition, EmptyConfiguration, multi_byte -from zarr_metadata.v3.data_type._integer import integer_fill_value_rules INT64_DATA_TYPE_NAME: Final = "int64" """The `data_type` value for the `int64` type.""" @@ -15,7 +16,7 @@ Int64DataTypeName = Literal["int64"] """Literal type of the `data_type` field for `int64`.""" -Int64FillValue = int +Int64FillValue = Annotated[int, Interval(ge=-(2**63), le=2**63 - 1)] """Permitted JSON shape of the `fill_value` field for `int64`: a JSON integer in [-2**63, 2**63 - 1].""" @@ -23,7 +24,6 @@ name=INT64_DATA_TYPE_NAME, configuration=EmptyConfiguration, fill_value=Int64FillValue, - fill_value_rules=integer_fill_value_rules(-(2**63), 2**63 - 1), storage=multi_byte, ) """The `int64` data type: a bare name, with nothing to configure; its fill value an integer in its range.""" diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/int8.py b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/int8.py index 7748a491f8..443b5f8a90 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/int8.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/int8.py @@ -4,10 +4,11 @@ See https://zarr-specs.readthedocs.io/en/latest/v3/data-types/index.html """ -from typing import Final, Literal +from typing import Annotated, Final, Literal + +from annotated_types import Interval from zarr_metadata.v3._definition import DataTypeDefinition, EmptyConfiguration, single_byte -from zarr_metadata.v3.data_type._integer import integer_fill_value_rules INT8_DATA_TYPE_NAME: Final = "int8" """The `data_type` value for the `int8` type.""" @@ -15,7 +16,7 @@ Int8DataTypeName = Literal["int8"] """Literal type of the `data_type` field for `int8`.""" -Int8FillValue = int +Int8FillValue = Annotated[int, Interval(ge=-(2**7), le=2**7 - 1)] """Permitted JSON shape of the `fill_value` field for `int8`: a JSON integer in [-128, 127].""" @@ -23,7 +24,6 @@ name=INT8_DATA_TYPE_NAME, configuration=EmptyConfiguration, fill_value=Int8FillValue, - fill_value_rules=integer_fill_value_rules(-(2**7), 2**7 - 1), storage=single_byte, ) """The `int8` data type: a bare name, with nothing to configure; its fill value an integer in its range.""" diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/numpy_datetime64.py b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/numpy_datetime64.py index f8b2e21280..365df13446 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/numpy_datetime64.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/numpy_datetime64.py @@ -10,8 +10,8 @@ from zarr_metadata.v3._definition import DataTypeDefinition from zarr_metadata.v3.data_type._numpy_time import ( - numpy_time_fill_value_rules, - numpy_time_rules, + NumpyTimeScaleFactor, + NumpyTimeTicks, numpy_time_storage, ) @@ -40,7 +40,7 @@ class NumpyDatetime64Configuration(TypedDict, closed=True): """ unit: ReadOnly[NumpyTimeUnit] - scale_factor: ReadOnly[int] + scale_factor: ReadOnly[NumpyTimeScaleFactor] class NumpyDatetime64(TypedDict, closed=True): @@ -51,7 +51,7 @@ class NumpyDatetime64(TypedDict, closed=True): must_understand: NotRequired[bool] -NumpyDatetime64FillValue = int | Literal["NaT"] +NumpyDatetime64FillValue = NumpyTimeTicks | Literal["NaT"] """Permitted JSON shape of the `fill_value` field for `numpy.datetime64`. Either a JSON integer (count of `unit * scale_factor` since the epoch), @@ -61,9 +61,7 @@ class NumpyDatetime64(TypedDict, closed=True): NUMPY_DATETIME64_DATA_TYPE: Final = DataTypeDefinition( name=NUMPY_DATETIME64_DATA_TYPE_NAME, configuration=NumpyDatetime64Configuration, - rules=numpy_time_rules, fill_value=NumpyDatetime64FillValue, - fill_value_rules=numpy_time_fill_value_rules, storage=numpy_time_storage, ) """The `numpy.datetime64` data type.""" diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/numpy_timedelta64.py b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/numpy_timedelta64.py index 5b0ee23626..2f11f45526 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/numpy_timedelta64.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/numpy_timedelta64.py @@ -10,8 +10,8 @@ from zarr_metadata.v3._definition import DataTypeDefinition from zarr_metadata.v3.data_type._numpy_time import ( - numpy_time_fill_value_rules, - numpy_time_rules, + NumpyTimeScaleFactor, + NumpyTimeTicks, numpy_time_storage, ) @@ -59,7 +59,7 @@ class NumpyTimedelta64Configuration(TypedDict, closed=True): """ unit: ReadOnly[NumpyTimeUnit] - scale_factor: ReadOnly[int] + scale_factor: ReadOnly[NumpyTimeScaleFactor] class NumpyTimedelta64(TypedDict, closed=True): @@ -70,7 +70,7 @@ class NumpyTimedelta64(TypedDict, closed=True): must_understand: NotRequired[bool] -NumpyTimedelta64FillValue = int | Literal["NaT"] +NumpyTimedelta64FillValue = NumpyTimeTicks | Literal["NaT"] """Permitted JSON shape of the `fill_value` field for `numpy.timedelta64`. Either a JSON integer (a count of `unit * scale_factor`), or the string @@ -80,9 +80,7 @@ class NumpyTimedelta64(TypedDict, closed=True): NUMPY_TIMEDELTA64_DATA_TYPE: Final = DataTypeDefinition( name=NUMPY_TIMEDELTA64_DATA_TYPE_NAME, configuration=NumpyTimedelta64Configuration, - rules=numpy_time_rules, fill_value=NumpyTimedelta64FillValue, - fill_value_rules=numpy_time_fill_value_rules, storage=numpy_time_storage, ) """The `numpy.timedelta64` data type.""" diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/raw.py b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/raw.py index e23e523f8c..b7475cb2f2 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/raw.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/raw.py @@ -21,7 +21,7 @@ Nested, StorageClass, ) -from zarr_metadata.v3.data_type._integer import byte_value_problems +from zarr_metadata.v3.data_type._byte import ByteValue RawBytesDataTypeName = NewType("RawBytesDataTypeName", str) """A spec-conformant `r` raw-bytes name (e.g. `"r8"`, `"r16"`). @@ -46,7 +46,7 @@ def raw_bytes_dtype_name(value: str) -> RawBytesDataTypeName: return RawBytesDataTypeName(value) -RawBytesFillValue = tuple[int, ...] +RawBytesFillValue = tuple[ByteValue, ...] """Permitted JSON shape of the `fill_value` field for `r`. A JSON array of N/8 integers in `[0, 255]` (one per byte). @@ -78,7 +78,7 @@ def _rules(configuration: RawBytesConfiguration, nested: Nested) -> Iterator[Val def _fill_value_rules( configuration: RawBytesConfiguration, nested: Nested, value: RawBytesFillValue ) -> Iterator[ValidationProblem]: - """One byte value, an integer in `[0, 255]`, for each 8 of the size. + """One byte value for each 8 of the size. The spec's text says `N` values for `r`, but `N` counts bits (https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v3/core/index.rst#L897-L900), @@ -89,7 +89,6 @@ def _fill_value_rules( yield ValidationProblem( (), f"expected {expected} byte values, got {len(value)}", "invalid_value" ) - yield from byte_value_problems(value) def _storage(configuration: RawBytesConfiguration, nested: Nested) -> StorageClass | None: diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/uint16.py b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/uint16.py index 3578ff03fe..fa461df280 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/uint16.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/uint16.py @@ -4,10 +4,11 @@ See https://zarr-specs.readthedocs.io/en/latest/v3/data-types/index.html """ -from typing import Final, Literal +from typing import Annotated, Final, Literal + +from annotated_types import Interval from zarr_metadata.v3._definition import DataTypeDefinition, EmptyConfiguration, multi_byte -from zarr_metadata.v3.data_type._integer import integer_fill_value_rules UINT16_DATA_TYPE_NAME: Final = "uint16" """The `data_type` value for the `uint16` type.""" @@ -15,7 +16,7 @@ Uint16DataTypeName = Literal["uint16"] """Literal type of the `data_type` field for `uint16`.""" -Uint16FillValue = int +Uint16FillValue = Annotated[int, Interval(ge=0, le=2**16 - 1)] """Permitted JSON shape of the `fill_value` field for `uint16`: a JSON integer in [0, 65535].""" @@ -23,7 +24,6 @@ name=UINT16_DATA_TYPE_NAME, configuration=EmptyConfiguration, fill_value=Uint16FillValue, - fill_value_rules=integer_fill_value_rules(0, 2**16 - 1), storage=multi_byte, ) """The `uint16` data type: a bare name, with nothing to configure; its fill value an integer in its range.""" diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/uint32.py b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/uint32.py index 771d86804d..4cb844acec 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/uint32.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/uint32.py @@ -4,10 +4,11 @@ See https://zarr-specs.readthedocs.io/en/latest/v3/data-types/index.html """ -from typing import Final, Literal +from typing import Annotated, Final, Literal + +from annotated_types import Interval from zarr_metadata.v3._definition import DataTypeDefinition, EmptyConfiguration, multi_byte -from zarr_metadata.v3.data_type._integer import integer_fill_value_rules UINT32_DATA_TYPE_NAME: Final = "uint32" """The `data_type` value for the `uint32` type.""" @@ -15,7 +16,7 @@ Uint32DataTypeName = Literal["uint32"] """Literal type of the `data_type` field for `uint32`.""" -Uint32FillValue = int +Uint32FillValue = Annotated[int, Interval(ge=0, le=2**32 - 1)] """Permitted JSON shape of the `fill_value` field for `uint32`: a JSON integer in [0, 2**32 - 1].""" @@ -23,7 +24,6 @@ name=UINT32_DATA_TYPE_NAME, configuration=EmptyConfiguration, fill_value=Uint32FillValue, - fill_value_rules=integer_fill_value_rules(0, 2**32 - 1), storage=multi_byte, ) """The `uint32` data type: a bare name, with nothing to configure; its fill value an integer in its range.""" diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/uint64.py b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/uint64.py index 4e72073b64..82005090e2 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/uint64.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/uint64.py @@ -4,10 +4,11 @@ See https://zarr-specs.readthedocs.io/en/latest/v3/data-types/index.html """ -from typing import Final, Literal +from typing import Annotated, Final, Literal + +from annotated_types import Interval from zarr_metadata.v3._definition import DataTypeDefinition, EmptyConfiguration, multi_byte -from zarr_metadata.v3.data_type._integer import integer_fill_value_rules UINT64_DATA_TYPE_NAME: Final = "uint64" """The `data_type` value for the `uint64` type.""" @@ -15,7 +16,7 @@ Uint64DataTypeName = Literal["uint64"] """Literal type of the `data_type` field for `uint64`.""" -Uint64FillValue = int +Uint64FillValue = Annotated[int, Interval(ge=0, le=2**64 - 1)] """Permitted JSON shape of the `fill_value` field for `uint64`: a JSON integer in [0, 2**64 - 1].""" @@ -23,7 +24,6 @@ name=UINT64_DATA_TYPE_NAME, configuration=EmptyConfiguration, fill_value=Uint64FillValue, - fill_value_rules=integer_fill_value_rules(0, 2**64 - 1), storage=multi_byte, ) """The `uint64` data type: a bare name, with nothing to configure; its fill value an integer in its range.""" diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/uint8.py b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/uint8.py index 54ea7c59a7..46eb1ff93c 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/uint8.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/uint8.py @@ -4,10 +4,11 @@ See https://zarr-specs.readthedocs.io/en/latest/v3/data-types/index.html """ -from typing import Final, Literal +from typing import Annotated, Final, Literal + +from annotated_types import Interval from zarr_metadata.v3._definition import DataTypeDefinition, EmptyConfiguration, single_byte -from zarr_metadata.v3.data_type._integer import integer_fill_value_rules UINT8_DATA_TYPE_NAME: Final = "uint8" """The `data_type` value for the `uint8` type.""" @@ -15,7 +16,7 @@ Uint8DataTypeName = Literal["uint8"] """Literal type of the `data_type` field for `uint8`.""" -Uint8FillValue = int +Uint8FillValue = Annotated[int, Interval(ge=0, le=2**8 - 1)] """Permitted JSON shape of the `fill_value` field for `uint8`: a JSON integer in [0, 255].""" @@ -23,7 +24,6 @@ name=UINT8_DATA_TYPE_NAME, configuration=EmptyConfiguration, fill_value=Uint8FillValue, - fill_value_rules=integer_fill_value_rules(0, 2**8 - 1), storage=single_byte, ) """The `uint8` data type: a bare name, with nothing to configure; its fill value an integer in its range.""" diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/definition.py b/packages/zarr-metadata/src/zarr_metadata/v3/definition.py index 8ce2f400bc..5d38478607 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/definition.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/definition.py @@ -12,10 +12,12 @@ checked configuration has it as its static type. It reads as the typing spec defines a TypedDict -- `total`, `Required`, `NotRequired`, `closed` and `extra_items` mean what they mean to a type checker, - whether or not its module postpones annotations; + whether or not its module postpones annotations -- and a member's type + carries its bounds, as pydantic reads them: a gzip `level` is + `Annotated[int, Interval(ge=0, le=9)]`; - `rules`, a function over that TypedDict yielding what the spec - disallows -- a bound, members read together -- located in the - configuration. + disallows that a type cannot say -- members read together -- located + in the configuration. What kind of metadata a definition defines is its type: `CodecDefinition` (with the codec's `kind`, and its `size`: whether the size of what it gives @@ -71,6 +73,8 @@ CodecDefinition, CORE_AND_EXTENSIONS) resolved # Refused(..., definition=CodecDefinition(name='gzip'), ...) problems[0].loc # ('configuration', 'level') + problems[0].input # 12 + dict(problems[0].ctx) # {'ge': 0, 'le': 9} resolved, problems = resolve({"name": "gzip", "configuration": {"level": 5}}, CodecDefinition, CORE_AND_EXTENSIONS) @@ -82,24 +86,33 @@ unknown key is still read -- the key reported, the configuration judged without it -- so a consumer that tolerates one filters by kind and uses what was read; the field is valid only when there is no problem at all. +Each problem carries what its message says as data, as pydantic's errors +and zod's issues do: `input`, what was found at `loc` -- the `12` above -- +and `ctx`, what was expected, where that is more than a type: the bounds +`{"ge": 0, "le": 9}`, or the values of a closed set. -**Writing an extension.** A TypedDict, a function for its rules, and a -definition; then a scope that holds it. The TypedDict is a -`typing_extensions.TypedDict`: `closed` and `extra_items` are PEP 728's, -which `typing.TypedDict` does not take on the versions this package -supports. The rules are handed the configuration and the fields it holds -as the scope read them: a field that is read keeps what it read inside it -as `Read.nested`, a `Nested` mapping by where each sits, so a struct's -rules reach its field types. `judge`, which reads in no scope, +**Writing an extension.** A TypedDict, which says what the +configuration's JSON is, bounds and all; a function for the rules a type +cannot say; and a definition; then a scope that holds it. The TypedDict +is a `typing_extensions.TypedDict`: `closed` and `extra_items` are PEP +728's, which `typing.TypedDict` does not take on the versions this +package supports. The rules are handed the configuration and the fields +it holds as the scope read them: a field that is read keeps what it read +inside it as `Read.nested`, a `Nested` mapping by where each sits, so a +struct's rules reach its field types. `judge`, which reads in no scope, hands them none. A rule's message shows a value as the package's own -messages do, as JSON, with `shown`: `null`, `[1, 2]`, `"C"`. Define each -function at a module's top level: a model holds the definitions that -read its fields, so it pickles, and compares equal once loaded, only -when they do -- a lambda or a closure does not pickle, and a -`functools.partial` pickles but compares unequal to itself loaded. +messages do, as JSON, with `shown`: `null`, `[1, 2]`, `"C"`. A rule +reports where a problem is; what is found there is the problem's +`input` without the rule saying so. Define each function at a module's +top level: a model holds the definitions that read its fields, so it +pickles, and compares equal once loaded, only when they do -- a lambda +or a closure does not pickle, and a `functools.partial` pickles but +compares unequal to itself loaded. from collections.abc import Iterator + from typing import Annotated, NotRequired + from annotated_types import Ge from typing_extensions import TypedDict from zarr_metadata.v3.definition import ( @@ -111,14 +124,18 @@ class AcmeLz4Configuration(TypedDict, closed=True): - acceleration: int + acceleration: Annotated[int, Ge(1)] + dictionary: NotRequired[str] + dictionary_size: NotRequired[Annotated[int, Ge(1)]] def acme_lz4_rules( configuration: AcmeLz4Configuration, nested: Nested ) -> Iterator[ValidationProblem]: - if configuration["acceleration"] < 1: - yield ValidationProblem(("acceleration",), "expected an integer >= 1", "invalid_value") + if "dictionary" in configuration and "dictionary_size" not in configuration: + yield ValidationProblem( + ("dictionary_size",), "a dictionary needs its size", "missing_key" + ) ACME_LZ4 = CodecDefinition( @@ -146,7 +163,20 @@ def acme_lz4_rules( word, so a definition refuses it. Its members are the shapes JSON takes: `int`, `float`, `bool`, `str`, `None`, `JSONValue`, a `Literal`, `tuple[T, ...]` and `tuple[T1, T2]`, a union, a TypedDict, -`Mapping[str, V]`, a `NewType` and a type alias. +`Mapping[str, V]`, a `NewType` and a type alias. A number's type may +carry bounds, as annotated-types spells them and pydantic reads them -- +`Gt`, `Ge`, `Lt`, `Le` and `Interval`, one from each side -- at any +depth: `tuple[Annotated[int, Ge(1)], ...]` bounds each element. A value +out of them is a problem, `invalid_value`, whose message says what the +type admits, "expected an integer >= 1, got 0", and whose `ctx` holds +the bounds. The rules are asked only of a configuration within its +bounds, so a rule relies on them, as pydantic's after-validators and +zod's refinements do: until a value out of bounds is fixed, it is the +one problem reported of the configuration. `Annotated` may also carry a +note, a string or a `Doc`. Any other metadata -- a `MinLen`, a +`Predicate`, pydantic's `Field` -- is refused when the definition is +built, since a type the checker does not hold its values to would say +what is not so. A member holding another metadata field is annotated with the field alias of its kind -- a shard's `codecs: tuple[CodecField, ...]` -- and read in @@ -163,10 +193,11 @@ def acme_lz4_rules( takes `EmptyConfiguration`, and is written with its name alone. A data type also says what its fill value is: `fill_value`, the JSON -shape of one as an annotation the checker reads -- `Int8FillValue` -- and -`fill_value_rules`, a function yielding what the spec disallows in a fill -value of that shape: an integer out of range, a hex string of another -width. The rules are handed the configuration, the fields it holds as the +shape of one as an annotation the checker reads -- `Int8FillValue`, +whose type carries the range -- and `fill_value_rules`, a function +yielding what the spec disallows in a fill value of that shape that the +type cannot say: a hex string of another width, a number of byte values +the size does not take. The rules are handed the configuration, the fields it holds as the scope read them, and the typed fill value, so a struct judges each field's fill value by that field's own type. `fill_value_problems(data_type, value)` judges a fill value against a data diff --git a/packages/zarr-metadata/tests/model/test_extension_points.py b/packages/zarr-metadata/tests/model/test_extension_points.py index b67216a309..e0a9fb3836 100644 --- a/packages/zarr-metadata/tests/model/test_extension_points.py +++ b/packages/zarr-metadata/tests/model/test_extension_points.py @@ -13,6 +13,7 @@ from typing import Any, cast import pytest +from typing_extensions import TypedDict from zarr_metadata._json import arrays_to_tuples from zarr_metadata.model import ( @@ -24,7 +25,6 @@ validate_array_metadata_v3, validate_group_metadata_v3, ) -from zarr_metadata.v3.codec.gzip import GZIP_CODEC from zarr_metadata.v3.definition import CORE, CORE_AND_EXTENSIONS, CodecDefinition, Context BYTES = {"name": "bytes", "configuration": {"endian": "little"}} @@ -36,12 +36,17 @@ def _document(**fields: object) -> dict[str, Any]: return cast("dict[str, Any]", arrays_to_tuples(document)) +class LenientGzipConfiguration(TypedDict, closed=True): + """A gzip configuration whose `level` is any integer: the bound is the type's, so taking any is a type of its own.""" + + level: int + + LENIENT_GZIP = CodecDefinition( name="gzip", - configuration=GZIP_CODEC.configuration, + configuration=LenientGzipConfiguration, kind="bytes_bytes", size="dynamic", - rules=lambda configuration, nested: [], ) """A reader's own gzip, which takes any level: a scope can grow, and substitute.""" diff --git a/packages/zarr-metadata/tests/test_problem_data.py b/packages/zarr-metadata/tests/test_problem_data.py new file mode 100644 index 0000000000..f6c1bfa026 --- /dev/null +++ b/packages/zarr-metadata/tests/test_problem_data.py @@ -0,0 +1,351 @@ +"""What a problem carries besides its message: what was found, and what was expected. + +As pydantic's errors and zod's issues do, a problem carries its message's +data: `input`, what the value handed in holds at the problem's `loc`, and +`ctx`, what was expected there, where that is more than a type -- a +type's bounds, a closed set's values. Every reader fills `input`, so a +rule says only where a problem is. +""" + +from __future__ import annotations + +import copy +import math +import pickle +from typing import TYPE_CHECKING, Annotated, Literal + +import pytest +from annotated_types import Le +from typing_extensions import TypedDict + +from zarr_metadata._json import arrays_to_tuples, is_canonical_json, value_at, with_input +from zarr_metadata._sentinel import UNSET +from zarr_metadata.model import ( + MetadataValidationError, + ValidationProblem, + ZarrV2ConsolidatedMetadata, + ZarrV3ArrayMetadata, + parse_array_metadata_v3, + read_array_metadata_v3, + validate_array_metadata_v2, + validate_array_metadata_v3, + validate_group_metadata_v2, + validate_group_metadata_v3, + validate_json, + validate_metadata_field_v3, + validate_node_metadata_v3, +) +from zarr_metadata.typed_json import check +from zarr_metadata.v3.codec.gzip import GZIP_CODEC +from zarr_metadata.v3.definition import ( + CORE_AND_EXTENSIONS, + CodecDefinition, + DataTypeDefinition, + fill_value_problems, + resolve, +) + +if TYPE_CHECKING: + from collections.abc import Callable, Sequence + + from zarr_metadata._common import JSONValue + + +@pytest.mark.parametrize( + ("found", "ctx"), + [ + (UNSET, {}), + (12, {"ge": 0, "le": 9}), + (None, {"expected": ["C", "F"]}), + ], + ids=["nothing-found", "a-bound", "a-null-found"], +) +def test_a_problem_carries_what_was_found_and_what_was_expected( + found: JSONValue | UNSET, ctx: dict[str, JSONValue] +) -> None: + written = dict(ctx) + problem = ValidationProblem(("level",), "bad level", "invalid_value", input=found, ctx=written) + assert problem.input is found + # Held as a read-only view of a copy, arrays as tuples. + assert dict(problem.ctx) == arrays_to_tuples(ctx) + # What it says, and where, is the problem; what it carries is detail. + bare = ValidationProblem(("level",), "bad level", "invalid_value") + assert problem == bare + assert hash(problem) == hash(bare) + assert repr(problem) == repr(bare) + assert str(problem) == "level: bad level" + # A raised error is a finished report. + written["more"] = 1 + assert dict(problem.ctx) == arrays_to_tuples(ctx) + with pytest.raises(TypeError): + problem.ctx["ge"] = 1 # pyright: ignore[reportIndexIssue] + + +@pytest.mark.parametrize( + "ctx", + [["ge", 0], {1: 0}, {"ge": {0}}, {"ge": math.nan}], + ids=["an-array", "a-key-that-is-not-a-string", "a-set", "nan"], +) +def test_error_a_problem_refuses_a_ctx_that_is_not_an_object_of_json_values(ctx: object) -> None: + with pytest.raises(TypeError, match="ctx is an object of JSON values"): + ValidationProblem(("level",), "bad level", "invalid_value", ctx=ctx) # pyright: ignore[reportArgumentType] + + +def test_an_error_about_what_is_not_json_pickles() -> None: + # A problem holds no input that is not JSON, which might not pickle, + # so the error it is raised in crosses to another process as it is. + value = {"shape": [lambda: 4]} + problems = validate_json(value) + assert [(problem.loc, problem.input) for problem in problems] == [(("shape", 0), UNSET)] + again = pickle.loads(pickle.dumps(MetadataValidationError(problems))) + assert again.problems == problems + + +def test_a_problem_pickles_and_copies_with_what_it_carries() -> None: + problem = ValidationProblem(("shape",), "bad", "invalid_value", input=[12], ctx={"ge": 0}) + error = MetadataValidationError([problem]) + for again in ( + pickle.loads(pickle.dumps(problem)), + copy.copy(problem), + copy.deepcopy(problem), + pickle.loads(pickle.dumps(error)).problems[0], + ): + assert again == problem + assert again.input == [12] + assert dict(again.ctx) == {"ge": 0} + with pytest.raises(TypeError): + again.ctx["ge"] = 1 # pyright: ignore[reportIndexIssue] + + +def _deep(levels: int) -> list[object]: + """An array nested `levels` deep, deeper than the interpreter walks.""" + value: list[object] = [] + for _ in range(levels): + value = [value] + return value + + +@pytest.mark.parametrize( + ("value", "at", "loc", "held", "found"), + [ + ({"a": [1, {"b": 2}]}, (), ("a", 1, "b"), UNSET, 2), + ({"a": None}, (), ("a",), UNSET, None), + ({"a": 1}, (), ("b",), UNSET, UNSET), + ({"a": [1]}, (), ("a", 1), UNSET, UNSET), + ({"a": "text"}, (), ("a", 0), UNSET, UNSET), + ({"a": [1]}, (), ("a", "b"), UNSET, UNSET), + ([1, 2], ("codecs",), ("codecs", 1), UNSET, 2), + ([1, 2], ("codecs",), ("storage_transformers", 1), UNSET, UNSET), + ({"a": {1, 2}}, (), ("a",), UNSET, UNSET), + ({"a": _deep(100_000)}, (), ("a",), UNSET, UNSET), + ({"a": 1}, (), ("a",), "what a reader inside found", 1), + ({"a": {1, 2}}, (), ("a",), [1, 2], [1, 2]), + ], + ids=[ + "nested", + "a-null", + "a-missing-key", + "past-the-end", + "a-string-is-no-array", + "an-array-has-no-keys", + "under-where-the-value-sits", + "not-under-it", + "what-is-not-json", + "too-deep-to-walk", + "the-value-handed-in-wins", + "what-is-not-json-keeps-what-it-holds", + ], +) +def test_a_problem_holds_what_its_value_holds_at_its_loc( + value: object, + at: tuple[str | int, ...], + loc: tuple[str | int, ...], + held: JSONValue | UNSET, + found: JSONValue | UNSET, +) -> None: + (problem,) = with_input([ValidationProblem(loc, "bad", "invalid_value", input=held)], value, at) + assert problem.input == found + assert (problem.input is UNSET) is (found is UNSET) + + +class GzipLevelOnly(TypedDict, closed=True): + level: int + + +def _raised(read: Callable[[], object]) -> Sequence[ValidationProblem]: + with pytest.raises(MetadataValidationError) as caught: + read() + return caught.value.problems + + +BAD_ARRAY: dict[str, object] = { + "zarr_format": 3, + "node_type": "array", + "shape": [4, 4], + "data_type": "uint8", + "chunk_grid": {"name": "regular", "configuration": {"chunk_shape": [2]}}, + "chunk_key_encoding": {"name": "default"}, + "fill_value": 300, + "codecs": [ + {"name": "transpose", "configuration": {"order": [0, 1, 2]}}, + {"name": "bytes"}, + {"name": "gzip", "configuration": {"level": 12, "extra": [1]}}, + ], + "dimension_names": ["x"], + "attributes": {}, +} +"""An array document as `json.loads` gives one, arrays as lists, with a problem in each part a reader reads.""" + +BAD_ARRAY_LOCS = { + ("chunk_grid", "configuration", "chunk_shape"), + ("fill_value",), + ("codecs", 0, "configuration", "order"), + ("codecs", 2, "configuration", "level"), + ("codecs", 2, "configuration", "extra"), + ("dimension_names",), +} + +READERS: list[tuple[Callable[[object], Sequence[ValidationProblem]], object]] = [ + (lambda value: check(value, GzipLevelOnly)[1], {"level": 12, "extra": [1]}), + (lambda value: GZIP_CODEC.judge(value)[1], {"level": 12, "extra": [1]}), + ( + lambda value: resolve(value, CodecDefinition, CORE_AND_EXTENSIONS)[1], + {"name": "gzip", "configuration": {"level": 12}, "must_understand": "yes"}, + ), + ( + lambda value: fill_value_problems( + resolve("uint8", DataTypeDefinition, CORE_AND_EXTENSIONS)[0], value + ), + 300, + ), + (validate_json, {"a": [math.inf]}), + ( + validate_metadata_field_v3, + {"name": 1, "configuration": {"a": [math.nan]}, "extra": [1]}, + ), + (validate_array_metadata_v3, BAD_ARRAY), + (lambda value: read_array_metadata_v3(value).problems, BAD_ARRAY), + (lambda value: _raised(lambda: parse_array_metadata_v3(value)), BAD_ARRAY), + (lambda value: _raised(lambda: ZarrV3ArrayMetadata.from_json(value)), BAD_ARRAY), + (validate_node_metadata_v3, BAD_ARRAY), + ( + validate_group_metadata_v3, + { + "zarr_format": 3, + "node_type": "group", + "consolidated_metadata": { + "kind": "inline", + "must_understand": [False], + "metadata": {"a": BAD_ARRAY}, + }, + }, + ), + ( + validate_array_metadata_v2, + {"zarr_format": 2, "shape": [4], "chunks": [2, 2], "order": "Q", "filters": [[1]]}, + ), + (validate_group_metadata_v2, {"zarr_format": 3, "extra": [1]}), + ( + lambda value: _raised(lambda: ZarrV2ConsolidatedMetadata.from_json(value)), + {"zarr_consolidated_format": [1], "metadata": {"a/.zarray": [math.inf]}}, + ), + ( + lambda value: _raised(lambda: ZarrV3ArrayMetadata.from_key_value(value)), # pyright: ignore[reportArgumentType] + {"zarr.json": b"{"}, + ), +] + + +@pytest.mark.parametrize( + ("read", "value"), + READERS, + ids=[ + "check", + "judge", + "resolve", + "fill-value-problems", + "validate-json", + "validate-metadata-field", + "validate-array", + "read-array", + "parse-array", + "array-from-json", + "validate-node", + "validate-group-and-what-it-holds", + "validate-array-v2", + "validate-group-v2", + "v2-consolidated-from-json", + "from-key-value", + ], +) +def test_every_problem_a_reader_finds_holds_what_the_value_handed_in_holds_there( + read: Callable[[object], Sequence[ValidationProblem]], value: object +) -> None: + # The value itself, not a copy a reader made on the way: the problems + # of a field, a pipeline, a chunk grid over the shape, and a document + # a group holds alike. A missing key holds none, and nor does what is + # not JSON, such as a store's bytes. + problems = read(value) + assert len(problems) != 0 + for problem in problems: + there = value_at(value, problem.loc) + assert problem.input is (there if is_canonical_json(there, finite=False) else UNSET), ( + problem + ) + if value is BAD_ARRAY: + assert {problem.loc for problem in problems} >= BAD_ARRAY_LOCS + + +class WideBits(TypedDict, closed=True): + bits: Annotated[int, Le(16)] + + +def test_a_problem_with_what_a_name_carries_is_the_field_s() -> None: + # `r24` carries `{"bits": 24}`, which this reader's own raw bits refuse: + # the problem is found at the field, where the name is what is there, + # and what was expected of `bits` is not expected of the name. + scope = CORE_AND_EXTENSIONS.extended_with(DataTypeDefinition(name="r*", configuration=WideBits)) + _, problems = resolve("r24", DataTypeDefinition, scope, ("data_type",)) + assert [(p.loc, p.input, dict(p.ctx)) for p in problems] == [(("data_type",), "r24", {})] + assert problems[0].message == "expected an integer <= 16, got 24" + + +class Ordered(TypedDict, closed=True): + order: Literal["F", "C"] + + +@pytest.mark.parametrize( + ("problems", "loc", "expected"), + [ + (check({"order": "Q"}, Ordered)[1], ("order",), ("C", "F")), + ( + validate_array_metadata_v3({**BAD_ARRAY, "zarr_format": 2}), + ("zarr_format",), + (3,), + ), + (validate_node_metadata_v3({"node_type": "dataset"}), ("node_type",), ("array", "group")), + (validate_array_metadata_v2({"order": "Q"}), ("order",), ("C", "F")), + ( + validate_group_metadata_v3( + { + "zarr_format": 3, + "node_type": "group", + "consolidated_metadata": { + "kind": "inline", + "must_understand": True, + "metadata": {}, + }, + } + ), + ("consolidated_metadata", "must_understand"), + (False,), + ), + ], + ids=["a-literal", "zarr-format", "node-type", "v2-order", "consolidated-must-understand"], +) +def test_a_value_outside_a_closed_set_is_told_the_set( + problems: Sequence[ValidationProblem], loc: tuple[str | int, ...], expected: tuple[object, ...] +) -> None: + # In the order the message lists them, as JSON writes them. + (problem,) = [problem for problem in problems if problem.loc == loc] + assert dict(problem.ctx) == {"expected": expected} diff --git a/packages/zarr-metadata/tests/test_typed_json.py b/packages/zarr-metadata/tests/test_typed_json.py index cedb2aec90..8e0fd54959 100644 --- a/packages/zarr-metadata/tests/test_typed_json.py +++ b/packages/zarr-metadata/tests/test_typed_json.py @@ -7,6 +7,7 @@ from __future__ import annotations +import dataclasses import importlib import itertools import math @@ -15,10 +16,33 @@ import textwrap import types from collections.abc import Mapping -from typing import TYPE_CHECKING, Generic, Literal, NewType, NotRequired, TypeVar, cast +from typing import ( + TYPE_CHECKING, + Annotated, + Generic, + Literal, + NewType, + NotRequired, + TypeVar, + cast, +) +import pydantic import pytest -from typing_extensions import ReadOnly, TypeAliasType, TypedDict, is_typeddict +from annotated_types import ( + BaseMetadata, + Ge, + Gt, + Interval, + Le, + Len, + Lt, + MinLen, + MultipleOf, + Predicate, + Timezone, +) +from typing_extensions import Doc, ReadOnly, TypeAliasType, TypedDict, is_typeddict import zarr_metadata.v2 import zarr_metadata.v3 @@ -913,6 +937,176 @@ def test_a_shape_is_described_as_a_message_would_name_it() -> None: assert describe(ZarrV2DataTypeMetadata) == "a ZarrV2DataTypeMetadata" +# --- constraints ------------------------------------------------------------ + +Digit = Annotated[int, Interval(ge=0, le=9)] + + +class Bounded(TypedDict, extra_items=Annotated[int, Ge(0)]): + level: Digit + shape: ReadOnly[NotRequired[tuple[Annotated[int, Ge(1)], ...]]] + name: Annotated[NotRequired[Annotated[str, "the inner note"]], Doc("the outer note")] + + +class Noted(TypedDict, extra_items=Annotated[object, "any JSON"]): + level: int + + +@pytest.mark.parametrize( + ("annotation", "value", "found"), + [ + (Digit, 9, []), + (Digit, 10, [((), "expected an integer in [0, 9], got 10", {"ge": 0, "le": 9})]), + (Digit, -1, [((), "expected an integer in [0, 9], got -1", {"ge": 0, "le": 9})]), + (Annotated[int, Gt(0)], 0, [((), "expected an integer > 0, got 0", {"gt": 0})]), + (Annotated[float, Lt(1)], 1, [((), "expected a number < 1, got 1", {"lt": 1})]), + (Annotated[float, Le(0.5)], 0.5, []), + ( + Annotated[float, Interval(gt=0, le=0.5)], + 0.0, + [((), "expected a number in (0, 0.5], got 0.0", {"gt": 0, "le": 0.5})], + ), + ( + Annotated[float, Interval(ge=0.0, le=9.0)], + 10, + [((), "expected a number in [0, 9], got 10", {"ge": 0, "le": 9})], + ), + (Annotated[int, Ge(True)], 0, [((), "expected an integer >= 1, got 0", {"ge": 1})]), + ( + tuple[Digit, ...], + [0, 10, 9], + [((1,), "expected an integer in [0, 9], got 10", {"ge": 0, "le": 9})], + ), + (Annotated[Width, Le(9)], 10, [((), "expected an integer <= 9, got 10", {"le": 9})]), + (Digit | str, "ten", []), + (Digit | str, 10, [((), "expected an integer in [0, 9], got 10", {"ge": 0, "le": 9})]), + (Digit, "nine", [((), 'expected an integer, got "nine"', {})]), + (Annotated[int, "a note", Doc("another")], -1, []), + ], + ids=[ + "within", + "above", + "below", + "exclusive-lower", + "exclusive-upper", + "inclusive-upper", + "half-open", + "an-integral-bound-is-an-integer", + "a-bool-bound-is-the-integer-it-equals", + "each-element-at-its-index", + "on-an-alias", + "a-branch-it-does-not-bound", + "the-branch-it-bounds", + "a-value-of-another-type-is-that-problem-alone", + "notes", + ], +) +def test_a_value_is_held_to_the_bounds_its_type_carries( + annotation: object, value: object, found: list[tuple[Loc, str, dict[str, JSONValue]]] +) -> None: + # annotated-types' bounds, as pydantic reads them: one problem for a + # value out of any, saying what the type admits, its bounds in the + # problem's `ctx`; and asked only of a value of the type. A bound is + # the number it equals, whichever of the equal ones Python's + # `Annotated` cache hands back. + _, problems = _read(annotation, value) + assert [(problem.loc, problem.message, dict(problem.ctx)) for problem in problems] == found + assert all(problem.kind == "invalid_value" for problem in problems if problem.ctx) + + +def test_a_key_s_type_keeps_the_metadata_its_annotation_carries() -> None: + # Qualifiers peeled, wherever they sit among the `Annotated` layers, + # and the metadata kept, an inner layer's first, as `Annotated` + # flattens nested layers; a note on `object` leaves the type open. + keys = typeddict_keys(Bounded) + assert keys.members["level"] == (Digit, True) + assert keys.members["shape"] == (tuple[Annotated[int, Ge(1)], ...], False) + assert keys.members["name"] == ( + Annotated[str, "the inner note", Doc("the outer note")], + False, + ) + assert keys.extra_items == Annotated[int, Ge(0)] + typed, problems = check({"level": 3, "shape": [0], "other": -1}, Bounded) + assert typed is None + assert [(problem.loc, problem.input, dict(problem.ctx)) for problem in problems] == [ + (("other",), -1, {"ge": 0}), + (("shape", 0), 0, {"ge": 1}), + ] + assert typeddict_keys(Noted).open + assert check({"level": 1, "more": [None]}, Noted) == ({"level": 1, "more": (None,)}, ()) + + +def test_error_a_bound_on_what_is_not_a_number() -> None: + with pytest.raises(TypeError, match="ge: a bound is on a number, and a string is not one"): + parser(Annotated[str, Ge(0)], no_leaf) + + +def test_error_a_bound_on_the_values_an_open_typeddict_takes() -> None: + class Loose(TypedDict, extra_items=Annotated[object, Ge(0)]): + level: int + + with pytest.raises(TypeError, match="ge: a bound is on a number, and a value is not one"): + parser(Loose, no_leaf) + + +@dataclasses.dataclass(frozen=True) +class Later(BaseMetadata): + """A constraint a later annotated-types might add.""" + + +@pytest.mark.parametrize( + "constraint", + [ + MinLen(1), + Len(1, 3), + MultipleOf(2), + Predicate(str.isdigit), + Timezone(None), + Later(), + pydantic.Field(ge=0), + pydantic.StringConstraints(pattern="^[a-z]+$"), + ], + ids=[ + "min-len", + "len", + "multiple-of", + "predicate", + "timezone", + "one-a-later-release-adds", + "pydantic-field", + "pydantic-string-constraints", + ], +) +def test_error_a_constraint_the_checker_does_not_read(constraint: object) -> None: + # A type the checker did not hold its values to would say what is not + # so, as a type pydantic reads and the checker does not would disagree. + with pytest.raises(TypeError, match="is not a constraint the checker reads"): + parser(Annotated[str, constraint], no_leaf) + + +@pytest.mark.parametrize( + "annotation", + [ + Annotated[Digit, Ge(1)], + Annotated[int, Ge(0), Gt(0)], + Annotated[int, Interval(le=9), Lt(5)], + ], + ids=["narrowing-an-alias", "gt-and-ge", "interval-and-lt"], +) +def test_error_a_second_bound_from_one_side(annotation: object) -> None: + # Pydantic reads the last one said; say one. + with pytest.raises(TypeError, match="is a second bound from (below|above)"): + parser(annotation, no_leaf) + + +@pytest.mark.parametrize( + "bound", ["0", math.nan, math.inf, None], ids=["string", "nan", "infinity", "null"] +) +def test_error_a_bound_that_is_not_a_finite_number(bound: object) -> None: + with pytest.raises(TypeError, match="a bound is a finite number"): + parser(Annotated[int, Ge(bound)], no_leaf) # pyright: ignore[reportArgumentType] + + # --- check: the public door ------------------------------------------------ _V3_DOCUMENT = dict(ZarrV3ArrayMetadata.create_default(shape=(4,)).to_json()) diff --git a/packages/zarr-metadata/tests/v3/test_definitions.py b/packages/zarr-metadata/tests/v3/test_definitions.py index f17c670529..a8a9271b5b 100644 --- a/packages/zarr-metadata/tests/v3/test_definitions.py +++ b/packages/zarr-metadata/tests/v3/test_definitions.py @@ -9,9 +9,10 @@ Mapping, # noqa: TC003 - a TypedDict's annotations are evaluated at run time ) from dataclasses import dataclass, replace -from typing import TYPE_CHECKING, Any, Final, NotRequired, cast +from typing import TYPE_CHECKING, Annotated, Any, Final, NotRequired, cast import pytest +from annotated_types import Ge, Predicate from typing_extensions import TypedDict from zarr_metadata.model import validate_array_metadata_v3 @@ -32,6 +33,7 @@ Context, DataTypeDefinition, Definition, + EmptyConfiguration, JSONValue, Nested, Read, @@ -1081,6 +1083,31 @@ def test_error_a_configuration_whose_annotations_do_not_resolve() -> None: ) +class AcmeUncheckedConfiguration(TypedDict, closed=True): + digits: Annotated[str, Predicate(str.isdigit)] + + +def test_error_a_configuration_with_a_constraint_the_checker_does_not_read() -> None: + # A bound the checker did not hold a value to would say what is not so. + with pytest.raises( + TypeError, + match=r"AcmeUncheckedConfiguration.digits: Predicate\(str.isdigit\) is not a constraint", + ): + CodecDefinition( + name="acme.unchecked", + configuration=AcmeUncheckedConfiguration, + kind="bytes_bytes", + size="dynamic", + ) + + +def test_error_a_fill_value_with_a_constraint_its_type_cannot_take() -> None: + with pytest.raises(TypeError, match="fill_value: ge: a bound is on a number, and a string"): + DataTypeDefinition( + name="acme.bounded", configuration=EmptyConfiguration, fill_value=Annotated[str, Ge(0)] + ) + + def test_error_a_definition_name_is_a_string() -> None: with pytest.raises(TypeError, match="a definition's name is a string"): CodecDefinition(name=5, configuration=Empty, kind="bytes_bytes", size="dynamic") # pyright: ignore[reportArgumentType] diff --git a/packages/zarr-metadata/tests/v3/test_every_definition.py b/packages/zarr-metadata/tests/v3/test_every_definition.py index 279521aa98..85e03dc329 100644 --- a/packages/zarr-metadata/tests/v3/test_every_definition.py +++ b/packages/zarr-metadata/tests/v3/test_every_definition.py @@ -12,6 +12,7 @@ import pytest +from zarr_metadata._json import value_at from zarr_metadata.v3.codec.gzip import GZIP_CODEC from zarr_metadata.v3.data_type.raw import RAW_BYTES_DATA_TYPE, RawBytesConfiguration from zarr_metadata.v3.definition import ( @@ -435,6 +436,17 @@ def test_error_blosc_typesize_is_missing_while_shuffling() -> None: assert _one("codecs:blosc", configuration) == [(("configuration", "typesize"), "missing_key")] +def test_a_rule_is_asked_only_of_a_configuration_within_its_bounds() -> None: + # As pydantic's after-validators and zod's refinements are: a rule + # relies on the bounds its type declares, so a value out of them is + # the one problem reported until it is fixed. + shuffled = {key: value for key, value in BLOSC.items() if key != "typesize"} + assert _one("codecs:blosc", {**shuffled, "clevel": 12}) == [ + (("configuration", "clevel"), "invalid_value") + ] + assert _one("codecs:blosc", shuffled) == [(("configuration", "typesize"), "missing_key")] + + def test_error_blosc_typesize_is_not_positive() -> None: assert _one("codecs:blosc", {**BLOSC, "typesize": 0}) == [ (("configuration", "typesize"), "invalid_value") @@ -464,6 +476,100 @@ def test_error_sharding_inner_chunk_extent_is_zero() -> None: ] +@pytest.mark.parametrize( + ("kind", "field", "loc", "ctx"), + [ + ( + CodecDefinition, + {"name": "gzip", "configuration": {"level": 10}}, + ("configuration", "level"), + {"ge": 0, "le": 9}, + ), + ( + CodecDefinition, + {"name": "zstd", "configuration": {"level": -131073}}, + ("configuration", "level"), + {"ge": -131072, "le": 22}, + ), + ( + CodecDefinition, + {"name": "blosc", "configuration": {**BLOSC, "clevel": -1}}, + ("configuration", "clevel"), + {"ge": 0, "le": 9}, + ), + ( + CodecDefinition, + {"name": "blosc", "configuration": {**BLOSC, "blocksize": -1}}, + ("configuration", "blocksize"), + {"ge": 0}, + ), + ( + CodecDefinition, + { + "name": "sharding_indexed", + "configuration": { + "chunk_shape": [2, 0], + "codecs": ["bytes"], + "index_codecs": ["bytes"], + }, + }, + ("configuration", "chunk_shape", 1), + {"ge": 1}, + ), + ( + ChunkGridDefinition, + {"name": "regular", "configuration": {"chunk_shape": [2, -1]}}, + ("configuration", "chunk_shape", 1), + {"ge": 0}, + ), + ( + ChunkGridDefinition, + {"name": "rectilinear", "configuration": {"kind": "inline", "chunk_shapes": [0]}}, + ("configuration", "chunk_shapes", 0), + {"ge": 1}, + ), + ( + ChunkGridDefinition, + { + "name": "rectilinear", + "configuration": {"kind": "inline", "chunk_shapes": [[4, [2, 0]]]}, + }, + ("configuration", "chunk_shapes", 0, 1, 1), + {"ge": 1}, + ), + ( + DataTypeDefinition, + {"name": "numpy.timedelta64", "configuration": {"unit": "s", "scale_factor": 2**31}}, + ("configuration", "scale_factor"), + {"ge": 1, "le": 2**31 - 1}, + ), + ], + ids=[ + "gzip-level", + "zstd-level", + "blosc-clevel", + "blosc-blocksize", + "sharding-inner-chunk-extent", + "regular-chunk-extent", + "rectilinear-extent", + "rectilinear-run-count", + "numpy-time-scale-factor", + ], +) +def test_a_bound_is_its_member_s_type_and_its_problem_holds_it( + kind: type[Definition[Any]], + field: dict[str, Any], + loc: tuple[str | int, ...], + ctx: dict[str, int], +) -> None: + # Declared on the TypedDict, as pydantic reads a bound, and not in a + # rule: the problem holds the bound, and what was found. + _, problems = resolve(field, kind, CORE_AND_EXTENSIONS) + assert [(p.loc, p.kind, p.input, dict(p.ctx)) for p in problems] == [ + (loc, "invalid_value", value_at(field, loc), ctx) + ] + + STATIC_SIZE = ("bytes", "cast_value", "crc32c", "scale_offset", "transpose") DYNAMIC_SIZE = ("blosc", "gzip", "sharding_indexed", "zstd") diff --git a/packages/zarr-metadata/tests/v3/test_fill_values.py b/packages/zarr-metadata/tests/v3/test_fill_values.py index 8757d02e62..7963d08d24 100644 --- a/packages/zarr-metadata/tests/v3/test_fill_values.py +++ b/packages/zarr-metadata/tests/v3/test_fill_values.py @@ -14,6 +14,7 @@ import pytest from typing_extensions import TypedDict +from zarr_metadata._json import value_at from zarr_metadata.model import validate_array_metadata_v3, validate_group_metadata_v3 from zarr_metadata.model._array import ZarrV3ArrayMetadata from zarr_metadata.v3.data_type.struct import STRUCT_DATA_TYPE @@ -202,6 +203,31 @@ def test_error_a_byte_value_out_of_range( assert _problems(data_type, fill_value) == [(loc, "invalid_value")] +@pytest.mark.parametrize( + ("data_type", "fill_value", "loc", "ctx"), + [ + ("int8", 128, (), {"ge": -128, "le": 127}), + ("uint16", -1, (), {"ge": 0, "le": 2**16 - 1}), + ("int64", 2**63, (), {"ge": -(2**63), "le": 2**63 - 1}), + ("uint64", 2**64, (), {"ge": 0, "le": 2**64 - 1}), + ("r16", [1, 256], (1,), {"ge": 0, "le": 255}), + ("bytes", [-1], (0,), {"ge": 0, "le": 255}), + (DATETIME, 2**63, (), {"ge": -(2**63), "le": 2**63 - 1}), + ], + ids=["int8", "uint16", "int64", "uint64", "raw-bits-byte", "bytes-byte", "numpy-time"], +) +def test_a_fill_value_s_range_is_its_type_s( + data_type: JSONValue, fill_value: object, loc: tuple[int, ...], ctx: dict[str, int] +) -> None: + # `Int8FillValue` is `Annotated[int, Interval(ge=-128, le=127)]`: the + # problem holds the range, and what was found. + resolved, _ = resolve(data_type, DataTypeDefinition, CORE_AND_EXTENSIONS) + problems = fill_value_problems(resolved, fill_value) + assert [(p.loc, p.input, dict(p.ctx)) for p in problems] == [ + (loc, value_at(fill_value, loc), ctx) + ] + + @pytest.mark.parametrize("fill_value", ["!!", "AQI"]) def test_error_a_bytes_fill_value_that_is_not_base64(fill_value: str) -> None: assert _problems("bytes", fill_value) == [((), "invalid_value")] diff --git a/packages/zarr-metadata/uv.lock b/packages/zarr-metadata/uv.lock index 07ad9a0ebe..f1d80de5cc 100644 --- a/packages/zarr-metadata/uv.lock +++ b/packages/zarr-metadata/uv.lock @@ -1135,6 +1135,7 @@ wheels = [ name = "zarr-metadata" source = { editable = "." } dependencies = [ + { name = "annotated-types" }, { name = "typing-extensions" }, ] @@ -1155,7 +1156,10 @@ test = [ ] [package.metadata] -requires-dist = [{ name = "typing-extensions", specifier = ">=4.16" }] +requires-dist = [ + { name = "annotated-types", specifier = ">=0.6" }, + { name = "typing-extensions", specifier = ">=4.16" }, +] [package.metadata.requires-dev] docs = [ From fd999e0d51011615fe5519b31cb2ca2799f2e2e3 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Tue, 29 Sep 2026 11:56:13 +0200 Subject: [PATCH 07/94] feat(zarr-metadata): with_problems groups a read's problems by field, and canonical_of spells a field read already `with_problems(fields, problems)` gives each field of one read -- a reading's `fields()` and `problems`, or `fields_of` a field and the problems `resolve` gave with it -- with the problems located in it, in the fields it holds too: those it was read with, and those the document found with it where it stands, its place in the pipeline, the chunk it is handed, the array's shape. A field with none is valid there. A function of the problems, as zod's `flattenError` is of the issues, grouping them at every depth; a problem in no field is in none's. `canonical_of(resolved, problems)` spells a field a scope has read already, given its problems, in its simplest equivalent spelling, without reading it again: None for a field with a problem, as `canonicalize` gives, so a key its TypedDict does not declare is never erased. `canonicalize` is `canonical_of(*resolve(...))`. Assisted-by: ClaudeCode:claude-opus-5-5 --- packages/zarr-metadata/README.md | 16 ++- packages/zarr-metadata/changes/377.feature.md | 11 ++ packages/zarr-metadata/docs/index.md | 16 ++- .../src/zarr_metadata/model/_validation.py | 3 +- .../src/zarr_metadata/v3/_definition.py | 69 ++++++++++-- .../src/zarr_metadata/v3/definition.py | 17 ++- .../tests/model/test_read_array_metadata.py | 105 +++++++++++++++++- .../tests/v3/test_every_definition.py | 26 ++++- 8 files changed, 239 insertions(+), 24 deletions(-) create mode 100644 packages/zarr-metadata/changes/377.feature.md diff --git a/packages/zarr-metadata/README.md b/packages/zarr-metadata/README.md index 5179986f96..217e3f6e01 100644 --- a/packages/zarr-metadata/README.md +++ b/packages/zarr-metadata/README.md @@ -97,13 +97,17 @@ definition that claims its name, `Unclaimed` when none does, or `Refused` -- with where it sits and the kind it was read as, each codec with the chunk it is handed, every problem, and the model when there is none; `from_json` is that model, or the problems raised. A consumer's -own policy is a walk over the fields, with nothing read twice. Which -fields go beyond the core spec, say -- a field that names nothing is a -problem already: +own policy is a walk over the fields, with nothing read twice: +`with_problems` gives each with its problems, those located in it and in +the fields it holds, as zod's `flattenError` groups issues, and +`canonical_of` spells a field with none in the fewest words, without +reading it again. Which fields go beyond the core spec, say -- a field +that names nothing is a problem already -- and how each is spelled most +simply: ```python from zarr_metadata.model import read_array_metadata_v3 -from zarr_metadata.v3.definition import CORE +from zarr_metadata.v3.definition import CORE, canonical_of, with_problems reading = read_array_metadata_v3(raw) beyond_core = [ @@ -111,6 +115,10 @@ beyond_core = [ for loc, field in reading.fields() if field.name is not None and CORE.claimant(field.read_as, field.name) is None ] +simplest = { + loc: canonical_of(field, problems) + for loc, field, problems in with_problems(reading.fields(), reading.problems) +} metadata = reading.metadata # None when reading.problems is not empty ``` diff --git a/packages/zarr-metadata/changes/377.feature.md b/packages/zarr-metadata/changes/377.feature.md new file mode 100644 index 0000000000..c7a01fee0f --- /dev/null +++ b/packages/zarr-metadata/changes/377.feature.md @@ -0,0 +1,11 @@ +`with_problems(fields, problems)` gives each field of one read -- a +reading's `fields()` and `problems`, or `fields_of` a field and the +problems `resolve` gave with it -- with the problems located in it, in +the fields it holds too: those it was read with, and those the document +found with it where it stands, its place in the pipeline, the chunk it +is handed, the array's shape. A field with none is valid there; a +problem in no field is in none's. It groups as zod's `flattenError` +groups issues, at every depth. `canonical_of(resolved, problems)` spells +a field a scope has read already, given its problems, in its simplest +equivalent spelling without reading it again, and gives what +`canonicalize` gives: None for a field with a problem. diff --git a/packages/zarr-metadata/docs/index.md b/packages/zarr-metadata/docs/index.md index c6fe748954..fa3c686217 100644 --- a/packages/zarr-metadata/docs/index.md +++ b/packages/zarr-metadata/docs/index.md @@ -112,13 +112,17 @@ definition that claims its name, `Unclaimed` when none does, or `Refused` -- with where it sits and the kind it was read as, each codec with the chunk it is handed, every problem, and the model when there is none; `from_json` is that model, or the problems raised. A consumer's -own policy is a walk over the fields, with nothing read twice. Which -fields go beyond the core spec, say -- a field that names nothing is a -problem already: +own policy is a walk over the fields, with nothing read twice: +`with_problems` gives each with its problems, those located in it and in +the fields it holds, as zod's `flattenError` groups issues, and +`canonical_of` spells a field with none in the fewest words, without +reading it again. Which fields go beyond the core spec, say -- a field +that names nothing is a problem already -- and how each is spelled most +simply: ```python from zarr_metadata.model import read_array_metadata_v3 -from zarr_metadata.v3.definition import CORE +from zarr_metadata.v3.definition import CORE, canonical_of, with_problems reading = read_array_metadata_v3(raw) beyond_core = [ @@ -126,6 +130,10 @@ beyond_core = [ for loc, field in reading.fields() if field.name is not None and CORE.claimant(field.read_as, field.name) is None ] +simplest = { + loc: canonical_of(field, problems) + for loc, field, problems in with_problems(reading.fields(), reading.problems) +} metadata = reading.metadata # None when reading.problems is not empty ``` diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_validation.py b/packages/zarr-metadata/src/zarr_metadata/model/_validation.py index 65e00161c4..2f8823cd7c 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_validation.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/_validation.py @@ -419,7 +419,8 @@ def fields(self) -> Iterator[tuple[Loc, Resolved[Any]]]: The extension points, then each codec and storage transformer at its index, each followed by the fields it holds, as `fields_of` gives - them: a shard's codecs, a struct's field types. + them: a shard's codecs, a struct's field types. `with_problems` + gives each with its problems. """ for key, field in ( ("data_type", self.data_type), diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py b/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py index fcf4dc9a6d..89410bb3ee 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py @@ -1037,6 +1037,35 @@ def fields_of(resolved: Resolved[Any], loc: Loc = ()) -> Iterator[tuple[Loc, Res yield from fields_of(inner, (*loc, "configuration", *place)) +def with_problems( + fields: Iterable[tuple[Loc, Resolved[Any]]], problems: Sequence[ValidationProblem] +) -> Iterator[tuple[Loc, Resolved[Any], Problems]]: + """Each of `fields`, with where it sits, and the problems among `problems` located in it, in the fields it holds too. + + `fields` and `problems` are one read's: a reading's `fields()` and + `problems`, or `fields_of` a field and the problems `resolve` gave + with it. A field's problems are those it was read with, and those the + document found with it where it stands -- its place in the pipeline, + the chunk it is handed, the array's shape -- so a field with none is + valid there, and `canonical_of` spells it. A function of the problems, + as zod's `flattenError` is of the issues, grouping them at every + depth: a problem with a shard's inner codec is the inner codec's, and + the shard's. Each field comes before the fields it holds, as + `fields_of` gives them, so the last field whose problems hold a + problem is the innermost field holding it. A problem in no field -- + with the fill value, with the shape -- is in none's. + """ + located = list(fields) + held: dict[Loc, list[ValidationProblem]] = {loc: [] for loc, _ in located} + for found in problems: + for depth in range(len(found.loc) + 1): + holder = held.get(found.loc[:depth]) + if holder is not None: + holder.append(found) + for loc, field in located: + yield loc, field, tuple(held[loc]) + + def configuration_of(resolved: Resolved[Any], definition: Definition[C]) -> C | None: """The configuration `resolved` holds, typed as `definition` declares it, if `definition` read it. @@ -1338,29 +1367,49 @@ def canonicalize( (https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v3/core/index.rst#L585-L592); and there is no `must_understand`, since `true` is what absence means. A name nothing in scope claims keeps the configuration it was written - with, since what it simplifies to is its own definition's call. + with, since what it simplifies to is its own definition's call. A + field a scope has read already is spelled so by `canonical_of`, + given its problems. """ resolved, problems = resolve(data, kind, context, loc) + return canonical_of(resolved, problems), problems + + +def canonical_of( + resolved: Resolved[Any], problems: Sequence[ValidationProblem] +) -> JSONValue | None: + """`resolved`, a field a scope read, in its simplest equivalent spelling, as `canonicalize` spells one; None when it has a problem. + + `problems` are the field's, as `resolve` gives them, or + `with_problems` gives each field of a reading. Only a field with none + has a simplest spelling: a simpler spelling of one with a problem + would erase what its author wrote -- a key its TypedDict does not + declare -- or spell what does not hold. What is spelled is what the + scope read, without reading the field again. + """ if len(problems) != 0: - return None, problems - return _simplest(resolved), () + return None + return _simplest(resolved) -def _simplest(field: Resolved[Any]) -> JSONValue: - """A field without problems in its simplest spelling: one nothing in scope claims as a document writes it.""" +def _simplest(field: Resolved[Any]) -> JSONValue | None: + """A field in its simplest spelling; None when it, or a field it holds, was refused, which has none.""" if isinstance(field, Read): return _canonical_field(field) if isinstance(field, Unclaimed): return field.to_json() - return field.json + return None -def _canonical_field(resolved: Read[Any]) -> JSONValue: - """A field that read, in its simplest equivalent spelling: the fields it holds first, then its own members.""" +def _canonical_field(resolved: Read[Any]) -> JSONValue | None: + """A field that read, in its simplest equivalent spelling: the fields it holds first, then its own members; None when one it holds was refused.""" definition, name = resolved.definition, resolved.name configuration: JSONValue = dict(resolved.configuration) for loc, inner in resolved.nested.items(): - configuration = _replaced(configuration, loc, _simplest(inner)) + simplest = _simplest(inner) + if simplest is None: + return None + configuration = _replaced(configuration, loc, simplest) simplified = cast("Mapping[str, JSONValue]", definition.canonical(configuration)) _, refused = definition.judge(simplified) if len(refused) != 0: @@ -1416,6 +1465,7 @@ def _replaced(value: JSONValue, path: Loc, new: JSONValue) -> JSONValue: "Unclaimed", "as_kind", "asked", + "canonical_of", "canonicalize", "chunk_grid_lengths", "configuration_of", @@ -1435,4 +1485,5 @@ def _replaced(value: JSONValue, path: Loc, new: JSONValue) -> JSONValue: "unknown_lengths", "unknown_storage", "variable_length", + "with_problems", ] diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/definition.py b/packages/zarr-metadata/src/zarr_metadata/v3/definition.py index 5d38478607..ccd233b56b 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/definition.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/definition.py @@ -258,10 +258,15 @@ def acme_lz4_rules( words every reader takes: a data type with nothing to configure is its bare name, any other field an object, `{"name": ...}`, as a Zarr v3.0 reader takes no short-hand name in `codecs`; raw bits write their size -back into the name, in decimal, so `r008` is `r8`. A field with any problem, an unknown key included, has -none: a simpler spelling of it would erase what its author wrote. What -`canonical` gives is judged again: one that does not hold is a -`ValueError`, a fault in the definition. +back into the name, in decimal, so `r008` is `r8`. A field with any +problem, an unknown key included, has none: a simpler spelling of it +would erase what its author wrote. What `canonical` gives is judged +again: one that does not hold is a `ValueError`, a fault in the +definition. `canonical_of(resolved, problems)` spells a field a scope +has read already, given its problems -- as `resolve` gives them, or +`with_problems` gives each field of a reading -- without reading it +again, and gives what `canonicalize` gives: None for a field with a +problem. A definition checks itself when it is built, and each of these is a `TypeError` saying what is wrong: a `configuration` that is not a @@ -307,6 +312,7 @@ class creation. StorageTransformerDefinition, StorageTransformerField, Unclaimed, + canonical_of, canonicalize, chunk_grid_lengths, configuration_of, @@ -314,6 +320,7 @@ class creation. fill_value_problems, resolve, storage_of, + with_problems, ) from zarr_metadata.v3._pipeline import Stage, read_pipeline from zarr_metadata.v3._registry import CORE, CORE_AND_EXTENSIONS, Context @@ -352,6 +359,7 @@ class creation. "Unclaimed", "ValidationProblem", "ZarrV3MetadataFieldJSON", + "canonical_of", "canonicalize", "check", "chunk_grid_lengths", @@ -362,4 +370,5 @@ class creation. "resolve", "shown", "storage_of", + "with_problems", ] diff --git a/packages/zarr-metadata/tests/model/test_read_array_metadata.py b/packages/zarr-metadata/tests/model/test_read_array_metadata.py index d25455ee6a..c3433572b8 100644 --- a/packages/zarr-metadata/tests/model/test_read_array_metadata.py +++ b/packages/zarr-metadata/tests/model/test_read_array_metadata.py @@ -18,6 +18,7 @@ ZarrV3ArrayMetadata, ZarrV3ArrayMetadataReading, read_array_metadata_v3, + read_group_metadata_v3, validate_array_metadata_v3, ) from zarr_metadata.v3.definition import ( @@ -33,10 +34,15 @@ Refused, StorageTransformerDefinition, Unclaimed, + fields_of, + resolve, + with_problems, ) if TYPE_CHECKING: - from zarr_metadata.v3.definition import Lengths, Loc + from collections.abc import Callable, Iterator + + from zarr_metadata.v3.definition import Lengths, Loc, Resolved LITTLE = {"name": "bytes", "configuration": {"endian": "little"}} ZSTD = {"name": "zstd", "configuration": {"level": 1}} @@ -228,6 +234,103 @@ def test_a_document_reads_as_each_field_where_it_sits_and_its_codecs_as_a_pipeli assert [None if s.incoming is None else s.incoming.lengths for s in reading.pipeline] == handed +_INNER_GZIP = {"name": "gzip", "configuration": {"level": 12}} +_WITH_PROBLEMS = _document( + (4, 4), + chunk_grid={"name": "regular", "configuration": {"chunk_shape": [4]}}, + fill_value="high", + codecs=[ + _shard([2, 2], [LITTLE, _INNER_GZIP]), + ZSTD, + {"name": "transpose", "configuration": {"order": [1, 0]}}, + ], +) +"""An array document with a problem in each place a field can have one, and one in no field.""" + + +def _array_fields( + document: object, +) -> Iterator[tuple[Loc, Resolved[Any], tuple[ValidationProblem, ...]]]: + reading = read_array_metadata_v3(document) + return with_problems(reading.fields(), reading.problems) + + +def _group_fields( + document: object, +) -> Iterator[tuple[Loc, Resolved[Any], tuple[ValidationProblem, ...]]]: + group = { + "zarr_format": 3, + "node_type": "group", + "consolidated_metadata": { + "kind": "inline", + "must_understand": False, + "metadata": {"a": document}, + }, + } + reading = read_group_metadata_v3(group) + return with_problems(reading.fields(), reading.problems) + + +def _field_fields( + document: object, +) -> Iterator[tuple[Loc, Resolved[Any], tuple[ValidationProblem, ...]]]: + resolved, problems = resolve( + cast("dict[str, Any]", document)["codecs"][0], + CodecDefinition, + CORE_AND_EXTENSIONS, + ("codecs", 0), + ) + return with_problems(fields_of(resolved, ("codecs", 0)), problems) + + +_INNER_LEVEL = ("codecs", 0, "configuration", "codecs", 1, "configuration", "level") +_EACH_FIELD_S = { + ("data_type",): [], + ("chunk_grid",): [("chunk_grid", "configuration", "chunk_shape")], + ("chunk_key_encoding",): [], + ("codecs", 0): [_INNER_LEVEL], + ("codecs", 0, "configuration", "codecs", 0): [], + ("codecs", 0, "configuration", "codecs", 1): [_INNER_LEVEL], + ("codecs", 0, "configuration", "index_codecs", 0): [], + ("codecs", 1): [], + ("codecs", 2): [("codecs", 2)], +} +"""Each field of `_WITH_PROBLEMS`, and where each of its problems is.""" +_IN_A_GROUP = ("consolidated_metadata", "metadata", "a") + + +@pytest.mark.parametrize( + ("fields", "expected"), + [ + (_array_fields, _EACH_FIELD_S), + ( + _group_fields, + { + (*_IN_A_GROUP, *loc): [(*_IN_A_GROUP, *problem) for problem in problems] + for loc, problems in _EACH_FIELD_S.items() + }, + ), + ( + _field_fields, + {loc: problems for loc, problems in _EACH_FIELD_S.items() if loc[:2] == ("codecs", 0)}, + ), + ], + ids=["an-array", "a-document-a-group-holds", "one-field"], +) +def test_each_field_comes_with_the_problems_located_in_it( + fields: Callable[[object], Iterator[tuple[Loc, Resolved[Any], tuple[ValidationProblem, ...]]]], + expected: dict[Loc, list[Loc]], +) -> None: + # Those it was read with and those the document found with it where + # it stands -- a transpose out of the pipeline's order, at its own + # place, and the chunk grid over the shape -- the fields it holds + # too: a shard with a bad inner codec has that problem as well. A + # problem in no field -- the fill value -- is in none's. + assert { + loc: [p.loc for p in problems] for loc, _, problems in fields(_WITH_PROBLEMS) + } == expected + + def test_a_document_with_no_problem_reads_as_its_model_holding_the_fields_read() -> None: # The fields the read made, not a second reading of them. document = _document(codecs=[{"name": "transpose", "configuration": {"order": [0]}}, LITTLE]) diff --git a/packages/zarr-metadata/tests/v3/test_every_definition.py b/packages/zarr-metadata/tests/v3/test_every_definition.py index 85e03dc329..8617799364 100644 --- a/packages/zarr-metadata/tests/v3/test_every_definition.py +++ b/packages/zarr-metadata/tests/v3/test_every_definition.py @@ -28,6 +28,7 @@ Refused, Unclaimed, ValidationProblem, + canonical_of, canonicalize, configuration_of, fill_value_problems, @@ -181,6 +182,8 @@ def test_every_example_reads_and_its_simplest_spelling_is_stable(key: str, field assert problems == () assert simplest is not None assert canonicalize(simplest, kind, CORE_AND_EXTENSIONS) == (simplest, ()) + # A field read already is spelled as its JSON is, with nothing read again. + assert canonical_of(*resolve(field, kind, CORE_AND_EXTENSIONS)) == simplest @pytest.mark.parametrize( @@ -271,6 +274,7 @@ def test_the_simplest_spelling(field: dict[str, Any], simplest: object) -> None: kinds = {"rectilinear": ChunkGridDefinition, "int8": DataTypeDefinition} kind = kinds.get(field["name"], CodecDefinition) assert canonicalize(field, kind, CORE_AND_EXTENSIONS) == (simplest, ()) + assert canonical_of(*resolve(field, kind, CORE_AND_EXTENSIONS)) == simplest @pytest.mark.parametrize( @@ -346,6 +350,11 @@ def test_an_unclaimed_field_keeps_its_configuration_in_its_kind_s_envelope( assert canonicalize(field, kind, CORE) == (simplest, ()) +def _shard(codecs: list[object], index_codecs: list[object]) -> dict[str, object]: + configuration = {"chunk_shape": [2], "codecs": codecs, "index_codecs": index_codecs} + return {"name": "sharding_indexed", "configuration": configuration} + + @pytest.mark.parametrize( ("field", "kind", "found"), [ @@ -378,6 +387,16 @@ def test_an_unclaimed_field_keeps_its_configuration_in_its_kind_s_envelope( [(("extra",), "invalid_value")], ), (None, CodecDefinition, [((), "invalid_type")]), + ( + _shard(["bytes", {"name": "gzip", "configuration": {"level": 12}}], ["bytes"]), + CodecDefinition, + [(("configuration", "codecs", 1, "configuration", "level"), "invalid_value")], + ), + ( + _shard(["bytes"], ["bytes", {"name": "gzip", "configuration": {"level": 1}}]), + CodecDefinition, + [(("configuration", "index_codecs", 1), "invalid_value")], + ), ], ids=[ "refused-value", @@ -386,16 +405,21 @@ def test_an_unclaimed_field_keeps_its_configuration_in_its_kind_s_envelope( "must-understand-false", "stray", "null", + "holding-a-refused-field", + "a-codec-of-dynamic-size-where-one-of-static-size-goes", ], ) def test_error_a_field_with_a_problem_has_no_simplest_spelling( field: dict[str, Any] | None, kind: type[Definition[Any]], found: list[object] ) -> None: # Whatever the author wrote stays theirs: a simpler spelling would - # drop the unknown key, the stray member or the `must_understand`. + # drop the unknown key, the stray member or the `must_understand`, or + # spell what does not hold. A field read already, given its problems, + # has none either. simplest, problems = canonicalize(field, kind, CORE_AND_EXTENSIONS) assert simplest is None assert [(problem.loc, problem.kind) for problem in problems] == found + assert canonical_of(*resolve(field, kind, CORE_AND_EXTENSIONS)) is None def test_error_a_canonical_that_does_not_hold_is_the_definitions_fault() -> None: From ee024357e0e5dfd4b6c25cdeebe3d92f636dafa7 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Tue, 29 Sep 2026 13:14:38 +0200 Subject: [PATCH 08/94] fix(zarr-metadata)!: a regular grid's chunk lengths are at least 1, as its specification says "Chunk sizes must be greater than zero" (https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v3/chunk-grids/regular-grid/index.rst#L40), along a dimension of length 0 too. `RegularChunkGridConfiguration` bounded its lengths with `Ge(0)`, and its shape rules took a 0 on an empty dimension, reading the core specification's "The chunk shape elements are non-zero when the corresponding dimensions of the arrays have non-zero length" as allowing it. That sentence says less than the grid's own. The bound is now `Ge(1)`, so a 0 is an `invalid_value` at its place wherever it is, and the shape rules check only that the grid has the array's dimensionality. `create_default` already wrote 1 there. BREAKING CHANGE: a regular grid with a chunk length of 0 on a dimension of length 0, as zarr-python 3.0 and 3.1 wrote one, is now refused. Assisted-by: ClaudeCode:claude-opus-5-5 --- packages/zarr-metadata/README.md | 14 +++++---- packages/zarr-metadata/changes/371.doc.md | 6 ++-- packages/zarr-metadata/changes/378.bugfix.md | 9 ++++++ .../zarr-metadata/changes/4443.feature.3.md | 9 +++--- packages/zarr-metadata/docs/index.md | 14 +++++---- .../src/zarr_metadata/model/_array.py | 4 +-- .../zarr_metadata/v3/chunk_grid/regular.py | 30 +++++++------------ .../tests/v3/test_definitions.py | 10 ------- .../tests/v3/test_every_definition.py | 11 +++++-- .../tests/v3/test_grid_shapes.py | 12 ++------ 10 files changed, 56 insertions(+), 63 deletions(-) create mode 100644 packages/zarr-metadata/changes/378.bugfix.md diff --git a/packages/zarr-metadata/README.md b/packages/zarr-metadata/README.md index 217e3f6e01..11eb9bac52 100644 --- a/packages/zarr-metadata/README.md +++ b/packages/zarr-metadata/README.md @@ -71,7 +71,7 @@ against the chunk it is handed, a shard's inner and index codecs too. The validators do no arithmetic on values: whether a fill value survives a `cast_value` round trip is not judged. -Three choices the specs' words leave open, or settle two ways: +Two choices the specs' words leave open, or settle two ways: - **`attributes` may hold `NaN`, `Infinity` and `-Infinity`.** The spec interprets no attribute, and zarr-python and xarray write those numbers @@ -85,11 +85,13 @@ Three choices the specs' words leave open, or settle two ways: reads wrong bytes as surely as one that skips a data type reads wrong values. It keeps its meaning on an unknown top-level member, which a reader can skip. -- **A chunk length of 0 is allowed along a dimension of length 0.** The - core spec asks for non-zero chunk lengths only "when the corresponding - dimensions of the arrays have non-zero length"; the regular grid spec - says chunk sizes are greater than zero. The package follows the core - spec, which zarr-python 3.0 and 3.1 wrote for an empty dimension. + +A regular grid's chunk lengths are at least 1, along a dimension of +length 0 too: "Chunk sizes must be greater than zero", the regular grid +spec says. The core spec's "non-zero when the corresponding dimensions +of the arrays have non-zero length" says less, and allows nothing more, +so a document with a 0 there, as zarr-python 3.0 and 3.1 wrote for an +empty dimension, is refused. `read_array_metadata_v3` reads a document once and returns everything the read found: each field as the scope read it -- `Read` by the diff --git a/packages/zarr-metadata/changes/371.doc.md b/packages/zarr-metadata/changes/371.doc.md index d0017875f5..94492433c2 100644 --- a/packages/zarr-metadata/changes/371.doc.md +++ b/packages/zarr-metadata/changes/371.doc.md @@ -2,6 +2,6 @@ The validation boundary says what the package decides where the specs leave it open or settle it two ways: `attributes` may hold `NaN`, `Infinity` and `-Infinity`, which `to_key_value` writes as bare tokens; `must_understand: false` is refused at every extension point, codecs -too; a chunk length of 0 is allowed along a dimension of length 0. The -models' `from_json`, `to_json`, `from_key_value` and `to_key_value` -have docstrings, and no public docstring names a private function. +too. The models' `from_json`, `to_json`, `from_key_value` and +`to_key_value` have docstrings, and no public docstring names a private +function. diff --git a/packages/zarr-metadata/changes/378.bugfix.md b/packages/zarr-metadata/changes/378.bugfix.md new file mode 100644 index 0000000000..bde372fb24 --- /dev/null +++ b/packages/zarr-metadata/changes/378.bugfix.md @@ -0,0 +1,9 @@ +The `regular` chunk grid refuses a chunk length of 0, along a dimension +of length 0 too, as its specification says: "Chunk sizes must be greater +than zero". `RegularChunkGridConfiguration.chunk_shape` is +`tuple[Annotated[int, Ge(1)], ...]`, so a 0 is an `invalid_value` at its +place, with the bound in its `ctx`. The package had read the core +specification's "The chunk shape elements are non-zero when the +corresponding dimensions of the arrays have non-zero length" as allowing +0 on an empty dimension, as zarr-python 3.0 and 3.1 wrote it; that +sentence says less than the grid's own, not something else. diff --git a/packages/zarr-metadata/changes/4443.feature.3.md b/packages/zarr-metadata/changes/4443.feature.3.md index b899c84f09..f7c8bcd78b 100644 --- a/packages/zarr-metadata/changes/4443.feature.3.md +++ b/packages/zarr-metadata/changes/4443.feature.3.md @@ -1,7 +1,6 @@ **Breaking:** a v3 array's `chunk_grid` is judged against its `shape`, by the grid's definition. A regular grid whose `chunk_shape` does not have -one length per dimension of the shape, or has a length of 0 for a -dimension that is not empty, and a rectilinear grid whose `chunk_shapes` -does not have one entry per dimension, or whose chunk lengths fall short -of their dimension, each have a problem in `chunk_grid.configuration`, -where the package accepted them before. +one length per dimension of the shape, and a rectilinear grid whose +`chunk_shapes` does not have one entry per dimension, or whose chunk +lengths fall short of their dimension, each have a problem in +`chunk_grid.configuration`, where the package accepted them before. diff --git a/packages/zarr-metadata/docs/index.md b/packages/zarr-metadata/docs/index.md index fa3c686217..8feaf6e95a 100644 --- a/packages/zarr-metadata/docs/index.md +++ b/packages/zarr-metadata/docs/index.md @@ -86,7 +86,7 @@ against the chunk it is handed, a shard's inner and index codecs too. The validators do no arithmetic on values: whether a fill value survives a `cast_value` round trip is not judged. -Three choices the specs' words leave open, or settle two ways: +Two choices the specs' words leave open, or settle two ways: - **`attributes` may hold `NaN`, `Infinity` and `-Infinity`.** The spec interprets no attribute, and zarr-python and xarray write those numbers @@ -100,11 +100,13 @@ Three choices the specs' words leave open, or settle two ways: reads wrong bytes as surely as one that skips a data type reads wrong values. It keeps its meaning on an unknown top-level member, which a reader can skip. -- **A chunk length of 0 is allowed along a dimension of length 0.** The - core spec asks for non-zero chunk lengths only "when the corresponding - dimensions of the arrays have non-zero length"; the regular grid spec - says chunk sizes are greater than zero. The package follows the core - spec, which zarr-python 3.0 and 3.1 wrote for an empty dimension. + +A regular grid's chunk lengths are at least 1, along a dimension of +length 0 too: "Chunk sizes must be greater than zero", the regular grid +spec says. The core spec's "non-zero when the corresponding dimensions +of the arrays have non-zero length" says less, and allows nothing more, +so a document with a 0 there, as zarr-python 3.0 and 3.1 wrote for an +empty dimension, is refused. `read_array_metadata_v3` reads a document once and returns everything the read found: each field as the scope read it -- `Read` by the diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_array.py b/packages/zarr-metadata/src/zarr_metadata/model/_array.py index 13d31ded89..4cbc93c597 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_array.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/_array.py @@ -152,8 +152,8 @@ def create_default( type of any fixed size. Overriding `shape` without `chunk_grid` derives a consistent default grid: one regular chunk covering the array (`chunk_shape` equal to `shape`, with a length of 1 for a - dimension of length 0, which every reader takes: the core spec - allows 0 there and the regular grid spec does not, + dimension of length 0, since "Chunk sizes must be greater than + zero", https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v3/chunk-grids/regular-grid/index.rst#L40). """ # The grid derives from a shape the read takes; one it refuses is diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/chunk_grid/regular.py b/packages/zarr-metadata/src/zarr_metadata/v3/chunk_grid/regular.py index 179fdcc5cc..299603475d 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/chunk_grid/regular.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/chunk_grid/regular.py @@ -23,14 +23,16 @@ class RegularChunkGridConfiguration(TypedDict, closed=True): """Configuration for the regular chunk grid. - No chunk extent is negative. "The chunk shape elements are non-zero - when the corresponding dimensions of the arrays have non-zero length": - an extent of 0 is right for a dimension of length 0, which zarr-python - 3.0 and 3.1 wrote, and which a grid alone cannot tell from one that is - not; the shape rules can, beside the array's shape. + Every chunk length is at least 1, along a dimension of length 0 too: + "Chunk sizes must be greater than zero" + (https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v3/chunk-grids/regular-grid/index.rst#L40). + The core spec's "The chunk shape elements are non-zero when the + corresponding dimensions of the arrays have non-zero length" + (https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v3/core/index.rst#L284-L285) + says less of an empty dimension, and allows nothing the grid does not. """ - chunk_shape: tuple[Annotated[int, Ge(0)], ...] + chunk_shape: tuple[Annotated[int, Ge(1)], ...] class RegularChunkGridObject(TypedDict, closed=True): @@ -54,14 +56,11 @@ class RegularChunkGridObject(TypedDict, closed=True): def _shape_rules( configuration: RegularChunkGridConfiguration, nested: Nested, shape: tuple[int, ...] ) -> Iterator[ValidationProblem]: - """A chunk length for each of the array's dimensions, and 0 only for a dimension of length 0. + """A chunk length for each of the array's dimensions. "The dimensionality of the grid is the same as the dimensionality of the array" - (https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v3/chunk-grids/regular-grid/index.rst#L29-L31), - and "The chunk shape elements are non-zero when the corresponding - dimensions of the arrays have non-zero length" - (https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v3/core/index.rst#L284-L285). + (https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v3/chunk-grids/regular-grid/index.rst#L29-L31). """ chunk_shape = configuration["chunk_shape"] if len(chunk_shape) != len(shape): @@ -70,15 +69,6 @@ def _shape_rules( f"expected one chunk length per dimension of shape, got {len(chunk_shape)}", "invalid_value", ) - return - for axis, (length, extent) in enumerate(zip(chunk_shape, shape, strict=True)): - if length == 0 and extent != 0: - yield ValidationProblem( - ("chunk_shape", axis), - f"expected a chunk length >= 1 for a dimension of length {extent}, got 0", - "invalid_value", - ctx={"ge": 1}, - ) def _chunk_lengths( diff --git a/packages/zarr-metadata/tests/v3/test_definitions.py b/packages/zarr-metadata/tests/v3/test_definitions.py index a8a9271b5b..2225b4d522 100644 --- a/packages/zarr-metadata/tests/v3/test_definitions.py +++ b/packages/zarr-metadata/tests/v3/test_definitions.py @@ -489,15 +489,6 @@ def test_a_field_is_written_as_every_reader_takes_it( {"chunk_shape": (2, 3)}, [], ), - # An extent of 0 is right on a dimension of length 0, which only the - # array's shape can tell. - ( - {"name": "regular", "configuration": {"chunk_shape": [0, 3]}}, - ChunkGridDefinition, - Read, - {"chunk_shape": (0, 3)}, - [], - ), # An unknown key is survivable: reported, left out of the # configuration, and the field still read. ( @@ -579,7 +570,6 @@ def test_a_field_is_written_as_every_reader_takes_it( "bytes-bare", "bytes-endian", "regular-grid", - "regular-grid-zero-extent", "unknown-key", "unknown-key-before-the-rules", "not-required-postponed", diff --git a/packages/zarr-metadata/tests/v3/test_every_definition.py b/packages/zarr-metadata/tests/v3/test_every_definition.py index 8617799364..5744860327 100644 --- a/packages/zarr-metadata/tests/v3/test_every_definition.py +++ b/packages/zarr-metadata/tests/v3/test_every_definition.py @@ -493,6 +493,13 @@ def test_error_scale_offset_scalar_is_null() -> None: ] +def test_error_regular_chunk_length_is_zero() -> None: + # Along a dimension of length 0 too: a grid's chunks have a size. + assert _one("chunk_grid:regular", {"chunk_shape": [4, 0]}) == [ + (("configuration", "chunk_shape", 1), "invalid_value") + ] + + def test_error_sharding_inner_chunk_extent_is_zero() -> None: configuration = {"chunk_shape": [0], "codecs": ["bytes"], "index_codecs": ["bytes"]} assert _one("codecs:sharding_indexed", configuration) == [ @@ -542,9 +549,9 @@ def test_error_sharding_inner_chunk_extent_is_zero() -> None: ), ( ChunkGridDefinition, - {"name": "regular", "configuration": {"chunk_shape": [2, -1]}}, + {"name": "regular", "configuration": {"chunk_shape": [2, 0]}}, ("configuration", "chunk_shape", 1), - {"ge": 0}, + {"ge": 1}, ), ( ChunkGridDefinition, diff --git a/packages/zarr-metadata/tests/v3/test_grid_shapes.py b/packages/zarr-metadata/tests/v3/test_grid_shapes.py index 6797fcb47e..93747545b5 100644 --- a/packages/zarr-metadata/tests/v3/test_grid_shapes.py +++ b/packages/zarr-metadata/tests/v3/test_grid_shapes.py @@ -44,9 +44,9 @@ def _problems(grid: JSONValue, shape: tuple[int, ...]) -> list[tuple[tuple[str | [ (_regular(), (), ()), (_regular(4, 4), (10, 3), ({4}, {4})), - # A chunk longer than its dimension, and a chunk length of 0 for a - # dimension of length 0. - (_regular(8, 0), (3, 0), ({8}, {0})), + # A chunk longer than its dimension, and a chunk over a dimension of + # length 0. + (_regular(8, 1), (3, 0), ({8}, {1})), # A bare integer repeats until it covers its dimension. (_rectilinear(4), (10,), ({4},)), (_rectilinear([4, 4, 2]), (10,), ({4, 2},)), @@ -79,12 +79,6 @@ def test_error_a_regular_grid_of_another_rank(grid: JSONValue, shape: tuple[int, assert _problems(grid, shape) == [(("configuration", "chunk_shape"), "invalid_value")] -def test_error_a_regular_chunk_length_of_0_for_a_dimension_that_is_not_empty() -> None: - assert _problems(_regular(4, 0), (4, 3)) == [ - (("configuration", "chunk_shape", 1), "invalid_value") - ] - - @pytest.mark.parametrize( ("grid", "shape"), [(_rectilinear(4), (10, 3)), (_rectilinear(4, 4), (1,))] ) From 887f460dd55730f21b97a422d98a836f2d151123 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Tue, 29 Sep 2026 14:00:46 +0200 Subject: [PATCH 09/94] feat(zarr-metadata): JSON Schemas of a TypedDict, a field in a scope, and a zarr.json As pydantic's `TypeAdapter(...).json_schema()` and zod's `toJSONSchema` write theirs, in draft 2020-12: - `json_schema(shape)`, in `zarr_metadata.typed_json`, writes a TypedDict as `check` reads it. A bound is JSON Schema's keyword for it, the stricter where a type and its `NewType` both say one, and a `Doc` the `description`. Each TypedDict and type alias is written once, in `$defs`, under its name. - `field_json_schema(kind, context)`, in `zarr_metadata.v3.definition`, writes a field of one kind as a scope reads it: each definition's field, and a name none of them is written with, with any configuration. Raw bits' name is matched to its end, so a validator that matches patterns as Python does takes no final newline for it. - `node_metadata_json_schema_v3(context=...)`, in `zarr_metadata.model`, writes a `zarr.json`: its fill value held to the data type it names, and a group's consolidated metadata holding documents. `Schemas` writes each shape, asking a caller's `SchemaLeaf` first, as the checker asks a `Leaf`, and hands back a schema sharing nothing. A schema says what the types say and not what the rules say, so every JSON document the package finds nothing wrong with, the schema accepts. Property tests hold the typed_json schema to the checker's reference, bounds in layers to what `check` holds a value to, and fields and documents changed in one or two places to soundness. JSON Schema takes `1.0` for an integer. Only a `Doc` is a description, as in zod: the package's docstrings are written for Python's readers. The six field aliases move to `v3._common`, so `ZarrV3ArrayMetadataJSON` says what each extension point is: `data_type: DataTypeField`, `codecs: tuple[CodecField, ...]`. To a type checker and to `check` they are the JSON they were, and the model's tables of extension points are read off the TypedDict. `shape` holds integers of at least 0, which `check` now holds it to. A data type whose `fill_value` holds a metadata field is refused when it is built, since the checker reads a fill value as a value and a schema would write it as a field. Assisted-by: ClaudeCode:claude-opus-5-5 --- packages/zarr-metadata/README.md | 38 +- packages/zarr-metadata/changes/378.feature.md | 21 + packages/zarr-metadata/docs/api/index.md | 10 +- packages/zarr-metadata/docs/index.md | 23 + .../src/zarr_metadata/_typed_json.py | 247 +++++ .../src/zarr_metadata/model/__init__.py | 6 +- .../src/zarr_metadata/model/_json_schema.py | 98 ++ .../src/zarr_metadata/model/_validation.py | 26 +- .../src/zarr_metadata/typed_json.py | 27 +- .../src/zarr_metadata/v3/_common.py | 33 +- .../src/zarr_metadata/v3/_definition.py | 173 +++- .../src/zarr_metadata/v3/array.py | 40 +- .../src/zarr_metadata/v3/definition.py | 48 +- .../zarr-metadata/tests/test_json_schema.py | 899 ++++++++++++++++++ .../zarr-metadata/tests/test_public_api.py | 1 + .../tests/test_typed_json_properties.py | 42 +- .../tests/v3/test_definitions.py | 25 +- 17 files changed, 1671 insertions(+), 86 deletions(-) create mode 100644 packages/zarr-metadata/changes/378.feature.md create mode 100644 packages/zarr-metadata/src/zarr_metadata/model/_json_schema.py create mode 100644 packages/zarr-metadata/tests/test_json_schema.py diff --git a/packages/zarr-metadata/README.md b/packages/zarr-metadata/README.md index 11eb9bac52..d91a85a443 100644 --- a/packages/zarr-metadata/README.md +++ b/packages/zarr-metadata/README.md @@ -136,13 +136,37 @@ build the model of either kind, as the models' own `from_json` and A member the spec does not define is not a field; the model's `must_understand_fields` names those a reader must understand. -The Pydantic integration's generated JSON Schemas express independently -checkable document structure and field constraints, but they are not a -replacement for runtime model validation. Standard JSON Schema treats a -mathematically integral number such as `1.0` as an integer, while the runtime -boundary requires Python `int` values, and it cannot express arbitrary -same-length relations such as `dimension_names` versus `shape` or v2 `chunks` -versus `shape`. Consumers should run the model parser after schema validation. +`node_metadata_json_schema_v3` writes what the validators read as a +JSON Schema, draft 2020-12, for an editor that checks a `zarr.json` as it +is written, or a validator in another language. Each extension point is +a field as its scope reads it: a configuration as its definition's +TypedDict says, bounds and all, and a name nothing in the scope claims +with any configuration. The fill value is what the data type it names +takes. `field_json_schema(kind, context)`, in +`zarr_metadata.v3.definition`, writes one field's schema, and +`json_schema`, in `zarr_metadata.typed_json`, any TypedDict's, as `check` +reads it. A schema says what each member is, and not what the rules say +of members together, so a document it accepts may still have a problem; +a JSON document the validators accept, it accepts. A validator reads +JSON as a parser gives it, arrays as lists: a model's `to_json` writes +tuples, which a Python validator does not take for arrays. + +```python +import json +from zarr_metadata.model import node_metadata_json_schema_v3 + +with open("zarr.schema.json", "w") as f: + json.dump(node_metadata_json_schema_v3(), f, indent=2) +``` + +The Pydantic integration's field types have JSON Schemas of their own, +for a model that holds them: an extension point there is a name and any +configuration, read in no scope, and v2 documents have one too. For a +`zarr.json`, use `node_metadata_json_schema_v3`. Neither replaces the +validators: JSON Schema takes a number such as `1.0` for an integer, +where the models require an `int`, and says nothing of what members read +together say, such as `dimension_names` against `shape` or v2 `chunks` +against `shape`. Run the model parser after schema validation. ## Scope diff --git a/packages/zarr-metadata/changes/378.feature.md b/packages/zarr-metadata/changes/378.feature.md new file mode 100644 index 0000000000..9943b052bd --- /dev/null +++ b/packages/zarr-metadata/changes/378.feature.md @@ -0,0 +1,21 @@ +JSON Schemas of what the package reads, draft 2020-12, as pydantic's +`TypeAdapter(...).json_schema()` and zod's `toJSONSchema` write theirs. +`node_metadata_json_schema_v3(context=...)`, in `zarr_metadata.model`, +is a `zarr.json`'s, an array's or a group's, for an editor or a +validator in another language: each extension point a field as its +scope reads it -- a configuration as its definition's TypedDict says, +bounds and all, or a name nothing in the scope claims, with any +configuration -- the fill value what the data type it names takes, and +a group's consolidated metadata the documents it holds. +`field_json_schema(kind, context)`, in `zarr_metadata.v3.definition`, is +one field's, and `json_schema(shape)`, in `zarr_metadata.typed_json`, +any TypedDict's, as `check` reads it. What the rules say of members +together is not in a schema, so a document it accepts may still have a +problem; a JSON document the validators accept, it accepts. JSON Schema +takes `1.0` for an integer, where the package wants `1`. A v3 array +document's extension points are annotated with the field aliases -- +`data_type: DataTypeField`, `codecs: tuple[CodecField, ...]` -- which +are the JSON a metadata field is, so a type checker and `check` read +them as before; its `shape` holds integers of at least 0, which `check` +now holds it to. A data type whose `fill_value` holds a metadata field +is refused when it is built: a fill value is a value of its data type. diff --git a/packages/zarr-metadata/docs/api/index.md b/packages/zarr-metadata/docs/api/index.md index 66c7bda75b..7f4c73c7dc 100644 --- a/packages/zarr-metadata/docs/api/index.md +++ b/packages/zarr-metadata/docs/api/index.md @@ -7,12 +7,14 @@ title: API reference The package is organized to mirror the structure of the Zarr specifications: - [`zarr_metadata.model`](model.md) — frozen-dataclass document models, - validators, loc-aware parsers, and the `UNSET` sentinel + validators, loc-aware parsers, a `zarr.json`'s JSON Schema, and the + `UNSET` sentinel - [`zarr_metadata.pydantic`](pydantic.md) — optional Pydantic field types over the models - [`zarr_metadata.typed_json`](typed_json.md) — `check`, which type-checks a JSON value against any of the package's `TypedDict`s, read as the - typing spec defines them, with every problem located + typing spec defines them, with every problem located, and + `json_schema`, which writes what `check` reads as a JSON Schema - [`zarr_metadata.v2`](v2.md) — `TypedDict` shapes for Zarr v2 documents (`.zarray`, `.zgroup`, `.zattrs`, `.zmetadata`) - [`zarr_metadata.v3`](v3/index.md) — `TypedDict` shapes for Zarr v3 @@ -22,8 +24,8 @@ The package is organized to mirror the structure of the Zarr specifications: - [`zarr_metadata.v3.definition`](v3/definition.md) — each extension's metadata as a definition: the TypedDict its configuration is, and the rules on it; check JSON against a TypedDict, judge a configuration, - read a whole field in a scope, or read a codec pipeline. Its module - docstring is the guide + read a whole field in a scope, read a codec pipeline, or write a + scope's fields as a JSON Schema. Its module docstring is the guide The document types, models, and spec vocabulary — including the store keys — are re-exported at the top level, so diff --git a/packages/zarr-metadata/docs/index.md b/packages/zarr-metadata/docs/index.md index 8feaf6e95a..627b37ed56 100644 --- a/packages/zarr-metadata/docs/index.md +++ b/packages/zarr-metadata/docs/index.md @@ -151,6 +151,29 @@ build the model of either kind, as the models' own `from_json` and A member the spec does not define is not a field; the model's `must_understand_fields` names those a reader must understand. +`node_metadata_json_schema_v3` writes what the validators read as a +JSON Schema, draft 2020-12, for an editor that checks a `zarr.json` as it +is written, or a validator in another language. Each extension point is +a field as its scope reads it: a configuration as its definition's +TypedDict says, bounds and all, and a name nothing in the scope claims +with any configuration. The fill value is what the data type it names +takes. `field_json_schema(kind, context)`, in +`zarr_metadata.v3.definition`, writes one field's schema, and +`json_schema`, in `zarr_metadata.typed_json`, any TypedDict's, as `check` +reads it. A schema says what each member is, and not what the rules say +of members together, so a document it accepts may still have a problem; +a JSON document the validators accept, it accepts. A validator reads +JSON as a parser gives it, arrays as lists: a model's `to_json` writes +tuples, which a Python validator does not take for arrays. + +```python +import json +from zarr_metadata.model import node_metadata_json_schema_v3 + +with open("zarr.schema.json", "w") as f: + json.dump(node_metadata_json_schema_v3(), f, indent=2) +``` + ## Scope At minimum, this library supports what Zarr-Python needs: the complete diff --git a/packages/zarr-metadata/src/zarr_metadata/_typed_json.py b/packages/zarr-metadata/src/zarr_metadata/_typed_json.py index a5c38f2476..d2dca722e7 100644 --- a/packages/zarr-metadata/src/zarr_metadata/_typed_json.py +++ b/packages/zarr-metadata/src/zarr_metadata/_typed_json.py @@ -28,6 +28,10 @@ depth; the parser it returns is used as it is. Parsers are compiled once per annotation and are pure functions of the value, so the branch of a union that did not match leaves nothing behind. + +`json_schema` writes the same reading as a JSON Schema: `Schemas` writes +each shape as the checker reads it, asking a caller's `SchemaLeaf` first, +as a parser asks a `Leaf`. """ from __future__ import annotations @@ -38,6 +42,7 @@ import sys import types import typing +import urllib.parse from collections.abc import Callable, Iterator, Mapping, Sequence from collections.abc import Set as AbstractSet from dataclasses import dataclass @@ -66,6 +71,7 @@ from zarr_metadata._json import ( ValidationProblem, choices, + copied, is_json, outside_of, refine_json, @@ -1135,14 +1141,253 @@ def check( return (cast("T", typed) if readable else None), with_input(found, value, loc) +# --- JSON Schema --------------------------------------------------------- + +JSONSchema: TypeAlias = dict[str, JSONValue] +"""A JSON Schema, as the JSON object it is: arrays as lists, as validators take them.""" + +SchemaLeaf: TypeAlias = Callable[[object, "Schemas"], "JSONSchema | None"] +"""A caller's own shapes, written into a schema: asked first for every annotation, as a `Leaf` is, None to decline. + +Handed the schema being written, so a shape of the caller's own can hold +others, written with `of`, or be written once, in `$defs`, with `defined`. +""" + +DIALECT: Final = "https://json-schema.org/draft/2020-12/schema" +"""The dialect every schema is written in: JSON Schema draft 2020-12, as pydantic and zod write theirs.""" + +_KEYWORDS: Final[Mapping[str, str]] = { + "gt": "exclusiveMinimum", + "ge": "minimum", + "lt": "exclusiveMaximum", + "le": "maximum", +} +"""JSON Schema's keyword for each bound.""" + +_STRICTER: Final[Mapping[str, Callable[[float, float], float]]] = { + "exclusiveMinimum": max, + "minimum": max, + "exclusiveMaximum": min, + "maximum": min, +} +"""Of two bounds a keyword says, the one a value in both keeps within.""" + + +def no_schema_leaf(annotation: object, schemas: Schemas) -> JSONSchema | None: + """The schema leaf of a caller with no shapes of its own.""" + return None + + +class Schemas: + """One JSON Schema being written, and the `$defs` it holds so far. + + `of` writes an annotation as the checker reads it, asking the leaf + first at every depth, as `parser_for` asks a `Leaf`. A TypedDict and a + type alias are each written once, in `$defs`, under its name -- or its + name and a number, when another holds that one -- and referred to + wherever they occur, so one that holds itself is a schema that refers + to itself. `document` is the whole schema. + """ + + __slots__ = ("_defs", "_leaf", "_names", "_uses") + + def __init__(self, leaf: SchemaLeaf = no_schema_leaf) -> None: + self._leaf = leaf + self._defs: dict[str, JSONSchema] = {} + self._names: dict[object, str] = {} + self._uses: dict[str, int] = {} + + def of(self, annotation: object) -> JSONSchema: + """`annotation` as the checker reads it: its type, with the bounds and notes `Annotated` carries. + + A bound is the keyword JSON Schema has for it -- `Ge(0)` is + `minimum` -- and a `Doc` is the `description`. A type's bounds and + the bounds its `NewType` holds are both kept, as the checker holds + a value to both: where the two say one keyword, the stricter. An + annotation the checker reads is written; any other is a + `TypeError`, which a caller that vetted it through `parser` never + meets. + """ + inner, metadata = strip_annotation(annotation) + schema = self._type(inner) + if len(metadata) == 0: + return schema + notes = [ + item.documentation + for item in _unpacked(metadata) + if isinstance(item, typing_extensions.Doc) + ] + if len(notes) != 0: + schema = {**schema, "description": "\n\n".join(notes)} + for name, bound in constraints_of(metadata).items(): + keyword = _KEYWORDS[name] + held = cast("int | float | None", schema.get(keyword)) + schema = {**schema, keyword: bound if held is None else _STRICTER[keyword](held, bound)} + return schema + + def object_of(self, typeddict: type) -> JSONSchema: + """`typeddict` written in place: its keys, those it requires, and what any other key may hold.""" + keys = typeddict_keys(typeddict) + schema: JSONSchema = {"type": "object"} + if len(keys.members) != 0: + schema["properties"] = { + key: self.of(annotation) for key, (annotation, _) in keys.members.items() + } + required: list[JSONValue] = [key for key in keys.members if key in keys.required] + if len(required) != 0: + schema["required"] = required + if keys.closed: + schema["additionalProperties"] = False + elif not keys.open: + extra = self.of(keys.extra_items) + if len(extra) != 0: + schema["additionalProperties"] = extra + return schema + + def defined(self, key: object, name: str, write: Callable[[], JSONSchema]) -> JSONSchema: + """A reference to the entry in `$defs` for `key`, which `write` writes the first time `key` is asked for. + + The entry is named `name`, or `name` and a number when another key + holds that name, and it is reserved before it is written, so a + schema that holds itself refers to itself. One `write` fails to + write is not left reserved: a later reference to it would be to an + empty schema, which takes anything. + """ + name_held = self._names.get(key) + if name_held is None: + name_held, number = name, 1 + while name_held in self._defs: + number += 1 + name_held = f"{name}{number}" + self._names[key] = name_held + self._defs[name_held] = {} + try: + self._defs[name_held] = write() + except BaseException: + del self._names[key], self._defs[name_held] + raise + self._uses[name_held] = self._uses.get(name_held, 0) + 1 + return {"$ref": _pointer(name_held)} + + def document(self, root: JSONSchema) -> JSONSchema: + """The whole schema: its dialect, `root`, and the `$defs`, by name. + + `root` is written in place when it refers to an entry nothing else + refers to, as pydantic writes a model that does not hold itself. + What comes back shares nothing with what was written, nor one part + of it with another, so a caller may change it where it likes. + """ + defs = dict(self._defs) + target = next((name for name in defs if root == {"$ref": _pointer(name)}), None) + if target is not None and self._uses[target] == 1: + root = defs.pop(target) + whole: JSONSchema = {"$schema": DIALECT, **root} + if len(defs) != 0: + whole["$defs"] = {name: defs[name] for name in sorted(defs)} + return cast("JSONSchema", copied(whole)) + + def _type(self, inner: object) -> JSONSchema: + """The schema of a type, `Annotated` peeled from it.""" + found = self._leaf(inner, self) + if found is not None: + return found + if inner is int: + return {"type": "integer"} + if inner is float: + return {"type": "number"} + if inner is bool: + return {"type": "boolean"} + if inner is str: + return {"type": "string"} + if inner is None or inner is types.NoneType: + return {"type": "null"} + if inner is JSONValue: + return {} + origin = get_origin(inner) + if origin is Literal: + # Sorted, as `_literal` sorts them: `get_args` reports a + # `Literal`'s values in the order the first one built wrote them. + values: list[JSONValue] = sorted(get_args(inner), key=repr) + return {"const": values[0]} if len(values) == 1 else {"enum": values} + if is_union(inner): + return {"anyOf": [self.of(branch) for branch in get_args(inner)]} + if origin is tuple: + return self._tuple(get_args(inner)) + if origin in (Mapping, dict): + value = self.of(get_args(inner)[1]) + if len(value) == 0: + return {"type": "object"} + return {"type": "object", "additionalProperties": value} + if isinstance(inner, type) and is_typeddict(inner): + typeddict = inner + return self.defined(typeddict, typeddict.__name__, lambda: self.object_of(typeddict)) + if isinstance(inner, NewType): + return self.of(inner.__supertype__) + if is_alias(inner): + alias = cast("typing_extensions.TypeAliasType", inner) + return self.defined(alias, alias.__name__, lambda: self.of(alias_value(alias))) + msg = f"{inner!r} is not a shape JSON takes" + raise TypeError(msg) + + def _tuple(self, arguments: tuple[object, ...]) -> JSONSchema: + if len(arguments) == 2 and arguments[1] is Ellipsis: + items = self.of(arguments[0]) + return {"type": "array"} if len(items) == 0 else {"type": "array", "items": items} + if len(arguments) == 0: + return {"type": "array", "maxItems": 0} + return { + "type": "array", + "prefixItems": [self.of(argument) for argument in arguments], + "items": False, + "minItems": len(arguments), + } + + +def _pointer(name: str) -> str: + """The reference to the entry in `$defs` named `name`: a JSON pointer, escaped as a URI fragment.""" + escaped = name.replace("~", "~0").replace("/", "~1") + return f"#/$defs/{urllib.parse.quote(escaped, safe='')}" + + +def json_schema(shape: type) -> JSONSchema: + """The JSON Schema of the JSON `check` finds no problem with as `shape`, a TypedDict. + + Draft 2020-12, as a JSON object: arrays as lists, and `$schema` + first. A TypedDict is an object of its keys, those it requires, and + what any other key may hold -- nothing, in a closed one; a bound is + the keyword JSON Schema has for it, `Ge(0)` a `minimum`; a `Doc` is + the `description`, which is all that says one, as zod writes only + what `.describe()` said: a docstring is written for Python's readers; + a `Literal` is its values; a union is `anyOf` its branches. A + TypedDict or type alias is written once in `$defs`, under its name, + and referred to wherever it occurs, but for `shape` itself, which is + written in place unless it holds itself. + + One difference is JSON Schema's own: it takes a number with no + fraction, `1.0`, for an integer, where `check` wants `1`. `TypeError` + for a `shape` that is not a TypedDict, or holds something no parser + reads, as `check` raises it. + """ + if not is_typeddict(shape): + msg = f"{shape!r} is not a TypedDict" + raise TypeError(msg) + _checker(shape) + schemas = Schemas() + return schemas.document(schemas.of(shape)) + + __all__ = [ + "DIALECT", "Branch", "Constraints", + "JSONSchema", "Leaf", "Loc", "Parsed", "Parser", "Qualifier", + "SchemaLeaf", + "Schemas", "Tag", "TypedDictKeys", "alias_value", @@ -1156,8 +1401,10 @@ def check( "is_alias", "is_integer", "is_union", + "json_schema", "mapping_of", "no_leaf", + "no_schema_leaf", "object_of", "one_of", "parser", diff --git a/packages/zarr-metadata/src/zarr_metadata/model/__init__.py b/packages/zarr-metadata/src/zarr_metadata/model/__init__.py index c5f45073fc..0b4ec7344f 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/__init__.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/__init__.py @@ -13,7 +13,9 @@ that narrows or raises `MetadataValidationError`; a v3 array or group document also gets `read_array_metadata_v3` or `read_group_metadata_v3`, one read that returns what it read, the problems, and the model when -there are none. Model `from_json` / `from_key_value` constructors raise +there are none. `node_metadata_json_schema_v3` writes what the v3 +validators read as a JSON Schema, but for the rules. Model `from_json` / +`from_key_value` constructors raise `MetadataValidationError` for every ingestion failure, including missing store keys and undecodable bytes, and the v3 ones take the same `context`. @@ -55,6 +57,7 @@ validate_group_metadata_v3, validate_node_metadata_v3, ) +from zarr_metadata.model._json_schema import node_metadata_json_schema_v3 from zarr_metadata.model._validation import ( ARRAY_METADATA_OPTIONAL_KEYS_V3, ARRAY_METADATA_REQUIRED_KEYS_V2, @@ -159,6 +162,7 @@ "is_metadata_field_v3", "node_metadata_from_json_v3", "node_metadata_from_key_value_v3", + "node_metadata_json_schema_v3", "parse_array_metadata_v2", "parse_array_metadata_v3", "parse_group_metadata_v2", diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_json_schema.py b/packages/zarr-metadata/src/zarr_metadata/model/_json_schema.py new file mode 100644 index 0000000000..3b6b015fd2 --- /dev/null +++ b/packages/zarr-metadata/src/zarr_metadata/model/_json_schema.py @@ -0,0 +1,98 @@ +"""The JSON Schema of a v3 `zarr.json`, as a scope reads one.""" + +from __future__ import annotations + +from typing import TYPE_CHECKING, Any, cast + +from zarr_metadata._typed_json import Schemas +from zarr_metadata.v3._definition import DataTypeDefinition, field_schemas, written_name +from zarr_metadata.v3._registry import CORE_AND_EXTENSIONS, Context +from zarr_metadata.v3.array import ZarrV3ArrayMetadataJSON +from zarr_metadata.v3.consolidated import ( + ZARR_V3_CONSOLIDATED_METADATA_KEY, + ZarrV3ConsolidatedMetadataJSON, +) +from zarr_metadata.v3.group import ZarrV3GroupMetadataJSON + +if TYPE_CHECKING: + from zarr_metadata._common import JSONValue + from zarr_metadata._typed_json import JSONSchema, SchemaLeaf + + +def node_metadata_json_schema_v3(*, context: Context = CORE_AND_EXTENSIONS) -> JSONSchema: + """The JSON Schema of a v3 `zarr.json` read in `context`: an array document or a group document, as `validate_node_metadata_v3` reads one, but for the rules. + + For an editor that validates a `zarr.json` as it is written, or a + validator in another language. JSON Schema draft 2020-12, as + `json_schema` writes one. Each extension point is a field as + `field_json_schema` writes one in `context`: one a definition in scope + reads, or a name none of them claims. The fill value is the JSON shape + the data type's definition declares for one -- an `int8`'s an integer + in [-128, 127] -- when the document names a data type in scope. A + group's `consolidated_metadata` holds array and group documents, by + path, or is `null`. Each document is in `$defs` under the name of its + TypedDict: `ZarrV3ArrayMetadataJSON` is an array's alone. + + A JSON Schema says what each member is, and what the rules say of + members read together is not in it: one dimension name per dimension + of the shape, a chunk grid that fits the shape, codecs in the order a + pipeline takes them, each against the chunk it is handed, and what a + definition's `rules` say. So a document it accepts may still have a + problem, and a JSON document `validate_node_metadata_v3` finds none + with, it accepts. A validator reads JSON as a parser gives it, arrays + as lists: a model's `to_json` writes tuples, which a Python validator + does not take for arrays. + """ + schemas = Schemas(_documents(context)) + array = schemas.of(ZarrV3ArrayMetadataJSON) + group = schemas.of(ZarrV3GroupMetadataJSON) + return schemas.document({"anyOf": [array, group]}) + + +def _documents(context: Context) -> SchemaLeaf: + """The schema leaf of the documents read in `context`: an array's fill value held to its data type, and a group's consolidated metadata; each field as `context` reads one.""" + fields = field_schemas(context) + + def leaf(annotation: object, schemas: Schemas) -> JSONSchema | None: + if annotation is ZarrV3ArrayMetadataJSON: + return schemas.defined( + annotation, annotation.__name__, lambda: _array(context, schemas) + ) + if annotation is ZarrV3GroupMetadataJSON: + return schemas.defined(annotation, annotation.__name__, lambda: _group(schemas)) + return fields(annotation, schemas) + + return leaf + + +def _array(context: Context, schemas: Schemas) -> JSONSchema: + """An array document: its TypedDict, and its fill value held to the data type it names, for each data type in scope.""" + schema = schemas.object_of(ZarrV3ArrayMetadataJSON) + held: list[JSONValue] = [] + for definition in context.tables.get(DataTypeDefinition, {}).values(): + fill_value = schemas.of(cast("DataTypeDefinition[Any]", definition).fill_value) + if len(fill_value) == 0: + continue # a data type that says nothing of its fill value takes any JSON + name = written_name(definition) + named: JSONSchema = { + "anyOf": [name, {"type": "object", "properties": {"name": name}, "required": ["name"]}] + } + condition: JSONSchema = { + "if": {"properties": {"data_type": named}, "required": ["data_type"]}, + "then": {"properties": {"fill_value": fill_value}}, + } + held.append(condition) + return schema if len(held) == 0 else {**schema, "allOf": held} + + +def _group(schemas: Schemas) -> JSONSchema: + """A group document: its TypedDict, and the consolidated metadata the model reads, which a historical zarr-python bug wrote as `null`.""" + schema = schemas.object_of(ZarrV3GroupMetadataJSON) + properties = cast("dict[str, JSONValue]", schema.get("properties", {})) + consolidated: JSONSchema = { + "anyOf": [schemas.of(ZarrV3ConsolidatedMetadataJSON), {"type": "null"}] + } + return {**schema, "properties": {**properties, ZARR_V3_CONSOLIDATED_METADATA_KEY: consolidated}} + + +__all__ = ["node_metadata_json_schema_v3"] diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_validation.py b/packages/zarr-metadata/src/zarr_metadata/model/_validation.py index 2f8823cd7c..4fc1c70f91 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_validation.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/_validation.py @@ -28,7 +28,7 @@ import json from collections.abc import Mapping, Sequence from dataclasses import dataclass -from typing import TYPE_CHECKING, Any, Final, TypeGuard, TypeVar, cast +from typing import TYPE_CHECKING, Any, Final, TypeGuard, TypeVar, cast, get_args, get_origin from zarr_metadata._json import ( MetadataValidationError, @@ -44,13 +44,13 @@ from zarr_metadata._json import is_canonical_json as _is_canonical_json from zarr_metadata._json import prefixed as _prefix from zarr_metadata._sentinel import UNSET +from zarr_metadata._typed_json import typeddict_keys from zarr_metadata.v2.array import ZarrV2ArrayMetadataJSON from zarr_metadata.v2.group import ZarrV2GroupMetadataJSON from zarr_metadata.v3._definition import ( Chunk, ChunkGridDefinition, ChunkKeyEncodingDefinition, - CodecDefinition, DataTypeDefinition, Definition, Lengths, @@ -59,6 +59,7 @@ StorageTransformerDefinition, Unclaimed, chunk_grid_lengths, + field_kind, fields_of, fill_value_problems, resolve, @@ -375,18 +376,21 @@ def attributes_of( return (attributes if len(problems) == 0 else None), tuple(problems) -_EXTENSION_POINTS_V3: Final[tuple[tuple[str, type[Definition[Any]]], ...]] = ( - ("data_type", DataTypeDefinition), - ("chunk_grid", ChunkGridDefinition), - ("chunk_key_encoding", ChunkKeyEncodingDefinition), +_MEMBERS_V3: Final = typeddict_keys(ZarrV3ArrayMetadataJSON).members + +_EXTENSION_POINTS_V3: Final[tuple[tuple[str, type[Definition[Any]]], ...]] = tuple( + (key, kind) + for key, (annotation, _) in _MEMBERS_V3.items() + if (kind := field_kind(annotation)) is not None ) -"""A v3 array document's single extension points, and the kind each is read as.""" +"""A v3 array document's single extension points, and the kind each is read as: each member its TypedDict annotates with a field alias.""" -_EXTENSION_LISTS_V3: Final[tuple[tuple[str, type[Definition[Any]]], ...]] = ( - ("codecs", CodecDefinition), - ("storage_transformers", StorageTransformerDefinition), +_EXTENSION_LISTS_V3: Final[tuple[tuple[str, type[Definition[Any]]], ...]] = tuple( + (key, kind) + for key, (annotation, _) in _MEMBERS_V3.items() + if get_origin(annotation) is tuple and (kind := field_kind(get_args(annotation)[0])) is not None ) -"""Its lists of extension points, and the kind each entry is read as.""" +"""Its lists of extension points, and the kind each entry is read as: each member it annotates as a tuple of a field alias.""" @dataclass(frozen=True, slots=True) diff --git a/packages/zarr-metadata/src/zarr_metadata/typed_json.py b/packages/zarr-metadata/src/zarr_metadata/typed_json.py index 097411a898..3ca22ad4f8 100644 --- a/packages/zarr-metadata/src/zarr_metadata/typed_json.py +++ b/packages/zarr-metadata/src/zarr_metadata/typed_json.py @@ -60,6 +60,22 @@ holds at `loc`, and `ctx`, what was expected there -- a type's bounds, or the values of a `Literal`. +`json_schema` writes what `check` reads as a JSON Schema, draft 2020-12, +for a validator in another language, or an editor: the JSON Schema of +the values `check` finds no problem with. A TypedDict is an object of its +keys, closed or not as it says; a bound is JSON Schema's keyword for it, +`Interval(ge=0, le=9)` a `minimum` and a `maximum`; a TypedDict or a type +alias is written once, in `$defs`, under its name. JSON Schema takes a +number with no fraction, `1.0`, for an integer, where `check` wants `1`: + + from zarr_metadata.typed_json import json_schema + + json_schema(GzipCodecConfiguration) + # {'$schema': 'https://json-schema.org/draft/2020-12/schema', + # 'type': 'object', + # 'properties': {'level': {'type': 'integer', 'minimum': 0, 'maximum': 9}}, + # 'required': ['level'], 'additionalProperties': False} + `check` reads the shapes JSON takes and no others -- `int`, `float` for any number, `bool`, `str`, `None`, `JSONValue`, a `Literal`, `tuple[T, ...]` and `tuple[T1, T2]`, a union, a TypedDict, @@ -77,14 +93,23 @@ from zarr_metadata._common import JSONValue from zarr_metadata._json import ProblemKind, ValidationProblem -from zarr_metadata._typed_json import Loc, TypedDictKeys, check, typeddict_keys +from zarr_metadata._typed_json import ( + JSONSchema, + Loc, + TypedDictKeys, + check, + json_schema, + typeddict_keys, +) __all__ = [ + "JSONSchema", "JSONValue", "Loc", "ProblemKind", "TypedDictKeys", "ValidationProblem", "check", + "json_schema", "typeddict_keys", ] diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/_common.py b/packages/zarr-metadata/src/zarr_metadata/v3/_common.py index 04a0874bf6..1347130a6b 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/_common.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/_common.py @@ -1,14 +1,17 @@ -"""The v3 metadata field: its JSON, and the validators that judge one on its own. +"""The v3 metadata field: its JSON, the aliases a member holding one is annotated with, and the validators that judge one on its own. Private, and below both readers of a field: the model, which judges the fields of a document, and the definitions, which read a field's configuration. Public consumers import `ZarrV3MetadataFieldJSON` from -`zarr_metadata.v3`, and the validators from `zarr_metadata.model`. +`zarr_metadata.v3`, the aliases from `zarr_metadata.v3.definition`, and +the validators from `zarr_metadata.model`. """ from collections.abc import Mapping from typing import TypeGuard, cast +from typing_extensions import TypeAliasType + from zarr_metadata._common import ZarrV3NamedConfigJSON from zarr_metadata._json import ( MetadataValidationError, @@ -31,6 +34,26 @@ """ +# A member holding a metadata field is annotated with the alias of its +# kind, which a scope reads it as. Each alias is the JSON a field is, so to +# a type checker, and to `check`, it is `ZarrV3MetadataFieldJSON`. + +DataTypeField = TypeAliasType("DataTypeField", ZarrV3MetadataFieldJSON) +"""A member holding a data type: a document's `data_type`, or a struct field's; read in the scope what holds it is read in.""" +ChunkGridField = TypeAliasType("ChunkGridField", ZarrV3MetadataFieldJSON) +"""A member holding a chunk grid: a document's `chunk_grid`.""" +ChunkKeyEncodingField = TypeAliasType("ChunkKeyEncodingField", ZarrV3MetadataFieldJSON) +"""A member holding a chunk key encoding: a document's `chunk_key_encoding`.""" +CodecField = TypeAliasType("CodecField", ZarrV3MetadataFieldJSON) +"""A member holding a codec: a document's `codecs` is `tuple[CodecField, ...]`, and so is a shard's.""" +StaticCodecField = TypeAliasType("StaticCodecField", ZarrV3MetadataFieldJSON) +"""A member holding a codec of static size: a shard's `index_codecs` is one, +since a reader finds the index by a size it knows before reading it. +""" +StorageTransformerField = TypeAliasType("StorageTransformerField", ZarrV3MetadataFieldJSON) +"""A member holding a storage transformer: a document's `storage_transformers` is `tuple[StorageTransformerField, ...]`.""" + + def validate_metadata_field_v3( value: object, *, allow_must_understand_false: bool = True ) -> tuple[ValidationProblem, ...]: @@ -142,6 +165,12 @@ def parse_metadata_field_v3(value: object) -> ZarrV3MetadataFieldJSON: __all__ = [ + "ChunkGridField", + "ChunkKeyEncodingField", + "CodecField", + "DataTypeField", + "StaticCodecField", + "StorageTransformerField", "ZarrV3MetadataFieldJSON", "envelope_problems", "is_metadata_field_v3", diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py b/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py index 89410bb3ee..0c589b9db5 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py @@ -54,16 +54,26 @@ from zarr_metadata._json import ValidationProblem, copied, refine_json, shown, with_input from zarr_metadata._sentinel import UNSET from zarr_metadata._typed_json import ( + JSONSchema, Loc, Parsed, Parser, - no_leaf, + SchemaLeaf, + Schemas, parser, problem, typeddict_keys, unread_in, ) -from zarr_metadata.v3._common import ZarrV3MetadataFieldJSON, envelope_problems +from zarr_metadata.v3._common import ( + ChunkGridField, + ChunkKeyEncodingField, + CodecField, + DataTypeField, + StaticCodecField, + StorageTransformerField, + envelope_problems, +) if TYPE_CHECKING: from collections.abc import Sequence @@ -357,8 +367,8 @@ def _refusal(self) -> str | None: @functools.cache def _fill_value_parser(annotation: object) -> Parser: - """The checker for a fill value's JSON shape, compiled once; `TypeError` naming what no checker reads.""" - return parser(annotation, no_leaf) + """The checker for a fill value's JSON shape, compiled once; `TypeError` naming what no checker reads, or a metadata field in it.""" + return parser(annotation, _no_field) @dataclass(frozen=True, kw_only=True, slots=True, repr=False) @@ -501,21 +511,6 @@ def as_kind(kind: object) -> type[Definition[Any]]: return found -DataTypeField = TypeAliasType("DataTypeField", ZarrV3MetadataFieldJSON) -"""A configuration member holding a data type, read in the scope the member's field is read in.""" -ChunkGridField = TypeAliasType("ChunkGridField", ZarrV3MetadataFieldJSON) -"""A configuration member holding a chunk grid.""" -ChunkKeyEncodingField = TypeAliasType("ChunkKeyEncodingField", ZarrV3MetadataFieldJSON) -"""A configuration member holding a chunk key encoding.""" -CodecField = TypeAliasType("CodecField", ZarrV3MetadataFieldJSON) -"""A configuration member holding a codec: a shard's `codecs` is `tuple[CodecField, ...]`.""" -StaticCodecField = TypeAliasType("StaticCodecField", ZarrV3MetadataFieldJSON) -"""A configuration member holding a codec of static size: a shard's `index_codecs` is one, -since a reader finds the index by a size it knows before reading it. -""" -StorageTransformerField = TypeAliasType("StorageTransformerField", ZarrV3MetadataFieldJSON) -"""A configuration member holding a storage transformer.""" - _FIELD_KINDS: Final[Mapping[object, type[Definition[Any]]]] = { DataTypeField: DataTypeDefinition, ChunkGridField: ChunkGridDefinition, @@ -529,6 +524,14 @@ def as_kind(kind: object) -> type[Definition[Any]]: """The field aliases whose codec must be of static size.""" +def field_kind(annotation: object) -> type[Definition[Any]] | None: + """The kind of metadata field a member annotated `annotation` holds -- `CodecDefinition` for `CodecField` -- or None when it holds none.""" + try: + return _FIELD_KINDS.get(annotation) + except TypeError: # an unhashable annotation is no field alias + return None + + @dataclass(frozen=True, slots=True) class _NestedField: """A metadata field the check met inside a configuration: where it sits, its kind, its JSON. @@ -553,10 +556,7 @@ def _field(annotation: object) -> Parser | None: judged, and the name related to a definition, by whoever reads the field -- `check` without a scope, `resolve` in one. """ - try: - kind = _FIELD_KINDS.get(annotation) - except TypeError: # an unhashable annotation is no field alias - return None + kind = field_kind(annotation) if kind is None: return None static = annotation in _STATIC_SIZE @@ -601,6 +601,15 @@ def _vetting(annotation: object) -> Parser | None: return _field(annotation) +def _no_field(annotation: object) -> Parser | None: + """A leaf refusing a field alias: a fill value is a value of its data type, and holds no metadata field.""" + if field_kind(annotation) is not None: + name = cast("TypeAliasType", annotation).__name__ + msg = f"{name} holds a metadata field, and a fill value is a value of its data type" + raise TypeError(msg) + return None + + @functools.cache def _vet(configuration: type) -> None: """Refuse a configuration no definition could read with, saying what is wrong with it.""" @@ -616,6 +625,120 @@ def _vet(configuration: type) -> None: raise TypeError(msg) +_KIND_FIELDS: Final[Mapping[type[Definition[Any]], object]] = { + kind: alias for alias, kind in _FIELD_KINDS.items() if alias not in _STATIC_SIZE +} +"""The field alias of each kind: the one a member holding any field of the kind is annotated with.""" + +_RAW_BYTES_SCHEMA_PATTERN: Final = f"^{RAW_BYTES_NAME_PATTERN.pattern}(?![\\s\\S])" +"""`RAW_BYTES_NAME_PATTERN`, matched whole, as a JSON Schema writes a pattern. + +Held to the end of the name by a lookahead for no character at all: a +`$` there would also match before a final newline in a validator that +matches patterns as Python does, so `"r16\\n"`, which names nothing, +would read as raw bits. +""" + + +def field_json_schema(kind: type[Definition[Any]], context: Context) -> JSONSchema: + """The JSON Schema of one metadata field read as `kind` in `context`: what `resolve` reads, but for the rules. + + A field one of the definitions in scope reads -- its name, its + configuration as the TypedDict says, a `must_understand` of `true` + if any, and its bare name when it needs no configuration -- or a + name none of them claims, with any configuration: what keeps the + format open. A field a configuration holds is written the same way, + in the same scope, and a member taking codecs of static size only + takes those. JSON Schema draft 2020-12, as `json_schema` writes one; + the fields it holds, and the configuration of each definition, are in + `$defs`, under the name of the field alias or TypedDict. What only a + rule says -- a blosc `typesize` against its `shuffle` -- is not in it, + so a field it accepts may still have a problem. + """ + schemas = Schemas(field_schemas(context)) + return schemas.document(schemas.of(_KIND_FIELDS[as_kind(kind)])) + + +def field_schemas(context: Context) -> SchemaLeaf: + """The schema leaf that writes each field alias as a field of its kind, as `context` reads one: `field_json_schema`'s.""" + + def leaf(annotation: object, schemas: Schemas) -> JSONSchema | None: + kind = field_kind(annotation) + if kind is None: + return None + alias = cast("TypeAliasType", annotation) + static = annotation in _STATIC_SIZE + return schemas.defined( + alias, alias.__name__, lambda: _field_schema(kind, static, context, schemas) + ) + + return leaf + + +def _field_schema( + kind: type[Definition[Any]], static: bool, context: Context, schemas: Schemas +) -> JSONSchema: + """A field of `kind` as `context` reads it: one a definition in scope reads, or one none of them claims.""" + table = context.tables.get(kind, {}) + branches: list[JSONValue] = [] + for definition in table.values(): + if static and cast("CodecDefinition[Any]", definition).size != "static": + continue + branches.extend(_read_by(definition, schemas)) + branches.extend(_unclaimed(table)) + return {"anyOf": branches} + + +def written_name(definition: Definition[Any]) -> JSONSchema: + """The JSON Schema of each name a document writes for `definition`: its name, or `r` and a size for raw bits.""" + if isinstance(definition, DataTypeDefinition) and definition.name == RAW_BYTES_NAME: + return {"type": "string", "pattern": _RAW_BYTES_SCHEMA_PATTERN} + return {"const": definition.name} + + +def _read_by(definition: Definition[Any], schemas: Schemas) -> list[JSONValue]: + """The fields `definition` reads: an object of its name and configuration, and its bare name when it needs no configuration. + + Raw bits' name carries their configuration, so what is written beside + it holds nothing. + """ + carried = isinstance(definition, DataTypeDefinition) and definition.name == RAW_BYTES_NAME + name = written_name(definition) + bare = carried or not definition.requires_configuration + envelope: JSONSchema = { + "type": "object", + "properties": { + "name": name, + "configuration": schemas.of( + EmptyConfiguration if carried else definition.configuration + ), + "must_understand": {"const": True}, + }, + "required": ["name"] if bare else ["name", "configuration"], + "additionalProperties": False, + } + return [name, envelope] if bare else [envelope] + + +def _unclaimed(table: Mapping[str, Definition[Any]]) -> list[JSONValue]: + """The fields no definition in `table` claims: a name none of them is written with, bare or with any configuration.""" + claimed: list[JSONValue] = [written_name(definition) for definition in table.values()] + name: JSONSchema = {"type": "string"} + if len(claimed) != 0: + name["not"] = {"anyOf": claimed} + envelope: JSONSchema = { + "type": "object", + "properties": { + "name": name, + "configuration": {"type": "object"}, + "must_understand": {"const": True}, + }, + "required": ["name"], + "additionalProperties": False, + } + return [name, envelope] + + @dataclass(frozen=True, slots=True) class _Checker: """A TypedDict's checker, compiled once, and whether a value of it can hold a nested field.""" @@ -1469,6 +1592,9 @@ def _replaced(value: JSONValue, path: Loc, new: JSONValue) -> JSONValue: "canonicalize", "chunk_grid_lengths", "configuration_of", + "field_json_schema", + "field_kind", + "field_schemas", "fill_value_problems", "kind_of", "multi_byte", @@ -1486,4 +1612,5 @@ def _replaced(value: JSONValue, path: Loc, new: JSONValue) -> JSONValue: "unknown_storage", "variable_length", "with_problems", + "written_name", ] diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/array.py b/packages/zarr-metadata/src/zarr_metadata/v3/array.py index 31a5f6b755..79328d4eb7 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/array.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/array.py @@ -1,12 +1,19 @@ """Zarr v3 array metadata types.""" from collections.abc import Mapping -from typing import Final, Literal, NotRequired, TypeAlias +from typing import Annotated, Final, Literal, NotRequired, TypeAlias +from annotated_types import Ge from typing_extensions import TypedDict from zarr_metadata._common import JSONValue -from zarr_metadata.v3._common import ZarrV3MetadataFieldJSON +from zarr_metadata.v3._common import ( + ChunkGridField, + ChunkKeyEncodingField, + CodecField, + DataTypeField, + StorageTransformerField, +) ZarrV3ExtensionField: TypeAlias = JSONValue """The JSON value of an unknown top-level v3 metadata field. @@ -21,21 +28,24 @@ class ZarrV3ArrayMetadataJSON(TypedDict, extra_items=ZarrV3ExtensionField): """ Zarr v3 array metadata document (the `zarr.json` content for an array). - Extra keys may contain arbitrary JSON values. + Extra keys may contain arbitrary JSON values. Each extension point is + annotated with the field alias of its kind -- `data_type` a + `DataTypeField`, each of `codecs` a `CodecField` -- which is the JSON + a metadata field is, and says what a scope reads it as. See https://zarr-specs.readthedocs.io/en/latest/v3/core/index.html#array-metadata """ zarr_format: Literal[3] node_type: Literal["array"] - data_type: ZarrV3MetadataFieldJSON - shape: tuple[int, ...] - chunk_grid: ZarrV3MetadataFieldJSON - chunk_key_encoding: ZarrV3MetadataFieldJSON + data_type: DataTypeField + shape: tuple[Annotated[int, Ge(0)], ...] + chunk_grid: ChunkGridField + chunk_key_encoding: ChunkKeyEncodingField fill_value: JSONValue - codecs: tuple[ZarrV3MetadataFieldJSON, ...] + codecs: tuple[CodecField, ...] attributes: NotRequired[Mapping[str, JSONValue]] - storage_transformers: NotRequired[tuple[ZarrV3MetadataFieldJSON, ...]] + storage_transformers: NotRequired[tuple[StorageTransformerField, ...]] dimension_names: NotRequired[tuple[str | None, ...]] @@ -64,14 +74,14 @@ class ZarrV3ArrayMetadataJSONPartial(TypedDict, total=False, extra_items=ZarrV3E zarr_format: Literal[3] node_type: Literal["array"] - data_type: ZarrV3MetadataFieldJSON - shape: tuple[int, ...] - chunk_grid: ZarrV3MetadataFieldJSON - chunk_key_encoding: ZarrV3MetadataFieldJSON + data_type: DataTypeField + shape: tuple[Annotated[int, Ge(0)], ...] + chunk_grid: ChunkGridField + chunk_key_encoding: ChunkKeyEncodingField fill_value: JSONValue - codecs: tuple[ZarrV3MetadataFieldJSON, ...] + codecs: tuple[CodecField, ...] attributes: NotRequired[Mapping[str, JSONValue]] - storage_transformers: NotRequired[tuple[ZarrV3MetadataFieldJSON, ...]] + storage_transformers: NotRequired[tuple[StorageTransformerField, ...]] dimension_names: NotRequired[tuple[str | None, ...]] diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/definition.py b/packages/zarr-metadata/src/zarr_metadata/v3/definition.py index ccd233b56b..65e766a55b 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/definition.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/definition.py @@ -268,6 +268,21 @@ def acme_lz4_rules( again, and gives what `canonicalize` gives: None for a field with a problem. +**JSON Schema.** `field_json_schema(CodecDefinition, SCOPE)` writes the +fields of one kind a scope reads as a JSON Schema, draft 2020-12, for a +validator in another language or an editor: each definition's field -- +its name, its configuration as its TypedDict says, bounds and all, a +`must_understand` of `true`, and its bare name when it needs no +configuration -- and a name nothing in scope claims, with any +configuration. A field a configuration holds is written in the same +scope, and a member taking codecs of static size only takes those. The +rules are not in it, so a field it accepts may still have a problem; +one `resolve` reads without a problem, it accepts, as JSON: arrays as +lists, as a parser gives them. Each configuration +TypedDict, and each field alias, is written once, in `$defs`, under its +name. `node_metadata_json_schema_v3`, in `zarr_metadata.model`, writes a +whole `zarr.json`, its fill value held to its data type's. + A definition checks itself when it is built, and each of these is a `TypeError` saying what is wrong: a `configuration` that is not a TypedDict, says nothing of the keys it does not declare, or has a member @@ -275,31 +290,36 @@ def acme_lz4_rules( that is not a string; a member declared as a function -- `rules`, `canonical`, `fill_value_rules`, `storage`, `shape_rules`, `chunk_lengths`, `chunk_rules`, `transition`, `pipelines` -- that is not -one; a data type's `fill_value` no checker reads; a codec `kind` that is -not one of the three, or a `size` that is not `"static"` or `"dynamic"`; -a function no codec of its kind is asked -- chunk rules or pipelines of -a bytes -> bytes codec, which is handed bytes, or a `transition` of a -codec that hands on bytes; a data type named as raw bits of one size are -written. A scope refuses a definition of no kind. Nothing happens at -class creation. +one; a data type's `fill_value` no checker reads, or one holding a +metadata field, which a value of the data type never is; a codec `kind` +that is not one of the three, or a `size` that is not `"static"` or +`"dynamic"`; a function no codec of its kind is asked -- chunk rules or +pipelines of a bytes -> bytes codec, which is handed bytes, or a +`transition` of a codec that hands on bytes; a data type named as raw +bits of one size are written. A scope refuses a definition of no kind. +Nothing happens at class creation. """ from zarr_metadata._common import JSONValue from zarr_metadata._json import MetadataValidationError, ProblemKind, ValidationProblem, shown from zarr_metadata._typed_json import Loc, check -from zarr_metadata.v3._common import ZarrV3MetadataFieldJSON +from zarr_metadata.v3._common import ( + ChunkGridField, + ChunkKeyEncodingField, + CodecField, + DataTypeField, + StaticCodecField, + StorageTransformerField, + ZarrV3MetadataFieldJSON, +) from zarr_metadata.v3._definition import ( Chunk, ChunkGridDefinition, - ChunkGridField, ChunkKeyEncodingDefinition, - ChunkKeyEncodingField, CodecDefinition, - CodecField, CodecKind, CodecSize, DataTypeDefinition, - DataTypeField, Definition, EmptyConfiguration, Lengths, @@ -307,15 +327,14 @@ class creation. Read, Refused, Resolved, - StaticCodecField, StorageClass, StorageTransformerDefinition, - StorageTransformerField, Unclaimed, canonical_of, canonicalize, chunk_grid_lengths, configuration_of, + field_json_schema, fields_of, fill_value_problems, resolve, @@ -364,6 +383,7 @@ class creation. "check", "chunk_grid_lengths", "configuration_of", + "field_json_schema", "fields_of", "fill_value_problems", "read_pipeline", diff --git a/packages/zarr-metadata/tests/test_json_schema.py b/packages/zarr-metadata/tests/test_json_schema.py new file mode 100644 index 0000000000..d2baa4b225 --- /dev/null +++ b/packages/zarr-metadata/tests/test_json_schema.py @@ -0,0 +1,899 @@ +"""JSON Schemas of what the package reads: a TypedDict, a field in a scope, a `zarr.json`. + +A schema is held to the reader it is written from: what the reader finds +nothing wrong with, the schema accepts, and what the type says -- a +bound, a key a closed TypedDict does not declare, a name a scope claims +-- the schema says too. What only a rule says, the schema does not. +""" + +from __future__ import annotations + +import json +import types +from collections.abc import Callable, Mapping +from typing import Annotated, Any, Literal, NewType, NotRequired, cast + +import pytest +from annotated_types import Ge, Gt, Interval, Le, Lt, MinLen +from hypothesis import HealthCheck, event, given, settings +from hypothesis import strategies as st +from jsonschema import Draft202012Validator +from typing_extensions import Doc, TypeAliasType, TypedDict + +from tests.v3.test_every_definition import CASES, KINDS +from zarr_metadata._common import JSONValue +from zarr_metadata._typed_json import Schemas +from zarr_metadata.model import node_metadata_json_schema_v3, validate_node_metadata_v3 +from zarr_metadata.typed_json import check, json_schema +from zarr_metadata.v3.codec.gzip import GzipCodecConfiguration +from zarr_metadata.v3.definition import ( + CORE, + CORE_AND_EXTENSIONS, + ChunkGridDefinition, + ChunkKeyEncodingDefinition, + CodecDefinition, + Context, + DataTypeDefinition, + Definition, + StorageTransformerDefinition, + field_json_schema, + resolve, +) + +DIALECT = "https://json-schema.org/draft/2020-12/schema" + +# Built at run time, where a type checker reads each special form's +# arguments as it would in an annotation. +_annotated: Any = Annotated +_new_type: Callable[[str, object], object] = cast("Any", NewType) +_alias_type: Callable[[str, object], object] = cast("Any", TypeAliasType) + +Level = TypeAliasType("Level", Annotated[int, Interval(ge=0, le=9)]) +Tree = TypeAliasType("Tree", "int | tuple[Tree, ...]") +Name = NewType("Name", str) +Count = _new_type("Count", Annotated[int, Ge(0)]) + + +class Inner(TypedDict, closed=True): + x: int + + +class Open(TypedDict, closed=False): + x: NotRequired[int] + + +class Extra(TypedDict, extra_items=int): + x: str + + +class ExtraJSON(TypedDict, extra_items=JSONValue): + x: str + + +class Node(TypedDict, closed=True): + children: tuple[Node, ...] + + +def _closed(name: str, annotations: dict[str, object]) -> type: + """A closed TypedDict of these required keys, made here, so its annotations resolve in this module.""" + base: object = TypedDict + namespace = {"__annotations__": annotations, "__module__": __name__} + return types.new_class(name, (base,), {"closed": True}, lambda body: body.update(namespace)) + + +def _holding(annotation: object) -> type: + """A closed TypedDict of one required key, `value`, holding `annotation`.""" + return _closed("Holder", {"value": annotation}) + + +# Another class of the name `Inner`, which the schema tells from the first. +Both = _closed("Both", {"first": Inner, "second": _closed("Inner", {"y": str})}) + + +def _held(member: dict[str, Any], defs: dict[str, Any] | None = None) -> dict[str, Any]: + """The schema of `_holding` an annotation whose schema is `member`.""" + schema: dict[str, Any] = { + "$schema": DIALECT, + "type": "object", + "properties": {"value": member}, + "required": ["value"], + "additionalProperties": False, + } + return schema if defs is None else {**schema, "$defs": defs} + + +INNER = { + "type": "object", + "properties": {"x": {"type": "integer"}}, + "required": ["x"], + "additionalProperties": False, +} + +SHAPES: list[tuple[str, type, dict[str, Any]]] = [ + ("int", _holding(int), _held({"type": "integer"})), + ("float", _holding(float), _held({"type": "number"})), + ("bool", _holding(bool), _held({"type": "boolean"})), + ("str", _holding(str), _held({"type": "string"})), + ("null", _holding(None), _held({"type": "null"})), + ("json", _holding(JSONValue), _held({})), + ("one-value", _holding(Literal["a"]), _held({"const": "a"})), + # Sorted as the checker sorts them, so the order written does not show. + ("values", _holding(Literal["b", "a", 1]), _held({"enum": ["a", "b", 1]})), + ("array", _holding(tuple[int, ...]), _held({"type": "array", "items": {"type": "integer"}})), + ("array-of-json", _holding(tuple[JSONValue, ...]), _held({"type": "array"})), + ( + "pair", + _holding(tuple[int, str]), + _held( + { + "type": "array", + "prefixItems": [{"type": "integer"}, {"type": "string"}], + "items": False, + "minItems": 2, + } + ), + ), + ("empty-array", _holding(tuple[()]), _held({"type": "array", "maxItems": 0})), + ( + "union", + _holding(int | None), + _held({"anyOf": [{"type": "integer"}, {"type": "null"}]}), + ), + ( + "mapping", + _holding(Mapping[str, int]), + _held({"type": "object", "additionalProperties": {"type": "integer"}}), + ), + ("mapping-of-json", _holding(Mapping[str, JSONValue]), _held({"type": "object"})), + ("newtype", _holding(Name), _held({"type": "string"})), + ( + "interval", + _holding(Annotated[int, Interval(ge=0, le=9)]), + _held({"type": "integer", "minimum": 0, "maximum": 9}), + ), + ( + "exclusive-bounds", + _holding(Annotated[float, Gt(0), Lt(1)]), + _held({"type": "number", "exclusiveMinimum": 0, "exclusiveMaximum": 1}), + ), + ( + "doc", + _holding(Annotated[int, Ge(0), Doc("a count")]), + _held({"type": "integer", "description": "a count", "minimum": 0}), + ), + # A note that is no `Doc` says nothing a schema writes. + ("note", _holding(Annotated[str, "a note"]), _held({"type": "string"})), + # A value is held to its type's bounds and to its `NewType`'s: of two + # of one keyword, the stricter. + ( + "bounds-on-a-bounded-newtype", + _holding(_annotated[Count, Ge(-5), Lt(9)]), + _held({"type": "integer", "minimum": 0, "exclusiveMaximum": 9}), + ), + ( + "alias", + _holding(Level), + _held( + {"$ref": "#/$defs/Level"}, + {"Level": {"type": "integer", "minimum": 0, "maximum": 9}}, + ), + ), + ( + "alias-holding-itself", + _holding(Tree), + _held( + {"$ref": "#/$defs/Tree"}, + { + "Tree": { + "anyOf": [ + {"type": "integer"}, + {"type": "array", "items": {"$ref": "#/$defs/Tree"}}, + ] + } + }, + ), + ), + ("typeddict", _holding(Inner), _held({"$ref": "#/$defs/Inner"}, {"Inner": INNER})), + ( + "open", + Open, + {"$schema": DIALECT, "type": "object", "properties": {"x": {"type": "integer"}}}, + ), + ( + "extra-items", + Extra, + { + "$schema": DIALECT, + "type": "object", + "properties": {"x": {"type": "string"}}, + "required": ["x"], + "additionalProperties": {"type": "integer"}, + }, + ), + ( + "extra-items-of-json", + ExtraJSON, + { + "$schema": DIALECT, + "type": "object", + "properties": {"x": {"type": "string"}}, + "required": ["x"], + }, + ), + # A TypedDict that holds itself is referred to, not written in place. + ( + "holding-itself", + Node, + { + "$schema": DIALECT, + "$ref": "#/$defs/Node", + "$defs": { + "Node": { + "type": "object", + "properties": { + "children": {"type": "array", "items": {"$ref": "#/$defs/Node"}} + }, + "required": ["children"], + "additionalProperties": False, + } + }, + }, + ), + ( + "one-name-two-classes", + Both, + { + "$schema": DIALECT, + "type": "object", + "properties": { + "first": {"$ref": "#/$defs/Inner"}, + "second": {"$ref": "#/$defs/Inner2"}, + }, + "required": ["first", "second"], + "additionalProperties": False, + "$defs": { + "Inner": INNER, + "Inner2": { + "type": "object", + "properties": {"y": {"type": "string"}}, + "required": ["y"], + "additionalProperties": False, + }, + }, + }, + ), + ( + "a-definition-s-configuration", + GzipCodecConfiguration, + { + "$schema": DIALECT, + "type": "object", + "properties": {"level": {"type": "integer", "minimum": 0, "maximum": 9}}, + "required": ["level"], + "additionalProperties": False, + }, + ), +] + + +@pytest.mark.parametrize( + ("shape", "expected"), [case[1:] for case in SHAPES], ids=[case[0] for case in SHAPES] +) +def test_json_schema_writes_what_check_reads(shape: type, expected: dict[str, Any]) -> None: + schema = json_schema(shape) + assert schema == expected + Draft202012Validator.check_schema(schema) + assert json.loads(json.dumps(schema)) == schema + + +def test_json_schema_refuses_what_is_not_a_typeddict() -> None: + with pytest.raises(TypeError, match="is not a TypedDict"): + json_schema(int) + + +def test_json_schema_refuses_what_check_cannot_read() -> None: + unread = _holding(bytes) + with pytest.raises(TypeError) as raised: + check({}, unread) + with pytest.raises(TypeError, match=str(raised.value)): + json_schema(unread) + + +def test_error_a_schema_that_fails_to_write_leaves_nothing_behind() -> None: + # Reserved before it is written, so that it can refer to itself, and + # given up when the writing fails, so that nothing refers to an empty + # schema, which takes anything. + schemas = Schemas() + unread = _closed("Unread", {"value": bytes}) + with pytest.raises(TypeError, match="is not a shape JSON takes"): + schemas.of(unread) + assert schemas.document({}) == {"$schema": DIALECT} + with pytest.raises(TypeError, match="is not a shape JSON takes"): + schemas.of(unread) + + +def test_json_schema_refuses_metadata_check_does_not_hold_a_value_to() -> None: + with pytest.raises(TypeError, match="is not a constraint the checker reads"): + json_schema(_holding(Annotated[tuple[int, ...], MinLen(1)])) + + +_PROPERTY = settings(max_examples=300, deadline=None, suppress_health_check=[HealthCheck.too_slow]) + +_MARKERS: dict[str, Callable[[int], object]] = {"ge": Ge, "gt": Gt, "le": Le, "lt": Lt} +_LOW = st.none() | st.tuples(st.sampled_from(("ge", "gt")), st.integers(-3, 3)) +_HIGH = st.none() | st.tuples(st.sampled_from(("le", "lt")), st.integers(-3, 3)) +_LAYERS = st.lists( + st.tuples(st.sampled_from(("annotated", "newtype", "alias")), _LOW, _HIGH), + min_size=1, + max_size=3, +) +_NUMBERS = st.integers(-5, 5) | st.floats(-5, 5, allow_nan=False) | st.sampled_from((True, "1")) + + +def _integral_as_int(value: object) -> object: + """`value` as JSON Schema reads a number: `1.0` is the integer 1.""" + return int(value) if isinstance(value, float) and value.is_integer() else value + + +@_PROPERTY +@given(base=st.sampled_from((int, float)), layers=_LAYERS, values=st.lists(_NUMBERS, max_size=8)) +def test_bounds_in_layers_are_written_as_check_holds_a_value_to_them( + base: type, layers: list[tuple[str, object, object]], values: list[object] +) -> None: + # A number's bounds, on it, on a `NewType` of it, on an alias of it, + # layer on layer: the schema accepts what `check` does, a number with no + # fraction read as the integer it equals. + annotation: object = base + for index, (how, low, high) in enumerate(layers): + sides = [cast("tuple[str, int]", side) for side in (low, high) if side is not None] + bounds = [_MARKERS[name](bound) for name, bound in sides] + bounded = _annotated[(annotation, *bounds)] if len(bounds) != 0 else annotation + if how == "newtype": + annotation = _new_type(f"Bounded{index}", bounded) + elif how == "alias": + annotation = _alias_type(f"Bounded{index}", bounded) + else: + annotation = bounded + holder = _holding(annotation) + try: + schema = json_schema(holder) + except TypeError: + # A second bound from one side, which neither reads. + event("refused") + with pytest.raises(TypeError): + check({}, holder) + return + validator = Draft202012Validator(schema) + for value in values: + accepted = check({"value": _integral_as_int(value)}, holder)[1] == () + event("accepted" if accepted else "refused a value") + assert validator.is_valid(cast("Any", {"value": value})) == accepted, (value, schema) + + +# --- values changed in one place -------------------------------------------- + +_GONE = object() +_NAMES = ("r16", "r16\n", "r*", "gzip", "bytes", "crc32c", "int8", "regular", "default", "acme.x") +_KEYS = ("name", "configuration", "must_understand", "level", "x") +_SCALARS = ( + st.none() + | st.booleans() + | st.integers(-3, 300) + | st.floats(-3, 3, allow_nan=False) + | st.sampled_from(_NAMES) + | st.text(max_size=3) +) +_VALUES = st.recursive( + _SCALARS, + lambda inner: ( + st.lists(inner, max_size=2) + | st.dictionaries(st.sampled_from(_KEYS) | st.text(max_size=2), inner, max_size=2) + ), + max_leaves=4, +) + + +def _places(value: object, path: tuple[str | int, ...] = ()) -> list[tuple[str | int, ...]]: + """Every place in `value`: itself, and each member or entry inside it, depth first.""" + found = [path] + if isinstance(value, dict): + for key, entry in cast("dict[str, object]", value).items(): + found += _places(entry, (*path, key)) + elif isinstance(value, list): + for index, entry in enumerate(cast("list[object]", value)): + found += _places(entry, (*path, index)) + return found + + +def _at(value: object, path: tuple[str | int, ...]) -> object: + for step in path: + value = cast("Any", value)[step] + return value + + +def _put(value: object, path: tuple[str | int, ...], new: object) -> object: + """`value` with what sits at `path` replaced by `new`, or removed when `new` is `_GONE`.""" + if len(path) == 0: + return new + step, rest = path[0], path[1:] + copy: Any = ( + dict(cast("dict[str, object]", value)) + if isinstance(value, dict) + else list(cast("list[object]", value)) + ) + if len(rest) == 0 and new is _GONE: + del copy[step] + else: + copy[step] = _put(copy[step], rest, new) + return copy + + +@st.composite +def _changed(draw: st.DrawFn, value: object) -> object: + """`value` changed in one or two places: something replaced, dropped or added, a string with a character more, or a field renamed.""" + for _ in range(draw(st.integers(1, 2))): + path = draw(st.sampled_from(_places(value))) + here = _at(value, path) + how = draw(st.sampled_from(("replace", "drop", "add", "extend", "rename"))) + if how == "drop" and len(path) != 0: + value = _put(value, path, _GONE) + elif how == "add" and isinstance(here, dict): + members = cast("dict[str, object]", here) + value = _put(value, path, {**members, draw(st.sampled_from(_KEYS)): draw(_VALUES)}) + elif how == "add" and isinstance(here, list): + value = _put(value, path, [*cast("list[object]", here), draw(_VALUES)]) + elif how == "extend" and isinstance(here, str): + value = _put(value, path, here + draw(st.sampled_from(("\n", " ", "0", "x")))) + elif how == "rename" and isinstance(here, dict) and "name" in here: + value = _put(value, (*path, "name"), draw(st.sampled_from(_NAMES))) + elif how == "rename" and isinstance(here, str): + value = _put(value, path, draw(st.sampled_from(_NAMES))) + else: + value = _put(value, path, draw(_VALUES)) + return value + + +_SCOPES = (CORE, CORE_AND_EXTENSIONS, Context.of()) +_FIELD_BASES: tuple[tuple[str, object], ...] = ( + *CASES, + # Names nothing claims, whose configuration nothing judges. + ("codecs:acme.x", {"name": "acme.x", "configuration": {"x": [1]}}), + ("data_type:acme.t", {"name": "acme.t", "configuration": {"bits": 8}}), + ("chunk_grid:acme.g", {"name": "acme.g", "configuration": {"chunk_shape": [0]}}), +) +_FIELD_SCHEMAS: dict[tuple[type[Definition[Any]], int], Draft202012Validator] = {} + + +def _field_validator(kind: type[Definition[Any]], scope: int) -> Draft202012Validator: + if (kind, scope) not in _FIELD_SCHEMAS: + schema = field_json_schema(kind, _SCOPES[scope]) + _FIELD_SCHEMAS[(kind, scope)] = Draft202012Validator(schema) + return _FIELD_SCHEMAS[(kind, scope)] + + +# --- a field in a scope ----------------------------------------------------- + +_INDEXED = {"chunk_shape": [2], "codecs": ["bytes"]} + +# Each field, the kind and scope it is read in, whether the schema accepts +# it, and whether the scope reads it without a problem. Where the two +# differ, a rule found what the schema cannot say. +FIELDS: list[tuple[str, object, type[Definition[Any]], Context, bool, bool]] = [ + ("object", {"name": "gzip", "configuration": {"level": 5}}, CodecDefinition, CORE, True, True), + ( + "out-of-bounds", + {"name": "gzip", "configuration": {"level": 12}}, + CodecDefinition, + CORE, + False, + False, + ), + ( + "unknown-key", + {"name": "gzip", "configuration": {"level": 5, "window": 15}}, + CodecDefinition, + CORE, + False, + False, + ), + ("bare-name-needing-a-configuration", "gzip", CodecDefinition, CORE, False, False), + ("bare-name", "bytes", CodecDefinition, CORE, True, True), + ( + "understood", + {"name": "gzip", "configuration": {"level": 5}, "must_understand": True}, + CodecDefinition, + CORE, + True, + True, + ), + ( + "not-understood", + {"name": "gzip", "configuration": {"level": 5}, "must_understand": False}, + CodecDefinition, + CORE, + False, + False, + ), + ( + "stray-member", + {"name": "gzip", "configuration": {"level": 5}, "version": 2}, + CodecDefinition, + CORE, + False, + False, + ), + ( + "unclaimed", + {"name": "zfpy", "configuration": {"mode": 4}}, + CodecDefinition, + CORE, + True, + True, + ), + ("unclaimed-bare", "zfpy", CodecDefinition, CORE, True, True), + ( + "unclaimed-not-understood", + {"name": "zfpy", "must_understand": False}, + CodecDefinition, + CORE, + False, + False, + ), + # Nothing in an empty scope claims gzip, so nothing judges it. + ( + "out-of-scope", + {"name": "gzip", "configuration": {"level": 12}}, + CodecDefinition, + Context.of(), + True, + True, + ), + ( + "a-rule", + { + "name": "blosc", + "configuration": {"cname": "lz4", "clevel": 1, "shuffle": "shuffle", "blocksize": 0}, + }, + CodecDefinition, + CORE, + True, + False, + ), + ( + "static-index-codec", + { + "name": "sharding_indexed", + "configuration": {**_INDEXED, "index_codecs": ["bytes", "crc32c"]}, + }, + CodecDefinition, + CORE, + True, + True, + ), + ( + "dynamic-index-codec", + { + "name": "sharding_indexed", + "configuration": { + **_INDEXED, + "index_codecs": [{"name": "gzip", "configuration": {"level": 1}}], + }, + }, + CodecDefinition, + CORE, + False, + False, + ), + ( + "unclaimed-index-codec", + { + "name": "sharding_indexed", + "configuration": {**_INDEXED, "index_codecs": ["bytes", "acme.sum"]}, + }, + CodecDefinition, + CORE, + True, + True, + ), + ( + "inner-codec-out-of-bounds", + { + "name": "sharding_indexed", + "configuration": { + **_INDEXED, + "codecs": ["bytes", {"name": "gzip", "configuration": {"level": 12}}], + "index_codecs": ["bytes"], + }, + }, + CodecDefinition, + CORE, + False, + False, + ), + ("raw-bits", "r16", DataTypeDefinition, CORE, True, True), + ("raw-bits-object", {"name": "r16", "configuration": {}}, DataTypeDefinition, CORE, True, True), + # What a raw-bits name carries is not written beside it. + ( + "raw-bits-configured", + {"name": "r16", "configuration": {"bits": 16}}, + DataTypeDefinition, + CORE, + False, + False, + ), + ("raw-bits-of-a-size-the-spec-refuses", "r12", DataTypeDefinition, CORE, True, False), + ("raw-bits-notation", "r*", DataTypeDefinition, CORE, True, True), + # Matched to the end of the name, as the package matches it, so a + # validator that matches as Python does takes no final newline for it: + # a name nothing claims, which any configuration goes with. + ( + "raw-bits-and-a-newline", + {"name": "r16\n", "configuration": {"x": 1}}, + DataTypeDefinition, + CORE, + True, + True, + ), + ( + "struct-field-out-of-bounds", + { + "name": "struct", + "configuration": { + "fields": [ + { + "name": "t", + "data_type": { + "name": "numpy.datetime64", + "configuration": {"unit": "s", "scale_factor": 0}, + }, + } + ] + }, + }, + DataTypeDefinition, + CORE_AND_EXTENSIONS, + False, + False, + ), + ( + "grid-out-of-bounds", + {"name": "regular", "configuration": {"chunk_shape": [0]}}, + ChunkGridDefinition, + CORE, + False, + False, + ), + ( + "encoding", + {"name": "v2", "configuration": {"separator": "/"}}, + ChunkKeyEncodingDefinition, + CORE, + True, + True, + ), + ("transformer", {"name": "acme.cache"}, StorageTransformerDefinition, CORE, True, True), +] + + +@pytest.mark.parametrize( + ("field", "kind", "scope", "accepted", "clean"), + [case[1:] for case in FIELDS], + ids=[case[0] for case in FIELDS], +) +def test_field_json_schema_says_what_the_scope_reads( + field: object, kind: type[Definition[Any]], scope: Context, accepted: bool, clean: bool +) -> None: + schema = field_json_schema(kind, scope) + Draft202012Validator.check_schema(schema) + assert Draft202012Validator(schema).is_valid(cast("Any", field)) == accepted + assert (resolve(field, kind, scope)[1] == ()) == clean + + +@pytest.mark.parametrize( + ("key", "field"), CASES, ids=[f"{key}:{index}" for index, (key, _) in enumerate(CASES)] +) +def test_every_example_field_is_one_its_schema_accepts(key: str, field: object) -> None: + schema = field_json_schema(KINDS[key.split(":")[0]], CORE_AND_EXTENSIONS) + assert Draft202012Validator(schema).is_valid(cast("Any", field)) + + +@_PROPERTY +@given(st.data()) +def test_a_field_read_without_a_problem_is_one_its_schema_accepts(data: st.DataObject) -> None: + # Each example changed in one place, in one of three scopes: whatever + # the scope still reads without a problem, the schema accepts. + key, example = data.draw(st.sampled_from(_FIELD_BASES), label="example") + kind = KINDS[key.split(":")[0]] + scope = data.draw(st.sampled_from(range(len(_SCOPES))), label="scope") + field = data.draw(_changed(json.loads(json.dumps(example))), label="field") + read = resolve(field, kind, _SCOPES[scope])[1] == () + event("read without a problem" if read else "a problem") + if read: + assert _field_validator(kind, scope).is_valid(cast("Any", field)) + + +def test_field_json_schema_refuses_what_is_not_a_kind() -> None: + with pytest.raises(TypeError, match="is not a kind of metadata"): + field_json_schema(Definition, CORE) + + +# --- a zarr.json ------------------------------------------------------------ + +ARRAY: dict[str, Any] = { + "zarr_format": 3, + "node_type": "array", + "shape": [4, 4], + "data_type": "int8", + "chunk_grid": {"name": "regular", "configuration": {"chunk_shape": [2, 2]}}, + "chunk_key_encoding": {"name": "default"}, + "fill_value": 0, + "codecs": [{"name": "bytes"}, {"name": "gzip", "configuration": {"level": 5}}], +} +FLOATS: dict[str, Any] = { + **ARRAY, + "data_type": "float32", + "codecs": [{"name": "bytes", "configuration": {"endian": "little"}}], +} +GROUP: dict[str, Any] = {"zarr_format": 3, "node_type": "group"} + + +def _consolidated(**metadata: object) -> dict[str, Any]: + return { + **GROUP, + "consolidated_metadata": {"kind": "inline", "must_understand": False, "metadata": metadata}, + } + + +# Each document, whether the schema accepts it, and whether +# `validate_node_metadata_v3` finds nothing wrong with it. +DOCUMENTS: list[tuple[str, dict[str, Any], bool, bool]] = [ + ("array", ARRAY, True, True), + ("group", {**GROUP, "attributes": {"a": [1, None]}}, True, True), + ("fill-value-out-of-range", {**ARRAY, "fill_value": 300}, False, False), + ("fill-value-of-the-wrong-type", {**ARRAY, "fill_value": "0"}, False, False), + ("float-fill-value", {**FLOATS, "fill_value": "NaN"}, True, True), + ("hex-fill-value", {**FLOATS, "fill_value": "0x7fc00000"}, True, True), + # A string that is no float's is a rule's to find. + ("fill-value-a-rule-refuses", {**FLOATS, "fill_value": "nan"}, True, False), + ("unclaimed-data-type", {**ARRAY, "data_type": "acme.int7", "fill_value": "any"}, True, True), + ( + "a-name-raw-bits-and-a-newline", + {**ARRAY, "data_type": "r16\n", "fill_value": "any"}, + True, + True, + ), + ("raw-bits", {**ARRAY, "data_type": "r16", "fill_value": [0, 255]}, True, True), + ( + "raw-bits-byte-out-of-range", + {**ARRAY, "data_type": "r16", "fill_value": [0, 256]}, + False, + False, + ), + ( + "codec-not-understood", + {**ARRAY, "codecs": [{"name": "bytes", "must_understand": False}]}, + False, + False, + ), + ( + "codec-out-of-bounds", + {**ARRAY, "codecs": [{"name": "bytes"}, {"name": "gzip", "configuration": {"level": 12}}]}, + False, + False, + ), + ("negative-shape", {**ARRAY, "shape": [-1, 4]}, False, False), + ("wrong-format", {**ARRAY, "zarr_format": 2}, False, False), + ( + "no-node-type", + {key: value for key, value in ARRAY.items() if key != "node_type"}, + False, + False, + ), + ("an-extension-field", {**ARRAY, "acme": {"must_understand": False}}, True, True), + # What members read together say is the rules'. + ("names-for-another-rank", {**ARRAY, "dimension_names": ["x"]}, True, False), + ( + "codecs-out-of-order", + {**ARRAY, "codecs": [{"name": "gzip", "configuration": {"level": 5}}, "bytes"]}, + True, + False, + ), + ("consolidated", _consolidated(a=ARRAY, b=GROUP), True, True), + ("consolidated-null", {**GROUP, "consolidated_metadata": None}, True, True), + ("consolidated-bad-array", _consolidated(a={**ARRAY, "fill_value": 300}), False, False), + ( + "consolidated-within-consolidated", + _consolidated(b=_consolidated(c={**ARRAY, "fill_value": 300})), + False, + False, + ), + ( + "consolidated-kind", + { + **GROUP, + "consolidated_metadata": {"kind": "sidecar", "must_understand": False, "metadata": {}}, + }, + False, + False, + ), +] + +NODE_SCHEMA = node_metadata_json_schema_v3() + + +@pytest.mark.parametrize( + ("document", "accepted", "clean"), + [case[1:] for case in DOCUMENTS], + ids=[case[0] for case in DOCUMENTS], +) +def test_node_metadata_json_schema_says_what_a_zarr_json_holds( + document: dict[str, Any], accepted: bool, clean: bool +) -> None: + assert Draft202012Validator(NODE_SCHEMA).is_valid(document) == accepted + assert (validate_node_metadata_v3(document) == ()) == clean + + +_BASES: tuple[dict[str, Any], ...] = ( + ARRAY, + FLOATS, + {**ARRAY, "data_type": "r16", "fill_value": [0, 0]}, + { + **ARRAY, + "codecs": [ + { + "name": "sharding_indexed", + "configuration": { + "chunk_shape": [1, 1], + "codecs": ["bytes", {"name": "gzip", "configuration": {"level": 1}}], + "index_codecs": ["bytes", "crc32c"], + }, + } + ], + }, + _consolidated(a=ARRAY, b=GROUP), +) +_NODE_SCHEMAS: dict[int, Draft202012Validator] = {} + + +def _node_validator(scope: int) -> Draft202012Validator: + if scope not in _NODE_SCHEMAS: + _NODE_SCHEMAS[scope] = Draft202012Validator( + node_metadata_json_schema_v3(context=_SCOPES[scope]) + ) + return _NODE_SCHEMAS[scope] + + +@_PROPERTY +@given(st.data()) +def test_a_document_read_without_a_problem_is_one_its_schema_accepts(data: st.DataObject) -> None: + # Each document changed in one place, in one of three scopes: whatever + # `validate_node_metadata_v3` finds nothing wrong with, the schema accepts. + base = data.draw(st.sampled_from(_BASES), label="base") + scope = data.draw(st.sampled_from(range(len(_SCOPES))), label="scope") + document = data.draw(_changed(json.loads(json.dumps(base))), label="document") + read = validate_node_metadata_v3(document, context=_SCOPES[scope]) == () + event("read without a problem" if read else "a problem") + if read: + assert _node_validator(scope).is_valid(cast("Any", document)) + + +@pytest.mark.parametrize( + "scope", + [CORE, CORE_AND_EXTENSIONS, Context.of()], + ids=["core", "core-and-extensions", "empty"], +) +def test_every_schema_is_a_json_schema_written_the_same_each_time(scope: Context) -> None: + schema = node_metadata_json_schema_v3(context=scope) + Draft202012Validator.check_schema(schema) + assert json.loads(json.dumps(schema)) == schema + assert schema == node_metadata_json_schema_v3(context=scope) + defs = cast("dict[str, JSONValue]", schema["$defs"]) + assert {"ZarrV3ArrayMetadataJSON", "ZarrV3GroupMetadataJSON"} <= defs.keys() + for kind in ( + CodecDefinition, + DataTypeDefinition, + ChunkGridDefinition, + ChunkKeyEncodingDefinition, + StorageTransformerDefinition, + ): + Draft202012Validator.check_schema(field_json_schema(kind, scope)) diff --git a/packages/zarr-metadata/tests/test_public_api.py b/packages/zarr-metadata/tests/test_public_api.py index 22fdd0c863..82448e356c 100644 --- a/packages/zarr-metadata/tests/test_public_api.py +++ b/packages/zarr-metadata/tests/test_public_api.py @@ -305,6 +305,7 @@ def test_all_is_grouped_and_unique() -> None: "HexFloat16", "HexFloat32", "HexFloat64", + "JSONSchema", "JSONValue", "MetadataValidationError", "NumpyDatetime64", diff --git a/packages/zarr-metadata/tests/test_typed_json_properties.py b/packages/zarr-metadata/tests/test_typed_json_properties.py index 00a137f560..b228546baf 100644 --- a/packages/zarr-metadata/tests/test_typed_json_properties.py +++ b/packages/zarr-metadata/tests/test_typed_json_properties.py @@ -13,27 +13,31 @@ Values are JSON, their depth capped: drawn from the spec, so they conform; drawn at large, so most do not; or drawn from the spec and changed in one place, so they nearly do. On every one the checker has to -agree with the reference, and the two builds with each other. +agree with the reference, and the two builds with each other; so does +the JSON Schema written of the type, but for JSON Schema's own reading of +a number with no fraction, `1.0`, as the integer it equals. """ from __future__ import annotations import copy import itertools +import json import sys import types import typing from collections.abc import Mapping from dataclasses import dataclass, field -from typing import Annotated, Literal, NewType, NotRequired, Required, TypeAlias, Union, cast +from typing import Annotated, Any, Literal, NewType, NotRequired, Required, TypeAlias, Union, cast import pytest from hypothesis import HealthCheck, event, find, given, note, settings from hypothesis import strategies as st +from jsonschema import Draft202012Validator from typing_extensions import ReadOnly, TypeAliasType, TypedDict from zarr_metadata._common import JSONValue -from zarr_metadata._typed_json import Loc, Parsed, no_leaf, parser, typeddict_keys +from zarr_metadata._typed_json import Loc, Parsed, Schemas, no_leaf, parser, typeddict_keys # --- specs ----------------------------------------------------------------- @@ -664,6 +668,38 @@ def test_the_checker_agrees_with_the_reference(data: st.DataObject) -> None: assert conforms(typed, spec) +def _as_json_schema_reads(value: object) -> object: + """`value` as JSON Schema reads it: a number with no fraction is an integer, `1.0` the `1` it equals.""" + if isinstance(value, float) and value.is_integer(): + return int(value) + if isinstance(value, list): + return [_as_json_schema_reads(entry) for entry in cast("list[object]", value)] + if isinstance(value, dict): + entries = cast("dict[str, object]", value) + return {key: _as_json_schema_reads(entry) for key, entry in entries.items()} + return value + + +@_EXAMPLES +@given(st.data()) +def test_the_json_schema_agrees_with_the_reference(data: st.DataObject) -> None: + # The schema of a type accepts exactly the values of it, as the checker + # does, but that JSON Schema takes `1.0` for the integer it equals, which + # the checker does not. + spec = data.draw(specs(_DEPTH), label="spec") + build = Build(data.draw(st.sampled_from(_MODES), label="mode")) + annotation = build.annotation(spec) + note(build.text_of()) + schemas = Schemas() + schema = schemas.document(schemas.of(annotation)) + note(json.dumps(schema, indent=1)) + Draft202012Validator.check_schema(schema) + value = data.draw(_values(spec), label="value") + expected = conforms(_as_json_schema_reads(value), spec) + event("conforms" if expected else "does not conform") + assert Draft202012Validator(schema).is_valid(cast("Any", value)) == expected + + @_EXAMPLES @given(st.data()) def test_a_postponed_typeddict_reads_as_an_evaluated_one(data: st.DataObject) -> None: diff --git a/packages/zarr-metadata/tests/v3/test_definitions.py b/packages/zarr-metadata/tests/v3/test_definitions.py index 2225b4d522..a262f152d8 100644 --- a/packages/zarr-metadata/tests/v3/test_definitions.py +++ b/packages/zarr-metadata/tests/v3/test_definitions.py @@ -856,17 +856,26 @@ def test_check_needs_nothing_but_the_value_and_a_typeddict() -> None: def test_check_reads_a_whole_array_document() -> None: - # A null in `dimension_names`, and a top-level key the document does - # not declare, typed by its `extra_items`. + # A null in `dimension_names`, a top-level key the document does not + # declare, typed by its `extra_items`, and a dimension of length 0. document = { - **ZarrV3ArrayMetadata.create_default(shape=(4,)).to_json(), - "dimension_names": [None], + **ZarrV3ArrayMetadata.create_default(shape=(0, 4)).to_json(), + "dimension_names": [None, "x"], "acme": {"must_understand": False}, } typed, problems = check(document, ZarrV3ArrayMetadataJSON) assert problems == () assert typed is not None - assert typed.get("dimension_names") == (None,) + assert typed.get("dimension_names") == (None, "x") + + +def test_error_check_holds_an_array_document_s_shape_to_its_bound() -> None: + document = {**ZarrV3ArrayMetadata.create_default(shape=(4,)).to_json(), "shape": [-1]} + typed, problems = check(document, ZarrV3ArrayMetadataJSON) + assert typed is None + assert [(found.loc, found.kind, dict(found.ctx)) for found in problems] == [ + (("shape", 0), "invalid_value", {"ge": 0}) + ] def test_configuration_of_types_what_its_definition_read() -> None: @@ -1119,6 +1128,12 @@ def test_error_a_data_type_fill_value_no_checker_reads() -> None: DataTypeDefinition(name="acme.set", configuration=Empty, fill_value=set[int]) +def test_error_a_data_type_fill_value_holding_a_metadata_field() -> None: + # A value of the data type, which no scope reads as a field. + with pytest.raises(TypeError, match="'acme.f': fill_value: CodecField holds a metadata field"): + DataTypeDefinition(name="acme.f", configuration=Empty, fill_value=tuple[CodecField, ...]) + + @pytest.mark.parametrize( ("kind", "member"), [ From d6673989725aaaf9846363a21a992af323d0d6d5 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Tue, 29 Sep 2026 14:41:13 +0200 Subject: [PATCH 10/94] feat(zarr-metadata)!: a model checks itself when it is built, and to_key_value writes it as it is As pydantic validates in `__init__`: a v3 or v2 model's constructor reads its document by its own fields, the read the writer made before, raises `MetadataValidationError` with every problem, an extra field named as a member among them, and holds its members as that read refines them, in containers of its own, as pydantic holds what its `__init__` coerced. So a model built by hand, or changed by `dataclasses.replace` or a v2 `update`, is refused at the change rather than at the write; none is built invalid, none shares a container with what it was built of, and each writes as JSON. Each field, a `Read` or an `Unclaimed`, is held as the scope read it. The reads build their models with a private `construct`, as pydantic's `model_construct` builds one, so a model a read builds is not read twice, and `to_key_value` serializes without reading again. A group checks its own members, since each document its consolidated metadata holds is a model that checked itself; a consolidated path that is not a string is a `TypeError`. `ZarrV2ArrayMetadata.create_default(chunks=...)` without a shape they fit is refused, as the v3 model refuses a grid its shape does not take. The clauses about the checking writer that this makes untrue are dropped from 373's and 4420's fragments. BREAKING CHANGE: building or replacing a model into one whose document has a problem raises, and a model holds the containers a read would, not the ones it was given; a container changed in place is not checked again. Assisted-by: ClaudeCode:claude-opus-5-5 --- packages/zarr-metadata/README.md | 12 + .../zarr-metadata/changes/373.feature.2.md | 12 +- packages/zarr-metadata/changes/373.removal.md | 3 +- packages/zarr-metadata/changes/379.feature.md | 16 ++ packages/zarr-metadata/changes/4420.bugfix.md | 6 +- packages/zarr-metadata/docs/index.md | 12 + .../src/zarr_metadata/model/__init__.py | 5 +- .../src/zarr_metadata/model/_array.py | 164 +++++++----- .../src/zarr_metadata/model/_group.py | 235 +++++++++------- .../src/zarr_metadata/model/_validation.py | 40 +++ .../zarr-metadata/tests/model/test_array.py | 38 --- .../tests/model/test_construction.py | 252 ++++++++++++++++++ .../zarr-metadata/tests/model/test_group.py | 16 -- 13 files changed, 586 insertions(+), 225 deletions(-) create mode 100644 packages/zarr-metadata/changes/379.feature.md create mode 100644 packages/zarr-metadata/tests/model/test_construction.py diff --git a/packages/zarr-metadata/README.md b/packages/zarr-metadata/README.md index d91a85a443..ed4a4783cd 100644 --- a/packages/zarr-metadata/README.md +++ b/packages/zarr-metadata/README.md @@ -136,6 +136,18 @@ build the model of either kind, as the models' own `from_json` and A member the spec does not define is not a field; the model's `must_understand_fields` names those a reader must understand. +A model checks itself when it is built, as a pydantic model does in +`__init__`: one built by hand, or changed by `dataclasses.replace`, whose +document has a problem raises `MetadataValidationError` with every +problem at the change, so no model is built invalid, and `to_key_value` +writes each as it is. It holds its members as that read refines them, in +containers of its own -- a list given for an array as a tuple -- as +pydantic holds what its `__init__` coerced, and each field, a `Read` or +an `Unclaimed`, as the scope read it: one built by hand is taken as +read. A model a read builds is not read a second time. Change a model by +building another: a container it holds, changed in place, is not +checked again. + `node_metadata_json_schema_v3` writes what the validators read as a JSON Schema, draft 2020-12, for an editor that checks a `zarr.json` as it is written, or a validator in another language. Each extension point is diff --git a/packages/zarr-metadata/changes/373.feature.2.md b/packages/zarr-metadata/changes/373.feature.2.md index efb6e69fba..29983eef86 100644 --- a/packages/zarr-metadata/changes/373.feature.2.md +++ b/packages/zarr-metadata/changes/373.feature.2.md @@ -3,10 +3,8 @@ scope, as a pydantic model holds no validation context: `update(context=..., **members)` reads only the members it is given, as JSON, in the scope it is given, keeps each field the model holds as it was read, and leaves out a member given as `UNSET`; a group keeps the -documents its consolidated metadata holds unless it is given others. -`to_key_value` reads the document it writes by the model's own fields, -and refuses one with a problem, so a model changed by hand is never -written invalid. The pydantic field types read in the scope their -validation context holds: the context itself, or its -`zarr_metadata_context` item. `create_default`'s codec is `bytes` with a -little `endian`, so the default of any data type of fixed size is valid. +documents its consolidated metadata holds unless it is given others. The +pydantic field types read in the scope their validation context holds: +the context itself, or its `zarr_metadata_context` item. +`create_default`'s codec is `bytes` with a little `endian`, so the +default of any data type of fixed size is valid. diff --git a/packages/zarr-metadata/changes/373.removal.md b/packages/zarr-metadata/changes/373.removal.md index 3eab927630..5ecdd43f76 100644 --- a/packages/zarr-metadata/changes/373.removal.md +++ b/packages/zarr-metadata/changes/373.removal.md @@ -7,5 +7,4 @@ members of the document as JSON, as `ZarrV3ArrayMetadataUpdate` and them. `Resolved` is no longer a class with a `resolution`: match on its variant, each built with keywords. `update` takes the scope it reads in, `context`, which has no default. `to_key_value` no longer takes a -`context`: it reads the document it writes by the model's own fields, -and refuses one with a problem. +`context`. diff --git a/packages/zarr-metadata/changes/379.feature.md b/packages/zarr-metadata/changes/379.feature.md new file mode 100644 index 0000000000..941dbd5310 --- /dev/null +++ b/packages/zarr-metadata/changes/379.feature.md @@ -0,0 +1,16 @@ +**Breaking:** a model checks itself when it is built, as pydantic's +`__init__` does, and `to_key_value` writes it as it is. A model built by +hand, or changed by `dataclasses.replace` or a v2 model's `update`, whose +document has a problem raises `MetadataValidationError` with every +problem at the change, where it was refused only when written, so no +model is built invalid; `ZarrV2ArrayMetadata.create_default` refuses +`chunks` its default shape does not take, as the v3 model refuses a +grid. A model holds its members as that read refines them, in containers +of its own, as pydantic holds what its `__init__` coerced: a list given +for an array is held as a tuple, and a dict given is not the dict held. +Each field, a `Read` or an `Unclaimed`, is held as the scope read it. A +model a read builds is not read a second time, and writing one reads +nothing again. Change a model by building another: a container it +holds, changed in place, is not checked again, as a pydantic model's is +not. A path in `ZarrV3ConsolidatedMetadata.metadata` that is not a string +is a `TypeError`. diff --git a/packages/zarr-metadata/changes/4420.bugfix.md b/packages/zarr-metadata/changes/4420.bugfix.md index a77d08aa32..11c7984099 100644 --- a/packages/zarr-metadata/changes/4420.bugfix.md +++ b/packages/zarr-metadata/changes/4420.bugfix.md @@ -14,10 +14,8 @@ failing to decode the document. `is_json`, `validate_json` and `parse_json` judge by RFC 8259 alone, so a document the node guards accept can hold a value `is_json` refuses. -`to_key_value` validates the document it writes as `from_key_value` -validates what it reads, and raises `MetadataValidationError` with every -problem instead of writing a document the reader refuses. A model built -by hand is not validated, so the writer is where these are caught: +A model whose document the reader refuses raises +`MetadataValidationError` with every problem, instead of being written: - a non-finite number outside attributes, which was refused before too; - a non-string attribute key, which was written as a string; diff --git a/packages/zarr-metadata/docs/index.md b/packages/zarr-metadata/docs/index.md index 627b37ed56..70608414dd 100644 --- a/packages/zarr-metadata/docs/index.md +++ b/packages/zarr-metadata/docs/index.md @@ -151,6 +151,18 @@ build the model of either kind, as the models' own `from_json` and A member the spec does not define is not a field; the model's `must_understand_fields` names those a reader must understand. +A model checks itself when it is built, as a pydantic model does in +`__init__`: one built by hand, or changed by `dataclasses.replace`, whose +document has a problem raises `MetadataValidationError` with every +problem at the change, so no model is built invalid, and `to_key_value` +writes each as it is. It holds its members as that read refines them, in +containers of its own -- a list given for an array as a tuple -- as +pydantic holds what its `__init__` coerced, and each field, a `Read` or +an `Unclaimed`, as the scope read it: one built by hand is taken as +read. A model a read builds is not read a second time. Change a model by +building another: a container it holds, changed in place, is not +checked again. + `node_metadata_json_schema_v3` writes what the validators read as a JSON Schema, draft 2020-12, for an editor that checks a `zarr.json` as it is written, or a validator in another language. Each extension point is diff --git a/packages/zarr-metadata/src/zarr_metadata/model/__init__.py b/packages/zarr-metadata/src/zarr_metadata/model/__init__.py index 0b4ec7344f..ad74f9f072 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/__init__.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/__init__.py @@ -18,7 +18,10 @@ `from_key_value` constructors raise `MetadataValidationError` for every ingestion failure, including missing store keys and undecodable bytes, and the v3 ones take the same -`context`. +`context`. A model checks itself when it is built, as a pydantic model +does in `__init__`, so one built by hand, or changed by +`dataclasses.replace`, is refused at the change when its document has a +problem, and `to_key_value` writes a model as it is. """ from zarr_metadata._json import ( diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_array.py b/packages/zarr-metadata/src/zarr_metadata/model/_array.py index 4cbc93c597..20f743471a 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_array.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/_array.py @@ -23,9 +23,11 @@ ArrayMembersV3, StoreKey, ZarrV3ArrayMetadataReading, + construct, dimension_lengths, dump_store_json, load_store_json, + overlapping, parse_array_metadata_v2, read_array_v3, ) @@ -114,13 +116,21 @@ class ZarrV3ArrayMetadata: A model holds no scope: each field keeps the definition that read it, and a scope is asked only to read new JSON -- by `from_json`, - `create_default`, and `update`, which each take one. `to_key_value` - reads the document it writes by the model's own fields, and refuses - one with a problem, so a model changed by hand, as - `dataclasses.replace` changes one, is never written invalid. `to_json` - writes each extension point as its readers take it, as `Read.to_json` - says. A model pickles when the definitions its fields hold do: ones - whose functions are defined at a module's top level. + `create_default`, and `update`, which each take one. A model checks + itself when it is built, as pydantic's `__init__` does: its document, + as its own fields read it, has no problem, or the constructor raises + `MetadataValidationError` with every one. So a model built by hand, or + changed as `dataclasses.replace` changes one, is refused at the + change, and none is built invalid. It holds its members as that read + refines them, in containers of its own -- a list given for an array + as a tuple -- as pydantic holds what its `__init__` coerced, and each + field as the scope read it: a field built by hand is taken as read. A + model a read builds is not read a second time. Change a model by + building another: a container it holds, changed in place, is not + checked again. `to_json` writes each extension point as its + readers take it, as `Read.to_json` says. A model pickles when the + definitions its fields hold do: ones whose functions are defined at a + module's top level. """ zarr_format: Literal[3] = field(default=3, init=False) @@ -192,19 +202,13 @@ def update( return type(self).from_json(document, context=context) def __post_init__(self) -> None: - overlap = set(self.extra_fields.keys()).intersection(ARRAY_METADATA_STANDARD_KEYS_V3) - if overlap: - raise MetadataValidationError( - [ - ValidationProblem( - ("extra_fields",), - "Extra fields cannot overlap with standard Zarr V3 array metadata fields", - "invalid_value", - ) - ] - ) - # The runtime half of the annotations: each extension point a field - # read as its kind, read or unclaimed, as a read gives it. + # The runtime half of the annotations: extra fields by name, and each + # extension point a field read as its kind, read or unclaimed, as a + # read gives it. + extra = cast("object", self.extra_fields) + if not isinstance(extra, Mapping): + msg = f"extra_fields: expected a mapping of names to JSON, got {extra!r}" + raise TypeError(msg) for key, kind, nodes in ( ("data_type", DataTypeDefinition, (self.data_type,)), ("chunk_grid", ChunkGridDefinition, (self.chunk_grid,)), @@ -216,6 +220,16 @@ def __post_init__(self) -> None: if not isinstance(node, (Read, Unclaimed)) or node.read_as is not kind: msg = f"{key}: expected a field read as a {kind.__name__}, got {node!r}" raise TypeError(msg) + # The rest held as the read of its document refines it, in containers + # of its own, as pydantic holds what its `__init__` coerced. + members = _members(self) + object.__setattr__(self, "shape", members.shape) + object.__setattr__(self, "fill_value", members.fill_value) + object.__setattr__(self, "dimension_names", members.dimension_names) + object.__setattr__(self, "attributes", members.attributes) + object.__setattr__(self, "extra_fields", members.extra_fields) + object.__setattr__(self, "codecs", tuple(self.codecs)) + object.__setattr__(self, "storage_transformers", tuple(self.storage_transformers)) def to_json(self) -> ZarrV3ArrayMetadataJSON: """The document as JSON, arrays as tuples, sharing no mutable state with the model. @@ -271,31 +285,36 @@ def to_key_value( ) -> Mapping[ZarrV3ArrayMetadataStoreKey, bytes]: """The document as a store holds it: JSON bytes at `zarr.json`, indented by `indent`. - `MetadataValidationError` when the document has a problem, as the - model's own fields read it, so no invalid document is written, even - of a model changed by hand. `NaN`, `Infinity` and `-Infinity` in - `attributes` are written as those bare tokens, as zarr-python - writes them, which a strict JSON parser refuses. + A model was checked when it was built, so its document is written + as it is. `NaN`, `Infinity` and `-Infinity` in `attributes` are + written as those bare tokens, as zarr-python writes them, which a + strict JSON parser refuses. """ - problems = array_problems(self) - if len(problems) != 0: - raise MetadataValidationError(problems) return {ZARR_V3_ARRAY_METADATA_STORE_KEY: dump_store_json(array_json(self), indent=indent)} -def array_problems(model: ZarrV3ArrayMetadata) -> tuple[ValidationProblem, ...]: - """What is wrong with `model`'s document, as the model's own fields read it: nothing, for a model a read built.""" - return read_array_v3(held_document(model), NO_SCOPE)[0].problems +def _members(model: ZarrV3ArrayMetadata) -> ArrayMembersV3: + """`model`'s members other than its fields, as the read of its document by its own fields refines them; `MetadataValidationError` with every problem that document has.""" + reading, members = read_array_v3(held_document(model), NO_SCOPE) + extra = overlapping(model.extra_fields, ARRAY_METADATA_STANDARD_KEYS_V3, "array") + problems = (*extra, *reading.problems) + if len(problems) != 0: + raise MetadataValidationError(problems) + return cast("ArrayMembersV3", members) def array_json(model: ZarrV3ArrayMetadata) -> ZarrV3ArrayMetadataJSON: """`model`'s document as JSON, holding the model's own values: what `to_key_value` serializes, which changes nothing, and `to_json` copies.""" - return cast("ZarrV3ArrayMetadataJSON", _document(model, document_json)) + return cast("ZarrV3ArrayMetadataJSON", _document(model, document_json, whole=False)) def held_document(model: ZarrV3ArrayMetadata) -> dict[str, object]: - """`model`'s document with each field as it was read, which a read takes as it is: what `update` and `to_key_value` read, reading no field again.""" - return _document(model, _as_read) + """`model`'s document with each field as it was read, which a read takes as it is: what `update` and the constructor read, reading no field again. + + Every member the model holds, an empty one too, so the read judges + each whatever it holds. + """ + return _document(model, _as_read, whole=True) def _as_read(field: Read[Any] | Unclaimed) -> object: @@ -303,9 +322,9 @@ def _as_read(field: Read[Any] | Unclaimed) -> object: def _document( - model: ZarrV3ArrayMetadata, write: Callable[[Read[Any] | Unclaimed], object] + model: ZarrV3ArrayMetadata, write: Callable[[Read[Any] | Unclaimed], object], *, whole: bool ) -> dict[str, object]: - """`model`'s document, each field as `write` gives it, the rest as the model holds it.""" + """`model`'s document, each field as `write` gives it, the rest as the model holds it; empty `attributes` and `storage_transformers` too when `whole`, where a writer leaves them out.""" out: dict[str, object] = { "zarr_format": model.zarr_format, "node_type": model.node_type, @@ -318,13 +337,19 @@ def _document( } if model.dimension_names is not UNSET: out["dimension_names"] = model.dimension_names - if len(model.attributes) > 0: + if whole or len(model.attributes) > 0: out["attributes"] = model.attributes - if len(model.storage_transformers) > 0: + if whole or len(model.storage_transformers) > 0: out["storage_transformers"] = tuple( write(transformer) for transformer in model.storage_transformers ) - out.update(model.extra_fields) + # An extra field named as a member the document declares is no member + # of it, which the constructor reports. + out.update( + (key, value) + for key, value in model.extra_fields.items() + if key not in ARRAY_METADATA_STANDARD_KEYS_V3 + ) return out @@ -351,8 +376,9 @@ def read_array_metadata_v3( def array_model( reading: ZarrV3ArrayMetadataReading, members: ArrayMembersV3 ) -> ZarrV3ArrayMetadata: - """The model of a document its reading found nothing wrong with: its fields as read, and its other members as the read refined them.""" - return ZarrV3ArrayMetadata( + """The model of a document its reading found nothing wrong with: its fields as read, and its other members as the read refined them, not read again.""" + return construct( + ZarrV3ArrayMetadata, shape=members.shape, fill_value=members.fill_value, data_type=cast("Read[DataTypeDefinition[Any]] | Unclaimed", reading.data_type), @@ -410,7 +436,10 @@ class ZarrV2ArrayMetadata: explicit empty `.zattrs`, which is `{}` and round-trips as a file. One spelling normalization: a `.zarray` that omits `dimension_separator` means `"."` by the v2 convention, and the model holds and re-emits that - value explicitly. + value explicitly. A model checks itself when it is built, as the v3 + models do: its document has no problem `validate_array_metadata_v2` + finds, or the constructor raises `MetadataValidationError`, so + `update` refuses a change that would make one. """ zarr_format: Literal[2] = field(default=2, init=False) @@ -428,6 +457,12 @@ class ZarrV2ArrayMetadata: dimension_separator: ZarrV2ArrayDimensionSeparator = field(default=".") attributes: dict[str, JSONValue] | UNSET + def __post_init__(self) -> None: + # Held as a read refines them, in containers of its own. + members = _v2_array_members(parse_array_metadata_v2(self.to_json())) + for name, value in members.items(): + object.__setattr__(self, name, value) + def update(self, **kwargs: Unpack[ZarrV2ArrayMetadataPartial]) -> ZarrV2ArrayMetadata: """ Return a new `ZarrV2ArrayMetadata` with the given fields updated. @@ -435,7 +470,8 @@ def update(self, **kwargs: Unpack[ZarrV2ArrayMetadataPartial]) -> ZarrV2ArrayMet Only the constructor-settable fields listed in `ZarrV2ArrayMetadataPartial` can be updated; the fixed `zarr_format` is rejected at the type level. Each given field fully replaces its previous - value. + value. `MetadataValidationError` when the document the change makes + has a problem, as the model checks itself when it is built. """ return dataclasses.replace(self, **kwargs) @@ -452,8 +488,9 @@ def create_default(cls, **overrides: Unpack[ZarrV2ArrayMetadataPartial]) -> Zarr The derivation is deliberately one-way, matching the v3 model: overriding `chunks` without `shape` keeps the scalar default - `shape=()`, and consistency between the two is the caller's - responsibility. + `shape=()`, which `chunks` of any other rank do not fit, so + `MetadataValidationError`, as the v3 model refuses a grid its + default shape does not take. """ if "shape" in overrides and "chunks" not in overrides: overrides["chunks"] = tuple(overrides["shape"]) @@ -505,17 +542,7 @@ def from_json(cls, data: object) -> ZarrV2ArrayMetadata: """ # A read model shares no mutable state with what it read. parsed = copy.deepcopy(parse_array_metadata_v2(data)) - return cls( - shape=parsed["shape"], - dtype=parsed["dtype"], - chunks=parsed["chunks"], - fill_value=parsed["fill_value"], - order=parsed["order"], - compressor=parsed["compressor"], - filters=parsed["filters"], - dimension_separator=parsed.get("dimension_separator", "."), - attributes=(dict(parsed["attributes"]) if "attributes" in parsed else UNSET), - ) + return construct(cls, **_v2_array_members(parsed)) @classmethod def from_key_value(cls, mapping: Mapping[StoreKey, bytes]) -> ZarrV2ArrayMetadata: @@ -543,15 +570,13 @@ def to_key_value( ) -> Mapping[ZarrV2ArrayMetadataStoreKey | ZarrV2AttributesStoreKey, bytes]: """The document as a store holds it: `.zarray` without the attributes, and `.zattrs` with them when they are set, even empty. - Validated first: a model that is not valid raises - `MetadataValidationError` with every problem, and nothing is written. + A model was checked when it was built, so its document is written + as it is. """ # Attributes live only in the sibling `.zattrs` file; the `.zarray` # document must exclude them. The `.zattrs` key is present exactly - # when attributes are set (even empty) — UNSET emits no file. A model - # built by hand is not validated: its document is written only if it - # reads as `from_json` reads one, and every problem is raised. - document = parse_array_metadata_v2(self.to_json()) + # when attributes are set (even empty) — UNSET emits no file. + document = self.to_json() zarray = {k: v for k, v in document.items() if k != "attributes"} out: dict[ZarrV2ArrayMetadataStoreKey | ZarrV2AttributesStoreKey, bytes] = { ZARR_V2_ARRAY_METADATA_STORE_KEY: dump_store_json(zarray, indent=indent) @@ -561,3 +586,18 @@ def to_key_value( document["attributes"], indent=indent ) return out + + +def _v2_array_members(document: ZarrV2ArrayMetadataJSON) -> dict[str, object]: + """The members of the v2 array model of `document`, which `parse_array_metadata_v2` gave: a missing `dimension_separator` is `"."`.""" + return { + "shape": document["shape"], + "dtype": document["dtype"], + "chunks": document["chunks"], + "fill_value": document["fill_value"], + "order": document["order"], + "compressor": document["compressor"], + "filters": document["filters"], + "dimension_separator": document.get("dimension_separator", "."), + "attributes": dict(document["attributes"]) if "attributes" in document else UNSET, + } diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_group.py b/packages/zarr-metadata/src/zarr_metadata/model/_group.py index e1b3542f1a..e9434b925c 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_group.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/_group.py @@ -28,7 +28,6 @@ ZarrV3ArrayMetadata, array_json, array_model, - held_document, must_understand_subset, read_array_metadata_v3, ) @@ -41,10 +40,12 @@ ZarrV3ArrayMetadataReading, attributes_of, check_literal, + construct, dump_store_json, load_store_json, missing_keys, other_members, + overlapping, parse_group_metadata_v2, read_array_v3, unexpected_keys, @@ -91,9 +92,11 @@ class ZarrV3GroupMetadata: document it holds; every other unknown top-level key lands in `extra_fields` verbatim. A model holds no scope, as `ZarrV3ArrayMetadata` holds none: `from_json`, `create_default` and - `update` each take the one they read new JSON in. `to_key_value` reads - the document it writes, and each document its consolidated metadata - holds, by the models' own fields, and refuses one with a problem. + `update` each take the one they read new JSON in. It checks itself + when it is built, as `ZarrV3ArrayMetadata` does, and holds its members + as the read refines them: its own members -- each document its + consolidated metadata holds is a model, which checked itself -- or the + constructor raises `MetadataValidationError` with every problem. """ zarr_format: Literal[3] = field(default=3, init=False) @@ -103,22 +106,24 @@ class ZarrV3GroupMetadata: extra_fields: dict[str, ZarrV3ExtensionField] def __post_init__(self) -> None: - reserved = GROUP_METADATA_STANDARD_KEYS_V3 | {ZARR_V3_CONSOLIDATED_METADATA_KEY} - if set(self.extra_fields.keys()).intersection(reserved): - raise MetadataValidationError( - [ - ValidationProblem( - ("extra_fields",), - "Extra fields cannot overlap with standard Zarr V3 group metadata fields", - "invalid_value", - ) - ] - ) # The runtime half of the annotations. + extra = cast("object", self.extra_fields) + if not isinstance(extra, Mapping): + msg = f"extra_fields: expected a mapping of names to JSON, got {extra!r}" + raise TypeError(msg) consolidated = cast("object", self.consolidated_metadata) if consolidated is not UNSET and not isinstance(consolidated, ZarrV3ConsolidatedMetadata): msg = f"consolidated_metadata: expected ZarrV3ConsolidatedMetadata or UNSET, got {consolidated!r}" raise TypeError(msg) + # Its own members, each document its consolidated metadata holds + # being a model that checked itself, held as the read refines them. + reading, members = read_group_v3(_own_document(self, whole=True), NO_SCOPE) + problems = (*overlapping(self.extra_fields, _RESERVED_V3, "group"), *reading.problems) + if len(problems) != 0: + raise MetadataValidationError(problems) + own = cast("GroupMembersV3", members) + object.__setattr__(self, "attributes", own.attributes) + object.__setattr__(self, "extra_fields", own.extra_fields) @classmethod def create_default( @@ -141,7 +146,7 @@ def update( of them again. `MetadataValidationError` when the document the members make has a problem. """ - document = {**_own_document(self), **members} + document = {**_own_document(self, whole=True), **members} for key, value in members.items(): if value is UNSET: del document[key] @@ -203,15 +208,11 @@ def to_key_value( ) -> Mapping[ZarrV3GroupMetadataStoreKey, bytes]: """The document as a store holds it: JSON bytes at `zarr.json`, indented by `indent`. - `MetadataValidationError` when the document, or one its - consolidated metadata holds, has a problem, as the models' own - fields read it. `NaN`, `Infinity` and `-Infinity` in `attributes` - are written as those bare tokens, as zarr-python writes them, - which a strict JSON parser refuses. + A model, and each it holds, was checked when it was built, so its + document is written as it is. `NaN`, `Infinity` and `-Infinity` in + `attributes` are written as those bare tokens, as zarr-python writes + them, which a strict JSON parser refuses. """ - problems = read_group_v3(_held_group(self), NO_SCOPE)[0].problems - if len(problems) != 0: - raise MetadataValidationError(problems) return {ZARR_V3_GROUP_METADATA_STORE_KEY: dump_store_json(group_json(self), indent=indent)} @@ -223,7 +224,8 @@ class ZarrV3ConsolidatedMetadata: is embedded as an extension field on a group's `zarr.json`. Each entry in `metadata` is the model of a complete child document, array or group. `must_understand` is typed permissively as `bool` to mirror the document - shape, but only `False` is valid; this is enforced at runtime. + shape, but only `False` is valid; this is enforced at runtime. Each + document it holds is a model, which checked itself when it was built. """ kind: Literal["inline"] = field(default="inline", init=False) @@ -244,9 +246,13 @@ def __post_init__(self) -> None: ) # The runtime half of the annotations. for path, node in cast("dict[object, object]", self.metadata).items(): + if not isinstance(path, str): + msg = f"metadata: a document's path is a string, got {path!r}" + raise TypeError(msg) if not isinstance(node, (ZarrV3ArrayMetadata, ZarrV3GroupMetadata)): msg = f"metadata[{path!r}]: expected a v3 array or group model, got {node!r}" raise TypeError(msg) + object.__setattr__(self, "metadata", dict(self.metadata)) def to_json(self) -> ZarrV3ConsolidatedMetadataJSON: """The `consolidated_metadata` member as JSON, sharing no mutable state with the model: its `kind`, `must_understand: false`, and each document by its path.""" @@ -265,7 +271,7 @@ def from_json( readings, members, problems = _read_consolidated_v3(data, context) if len(problems) != 0: raise MetadataValidationError(problems) - return cls(metadata=_models(readings, members)[1]) + return construct(cls, metadata=_models(readings, members)[1]) def group_json(model: ZarrV3GroupMetadata) -> ZarrV3GroupMetadataJSON: @@ -280,27 +286,31 @@ def consolidated_json(model: ZarrV3ConsolidatedMetadata) -> ZarrV3ConsolidatedMe ) -def _held_group(model: ZarrV3GroupMetadata) -> dict[str, object]: - """`model`'s document with each field of each document it holds as it was read, which a read takes as it is, as `held_document` gives an array's.""" - return _group_document(model, held_document, _held_group) - +def _own_document(model: ZarrV3GroupMetadata, *, whole: bool) -> dict[str, object]: + """`model`'s document without its consolidated metadata: the members that are the group's own. -def _own_document(model: ZarrV3GroupMetadata) -> dict[str, object]: - """`model`'s document without its consolidated metadata: the members that are the group's own.""" + Empty `attributes` too when `whole`, as the constructor reads them, + where a writer leaves them out. An extra field named as a member the + document declares is no member of it, which the constructor reports. + """ out: dict[str, object] = {"zarr_format": model.zarr_format, "node_type": model.node_type} - if len(model.attributes) > 0: + if whole or len(model.attributes) > 0: out["attributes"] = model.attributes - out.update(model.extra_fields) + out.update((key, value) for key, value in model.extra_fields.items() if key not in _RESERVED_V3) return out +_RESERVED_V3: Final = GROUP_METADATA_STANDARD_KEYS_V3 | {ZARR_V3_CONSOLIDATED_METADATA_KEY} +"""The members of a v3 group document the model holds apart from its extra fields.""" + + def _group_document( model: ZarrV3GroupMetadata, array: Callable[[ZarrV3ArrayMetadata], object], group: Callable[[ZarrV3GroupMetadata], object], ) -> dict[str, object]: """`model`'s document, each document its consolidated metadata holds as `array` or `group` gives it.""" - out = _own_document(model) + out = _own_document(model, whole=False) if model.consolidated_metadata is not UNSET: # Consolidated metadata is a known non-core top-level JSON field. out[ZARR_V3_CONSOLIDATED_METADATA_KEY] = _consolidated_document( @@ -588,10 +598,13 @@ def _with_models( ) if len(reading.problems) != 0: return dataclasses.replace(reading, consolidated=readings) - model = ZarrV3GroupMetadata( + model = construct( + ZarrV3GroupMetadata, attributes=cast("dict[str, JSONValue]", members.attributes), consolidated_metadata=( - UNSET if members.consolidated is UNSET else ZarrV3ConsolidatedMetadata(metadata=models) + UNSET + if members.consolidated is UNSET + else construct(ZarrV3ConsolidatedMetadata, metadata=models) ), extra_fields=members.extra_fields, ) @@ -682,12 +695,20 @@ class ZarrV2GroupMetadata: (mirroring the merged `ZarrV2GroupMetadataJSON` document form). `attributes` is `UNSET` when no `.zattrs` file (or merged `attributes` key) exists — distinct from an explicit empty `.zattrs`, which is `{}` and round-trips - as a file. + as a file. A model checks itself when it is built: its document has no + problem `validate_group_metadata_v2` finds, or the constructor raises + `MetadataValidationError`. """ zarr_format: Literal[2] = field(default=2, init=False) attributes: dict[str, JSONValue] | UNSET + def __post_init__(self) -> None: + # Held as a read refines them, in containers of its own. + object.__setattr__( + self, "attributes", _v2_attributes(parse_group_metadata_v2(self.to_json())) + ) + @classmethod def create_default(cls, **overrides: Unpack[ZarrV2GroupMetadataPartial]) -> ZarrV2GroupMetadata: """ @@ -707,7 +728,9 @@ def update(self, **kwargs: Unpack[ZarrV2GroupMetadataPartial]) -> ZarrV2GroupMet Only the constructor-settable fields listed in `ZarrV2GroupMetadataPartial` can be updated; the fixed `zarr_format` is rejected at the type level. Each given field fully replaces its - previous value. + previous value. `MetadataValidationError` when the document the + change makes has a problem, as the model checks itself when it is + built. """ return dataclasses.replace(self, **kwargs) @@ -735,7 +758,7 @@ def from_json(cls, data: object) -> ZarrV2GroupMetadata: """ # A read model shares no mutable state with what it read. parsed = copy.deepcopy(parse_group_metadata_v2(data)) - return cls(attributes=(dict(parsed["attributes"]) if "attributes" in parsed else UNSET)) + return construct(cls, attributes=_v2_attributes(parsed)) @classmethod def from_key_value(cls, mapping: Mapping[StoreKey, bytes]) -> ZarrV2GroupMetadata: @@ -763,15 +786,13 @@ def to_key_value( ) -> Mapping[ZarrV2GroupMetadataStoreKey | ZarrV2AttributesStoreKey, bytes]: """The document as a store holds it: `.zgroup` without the attributes, and `.zattrs` with them when they are set, even empty. - Validated first: a model that is not valid raises - `MetadataValidationError` with every problem, and nothing is written. + A model was checked when it was built, so its document is written + as it is. """ # Attributes live only in the sibling `.zattrs` file; the `.zgroup` # document must exclude them. The `.zattrs` key is present exactly - # when attributes are set (even empty) — UNSET emits no file. A model - # built by hand is not validated: its document is written only if it - # reads as `from_json` reads one, and every problem is raised. - document = parse_group_metadata_v2(self.to_json()) + # when attributes are set (even empty) — UNSET emits no file. + document = self.to_json() zgroup = {k: v for k, v in document.items() if k != "attributes"} out: dict[ZarrV2GroupMetadataStoreKey | ZarrV2AttributesStoreKey, bytes] = { ZARR_V2_GROUP_METADATA_STORE_KEY: dump_store_json(zgroup, indent=indent) @@ -791,12 +812,22 @@ class ZarrV2ConsolidatedMetadata: `"path/.zattrs"`, ...) verbatim, preserving the normalized JSON tree. Entries are deliberately NOT merged into per-node models: which nodes had a `.zattrs` file at all is information the canonical representation must - keep. Interpreting entries into node models is consumer work. + keep. Interpreting entries into node models is consumer work. A model + checks itself when it is built, as `from_json` checks a document, or + the constructor raises `MetadataValidationError`. """ zarr_consolidated_format: Literal[1] = field(default=1, init=False) metadata: dict[str, JSONValue] + def __post_init__(self) -> None: + # Held as a read refines them, in containers of its own. + document = {"zarr_consolidated_format": 1, "metadata": self.metadata} + refined, problems = _read_consolidated_v2(document) + if len(problems) != 0: + raise MetadataValidationError(problems) + object.__setattr__(self, "metadata", refined) + def to_json(self) -> dict[str, JSONValue]: """The `.zmetadata` document as JSON, sharing no mutable state with the model.""" # to_json output shares no mutable state with the model. @@ -814,49 +845,10 @@ def from_json(cls, data: object) -> ZarrV2ConsolidatedMetadata: `.zattrs` entry is user data, and may hold `NaN`, `Infinity` and `-Infinity`. """ - normalized = arrays_to_tuples(data) - if not isinstance(normalized, Mapping): - raise MetadataValidationError(not_an_object(data)) - doc = cast("Mapping[object, object]", normalized) - problems: list[ValidationProblem] = [ - ValidationProblem((key,), "missing required key", "missing_key") - for key in ("zarr_consolidated_format", "metadata") - if key not in doc - ] - problems.extend( - ValidationProblem((key,), "unexpected document member", "invalid_value") - if isinstance(key, str) - else ValidationProblem((), f"non-string document key {key!r}", "invalid_type") - for key in doc - if key not in {"zarr_consolidated_format", "metadata"} - ) - problems.extend(check_literal(doc, "zarr_consolidated_format", 1)) - refined: dict[str, JSONValue] = {} - if "metadata" in doc: - entries = doc["metadata"] - if not isinstance(entries, Mapping) or not all( - isinstance(k, str) for k in cast("Mapping[object, object]", entries) - ): - problems.append( - ValidationProblem( - ("metadata",), "expected an object with string keys", "invalid_type" - ) - ) - else: - # Each entry is the document its key names: a `.zattrs` is - # user data, and any other is JSON by RFC 8259. - for key, value in cast("Mapping[str, object]", entries).items(): - refine = ( - refine_user_data - if key.rsplit("/", 1)[-1] == ZARR_V2_ATTRIBUTES_STORE_KEY - else refine_json - ) - entry, found = refine(value, ("metadata", key)) - problems.extend(found) - refined[key] = entry + refined, problems = _read_consolidated_v2(data) if len(problems) != 0: - raise MetadataValidationError(with_input(problems, data)) - return cls(metadata=refined) + raise MetadataValidationError(problems) + return construct(cls, metadata=refined) @classmethod def from_key_value(cls, mapping: Mapping[StoreKey, bytes]) -> ZarrV2ConsolidatedMetadata: @@ -872,10 +864,63 @@ def to_key_value( ) -> Mapping[ZarrV2ConsolidatedMetadataStoreKey, bytes]: """The document as a store holds it: JSON bytes at `.zmetadata`, indented by `indent`. - Validated first: a model that is not valid raises - `MetadataValidationError` with every problem, and nothing is written. + A model was checked when it was built, so its document is written + as it is. """ - # A model built by hand is not validated: it is written only as - # `from_json` reads it, and every problem is raised. - document = ZarrV2ConsolidatedMetadata.from_json(self.to_json()).to_json() - return {ZARR_V2_CONSOLIDATED_METADATA_STORE_KEY: dump_store_json(document, indent=indent)} + return { + ZARR_V2_CONSOLIDATED_METADATA_STORE_KEY: dump_store_json(self.to_json(), indent=indent) + } + + +def _read_consolidated_v2( + data: object, +) -> tuple[dict[str, JSONValue], tuple[ValidationProblem, ...]]: + """`data`, a `.zmetadata` document: each entry as read, and every problem, located in the document. + + Each entry is the document its key names: a `.zattrs` is user data, and + any other is JSON by RFC 8259. + """ + normalized = arrays_to_tuples(data) + if not isinstance(normalized, Mapping): + return {}, not_an_object(data) + doc = cast("Mapping[object, object]", normalized) + problems: list[ValidationProblem] = [ + ValidationProblem((key,), "missing required key", "missing_key") + for key in ("zarr_consolidated_format", "metadata") + if key not in doc + ] + problems.extend( + ValidationProblem((key,), "unexpected document member", "invalid_value") + if isinstance(key, str) + else ValidationProblem((), f"non-string document key {key!r}", "invalid_type") + for key in doc + if key not in {"zarr_consolidated_format", "metadata"} + ) + problems.extend(check_literal(doc, "zarr_consolidated_format", 1)) + refined: dict[str, JSONValue] = {} + if "metadata" in doc: + entries = doc["metadata"] + if not isinstance(entries, Mapping) or not all( + isinstance(k, str) for k in cast("Mapping[object, object]", entries) + ): + problems.append( + ValidationProblem( + ("metadata",), "expected an object with string keys", "invalid_type" + ) + ) + else: + for key, value in cast("Mapping[str, object]", entries).items(): + refine = ( + refine_user_data + if key.rsplit("/", 1)[-1] == ZARR_V2_ATTRIBUTES_STORE_KEY + else refine_json + ) + entry, found = refine(value, ("metadata", key)) + problems.extend(found) + refined[key] = entry + return refined, with_input(problems, data) + + +def _v2_attributes(document: ZarrV2GroupMetadataJSON) -> dict[str, JSONValue] | UNSET: + """The attributes of the v2 group model of `document`, which `parse_group_metadata_v2` gave: `UNSET` when it holds none.""" + return dict(document["attributes"]) if "attributes" in document else UNSET diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_validation.py b/packages/zarr-metadata/src/zarr_metadata/model/_validation.py index 4fc1c70f91..f08312ced3 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_validation.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/_validation.py @@ -442,6 +442,46 @@ def fields(self) -> Iterator[tuple[Loc, Resolved[Any]]]: NO_SCOPE: Final = Context.of() """A scope of no definitions, which a model's own document is read in: its fields are read already, and nothing else in it is a field.""" +M = TypeVar("M") + + +def overlapping( + extra_fields: Mapping[str, object], reserved: frozenset[str], node: str +) -> tuple[ValidationProblem, ...]: + """The problem of a model's extra fields named as a member its document declares, which the model holds apart; none when they are not.""" + if set(extra_fields).isdisjoint(reserved): + return () + message = f"Extra fields cannot overlap with standard Zarr V3 {node} metadata fields" + return (ValidationProblem(("extra_fields",), message, "invalid_value"),) + + +def construct(model: type[M], /, **members: object) -> M: + """A `model` of `members` a read found nothing wrong with: as its constructor builds one, without checking them again. + + What pydantic's `model_construct` is to its `__init__`. A model checks + itself when it is built; a read has checked what it read already, so + the model it builds is not checked twice. A member not given, or one + the constructor does not take, is its default. + """ + built = object.__new__(model) + declared = dataclasses.fields(cast("Any", model)) + unknown = members.keys() - {member.name for member in declared if member.init} + if len(unknown) != 0: + msg = f"{model.__name__} has no member {sorted(unknown)!r} to build" + raise TypeError(msg) + for member in declared: + if member.init and member.name in members: + value = members[member.name] + elif member.default is not dataclasses.MISSING: + value = member.default + elif member.default_factory is not dataclasses.MISSING: + value = member.default_factory() + else: + msg = f"{model.__name__} is built with {member.name!r}" + raise TypeError(msg) + object.__setattr__(built, member.name, value) + return built + @dataclass(frozen=True, slots=True) class ArrayMembersV3: diff --git a/packages/zarr-metadata/tests/model/test_array.py b/packages/zarr-metadata/tests/model/test_array.py index 9083fee2d0..90a2510fd1 100644 --- a/packages/zarr-metadata/tests/model/test_array.py +++ b/packages/zarr-metadata/tests/model/test_array.py @@ -1691,34 +1691,6 @@ def test_error_update_refuses_to_leave_out_a_member_a_document_holds() -> None: assert [(p.loc, p.kind) for p in raised.value.problems] == [(("shape",), "missing_key")] -@pytest.mark.parametrize( - ("changes", "problems"), - [ - ({"fill_value": math.nan}, [(("fill_value",), "invalid_value")]), - ({"dimension_names": ("x", "y")}, [(("dimension_names",), "invalid_value")]), - ( - {"shape": (4, 4)}, - [(("chunk_grid", "configuration", "chunk_shape"), "invalid_value")], - ), - ({"attributes": {1: "a"}}, [(("attributes",), "invalid_type")]), - ], - ids=[ - "fill-value-not-json", - "names-for-another-rank", - "shape-without-its-grid", - "attribute-key", - ], -) -def test_error_to_key_value_refuses_a_model_changed_by_hand_into_an_invalid_one( - changes: dict[str, object], problems: list[tuple[tuple[str | int, ...], str]] -) -> None: - """As its own fields read it: no invalid document is written, however the model came to be.""" - model = dataclasses.replace(ZarrV3ArrayMetadata.create_default(shape=(4,)), **changes) - with pytest.raises(MetadataValidationError) as raised: - model.to_key_value() - assert [(p.loc, p.kind) for p in raised.value.problems] == problems - - @pytest.mark.parametrize( ("shape", "kind"), [ @@ -2070,16 +2042,6 @@ def test_v3_create_default_zero_length_dimensions() -> None: assert validate_array_metadata_v3(model.to_json()) == () -def test_v2_create_default_derivation_is_one_way() -> None: - """Overriding chunks without shape leaves the scalar default shape=() - untouched. The v3 model does not derive a shape from its grid either, and - refuses a grid the default shape does not fit: see - `test_error_create_default_refuses_a_document_with_a_problem`.""" - v2 = ZarrV2ArrayMetadata.create_default(chunks=(10, 10)) - assert v2.shape == () - assert v2.chunks == (10, 10) - - # --- v2 dimension_separator default (roborev job 426) ------------------------- diff --git a/packages/zarr-metadata/tests/model/test_construction.py b/packages/zarr-metadata/tests/model/test_construction.py new file mode 100644 index 0000000000..8190d57859 --- /dev/null +++ b/packages/zarr-metadata/tests/model/test_construction.py @@ -0,0 +1,252 @@ +"""A model checks itself when it is built, as pydantic's `__init__` does. + +Built by hand, or changed as `dataclasses.replace` changes one, a model +whose document has a problem is refused at the change, with every problem, +so none is ever invalid. A model a read builds is not read a second time, +and `to_key_value` writes a model as it is. +""" + +from __future__ import annotations + +import dataclasses +import math +from collections import UserDict +from typing import TYPE_CHECKING, Any, cast + +import pytest + +from zarr_metadata.model import ( + UNSET, + MetadataValidationError, + ValidationProblem, + ZarrV2ArrayMetadata, + ZarrV2ConsolidatedMetadata, + ZarrV2GroupMetadata, + ZarrV3ArrayMetadata, + ZarrV3ConsolidatedMetadata, + ZarrV3GroupMetadata, +) +from zarr_metadata.model._validation import construct +from zarr_metadata.v3.data_type.int8 import INT8_DATA_TYPE +from zarr_metadata.v3.definition import CORE_AND_EXTENSIONS + +if TYPE_CHECKING: + from collections.abc import Iterator + + from zarr_metadata.v3.definition import Nested + +ARRAY = ZarrV3ArrayMetadata.create_default( + shape=(4,), + attributes={"a": [1, None]}, + dimension_names=("x",), + acme={"must_understand": False}, +) +GROUP = ZarrV3GroupMetadata.from_json( + { + "zarr_format": 3, + "node_type": "group", + "attributes": {"a": 1}, + "consolidated_metadata": { + "kind": "inline", + "must_understand": False, + "metadata": { + "x": ARRAY.to_json(), + "y": {"zarr_format": 3, "node_type": "group"}, + }, + }, + } +) +V2_ARRAY = ZarrV2ArrayMetadata.create_default(shape=(4,), attributes={"a": 1}) +V2_GROUP = ZarrV2GroupMetadata.create_default(attributes={}) +V2_CONSOLIDATED = ZarrV2ConsolidatedMetadata.from_json( + {"zarr_consolidated_format": 1, "metadata": {"a/.zattrs": {"x": 1}}} +) + + +@pytest.mark.parametrize( + "model", + [ + ARRAY, + GROUP, + GROUP.consolidated_metadata, + V2_ARRAY, + V2_GROUP, + V2_CONSOLIDATED, + ], + ids=["array", "group", "consolidated", "v2-array", "v2-group", "v2-consolidated"], +) +def test_a_model_is_built_as_a_read_builds_it(model: object) -> None: + # The constructor checks what the read checked, and of the same members + # builds the same model. + held = cast("Any", model) + members = { + member.name: getattr(held, member.name) + for member in dataclasses.fields(held) + if member.init + } + assert type(held)(**members) == model + + +@pytest.mark.parametrize( + ("model", "changes"), + [ + (ARRAY, {"shape": [4], "dimension_names": ["x"]}), + (ARRAY, {"attributes": UserDict({"a": [1, None]})}), + (GROUP, {"attributes": UserDict({"a": 1})}), + (V2_ARRAY, {"shape": range(4, 5), "chunks": [4], "attributes": {"a": 1}}), + (V2_CONSOLIDATED, {"metadata": {"a/.zattrs": UserDict({"x": 1})}}), + ], + ids=[ + "array-lists", + "array-mapping", + "group-mapping", + "v2-sequences", + "v2-consolidated-mapping", + ], +) +def test_a_model_holds_its_members_as_a_read_refines_them( + model: object, changes: dict[str, object] +) -> None: + # Arrays as tuples and objects as dicts, as the read holds them, so a + # model built of other containers is the model a read builds. + changed = dataclasses.replace(cast("Any", model), **changes) + assert changed == model + assert type(changed).from_key_value(changed.to_key_value()) == changed + + +@pytest.mark.parametrize( + ("model", "member"), + [ + (ARRAY, "extra_fields"), + (GROUP, "attributes"), + (V2_GROUP, "attributes"), + (V2_CONSOLIDATED, "metadata"), + ], + ids=["array-extra-fields", "group-attributes", "v2-group-attributes", "v2-consolidated"], +) +def test_a_model_shares_no_container_with_what_it_was_built_of(model: object, member: str) -> None: + held: dict[str, object] = {"acme.x": {"must_understand": False}} + built = dataclasses.replace(cast("Any", model), **{member: held}) + held["acme.y"] = math.nan + cast("dict[str, object]", held["acme.x"])["z"] = math.nan + assert getattr(built, member) == {"acme.x": {"must_understand": False}} + assert type(built).from_key_value(built.to_key_value()) == built + + +def test_a_model_is_read_once_and_written_as_it_is() -> None: + values: list[object] = [] + + def counted( + configuration: object, nested: Nested, value: object + ) -> Iterator[ValidationProblem]: + values.append(value) + yield from () + + counting = dataclasses.replace(INT8_DATA_TYPE, fill_value_rules=counted) + scope = CORE_AND_EXTENSIONS.extended_with(counting) + model = ZarrV3ArrayMetadata.create_default(context=scope, data_type="int8", fill_value=3) + model.to_key_value() + # Built by hand, a model is read once, as its own fields read it. + dataclasses.replace(model, fill_value=4) + assert values == [3, 4] + # A group checks its own members: each document it holds checked itself. + group = ZarrV3GroupMetadata( + attributes={}, + consolidated_metadata=ZarrV3ConsolidatedMetadata(metadata={"a": model}), + extra_fields={}, + ) + dataclasses.replace(group, attributes={"b": 1}).to_key_value() + assert values == [3, 4] + + +@pytest.mark.parametrize( + ("model", "changes", "problems"), + [ + (ARRAY, {"fill_value": math.nan}, [(("fill_value",), "invalid_value")]), + (ARRAY, {"dimension_names": ("x", "y")}, [(("dimension_names",), "invalid_value")]), + # Every problem: the grid and the names are each for one dimension. + ( + ARRAY, + {"shape": (4, 4)}, + [ + (("chunk_grid", "configuration", "chunk_shape"), "invalid_value"), + (("dimension_names",), "invalid_value"), + ], + ), + (ARRAY, {"attributes": {1: "a"}}, [(("attributes",), "invalid_type")]), + # Empty or not, a value that is no object is judged as one. + (ARRAY, {"attributes": []}, [(("attributes",), "invalid_type")]), + (ARRAY, {"attributes": None}, [(("attributes",), "invalid_type")]), + # Every problem: an extra field named as a member, and the rest. + ( + ARRAY, + {"extra_fields": {"shape": [1]}, "fill_value": math.nan}, + [(("extra_fields",), "invalid_value"), (("fill_value",), "invalid_value")], + ), + (GROUP, {"attributes": {1: "a"}}, [(("attributes",), "invalid_type")]), + (GROUP, {"attributes": ()}, [(("attributes",), "invalid_type")]), + (GROUP, {"extra_fields": {"acme": math.nan}}, [(("acme",), "invalid_value")]), + (V2_ARRAY, {"order": "Q"}, [(("order",), "invalid_value")]), + (V2_ARRAY, {"chunks": (4, 4)}, [(("chunks",), "invalid_value")]), + (V2_GROUP, {"attributes": {1: "a"}}, [(("attributes",), "invalid_type")]), + ( + V2_CONSOLIDATED, + {"metadata": {"a/.zarray": {"x": math.nan}}}, + [(("metadata", "a/.zarray", "x"), "invalid_value")], + ), + ], + ids=[ + "fill-value-not-json", + "names-for-another-rank", + "shape-without-its-grid", + "attribute-key", + "attributes-empty-and-no-object", + "attributes-null", + "extra-field-named-as-a-member-and-more", + "group-attribute-key", + "group-attributes-empty-and-no-object", + "group-extension-not-json", + "v2-order", + "v2-chunks-for-another-rank", + "v2-group-attribute-key", + "v2-consolidated-entry-not-json", + ], +) +def test_error_a_model_changed_by_hand_into_an_invalid_one_is_refused_at_the_change( + model: object, changes: dict[str, object], problems: list[tuple[tuple[str | int, ...], str]] +) -> None: + """As a read reads its document: no model is invalid, however it came to be.""" + with pytest.raises(MetadataValidationError) as raised: + dataclasses.replace(cast("Any", model), **changes) + assert [(found.loc, found.kind) for found in raised.value.problems] == problems + + +def test_error_v2_create_default_refuses_chunks_its_default_shape_does_not_take() -> None: + # Overriding chunks without shape keeps the scalar default shape (), as + # the v3 model keeps its default shape, and refuses a grid it does not fit. + with pytest.raises(MetadataValidationError) as raised: + ZarrV2ArrayMetadata.create_default(chunks=(10, 10)) + assert [(found.loc, found.kind) for found in raised.value.problems] == [ + (("chunks",), "invalid_value") + ] + + +def test_error_construct_refuses_a_member_the_model_does_not_take() -> None: + with pytest.raises(TypeError, match="ZarrV2GroupMetadata has no member \\['bogus'\\] to build"): + construct(ZarrV2GroupMetadata, attributes=UNSET, bogus=1) + + +def test_error_construct_refuses_a_model_missing_a_member() -> None: + with pytest.raises(TypeError, match="ZarrV2GroupMetadata is built with 'attributes'"): + construct(ZarrV2GroupMetadata) + + +@pytest.mark.parametrize("model", [ARRAY, GROUP], ids=["array", "group"]) +def test_error_extra_fields_are_a_mapping(model: object) -> None: + with pytest.raises(TypeError, match="extra_fields: expected a mapping of names to JSON"): + dataclasses.replace(cast("Any", model), extra_fields=[("acme", 1)]) + + +def test_error_consolidated_metadata_paths_are_strings() -> None: + with pytest.raises(TypeError, match="a document's path is a string, got 1"): + ZarrV3ConsolidatedMetadata(metadata=cast("Any", {1: ARRAY})) diff --git a/packages/zarr-metadata/tests/model/test_group.py b/packages/zarr-metadata/tests/model/test_group.py index 59c17c6d3d..27d05e603a 100644 --- a/packages/zarr-metadata/tests/model/test_group.py +++ b/packages/zarr-metadata/tests/model/test_group.py @@ -3,7 +3,6 @@ import copy import dataclasses import json -import math from collections import UserDict from collections.abc import Callable, Iterator from typing import cast @@ -437,21 +436,6 @@ def test_group_update_reads_the_documents_it_is_given_in_its_scope() -> None: assert removed.consolidated_metadata is UNSET -def test_error_to_key_value_refuses_a_group_holding_a_document_with_a_problem() -> None: - """A document it holds changed by hand into an invalid one, as its own fields read it.""" - child = dataclasses.replace(ZarrV3ArrayMetadata.create_default(shape=(4,)), fill_value=math.nan) - group = ZarrV3GroupMetadata( - attributes={}, - consolidated_metadata=ZarrV3ConsolidatedMetadata(metadata={"a": child}), - extra_fields={}, - ) - with pytest.raises(MetadataValidationError) as raised: - group.to_key_value() - assert [(p.loc, p.kind) for p in raised.value.problems] == [ - ((*A, "fill_value"), "invalid_value") - ] - - def _fields_of_an_array(*at: str | int) -> list[tuple[str | int, ...]]: """Where a default array's fields sit, under `at`.""" points = ("data_type", "chunk_grid", "chunk_key_encoding") From e7b934e45fa399841bc4df731787bd8e6ad19daf Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Tue, 29 Sep 2026 18:33:54 +0200 Subject: [PATCH 11/94] fix(zarr-metadata): a member a closed object does not declare is an unknown key wherever it sits `ProblemKind` defines `unknown_key` as a key an object's type does not declare where the type is closed, and a v3 configuration reported one so. A v2 `.zgroup`, a `.zmetadata` envelope, a group's `consolidated_metadata` and a metadata field's envelope reported the same case as `invalid_value`, so a reader that tolerates what another writer added -- netCDF-C's `_nczarr_*` keys (zarr-python#2296) -- could not filter it by kind. Each is now `unknown_key`, with the checker's message, "unexpected key 'x'". A definition's `check` and `judge` give a configuration back when a field it holds has a stray member, reported and left out, as the checker leaves out a key a closed TypedDict does not declare; a `must_understand` of `false` still refuses it. Assisted-by: ClaudeCode:claude-opus-5-5 --- packages/zarr-metadata/changes/381.bugfix.md | 9 +++ .../src/zarr_metadata/model/_group.py | 11 +--- .../src/zarr_metadata/model/_validation.py | 6 +- .../src/zarr_metadata/v3/_common.py | 12 ++-- .../src/zarr_metadata/v3/_definition.py | 22 +++++-- .../zarr-metadata/tests/model/test_array.py | 2 +- .../zarr-metadata/tests/model/test_group.py | 60 +++++++++++++++++-- .../tests/v3/test_definitions.py | 17 ++++++ .../tests/v3/test_every_definition.py | 2 +- 9 files changed, 112 insertions(+), 29 deletions(-) create mode 100644 packages/zarr-metadata/changes/381.bugfix.md diff --git a/packages/zarr-metadata/changes/381.bugfix.md b/packages/zarr-metadata/changes/381.bugfix.md new file mode 100644 index 0000000000..5b3ebacb69 --- /dev/null +++ b/packages/zarr-metadata/changes/381.bugfix.md @@ -0,0 +1,9 @@ +A member a closed object does not declare is an `unknown_key` wherever +it sits, as `ProblemKind` defines one: in a v2 `.zgroup`, a `.zmetadata` +envelope, a group's `consolidated_metadata` and a metadata field's +envelope, where it was an `invalid_value`, as it already was inside a v3 +configuration. So a reader that tolerates a key another writer added, +such as netCDF-C's `_nczarr_*` keys in a `.zgroup`, can filter by kind. +A definition's `check` and `judge` give a configuration back when a +field it holds has a stray member, which they report and leave out, as +the checker does a key a closed TypedDict does not declare. diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_group.py b/packages/zarr-metadata/src/zarr_metadata/model/_group.py index e9434b925c..7202ecf1d7 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_group.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/_group.py @@ -772,8 +772,9 @@ def from_key_value(cls, mapping: Mapping[StoreKey, bytes]) -> ZarrV2GroupMetadat return cls.from_json(zgroup_raw) zgroup = cast("Mapping[str, object]", zgroup_raw) if "attributes" in zgroup: + # A key `.zgroup` does not declare: its attributes are `.zattrs`. refused = ValidationProblem( - ("attributes",), "unexpected document member", "invalid_value" + ("attributes",), "unexpected key 'attributes'", "unknown_key" ) raise MetadataValidationError(with_input((refused,), zgroup)) if ZARR_V2_ATTRIBUTES_STORE_KEY in mapping: @@ -889,13 +890,7 @@ def _read_consolidated_v2( for key in ("zarr_consolidated_format", "metadata") if key not in doc ] - problems.extend( - ValidationProblem((key,), "unexpected document member", "invalid_value") - if isinstance(key, str) - else ValidationProblem((), f"non-string document key {key!r}", "invalid_type") - for key in doc - if key not in {"zarr_consolidated_format", "metadata"} - ) + problems.extend(unexpected_keys(frozenset({"zarr_consolidated_format", "metadata"}), doc)) problems.extend(check_literal(doc, "zarr_consolidated_format", 1)) refined: dict[str, JSONValue] = {} if "metadata" in doc: diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_validation.py b/packages/zarr-metadata/src/zarr_metadata/model/_validation.py index f08312ced3..090de80b0d 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_validation.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/_validation.py @@ -135,7 +135,7 @@ def missing_keys( def unexpected_keys( allowed: frozenset[str], doc: Mapping[object, object] ) -> tuple[ValidationProblem, ...]: - """One problem per member outside a closed document's declared shape.""" + """One problem per member outside a closed document's declared shape: `unknown_key`, as a closed TypedDict's checker reports one, so a caller who tolerates a member another writer added can tell it from a wrong value.""" problems: list[ValidationProblem] = [] for key in doc: if not isinstance(key, str): @@ -143,9 +143,7 @@ def unexpected_keys( ValidationProblem((), f"non-string document key {key!r}", "invalid_type") ) elif key not in allowed: - problems.append( - ValidationProblem((key,), "unexpected document member", "invalid_value") - ) + problems.append(ValidationProblem((key,), f"unexpected key {key!r}", "unknown_key")) return tuple(problems) diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/_common.py b/packages/zarr-metadata/src/zarr_metadata/v3/_common.py index 1347130a6b..3ccc21e1e8 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/_common.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/_common.py @@ -68,6 +68,10 @@ def validate_metadata_field_v3( return with_input((*envelope, *_configuration_json_problems(value)), value) +ENVELOPE_KEYS: frozenset[str] = frozenset({"name", "configuration", "must_understand"}) +"""The members a metadata field's envelope declares: a key beside them is an unknown key.""" + + def envelope_problems( value: object, *, allow_must_understand_false: bool ) -> tuple[ValidationProblem, ...]: @@ -90,16 +94,13 @@ def envelope_problems( ) field = cast("Mapping[object, object]", value) problems: list[ValidationProblem] = [] - allowed_keys = frozenset({"name", "configuration", "must_understand"}) for key in field: if not isinstance(key, str): problems.append( ValidationProblem((), f"non-string metadata field key {key!r}", "invalid_type") ) - elif key not in allowed_keys: - problems.append( - ValidationProblem((key,), "unexpected metadata field member", "invalid_value") - ) + elif key not in ENVELOPE_KEYS: + problems.append(ValidationProblem((key,), f"unexpected key {key!r}", "unknown_key")) if not isinstance(field.get("name"), str): problems.append(ValidationProblem(("name",), "expected a string name", "invalid_type")) if "configuration" in field: @@ -165,6 +166,7 @@ def parse_metadata_field_v3(value: object) -> ZarrV3MetadataFieldJSON: __all__ = [ + "ENVELOPE_KEYS", "ChunkGridField", "ChunkKeyEncodingField", "CodecField", diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py b/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py index 0c589b9db5..d8d149cd81 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py @@ -66,6 +66,7 @@ unread_in, ) from zarr_metadata.v3._common import ( + ENVELOPE_KEYS, ChunkGridField, ChunkKeyEncodingField, CodecField, @@ -799,6 +800,14 @@ def _as_written(field: _NestedField) -> JSONValue: return field.json +def _declared(field: _NestedField) -> JSONValue: + """A nested field put back as it was written, but for the members its envelope does not declare, which are reported and left out, as the checker leaves out a key a closed TypedDict does not declare.""" + if not isinstance(field.json, Mapping): + return field.json + envelope = cast("Mapping[str, JSONValue]", field.json) + return {key: value for key, value in envelope.items() if key in ENVELOPE_KEYS} + + def _put_back(value: object, put: Callable[[_NestedField], JSONValue]) -> object: """`value` with each nested field the checker handed back put back as `put` gives it.""" if isinstance(value, _NestedField): @@ -864,15 +873,20 @@ def _configuration_checked( The step a definition's `judge` starts from. A member typed with a field alias holds a metadata field, whose envelope is judged as a - document's is: a stray member, or a `must_understand` of `false`, is a - problem of the configuration, and the value does not come back. + document's is: a `must_understand` of `false` is a problem of the + configuration, and the value does not come back; a stray member is an + unknown key, reported and left out, and the value still comes back, + as the checker reports and leaves out a key a closed TypedDict does + not declare. """ refined, problems = refine_json(value, loc) if len(problems) != 0: return None, problems - typed, found, nested = _checked(shape, refined, loc) + typed, found, nested = _typed(shape, refined, loc) problems = (*found, *(problem for field in nested for problem in _envelope(field))) - return (cast("T", typed) if _usable(problems) else None), problems + if not _usable(problems): + return None, problems + return cast("T", _put_back(typed, _declared)), problems def named_configuration( diff --git a/packages/zarr-metadata/tests/model/test_array.py b/packages/zarr-metadata/tests/model/test_array.py index 90a2510fd1..f7cde5c83b 100644 --- a/packages/zarr-metadata/tests/model/test_array.py +++ b/packages/zarr-metadata/tests/model/test_array.py @@ -826,7 +826,7 @@ def test_metadata_field_must_understand_must_be_boolean(value: object) -> None: def test_metadata_field_rejects_unknown_envelope_member() -> None: """Unknown envelope keys cannot be silently discarded during normalization.""" problems = validate_metadata_field_v3({"name": "x", "typo": 1}) - assert [(problem.loc, problem.kind) for problem in problems] == [(("typo",), "invalid_value")] + assert [(problem.loc, problem.kind) for problem in problems] == [(("typo",), "unknown_key")] @pytest.mark.parametrize("field", ["codecs", "storage_transformers"]) diff --git a/packages/zarr-metadata/tests/model/test_group.py b/packages/zarr-metadata/tests/model/test_group.py index 27d05e603a..6d364322ef 100644 --- a/packages/zarr-metadata/tests/model/test_group.py +++ b/packages/zarr-metadata/tests/model/test_group.py @@ -57,6 +57,7 @@ Nested, Refused, Unclaimed, + resolve, ) from zarr_metadata.v3.group import ZarrV3GroupMetadataJSONPartial @@ -157,11 +158,58 @@ def test_group_zarr_format_rejects_float( assert [(p.loc, p.kind) for p in validate(document)] == [(("zarr_format",), "invalid_value")] -def test_group_v2_rejects_unknown_document_member() -> None: - """The closed v2 merged-document shape rejects undeclared members.""" - assert [(p.loc, p.kind) for p in validate_group_metadata_v2({"zarr_format": 2, "x": 1})] == [ - (("x",), "invalid_value") - ] +def _v2_consolidated_problems(document: object) -> list[tuple[tuple[str | int, ...], str]]: + with pytest.raises(MetadataValidationError) as raised: + ZarrV2ConsolidatedMetadata.from_json(document) + return [(p.loc, p.kind) for p in raised.value.problems] + + +def _v3_field_problems(field: object) -> list[tuple[tuple[str | int, ...], str]]: + return [(p.loc, p.kind) for p in resolve(field, CodecDefinition, CORE_AND_EXTENSIONS)[1]] + + +@pytest.mark.parametrize( + ("problems", "expected"), + [ + ( + lambda: [ + (p.loc, p.kind) for p in validate_group_metadata_v2({"zarr_format": 2, "x": 1}) + ], + [(("x",), "unknown_key")], + ), + ( + lambda: _v2_consolidated_problems( + {"zarr_consolidated_format": 1, "metadata": {}, "x": 1} + ), + [(("x",), "unknown_key")], + ), + ( + lambda: [ + (p.loc, p.kind) + for p in validate_group_metadata_v3( + _group(consolidated_metadata={**_inline(), "x": 1}) + ) + ], + [(("consolidated_metadata", "x"), "unknown_key")], + ), + ( + lambda: _v3_field_problems({"name": "gzip", "configuration": {"level": 1}, "x": 1}), + [(("x",), "unknown_key")], + ), + ( + lambda: _v3_field_problems({"name": "gzip", "configuration": {"level": 1, "x": 1}}), + [(("configuration", "x"), "unknown_key")], + ), + ], + ids=["v2-group", "v2-consolidated", "v3-consolidated", "v3-field", "v3-configuration"], +) +def test_a_member_a_closed_object_does_not_declare_is_an_unknown_key( + problems: Callable[[], list[tuple[tuple[str | int, ...], str]]], + expected: list[tuple[tuple[str | int, ...], str]], +) -> None: + # Wherever it sits, so a reader that tolerates what another writer + # added -- NCZarr's `_nczarr_*` keys -- filters by kind. + assert problems() == expected @pytest.mark.parametrize( @@ -254,7 +302,7 @@ def test_v2_group_from_key_value_rejects_zgroup_extra_members(extra_key: str) -> ZarrV2GroupMetadata.from_key_value({".zgroup": json.dumps(doc).encode()}) assert [(problem.loc, problem.kind) for problem in exc_info.value.problems] == [ - ((extra_key,), "invalid_value") + ((extra_key,), "unknown_key") ] diff --git a/packages/zarr-metadata/tests/v3/test_definitions.py b/packages/zarr-metadata/tests/v3/test_definitions.py index a262f152d8..839fe9d309 100644 --- a/packages/zarr-metadata/tests/v3/test_definitions.py +++ b/packages/zarr-metadata/tests/v3/test_definitions.py @@ -898,6 +898,23 @@ def test_error_check_judges_a_nested_envelope() -> None: assert [found.loc for found in problems] == [("codecs", 0, "configuration")] +def test_judge_leaves_out_a_member_a_nested_field_s_envelope_does_not_declare() -> None: + # Reported as an unknown key and left out, as the checker leaves out a + # key a closed TypedDict does not declare: the configuration comes back. + for given in (ACME_STACK.check, ACME_STACK.judge): + typed, problems = given({"codecs": [{"name": "crc32c", "x": 1}]}) + assert typed == {"codecs": ({"name": "crc32c"},)} + assert _locs(problems) == [(("codecs", 0, "x"), "unknown_key")] + + +def test_error_judge_refuses_a_nested_field_that_need_not_be_understood() -> None: + # A `must_understand` of false is no unknown key: the configuration + # does not come back. + typed, problems = ACME_STACK.judge({"codecs": [{"name": "crc32c", "must_understand": False}]}) + assert typed is None + assert _locs(problems) == [(("codecs", 0, "must_understand"), "invalid_value")] + + def test_judge_is_the_check_and_then_the_rules() -> None: # A caller holding one configuration: the rules are asked only of a # configuration that type-checked, so they never meet a wrong type. diff --git a/packages/zarr-metadata/tests/v3/test_every_definition.py b/packages/zarr-metadata/tests/v3/test_every_definition.py index 5744860327..91b3a76849 100644 --- a/packages/zarr-metadata/tests/v3/test_every_definition.py +++ b/packages/zarr-metadata/tests/v3/test_every_definition.py @@ -384,7 +384,7 @@ def _shard(codecs: list[object], index_codecs: list[object]) -> dict[str, object ( {"name": "zfpy", "configuration": {}, "extra": 1}, CodecDefinition, - [(("extra",), "invalid_value")], + [(("extra",), "unknown_key")], ), (None, CodecDefinition, [((), "invalid_type")]), ( From 3b96e780b7f4d73c921eef131b49fdfb14329be1 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Tue, 29 Sep 2026 18:35:03 +0200 Subject: [PATCH 12/94] fix(zarr-metadata): a zarr.json of no node type says what format it is in `read_node_metadata_v3` reads a document by the node type it names, and one that names none was reported only as missing its `node_type`. A crawler meeting the root `zarr.json` of zarr-python 2's draft of v3 (zarr-python#2982), whose `zarr_format` is a URL, learned "no node type", not "not v3". A document the dispatch cannot follow is now judged by its `zarr_format` as well: missing, or other than 3, is a problem beside the node type's. Assisted-by: ClaudeCode:claude-opus-5-5 --- packages/zarr-metadata/README.md | 5 ++- .../zarr-metadata/changes/381.bugfix.1.md | 6 +++ packages/zarr-metadata/docs/index.md | 5 ++- .../src/zarr_metadata/model/_group.py | 32 ++++++++++++---- .../zarr-metadata/tests/model/test_group.py | 37 ++++++++++++++++++- 5 files changed, 72 insertions(+), 13 deletions(-) create mode 100644 packages/zarr-metadata/changes/381.bugfix.1.md diff --git a/packages/zarr-metadata/README.md b/packages/zarr-metadata/README.md index ed4a4783cd..b454cd93d4 100644 --- a/packages/zarr-metadata/README.md +++ b/packages/zarr-metadata/README.md @@ -128,8 +128,9 @@ metadata = reading.metadata # None when reading.problems is not empty its consolidated metadata holds once. `read_node_metadata_v3` reads a `zarr.json` of either kind as the node its `node_type` says it is, as a discriminated union reads its tag: a document that says neither reads -as `ZarrV3UnknownNodeReading`, with the problem, and nothing else of it -is read. `node_metadata_from_json_v3` and `node_metadata_from_key_value_v3` +as `ZarrV3UnknownNodeReading`, with the problems, and nothing else of it +is read but its `zarr_format`, so a document of another format says it +is not v3. `node_metadata_from_json_v3` and `node_metadata_from_key_value_v3` build the model of either kind, as the models' own `from_json` and `from_key_value` build one. diff --git a/packages/zarr-metadata/changes/381.bugfix.1.md b/packages/zarr-metadata/changes/381.bugfix.1.md new file mode 100644 index 0000000000..6bfaeb669a --- /dev/null +++ b/packages/zarr-metadata/changes/381.bugfix.1.md @@ -0,0 +1,6 @@ +A `zarr.json` that says no node type the spec defines is judged by its +`zarr_format` too, so a document of another format says it is not v3. +The root `zarr.json` of zarr-python 2's draft of v3, whose `zarr_format` +is a URL, was reported only as missing its `node_type`; it now also has +an `invalid_type` problem at `zarr_format`, and a v2 document read as a +`zarr.json` an `invalid_value` there. diff --git a/packages/zarr-metadata/docs/index.md b/packages/zarr-metadata/docs/index.md index 70608414dd..ee989b2a79 100644 --- a/packages/zarr-metadata/docs/index.md +++ b/packages/zarr-metadata/docs/index.md @@ -143,8 +143,9 @@ metadata = reading.metadata # None when reading.problems is not empty its consolidated metadata holds once. `read_node_metadata_v3` reads a `zarr.json` of either kind as the node its `node_type` says it is, as a discriminated union reads its tag: a document that says neither reads -as `ZarrV3UnknownNodeReading`, with the problem, and nothing else of it -is read. `node_metadata_from_json_v3` and `node_metadata_from_key_value_v3` +as `ZarrV3UnknownNodeReading`, with the problems, and nothing else of it +is read but its `zarr_format`, so a document of another format says it +is not v3. `node_metadata_from_json_v3` and `node_metadata_from_key_value_v3` build the model of either kind, as the models' own `from_json` and `from_key_value` build one. diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_group.py b/packages/zarr-metadata/src/zarr_metadata/model/_group.py index 7202ecf1d7..f09419a322 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_group.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/_group.py @@ -364,7 +364,12 @@ def fields(self) -> Iterator[tuple[Loc, Resolved[Any]]]: @dataclass(frozen=True, slots=True) class ZarrV3UnknownNodeReading: - """A v3 document of no node type the spec defines -- its `node_type` missing, or neither `"array"` nor `"group"` -- or not an object at all: nothing else of it is read, as its problem says.""" + """A v3 document of no node type the spec defines -- its `node_type` missing, or neither `"array"` nor `"group"` -- or not an object at all: nothing else of it is read but its `zarr_format`, as its problems say. + + So a document of another format says so: zarr-python 2's draft of v3 + wrote a root `zarr.json` whose `zarr_format` is a URL, and a v2 + document names format 2. + """ problems: tuple[ValidationProblem, ...] """Why it is no node.""" @@ -395,8 +400,9 @@ def read_node_metadata_v3( zod's discriminated union read one: an array is read as `read_array_metadata_v3` reads it, a group as `read_group_metadata_v3` does, and a document that says neither, or is not an object, is - `ZarrV3UnknownNodeReading`, with the problem. So no caller reads - `node_type` from JSON it has not read. + `ZarrV3UnknownNodeReading`, with the problems, its `zarr_format`'s + among them. So no caller reads `node_type` from JSON it has not read, + and a document of another format says it is not v3. """ node_type, problems = _node_type(value) if node_type == "array": @@ -465,16 +471,26 @@ def _read_node_v3( def _node_type(value: object) -> tuple[str | None, tuple[ValidationProblem, ...]]: - """The node type `value` says it is, one of `_NODE_TYPES`; None, with the problem, when it says none of them, or is not an object.""" + """The node type `value` says it is, one of `_NODE_TYPES`; None, with the problems, when it says none of them, or is not an object. + + A document that says none is judged by its `zarr_format` too, so one + of another format says it is not v3. + """ if not isinstance(value, Mapping): return None, not_an_object(value) document = cast("Mapping[object, object]", value) - if "node_type" not in document: - return None, (ValidationProblem(("node_type",), "missing required key", "missing_key"),) - node_type = document["node_type"] + node_type = document.get("node_type") if isinstance(node_type, str) and node_type in _NODE_TYPES: return node_type, () - return None, with_input((outside_of(("node_type",), node_type, _NODE_TYPES),), document) + problems = [ + *missing_keys(frozenset({"zarr_format"}), document), + *check_literal(document, "zarr_format", 3), + ] + if "node_type" not in document: + problems.append(ValidationProblem(("node_type",), "missing required key", "missing_key")) + else: + problems.append(outside_of(("node_type",), node_type, _NODE_TYPES)) + return None, with_input(problems, document) @dataclass(frozen=True, slots=True) diff --git a/packages/zarr-metadata/tests/model/test_group.py b/packages/zarr-metadata/tests/model/test_group.py index 6d364322ef..3563badf0d 100644 --- a/packages/zarr-metadata/tests/model/test_group.py +++ b/packages/zarr-metadata/tests/model/test_group.py @@ -629,7 +629,7 @@ def test_a_node_is_read_as_the_node_its_node_type_says( ids=["another-kind", "a-number", "null"], ) def test_error_a_node_type_the_spec_does_not_define(node_type: object, kind: str) -> None: - """Nothing else of the document is read, so nothing else is judged.""" + """Nothing else of the document is read but its `zarr_format`, which is 3 here, so nothing else is judged.""" read = read_node_metadata_v3({**_array(), "node_type": node_type, "shape": "not a shape"}) assert isinstance(read, ZarrV3UnknownNodeReading) assert [(p.loc, p.kind) for p in read.problems] == [(("node_type",), kind)] @@ -644,6 +644,41 @@ def test_error_a_document_without_a_node_type() -> None: assert [(p.loc, p.kind) for p in read.problems] == [(("node_type",), "missing_key")] +@pytest.mark.parametrize( + ("document", "problems"), + [ + # zarr-python 2's draft of v3 (zarr-python#2982): a root `zarr.json` + # naming its format by URL, and no node type. + ( + { + "zarr_format": "https://purl.org/zarr/spec/protocol/core/3.0", + "metadata_encoding": "https://purl.org/zarr/spec/protocol/core/3.0", + "metadata_key_suffix": ".json", + "extensions": [], + }, + [(("zarr_format",), "invalid_type"), (("node_type",), "missing_key")], + ), + ( + {"zarr_format": 2, "shape": [4], "chunks": [4], "dtype": "|u1"}, + [(("zarr_format",), "invalid_value"), (("node_type",), "missing_key")], + ), + ( + {"node_type": "dataset"}, + [(("zarr_format",), "missing_key"), (("node_type",), "invalid_value")], + ), + ], + ids=["v3-draft", "v2", "no-format"], +) +def test_error_a_document_of_another_format_says_so( + document: dict[str, object], problems: list[tuple[tuple[str | int, ...], str]] +) -> None: + read = read_node_metadata_v3(document) + assert isinstance(read, ZarrV3UnknownNodeReading) + assert [(p.loc, p.kind) for p in read.problems] == problems + # The validator reads a node's type as the reader does. + assert [(p.loc, p.kind) for p in validate_node_metadata_v3(document)] == problems + + def test_error_a_node_that_is_not_an_object() -> None: read = read_node_metadata_v3([_array()]) assert isinstance(read, ZarrV3UnknownNodeReading) From 8d773520a508cbf3cef8e99cf58521cb2857a992 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Tue, 29 Sep 2026 18:46:09 +0200 Subject: [PATCH 13/94] fix(zarr-metadata): consolidated metadata is judged as the hierarchy below its group A group's consolidated metadata holds the hierarchy below the group: zarr-python keeps the document of the node at /a/b of that hierarchy, the group its root, at the key a/b. Nothing judged the keys or the tree they make. A key that is no node's path ("", "__a", "a/../b", "/a"), a document below an array's, and a document whose group is missing were read without a problem; zarr-python raises a bare KeyError on three of them and keeps the rest. Each is now a problem at its entry, a missing group a missing_key, and ZarrV3ConsolidatedMetadata refuses them when it is built. A listed group's own listing lists what the group lists, at the joined key and of the same node type: a node it lists alone is dropped by zarr-python's reader, which keeps the flat listing. The rules are a hierarchy's, not consolidated metadata's: a private hierarchy_problems judges the node type of each node, by its node path, as the spec's tree, so a model of a whole hierarchy can use it too. Every problem is one per value, saying its reasons at once -- a path of a thousand bad names is one problem, a chain of missing groups one missing_key at the nearest -- so what is reported weighs what was read. NodeName and NodePath, in zarr_metadata.v3, are the spec's node names and paths, modelled on zarrs' types, and zarr_metadata.model judges a string by them with validate/is/parse_node_name_v3 and their node_path twins. Assisted-by: ClaudeCode:claude-opus-5-5 Assisted-by: ClaudeCode:claude-fable-5-1 --- packages/zarr-metadata/README.md | 9 + .../zarr-metadata/changes/381.bugfix.2.md | 9 + packages/zarr-metadata/changes/381.feature.md | 5 + packages/zarr-metadata/docs/api/v3/index.md | 4 + packages/zarr-metadata/docs/index.md | 9 + .../src/zarr_metadata/model/__init__.py | 14 + .../src/zarr_metadata/model/_group.py | 153 +++++++++- .../src/zarr_metadata/model/_json_schema.py | 5 +- .../src/zarr_metadata/v3/__init__.py | 3 + .../src/zarr_metadata/v3/_hierarchy.py | 287 ++++++++++++++++++ .../tests/model/test_construction.py | 11 + .../zarr-metadata/tests/model/test_group.py | 192 +++++++++++- .../zarr-metadata/tests/test_problem_data.py | 8 +- .../zarr-metadata/tests/test_public_api.py | 2 + .../zarr-metadata/tests/v3/test_hierarchy.py | 256 ++++++++++++++++ 15 files changed, 955 insertions(+), 12 deletions(-) create mode 100644 packages/zarr-metadata/changes/381.bugfix.2.md create mode 100644 packages/zarr-metadata/changes/381.feature.md create mode 100644 packages/zarr-metadata/src/zarr_metadata/v3/_hierarchy.py create mode 100644 packages/zarr-metadata/tests/v3/test_hierarchy.py diff --git a/packages/zarr-metadata/README.md b/packages/zarr-metadata/README.md index b454cd93d4..42480c7ee4 100644 --- a/packages/zarr-metadata/README.md +++ b/packages/zarr-metadata/README.md @@ -134,6 +134,15 @@ is not v3. `node_metadata_from_json_v3` and `node_metadata_from_key_value_v3` build the model of either kind, as the models' own `from_json` and `from_key_value` build one. +Consolidated metadata holds the hierarchy below its group, the group its +root: the document of the node at `/a/b` sits at the key `a/b`, and the +documents and the group make a tree in which only groups hold nodes and +each node's parent is held. `NodeName` and `NodePath`, in +`zarr_metadata.v3`, are the strings the spec's rules for node names and +paths hold of, modelled on zarrs' types of those names, and +`validate_node_name_v3`, `is_node_name_v3` and `parse_node_name_v3`, and +their `node_path` twins, judge a string by them. + A member the spec does not define is not a field; the model's `must_understand_fields` names those a reader must understand. diff --git a/packages/zarr-metadata/changes/381.bugfix.2.md b/packages/zarr-metadata/changes/381.bugfix.2.md new file mode 100644 index 0000000000..9c5b0f86bb --- /dev/null +++ b/packages/zarr-metadata/changes/381.bugfix.2.md @@ -0,0 +1,9 @@ +A group's consolidated metadata is judged as the hierarchy below the +group, the group its root: zarr-python keeps the document of the node at +`/a/b` at the key `a/b`, and the documents and the group make a tree in +which only groups hold nodes and each node's parent is held. A key that +is no node's path below the group, such as `""`, `"__a"` or `"a/../b"`, +and a document below an array's were read without a problem, and so was +a document whose group is missing, which zarr-python fails to read. Each +is a problem now, a missing group a `missing_key`, and +`ZarrV3ConsolidatedMetadata` refuses them when it is built. diff --git a/packages/zarr-metadata/changes/381.feature.md b/packages/zarr-metadata/changes/381.feature.md new file mode 100644 index 0000000000..a526471296 --- /dev/null +++ b/packages/zarr-metadata/changes/381.feature.md @@ -0,0 +1,5 @@ +`NodeName` and `NodePath`, in `zarr_metadata.v3`, are the names and paths +of the nodes of a v3 hierarchy, modelled on zarrs' types of those names, +and `zarr_metadata.model` judges a string by the spec's rules for them: +`validate_node_name_v3`, `is_node_name_v3` and `parse_node_name_v3`, and +their `node_path` twins. diff --git a/packages/zarr-metadata/docs/api/v3/index.md b/packages/zarr-metadata/docs/api/v3/index.md index f20267d372..b0f5c2ab4e 100644 --- a/packages/zarr-metadata/docs/api/v3/index.md +++ b/packages/zarr-metadata/docs/api/v3/index.md @@ -8,6 +8,10 @@ title: v3 ::: zarr_metadata.v3.ZarrV3MetadataFieldJSON +::: zarr_metadata.v3.NodeName + +::: zarr_metadata.v3.NodePath + ::: zarr_metadata.v3.array ::: zarr_metadata.v3.group diff --git a/packages/zarr-metadata/docs/index.md b/packages/zarr-metadata/docs/index.md index ee989b2a79..6597cedc87 100644 --- a/packages/zarr-metadata/docs/index.md +++ b/packages/zarr-metadata/docs/index.md @@ -149,6 +149,15 @@ is not v3. `node_metadata_from_json_v3` and `node_metadata_from_key_value_v3` build the model of either kind, as the models' own `from_json` and `from_key_value` build one. +Consolidated metadata holds the hierarchy below its group, the group its +root: the document of the node at `/a/b` sits at the key `a/b`, and the +documents and the group make a tree in which only groups hold nodes and +each node's parent is held. `NodeName` and `NodePath`, in +`zarr_metadata.v3`, are the strings the spec's rules for node names and +paths hold of, modelled on zarrs' types of those names, and +`validate_node_name_v3`, `is_node_name_v3` and `parse_node_name_v3`, and +their `node_path` twins, judge a string by them. + A member the spec does not define is not a field; the model's `must_understand_fields` names those a reader must understand. diff --git a/packages/zarr-metadata/src/zarr_metadata/model/__init__.py b/packages/zarr-metadata/src/zarr_metadata/model/__init__.py index ad74f9f072..f967996b82 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/__init__.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/__init__.py @@ -106,6 +106,14 @@ parse_metadata_field_v3, validate_metadata_field_v3, ) +from zarr_metadata.v3._hierarchy import ( + is_node_name_v3, + is_node_path_v3, + parse_node_name_v3, + parse_node_path_v3, + validate_node_name_v3, + validate_node_path_v3, +) from zarr_metadata.v3.array import ( ZARR_V3_ARRAY_METADATA_STORE_KEY, ZarrV3ArrayMetadataStoreKey, @@ -163,6 +171,8 @@ "is_group_metadata_v3", "is_json", "is_metadata_field_v3", + "is_node_name_v3", + "is_node_path_v3", "node_metadata_from_json_v3", "node_metadata_from_key_value_v3", "node_metadata_json_schema_v3", @@ -172,6 +182,8 @@ "parse_group_metadata_v3", "parse_json", "parse_metadata_field_v3", + "parse_node_name_v3", + "parse_node_path_v3", "read_array_metadata_v3", "read_group_metadata_v3", "read_node_metadata_v3", @@ -182,4 +194,6 @@ "validate_json", "validate_metadata_field_v3", "validate_node_metadata_v3", + "validate_node_name_v3", + "validate_node_path_v3", ] diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_group.py b/packages/zarr-metadata/src/zarr_metadata/model/_group.py index f09419a322..62bc8c09b2 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_group.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/_group.py @@ -6,7 +6,7 @@ import dataclasses from collections.abc import Callable, Mapping from dataclasses import dataclass, field -from typing import TYPE_CHECKING, Any, Final, Literal, TypeGuard, cast +from typing import TYPE_CHECKING, Any, Final, Literal, TypeGuard, TypeVar, cast from typing_extensions import TypeAliasType, TypedDict, Unpack @@ -20,6 +20,7 @@ outside_of, refine_json, refine_user_data, + shown, with_input, ) from zarr_metadata._json import prefixed as _prefix @@ -53,6 +54,7 @@ from zarr_metadata.v2.attributes import ZARR_V2_ATTRIBUTES_STORE_KEY from zarr_metadata.v2.consolidated import ZARR_V2_CONSOLIDATED_METADATA_STORE_KEY from zarr_metadata.v2.group import ZARR_V2_GROUP_METADATA_STORE_KEY +from zarr_metadata.v3._hierarchy import NodeType, hierarchy_problems, path_faults, said from zarr_metadata.v3._registry import CORE_AND_EXTENSIONS, Context from zarr_metadata.v3.array import ZarrV3ExtensionField from zarr_metadata.v3.consolidated import ZARR_V3_CONSOLIDATED_METADATA_KEY @@ -226,6 +228,9 @@ class ZarrV3ConsolidatedMetadata: `must_understand` is typed permissively as `bool` to mirror the document shape, but only `False` is valid; this is enforced at runtime. Each document it holds is a model, which checked itself when it was built. + The documents and the group make the hierarchy below the group, the + group its root, each at its node's path in it without the leading `/`: + the node at `/a/b` at `a/b`. """ kind: Literal["inline"] = field(default="inline", init=False) @@ -252,6 +257,23 @@ def __post_init__(self) -> None: if not isinstance(node, (ZarrV3ArrayMetadata, ZarrV3GroupMetadata)): msg = f"metadata[{path!r}]: expected a v3 array or group model, got {node!r}" raise TypeError(msg) + problems: list[ValidationProblem] = [] + node_types: dict[str, NodeType | None] = {} + for path, node in self.metadata.items(): + faults = _key_problems(path) + problems.extend(faults) + if len(faults) == 0: + node_types[path] = _model_node_type(node) + problems.extend(_hierarchy_problems(node_types)) + for path, node in self.metadata.items(): + if path in node_types: + problems.extend( + _nested_listing_problems( + path, _model_listing(node), node_types, _model_node_type + ) + ) + if len(problems) != 0: + raise MetadataValidationError(problems) object.__setattr__(self, "metadata", dict(self.metadata)) def to_json(self) -> ZarrV3ConsolidatedMetadataJSON: @@ -560,6 +582,8 @@ def read_group_v3( return reading, GroupMembersV3(attributes, extra_fields, held) +T = TypeVar("T") + _CONSOLIDATED_MEMBERS: Final = ("kind", "must_understand", "metadata") """The members of an inline `consolidated_metadata`, in the order the convention declares them.""" @@ -586,6 +610,7 @@ def _read_consolidated_v3( problems.extend(check_literal(env, "must_understand", False)) readings: dict[str, ZarrV3NodeMetadataReading] = {} members: dict[str, ArrayMembersV3 | GroupMembersV3] = {} + node_types: dict[str, NodeType | None] = {} entries = env.get("metadata") if "metadata" in env and not isinstance(entries, Mapping): problems.append(ValidationProblem(("metadata",), "expected an object", "invalid_type")) @@ -596,13 +621,139 @@ def _read_consolidated_v3( ValidationProblem(("metadata",), f"non-string key {key!r}", "invalid_type") ) continue + faults = _key_problems(key) + problems.extend(faults) readings[key], child = _read_node_v3(entry, context) if child is not None: members[key] = child problems.extend(_prefix("metadata", _prefix(key, readings[key].problems))) + if len(faults) == 0: + node_types[key] = _node_type_of(readings[key]) + problems.extend(_hierarchy_problems(node_types)) + for key in node_types: + problems.extend( + _nested_listing_problems( + key, _reading_listing(readings[key]), node_types, _node_type_of + ) + ) return readings, members, tuple(problems) +def _key_problems(key: str) -> list[ValidationProblem]: + """What keeps `key` from being where consolidated metadata keeps a document, said in one problem at the key. + + Consolidated metadata holds the hierarchy below its group, the group its + root, and the reference implementation keeps the document of each node + at the node's path in that hierarchy without its leading `/`: the node + at `/a/b` at the key `a/b`. + """ + faults = _below_faults(key) + if len(faults) == 0: + return [] + message = f"expected the path of a node below the group, got {shown(key)}, which {said(faults)}" + return [ValidationProblem(("metadata", key), message, "invalid_value")] + + +def _nested_listing_problems( + key: str, + listing: Mapping[str, T], + node_types: Mapping[str, NodeType | None], + node_type: Callable[[T], NodeType | None], +) -> list[ValidationProblem]: + """What is wrong with `listing`, the own consolidated listing of the group at `key`, against `node_types`, the group's flat listing: each problem at the nested entry. + + The reference implementation lists every node below the group in the + group's own listing, flat, and gives each group it lists an empty + listing of its own. So a node a listed group lists is one the group + lists too, at the joined key, of the same node type: one it lists + alone would be dropped by the reference reader, and one it lists as + another type contradicts the tree. Only the listing's own entries are + judged: what a group listed there lists in turn is that group's own to + judge, when its document is read. `node_type` says what each entry is, + so readings and models are judged alike. + """ + problems: list[ValidationProblem] = [] + for path, entry in listing.items(): + if len(_below_faults(path)) != 0: + # Its own reader reports a key that is no node's path. + continue + joined = f"{key}/{path}" + here = ("metadata", key, ZARR_V3_CONSOLIDATED_METADATA_KEY, "metadata", path) + if joined not in node_types: + message = ( + f"expected a node the group lists, got {shown(f'/{joined}')}, which " + f"{shown(f'/{key}')} lists alone" + ) + problems.append(ValidationProblem(here, message, "invalid_value")) + continue + listed, nested = node_types[joined], node_type(entry) + if listed is not None and nested is not None and listed != nested: + message = ( + f"expected {_an(listed)}, as the group lists {shown(f'/{joined}')}, " + f"got {_an(nested)}" + ) + problems.append(ValidationProblem(here, message, "invalid_value")) + return problems + + +def _an(node_type: NodeType) -> str: + return "an array" if node_type == "array" else "a group" + + +def _model_node_type(node: ZarrV3ArrayMetadata | ZarrV3GroupMetadata) -> NodeType: + """The node type a model is.""" + return "array" if isinstance(node, ZarrV3ArrayMetadata) else "group" + + +def _model_listing( + node: ZarrV3ArrayMetadata | ZarrV3GroupMetadata, +) -> Mapping[str, ZarrV3ArrayMetadata | ZarrV3GroupMetadata]: + """What a model lists in its own consolidated metadata: nothing, for an array or a group with none.""" + if isinstance(node, ZarrV3GroupMetadata) and node.consolidated_metadata is not UNSET: + return node.consolidated_metadata.metadata + return {} + + +def _reading_listing(reading: ZarrV3NodeMetadataReading) -> Mapping[str, ZarrV3NodeMetadataReading]: + """What a reading lists in its own consolidated metadata: nothing, for an array's or one of no node type.""" + if isinstance(reading, ZarrV3GroupMetadataReading): + return reading.consolidated + return {} + + +def _hierarchy_problems(node_types: Mapping[str, NodeType | None]) -> list[ValidationProblem]: + """What keeps the documents consolidated metadata keeps and its group from making a hierarchy, the group its root, as `hierarchy_problems` judges one: each at the key it is about. + + `node_types` gives the node type of each document by its key, None for + a document of no node type the spec defines, and holds only keys + `_key_problems` finds nothing wrong with. + """ + nodes: dict[str, NodeType | None] = {"/": "group"} + nodes.update((f"/{key}", node_type) for key, node_type in node_types.items()) + return [ + ValidationProblem(("metadata", cast("str", found.loc[0])[1:]), found.message, found.kind) + for found in hierarchy_problems(nodes) + ] + + +def _node_type_of(reading: ZarrV3NodeMetadataReading) -> NodeType | None: + """The node type a document says it is, as its reading tells; None when it says none.""" + if isinstance(reading, ZarrV3ArrayMetadataReading): + return "array" + if isinstance(reading, ZarrV3GroupMetadataReading): + return "group" + return None + + +def _below_faults(path: str) -> list[str]: + """What keeps `path` from being the path of a node below a group, relative to the group, each said.""" + if path == "": + return ["is the group's own"] + if path.startswith("/"): + return ['starts with "/"'] + return path_faults(f"/{path}") + + def _with_models( reading: ZarrV3GroupMetadataReading, members: GroupMembersV3 ) -> ZarrV3GroupMetadataReading: diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_json_schema.py b/packages/zarr-metadata/src/zarr_metadata/model/_json_schema.py index 3b6b015fd2..227a5a1916 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_json_schema.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/_json_schema.py @@ -36,8 +36,9 @@ def node_metadata_json_schema_v3(*, context: Context = CORE_AND_EXTENSIONS) -> J A JSON Schema says what each member is, and what the rules say of members read together is not in it: one dimension name per dimension of the shape, a chunk grid that fits the shape, codecs in the order a - pipeline takes them, each against the chunk it is handed, and what a - definition's `rules` say. So a document it accepts may still have a + pipeline takes them, each against the chunk it is handed, the + hierarchy the documents of consolidated metadata make below their + group, and what a definition's `rules` say. So a document it accepts may still have a problem, and a JSON document `validate_node_metadata_v3` finds none with, it accepts. A validator reads JSON as a parser gives it, arrays as lists: a model's `to_json` writes tuples, which a Python validator diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/__init__.py b/packages/zarr-metadata/src/zarr_metadata/v3/__init__.py index 4e335f9573..9879db759f 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/__init__.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/__init__.py @@ -1,11 +1,14 @@ """Zarr v3 metadata types.""" from zarr_metadata.v3._common import ZarrV3MetadataFieldJSON +from zarr_metadata.v3._hierarchy import NodeName, NodePath from zarr_metadata.v3.array import ZarrV3ArrayMetadataJSON, ZarrV3ExtensionField from zarr_metadata.v3.consolidated import ZarrV3ConsolidatedMetadataJSON from zarr_metadata.v3.group import ZarrV3GroupMetadataJSON __all__ = [ + "NodeName", + "NodePath", "ZarrV3ArrayMetadataJSON", "ZarrV3ConsolidatedMetadataJSON", "ZarrV3ExtensionField", diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/_hierarchy.py b/packages/zarr-metadata/src/zarr_metadata/v3/_hierarchy.py new file mode 100644 index 0000000000..7b6cac919c --- /dev/null +++ b/packages/zarr-metadata/src/zarr_metadata/v3/_hierarchy.py @@ -0,0 +1,287 @@ +"""A Zarr v3 hierarchy: the names and paths of its nodes, and the tree they make. + +`NodeName` and `NodePath` are modelled on zarrs' types of those names, and +hold the spec's rules, which reserve `zarr.json` too. `hierarchy_problems` +judges the node type of each node of a hierarchy, by its path, as a tree. + +Every problem here is one per value, and says its reasons at once, so +what is reported stays proportional to what was read: a path of a +thousand bad names is one problem, not a thousand each repeating the +path. + +Private: consumers import `NodeName` and `NodePath` from `zarr_metadata.v3`, +and the validators from `zarr_metadata.model`. +""" + +from __future__ import annotations + +from typing import TYPE_CHECKING, Literal, NewType, TypeGuard, cast + +from zarr_metadata._json import MetadataValidationError, ValidationProblem, shown, with_input + +if TYPE_CHECKING: + from collections.abc import Mapping, Sequence + +NodeName = NewType("NodeName", str) +"""The name of a node in a Zarr v3 hierarchy. + +The root's is `""`. Any other is not empty, holds no `/`, is not periods +alone -- `.`, `..` -- does not start with the reserved `__`, and is not +`zarr.json`. Case matters: `foo` and `FOO` are two names +(https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v3/core/index.rst#L818-L837). +""" + +NodePath = NewType("NodePath", str) +"""The path of a node in a Zarr v3 hierarchy. + +The root's is `/`. Any other's is its parent's path, a `/` unless the +parent is the root, and its name: so a path starts with `/`, does not end +with one, and holds a node name between each two +(https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v3/core/index.rst#L211-L229). +""" + +NodeType = Literal["array", "group"] +"""The two kinds of node a hierarchy holds.""" + + +def said(faults: Sequence[str]) -> str: + """`faults`, each a reason, as one clause: `a`, `a and b`, `a, b and c`.""" + if len(faults) <= 1: + return "".join(faults) + return f"{', '.join(faults[:-1])} and {faults[-1]}" + + +def name_faults(name: str) -> list[str]: + """What keeps `name` from being a node name, each said; none for a node name, the root's `""` among them.""" + faults: list[str] = [] + if "/" in name: + faults.append('holds "/"') + if name != "" and set(name) == {"."}: + faults.append("is periods alone") + if name.startswith("__"): + faults.append('starts with the reserved "__"') + if name == "zarr.json": + faults.append('is the reserved "zarr.json"') + return faults + + +def names_faults(names: list[str]) -> list[str]: + """What keeps `names`, the names a path holds between its `/`, from being node names below the root: the first that is not one, said, and how many more there are.""" + faulty = [name for name in names if name == "" or len(name_faults(name)) != 0] + if len(faulty) == 0: + return [] + first = faulty[0] + faults = ( + ['holds an empty name between two "/"'] + if first == "" + else [f"holds {shown(first)}, a name that {said(name_faults(first))}"] + ) + if len(faulty) > 1: + more = len(faulty) - 1 + faults.append( + f"holds {more} more name{'s' if more != 1 else ''} that {'are' if more != 1 else 'is'} not a node name" + ) + return faults + + +def path_faults(path: str) -> list[str]: + """What keeps `path` from being a node path, each said; none for a node path, the root's `/` among them.""" + if path == "/": + return [] + if not path.startswith("/"): + return ['does not start with "/"'] + faults: list[str] = [] + names = path[1:].split("/") + if path.endswith("/"): + faults.append('ends with "/"') + names = names[:-1] + return [*faults, *names_faults(names)] + + +def hierarchy_problems(nodes: Mapping[str, NodeType | None]) -> tuple[ValidationProblem, ...]: + """What keeps `nodes`, the node type of each node by its path, from being a Zarr v3 hierarchy: one problem per node it is about, at that node's path. + + "A Zarr hierarchy is a tree structure, where each node in the tree is + either a group or an array. Group nodes may have children but array + nodes may not" + (https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v3/core/index.rst#L177-L181). + So each path is a node path, and each node but the root has its parent + among them, a group: a node below an array is an `invalid_value` at its + path, and the nearest group missing above a node is a `missing_key` at + that group's path, once, saying how many more are missing above it, the + root's among them. A node of no type known, None, is taken as a group: + what makes its type unknown is its own problem, reported where it sits. + A mapping holds each path once, so no two siblings share a name. + """ + problems: list[ValidationProblem] = [] + well_formed: list[str] = [] + for path in nodes: + faults = path_faults(path) + if len(faults) == 0: + well_formed.append(path) + else: + message = f"expected a node path, got {shown(path)}, which {said(faults)}" + problems.append(ValidationProblem((path,), message, "invalid_value")) + tree = _Tree() + for path in well_formed: + tree.hold(path, nodes[path]) + reported: set[str] = set() + for path in well_formed: + if path == "/": + continue + holder, missing = tree.above(path) + if holder == "array": + # No node is below an array, however many groups are missing + # between them. + message = ( + f"expected a node below a group, got {shown(path)}, below the array " + f"{shown(tree.holder_of(path))}" + ) + problems.append(ValidationProblem((path,), message, "invalid_value")) + elif missing != 0: + nearest = path.rpartition("/")[0] or "/" + if nearest not in reported: + reported.add(nearest) + above = missing - 1 + more = ( + "" if above == 0 else f", and {above} group{'s' if above != 1 else ''} above it" + ) + message = f"missing the group holding {shown(path)}{more}" + problems.append(ValidationProblem((nearest,), message, "missing_key")) + return tuple(problems) + + +class _Branch: + """A name in the tree of held paths: whether a node is held there, its type, and the names below it.""" + + __slots__ = ("children", "held", "node_type") + + def __init__(self) -> None: + self.held = False + self.node_type: NodeType | None = None + self.children: dict[str, _Branch] = {} + + +class _Tree: + """The held paths as a tree of their names, so each path is walked once, in time proportional to its length. + + Walking up a path by its ancestors' strings costs the path's length + for each ancestor, which a hostile key of a million characters makes + a quadratic wait; here a path is split once and walked name by name. + """ + + __slots__ = ("root",) + + def __init__(self) -> None: + self.root = _Branch() + + def hold(self, path: str, node_type: NodeType | None) -> None: + """Mark the node at `path`, a node path, as held, of `node_type`.""" + branch = self.root + for name in _names(path): + branch = branch.children.setdefault(name, _Branch()) + branch.held = True + branch.node_type = node_type + + def above(self, path: str) -> tuple[NodeType | None, int]: + """Of the node at `path`, a node path below the root: the type of the nearest held node above it, `"group"` for one of no type known and None when none is held, and how many groups are missing between them.""" + names = _names(path) + branch: _Branch | None = self.root + holder: NodeType | None = None + held_depth = -1 + for depth, name in enumerate(names[:-1], start=0): + if branch is not None and branch.held: + holder, held_depth = branch.node_type or "group", depth + branch = None if branch is None else branch.children.get(name) + # The parent, at the last depth walked, may be held too. + if branch is not None and branch.held: + holder, held_depth = branch.node_type or "group", len(names) - 1 + return holder, len(names) - 1 - held_depth + + def holder_of(self, path: str) -> str: + """The path of the nearest held node above the node at `path`, a node path below the root that has one.""" + names = _names(path) + branch: _Branch | None = self.root + holder = "/" + for depth, name in enumerate(names[:-1]): + if branch is not None and branch.held: + holder = "/" + "/".join(names[:depth]) if depth != 0 else "/" + branch = None if branch is None else branch.children.get(name) + if branch is not None and branch.held: + holder = "/" + "/".join(names[:-1]) + return holder + + +def _names(path: str) -> list[str]: + """The names a node path holds between its `/`: none for the root's.""" + return [] if path == "/" else path[1:].split("/") + + +def _string_problems(value: object, what: str) -> tuple[ValidationProblem, ...]: + return (ValidationProblem((), f"expected {what}, got {shown(value)}", "invalid_type"),) + + +def validate_node_name_v3(value: object) -> tuple[ValidationProblem, ...]: + """Every reason `value` is not a v3 node name, said in one `invalid_value`, or an `invalid_type` for what is not a string.""" + if not isinstance(value, str): + return with_input(_string_problems(value, "a node name"), value) + faults = name_faults(value) + if len(faults) == 0: + return () + message = f"expected a node name, got {shown(value)}, which {said(faults)}" + return with_input((ValidationProblem((), message, "invalid_value"),), value) + + +def is_node_name_v3(value: object) -> TypeGuard[NodeName]: + """Whether `value` is a v3 node name `validate_node_name_v3` finds nothing wrong with.""" + return len(validate_node_name_v3(value)) == 0 + + +def parse_node_name_v3(value: object) -> NodeName: + """`value` as a `NodeName`, or `MetadataValidationError` with every reason it is not one.""" + problems = validate_node_name_v3(value) + if len(problems) != 0: + raise MetadataValidationError(problems) + return NodeName(cast("str", value)) + + +def validate_node_path_v3(value: object) -> tuple[ValidationProblem, ...]: + """Every reason `value` is not a v3 node path, said in one `invalid_value`, or an `invalid_type` for what is not a string.""" + if not isinstance(value, str): + return with_input(_string_problems(value, "a node path"), value) + faults = path_faults(value) + if len(faults) == 0: + return () + message = f"expected a node path, got {shown(value)}, which {said(faults)}" + return with_input((ValidationProblem((), message, "invalid_value"),), value) + + +def is_node_path_v3(value: object) -> TypeGuard[NodePath]: + """Whether `value` is a v3 node path `validate_node_path_v3` finds nothing wrong with.""" + return len(validate_node_path_v3(value)) == 0 + + +def parse_node_path_v3(value: object) -> NodePath: + """`value` as a `NodePath`, or `MetadataValidationError` with every reason it is not one.""" + problems = validate_node_path_v3(value) + if len(problems) != 0: + raise MetadataValidationError(problems) + return NodePath(cast("str", value)) + + +__all__ = [ + "NodeName", + "NodePath", + "NodeType", + "hierarchy_problems", + "is_node_name_v3", + "is_node_path_v3", + "name_faults", + "names_faults", + "parse_node_name_v3", + "parse_node_path_v3", + "path_faults", + "said", + "validate_node_name_v3", + "validate_node_path_v3", +] diff --git a/packages/zarr-metadata/tests/model/test_construction.py b/packages/zarr-metadata/tests/model/test_construction.py index 8190d57859..29a29402ec 100644 --- a/packages/zarr-metadata/tests/model/test_construction.py +++ b/packages/zarr-metadata/tests/model/test_construction.py @@ -52,6 +52,7 @@ "metadata": { "x": ARRAY.to_json(), "y": {"zarr_format": 3, "node_type": "group"}, + "y/z": ARRAY.to_json(), }, }, } @@ -186,6 +187,15 @@ def counted( (GROUP, {"attributes": {1: "a"}}, [(("attributes",), "invalid_type")]), (GROUP, {"attributes": ()}, [(("attributes",), "invalid_type")]), (GROUP, {"extra_fields": {"acme": math.nan}}, [(("acme",), "invalid_value")]), + ( + GROUP.consolidated_metadata, + {"metadata": {"x": ARRAY, "x/a": ARRAY, "__b": ARRAY, "c/d": ARRAY}}, + [ + (("metadata", "__b"), "invalid_value"), + (("metadata", "x/a"), "invalid_value"), + (("metadata", "c"), "missing_key"), + ], + ), (V2_ARRAY, {"order": "Q"}, [(("order",), "invalid_value")]), (V2_ARRAY, {"chunks": (4, 4)}, [(("chunks",), "invalid_value")]), (V2_GROUP, {"attributes": {1: "a"}}, [(("attributes",), "invalid_type")]), @@ -206,6 +216,7 @@ def counted( "group-attribute-key", "group-attributes-empty-and-no-object", "group-extension-not-json", + "consolidated-paths", "v2-order", "v2-chunks-for-another-rank", "v2-group-attribute-key", diff --git a/packages/zarr-metadata/tests/model/test_group.py b/packages/zarr-metadata/tests/model/test_group.py index 3563badf0d..3542a3c24e 100644 --- a/packages/zarr-metadata/tests/model/test_group.py +++ b/packages/zarr-metadata/tests/model/test_group.py @@ -501,19 +501,35 @@ def _fields_of_an_array(*at: str | int) -> list[tuple[str | int, ...]]: ["a", "g"], _fields_of_an_array(*A), ), - # A group's consolidated metadata in a group's: each field located - # from the root of the outer document. + # Every node below the group, each below a group, at its path. + ( + _group(consolidated_metadata=_inline(g=_group(), **{"g/b": _array()})), + ["g", "g/b"], + _fields_of_an_array("consolidated_metadata", "metadata", "g/b"), + ), + # A group's consolidated metadata in a group's, listing what the group + # lists too: each field located from the root of the outer document. ( _group( - consolidated_metadata=_inline(g=_group(consolidated_metadata=_inline(b=_array()))) - ), - ["g"], - _fields_of_an_array( - "consolidated_metadata", "metadata", "g", "consolidated_metadata", "metadata", "b" + consolidated_metadata=_inline( + g=_group(consolidated_metadata=_inline(b=_array())), **{"g/b": _array()} + ) ), + ["g", "g/b"], + [ + *_fields_of_an_array( + "consolidated_metadata", + "metadata", + "g", + "consolidated_metadata", + "metadata", + "b", + ), + *_fields_of_an_array("consolidated_metadata", "metadata", "g/b"), + ], ), ], - ids=["no-consolidated-metadata", "null", "an-array-and-a-group", "nested"], + ids=["no-consolidated-metadata", "null", "an-array-and-a-group", "paths", "nested"], ) def test_a_group_reads_each_document_its_consolidated_metadata_holds( document: dict[str, object], paths: list[str], locs: list[tuple[str | int, ...]] @@ -531,6 +547,166 @@ def test_a_group_reads_each_document_its_consolidated_metadata_holds( assert all(held[path] is reading.consolidated[path].metadata for path in paths) +@pytest.mark.parametrize( + ("path", "fault"), + [ + ("", "is the group's own"), + ("/a", 'starts with "/"'), + ("a/", 'ends with "/"'), + ("a//b", 'holds an empty name between two "/"'), + (".", 'holds ".", a name that is periods alone'), + ("a/../b", 'holds "..", a name that is periods alone'), + ("__a", 'holds "__a", a name that starts with the reserved "__"'), + ("zarr.json", 'holds "zarr.json", a name that is the reserved "zarr.json"'), + ], +) +def test_error_a_document_in_consolidated_metadata_is_at_a_node_s_path_below_the_group( + path: str, fault: str +) -> None: + # Its node names, joined by "/", as the reference implementation keeps + # it: the group's own path, "/", and it make the node's. + document = _group(consolidated_metadata=_inline(**{path: _group()})) + message = f"expected the path of a node below the group, got {json.dumps(path)}, which {fault}" + assert validate_group_metadata_v3(document) == ( + ValidationProblem(("consolidated_metadata", "metadata", path), message, "invalid_value"), + ) + + +@pytest.mark.parametrize( + ("documents", "path"), + [ + ({"a": _array(), "a/b": _group()}, "a/b"), + # However many groups are missing between them: none would help. + ({"a": _array(), "a/b/c": _array()}, "a/b/c"), + ], + ids=["child", "descendant"], +) +def test_error_no_document_in_consolidated_metadata_is_below_an_array( + documents: dict[str, object], path: str +) -> None: + # "Group nodes may have children but array nodes may not." A message + # names each node by its path in the hierarchy below the group. + document = _group(consolidated_metadata=_inline(**documents)) + message = f'expected a node below a group, got "/{path}", below the array "/a"' + assert validate_group_metadata_v3(document) == ( + ValidationProblem(("consolidated_metadata", "metadata", path), message, "invalid_value"), + ) + + +def test_a_document_of_no_node_type_is_taken_as_a_group_s() -> None: + # Its own problem is reported where it sits, and the nodes below it are + # not refused for it. + document = _group( + consolidated_metadata=_inline(a={"zarr_format": 3, "node_type": "x"}, **{"a/b": _array()}) + ) + assert [(p.loc, p.kind) for p in validate_group_metadata_v3(document)] == [ + (("consolidated_metadata", "metadata", "a", "node_type"), "invalid_value") + ] + + +def test_error_consolidated_metadata_holds_the_group_holding_each_document() -> None: + # The nearest group missing above a node, once, counting those above it + # up to the group holding the documents. + document = _group(consolidated_metadata=_inline(**{"a/b/c": _array(), "a/b/d": _array()})) + assert [(p.loc, p.message, p.kind) for p in validate_group_metadata_v3(document)] == [ + ( + ("consolidated_metadata", "metadata", "a/b"), + 'missing the group holding "/a/b/c", and 1 group above it', + "missing_key", + ), + ] + + +def test_error_a_consolidated_metadata_key_that_is_not_a_string() -> None: + listing: dict[object, object] = {"a": _array(), 1: _array()} + document = _group(consolidated_metadata={**_inline(), "metadata": listing}) + assert [(p.loc, p.kind) for p in validate_group_metadata_v3(document)] == [ + (("consolidated_metadata", "metadata"), "invalid_type") + ] + + +NESTED = ("consolidated_metadata", "metadata", "g", "consolidated_metadata", "metadata") +"""Where the own listing of the group at `g` sits in the outer document.""" + + +@pytest.mark.parametrize( + ("listed", "flat", "expected"), + [ + # What a listed group lists itself is what the group lists, too. + ({"b": _array()}, {"g/b": _array()}, []), + # A node the listed group lists alone would be dropped by the + # reference reader, which keeps the flat listing. + ( + {"b": _array()}, + {}, + [ + ( + (*NESTED, "b"), + 'expected a node the group lists, got "/g/b", which "/g" lists alone', + "invalid_value", + ) + ], + ), + # Nor may the two listings disagree on what a node is. + ( + {"b": _array()}, + {"g/b": _group()}, + [ + ( + (*NESTED, "b"), + 'expected a group, as the group lists "/g/b", got an array', + "invalid_value", + ) + ], + ), + # A deeper listing is the listed group's own to judge, when its + # document is read: its problem is that document's, at its place. + ( + {"h": _group(consolidated_metadata=_inline(x=_array()))}, + {"g/h": _group()}, + [ + ( + (*NESTED, "h", "consolidated_metadata", "metadata", "x"), + 'expected a node the group lists, got "/h/x", which "/h" lists alone', + "invalid_value", + ) + ], + ), + ], + ids=["agreeing", "listed-alone", "contradicting", "deeper"], +) +def test_a_listed_group_s_own_listing_lists_what_the_group_lists( + listed: dict[str, object], + flat: dict[str, object], + expected: list[tuple[tuple[str, ...], str, str]], +) -> None: + document = _group( + consolidated_metadata=_inline(g=_group(consolidated_metadata=_inline(**listed)), **flat) + ) + problems = validate_group_metadata_v3(document) + assert [(p.loc, p.message, p.kind) for p in problems] == expected + # The constructor refuses what the reader reports, built of models: a + # listed group whose own listing is wrong is refused as it is built. + listing = {"g": _group(consolidated_metadata=_inline(**listed)), **flat} + deeper = [(loc, message, kind) for loc, message, kind in expected if len(loc) > len(NESTED) + 1] + if len(deeper) != 0: + with pytest.raises(MetadataValidationError) as inner: + node_metadata_from_json_v3(listing["g"]) + assert [(p.loc, p.message, p.kind) for p in inner.value.problems] == [ + (loc[3:], message, kind) for loc, message, kind in deeper + ] + return + members = {key: node_metadata_from_json_v3(value) for key, value in listing.items()} + if expected == []: + assert ZarrV3ConsolidatedMetadata(metadata=members).metadata.keys() == members.keys() + return + with pytest.raises(MetadataValidationError) as raised: + ZarrV3ConsolidatedMetadata(metadata=members) + assert [(p.loc[1:], p.message, p.kind) for p in raised.value.problems] == [ + (loc[2:], message, kind) for loc, message, kind in expected + ] + + def test_each_document_its_consolidated_metadata_holds_is_read_once() -> None: reads: list[GzipCodecConfiguration] = [] diff --git a/packages/zarr-metadata/tests/test_problem_data.py b/packages/zarr-metadata/tests/test_problem_data.py index f6c1bfa026..9541756ed0 100644 --- a/packages/zarr-metadata/tests/test_problem_data.py +++ b/packages/zarr-metadata/tests/test_problem_data.py @@ -34,6 +34,8 @@ validate_json, validate_metadata_field_v3, validate_node_metadata_v3, + validate_node_name_v3, + validate_node_path_v3, ) from zarr_metadata.typed_json import check from zarr_metadata.v3.codec.gzip import GZIP_CODEC @@ -223,6 +225,8 @@ def _raised(read: Callable[[], object]) -> Sequence[ValidationProblem]: validate_metadata_field_v3, {"name": 1, "configuration": {"a": [math.nan]}, "extra": [1]}, ), + (validate_node_name_v3, "__/"), + (validate_node_path_v3, "a//"), (validate_array_metadata_v3, BAD_ARRAY), (lambda value: read_array_metadata_v3(value).problems, BAD_ARRAY), (lambda value: _raised(lambda: parse_array_metadata_v3(value)), BAD_ARRAY), @@ -236,7 +240,7 @@ def _raised(read: Callable[[], object]) -> Sequence[ValidationProblem]: "consolidated_metadata": { "kind": "inline", "must_understand": [False], - "metadata": {"a": BAD_ARRAY}, + "metadata": {"a": BAD_ARRAY, "__b": BAD_ARRAY, "a/c": BAD_ARRAY, "d/e": BAD_ARRAY}, }, }, ), @@ -266,6 +270,8 @@ def _raised(read: Callable[[], object]) -> Sequence[ValidationProblem]: "fill-value-problems", "validate-json", "validate-metadata-field", + "validate-node-name", + "validate-node-path", "validate-array", "read-array", "parse-array", diff --git a/packages/zarr-metadata/tests/test_public_api.py b/packages/zarr-metadata/tests/test_public_api.py index 82448e356c..03e98d268f 100644 --- a/packages/zarr-metadata/tests/test_public_api.py +++ b/packages/zarr-metadata/tests/test_public_api.py @@ -291,6 +291,8 @@ def test_all_is_grouped_and_unique() -> None: "Lengths", "Loc", "Nested", + "NodeName", + "NodePath", # What a scope made of a field: `Read` by the definition that claims # its name, `Unclaimed`, or `Refused`; `Resolved` is the three. "Read", diff --git a/packages/zarr-metadata/tests/v3/test_hierarchy.py b/packages/zarr-metadata/tests/v3/test_hierarchy.py new file mode 100644 index 0000000000..4b9944bf2b --- /dev/null +++ b/packages/zarr-metadata/tests/v3/test_hierarchy.py @@ -0,0 +1,256 @@ +"""A Zarr v3 hierarchy: the names and paths of its nodes, and the tree they make, as the core specification constrains them. + +Names: +https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v3/core/index.rst#L818-L837 +Paths: +https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v3/core/index.rst#L211-L229 +The tree: +https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v3/core/index.rst#L177-L181 +""" + +from __future__ import annotations + +import time +from typing import TYPE_CHECKING, Literal, cast + +import pytest + +from zarr_metadata.model import ( + MetadataValidationError, + ValidationProblem, + is_node_name_v3, + is_node_path_v3, + parse_node_name_v3, + parse_node_path_v3, + validate_node_name_v3, + validate_node_path_v3, +) +from zarr_metadata.v3._hierarchy import hierarchy_problems + +if TYPE_CHECKING: + from collections.abc import Callable + + +@pytest.mark.parametrize( + ("value", "message"), + [ + # The root's name, and names of any code points but the reserved. + ("", None), + ("a", None), + ("FOO", None), + ("a.b", None), + ("...a", None), + ("_a", None), + ("a__", None), + ("zarr.json.bak", None), + ("ñ", None), + ("/", 'expected a node name, got "/", which holds "/"'), + ("a/b", 'expected a node name, got "a/b", which holds "/"'), + (".", 'expected a node name, got ".", which is periods alone'), + ("..", 'expected a node name, got "..", which is periods alone'), + ("__a", 'expected a node name, got "__a", which starts with the reserved "__"'), + ("zarr.json", 'expected a node name, got "zarr.json", which is the reserved "zarr.json"'), + # Every reason a name is not one, in one problem. + ( + "__/", + 'expected a node name, got "__/", which holds "/" and starts with the reserved "__"', + ), + ], +) +def test_node_names(value: str, message: str | None) -> None: + problems = validate_node_name_v3(value) + assert [problem.message for problem in problems] == ([] if message is None else [message]) + assert {(problem.loc, problem.kind, problem.input) for problem in problems} <= { + ((), "invalid_value", value) + } + assert is_node_name_v3(value) is (message is None) + if message is None: + assert parse_node_name_v3(value) == value + + +@pytest.mark.parametrize( + ("value", "message"), + [ + # The root's path, and the paths below it. + ("/", None), + ("/a", None), + ("/a/b", None), + ("/a/.b/c..", None), + ("", 'expected a node path, got "", which does not start with "/"'), + ("a/b", 'expected a node path, got "a/b", which does not start with "/"'), + ("/a/", 'expected a node path, got "/a/", which ends with "/"'), + ( + "//", + 'expected a node path, got "//", which ends with "/" and holds an empty name between two "/"', + ), + ("/a//b", 'expected a node path, got "/a//b", which holds an empty name between two "/"'), + ( + "/a/../b", + 'expected a node path, got "/a/../b", which holds "..", a name that is periods alone', + ), + ( + "/__a", + ( + 'expected a node path, got "/__a", which holds "__a", a name that starts with the ' + 'reserved "__"' + ), + ), + ( + "/a/zarr.json", + ( + 'expected a node path, got "/a/zarr.json", which holds "zarr.json", a name that is ' + 'the reserved "zarr.json"' + ), + ), + # Every reason a path is not one, in one problem: the first name that + # is not a node name is said, and the rest counted. + ( + "/./__a/", + ( + 'expected a node path, got "/./__a/", which ends with "/", holds ".", a name that is ' + "periods alone and holds 1 more name that is not a node name" + ), + ), + ], +) +def test_node_paths(value: str, message: str | None) -> None: + problems = validate_node_path_v3(value) + assert [problem.message for problem in problems] == ([] if message is None else [message]) + assert {(problem.loc, problem.kind, problem.input) for problem in problems} <= { + ((), "invalid_value", value) + } + assert is_node_path_v3(value) is (message is None) + if message is None: + assert parse_node_path_v3(value) == value + + +def test_error_parse_node_name_raises_every_reason_a_string_is_not_a_node_name() -> None: + with pytest.raises(MetadataValidationError) as raised: + parse_node_name_v3("__/") + assert raised.value.problems == validate_node_name_v3("__/") + assert len(raised.value.problems) == 1 + assert 'holds "/"' in raised.value.problems[0].message + assert 'starts with the reserved "__"' in raised.value.problems[0].message + + +def test_error_parse_node_path_raises_every_reason_a_string_is_not_a_node_path() -> None: + with pytest.raises(MetadataValidationError) as raised: + parse_node_path_v3("/./__a/") + assert raised.value.problems == validate_node_path_v3("/./__a/") + assert len(raised.value.problems) == 1 + assert 'ends with "/"' in raised.value.problems[0].message + assert "1 more name" in raised.value.problems[0].message + + +@pytest.mark.parametrize( + ("validate", "parse", "what"), + [ + (validate_node_name_v3, parse_node_name_v3, "a node name"), + (validate_node_path_v3, parse_node_path_v3, "a node path"), + ], + ids=["name", "path"], +) +def test_error_a_node_name_or_path_is_a_string( + validate: Callable[[object], tuple[ValidationProblem, ...]], + parse: Callable[[object], str], + what: str, +) -> None: + problems = validate(3) + assert problems == (ValidationProblem((), f"expected {what}, got 3", "invalid_type"),) + assert problems[0].input == 3 + with pytest.raises(MetadataValidationError) as raised: + parse(b"a") + assert [(problem.loc, problem.kind) for problem in raised.value.problems] == [ + ((), "invalid_type") + ] + + +@pytest.mark.parametrize( + "nodes", + [ + # No nodes, a root alone, and a tree of groups whose leaves are + # arrays; the root may be an array, which is then the hierarchy. + {}, + {"/": "group"}, + {"/": "array"}, + {"/": "group", "/a": "array", "/g": "group", "/g/b": "array", "/g/h": "group"}, + # A node of no type known is taken as a group. + {"/": "group", "/x": None, "/x/a": "array"}, + ], + ids=["empty", "root-group", "root-array", "tree", "unknown-type"], +) +def test_a_hierarchy(nodes: dict[str, Literal["array", "group"] | None]) -> None: + assert hierarchy_problems(nodes) == () + + +def test_error_a_hierarchy_holds_its_nodes_at_node_paths() -> None: + assert hierarchy_problems({"/": "group", "a": "array", "/b/": "array"}) == ( + ValidationProblem( + ("a",), 'expected a node path, got "a", which does not start with "/"', "invalid_value" + ), + ValidationProblem( + ("/b/",), 'expected a node path, got "/b/", which ends with "/"', "invalid_value" + ), + ) + + +@pytest.mark.parametrize( + ("nodes", "path"), + [ + ({"/": "group", "/a": "array", "/a/b": "group"}, "/a/b"), + # However many groups are missing between them: none would help. + ({"/": "group", "/a": "array", "/a/b/c": "array"}, "/a/b/c"), + # The root is an array: the hierarchy is that array alone. + ({"/": "array", "/a": "group"}, "/a"), + ], + ids=["child", "descendant", "below-the-root"], +) +def test_error_no_node_is_below_an_array( + nodes: dict[str, Literal["array", "group"] | None], path: str +) -> None: + holder = next(ancestor for ancestor in ("/a", "/") if nodes.get(ancestor) == "array") + message = f'expected a node below a group, got "{path}", below the array "{holder}"' + assert hierarchy_problems(nodes) == (ValidationProblem((path,), message, "invalid_value"),) + + +def test_error_a_hierarchy_holds_the_group_holding_each_node() -> None: + # The nearest group missing above a node, once, counting those above it: + # the root's among them. + assert hierarchy_problems({"/a/b/c": "array", "/a/b/d": "array"}) == ( + ValidationProblem( + ("/a/b",), 'missing the group holding "/a/b/c", and 2 groups above it', "missing_key" + ), + ) + assert hierarchy_problems({"/": "group", "/a/b": "array"}) == ( + ValidationProblem(("/a",), 'missing the group holding "/a/b"', "missing_key"), + ) + + +def test_problems_stay_proportional_to_the_path() -> None: + # One problem per value, however many names are wrong and however many + # groups are missing, so a hostile path costs what it weighs. + path = "/" * 10_000 + problems = validate_node_path_v3(path) + assert len(problems) == 1 + assert len(problems[0].message) < len(path) + 200 + deep = "/a" * 5_000 + found = hierarchy_problems({deep: "array"}) + assert [(problem.loc, problem.kind) for problem in found] == [ + ((deep.rpartition("/")[0],), "missing_key") + ] + assert found[0].message.endswith(", and 4999 groups above it") + assert len(found[0].message) < len(deep) + 200 + + +def test_a_hierarchy_is_walked_in_time_proportional_to_its_paths() -> None: + # Each path split once and walked name by name: a key of a million + # characters, and thousands of keys under one long missing prefix, + # each cost what they weigh, where walking up by ancestor strings + # costs the square. + started = time.perf_counter() + long = "/a" * 500_000 + assert len(hierarchy_problems({long: "array"})) == 1 + prefix = "/a" * 5_000 + many = {f"{prefix}/{index}": "array" for index in range(2_000)} + assert len(hierarchy_problems(cast("dict[str, Literal['array', 'group'] | None]", many))) == 1 + assert time.perf_counter() - started < 10 From f5753be1d770fab5fad38efdc7b11740768be640 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Tue, 29 Sep 2026 19:03:36 +0200 Subject: [PATCH 14/94] fix(zarr-metadata): two models are equal when they mean the same document A v3 array model compared its members with ==, so "NaN" and "0x7fc00000" were two float32 fill values, 0.0 and -0.0 one, a blosc with and without the typesize that noshuffle ignores two codecs though canonical_of spelled them alike, and a model holding a NaN attribute never equalled its own copy. Two models are now equal when they mean the same document: what the package interprets compares by its canonical spelling, and what it does not compares as JSON text, which tells true from 1 and -0.0 from 0.0 and takes NaN for itself. A data type's definition says which spelling of a fill value is its value's own, DataTypeDefinition.fill_value_canonical, and canonical_fill_value spells one, or gives UNSET for a fill value with a problem: the float types read a number as a float64, as JSON parsers and numpy do, and spell a value by its shortest number, a named value, or a NaN's bits; complex by part; the numpy time types' -2**63 as "NaT"; bytes as base64; a struct field by field. A field, Read, Unclaimed or Refused, compares by field_key: its definition and its canonical configuration as text, with the fields it holds by their own keys; default, v2, sharding_indexed and zstd gain a canonical folding their spec defaults. Every model and field hashes as it compares. Assisted-by: ClaudeCode:claude-opus-5-5 Assisted-by: ClaudeCode:claude-fable-5-1 --- packages/zarr-metadata/README.md | 14 ++ .../zarr-metadata/changes/381.bugfix.3.md | 14 ++ .../zarr-metadata/changes/381.feature.1.md | 10 ++ packages/zarr-metadata/docs/index.md | 14 ++ .../zarr-metadata/src/zarr_metadata/_json.py | 10 ++ .../src/zarr_metadata/model/_array.py | 66 +++++++ .../src/zarr_metadata/model/_group.py | 58 +++++++ .../src/zarr_metadata/v3/_definition.py | 164 +++++++++++++++++- .../v3/chunk_key_encoding/default.py | 21 ++- .../zarr_metadata/v3/chunk_key_encoding/v2.py | 19 +- .../v3/codec/sharding_indexed.py | 20 ++- .../src/zarr_metadata/v3/codec/zstd.py | 19 +- .../src/zarr_metadata/v3/data_type/_float.py | 147 +++++++++++++++- .../zarr_metadata/v3/data_type/_numpy_time.py | 20 +++ .../src/zarr_metadata/v3/data_type/bytes.py | 16 ++ .../zarr_metadata/v3/data_type/complex128.py | 6 +- .../zarr_metadata/v3/data_type/complex64.py | 6 +- .../src/zarr_metadata/v3/data_type/float16.py | 7 +- .../src/zarr_metadata/v3/data_type/float32.py | 7 +- .../src/zarr_metadata/v3/data_type/float64.py | 7 +- .../v3/data_type/numpy_datetime64.py | 2 + .../v3/data_type/numpy_timedelta64.py | 2 + .../src/zarr_metadata/v3/data_type/struct.py | 19 ++ .../src/zarr_metadata/v3/definition.py | 11 ++ .../zarr-metadata/tests/model/test_array.py | 143 +++++++++++++++ .../tests/v3/test_definitions.py | 82 ++++++++- .../tests/v3/test_every_definition.py | 44 ++++- .../tests/v3/test_fill_values.py | 154 +++++++++++++++- 28 files changed, 1076 insertions(+), 26 deletions(-) create mode 100644 packages/zarr-metadata/changes/381.bugfix.3.md create mode 100644 packages/zarr-metadata/changes/381.feature.1.md diff --git a/packages/zarr-metadata/README.md b/packages/zarr-metadata/README.md index 42480c7ee4..75bae98677 100644 --- a/packages/zarr-metadata/README.md +++ b/packages/zarr-metadata/README.md @@ -158,6 +158,20 @@ read. A model a read builds is not read a second time. Change a model by building another: a container it holds, changed in place, is not checked again. +Two models are equal when they mean the same document, however each is +spelled. What the package interprets -- each field, and the fill value +against its data type -- compares by its canonical spelling, as +`canonical_of` and `canonical_fill_value` give it: `"NaN"` and +`"0x7fc00000"` are one `float32` fill value, `0.0` and `-0.0` two, and a +blosc with and without the `typesize` that `noshuffle` ignores one +codec. What it does not interpret -- attributes, extra fields, the +configuration of a field nothing in scope claims, and every member of a +v2 document -- compares as JSON text, which tells `true` from `1` and +`-0.0` from `0.0`, and takes `NaN` for itself. Equal models hash alike, +and may write two documents: `to_json` writes each as it was given. A +model's hash is of what its containers held when it was hashed, so a +model in a set, or a key of a dict, is not changed in place. + `node_metadata_json_schema_v3` writes what the validators read as a JSON Schema, draft 2020-12, for an editor that checks a `zarr.json` as it is written, or a validator in another language. Each extension point is diff --git a/packages/zarr-metadata/changes/381.bugfix.3.md b/packages/zarr-metadata/changes/381.bugfix.3.md new file mode 100644 index 0000000000..bc3f7ef3ed --- /dev/null +++ b/packages/zarr-metadata/changes/381.bugfix.3.md @@ -0,0 +1,14 @@ +Two models are equal when they mean the same document, however each is +spelled. What the package interprets compares by its canonical spelling: +`"NaN"` and `"0x7fc00000"` were two `float32` fill values and `0.0` and +`-0.0` one, and a blosc with and without the `typesize` that `noshuffle` +ignores, a key encoding with and without its default separator, a shard +with and without `index_location: "end"` and a zstd with and without +`checksum: false` were two fields each, though `canonical_of` spelled +them alike; each is one now, and `default`, `v2`, `sharding_indexed` +and `zstd` fold their spec defaults in a `canonical` of their own. What +it does not interpret -- attributes, extra fields, unclaimed +configurations, every member of a v2 document -- compares as JSON text, +which tells `true` from `1` and `-0.0` from `0.0` and takes `NaN` for +itself, so a model holding a NaN attribute equals its own copy. Equal +fields and models hash alike. diff --git a/packages/zarr-metadata/changes/381.feature.1.md b/packages/zarr-metadata/changes/381.feature.1.md new file mode 100644 index 0000000000..48a0348c8b --- /dev/null +++ b/packages/zarr-metadata/changes/381.feature.1.md @@ -0,0 +1,10 @@ +A data type's definition says which spelling of a fill value is its +value's own: `DataTypeDefinition.fill_value_canonical`, and +`canonical_fill_value(data_type, value)` in `zarr_metadata.v3.definition` +spells one, or gives `UNSET` for a fill value with a problem. The float +types read a number as a float64, as JSON parsers and numpy do, and +spell a value by its shortest number, a named value, or the bits of a +NaN the spec does not name; the complex types spell each part so; the +numpy time types spell `-2**63` as `"NaT"`; `bytes` spells its bytes as +base64; and a struct spells each field's fill value by that field's +type. diff --git a/packages/zarr-metadata/docs/index.md b/packages/zarr-metadata/docs/index.md index 6597cedc87..7e99d31144 100644 --- a/packages/zarr-metadata/docs/index.md +++ b/packages/zarr-metadata/docs/index.md @@ -173,6 +173,20 @@ read. A model a read builds is not read a second time. Change a model by building another: a container it holds, changed in place, is not checked again. +Two models are equal when they mean the same document, however each is +spelled. What the package interprets -- each field, and the fill value +against its data type -- compares by its canonical spelling, as +`canonical_of` and `canonical_fill_value` give it: `"NaN"` and +`"0x7fc00000"` are one `float32` fill value, `0.0` and `-0.0` two, and a +blosc with and without the `typesize` that `noshuffle` ignores one +codec. What it does not interpret -- attributes, extra fields, the +configuration of a field nothing in scope claims, and every member of a +v2 document -- compares as JSON text, which tells `true` from `1` and +`-0.0` from `0.0`, and takes `NaN` for itself. Equal models hash alike, +and may write two documents: `to_json` writes each as it was given. A +model's hash is of what its containers held when it was hashed, so a +model in a set, or a key of a dict, is not changed in place. + `node_metadata_json_schema_v3` writes what the validators read as a JSON Schema, draft 2020-12, for an editor that checks a `zarr.json` as it is written, or a validator in another language. Each extension point is diff --git a/packages/zarr-metadata/src/zarr_metadata/_json.py b/packages/zarr-metadata/src/zarr_metadata/_json.py index 99c66e5b5c..de7ac9b92c 100644 --- a/packages/zarr-metadata/src/zarr_metadata/_json.py +++ b/packages/zarr-metadata/src/zarr_metadata/_json.py @@ -334,6 +334,16 @@ def _refine(value: object, loc: tuple[str | int, ...], *, finite: bool) -> _Refi ) +def json_text(value: JSONValue) -> str: + """`value` as the JSON text `json.dumps` writes for it, an object's keys sorted: what `==` compares of a JSON value the package does not interpret. + + Not Python's `==` on the value, which takes `true` for `1`, `-0.0` for + `0.0` and `NaN` for no value at all, but what a document writes: two + values written alike are one value to every reader. + """ + return json.dumps(value, sort_keys=True, ensure_ascii=False) + + def shown(value: object) -> str: """`value` as a problem's message shows it: as the JSON a document writes, `null` and `[1, 2]`, or by its repr when it is not JSON.""" refined, problems = _refine(value, (), finite=False) diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_array.py b/packages/zarr-metadata/src/zarr_metadata/model/_array.py index 20f743471a..9c65362e5e 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_array.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/_array.py @@ -14,6 +14,7 @@ MetadataValidationError, ValidationProblem, copied, + json_text, with_input, ) from zarr_metadata._sentinel import UNSET @@ -42,6 +43,8 @@ StorageTransformerDefinition, Unclaimed, document_json, + field_key, + spelled_canonically, ) from zarr_metadata.v3._registry import CORE_AND_EXTENSIONS, Context from zarr_metadata.v3.array import ZARR_V3_ARRAY_METADATA_STORE_KEY, ZarrV3ExtensionField @@ -231,6 +234,27 @@ def __post_init__(self) -> None: object.__setattr__(self, "codecs", tuple(self.codecs)) object.__setattr__(self, "storage_transformers", tuple(self.storage_transformers)) + def __eq__(self, other: object) -> bool: + """Whether `other` models the same array: the same document, however each is spelled. + + Compared as `array_key` says: what the package interprets -- each + field, and the fill value against the data type -- in its canonical + spelling, so `"NaN"` and `"0x7fc00000"` are one `float32` fill value + and a blosc with and without the `typesize` that `noshuffle` ignores + one codec; and what it does not interpret -- the attributes, the + extra fields, the fill value of a data type nothing in scope claims + -- as JSON text, which tells `true` from `1` and `-0.0` from `0.0`, + and takes `NaN` for itself. So two equal models may write two + documents: `to_json` writes each as it was given. Equal models hash + alike. + """ + if type(other) is not type(self): + return NotImplemented + return array_key(self) == array_key(cast("ZarrV3ArrayMetadata", other)) + + def __hash__(self) -> int: + return hash(array_key(self)) + def to_json(self) -> ZarrV3ArrayMetadataJSON: """The document as JSON, arrays as tuples, sharing no mutable state with the model. @@ -293,6 +317,34 @@ def to_key_value( return {ZARR_V3_ARRAY_METADATA_STORE_KEY: dump_store_json(array_json(self), indent=indent)} +def array_key(model: ZarrV3ArrayMetadata) -> tuple[object, ...]: + """What `==` and `hash` compare of a v3 array model: what its document means. + + Each field by its `field_key`, the fill value in its canonical spelling + as JSON text when a definition in scope read the data type, and every + other member as it is, the JSON ones as text. + """ + return ( + model.shape, + _fill_value_key(model), + field_key(model.data_type), + field_key(model.chunk_grid), + tuple(field_key(codec) for codec in model.codecs), + field_key(model.chunk_key_encoding), + model.dimension_names, + json_text(model.attributes), + tuple(field_key(transformer) for transformer in model.storage_transformers), + json_text(model.extra_fields), + ) + + +def _fill_value_key(model: ZarrV3ArrayMetadata) -> str: + """What `==` compares of `model`'s fill value: its canonical spelling as JSON text when a definition in scope read the data type, and the fill value as written when none did.""" + if isinstance(model.data_type, Read): + return json_text(spelled_canonically(model.data_type, model.fill_value)) + return json_text(model.fill_value) + + def _members(model: ZarrV3ArrayMetadata) -> ArrayMembersV3: """`model`'s members other than its fields, as the read of its document by its own fields refines them; `MetadataValidationError` with every problem that document has.""" reading, members = read_array_v3(held_document(model), NO_SCOPE) @@ -506,6 +558,20 @@ def create_default(cls, **overrides: Unpack[ZarrV2ArrayMetadataPartial]) -> Zarr ) return default.update(**overrides) + def __eq__(self, other: object) -> bool: + """Whether `other` models the same array: the same document, as JSON text. + + Nothing in a v2 document is interpreted, so two models are one when + their documents are written alike, which tells `0` from `0.0` and + `-0.0`, and takes `NaN` for itself. Equal models hash alike. + """ + if type(other) is not type(self): + return NotImplemented + return json_text(self.to_json()) == json_text(cast("ZarrV2ArrayMetadata", other).to_json()) + + def __hash__(self) -> int: + return hash(json_text(self.to_json())) + def to_json(self) -> ZarrV2ArrayMetadataJSON: """Return the merged in-memory document form. diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_group.py b/packages/zarr-metadata/src/zarr_metadata/model/_group.py index 62bc8c09b2..ff900f4062 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_group.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/_group.py @@ -16,6 +16,7 @@ arrays_to_tuples, copied, is_canonical_json, + json_text, not_an_object, outside_of, refine_json, @@ -28,6 +29,7 @@ from zarr_metadata.model._array import ( ZarrV3ArrayMetadata, array_json, + array_key, array_model, must_understand_subset, read_array_metadata_v3, @@ -157,6 +159,15 @@ def update( return updated return dataclasses.replace(updated, consolidated_metadata=self.consolidated_metadata) + def __eq__(self, other: object) -> bool: + """Whether `other` models the same group: the same document, however each is spelled, as `group_key` says; equal models hash alike.""" + if type(other) is not type(self): + return NotImplemented + return group_key(self) == group_key(cast("ZarrV3GroupMetadata", other)) + + def __hash__(self) -> int: + return hash(group_key(self)) + def to_json(self) -> ZarrV3GroupMetadataJSON: """The document as JSON, sharing no mutable state with the model. @@ -276,6 +287,15 @@ def __post_init__(self) -> None: raise MetadataValidationError(problems) object.__setattr__(self, "metadata", dict(self.metadata)) + def __eq__(self, other: object) -> bool: + """Whether `other` holds the same documents at the same paths, each as its model compares; equal ones hash alike.""" + if type(other) is not type(self): + return NotImplemented + return consolidated_key(self) == consolidated_key(cast("ZarrV3ConsolidatedMetadata", other)) + + def __hash__(self) -> int: + return hash(consolidated_key(self)) + def to_json(self) -> ZarrV3ConsolidatedMetadataJSON: """The `consolidated_metadata` member as JSON, sharing no mutable state with the model: its `kind`, `must_understand: false`, and each document by its path.""" return cast( @@ -639,6 +659,24 @@ def _read_consolidated_v3( return readings, members, tuple(problems) +def group_key(model: ZarrV3GroupMetadata) -> tuple[object, ...]: + """What `==` and `hash` compare of a v3 group model: its attributes and extra fields as JSON text, and what its consolidated metadata holds, by `consolidated_key`.""" + consolidated = model.consolidated_metadata + return ( + json_text(model.attributes), + UNSET if consolidated is UNSET else consolidated_key(consolidated), + json_text(model.extra_fields), + ) + + +def consolidated_key(model: ZarrV3ConsolidatedMetadata) -> tuple[object, ...]: + """What `==` and `hash` compare of consolidated metadata: each document's key, by its path, in path order.""" + return tuple( + (path, array_key(node) if isinstance(node, ZarrV3ArrayMetadata) else group_key(node)) + for path, node in sorted(model.metadata.items(), key=lambda item: item[0]) + ) + + def _key_problems(key: str) -> list[ValidationProblem]: """What keeps `key` from being where consolidated metadata keeps a document, said in one problem at the key. @@ -901,6 +939,15 @@ def update(self, **kwargs: Unpack[ZarrV2GroupMetadataPartial]) -> ZarrV2GroupMet """ return dataclasses.replace(self, **kwargs) + def __eq__(self, other: object) -> bool: + """Whether `other` models the same group: the same document, as JSON text, which takes `NaN` for itself; equal models hash alike.""" + if type(other) is not type(self): + return NotImplemented + return json_text(self.to_json()) == json_text(cast("ZarrV2GroupMetadata", other).to_json()) + + def __hash__(self) -> int: + return hash(json_text(self.to_json())) + def to_json(self) -> ZarrV2GroupMetadataJSON: """Return the merged in-memory document form. @@ -996,6 +1043,17 @@ def __post_init__(self) -> None: raise MetadataValidationError(problems) object.__setattr__(self, "metadata", refined) + def __eq__(self, other: object) -> bool: + """Whether `other` holds the same document: the same JSON text; equal ones hash alike.""" + if type(other) is not type(self): + return NotImplemented + return json_text(self.to_json()) == json_text( + cast("ZarrV2ConsolidatedMetadata", other).to_json() + ) + + def __hash__(self) -> int: + return hash(json_text(self.to_json())) + def to_json(self) -> dict[str, JSONValue]: """The `.zmetadata` document as JSON, sharing no mutable state with the model.""" # to_json output shares no mutable state with the model. diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py b/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py index d8d149cd81..4e2c154d73 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py @@ -51,7 +51,14 @@ from typing_extensions import TypeAliasType, TypedDict, TypeVar, is_typeddict from zarr_metadata._common import JSONValue, ZarrV3NamedConfigJSON -from zarr_metadata._json import ValidationProblem, copied, refine_json, shown, with_input +from zarr_metadata._json import ( + ValidationProblem, + copied, + json_text, + refine_json, + shown, + with_input, +) from zarr_metadata._sentinel import UNSET from zarr_metadata._typed_json import ( JSONSchema, @@ -106,6 +113,11 @@ def unchanged(configuration: T) -> T: return configuration +def fill_value_as_written(configuration: object, nested: object, value: T) -> T: + """The canonical spelling of a fill value of a data type that spells each of its values one way: the fill value as written.""" + return value + + def unknown_lengths(configuration: object, nested: object, shape: tuple[int, ...]) -> Lengths: """The chunk lengths of a grid that says nothing of them: unknown, along each axis of the array.""" return (None,) * len(shape) @@ -334,6 +346,14 @@ class DataTypeDefinition(Definition[C]): scope read them (a struct's field types), and the typed fill value. A data type that says nothing of its fill value takes any JSON. + `fill_value_canonical` spells a fill value that has no problem -- + well typed, and allowed by the rules -- in the one spelling its value + has, so two fill values are one value of the type exactly when their + canonical spellings are written alike: `"NaN"` and `"0x7fc00000"` are + one `float32`, and `0.0` and `-0.0` two. It is handed what the rules + are handed. A data type that says nothing of it spells each of its + values one way: as written. + `storage` says how its values are stored -- in single bytes, in several bytes at a time, or each in as many as it needs -- which is what the `bytes` codec asks of the data type it is handed: an @@ -350,6 +370,8 @@ class DataTypeDefinition(Definition[C]): """The JSON shape of a fill value, as an annotation: `Int8FillValue`.""" fill_value_rules: Callable[[C, Nested, Any], Iterable[ValidationProblem]] = no_rules """What the spec disallows in a fill value of that shape, located in it.""" + fill_value_canonical: Callable[[C, Nested, Any], JSONValue] = fill_value_as_written + """A fill value that has no problem, in the one spelling its value has.""" storage: Callable[[C, Nested], StorageClass | None] = unknown_storage """How its values are stored, given the configuration and the fields it holds; None when unknown.""" @@ -933,12 +955,14 @@ class Read(Generic[D]): `must_understand` of `false` -- is reported with the field and leaves it read; so is a problem of a field its configuration holds, which is that field's own, as `nested` says. Two fields are equal when they - read the same, however each was spelled: `"bytes"` and - `{"name": "bytes"}` are one field. + read the same, however each was spelled, as `field_key` compares + them: `"bytes"` and `{"name": "bytes"}` are one field, and so are a + blosc with and without the `typesize` that `noshuffle` ignores, which + the definition's `canonical` folds. Equal fields hash alike. """ - json: JSONValue = dataclasses.field(compare=False) - """The field as written, refined: arrays as tuples.""" + json: JSONValue + """The field as written, refined: arrays as tuples; it takes no part in equality.""" name: str """The name it is written with: `"r16"`, though its definition is filed under `r*`.""" definition: D @@ -958,9 +982,17 @@ class Read(Generic[D]): codecs at `("codecs", 0)`: what a definition's functions consult about the fields inside its own. """ - read_as: type[Definition[Any]] = dataclasses.field(init=False, repr=False, compare=False) + read_as: type[Definition[Any]] = dataclasses.field(init=False, repr=False) """The kind of metadata it was read as: its definition's.""" + def __eq__(self, other: object) -> bool: + if not isinstance(other, _FIELDS): + return NotImplemented + return field_key(self) == field_key(cast("Resolved[Any]", other)) + + def __hash__(self) -> int: + return hash(field_key(self)) + def __post_init__(self) -> None: # The runtime half of the annotations: a field read by hand, as an # extension's may be, fails here rather than where a function trusts it. @@ -995,11 +1027,12 @@ class Unclaimed: """A field nothing in scope claims: an extension the scope leaves unjudged, which is what keeps the format open. Equal to another when it is written with the same name and - configuration, however each was spelled. + configuration, however each was spelled: nothing in scope interprets + its configuration, so it compares as JSON text, as `field_key` says. """ - json: JSONValue = dataclasses.field(compare=False) - """The field as written, refined: arrays as tuples.""" + json: JSONValue + """The field as written, refined: arrays as tuples; it takes no part in equality.""" name: str """The name nothing in scope claims.""" read_as: type[Definition[Any]] @@ -1007,6 +1040,14 @@ class Unclaimed: configuration: Mapping[str, JSONValue] = dataclasses.field(init=False) """The configuration as written, which nothing judged; empty when none is written.""" + def __eq__(self, other: object) -> bool: + if not isinstance(other, _FIELDS): + return NotImplemented + return field_key(self) == field_key(cast("Resolved[Any]", other)) + + def __hash__(self) -> int: + return hash(field_key(self)) + def __post_init__(self) -> None: # The runtime half of the annotations; `read_as` with its type # arguments dropped, as `resolve` drops them. @@ -1049,6 +1090,14 @@ class Refused(Generic[D]): nested: Nested = dataclasses.field(default_factory=_nothing_nested) """The fields its configuration holds, each as the scope read it; empty when its configuration was not checked against its TypedDict.""" + def __eq__(self, other: object) -> bool: + if not isinstance(other, _FIELDS): + return NotImplemented + return field_key(self) == field_key(cast("Resolved[Any]", other)) + + def __hash__(self) -> int: + return hash(field_key(self)) + def __post_init__(self) -> None: # The runtime half of the annotations; `read_as` with its type # arguments dropped, as `resolve` drops them. @@ -1077,6 +1126,50 @@ def _misread(definition: object, kind: type[Definition[Any]], name: object) -> s return None +def field_key(field: Resolved[Any]) -> tuple[object, ...]: + """What `==` and `hash` compare of a field: what it means, not how it was spelled. + + A field read compares by the definition that read it, its + configuration in the canonical spelling the definition gives it -- what + `canonical_of` spells -- as JSON text, and the fields it holds, each by + its own key. A field nothing claims compares by its name and its + configuration as written, as JSON text: nothing interprets it. A field + refused compares by what was written, and by what refused it. So two + fields are one when what a reader understands of them reads the same, + and what none interprets is written alike. + """ + if isinstance(field, Read): + definition = cast("Definition[Any]", field.definition) + configuration: JSONValue = dict(field.configuration) + # A field it holds is compared by its own key, so its place holds + # nothing before the definition's `canonical` sees the rest, as + # `_canonical_field` orders it: `canonical` folds only the + # definition's own members. + for loc in field.nested: + configuration = _replaced(configuration, loc, None) + spelled = asked( + definition, + "canonical", + lambda: json_text(cast("JSONValue", definition.canonical(configuration))), + ) + return ("read", definition, spelled, _nested_key(field.nested)) + if isinstance(field, Unclaimed): + return ("unclaimed", field.read_as, field.name, json_text(field.configuration)) + return ( + "refused", + field.read_as, + field.name, + field.definition, + json_text(field.json), + _nested_key(field.nested), + ) + + +def _nested_key(nested: Nested) -> tuple[tuple[Loc, tuple[object, ...]], ...]: + """The fields a configuration holds, each by its key, where it sits.""" + return tuple((loc, field_key(inner)) for loc, inner in nested.items()) + + def document_json(field: Resolved[Any]) -> JSONValue: """A field as a document writes it, holding the field's own values: as `to_json` writes it, or as it was written when it was refused. @@ -1246,6 +1339,55 @@ def fill_value_problems( return with_input((*problems, *refused), value, loc) +def canonical_fill_value( + data_type: Resolved[DataTypeDefinition[Any]], value: object +) -> JSONValue | UNSET: + """`value`, a fill value of `data_type`, a data type field a scope read, in the one spelling its value has; `UNSET` when it has a problem. + + As the data type's `fill_value_canonical` spells it, so two fill + values of a data type are one value exactly when their canonical + spellings are written alike -- the same JSON, as `json.dumps` writes + it, which `==` is not: it takes `-0.0` for `0.0`. A fill value + `fill_value_problems` finds a problem with has no canonical spelling, + as `canonical_of` gives a field with a problem none: `UNSET`, since + `None` is the JSON `null`, a fill value of a data type the scope did + not read, which spells a fill value as written. + """ + if len(fill_value_problems(data_type, value)) != 0: + return UNSET + refined, _ = refine_json(value, ()) + return spelled_canonically(data_type, refined) + + +def spelled_canonically( + data_type: Resolved[DataTypeDefinition[Any]], value: JSONValue +) -> JSONValue: + """`value`, a fill value of `data_type` with no problem, in its canonical spelling, as `canonical_fill_value` gives it, without judging it again. + + Its `fill_value_canonical` is the extension author's code: what it + gives is checked to be JSON, and an error it raises says which data + type's canonical spelling raised it. + """ + if not isinstance(data_type, Read): + return value + definition, configuration = data_type.definition, data_type.configuration + spelled = asked( + definition, + "fill_value_canonical", + lambda: cast( + "object", definition.fill_value_canonical(configuration, data_type.nested, value) + ), + ) + refined, problems = refine_json(spelled, ()) + if len(problems) != 0: + msg = ( + f"{definition.name!r}: its fill_value_canonical gives JSON, got {spelled!r}: " + f"{problems[0].message}" + ) + raise TypeError(msg) + return refined + + def storage_of(data_type: Resolved[DataTypeDefinition[Any]]) -> StorageClass | None: """How the values of `data_type`, a data type field a scope read, are stored; None when unknown. @@ -1602,13 +1744,16 @@ def _replaced(value: JSONValue, path: Loc, new: JSONValue) -> JSONValue: "Unclaimed", "as_kind", "asked", + "canonical_fill_value", "canonical_of", "canonicalize", "chunk_grid_lengths", "configuration_of", "field_json_schema", + "field_key", "field_kind", "field_schemas", + "fill_value_as_written", "fill_value_problems", "kind_of", "multi_byte", @@ -1619,6 +1764,7 @@ def _replaced(value: JSONValue, path: Loc, new: JSONValue) -> JSONValue: "ruled", "single_byte", "spelled", + "spelled_canonically", "storage_of", "unchanged", "unknown_chunk", diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/chunk_key_encoding/default.py b/packages/zarr-metadata/src/zarr_metadata/v3/chunk_key_encoding/default.py index c0e49ee543..cc6ab57062 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/chunk_key_encoding/default.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/chunk_key_encoding/default.py @@ -7,7 +7,7 @@ See https://zarr-specs.readthedocs.io/en/latest/v3/core/index.html#chunk-key-encoding """ -from typing import Final, Literal, NotRequired +from typing import Final, Literal, NotRequired, cast from typing_extensions import TypedDict @@ -54,8 +54,25 @@ class DefaultChunkKeyEncodingObject(TypedDict, closed=True): so the short-hand-name form is permitted in addition to the object form. """ + +def _canonical( + configuration: DefaultChunkKeyEncodingConfiguration, +) -> DefaultChunkKeyEncodingConfiguration: + """Without a `separator` of `/`, which is what an absent one means. + + "If not specified, `separator` defaults to `/`" + (https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v3/chunk-key-encodings/default/index.rst#L27-L29), + so the two spellings are one encoding. + """ + if configuration.get("separator") != "/": + return configuration + return cast("DefaultChunkKeyEncodingConfiguration", {}) + + DEFAULT_CHUNK_KEY_ENCODING: Final = ChunkKeyEncodingDefinition( - name=DEFAULT_CHUNK_KEY_ENCODING_NAME, configuration=DefaultChunkKeyEncodingConfiguration + name=DEFAULT_CHUNK_KEY_ENCODING_NAME, + configuration=DefaultChunkKeyEncodingConfiguration, + canonical=_canonical, ) """The `default` chunk key encoding; its `separator` is typed, so it has no rule of its own.""" diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/chunk_key_encoding/v2.py b/packages/zarr-metadata/src/zarr_metadata/v3/chunk_key_encoding/v2.py index 7c178a599e..4d9bc284c4 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/chunk_key_encoding/v2.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/chunk_key_encoding/v2.py @@ -13,7 +13,7 @@ See https://zarr-specs.readthedocs.io/en/latest/v3/core/index.html#chunk-key-encoding """ -from typing import Final, Literal, NotRequired +from typing import Final, Literal, NotRequired, cast from typing_extensions import TypedDict @@ -60,8 +60,23 @@ class V2ChunkKeyEncodingObject(TypedDict, closed=True): so the short-hand-name form is permitted in addition to the object form. """ + +def _canonical(configuration: V2ChunkKeyEncodingConfiguration) -> V2ChunkKeyEncodingConfiguration: + """Without a `separator` of `.`, which is what an absent one means. + + "If not specified, `separator` defaults to `.`" + (https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v3/chunk-key-encodings/v2/index.rst#L27-L29), + so the two spellings are one encoding. + """ + if configuration.get("separator") != ".": + return configuration + return cast("V2ChunkKeyEncodingConfiguration", {}) + + V2_CHUNK_KEY_ENCODING: Final = ChunkKeyEncodingDefinition( - name=V2_CHUNK_KEY_ENCODING_NAME, configuration=V2ChunkKeyEncodingConfiguration + name=V2_CHUNK_KEY_ENCODING_NAME, + configuration=V2ChunkKeyEncodingConfiguration, + canonical=_canonical, ) """The `v2` chunk key encoding; its `separator` is typed, so it has no rule of its own.""" diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/codec/sharding_indexed.py b/packages/zarr-metadata/src/zarr_metadata/v3/codec/sharding_indexed.py index 2bf5bd8020..8a56d50a16 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/codec/sharding_indexed.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/codec/sharding_indexed.py @@ -5,7 +5,7 @@ """ from collections.abc import Iterator, Mapping -from typing import Annotated, Final, Literal, NotRequired +from typing import Annotated, Final, Literal, NotRequired, cast from annotated_types import Ge from typing_extensions import TypedDict @@ -165,9 +165,27 @@ def _per_shard(chunk_shape: tuple[int, ...], lengths: Lengths | None) -> Lengths ) +def _canonical( + configuration: ShardingIndexedCodecConfiguration, +) -> ShardingIndexedCodecConfiguration: + """Without an `index_location` of `end`, which is what an absent one means. + + "If the parameter is not present, the value defaults to `end`" + (https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v3/codecs/sharding-indexed/index.rst#L157-L161), + so the two spellings are one codec. + """ + if configuration.get("index_location") != "end": + return configuration + return cast( + "ShardingIndexedCodecConfiguration", + {key: value for key, value in configuration.items() if key != "index_location"}, + ) + + SHARDING_INDEXED_CODEC: Final = CodecDefinition( name=SHARDING_INDEXED_CODEC_NAME, configuration=ShardingIndexedCodecConfiguration, + canonical=_canonical, kind="array_bytes", size="dynamic", chunk_rules=_chunk_rules, diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/codec/zstd.py b/packages/zarr-metadata/src/zarr_metadata/v3/codec/zstd.py index 61a0695b61..55cd471dae 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/codec/zstd.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/codec/zstd.py @@ -6,7 +6,7 @@ proposed the codec, was never merged). """ -from typing import Annotated, Final, Literal, NotRequired +from typing import Annotated, Final, Literal, NotRequired, cast from annotated_types import Interval from typing_extensions import TypedDict @@ -56,9 +56,26 @@ class ZstdCodecObject(TypedDict, closed=True): https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v3/core/index.rst#L1562-L1564 (short-hand names only "if no configuration metadata is required") """ + +def _canonical(configuration: ZstdCodecConfiguration) -> ZstdCodecConfiguration: + """Without a `checksum` of `false`, which is what an absent one means. + + The spec says of `checksum` that it "should be omitted if false" + (https://github.com/zarr-developers/zarr-extensions/blob/4da7b37a84f76e660902f6d3de3eaef0e0febae6/codecs/zstd/README.md?plain=1#L17-L19), + so the two spellings are one codec. + """ + if configuration.get("checksum") is not False: + return configuration + return cast( + "ZstdCodecConfiguration", + {key: value for key, value in configuration.items() if key != "checksum"}, + ) + + ZSTD_CODEC: Final = CodecDefinition( name=ZSTD_CODEC_NAME, configuration=ZstdCodecConfiguration, + canonical=_canonical, kind="bytes_bytes", size="dynamic", ) diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/_float.py b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/_float.py index 3f7b56b5da..f30b842699 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/_float.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/_float.py @@ -9,10 +9,13 @@ (https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v3/data-types/index.rst#L88-L91). """ +import struct from collections.abc import Callable, Iterable, Iterator from dataclasses import dataclass -from typing import Any, Literal, get_args +from decimal import ROUND_HALF_EVEN, Decimal, localcontext +from typing import Any, Final, Literal, get_args +from zarr_metadata._common import JSONValue from zarr_metadata._json import ValidationProblem, choices, shown from zarr_metadata.v3._definition import EmptyConfiguration, Nested @@ -56,6 +59,116 @@ def __call__( ) +FloatWidth = Literal[16, 32, 64] +"""The widths, in bits, of the IEEE 754 binary formats the float types store.""" + +_FORMATS: Final[dict[FloatWidth, tuple[str, int]]] = {16: ("e", 10), 32: ("f", 23), 64: ("d", 52)} +"""Each width's `struct` format, and how many bits of it hold the fraction.""" + + +def float_fill_value_canonical( + width: FloatWidth, +) -> Callable[[EmptyConfiguration, Nested, float | str], float | str]: + """The canonical spelling of a fill value of the floating-point type `width` bits wide. + + A fill value spells a value of the type, as `float_bits` reads one: a + number, a named value, or the value's bits. Its canonical spelling is + the named value for an infinity, or for the NaN the spec names + `"NaN"`; the hex string of its bits, in lower case, for any other NaN; + and, for any other value, the shortest number that rounds to it, the + nearest of those, as numpy and `repr` spell one -- `0.1` for the + `float32` nearest `0.1` -- `-0.0`, a value of its own, among them. + """ + return _FloatCanonical(width) + + +@dataclass(frozen=True, slots=True) +class _FloatCanonical: + """A float fill value's canonical spelling: a value rather than a closure, as `_FloatFillValue` is.""" + + width: FloatWidth + + def __call__( + self, configuration: EmptyConfiguration, nested: Nested, value: float | str + ) -> float | str: + return _spelled(float_bits(value, self.width), self.width) + + +def float_bits(value: float | str, width: FloatWidth) -> int: + """The bits of the value `value`, a float fill value the rules allow, spells in the type `width` bits wide. + + A number is read as a float64, as a JSON parser reads one -- an integer + of more digits than a float64 holds is rounded to one -- and rounded to + the nearest value the type represents, ties to even, and to an infinity + past the largest, as numpy casts a float64. So an integer and a number + with a fraction that read as one float64 spell one value. + """ + code, fraction = _FORMATS[width] + exponent = width - 1 - fraction + infinity = ((1 << exponent) - 1) << fraction + sign = 1 << (width - 1) + if isinstance(value, str): + named = { + "NaN": infinity | (1 << (fraction - 1)), + "Infinity": infinity, + "-Infinity": sign | infinity, + } + return named[value] if value in named else int(value, 16) + try: + held = float(value) + except OverflowError: + # An integer past the largest float64 reads as an infinity. + return (sign if value < 0 else 0) | infinity + try: + return int.from_bytes(struct.pack(f">{code}", held), "big") + except OverflowError: + # `struct` refuses a float64 that rounds past the type's largest value. + return (sign if held < 0 else 0) | infinity + + +def _spelled(bits: int, width: FloatWidth) -> float | str: + """The canonical spelling of the value whose bits, in the type `width` bits wide, are `bits`.""" + code, fraction = _FORMATS[width] + exponent = width - 1 - fraction + infinity = ((1 << exponent) - 1) << fraction + sign = 1 << (width - 1) + if bits & infinity == infinity: + if bits & ((1 << fraction) - 1) == 0: + return "-Infinity" if bits & sign else "Infinity" + if bits == infinity | (1 << (fraction - 1)): + return "NaN" + return f"0x{bits:0{width // 4}x}" + value: float = struct.unpack(f">{code}", bits.to_bytes(width // 8, "big"))[0] + if width == 64 or value == 0: + # A float64 is its own shortest spelling, as `repr` writes it, and + # so is a zero of either sign. + return value + return _shortest(value, bits, width) + + +def _shortest(value: float, bits: int, width: FloatWidth) -> float: + """The shortest number that rounds to `value`, whose bits in the type `width` bits wide are `bits`: of the fewest significant digits, the nearest to it. + + Of each number of digits the nearest is tried, and then the one past + `value` from it: at a power of two the numbers that round to it reach + twice as far above it as below, so the nearest may miss where the next + one does not. + """ + exact = Decimal(value) + for digits in range(1, 18): + with localcontext() as context: + context.prec = digits + context.rounding = ROUND_HALF_EVEN + nearest = +exact + step = Decimal((0, (1,), nearest.adjusted() - digits + 1)) + beyond = nearest + step if nearest < exact else nearest - step + for candidate in (nearest, beyond): + if float_bits(float(candidate), width) == bits: + return float(candidate) + # Seventeen significant digits tell every float64 from every other. + return value + + def complex_fill_value_rules( component: Callable[[EmptyConfiguration, Nested, Any], Iterable[ValidationProblem]], ) -> Callable[ @@ -85,4 +198,34 @@ def __call__( yield ValidationProblem((index, *found.loc), found.message, found.kind) -__all__ = ["FloatSpecialFillValue", "complex_fill_value_rules", "float_fill_value_rules"] +def complex_fill_value_canonical( + component: Callable[[EmptyConfiguration, Nested, Any], JSONValue], +) -> Callable[[EmptyConfiguration, Nested, tuple[float | str, float | str]], JSONValue]: + """The canonical spelling of a complex fill value: each component in the canonical spelling `component`, its float type's, gives it.""" + return _ComplexCanonical(component) + + +@dataclass(frozen=True, slots=True) +class _ComplexCanonical: + """A complex fill value's canonical spelling: a value rather than a closure, as `_FloatFillValue` is.""" + + component: Callable[[EmptyConfiguration, Nested, Any], JSONValue] + + def __call__( + self, + configuration: EmptyConfiguration, + nested: Nested, + value: tuple[float | str, float | str], + ) -> JSONValue: + return tuple(self.component(configuration, nested, part) for part in value) + + +__all__ = [ + "FloatSpecialFillValue", + "FloatWidth", + "complex_fill_value_canonical", + "complex_fill_value_rules", + "float_bits", + "float_fill_value_canonical", + "float_fill_value_rules", +] diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/_numpy_time.py b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/_numpy_time.py index ffc0ce2aab..de5c0c7d09 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/_numpy_time.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/_numpy_time.py @@ -24,6 +24,24 @@ https://github.com/zarr-developers/zarr-extensions/blob/4da7b37a84f76e660902f6d3de3eaef0e0febae6/data-types/numpy.datetime64/README.md?plain=1#L109-L112 """ +NOT_A_TIME_TICKS: Final = -(2**63) +"""The tick count `NaT` is stored as, which a fill value may write for it. + +"`"fill_value": "NaT"` and `"fill_value": -9223372036854775808` should +be treated as equivalent representations of the same scalar value" +(https://github.com/zarr-developers/zarr-extensions/blob/4da7b37a84f76e660902f6d3de3eaef0e0febae6/data-types/numpy.datetime64/README.md?plain=1#L114-L116); +`numpy.timedelta64` says the same +(https://github.com/zarr-developers/zarr-extensions/blob/4da7b37a84f76e660902f6d3de3eaef0e0febae6/data-types/numpy.timedelta64/README.md?plain=1#L117-L119). +""" + + +def numpy_time_fill_value_canonical( + configuration: object, nested: object, value: int | str +) -> int | str: + """The canonical spelling of a numpy time fill value: `"NaT"` for `NaT`, however it is written, and any other count of ticks as written.""" + return "NaT" if value in ("NaT", NOT_A_TIME_TICKS) else value + + numpy_time_storage: Final = multi_byte """Signed 64-bit integers, in the byte order the codecs say. @@ -38,8 +56,10 @@ __all__ = [ + "NOT_A_TIME_TICKS", "NUMPY_TIME_MAX_SCALE_FACTOR", "NumpyTimeScaleFactor", "NumpyTimeTicks", + "numpy_time_fill_value_canonical", "numpy_time_storage", ] diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/bytes.py b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/bytes.py index 62a266ffc9..aa83447ef3 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/bytes.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/bytes.py @@ -4,6 +4,7 @@ See https://github.com/zarr-developers/zarr-extensions/blob/4da7b37a84f76e660902f6d3de3eaef0e0febae6/data-types/bytes/README.md """ +import base64 import re from collections.abc import Iterator from typing import Final, Literal, NewType @@ -63,11 +64,26 @@ def _fill_value_rules( ) +def _fill_value_canonical( + configuration: EmptyConfiguration, nested: Nested, value: BytesFillValue +) -> str: + """The bytes, however written, as the base64 that encodes them. + + The string "is more compact" + (https://github.com/zarr-developers/zarr-extensions/blob/4da7b37a84f76e660902f6d3de3eaef0e0febae6/data-types/bytes/README.md?plain=1#L11), + and encoding the bytes again spells them one way: `"QR=="` decodes to + the one byte `"QQ=="` does. + """ + data = base64.b64decode(value) if isinstance(value, str) else bytes(value) + return base64.b64encode(data).decode("ascii") + + BYTES_DATA_TYPE: Final = DataTypeDefinition( name=BYTES_DATA_TYPE_NAME, configuration=EmptyConfiguration, fill_value=BytesFillValue, fill_value_rules=_fill_value_rules, + fill_value_canonical=_fill_value_canonical, storage=variable_length, ) """The `bytes` data type: a bare name, with nothing to configure. diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/complex128.py b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/complex128.py index 509b12f30a..bf3d55de87 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/complex128.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/complex128.py @@ -7,7 +7,10 @@ from typing import Final, Literal from zarr_metadata.v3._definition import DataTypeDefinition, EmptyConfiguration, multi_byte -from zarr_metadata.v3.data_type._float import complex_fill_value_rules +from zarr_metadata.v3.data_type._float import ( + complex_fill_value_canonical, + complex_fill_value_rules, +) from zarr_metadata.v3.data_type.float64 import FLOAT64_DATA_TYPE, Float64FillValue COMPLEX128_DATA_TYPE_NAME: Final = "complex128" @@ -36,6 +39,7 @@ configuration=EmptyConfiguration, fill_value=Complex128FillValue, fill_value_rules=complex_fill_value_rules(FLOAT64_DATA_TYPE.fill_value_rules), + fill_value_canonical=complex_fill_value_canonical(FLOAT64_DATA_TYPE.fill_value_canonical), storage=multi_byte, ) """The `complex128` data type: a bare name, with nothing to configure; its fill value a pair of `float64` components.""" diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/complex64.py b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/complex64.py index ceb1576473..5ddd0559f8 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/complex64.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/complex64.py @@ -7,7 +7,10 @@ from typing import Final, Literal from zarr_metadata.v3._definition import DataTypeDefinition, EmptyConfiguration, multi_byte -from zarr_metadata.v3.data_type._float import complex_fill_value_rules +from zarr_metadata.v3.data_type._float import ( + complex_fill_value_canonical, + complex_fill_value_rules, +) from zarr_metadata.v3.data_type.float32 import FLOAT32_DATA_TYPE, Float32FillValue COMPLEX64_DATA_TYPE_NAME: Final = "complex64" @@ -36,6 +39,7 @@ configuration=EmptyConfiguration, fill_value=Complex64FillValue, fill_value_rules=complex_fill_value_rules(FLOAT32_DATA_TYPE.fill_value_rules), + fill_value_canonical=complex_fill_value_canonical(FLOAT32_DATA_TYPE.fill_value_canonical), storage=multi_byte, ) """The `complex64` data type: a bare name, with nothing to configure; its fill value a pair of `float32` components.""" diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/float16.py b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/float16.py index 33ba182dc4..1a53e8bca1 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/float16.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/float16.py @@ -8,7 +8,11 @@ from typing import Final, Literal, NewType from zarr_metadata.v3._definition import DataTypeDefinition, EmptyConfiguration, multi_byte -from zarr_metadata.v3.data_type._float import FloatSpecialFillValue, float_fill_value_rules +from zarr_metadata.v3.data_type._float import ( + FloatSpecialFillValue, + float_fill_value_canonical, + float_fill_value_rules, +) FLOAT16_DATA_TYPE_NAME: Final = "float16" """The `data_type` value for the `float16` type.""" @@ -69,6 +73,7 @@ def hex_float16(value: str) -> HexFloat16: configuration=EmptyConfiguration, fill_value=Float16FillValue, fill_value_rules=float_fill_value_rules("float16", hex_float16), + fill_value_canonical=float_fill_value_canonical(16), storage=multi_byte, ) """The `float16` data type: a bare name, with nothing to configure; its fill value a number, a named value or a hex string.""" diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/float32.py b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/float32.py index 7eb923d581..35e9706ae0 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/float32.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/float32.py @@ -8,7 +8,11 @@ from typing import Final, Literal, NewType from zarr_metadata.v3._definition import DataTypeDefinition, EmptyConfiguration, multi_byte -from zarr_metadata.v3.data_type._float import FloatSpecialFillValue, float_fill_value_rules +from zarr_metadata.v3.data_type._float import ( + FloatSpecialFillValue, + float_fill_value_canonical, + float_fill_value_rules, +) FLOAT32_DATA_TYPE_NAME: Final = "float32" """The `data_type` value for the `float32` type.""" @@ -69,6 +73,7 @@ def hex_float32(value: str) -> HexFloat32: configuration=EmptyConfiguration, fill_value=Float32FillValue, fill_value_rules=float_fill_value_rules("float32", hex_float32), + fill_value_canonical=float_fill_value_canonical(32), storage=multi_byte, ) """The `float32` data type: a bare name, with nothing to configure; its fill value a number, a named value or a hex string.""" diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/float64.py b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/float64.py index 0363439cfa..a086ed50a5 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/float64.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/float64.py @@ -8,7 +8,11 @@ from typing import Final, Literal, NewType from zarr_metadata.v3._definition import DataTypeDefinition, EmptyConfiguration, multi_byte -from zarr_metadata.v3.data_type._float import FloatSpecialFillValue, float_fill_value_rules +from zarr_metadata.v3.data_type._float import ( + FloatSpecialFillValue, + float_fill_value_canonical, + float_fill_value_rules, +) FLOAT64_DATA_TYPE_NAME: Final = "float64" """The `data_type` value for the `float64` type.""" @@ -70,6 +74,7 @@ def hex_float64(value: str) -> HexFloat64: configuration=EmptyConfiguration, fill_value=Float64FillValue, fill_value_rules=float_fill_value_rules("float64", hex_float64), + fill_value_canonical=float_fill_value_canonical(64), storage=multi_byte, ) """The `float64` data type: a bare name, with nothing to configure; its fill value a number, a named value or a hex string.""" diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/numpy_datetime64.py b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/numpy_datetime64.py index 365df13446..4464c93d4d 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/numpy_datetime64.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/numpy_datetime64.py @@ -12,6 +12,7 @@ from zarr_metadata.v3.data_type._numpy_time import ( NumpyTimeScaleFactor, NumpyTimeTicks, + numpy_time_fill_value_canonical, numpy_time_storage, ) @@ -62,6 +63,7 @@ class NumpyDatetime64(TypedDict, closed=True): name=NUMPY_DATETIME64_DATA_TYPE_NAME, configuration=NumpyDatetime64Configuration, fill_value=NumpyDatetime64FillValue, + fill_value_canonical=numpy_time_fill_value_canonical, storage=numpy_time_storage, ) """The `numpy.datetime64` data type.""" diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/numpy_timedelta64.py b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/numpy_timedelta64.py index 2f11f45526..0aa75d3559 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/numpy_timedelta64.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/numpy_timedelta64.py @@ -12,6 +12,7 @@ from zarr_metadata.v3.data_type._numpy_time import ( NumpyTimeScaleFactor, NumpyTimeTicks, + numpy_time_fill_value_canonical, numpy_time_storage, ) @@ -81,6 +82,7 @@ class NumpyTimedelta64(TypedDict, closed=True): name=NUMPY_TIMEDELTA64_DATA_TYPE_NAME, configuration=NumpyTimedelta64Configuration, fill_value=NumpyTimedelta64FillValue, + fill_value_canonical=numpy_time_fill_value_canonical, storage=numpy_time_storage, ) """The `numpy.timedelta64` data type.""" diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/struct.py b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/struct.py index 140f8ae847..0aa0f55257 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/struct.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/struct.py @@ -17,6 +17,7 @@ Nested, StorageClass, fill_value_problems, + spelled_canonically, storage_of, ) @@ -128,6 +129,23 @@ def _fill_value_rules( yield ValidationProblem((key,), f"no struct field is named {key!r}", "unknown_key") +def _fill_value_canonical( + configuration: StructConfiguration, nested: Nested, value: StructFillValue +) -> dict[str, JSONValue]: + """Each field's fill value in the canonical spelling of that field's own type, in the order the fields are declared. + + A field type the scope did not read, or whose reading the struct's does + not hold, spells its fill value as written. + """ + spelled: dict[str, JSONValue] = {} + for index, member in enumerate(configuration["fields"]): + name = member["name"] + field_type = nested.get(("fields", index, "data_type")) + held = value[name] + spelled[name] = held if field_type is None else spelled_canonically(field_type, held) + return spelled + + def _storage(configuration: StructConfiguration, nested: Nested) -> StorageClass | None: """Its fields' values, packed together: numbers of several bytes if any field holds them, single bytes if every field is made of them. @@ -160,6 +178,7 @@ def _storage(configuration: StructConfiguration, nested: Nested) -> StorageClass rules=_rules, fill_value=StructFillValue, fill_value_rules=_fill_value_rules, + fill_value_canonical=_fill_value_canonical, storage=_storage, ) """The `struct` data type: a record of named fields, each field's type a nested field.""" diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/definition.py b/packages/zarr-metadata/src/zarr_metadata/v3/definition.py index 65e766a55b..5f112e6b18 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/definition.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/definition.py @@ -203,6 +203,15 @@ def acme_lz4_rules( `fill_value_problems(data_type, value)` judges a fill value against a data type field the scope read; one nothing in scope claims leaves it unjudged. A data type that says nothing of its fill value takes any JSON. +A fill value may be spelled more ways than one -- `"NaN"` and +`"0x7fc00000"` are one `float32` -- so a data type says which spelling +is its value's own: `fill_value_canonical`, handed what the rules are +handed and a fill value they allow. Two fill values are one value of the +type exactly when their canonical spellings are written alike: the same +JSON, as `json.dumps` writes it, which `==` is not -- it takes `-0.0`, +a `float32` of its own, for `0.0`. `canonical_fill_value(data_type, +value)` spells one, and gives `UNSET` for a fill value with a problem; a +data type that says nothing of it spells each value as written. A data type says how its values are stored, too: `storage`, a function of its configuration and the fields it holds, giving a `StorageClass` -- @@ -330,6 +339,7 @@ def acme_lz4_rules( StorageClass, StorageTransformerDefinition, Unclaimed, + canonical_fill_value, canonical_of, canonicalize, chunk_grid_lengths, @@ -378,6 +388,7 @@ def acme_lz4_rules( "Unclaimed", "ValidationProblem", "ZarrV3MetadataFieldJSON", + "canonical_fill_value", "canonical_of", "canonicalize", "check", diff --git a/packages/zarr-metadata/tests/model/test_array.py b/packages/zarr-metadata/tests/model/test_array.py index f7cde5c83b..34fd225c69 100644 --- a/packages/zarr-metadata/tests/model/test_array.py +++ b/packages/zarr-metadata/tests/model/test_array.py @@ -24,6 +24,8 @@ ZarrV2ArrayMetadata, ZarrV2ArrayMetadataPartial, ZarrV3ArrayMetadata, + ZarrV3ConsolidatedMetadata, + ZarrV3GroupMetadata, is_array_metadata_v2, is_array_metadata_v3, is_group_metadata_v2, @@ -515,6 +517,147 @@ def test_a_model_is_what_its_document_says_not_how_it_is_spelled( assert ZarrV3ArrayMetadata.from_json(written) == model +_DATETIME = {"name": "numpy.datetime64", "configuration": {"unit": "s", "scale_factor": 1}} +_LITTLE = {"name": "bytes", "configuration": {"endian": "little"}} + + +@pytest.mark.parametrize( + ("data_type", "left", "right", "same"), + [ + ("float32", "NaN", "0x7fc00000", True), + ("float32", 1, 1.0, True), + ("float32", 0.1, 0.10000000149011612, True), + ("float32", 0.0, -0.0, False), + ("float32", "NaN", "0xffc00000", False), + (_DATETIME, "NaT", -(2**63), True), + ("bytes", [65], "QR==", True), + # The fill value of a data type nothing in scope claims is not + # interpreted, and is compared as JSON text. + ("acme.decimal", {"a": 1, "b": 2}, {"b": 2, "a": 1}, True), + ("acme.decimal", [1, 2], [2, 1], False), + ("acme.decimal", True, 1, False), + ("acme.decimal", 0.0, -0.0, False), + ], +) +def test_two_models_are_one_array_when_their_fill_values_are_one_value( + data_type: object, left: object, right: object, same: bool +) -> None: + """As two fields are one when they read the same, however each is spelled.""" + codecs = [{"name": "vlen-bytes"} if data_type == "bytes" else _LITTLE] + document = {**ZarrV3ArrayMetadata.create_default(shape=(2,)).to_json(), "codecs": codecs} + models = [ + ZarrV3ArrayMetadata.from_json({**document, "data_type": data_type, "fill_value": value}) + for value in (left, right) + ] + assert (models[0] == models[1]) is same + assert (models[1] == models[0]) is same + assert models[0] == ZarrV3ArrayMetadata.from_json(models[0].to_json()) + # Equal models hash alike. + if same: + assert hash(models[0]) == hash(models[1]) + # A group holding the arrays says so too. + groups = [ + ZarrV3GroupMetadata( + attributes={}, + consolidated_metadata=ZarrV3ConsolidatedMetadata(metadata={"a": model}), + extra_fields={}, + ) + for model in models + ] + assert (groups[0] == groups[1]) is same + + +_NOSHUFFLE = {"cname": "lz4", "clevel": 5, "shuffle": "noshuffle", "blocksize": 0} + + +@pytest.mark.parametrize( + ("member", "left", "right"), + [ + ( + "codecs", + [_LITTLE, {"name": "blosc", "configuration": _NOSHUFFLE}], + [_LITTLE, {"name": "blosc", "configuration": {**_NOSHUFFLE, "typesize": 4}}], + ), + ( + "codecs", + [_LITTLE, {"name": "zstd", "configuration": {"level": 1}}], + [_LITTLE, {"name": "zstd", "configuration": {"level": 1, "checksum": False}}], + ), + ( + "chunk_key_encoding", + "default", + {"name": "default", "configuration": {"separator": "/"}}, + ), + ], + ids=["blosc-typesize-noshuffle", "zstd-checksum-false", "default-separator"], +) +def test_two_models_are_one_array_when_their_fields_read_the_same( + member: str, left: object, right: object +) -> None: + """The spec's equivalences, which each definition's `canonical` folds, hold in a model's `==`, though `to_json` writes each as given.""" + document = ZarrV3ArrayMetadata.create_default(shape=(2,)).to_json() + models = [ZarrV3ArrayMetadata.from_json({**document, member: value}) for value in (left, right)] + assert models[0] == models[1] + assert hash(models[0]) == hash(models[1]) + assert models[0].to_json() != models[1].to_json() + + +@pytest.mark.parametrize( + ("left", "right", "same"), + [ + ({"a": math.nan}, {"a": math.nan}, True), + ({"a": 1, "b": 2}, {"b": 2, "a": 1}, True), + ({"a": True}, {"a": 1}, False), + ({"a": 0.0}, {"a": -0.0}, False), + ({"a": 1}, {"a": 1.0}, False), + ], + ids=["nan", "key-order", "bool-vs-int", "signed-zero", "int-vs-float"], +) +def test_user_json_compares_as_text( + left: "dict[str, JSONValue]", right: "dict[str, JSONValue]", same: bool +) -> None: + """Attributes, which nothing interprets, compare as a document writes them: `NaN` is itself, `true` is not `1`, `-0.0` is not `0.0`.""" + arrays = [ZarrV3ArrayMetadata.create_default(attributes=held) for held in (left, right)] + groups = [ + ZarrV3GroupMetadata(attributes=held, consolidated_metadata=UNSET, extra_fields={}) + for held in (left, right) + ] + v2 = [ZarrV2ArrayMetadata.create_default(attributes=held) for held in (left, right)] + for models in (arrays, groups, v2): + assert (models[0] == models[1]) is same + if same: + assert hash(models[0]) == hash(models[1]) + + +def test_a_model_holding_nan_user_data_equals_its_copies() -> None: + # What Python's `==` on the value denies: `nan != nan`. + model = ZarrV3ArrayMetadata.create_default(attributes={"_FillValue": math.nan}) + assert ZarrV3ArrayMetadata.from_key_value(model.to_key_value()) == model + assert pickle.loads(pickle.dumps(model)) == model + assert ZarrV3ArrayMetadata.from_json(json.loads(json.dumps(model.to_json()))) == model + + +@pytest.mark.parametrize( + ("left", "right", "same"), + [ + (0.0, 0.0, True), + ("NaN", "NaN", True), + (0.0, -0.0, False), + (1, 1.0, False), + (0, False, False), + ], +) +def test_two_v2_models_are_one_array_when_their_documents_are_written_alike( + left: object, right: object, same: bool +) -> None: + """A v2 data type is not interpreted, so nor is its fill value: the document as text decides.""" + model = ZarrV2ArrayMetadata.create_default(shape=(2,), chunks=(2,), dtype=" None: model = ZarrV3ArrayMetadata.create_default( codecs=( diff --git a/packages/zarr-metadata/tests/v3/test_definitions.py b/packages/zarr-metadata/tests/v3/test_definitions.py index 839fe9d309..3477e66be8 100644 --- a/packages/zarr-metadata/tests/v3/test_definitions.py +++ b/packages/zarr-metadata/tests/v3/test_definitions.py @@ -42,6 +42,7 @@ Unclaimed, ValidationProblem, ZarrV3MetadataFieldJSON, + canonical_of, check, configuration_of, resolve, @@ -958,6 +959,7 @@ def test_a_scope_pickles_as_its_definitions_and_copies_as_itself() -> None: LE: Final = {"name": "bytes", "configuration": {"endian": "little"}} SHARD: Final = {"chunk_shape": [2], "codecs": [LE], "index_location": "end"} +NOSHUFFLE: Final = {"cname": "lz4", "clevel": 5, "shuffle": "noshuffle", "blocksize": 0} @pytest.mark.parametrize( @@ -997,6 +999,71 @@ def test_a_scope_pickles_as_its_definitions_and_copies_as_itself() -> None: False, ), ("int8", "uint8", DataTypeDefinition, False), + # The spec's equivalences, which each definition's `canonical` + # folds: a member at its default, a run-length encoding, leading + # zeros in a size. + ( + {"name": "blosc", "configuration": {**NOSHUFFLE}}, + {"name": "blosc", "configuration": {**NOSHUFFLE, "typesize": 4}}, + CodecDefinition, + True, + ), + ( + "default", + {"name": "default", "configuration": {"separator": "/"}}, + ChunkKeyEncodingDefinition, + True, + ), + ( + "default", + {"name": "default", "configuration": {"separator": "."}}, + ChunkKeyEncodingDefinition, + False, + ), + ( + {"name": "v2"}, + {"name": "v2", "configuration": {"separator": "."}}, + ChunkKeyEncodingDefinition, + True, + ), + ( + { + "name": "sharding_indexed", + "configuration": {**SHARD, "index_codecs": [LE, "crc32c"]}, + }, + { + "name": "sharding_indexed", + "configuration": { + "chunk_shape": [2], + "codecs": [LE], + "index_codecs": [LE, "crc32c"], + }, + }, + CodecDefinition, + True, + ), + ( + {"name": "zstd", "configuration": {"level": 3}}, + {"name": "zstd", "configuration": {"level": 3, "checksum": False}}, + CodecDefinition, + True, + ), + ( + {"name": "zstd", "configuration": {"level": 3}}, + {"name": "zstd", "configuration": {"level": 3, "checksum": True}}, + CodecDefinition, + False, + ), + ( + {"name": "rectilinear", "configuration": {"kind": "inline", "chunk_shapes": [[2, 2]]}}, + { + "name": "rectilinear", + "configuration": {"kind": "inline", "chunk_shapes": [[[2, 2]]]}, + }, + ChunkGridDefinition, + True, + ), + ("r008", "r8", DataTypeDefinition, True), ], ids=[ "bare-or-object", @@ -1007,18 +1074,29 @@ def test_a_scope_pickles_as_its_definitions_and_copies_as_itself() -> None: "another-configuration", "unclaimed-another-configuration", "another-name", + "blosc-typesize-noshuffle", + "default-separator", + "default-other-separator", + "v2-separator", + "sharding-index-location", + "zstd-checksum-false", + "zstd-checksum-true", + "rectilinear-rle", + "raw-bits-leading-zeros", ], ) def test_two_fields_are_equal_when_they_read_the_same( one: JSONValue, other: JSONValue, kind: type[Definition[Any]], equal: bool ) -> None: - """However each was spelled; each is written as it reads, and equal fields are written the same.""" + """However each was spelled: equal fields have one simplest spelling and one hash, though each is written as read.""" first, _ = resolve(one, kind, CORE_AND_EXTENSIONS) second, _ = resolve(other, kind, CORE_AND_EXTENSIONS) assert isinstance(first, (Read, Unclaimed)) assert isinstance(second, (Read, Unclaimed)) assert (first == second) is equal - assert (first.to_json() == second.to_json()) is equal + assert (canonical_of(first, ()) == canonical_of(second, ())) is equal + if equal: + assert hash(first) == hash(second) def test_a_field_copied_or_pickled_is_read_by_a_definition_equal_to_its_own() -> None: diff --git a/packages/zarr-metadata/tests/v3/test_every_definition.py b/packages/zarr-metadata/tests/v3/test_every_definition.py index 91b3a76849..e7138ba75c 100644 --- a/packages/zarr-metadata/tests/v3/test_every_definition.py +++ b/packages/zarr-metadata/tests/v3/test_every_definition.py @@ -260,6 +260,39 @@ def test_every_example_reads_and_its_simplest_spelling_is_stable(key: str, field "configuration": {"kind": "inline", "chunk_shapes": (((32, 3), 16), 8)}, }, ), + # A member at its spec default is left out: a key encoding's + # separator, a shard's index location, zstd's checksum. + ({"name": "default", "configuration": {"separator": "/"}}, {"name": "default"}), + ({"name": "v2", "configuration": {"separator": "."}}, {"name": "v2"}), + ( + { + "name": "sharding_indexed", + "configuration": { + "chunk_shape": [2], + "codecs": ["bytes"], + "index_codecs": [ + {"name": "bytes", "configuration": {"endian": "little"}}, + "crc32c", + ], + "index_location": "end", + }, + }, + { + "name": "sharding_indexed", + "configuration": { + "chunk_shape": (2,), + "codecs": ({"name": "bytes"},), + "index_codecs": ( + {"name": "bytes", "configuration": {"endian": "little"}}, + {"name": "crc32c"}, + ), + }, + }, + ), + ( + {"name": "zstd", "configuration": {"level": 3, "checksum": False}}, + {"name": "zstd", "configuration": {"level": 3}}, + ), ], ids=[ "blosc-noshuffle", @@ -268,10 +301,19 @@ def test_every_example_reads_and_its_simplest_spelling_is_stable(key: str, field "sharding-nested", "cast-value-target", "rectilinear-rle", + "default-separator", + "v2-separator", + "sharding-index-location", + "zstd-checksum", ], ) def test_the_simplest_spelling(field: dict[str, Any], simplest: object) -> None: - kinds = {"rectilinear": ChunkGridDefinition, "int8": DataTypeDefinition} + kinds = { + "rectilinear": ChunkGridDefinition, + "int8": DataTypeDefinition, + "default": ChunkKeyEncodingDefinition, + "v2": ChunkKeyEncodingDefinition, + } kind = kinds.get(field["name"], CodecDefinition) assert canonicalize(field, kind, CORE_AND_EXTENSIONS) == (simplest, ()) assert canonical_of(*resolve(field, kind, CORE_AND_EXTENSIONS)) == simplest diff --git a/packages/zarr-metadata/tests/v3/test_fill_values.py b/packages/zarr-metadata/tests/v3/test_fill_values.py index 7963d08d24..7334bbcc82 100644 --- a/packages/zarr-metadata/tests/v3/test_fill_values.py +++ b/packages/zarr-metadata/tests/v3/test_fill_values.py @@ -8,15 +8,21 @@ from __future__ import annotations +import dataclasses +import json import math -from typing import TYPE_CHECKING +from typing import TYPE_CHECKING, Any, cast, get_args import pytest +from hypothesis import given +from hypothesis import strategies as st from typing_extensions import TypedDict from zarr_metadata._json import value_at +from zarr_metadata._sentinel import UNSET from zarr_metadata.model import validate_array_metadata_v3, validate_group_metadata_v3 from zarr_metadata.model._array import ZarrV3ArrayMetadata +from zarr_metadata.v3.data_type._float import FloatWidth, float_bits from zarr_metadata.v3.data_type.struct import STRUCT_DATA_TYPE from zarr_metadata.v3.definition import ( CORE_AND_EXTENSIONS, @@ -26,6 +32,7 @@ Nested, Read, ValidationProblem, + canonical_fill_value, fill_value_problems, resolve, ) @@ -315,3 +322,148 @@ def test_a_struct_read_without_its_field_types_leaves_its_fields_unjudged() -> N json=STRUCT, name="struct", definition=STRUCT_DATA_TYPE, configuration=configuration ) assert fill_value_problems(struct, {"a": 300}) == () + + +# --- canonical spellings --------------------------------------------------- + + +def _alike(left: object, right: object) -> bool: + """Whether two JSON values are written alike: `==` takes `-0.0` for `0.0`.""" + return json.dumps(left, sort_keys=True) == json.dumps(right, sort_keys=True) + + +def _read(data_type: JSONValue) -> Read[DataTypeDefinition[Any]]: + resolved, found = resolve(data_type, DataTypeDefinition, CORE_AND_EXTENSIONS) + assert found == () + assert isinstance(resolved, Read) + return resolved + + +@pytest.mark.parametrize( + ("data_type", "spellings", "canonical"), + [ + ("float32", ["NaN", "0x7fc00000", "0x7FC00000"], "NaN"), + # Any other NaN is its bits: the spec names the one. + ("float32", ["0xffc00000"], "0xffc00000"), + ("float32", ["0x7fc00001", "0x7FC00001"], "0x7fc00001"), + ("float32", ["Infinity", "0x7f800000", 3.5e38, 10**39], "Infinity"), + ("float32", ["-Infinity", "0xff800000", -(10**39)], "-Infinity"), + ("float32", [0, 0.0, "0x00000000", 1e-46], 0.0), + # Zero's sign is a value of its own. + ("float32", [-0.0, "0x80000000"], -0.0), + ("float32", [1, 1.0, "0x3f800000"], 1.0), + # The number of the fewest digits that rounds to the value. + ("float32", [0.1, 0.10000000149011612, "0x3dcccccd"], 0.1), + # A number is read as a float64, as a JSON parser reads one, and + # then rounded to the type, as numpy rounds it: an integer and a + # number with a fraction that read as one float64 are one value. + ("float32", [2**53 + 2**29 + 1, 9007199791611905.0, 2**53], 9007199000000000.0), + # Of the numbers of the fewest digits, the one past the nearest: at + # a power of two, those that round to it reach further above it. + ("float16", [0.015625, "0x2400"], 0.01563), + ("float32", ["0x6b000000"], 1.5474251e26), + ("float16", [65504, 65519, "0x7bff"], 65500.0), + ("float16", [65520, "Infinity", "0x7c00"], "Infinity"), + # Halfway rounds to the even value. + ("float16", [2049, 2048], 2048.0), + ("float16", [2051, 2052], 2052.0), + ("float16", ["NaN", "0x7e00", "0x7E00"], "NaN"), + ("float16", [-0.0, "0x8000"], -0.0), + ("float64", [2**53 + 1, 2**53], float(2**53)), + ("float64", [10**400, "Infinity", "0x7ff0000000000000"], "Infinity"), + ("float64", [-(10**400), "-Infinity", "0xfff0000000000000"], "-Infinity"), + ("float64", ["NaN", "0x7ff8000000000000"], "NaN"), + ("complex64", [["NaN", -0.0], ["0x7fc00000", "0x80000000"]], ("NaN", -0.0)), + ("complex128", [[0.1, 1], ["0x3fb999999999999a", 1.0]], (0.1, 1.0)), + ("bytes", [[65], "QQ==", "QR=="], "QQ=="), + ("bytes", [[], ""], ""), + (DATETIME, ["NaT", -(2**63)], "NaT"), + (TIMEDELTA, ["NaT", -(2**63)], "NaT"), + (TIMEDELTA, [5], 5), + (STRUCT, [{"a": 1, "b": "0x7fc00000"}, {"b": "NaN", "a": 1}], {"a": 1, "b": "NaN"}), + # A field type nothing in scope claims spells its fill value as written. + ( + { + "name": "struct", + "configuration": {"fields": [{"name": "a", "data_type": "acme.decimal"}]}, + }, + [{"a": -0.0}], + {"a": -0.0}, + ), + # A type that spells each value one way spells it as written. + ("int8", [-128], -128), + ("bool", [True], True), + ("string", ["NaN"], "NaN"), + ("r16", [[0, 1]], (0, 1)), + ], +) +def test_a_fill_value_s_canonical_spelling_is_its_value_s( + data_type: JSONValue, spellings: list[object], canonical: JSONValue +) -> None: + read = _read(data_type) + for spelling in spellings: + spelled = canonical_fill_value(read, spelling) + assert _alike(spelled, canonical), (spelling, spelled) + # A fill value of the type, spelled as itself. + assert fill_value_problems(read, canonical) == () + assert _alike(canonical_fill_value(read, canonical), canonical) + + +def test_a_data_type_nothing_in_scope_claims_spells_a_fill_value_as_written() -> None: + unclaimed, _ = resolve("acme.decimal", DataTypeDefinition, CORE_AND_EXTENSIONS) + values: list[JSONValue] = [-0.0, True, 1, [1.0], None] + for value in values: + assert _alike(canonical_fill_value(unclaimed, value), value) + + +WIDTHS: tuple[FloatWidth, ...] = get_args(FloatWidth) + + +@given(st.data()) +def test_a_float_s_canonical_spelling_spells_its_bits(data: st.DataObject) -> None: + # Every value of each float type, NaNs among them: its canonical + # spelling is a fill value of the type, spelling the same bits, and + # its own canonical spelling. + width = data.draw(st.sampled_from(WIDTHS)) + bits = data.draw(st.integers(0, 2**width - 1)) + read = _read(f"float{width}") + spelled = canonical_fill_value(read, f"0x{bits:0{width // 4}x}") + assert spelled is not None + assert fill_value_problems(read, spelled) == () + assert float_bits(cast("float | str", spelled), width) == bits + assert _alike(canonical_fill_value(read, spelled), spelled) + + +def test_error_a_fill_value_with_a_problem_has_no_canonical_spelling() -> None: + # `UNSET`, since `None` is the JSON null, a fill value of a data type + # the scope did not read. + assert canonical_fill_value(_read("float32"), "0x7fc0") is UNSET + assert canonical_fill_value(_read("int8"), 1.0) is UNSET + assert canonical_fill_value(_read("int8"), math.nan) is UNSET + unclaimed, _ = resolve("acme.decimal", DataTypeDefinition, CORE_AND_EXTENSIONS) + assert canonical_fill_value(unclaimed, None) is None + + +def test_error_a_canonical_spelling_that_raises_says_which_data_type_raised_it() -> None: + def refuses(configuration: EmptyConfiguration, nested: Nested, value: object) -> JSONValue: + raise ValueError("no") + + scope = CORE_AND_EXTENSIONS.extended_with( + dataclasses.replace(ACME_POINT, fill_value_canonical=refuses) + ) + resolved, _ = resolve("acme.point", DataTypeDefinition, scope) + with pytest.raises(ValueError, match="no") as raised: + canonical_fill_value(resolved, {"x": 1}) + assert raised.value.__notes__ == ["raised by the fill_value_canonical of 'acme.point'"] + + +def test_error_a_canonical_spelling_that_is_not_json_is_refused() -> None: + def not_json(configuration: EmptyConfiguration, nested: Nested, value: object) -> JSONValue: + return cast("JSONValue", {1, 2}) + + scope = CORE_AND_EXTENSIONS.extended_with( + dataclasses.replace(ACME_POINT, fill_value_canonical=not_json) + ) + resolved, _ = resolve("acme.point", DataTypeDefinition, scope) + with pytest.raises(TypeError, match="'acme.point': its fill_value_canonical gives JSON"): + canonical_fill_value(resolved, {"x": 1}) From c2e5638d01134da0b9d52e9356ee9a527a7d7e0d Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Tue, 29 Sep 2026 23:09:59 +0200 Subject: [PATCH 15/94] fix(zarr-metadata): fix the problems found by the adversarial review of the readers This commit fixes what six review passes found, beyond the fixes already in #381. Nesting depth. A reader now walks at most 256 levels (JSON_DEPTH) and reports a deeper level as an invalid_value problem. Before, a small document nested a thousand levels deep made every validator, reader, parser and from_key_value raise RecursionError, and json.loads on store bytes made of 100,000 "[" did the same (now an invalid_json problem). A chain of groups, each holding the next in its consolidated metadata, was also unbounded. Each document is now read from its position in the outer document, so depth is counted from the outer root. is_json and the is_* guards stop at the same depth. A v2 dtype of field records is checked as JSON before its shape is read. Readers, writers and comparisons use one stack frame per level; the v2 models no longer use copy.deepcopy, which used two. validate_json takes the location of its value, so the v2 validator counts depth from the document root as the v3 readers do. arrays_to_tuples uses one frame per level. A problem copied by with_input or prefixed is not re-checked. A struct fill value is judged in linear time. An integer too long to print is shown by its bit length, a value too deeply nested is described instead of printed, and a non-string key is shown with Python's repr. Hand-built field objects. A Read or Unclaimed object placed inside a document passed to a reader is not JSON and is now refused. Before, it passed validation as if it had been read, so a model could write a document that the same validator then refused. Only a model's own fields are treated as already read. Absent values. A reading's data_type, chunk_grid and chunk_key_encoding are UNSET when the document has no such key, and Refused.json is UNSET when the field was not JSON. Both used to be None, which is also what a JSON null becomes. A field with no name is a missing_key problem. A consolidated_metadata of null is an invalid_type problem; it is no longer read as absent, and neither JSON Schema admits it. ZarrV3ConsolidatedMetadata declares must_understand as False. Extension names. An extension name must match ^[a-z][a-z0-9-_.]+$ or be a URI in RFC 3986 characters, so Python's re and ECMA-262 read the schema pattern the same way. Any other name is refused before a definition is looked up; before, any string was read as an unknown extension. well_named checks this, validate_metadata_field_v3 enforces it, the JSON Schema's unclaimed branch carries the pattern, and a definition with an invalid name raises TypeError when built. A raw-bits name with more than 100 digits is treated as an extension name, not a size. Definition functions always receive a read-only view of the configuration, and an error raised by a canonical function names its definition. Pydantic. The pydantic types raise ValidationError with one line error per problem, carrying type, loc, input and ctx. For a missing key the input is the object that lacks it. A message containing a ctx placeholder like {expected} is passed as the ctx's "message" entry, because pydantic renders messages as templates with no escaping. Tests. New tests pin each behavior the mutation review found unpinned: the stricter of two bounds per keyword, the unclaimed envelope in the schema, no empty attributes written for a group, value_at at index 0, every integer type's range, union naming, layered Annotated metadata, and construct's default factory. The float-bits property test draws sign, exponent and mantissa separately so it reaches normal values. A shard nested to the depth cap is read, written, hashed, pickled, deep-copied and walked; one level deeper reports the depth problem. The same holds for a v2 document and a v3 fill value at the cap. Docs: two spec anchors fixed, the struct data type rendered, a note on the JSONValue export, a typed TypeAdapter example, with_problems compared to zod's treeifyError, and the float fill value rounding cited to zarrs and tensorstore. Assisted-by: ClaudeCode:claude-fable-5-1 --- packages/zarr-metadata/README.md | 16 +- .../zarr-metadata/changes/382.bugfix.1.md | 5 + .../zarr-metadata/changes/382.bugfix.2.md | 8 + .../zarr-metadata/changes/382.bugfix.3.md | 6 + .../zarr-metadata/changes/382.bugfix.4.md | 12 + packages/zarr-metadata/changes/382.bugfix.md | 28 ++ .../zarr-metadata/changes/382.feature.1.md | 12 + packages/zarr-metadata/changes/382.feature.md | 9 + packages/zarr-metadata/docs/api/index.md | 5 +- .../zarr-metadata/docs/api/v3/data_type.md | 2 + packages/zarr-metadata/docs/index.md | 16 +- .../zarr-metadata/src/zarr_metadata/_json.py | 189 ++++++++++--- .../src/zarr_metadata/_pydantic_schema.py | 19 +- .../src/zarr_metadata/model/_array.py | 39 ++- .../src/zarr_metadata/model/_group.py | 136 +++++---- .../src/zarr_metadata/model/_json_schema.py | 6 +- .../src/zarr_metadata/model/_validation.py | 154 ++++++---- .../src/zarr_metadata/pydantic.py | 68 ++++- .../src/zarr_metadata/v3/_common.py | 85 +++++- .../src/zarr_metadata/v3/_definition.py | 107 +++++-- .../src/zarr_metadata/v3/_pipeline.py | 9 +- .../v3/codec/sharding_indexed.py | 3 +- .../src/zarr_metadata/v3/data_type/_byte.py | 3 +- .../src/zarr_metadata/v3/data_type/_float.py | 11 +- .../src/zarr_metadata/v3/data_type/struct.py | 3 +- .../zarr-metadata/tests/model/conftest.py | 22 ++ .../zarr-metadata/tests/model/test_array.py | 266 ++++++++++++++++-- .../tests/model/test_construction.py | 10 +- .../zarr-metadata/tests/model/test_group.py | 248 ++++++++++++++-- .../tests/model/test_pydantic_module.py | 124 +++++++- .../tests/model/test_read_array_metadata.py | 14 +- .../tests/model/test_refine_json.py | 33 ++- .../tests/model/test_store_json.py | 8 + .../zarr-metadata/tests/test_json_schema.py | 116 +++++++- .../zarr-metadata/tests/test_problem_data.py | 47 +++- .../zarr-metadata/tests/test_typed_json.py | 17 ++ .../tests/v3/test_definitions.py | 157 ++++++++++- .../tests/v3/test_every_definition.py | 14 +- .../tests/v3/test_fill_values.py | 50 +++- 39 files changed, 1780 insertions(+), 297 deletions(-) create mode 100644 packages/zarr-metadata/changes/382.bugfix.1.md create mode 100644 packages/zarr-metadata/changes/382.bugfix.2.md create mode 100644 packages/zarr-metadata/changes/382.bugfix.3.md create mode 100644 packages/zarr-metadata/changes/382.bugfix.4.md create mode 100644 packages/zarr-metadata/changes/382.bugfix.md create mode 100644 packages/zarr-metadata/changes/382.feature.1.md create mode 100644 packages/zarr-metadata/changes/382.feature.md create mode 100644 packages/zarr-metadata/tests/model/conftest.py diff --git a/packages/zarr-metadata/README.md b/packages/zarr-metadata/README.md index 75bae98677..02d132fd81 100644 --- a/packages/zarr-metadata/README.md +++ b/packages/zarr-metadata/README.md @@ -44,7 +44,8 @@ parser and returns the same normalized model class: from pydantic import TypeAdapter import zarr_metadata.pydantic as zmp -metadata = TypeAdapter(zmp.ZarrV3ArrayMetadata).validate_python(raw) +adapter: TypeAdapter[zmp.ZarrV3ArrayMetadata] = TypeAdapter(zmp.ZarrV3ArrayMetadata) +metadata = adapter.validate_python(raw) encoded = metadata.to_key_value()["zarr.json"] ``` @@ -79,6 +80,19 @@ Two choices the specs' words leave open, or settle two ways: writes them back as those bare tokens, which a strict JSON parser refuses. `check`, from `zarr_metadata.typed_json`, refuses a non-finite number wherever it is, attributes included. +- **A reader walks 256 levels of nesting.** A value nested deeper is an + `invalid_value` at the level past the last, wherever it sits. Every + reader, writer and comparison takes one frame for each level, and + `copy.deepcopy`, and `pickle` before Python 3.12, two: a document at + the cap takes about half of the interpreter's default limit, and the + rest is the caller's. +- **`consolidated_metadata: null` is a problem.** A zarr-python 3.0.x bug + wrote it; the spec says an object, and the package models nothing else + as right. A reader of those stores strips the key before reading. +- **An extension is named as the spec names one**, `^[a-z][a-z0-9-_.]+$`, + or by a URI, which earlier versions of the spec required; any other + name is refused before a definition is asked, so `""` and `"foo/bar"` + are not unknown extensions but problems. - **`must_understand: false` is refused at every extension point**, codecs and storage transformers too, though the core spec names only the data type, chunk grid and chunk key encoding: a reader that skips a codec diff --git a/packages/zarr-metadata/changes/382.bugfix.1.md b/packages/zarr-metadata/changes/382.bugfix.1.md new file mode 100644 index 0000000000..c4ea1234a5 --- /dev/null +++ b/packages/zarr-metadata/changes/382.bugfix.1.md @@ -0,0 +1,5 @@ +A field object -- a `Read` or `Unclaimed` built by hand -- inside a +document a caller hands in is not JSON, and is refused as such, where it +passed the validators as read and let a model write a document the same +validator refuses. Only a model's own fields are held as read: its +constructor and `update` pass them so. diff --git a/packages/zarr-metadata/changes/382.bugfix.2.md b/packages/zarr-metadata/changes/382.bugfix.2.md new file mode 100644 index 0000000000..7394646791 --- /dev/null +++ b/packages/zarr-metadata/changes/382.bugfix.2.md @@ -0,0 +1,8 @@ +What is absent is `UNSET`, and `None` is a JSON null the document wrote: +a reading's `data_type`, `chunk_grid` and `chunk_key_encoding` are `UNSET` +for a key the document does not hold, where they were `None`, which a +`data_type: null` reads as too; and `Refused.json` is `UNSET` for a field +that was not JSON. A metadata field without a `name` is a `missing_key` +at `name`, where it was an `invalid_type`. `ZarrV3ConsolidatedMetadata` +declares `must_understand` as `False`, as it declares `kind`, rather than +taking and refusing a value. diff --git a/packages/zarr-metadata/changes/382.bugfix.3.md b/packages/zarr-metadata/changes/382.bugfix.3.md new file mode 100644 index 0000000000..5c022a5071 --- /dev/null +++ b/packages/zarr-metadata/changes/382.bugfix.3.md @@ -0,0 +1,6 @@ +A group's `consolidated_metadata` of `null` is a problem, `invalid_type` +at the key, and no longer read as absent, nor admitted by either JSON +Schema, the package's or pydantic's: +the spec says an object, and the package models nothing else as right. A +zarr-python 3.0.x bug wrote it; a reader of those stores strips the key +before reading. diff --git a/packages/zarr-metadata/changes/382.bugfix.4.md b/packages/zarr-metadata/changes/382.bugfix.4.md new file mode 100644 index 0000000000..8196eeaefa --- /dev/null +++ b/packages/zarr-metadata/changes/382.bugfix.4.md @@ -0,0 +1,12 @@ +A definition's functions are handed a read-only view of a field's +configuration on every path that asks them -- `resolve` and `judge` the +rules, `canonical_of`, `==` and `hash` the `canonical` -- so a rule that +assigns one of its members fails there, with the note saying whose rule; +an error a `canonical` raises says which definition's it is, as every +other function's does. A raw-bits name of more than a hundred digits is +an extension's name, not a size, where `r` and 4,301 digits raised +`ValueError` from every reader; an integer too long for the interpreter +to write is shown by its size in a message, and a value nested too deep +for it to write by saying so, where each raised; a key no JSON object +holds -- a tuple, `None` -- is shown as Python shows it, not as the JSON +it is not. diff --git a/packages/zarr-metadata/changes/382.bugfix.md b/packages/zarr-metadata/changes/382.bugfix.md new file mode 100644 index 0000000000..1837af2952 --- /dev/null +++ b/packages/zarr-metadata/changes/382.bugfix.md @@ -0,0 +1,28 @@ +A reader walks 256 levels of nesting, and a value nested deeper is an +`invalid_value` at the level past the last, wherever it sits. A 2 KB +document nested a thousand deep took every validator, reader, parser and +`from_key_value` past the interpreter's recursion limit, which escaped as +`RecursionError`; so did `json.loads` on store bytes of a hundred thousand +`[`, now an `invalid_json` problem like any other undecodable bytes; and +so did a chain of groups each holding the next in its consolidated +metadata, which a reader descended without bound: each document is now +read from where it sits in the one handed in, its levels counted from +that one's root, and the reader judges each container it walks -- a +document's members, the consolidated metadata it descends into -- where +it sits, as `refine_json` of the whole would, and +`ZarrV3ConsolidatedMetadata.from_json` reads the member from where it +sits, under a group's key; `is_json` and the `is_*` guards walk no deeper +than a reader either, and a v2 `dtype` of field records is read as JSON +first, so no path recurses past the cap. Every reader, writer and +comparison takes one frame for each level -- the v2 models copied with +`copy.deepcopy`, which takes two, and overflowed on documents their +validators accept -- and `copy.deepcopy` of a model, and `pickle` before +Python 3.12, take two, so a document at the cap takes about half of the +interpreter's default limit, and the rest is the caller's. +`validate_json` takes the `loc` its value sits at, and the v2 validator +hands it one for a compressor, a filter and a fill value, so their +levels are counted from the document's root too, as every v3 reader +counts them. `arrays_to_tuples` walks one frame per level, as `copied` +does, so `parse_*` reads as deep as `validate_*`. A problem copied by +`with_input` or `prefixed` is not checked again, and a struct's fill +value is judged in time linear in its fields. diff --git a/packages/zarr-metadata/changes/382.feature.1.md b/packages/zarr-metadata/changes/382.feature.1.md new file mode 100644 index 0000000000..21e65f1424 --- /dev/null +++ b/packages/zarr-metadata/changes/382.feature.1.md @@ -0,0 +1,12 @@ +The pydantic types raise a `ValidationError` with one line error per +problem, as pydantic reports its own: its `type` the problem's `kind`, at +the problem's `loc` under the field's, with the `input` found there -- +for a key that is missing, the object that lacks it, as pydantic's own +`missing` reports; for what a problem could not hold, not being JSON a +reader walks, what sits at its loc -- and the problem's `ctx`, where +every problem was concatenated into one `value_error` at the field. +pydantic renders a message as a template of its ctx, key by key, with no +escape, so a message that holds a placeholder -- a document value +written `{expected}`, shown in it -- rides whole as the ctx's `message`, +last, and `msg` is the problem's message whatever it holds; a ctx member +of that name yields to it. diff --git a/packages/zarr-metadata/changes/382.feature.md b/packages/zarr-metadata/changes/382.feature.md new file mode 100644 index 0000000000..9bfb35ebe3 --- /dev/null +++ b/packages/zarr-metadata/changes/382.feature.md @@ -0,0 +1,9 @@ +An extension is named as the spec names one -- `^[a-z][a-z0-9-_.]+$`, or +a URI, which earlier versions of the spec required, in the characters +RFC 3986 writes one in -- and any other name, `""` or `"foo/bar"`, is +refused before a definition is asked, an `invalid_value` at `name`, +where every string read as an unknown extension. `well_named` says so of +a string, `validate_metadata_field_v3` refuses it as the envelope rule it +is, the JSON Schema's unclaimed branch carries the same pattern, +and a definition given another name is a `TypeError` when it is built, +since no field could name it. diff --git a/packages/zarr-metadata/docs/api/index.md b/packages/zarr-metadata/docs/api/index.md index 7f4c73c7dc..f7ec11a57f 100644 --- a/packages/zarr-metadata/docs/api/index.md +++ b/packages/zarr-metadata/docs/api/index.md @@ -36,8 +36,9 @@ imported from [`zarr_metadata.model`](model.md) directly. ## Common types -A few cross-cutting aliases are exported only from the top-level -`zarr_metadata` namespace: +A few cross-cutting aliases are exported from the top-level +`zarr_metadata` namespace (`JSONValue` from `zarr_metadata.typed_json` +and `zarr_metadata.v3.definition` too): ::: zarr_metadata.JSONValue diff --git a/packages/zarr-metadata/docs/api/v3/data_type.md b/packages/zarr-metadata/docs/api/v3/data_type.md index f482c33201..3f8444625a 100644 --- a/packages/zarr-metadata/docs/api/v3/data_type.md +++ b/packages/zarr-metadata/docs/api/v3/data_type.md @@ -43,3 +43,5 @@ title: data_type ::: zarr_metadata.v3.data_type.numpy_datetime64 ::: zarr_metadata.v3.data_type.numpy_timedelta64 + +::: zarr_metadata.v3.data_type.struct diff --git a/packages/zarr-metadata/docs/index.md b/packages/zarr-metadata/docs/index.md index 7e99d31144..027800cb77 100644 --- a/packages/zarr-metadata/docs/index.md +++ b/packages/zarr-metadata/docs/index.md @@ -59,7 +59,8 @@ parser and returns the same normalized model class: from pydantic import TypeAdapter import zarr_metadata.pydantic as zmp -metadata = TypeAdapter(zmp.ZarrV3ArrayMetadata).validate_python(raw) +adapter: TypeAdapter[zmp.ZarrV3ArrayMetadata] = TypeAdapter(zmp.ZarrV3ArrayMetadata) +metadata = adapter.validate_python(raw) encoded = metadata.to_key_value()["zarr.json"] ``` @@ -94,6 +95,19 @@ Two choices the specs' words leave open, or settle two ways: writes them back as those bare tokens, which a strict JSON parser refuses. `check`, from `zarr_metadata.typed_json`, refuses a non-finite number wherever it is, attributes included. +- **A reader walks 256 levels of nesting.** A value nested deeper is an + `invalid_value` at the level past the last, wherever it sits. Every + reader, writer and comparison takes one frame for each level, and + `copy.deepcopy`, and `pickle` before Python 3.12, two: a document at + the cap takes about half of the interpreter's default limit, and the + rest is the caller's. +- **`consolidated_metadata: null` is a problem.** A zarr-python 3.0.x bug + wrote it; the spec says an object, and the package models nothing else + as right. A reader of those stores strips the key before reading. +- **An extension is named as the spec names one**, `^[a-z][a-z0-9-_.]+$`, + or by a URI, which earlier versions of the spec required; any other + name is refused before a definition is asked, so `""` and `"foo/bar"` + are not unknown extensions but problems. - **`must_understand: false` is refused at every extension point**, codecs and storage transformers too, though the core spec names only the data type, chunk grid and chunk key encoding: a reader that skips a codec diff --git a/packages/zarr-metadata/src/zarr_metadata/_json.py b/packages/zarr-metadata/src/zarr_metadata/_json.py index de7ac9b92c..4a97f7c9a7 100644 --- a/packages/zarr-metadata/src/zarr_metadata/_json.py +++ b/packages/zarr-metadata/src/zarr_metadata/_json.py @@ -12,9 +12,8 @@ import dataclasses import json import math -from collections.abc import Callable, Mapping, Sequence +from collections.abc import Callable, Iterator, Mapping, Sequence from dataclasses import dataclass -from types import MappingProxyType from typing import Final, Literal, TypeGuard, cast, get_args from zarr_metadata._common import JSONValue @@ -39,7 +38,49 @@ """ -_NO_CTX: Final[Mapping[str, JSONValue]] = MappingProxyType({}) +JSON_DEPTH: Final = 256 +"""How many levels of nesting a reader walks. + +A value nested deeper is a problem at the level past the last, so no +document, however deep, takes a reader past what the interpreter allows: +every value the package reads is refined first, and refining stops here. +Every reader, writer and comparison takes one frame for each level, and +`copy.deepcopy`, and `pickle` before Python 3.12, two: so a document at +the cap takes about half of the interpreter's default limit, a thousand +frames, and the rest is the caller's. +""" + + +class _Ctx(Mapping[str, JSONValue]): + """What a problem holds as its `ctx`: a copy of its own, arrays as tuples, checked to be JSON when it was made, which nothing edits after. + + Its own type, so a problem built of another's `ctx` -- as `replace` + builds one -- knows it was checked, and skips the check; a mapping of + any other type is checked and copied. + """ + + __slots__ = ("_held",) + + def __init__(self, held: dict[str, JSONValue]) -> None: + self._held = held + + def __getitem__(self, key: str) -> JSONValue: + return self._held[key] + + def __iter__(self) -> Iterator[str]: + return iter(self._held) + + def __len__(self) -> int: + return len(self._held) + + def __repr__(self) -> str: + return repr(self._held) + + def __reduce__(self) -> tuple[type[_Ctx], tuple[dict[str, JSONValue]]]: + return _Ctx, (self._held,) + + +_NO_CTX: Final[Mapping[str, JSONValue]] = _Ctx({}) def _no_ctx() -> Mapping[str, JSONValue]: @@ -111,7 +152,9 @@ def __post_init__(self) -> None: msg = f"a ValidationProblem's kind is one of {get_args(ProblemKind)!r}, got {kind!r}" raise TypeError(msg) ctx = cast("object", self.ctx) - if ctx is _NO_CTX: + if isinstance(ctx, _Ctx): + # A problem's own, checked when it was made: what `replace` + # hands a copy. return if ( not isinstance(ctx, Mapping) @@ -120,12 +163,14 @@ def __post_init__(self) -> None: ): msg = f"a ValidationProblem's ctx is an object of JSON values, got {ctx!r}" raise TypeError(msg) - # Held as a view of a copy of its own, arrays as tuples, so a raised - # error, a finished report, cannot be edited through it. + # Held as a view of a copy of its own at every level, arrays as + # tuples, so a raised error, a finished report, cannot be edited + # through it, nor through what was handed in. held = cast( - "dict[str, JSONValue]", arrays_to_tuples(dict(cast("Mapping[str, object]", ctx))) + "dict[str, JSONValue]", + copied(cast("JSONValue", arrays_to_tuples(dict(cast("Mapping[str, object]", ctx))))), ) - object.__setattr__(self, "ctx", MappingProxyType(held)) + object.__setattr__(self, "ctx", _Ctx(held)) def __str__(self) -> str: location = ".".join(str(part) for part in self.loc) if self.loc else "" @@ -239,16 +284,13 @@ def with_input( def _holdable(value: object) -> bool: - """Whether a problem can hold `value` as its input: JSON, and shallow enough to walk. + """Whether a problem can hold `value` as its input: JSON, nested no deeper than a reader walks. What a validator did not walk -- a member it only reports -- may be - deeper than the interpreter walks; such a value would not pickle - either, and is held as nothing. + deeper than that; such a value is held as nothing, as + `is_canonical_json` says. """ - try: - return is_canonical_json(value, finite=False) - except RecursionError: - return False + return is_canonical_json(value, finite=False) def not_an_object(value: object) -> tuple[ValidationProblem, ...]: @@ -256,9 +298,9 @@ def not_an_object(value: object) -> tuple[ValidationProblem, ...]: return with_input((ValidationProblem((), "expected an object", "invalid_type"),), value) -def validate_json(value: object) -> tuple[ValidationProblem, ...]: - """Return every reason `value` is not JSON, each where it sits: a float that is not finite, a key that is not a string, a value of no JSON type.""" - return with_input(refine_json(value)[1], value) +def validate_json(value: object, loc: tuple[str | int, ...] = ()) -> tuple[ValidationProblem, ...]: + """Return every reason `value`, which sits at `loc`, is not JSON, each where it sits: a float that is not finite, a key that is not a string, a value of no JSON type, a level of nesting past `JSON_DEPTH`, counted from the document's root, which `loc` is below.""" + return with_input(refine_json(value, loc)[1], value, loc) def refine_json( @@ -294,6 +336,38 @@ def refine_user_data( _Refined = tuple[JSONValue | None, tuple[ValidationProblem, ...]] +def nested_past_the_levels(value: object, loc: tuple[str | int, ...]) -> ValidationProblem | None: + """The problem `value` is when it is a container -- an object or an array -- at `loc`, past the levels a reader walks, `JSON_DEPTH` of them; None for a scalar, or within them. + + `_refine` asks it of every value it reaches, and a reader of each + container it walks without refining -- a document's members, the + consolidated metadata it descends into -- so a chain of documents is + bounded as any other nesting is, and every container is judged where + it sits, as `refine_json` of the whole document would judge it. + """ + if len(loc) < JSON_DEPTH or isinstance(value, (str, int, float, bool)) or value is None: + return None + if not isinstance(value, (Mapping, Sequence)) or isinstance(value, (bytes, bytearray)): + return None + message = f"nested deeper than the {JSON_DEPTH} levels a reader walks" + return ValidationProblem(loc, message, "invalid_value") + + +def within( + problems: Sequence[ValidationProblem], at: tuple[str | int, ...] +) -> tuple[ValidationProblem, ...]: + """`problems`, found in a value that sits at `at`, located from that value: the reverse of `prefixed`, for a reader that counts the levels it walks from the document handed in, but reports where a problem sits in the one it reads. + + A problem not below `at` is a `TypeError`: a reader that located one + from the wrong root would otherwise report it in the wrong place. + """ + for problem in problems: + if problem.loc[: len(at)] != at: + msg = f"a problem at {problem.loc!r} does not sit below {at!r}" + raise TypeError(msg) + return tuple(dataclasses.replace(p, loc=p.loc[len(at) :]) for p in problems) + + def _refine(value: object, loc: tuple[str | int, ...], *, finite: bool) -> _Refined: """`refine_json`, a non-finite number being JSON unless `finite`.""" if isinstance(value, float): @@ -304,15 +378,19 @@ def _refine(value: object, loc: tuple[str | int, ...], *, finite: bool) -> _Refi ) if isinstance(value, (str, int, bool)) or value is None: return value, () + if (past := nested_past_the_levels(value, loc)) is not None: + return None, (past,) if isinstance(value, Mapping): # Walked here rather than through `_refine_members`, so that each - # level of nesting costs one frame, as deep as the interpreter goes. + # level of nesting costs one frame, `JSON_DEPTH` of them at most. members: dict[str, JSONValue] = {} found_in_members: list[ValidationProblem] = [] for key, item in cast("Mapping[object, object]", value).items(): if not isinstance(key, str): found_in_members.append( - ValidationProblem(loc, f"non-string key {key!r} in JSON object", "invalid_type") + ValidationProblem( + loc, f"non-string key {shown_key(key)} in JSON object", "invalid_type" + ) ) continue member, found = _refine(item, (*loc, key), finite=finite) @@ -330,7 +408,9 @@ def _refine(value: object, loc: tuple[str | int, ...], *, finite: bool) -> _Refi entries.append(entry) return (tuple(entries) if len(found_in_entries) == 0 else None), tuple(found_in_entries) return None, ( - ValidationProblem(loc, f"not a JSON-serializable value: {value!r}", "invalid_type"), + ValidationProblem( + loc, f"not a JSON-serializable value: {shown_by_python(value)}", "invalid_type" + ), ) @@ -345,11 +425,40 @@ def json_text(value: JSONValue) -> str: def shown(value: object) -> str: - """`value` as a problem's message shows it: as the JSON a document writes, `null` and `[1, 2]`, or by its repr when it is not JSON.""" + """`value` as a problem's message shows it: as the JSON a document writes, `null` and `[1, 2]`, or by its repr when it is not JSON; what the interpreter will not write, an integer of too many digits or a value nested too deep, by saying so.""" refined, problems = _refine(value, (), finite=False) if len(problems) != 0: + # Not JSON, or nested past the levels a reader walks. + return shown_by_python(value) + try: + return json.dumps(refined, ensure_ascii=False) + except ValueError: + # An integer of more digits than the interpreter converts to text: + # the value itself, by its size, or one a container holds, which + # is then what Python will not write either. + if isinstance(value, int): + return f"an integer of {value.bit_length()} bits" + return shown_by_python(value) + + +def shown_key(key: object) -> str: + """A key that is not a string, as a message shows it: as Python shows it, since no JSON object holds such a key to write; an integer of more digits than the interpreter writes, by its size.""" + if isinstance(key, int) and not isinstance(key, bool): + try: + return repr(key) + except ValueError: + return f"an integer of {key.bit_length()} bits" + return shown_by_python(key) + + +def shown_by_python(value: object) -> str: + """`value` as Python shows it, for what is not JSON a reader walks; what the interpreter will not write -- nested too deep for its repr, or holding an integer of more digits than it writes -- said so.""" + try: return repr(value) - return json.dumps(refined, ensure_ascii=False) + except RecursionError: + return "a value nested too deep to show" + except ValueError: + return f"a value of type {type(value).__name__} the interpreter will not write" def choices(allowed: Sequence[object]) -> str: @@ -402,25 +511,33 @@ def refused_kind(value: object, allowed: Sequence[object]) -> ProblemKind: def is_canonical_json(value: object, *, finite: bool = True) -> TypeGuard[JSONValue]: - """Whether `value` already uses the concrete containers in `JSONValue`. + """Whether `value` already uses the concrete containers in `JSONValue`, nested no deeper than a reader walks. A non-finite number counts only when `finite` is false, as a document's guard passes it: where one may be is the document's validator's to say. - One frame per level of nesting, as `refine_json` takes, so a value - `refine_json` reads is one this can walk. + One frame per level of nesting, `JSON_DEPTH` of them at most, as + `refine_json` takes: a container past them is no JSON a reader walks, + so it is none to this either, and the walk stops there. """ + return _is_canonical(value, 0, finite=finite) + + +def _is_canonical(value: object, depth: int, *, finite: bool) -> bool: + """`is_canonical_json` of `value`, which sits `depth` levels down.""" if isinstance(value, float): return not finite or math.isfinite(value) if isinstance(value, (str, int, bool)) or value is None: return True + if depth >= JSON_DEPTH: + return False if isinstance(value, (list, tuple)): for item in cast("list[object] | tuple[object, ...]", value): - if not is_canonical_json(item, finite=finite): + if not _is_canonical(item, depth + 1, finite=finite): return False return True if isinstance(value, dict): for key, item in cast("dict[object, object]", value).items(): - if not isinstance(key, str) or not is_canonical_json(item, finite=finite): + if not isinstance(key, str) or not _is_canonical(item, depth + 1, finite=finite): return False return True return False @@ -464,7 +581,12 @@ def arrays_to_tuples(obj: object) -> object: """Recursively materialize mappings and convert array-like values to tuples.""" if isinstance(obj, Sequence) and not isinstance(obj, (str, bytes, bytearray)): sequence = cast("Sequence[object]", obj) - converted_sequence = tuple(arrays_to_tuples(item) for item in sequence) + # Loops, not comprehensions, which are a frame of their own before + # Python 3.12: one frame for each level, as `copied` takes. + converted_items: list[object] = [] + for item in sequence: + converted_items.append(arrays_to_tuples(item)) # noqa: PERF401 + converted_sequence = tuple(converted_items) if isinstance(obj, tuple) and all( converted is original for converted, original in zip(converted_sequence, sequence, strict=True) @@ -473,9 +595,9 @@ def arrays_to_tuples(obj: object) -> object: return converted_sequence if isinstance(obj, Mapping): mapping = cast("Mapping[object, object]", obj) - converted: dict[object, object] = { - key: arrays_to_tuples(value) for key, value in mapping.items() - } + converted: dict[object, object] = {} + for key, value in mapping.items(): + converted[key] = arrays_to_tuples(value) if isinstance(obj, dict) and all(converted[key] is value for key, value in mapping.items()): return mapping return converted @@ -493,6 +615,7 @@ def arrays_to_tuples(obj: object) -> object: "is_json", "json_type", "listed", + "nested_past_the_levels", "not_an_object", "outside_of", "parse_json", @@ -501,7 +624,9 @@ def arrays_to_tuples(obj: object) -> object: "refine_user_data", "refused_kind", "shown", + "shown_key", "validate_json", "value_at", "with_input", + "within", ] diff --git a/packages/zarr-metadata/src/zarr_metadata/_pydantic_schema.py b/packages/zarr-metadata/src/zarr_metadata/_pydantic_schema.py index 8b0e354bc4..810c9d69ae 100644 --- a/packages/zarr-metadata/src/zarr_metadata/_pydantic_schema.py +++ b/packages/zarr-metadata/src/zarr_metadata/_pydantic_schema.py @@ -2,6 +2,7 @@ from __future__ import annotations +import re from collections.abc import Mapping # noqa: TC003 # resolved by Pydantic at runtime from typing import Annotated, Literal, NotRequired @@ -15,14 +16,22 @@ from zarr_metadata.v2.codec import ( # resolved by Pydantic at runtime ZarrV2CodecMetadata, ) +from zarr_metadata.v3._common import EXTENSION_NAME_SCHEMA_PATTERN NonNegativeInt = Annotated[int, Field(ge=0)] +ExtensionName = Annotated[str, Field(pattern=re.compile(EXTENSION_NAME_SCHEMA_PATTERN))] +"""A name as the spec names an extension, the pattern `well_named` accepts, so the schema pydantic generates refuses what the reader refuses. + +Compiled, so pydantic reads it with Python's `re`, which has the look-ahead +the pattern ends in where its own engine has none, and writes it into the +schema as it is. +""" class ZarrV3NamedConfigJSON(TypedDict, closed=True): """Closed v3 named configuration read on its own, outside a document, where `must_understand` may be `false`.""" - name: str + name: ExtensionName configuration: NotRequired[Mapping[str, JSONValue]] must_understand: NotRequired[bool] @@ -30,13 +39,13 @@ class ZarrV3NamedConfigJSON(TypedDict, closed=True): class ZarrV3MandatoryNamedConfigJSON(TypedDict, closed=True): """Closed named configuration at an extension point of a document, where understanding is mandatory.""" - name: str + name: ExtensionName configuration: NotRequired[Mapping[str, JSONValue]] must_understand: NotRequired[Literal[True]] -ZarrV3MetadataFieldJSON = str | ZarrV3NamedConfigJSON -ZarrV3MandatoryMetadataFieldJSON = str | ZarrV3MandatoryNamedConfigJSON +ZarrV3MetadataFieldJSON = ExtensionName | ZarrV3NamedConfigJSON +ZarrV3MandatoryMetadataFieldJSON = ExtensionName | ZarrV3MandatoryNamedConfigJSON ZarrV3CodecPipelineJSON = Annotated[ tuple[ZarrV3MandatoryMetadataFieldJSON, ...], Field(min_length=1) ] @@ -73,7 +82,7 @@ class ZarrV3GroupMetadataJSON(TypedDict, extra_items=JSONValue): zarr_format: Literal[3] node_type: Literal["group"] attributes: NotRequired[Mapping[str, JSONValue]] - consolidated_metadata: NotRequired[ZarrV3ConsolidatedMetadataJSON | None] + consolidated_metadata: NotRequired[ZarrV3ConsolidatedMetadataJSON] class ZarrV2ArrayMetadataJSON(TypedDict, extra_items=JSONValue): diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_array.py b/packages/zarr-metadata/src/zarr_metadata/model/_array.py index 9c65362e5e..f78ebe0316 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_array.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/_array.py @@ -2,7 +2,6 @@ from __future__ import annotations -import copy import dataclasses from collections.abc import Callable, Mapping from dataclasses import dataclass, field @@ -202,7 +201,12 @@ def update( for key, value in members.items(): if value is UNSET: del document[key] - return type(self).from_json(document, context=context) + # The model's own fields are held, read already; the members given + # are JSON, read in `context`, a field object among them refused. + reading, refined = read_array_v3(document, context, held=own_fields(self)) + if refined is None: + raise MetadataValidationError(reading.problems) + return array_model(reading, refined) def __post_init__(self) -> None: # The runtime half of the annotations: extra fields by name, and each @@ -347,7 +351,7 @@ def _fill_value_key(model: ZarrV3ArrayMetadata) -> str: def _members(model: ZarrV3ArrayMetadata) -> ArrayMembersV3: """`model`'s members other than its fields, as the read of its document by its own fields refines them; `MetadataValidationError` with every problem that document has.""" - reading, members = read_array_v3(held_document(model), NO_SCOPE) + reading, members = read_array_v3(held_document(model), NO_SCOPE, held=own_fields(model)) extra = overlapping(model.extra_fields, ARRAY_METADATA_STANDARD_KEYS_V3, "array") problems = (*extra, *reading.problems) if len(problems) != 0: @@ -360,6 +364,17 @@ def array_json(model: ZarrV3ArrayMetadata) -> ZarrV3ArrayMetadataJSON: return cast("ZarrV3ArrayMetadataJSON", _document(model, document_json, whole=False)) +def own_fields(model: ZarrV3ArrayMetadata) -> tuple[Read[Any] | Unclaimed, ...]: + """The fields `model` holds, each as a scope read it: what a read of the model's own document takes as read.""" + return ( + model.data_type, + model.chunk_grid, + model.chunk_key_encoding, + *model.codecs, + *model.storage_transformers, + ) + + def held_document(model: ZarrV3ArrayMetadata) -> dict[str, object]: """`model`'s document with each field as it was read, which a read takes as it is: what `update` and the constructor read, reading no field again. @@ -581,22 +596,22 @@ def to_json(self) -> ZarrV2ArrayMetadataJSON: `to_key_value` to produce the spec-conforming split for storage (https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v2/v2.0.rst#L323-L330). """ - # to_json output shares no mutable state with the model: every value - # that can hold a mutable container is deep-copied. + # to_json output shares no mutable state with the model: the document + # is copied whole, one frame for each level of nesting. out: ZarrV2ArrayMetadataJSON = { "zarr_format": self.zarr_format, "shape": self.shape, "dtype": self.dtype, "order": self.order, "chunks": self.chunks, - "fill_value": copy.deepcopy(self.fill_value), + "fill_value": self.fill_value, "dimension_separator": self.dimension_separator, - "compressor": copy.deepcopy(self.compressor), - "filters": copy.deepcopy(self.filters), + "compressor": self.compressor, + "filters": self.filters, } if self.attributes is not UNSET: - out["attributes"] = copy.deepcopy(self.attributes) - return out + out["attributes"] = self.attributes + return cast("ZarrV2ArrayMetadataJSON", copied(cast("JSONValue", out))) @classmethod def from_json(cls, data: object) -> ZarrV2ArrayMetadata: @@ -607,7 +622,9 @@ def from_json(cls, data: object) -> ZarrV2ArrayMetadata: written back. The model shares no mutable state with `data`. """ # A read model shares no mutable state with what it read. - parsed = copy.deepcopy(parse_array_metadata_v2(data)) + parsed = cast( + "ZarrV2ArrayMetadataJSON", copied(cast("JSONValue", parse_array_metadata_v2(data))) + ) return construct(cls, **_v2_array_members(parsed)) @classmethod diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_group.py b/packages/zarr-metadata/src/zarr_metadata/model/_group.py index ff900f4062..32ae4b9b4f 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_group.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/_group.py @@ -2,7 +2,6 @@ from __future__ import annotations -import copy import dataclasses from collections.abc import Callable, Mapping from dataclasses import dataclass, field @@ -17,12 +16,15 @@ copied, is_canonical_json, json_text, + nested_past_the_levels, not_an_object, outside_of, refine_json, refine_user_data, shown, + shown_key, with_input, + within, ) from zarr_metadata._json import prefixed as _prefix from zarr_metadata._sentinel import UNSET @@ -46,6 +48,7 @@ construct, dump_store_json, load_store_json, + members_past_the_levels, missing_keys, other_members, overlapping, @@ -183,8 +186,11 @@ def from_json( """The model of `data`, a v3 group document read in `context`, with each document its consolidated metadata holds. `MetadataValidationError` with every problem the read finds. A - `consolidated_metadata` of `null` is read as none, and not written - back. A member the spec does not define is held in `extra_fields`. + `consolidated_metadata` of `null`, which a zarr-python 3.0.x bug + wrote, is a value the document wrote, and no object: a problem, + as the spec says an object; a reader of such a store strips the + key first. A member the spec does not define is held in + `extra_fields`. """ reading = read_group_metadata_v3(data, context=context) if reading.metadata is None: @@ -236,30 +242,25 @@ class ZarrV3ConsolidatedMetadata: Models the reference-implementation convention where consolidated metadata is embedded as an extension field on a group's `zarr.json`. Each entry in `metadata` is the model of a complete child document, array or group. - `must_understand` is typed permissively as `bool` to mirror the document - shape, but only `False` is valid; this is enforced at runtime. Each - document it holds is a model, which checked itself when it was built. + `must_understand` is `False` by declaration, not given, as `kind` is + its literal. Each document it holds is a model, which checked itself + when it was built, at its own root: so a model built by hand of models + can hold one that sits, as a whole, deeper than a reader walks, and + write a document its validator refuses there, as one holding a + hand-built `Read` can, and a chain of them, each holding the last, is + bounded by nothing but the interpreter, which `to_json`, `==`, `hash` + and pickle walk a frame a level; `from_json` builds only what a reader + read. The documents and the group make the hierarchy below the group, the group its root, each at its node's path in it without the leading `/`: the node at `/a/b` at `a/b`. """ kind: Literal["inline"] = field(default="inline", init=False) - must_understand: bool = False + must_understand: Literal[False] = field(default=False, init=False) metadata: dict[str, ZarrV3ArrayMetadata | ZarrV3GroupMetadata] def __post_init__(self) -> None: - if self.must_understand is not False: - raise MetadataValidationError( - [ - ValidationProblem( - ("must_understand",), - f"Invalid value for 'must_understand'. Expected False. " - f"Got {self.must_understand!r}.", - "invalid_value", - ) - ] - ) # The runtime half of the annotations. for path, node in cast("dict[object, object]", self.metadata).items(): if not isinstance(path, str): @@ -310,7 +311,11 @@ def from_json( `MetadataValidationError` with every problem found. """ - readings, members, problems = _read_consolidated_v3(data, context) + # The member sits under a group's key wherever it is read, so the + # levels a reader walks are counted from there, as in the group. + readings, members, problems = _read_consolidated_v3( + data, context, (ZARR_V3_CONSOLIDATED_METADATA_KEY,) + ) if len(problems) != 0: raise MetadataValidationError(problems) return construct(cls, metadata=_models(readings, members)[1]) @@ -367,8 +372,7 @@ def _consolidated_document( group: Callable[[ZarrV3GroupMetadata], object], ) -> dict[str, object]: """The member, each document it holds as `array` or `group` gives it.""" - # `must_understand` is emitted as the literal False: the field is typed - # permissively as `bool`, but `__post_init__` guarantees the value. + # `must_understand` is False by declaration, as `kind` is its literal. return { "kind": model.kind, "must_understand": False, @@ -497,14 +501,22 @@ def validate_node_metadata_v3( def _read_node_v3( - value: object, context: Context + value: object, context: Context, at: tuple[str | int, ...] = () ) -> tuple[ZarrV3NodeMetadataReading, ArrayMembersV3 | GroupMembersV3 | None]: - """`value` read as `read_node_metadata_v3` reads it, without models, and its members refined.""" + """`value` read as `read_node_metadata_v3` reads it, without models, and its members refined. + + `at` is where it sits in the document handed in, so the levels a + reader walks are counted from that one's root: a document past them + is the problem `_refine` reports for a container there, and not read. + """ + past = nested_past_the_levels(value, at) + if past is not None: + return ZarrV3UnknownNodeReading(within((past,), at)), None node_type, problems = _node_type(value) if node_type == "array": - return read_array_v3(value, context) + return read_array_v3(value, context, at=at) if node_type == "group": - return read_group_v3(value, context) + return read_group_v3(value, context, at=at) return ZarrV3UnknownNodeReading(problems), None @@ -566,39 +578,47 @@ def read_group_metadata_v3( def read_group_v3( - value: object, context: Context + value: object, context: Context, *, at: tuple[str | int, ...] = () ) -> tuple[ZarrV3GroupMetadataReading, GroupMembersV3 | None]: """`value`, a v3 group document, as `context` read it, without models, and its members refined; None when it is not an object. - A field a scope already read -- a model's own, handed back -- is taken - as it is, as `read_array_v3` takes one. + Every value is JSON: a field object built by hand, in a document + `consolidated_metadata` holds, is not, and is refused as + `read_array_v3` refuses one it is not told to hold. `at` is where the + document sits in the one handed in, as `read_array_v3` takes it: each + document its consolidated metadata holds is read from where it sits, + so a chain of them is bounded by the levels a reader walks, counted + from the outermost root. """ if not isinstance(value, Mapping): return ZarrV3GroupMetadataReading(problems=not_an_object(value)), None doc = cast("Mapping[object, object]", value) found: list[ValidationProblem] = list(missing_keys(GROUP_METADATA_REQUIRED_KEYS_V3, doc)) + past = members_past_the_levels(doc, at) + found.extend(past.values()) + whole, doc = doc, {key: item for key, item in doc.items() if key not in past} extra_fields, others = other_members( doc, GROUP_METADATA_STANDARD_KEYS_V3, additional_reserved_keys=frozenset({ZARR_V3_CONSOLIDATED_METADATA_KEY}), + at=at, ) found.extend(others) found.extend(check_literal(doc, "zarr_format", 3)) found.extend(check_literal(doc, "node_type", "group")) attributes: dict[str, JSONValue] | None = {} if "attributes" in doc: - attributes, problems = attributes_of(doc["attributes"]) + attributes, problems = attributes_of(doc["attributes"], at) found.extend(problems) - # consolidated_metadata: null, which a historical zarr-python bug wrote, - # is read as none, so those stores stay readable; the model never - # writes it back. - raw = doc.get(ZARR_V3_CONSOLIDATED_METADATA_KEY) + raw = doc.get(ZARR_V3_CONSOLIDATED_METADATA_KEY, UNSET) consolidated: dict[str, ZarrV3NodeMetadataReading] = {} held: Mapping[str, ArrayMembersV3 | GroupMembersV3] | UNSET = UNSET - if raw is not None: - consolidated, held, inside = _read_consolidated_v3(raw, context) + if raw is not UNSET: + consolidated, held, inside = _read_consolidated_v3( + raw, context, (*at, ZARR_V3_CONSOLIDATED_METADATA_KEY) + ) found.extend(_prefix(ZARR_V3_CONSOLIDATED_METADATA_KEY, inside)) - reading = ZarrV3GroupMetadataReading(consolidated, with_input(found, doc)) + reading = ZarrV3GroupMetadataReading(consolidated, with_input(found, whole)) return reading, GroupMembersV3(attributes, extra_fields, held) @@ -609,13 +629,22 @@ def read_group_v3( def _read_consolidated_v3( - value: object, context: Context + value: object, context: Context, at: tuple[str | int, ...] = () ) -> tuple[ dict[str, ZarrV3NodeMetadataReading], dict[str, ArrayMembersV3 | GroupMembersV3], tuple[ValidationProblem, ...], ]: - """An inline `consolidated_metadata` member, as `context` read it: each document it holds, read once, as `read_node_metadata_v3` reads one, by its path; the members of each a model can be built of; and every problem, located in the member.""" + """An inline `consolidated_metadata` member, as `context` read it: each document it holds, read once, as `read_node_metadata_v3` reads one, by its path; the members of each a model can be built of; and every problem, located in the member. + + `at` is where the member sits in the document handed in. Each container + the reader descends into -- the member, its `metadata`, each document + -- is judged where it sits, as `_refine` judges one, so a chain of + documents is bounded by the levels a reader walks. + """ + past = nested_past_the_levels(value, at) + if past is not None: + return {}, {}, within((past,), at) if not isinstance(value, Mapping): return {}, {}, (ValidationProblem((), "expected an object", "invalid_type"),) env = cast("Mapping[object, object]", value) @@ -632,18 +661,23 @@ def _read_consolidated_v3( members: dict[str, ArrayMembersV3 | GroupMembersV3] = {} node_types: dict[str, NodeType | None] = {} entries = env.get("metadata") + past = nested_past_the_levels(entries, (*at, "metadata")) if "metadata" in env and not isinstance(entries, Mapping): problems.append(ValidationProblem(("metadata",), "expected an object", "invalid_type")) + elif isinstance(entries, Mapping) and past is not None: + problems.extend(within((past,), at)) elif isinstance(entries, Mapping): for key, entry in cast("Mapping[object, object]", entries).items(): if not isinstance(key, str): problems.append( - ValidationProblem(("metadata",), f"non-string key {key!r}", "invalid_type") + ValidationProblem( + ("metadata",), f"non-string key {shown_key(key)}", "invalid_type" + ) ) continue faults = _key_problems(key) problems.extend(faults) - readings[key], child = _read_node_v3(entry, context) + readings[key], child = _read_node_v3(entry, context, (*at, "metadata", key)) if child is not None: members[key] = child problems.extend(_prefix("metadata", _prefix(key, readings[key].problems))) @@ -957,11 +991,12 @@ def to_json(self) -> ZarrV2GroupMetadataJSON: `to_key_value` to produce the spec-conforming split for storage (https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v2/v2.0.rst#L313; https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v2/v2.0.rst#L323-L330). """ - # to_json output shares no mutable state with the model. + # to_json output shares no mutable state with the model: the document + # is copied whole, one frame for each level of nesting. out: ZarrV2GroupMetadataJSON = {"zarr_format": self.zarr_format} if self.attributes is not UNSET: - out["attributes"] = copy.deepcopy(self.attributes) - return out + out["attributes"] = self.attributes + return cast("ZarrV2GroupMetadataJSON", copied(cast("JSONValue", out))) @classmethod def from_json(cls, data: object) -> ZarrV2GroupMetadata: @@ -971,7 +1006,9 @@ def from_json(cls, data: object) -> ZarrV2GroupMetadata: finds. The model shares no mutable state with `data`. """ # A read model shares no mutable state with what it read. - parsed = copy.deepcopy(parse_group_metadata_v2(data)) + parsed = cast( + "ZarrV2GroupMetadataJSON", copied(cast("JSONValue", parse_group_metadata_v2(data))) + ) return construct(cls, attributes=_v2_attributes(parsed)) @classmethod @@ -1059,7 +1096,7 @@ def to_json(self) -> dict[str, JSONValue]: # to_json output shares no mutable state with the model. return { "zarr_consolidated_format": self.zarr_consolidated_format, - "metadata": copy.deepcopy(self.metadata), + "metadata": copied(self.metadata), } @classmethod @@ -1106,10 +1143,11 @@ def _read_consolidated_v2( Each entry is the document its key names: a `.zattrs` is user data, and any other is JSON by RFC 8259. """ - normalized = arrays_to_tuples(data) - if not isinstance(normalized, Mapping): + # Each entry is refined below, arrays as tuples, no deeper than a + # reader walks; the document around them is walked as it is. + if not isinstance(data, Mapping): return {}, not_an_object(data) - doc = cast("Mapping[object, object]", normalized) + doc = cast("Mapping[object, object]", data) problems: list[ValidationProblem] = [ ValidationProblem((key,), "missing required key", "missing_key") for key in ("zarr_consolidated_format", "metadata") @@ -1138,7 +1176,7 @@ def _read_consolidated_v2( entry, found = refine(value, ("metadata", key)) problems.extend(found) refined[key] = entry - return refined, with_input(problems, data) + return refined, with_input(problems, doc) def _v2_attributes(document: ZarrV2GroupMetadataJSON) -> dict[str, JSONValue] | UNSET: diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_json_schema.py b/packages/zarr-metadata/src/zarr_metadata/model/_json_schema.py index 227a5a1916..b8b07e0a90 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_json_schema.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/_json_schema.py @@ -87,12 +87,10 @@ def _array(context: Context, schemas: Schemas) -> JSONSchema: def _group(schemas: Schemas) -> JSONSchema: - """A group document: its TypedDict, and the consolidated metadata the model reads, which a historical zarr-python bug wrote as `null`.""" + """A group document: its TypedDict, and the consolidated metadata the model reads.""" schema = schemas.object_of(ZarrV3GroupMetadataJSON) properties = cast("dict[str, JSONValue]", schema.get("properties", {})) - consolidated: JSONSchema = { - "anyOf": [schemas.of(ZarrV3ConsolidatedMetadataJSON), {"type": "null"}] - } + consolidated: JSONSchema = schemas.of(ZarrV3ConsolidatedMetadataJSON) return {**schema, "properties": {**properties, ZARR_V3_CONSOLIDATED_METADATA_KEY: consolidated}} diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_validation.py b/packages/zarr-metadata/src/zarr_metadata/model/_validation.py index 090de80b0d..989d6097ae 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_validation.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/_validation.py @@ -26,7 +26,7 @@ import dataclasses import json -from collections.abc import Mapping, Sequence +from collections.abc import Collection, Mapping, Sequence from dataclasses import dataclass from typing import TYPE_CHECKING, Any, Final, TypeGuard, TypeVar, cast, get_args, get_origin @@ -34,15 +34,17 @@ MetadataValidationError, ValidationProblem, arrays_to_tuples, + nested_past_the_levels, not_an_object, outside_of, refine_json, refine_user_data, + shown_key, validate_json, with_input, + within, ) from zarr_metadata._json import is_canonical_json as _is_canonical_json -from zarr_metadata._json import prefixed as _prefix from zarr_metadata._sentinel import UNSET from zarr_metadata._typed_json import typeddict_keys from zarr_metadata.v2.array import ZarrV2ArrayMetadataJSON @@ -140,7 +142,7 @@ def unexpected_keys( for key in doc: if not isinstance(key, str): problems.append( - ValidationProblem((), f"non-string document key {key!r}", "invalid_type") + ValidationProblem((), f"non-string document key {shown_key(key)}", "invalid_type") ) elif key not in allowed: problems.append(ValidationProblem((key,), f"unexpected key {key!r}", "unknown_key")) @@ -175,6 +177,7 @@ def other_members( standard_keys: frozenset[str], *, additional_reserved_keys: frozenset[str] = frozenset(), + at: Loc = (), ) -> tuple[dict[str, JSONValue], tuple[ValidationProblem, ...]]: """Each member outside `standard_keys` refined to JSON, and every problem, as `other_members_problems` finds them; a member that is not JSON is left out.""" members: dict[str, JSONValue] = {} @@ -183,13 +186,13 @@ def other_members( for key, value in doc.items(): if not isinstance(key, str): problems.append( - ValidationProblem((), f"non-string top-level key {key!r}", "invalid_type") + ValidationProblem((), f"non-string top-level key {shown_key(key)}", "invalid_type") ) continue if key in reserved_keys: continue - refined, found = refine_json(value, (key,)) - problems.extend(found) + refined, found = refine_json(value, (*at, key)) + problems.extend(within(found, at)) if len(found) == 0: members[key] = refined return members, tuple(problems) @@ -330,15 +333,15 @@ def _is_codec_v2(value: object) -> bool: ) -def _validate_codec_v2(value: object) -> tuple[ValidationProblem, ...]: - """Validate a v2 codec's required shape and JSON-valued configuration.""" +def _validate_codec_v2(value: object, loc: tuple[str | int, ...]) -> tuple[ValidationProblem, ...]: + """Validate a v2 codec's required shape and JSON-valued configuration, `value` sitting at `loc`.""" if not _is_codec_v2(value): return ( ValidationProblem( - (), "expected a codec configuration with a string 'id'", "invalid_type" + loc, "expected a codec configuration with a string 'id'", "invalid_type" ), ) - return validate_json(value) + return validate_json(value, loc) def validate_attributes(value: object) -> tuple[ValidationProblem, ...]: @@ -353,10 +356,33 @@ def validate_attributes(value: object) -> tuple[ValidationProblem, ...]: return attributes_of(value)[1] +def members_past_the_levels( + doc: Mapping[object, object], at: Loc +) -> dict[object, ValidationProblem]: + """Each member of `doc`, a document that sits at `at`, that is a container past the levels a reader walks, with the problem it is, located in the document. + + A reader reports them and walks no further into them, as `refine_json` + of the whole document would judge them: a document a level before the + cap holds its scalars, and nothing else. + """ + past: dict[object, ValidationProblem] = {} + for key, item in doc.items(): + if isinstance(key, str): + problem = nested_past_the_levels(item, (*at, key)) + if problem is not None: + (past[key],) = within((problem,), at) + return past + + def attributes_of( - value: object, + value: object, at: Loc = () ) -> tuple[dict[str, JSONValue] | None, tuple[ValidationProblem, ...]]: - """An `attributes` value refined as user data, and every problem `validate_attributes` finds; None when it is not an object with string keys, or holds a value that is not JSON.""" + """An `attributes` value refined as user data, and every problem `validate_attributes` finds; None when it is not an object with string keys, or holds a value that is not JSON. + + `at` is where the document holding it sits in the one handed in, so + the levels a reader walks are counted from that one's root; the + problems are located in the document holding it. + """ if not isinstance(value, Mapping) or not all( isinstance(k, str) for k in cast("Mapping[object, object]", value) ): @@ -368,10 +394,10 @@ def attributes_of( attributes: dict[str, JSONValue] = {} problems: list[ValidationProblem] = [] for key, item in cast("Mapping[str, object]", value).items(): - refined, found = refine_user_data(item, ("attributes", key)) + refined, found = refine_user_data(item, (*at, "attributes", key)) problems.extend(found) attributes[key] = refined - return (attributes if len(problems) == 0 else None), tuple(problems) + return (attributes if len(problems) == 0 else None), within(problems, at) _MEMBERS_V3: Final = typeddict_keys(ZarrV3ArrayMetadataJSON).members @@ -395,15 +421,15 @@ def attributes_of( class ZarrV3ArrayMetadataReading: """A v3 array document as a scope read it, whatever it holds: each extension point, its codecs as a pipeline, every problem, and the model when there is none. - A field the document does not hold is None, and a list of them it does + A field the document does not hold is `UNSET`, and a list of them it does not hold as a list is empty. """ - data_type: Resolved[DataTypeDefinition[Any]] | None = None + data_type: Resolved[DataTypeDefinition[Any]] | UNSET = UNSET """The data type, as the scope read it.""" - chunk_grid: Resolved[ChunkGridDefinition[Any]] | None = None + chunk_grid: Resolved[ChunkGridDefinition[Any]] | UNSET = UNSET """The chunk grid, as the scope read it.""" - chunk_key_encoding: Resolved[ChunkKeyEncodingDefinition[Any]] | None = None + chunk_key_encoding: Resolved[ChunkKeyEncodingDefinition[Any]] | UNSET = UNSET """The chunk key encoding, as the scope read it.""" chunk: Chunk = dataclasses.field(default_factory=Chunk) """The chunks the codecs are handed: the lengths the grid's chunks take along each axis of the shape, of the data type.""" @@ -429,8 +455,8 @@ def fields(self) -> Iterator[tuple[Loc, Resolved[Any]]]: ("chunk_grid", self.chunk_grid), ("chunk_key_encoding", self.chunk_key_encoding), ): - if field is not None: - yield from fields_of(field, (key,)) + if field is not UNSET: + yield from fields_of(cast("Resolved[Any]", field), (key,)) for index, stage in enumerate(self.pipeline): yield from fields_of(stage.codec, ("codecs", index)) for index, transformer in enumerate(self.storage_transformers): @@ -493,16 +519,23 @@ class ArrayMembersV3: def read_field( - value: object, kind: type[Definition[Any]], context: Context, loc: Loc + value: object, + kind: type[Definition[Any]], + context: Context, + loc: Loc, + *, + held: Collection[object], ) -> tuple[Resolved[Any], tuple[ValidationProblem, ...]]: - """`value`, one of a document's fields, as `context` reads it -- or as it is, when a scope has read it already. - - A model's own fields come back this way, so `update` reads only the - members it is given, and a field read in one scope keeps the - definition that read it, however the scope it is handed on in differs: - a pydantic model instance is taken as it is, too. + """`value`, one of a document's fields, as `context` reads it -- or as it is, when it is one of `held`: the fields a model holds, read already. + + A model's own fields come back this way, so a model checks itself + without reading them again, and a field read in one scope keeps the + definition that read it, however the scope it is handed on in differs. + Any other field object -- in a document a caller hands in, or among + the members `update` is given -- is not JSON, and is refused as such, + so nothing built by hand passes as read but what a model holds. """ - if _read_as(value, kind): + if _read_as(value, kind) and any(value is field for field in held): return cast("Resolved[Any]", value), () return resolve(value, kind, context, loc) @@ -513,18 +546,26 @@ def _read_as(value: object, kind: type[Definition[Any]]) -> bool: def read_array_v3( - value: object, context: Context + value: object, context: Context, *, held: Collection[object] = (), at: Loc = () ) -> tuple[ZarrV3ArrayMetadataReading, ArrayMembersV3 | None]: """`value`, a v3 array document, as `context` read it, without its model, and its other members refined; None when it has a problem. - `read_array_metadata_v3` builds the model from the two. A field a - scope already read -- a model's own, handed back -- is taken as it is. + `read_array_metadata_v3` builds the model from the two. `held` are the + fields a model holds, read already, which are taken as they are where + the document holds them; any other field object in the document is + refused as not JSON. `at` is where the document sits in the one handed + in -- a document consolidated metadata holds sits three levels below + its group's -- so the levels a reader walks are counted from that + one's root; the problems are located in this document. """ if not isinstance(value, Mapping): return ZarrV3ArrayMetadataReading(problems=not_an_object(value)), None doc = cast("Mapping[object, object]", value) problems: list[ValidationProblem] = list(missing_keys(ARRAY_METADATA_REQUIRED_KEYS_V3, doc)) - extra_fields, found = other_members(doc, ARRAY_METADATA_STANDARD_KEYS_V3) + past = members_past_the_levels(doc, at) + problems.extend(past.values()) + whole, doc = doc, {key: item for key, item in doc.items() if key not in past} + extra_fields, found = other_members(doc, ARRAY_METADATA_STANDARD_KEYS_V3, at=at) problems.extend(found) problems.extend(check_literal(doc, "zarr_format", 3)) problems.extend(check_literal(doc, "node_type", "array")) @@ -543,15 +584,15 @@ def read_array_v3( read: dict[str, Resolved[Any]] = {} for key, kind in _EXTENSION_POINTS_V3: if key in doc: - read[key], found = read_field(doc[key], kind, context, (key,)) - problems.extend(found) + read[key], found = read_field(doc[key], kind, context, (*at, key), held=held) + problems.extend(within(found, at)) # The fill value is JSON, and judged by the data type the scope read, # when there is one: a data type nothing in scope claims leaves it # unjudged. fill_value: JSONValue = None if "fill_value" in doc: - fill_value, found = refine_json(doc["fill_value"], ("fill_value",)) - problems.extend(found) + fill_value, found = refine_json(doc["fill_value"], (*at, "fill_value")) + problems.extend(within(found, at)) if len(found) == 0 and "data_type" in read: problems.extend(fill_value_problems(read["data_type"], fill_value, ("fill_value",))) # The chunk grid is judged against the shape, once both are read, and @@ -570,9 +611,9 @@ def read_array_v3( else: listed[key] = [] for index, entry in enumerate(entries): - resolved, found = read_field(entry, kind, context, (key, index)) + resolved, found = read_field(entry, kind, context, (*at, key, index), held=held) listed[key].append(resolved) - problems.extend(found) + problems.extend(within(found, at)) # The codecs are read as a pipeline, the first handed the grid's chunks # of the array's data type: in order, each judged against the chunk it # is handed. That holds one array -> bytes codec, so it is not empty. @@ -583,7 +624,7 @@ def read_array_v3( problems.extend(found) attributes: dict[str, JSONValue] | None = {} if "attributes" in doc: - attributes, found = attributes_of(doc["attributes"]) + attributes, found = attributes_of(doc["attributes"], at) problems.extend(found) if "dimension_names" in doc: # Simple typed sequences (dimension_names, shape, chunks) report a single @@ -609,13 +650,13 @@ def read_array_v3( ) ) reading = ZarrV3ArrayMetadataReading( - data_type=read.get("data_type"), - chunk_grid=read.get("chunk_grid"), - chunk_key_encoding=read.get("chunk_key_encoding"), + data_type=read.get("data_type", UNSET), + chunk_grid=read.get("chunk_grid", UNSET), + chunk_key_encoding=read.get("chunk_key_encoding", UNSET), chunk=chunk, pipeline=pipeline, storage_transformers=tuple(listed.get("storage_transformers", ())), - problems=with_input(problems, doc), + problems=with_input(problems, whole), ) if len(problems) != 0 or shape is None or attributes is None: return reading, None @@ -709,20 +750,25 @@ def validate_array_metadata_v2(value: object) -> tuple[ValidationProblem, ...]: "invalid_value", ) ) - if "dtype" in doc and not _is_dtype_v2(doc["dtype"]): - problems.append( - ValidationProblem( - ("dtype",), - "expected a v2 dtype string or an array of field records", - "invalid_type", + if "dtype" in doc: + # JSON first, nested no deeper than a reader walks, then its shape: + # field records nest a dtype two levels a record, without bound. + found = validate_json(doc["dtype"], ("dtype",)) + problems.extend(found) + if len(found) == 0 and not _is_dtype_v2(doc["dtype"]): + problems.append( + ValidationProblem( + ("dtype",), + "expected a v2 dtype string or an array of field records", + "invalid_type", + ) ) - ) if "order" in doc and doc["order"] not in ("C", "F"): problems.append(outside_of(("order",), doc["order"], ("C", "F"))) if "compressor" in doc: compressor = doc["compressor"] if compressor is not None: - problems.extend(_prefix("compressor", _validate_codec_v2(compressor))) + problems.extend(_validate_codec_v2(compressor, ("compressor",))) if "filters" in doc: filters = doc["filters"] if filters is not None and ( @@ -739,13 +785,13 @@ def validate_array_metadata_v2(value: object) -> tuple[ValidationProblem, ...]: # "A list of JSON objects providing codec configurations, or # null" (https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v2/v2.0.rst#L76-L79): an empty list is a list. for index, item in enumerate(filters): - problems.extend(_prefix("filters", _prefix(index, validate_json(item)))) + problems.extend(validate_json(item, ("filters", index))) if "dimension_separator" in doc and doc["dimension_separator"] not in (".", "/"): problems.append( outside_of(("dimension_separator",), doc["dimension_separator"], (".", "/")) ) if "fill_value" in doc: - problems.extend(_prefix("fill_value", validate_json(doc["fill_value"]))) + problems.extend(validate_json(doc["fill_value"], ("fill_value",))) if "attributes" in doc: problems.extend(validate_attributes(doc["attributes"])) return with_input(problems, doc) @@ -835,7 +881,7 @@ def load_store_json(mapping: Mapping[StoreKey, bytes], key: str) -> object: raise MetadataValidationError(with_input((refused,), stored)) try: return json.loads(raw) - except (UnicodeDecodeError, ValueError) as exc: + except (UnicodeDecodeError, ValueError, RecursionError) as exc: raise MetadataValidationError( [ValidationProblem((key,), f"invalid JSON: {exc}", "invalid_json")] ) from exc diff --git a/packages/zarr-metadata/src/zarr_metadata/pydantic.py b/packages/zarr-metadata/src/zarr_metadata/pydantic.py index adf285f5f4..9c436a2598 100644 --- a/packages/zarr-metadata/src/zarr_metadata/pydantic.py +++ b/packages/zarr-metadata/src/zarr_metadata/pydantic.py @@ -18,9 +18,13 @@ ArrayManifest.model_validate(data, context={"zarr_metadata_context": SCOPE}) An existing model instance passes through unchanged, as pydantic's does, and -serialization emits the canonical document via `to_json`. `MetadataValidationError` subclasses `ValueError`, -so a failed parse surfaces as a pydantic `ValidationError` carrying the -loc-annotated problem messages. +serialization emits the canonical document via `to_json`. A failed parse +surfaces as a pydantic `ValidationError` with one line error per problem, +as pydantic reports its own: its `type` the problem's `kind`, at the +problem's `loc` under the field's, with the `input` found there and the +problem's `ctx` (a message holding a ctx placeholder rides in the ctx as +`message`, as `_line_error` says). Annotate a field with these types, not the core classes, +which pydantic cannot build a schema for. A node's attributes may hold `NaN`, `Infinity` or `-Infinity`, which the models read and write as zarr-python does. Pydantic writes JSON by its own @@ -44,11 +48,13 @@ class ArrayManifest(BaseModel): from __future__ import annotations from collections.abc import Mapping -from typing import TYPE_CHECKING, Annotated, Final, Protocol, TypeVar, cast +from typing import TYPE_CHECKING, Annotated, Final, LiteralString, Protocol, TypeVar, cast from pydantic import BeforeValidator, InstanceOf, PlainSerializer, ValidationInfo +from pydantic_core import InitErrorDetails, PydanticCustomError, ValidationError from zarr_metadata import model as _model +from zarr_metadata._json import MetadataValidationError, ValidationProblem, value_at from zarr_metadata._pydantic_schema import ( ZarrV2ArrayMetadataJSON as _ZarrV2ArrayMetadataSchema, ) @@ -67,6 +73,7 @@ class ArrayManifest(BaseModel): from zarr_metadata._pydantic_schema import ( ZarrV3GroupMetadataJSON as _ZarrV3GroupMetadataSchema, ) +from zarr_metadata._sentinel import UNSET from zarr_metadata.v3._registry import CORE_AND_EXTENSIONS, Context if TYPE_CHECKING: @@ -76,16 +83,63 @@ class ArrayManifest(BaseModel): def _coerce_to(cls: type[_M], parse: Callable[[object], _M]) -> Callable[[object], _M]: - """A validator that passes instances of `cls` through and parses anything else.""" + """A validator that passes instances of `cls` through and parses anything else, each problem a line error of pydantic's.""" def coerce(value: object) -> _M: if isinstance(value, cls): return value - return parse(value) + return _as_pydantic_raises(cls, value, lambda: parse(value)) return coerce +def _as_pydantic_raises(cls: type[_M], value: object, read: Callable[[], _M]) -> _M: + """What `read` gives of `value`, or the `MetadataValidationError` it raises as a `pydantic_core.ValidationError`: one line error per problem, its type the problem's kind, at the problem's loc, with its input and ctx. + + A missing key's problem reports the object missing it, as pydantic's + own `missing` error does; one that holds nothing else, what it found + not being JSON a reader walks, reports what sits at its loc. + """ + try: + return read() + except MetadataValidationError as error: + line_errors = [_line_error(problem, value) for problem in error.problems] + raise ValidationError.from_exception_data(cls.__name__, line_errors) from error + + +def _line_error(problem: ValidationProblem, value: object) -> InitErrorDetails: + """`problem` as one of a `ValidationError`'s line errors: its type the problem's kind, at its loc, with its input and ctx, and its message as `msg`. + + pydantic renders a message as a template of its ctx, each `{key}` the + ctx holds replaced, one key at a time in the ctx's order, with no way + to escape one. A message that holds one -- a document value written + `{expected}`, shown in it -- is handed whole as the ctx's `message`, + last, where what it carries is scanned for no key after it, and the + template is that placeholder, so `msg` is the problem's message + whatever it holds; a ctx member of that name yields to it. + """ + template, ctx = problem.message, dict(problem.ctx) + if any(f"{{{key}}}" in template for key in ctx): + ctx.pop("message", None) + template, ctx = "{message}", {**ctx, "message": problem.message} + return InitErrorDetails( + type=PydanticCustomError(problem.kind, cast("LiteralString", template), ctx), + loc=problem.loc, + input=_input_of(problem, value), + ) + + +def _input_of(problem: ValidationProblem, value: object) -> object: + """What a line error reports as its input: what the problem found; for a key that is missing, what `value` holds at the loc above, as pydantic's own `missing` reports the object; for what the problem could not hold, not being JSON a reader walks, what sits at its loc; and `value` itself where that is nothing either.""" + if problem.input is not UNSET: + return problem.input + if problem.kind == "missing_key": + above = value_at(value, problem.loc[:-1]) if len(problem.loc) != 0 else UNSET + return value if above is UNSET else above + there = value_at(value, problem.loc) + return value if there is UNSET else there + + CONTEXT_KEY: Final = "zarr_metadata_context" """The item of a mapping pydantic's validation context is that holds the scope a v3 field type reads in.""" @@ -103,7 +157,7 @@ def _read_in_scope(cls: type[_M], read: _Reads[_M]) -> Callable[[object, Validat def coerce(value: object, info: ValidationInfo) -> _M: if isinstance(value, cls): return value - return read(value, context=_scope(info.context)) + return _as_pydantic_raises(cls, value, lambda: read(value, context=_scope(info.context))) return coerce diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/_common.py b/packages/zarr-metadata/src/zarr_metadata/v3/_common.py index 3ccc21e1e8..c55a5123b3 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/_common.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/_common.py @@ -7,8 +7,9 @@ the validators from `zarr_metadata.model`. """ +import re from collections.abc import Mapping -from typing import TypeGuard, cast +from typing import Final, TypeGuard, cast from typing_extensions import TypeAliasType @@ -18,7 +19,8 @@ ValidationProblem, arrays_to_tuples, is_canonical_json, - prefixed, + shown, + shown_key, validate_json, with_input, ) @@ -60,9 +62,9 @@ def validate_metadata_field_v3( """Return every reason `value` is not a v3 metadata field. A metadata field is a bare name, or an envelope around a configuration - whose members are JSON: an object of a string `name`, a `configuration` - that is an object of string keys, a boolean `must_understand`, and - nothing else. + whose members are JSON: an object of a `name` as the spec names an + extension, a `configuration` that is an object of string keys, a + boolean `must_understand`, and nothing else. """ envelope = envelope_problems(value, allow_must_understand_false=allow_must_understand_false) return with_input((*envelope, *_configuration_json_problems(value)), value) @@ -72,18 +74,62 @@ def validate_metadata_field_v3( """The members a metadata field's envelope declares: a key beside them is an unknown key.""" +_EXTENSION_NAME: Final = re.compile(r"[a-z][a-z0-9_.-]+") +"""A registered extension name: the spec's `^[a-z][a-z0-9-_.]+$`, its hyphen last so no engine reads it as a range.""" + +_URI: Final = re.compile(r"[A-Za-z][A-Za-z0-9+.-]*:[A-Za-z0-9._~:/?#\[\]@!$&'()*+,;=%-]+") +"""A URI, as far as a name is judged: RFC 3986's scheme, its colon, and one or more of the characters a URI is written in -- its unreserved and reserved sets, and the `%` of percent-encoding (https://www.rfc-editor.org/rfc/rfc3986#section-2) -- the names earlier versions of the spec required. + +The characters are listed rather than written `\\S`, which Python's `re` +and ECMA-262, the dialect a JSON Schema `pattern` is read in, disagree +on: `\\x1c`-`\\x1f` and `\\x85` are whitespace to one and `\\ufeff` to the +other, so a name this reader refused, a validator of the schema would +accept, or the other way round. +""" + +EXTENSION_NAME_SCHEMA_PATTERN: Final = rf"^({_EXTENSION_NAME.pattern}|{_URI.pattern})(?![\s\S])" +r"""What `well_named` accepts, as a JSON Schema `pattern`: the two patterns it matches whole, `(?![\s\S])` where `$` would take a final newline, as `_RAW_BYTES_SCHEMA_PATTERN` has it.""" + + +def well_named(name: str) -> bool: + """Whether `name` is an extension name as the spec names one. + + A registered name "MUST start with one lower case letter a-z and then + be followed by only lower case letters a-z, numerals 0-9, underscores, + dots and dashes", regex `^[a-z][a-z0-9-_.]+$` + (https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v3/core/index.rst#L1603-L1606); + a URI, which earlier versions of the spec required, is "still + permitted" + (https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v3/core/index.rst#L1614-L1618). + """ + return _EXTENSION_NAME.fullmatch(name) is not None or _URI.fullmatch(name) is not None + + +def name_problem(name: str, at: tuple[str | int, ...]) -> ValidationProblem | None: + """The problem `name`, at `at`, is when the spec gives no extension such a name; None when it does, as `well_named` says.""" + if well_named(name): + return None + message = ( + "expected an extension name -- lower-case letters, digits, '-', '_' and '.', " + f"starting with a letter -- or a URI, got {shown(name)}" + ) + return ValidationProblem(at, message, "invalid_value") + + def envelope_problems( value: object, *, allow_must_understand_false: bool ) -> tuple[ValidationProblem, ...]: """Every reason `value` is not a v3 metadata field's envelope, what its configuration holds left unjudged. - The envelope is what sits around the configuration: a string `name`, a - `configuration` that is an object of string keys, a boolean - `must_understand`, and nothing else. `resolve` asks this of a field it - has refined to JSON already, so a configuration is walked once. + The envelope is what sits around the configuration: a `name` as the + spec names an extension, a `configuration` that is an object of string + keys, a boolean `must_understand`, and nothing else. `resolve` asks + this of a field it has refined to JSON already, so a configuration is + walked once. """ if isinstance(value, str): - return () + bad = name_problem(value, ()) + return () if bad is None else (bad,) if not isinstance(value, Mapping): return ( ValidationProblem( @@ -97,12 +143,18 @@ def envelope_problems( for key in field: if not isinstance(key, str): problems.append( - ValidationProblem((), f"non-string metadata field key {key!r}", "invalid_type") + ValidationProblem( + (), f"non-string metadata field key {shown_key(key)}", "invalid_type" + ) ) elif key not in ENVELOPE_KEYS: problems.append(ValidationProblem((key,), f"unexpected key {key!r}", "unknown_key")) - if not isinstance(field.get("name"), str): + if "name" not in field: + problems.append(ValidationProblem(("name",), "missing required key", "missing_key")) + elif not isinstance(field["name"], str): problems.append(ValidationProblem(("name",), "expected a string name", "invalid_type")) + elif (bad := name_problem(field["name"], ("name",))) is not None: + problems.append(bad) if "configuration" in field: configuration = field["configuration"] if not isinstance(configuration, Mapping): @@ -143,14 +195,14 @@ def _configuration_json_problems(value: object) -> tuple[ValidationProblem, ...] return tuple( found for key, item in cast("Mapping[str, object]", members).items() - for found in prefixed("configuration", prefixed(key, validate_json(item))) + for found in validate_json(item, ("configuration", key)) ) def is_metadata_field_v3(value: object) -> TypeGuard[ZarrV3MetadataFieldJSON]: - """Whether `value` is a v3 metadata field: a bare name or a named config.""" + """Whether `value` is a v3 metadata field: a bare name as the spec names an extension, or a named config.""" if isinstance(value, str): - return True + return well_named(value) if not isinstance(value, dict): return False field = cast("dict[object, object]", value) @@ -167,6 +219,7 @@ def parse_metadata_field_v3(value: object) -> ZarrV3MetadataFieldJSON: __all__ = [ "ENVELOPE_KEYS", + "EXTENSION_NAME_SCHEMA_PATTERN", "ChunkGridField", "ChunkKeyEncodingField", "CodecField", @@ -176,6 +229,8 @@ def parse_metadata_field_v3(value: object) -> ZarrV3MetadataFieldJSON: "ZarrV3MetadataFieldJSON", "envelope_problems", "is_metadata_field_v3", + "name_problem", "parse_metadata_field_v3", "validate_metadata_field_v3", + "well_named", ] diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py b/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py index 4e2c154d73..3f76daa781 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py @@ -34,6 +34,7 @@ import re from collections.abc import Callable, Iterable, Iterator, Mapping from dataclasses import dataclass +from types import MappingProxyType from typing import ( TYPE_CHECKING, Any, @@ -74,6 +75,7 @@ ) from zarr_metadata.v3._common import ( ENVELOPE_KEYS, + EXTENSION_NAME_SCHEMA_PATTERN, ChunkGridField, ChunkKeyEncodingField, CodecField, @@ -81,6 +83,8 @@ StaticCodecField, StorageTransformerField, envelope_problems, + name_problem, + well_named, ) if TYPE_CHECKING: @@ -185,6 +189,11 @@ class Definition(Generic[C]): struct's field types -- which is nothing when no scope read it. `judge` is the two, for a caller holding JSON. + Each function is handed the configuration as a read-only view, no + `dict`: `copy.deepcopy` and `json.dumps` refuse it, and a function that + folds a spelling builds a new mapping, `{**configuration}` without the + member, rather than editing what it was handed. + `canonical` is where two spellings of the configuration that mean the same thing are made one. @@ -255,7 +264,7 @@ def judge(self, value: object, loc: Loc = ()) -> tuple[C | None, Problems]: configuration, problems = self.check(value, loc) if configuration is None: return None, problems - refused = ruled(self, lambda: self.rules(configuration, _nothing_nested()), loc) + refused = ruled(self, lambda: self.rules(read_only(configuration), _nothing_nested()), loc) return (configuration if len(refused) == 0 else None), ( *problems, *with_input(refused, value, loc), @@ -271,6 +280,12 @@ def _malformed(definition: Definition[Any]) -> str | None: value = getattr(definition, member) if not callable(value): return f"{name!r}: {member} is a function, got {value!r}" + if definition.name != RAW_BYTES_NAME and not well_named(definition.name): + return ( + f"{definition.name!r} is not a name the spec gives an extension -- lower-case " + "letters, digits, '-', '_' and '.', starting with a letter, or a URI -- so no " + "document names it, and nothing would ever read with this definition" + ) return None @@ -294,8 +309,8 @@ def _function_members(kind: type[Definition[Any]]) -> tuple[str, ...]: (https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v3/data-types/index.rst#L46-L47). """ -RAW_BYTES_NAME_PATTERN: Final = re.compile(r"r([0-9]+)") -"""A name that writes raw bits: `r` and a size in bits, matched whole, whether or not the size is allowed. +RAW_BYTES_NAME_PATTERN: Final = re.compile(r"r([0-9]{1,100})") +"""A name that writes raw bits: `r` and a size in bits of up to a hundred digits, matched whole, whether or not the size is allowed. ASCII digits only: `\\d` would also match every other Unicode decimal, so `r\uff11\uff16` would be read as sixteen bits, and a third-party name @@ -746,7 +761,7 @@ def _read_by(definition: Definition[Any], schemas: Schemas) -> list[JSONValue]: def _unclaimed(table: Mapping[str, Definition[Any]]) -> list[JSONValue]: """The fields no definition in `table` claims: a name none of them is written with, bare or with any configuration.""" claimed: list[JSONValue] = [written_name(definition) for definition in table.values()] - name: JSONSchema = {"type": "string"} + name: JSONSchema = {"type": "string", "pattern": EXTENSION_NAME_SCHEMA_PATTERN} if len(claimed) != 0: name["not"] = {"anyOf": claimed} envelope: JSONSchema = { @@ -847,6 +862,16 @@ def _usable(problems: Sequence[ValidationProblem]) -> bool: return all(found.kind == "unknown_key" for found in problems) +def read_only(configuration: C) -> C: + """`configuration` as a definition's functions are handed it: a read-only view of the field's own, so a function that assigns a member fails there. + + The view is of the members: what a member holds, an object among a + struct's `fields` say, is the field's own, and a function that writes + into one writes into the field. + """ + return cast("C", MappingProxyType(cast("Mapping[str, JSONValue]", configuration))) + + def asked(definition: Definition[Any], what: str, ask: Callable[[], T], at: Loc | None = None) -> T: """What `ask`, a call of `definition`'s `what`, gives. @@ -1019,7 +1044,7 @@ def to_json(self) -> JSONValue: A name that carries its configuration, as raw bits' does, is written alone. """ - return copied(document_json(self)) + return copied(cast("JSONValue", document_json(self))) @dataclass(frozen=True, slots=True, kw_only=True) @@ -1053,8 +1078,8 @@ def __post_init__(self) -> None: # arguments dropped, as `resolve` drops them. object.__setattr__(self, "read_as", as_kind(self.read_as)) name = cast("object", self.name) - if not isinstance(name, str): - msg = f"a field nothing in scope claims is named, got {name!r}" + if not isinstance(name, str) or not well_named(name): + msg = f"a field nothing in scope claims is named as the spec names an extension, got {name!r}" raise TypeError(msg) _, written, _ = named_configuration(self.json) configuration: Mapping[str, object] = {} if written is None else written @@ -1072,15 +1097,15 @@ def nested(self) -> Nested: def to_json(self) -> JSONValue: """The field as a document writes it, sharing nothing with the field: its configuration as written, in the envelope every reader takes, as `Read.to_json` writes one.""" - return copied(document_json(self)) + return copied(cast("JSONValue", document_json(self))) @dataclass(frozen=True, slots=True, kw_only=True) class Refused(Generic[D]): """A field that could not be read -- not a field at all, not JSON, or refused by the definition that claims its name -- as its problems say.""" - json: JSONValue | None - """The field as written, refined: arrays as tuples; None when it is not JSON.""" + json: JSONValue | UNSET + """The field as written, refined: arrays as tuples; `UNSET` when it is not JSON, which no document holds.""" name: str | None """The name it is written with; None when it names none.""" read_as: type[Definition[Any]] @@ -1147,10 +1172,11 @@ def field_key(field: Resolved[Any]) -> tuple[object, ...]: # definition's own members. for loc in field.nested: configuration = _replaced(configuration, loc, None) + view = read_only(cast("Mapping[str, JSONValue]", configuration)) spelled = asked( definition, "canonical", - lambda: json_text(cast("JSONValue", definition.canonical(configuration))), + lambda: json_text(cast("JSONValue", dict(definition.canonical(view)))), ) return ("read", definition, spelled, _nested_key(field.nested)) if isinstance(field, Unclaimed): @@ -1160,7 +1186,7 @@ def field_key(field: Resolved[Any]) -> tuple[object, ...]: field.read_as, field.name, field.definition, - json_text(field.json), + UNSET if field.json is UNSET else json_text(field.json), _nested_key(field.nested), ) @@ -1170,8 +1196,8 @@ def _nested_key(nested: Nested) -> tuple[tuple[Loc, tuple[object, ...]], ...]: return tuple((loc, field_key(inner)) for loc, inner in nested.items()) -def document_json(field: Resolved[Any]) -> JSONValue: - """A field as a document writes it, holding the field's own values: as `to_json` writes it, or as it was written when it was refused. +def document_json(field: Resolved[Any]) -> JSONValue | UNSET: + """A field as a document writes it, holding the field's own values: as `to_json` writes it, or as it was written when it was refused, `UNSET` for one that was not JSON. What a writer serializes, which changes nothing, so it copies nothing; `to_json` is this, copied. @@ -1216,7 +1242,7 @@ class Chunk: lengths: Lengths | None = None """Per axis, the lengths the chunks take along it; None when not even the number of axes is known.""" data_type: Resolved[DataTypeDefinition[Any]] | None = None - """The data type field of the values, as a scope read it; None when no field says what they are.""" + """The data type field of the values, as a scope read it; None when no field says what they are: a document naming none, which its reading holds as `UNSET`, hands the pipeline a chunk of no known type.""" def __post_init__(self) -> None: lengths = cast("object", self.lengths) @@ -1278,7 +1304,7 @@ def with_problems( document found with it where it stands -- its place in the pipeline, the chunk it is handed, the array's shape -- so a field with none is valid there, and `canonical_of` spells it. A function of the problems, - as zod's `flattenError` is of the issues, grouping them at every + as zod's `treeifyError` is of the issues, grouping them at every depth: a problem with a shard's inner codec is the inner codec's, and the shard's. Each field comes before the fields it holds, as `fields_of` gives them, so the last field whose problems hold a @@ -1333,7 +1359,7 @@ def fill_value_problems( return with_input(problems, value, loc) refused = ruled( definition, - lambda: definition.fill_value_rules(configuration, data_type.nested, typed), + lambda: definition.fill_value_rules(read_only(configuration), data_type.nested, typed), loc, ) return with_input((*problems, *refused), value, loc) @@ -1375,7 +1401,8 @@ def spelled_canonically( definition, "fill_value_canonical", lambda: cast( - "object", definition.fill_value_canonical(configuration, data_type.nested, value) + "object", + definition.fill_value_canonical(read_only(configuration), data_type.nested, value), ), ) refined, problems = refine_json(spelled, ()) @@ -1402,7 +1429,7 @@ def storage_of(data_type: Resolved[DataTypeDefinition[Any]]) -> StorageClass | N found = asked( definition, "storage", - lambda: cast("object", definition.storage(configuration, data_type.nested)), + lambda: cast("object", definition.storage(read_only(configuration), data_type.nested)), ) if found is not None and found not in get_args(StorageClass): msg = ( @@ -1437,14 +1464,18 @@ def chunk_grid_lengths( definition, configuration = chunk_grid.definition, chunk_grid.configuration at = (*loc, "configuration") problems = ruled( - definition, lambda: definition.shape_rules(configuration, chunk_grid.nested, shape), at + definition, + lambda: definition.shape_rules(read_only(configuration), chunk_grid.nested, shape), + at, ) if len(problems) != 0: return unknown, with_input(problems, chunk_grid.json, loc) lengths = asked( definition, "chunk lengths", - lambda: cast("object", definition.chunk_lengths(configuration, chunk_grid.nested, shape)), + lambda: cast( + "object", definition.chunk_lengths(read_only(configuration), chunk_grid.nested, shape) + ), at, ) if not _is_lengths(lengths): @@ -1487,11 +1518,14 @@ def resolve( refined, problems = refine_json(data, loc) if len(problems) != 0: # Not JSON, so not read; its name, if it has one, still says what - # claims it. + # claims it, and one the spec does not give an extension is a + # problem here as on the other path, asked of no definition. name = named_configuration(data)[0] - claimant = None if name is None else context.claimant(asked, name) - refused = Refused(json=None, name=name, read_as=asked, definition=claimant) - return cast("Resolved[D]", refused), with_input(problems, data, loc) + bad = None if name is None else name_problem(name, (*loc, "name")) + claimant = None if name is None or bad is not None else context.claimant(asked, name) + refused = Refused(json=UNSET, name=name, read_as=asked, definition=claimant) + found = problems if bad is None else (bad, *problems) + return cast("Resolved[D]", refused), with_input(found, data, loc) resolved, found = _resolve_field(refined, asked, context, loc) return cast("Resolved[D]", resolved), with_input(found, data, loc) @@ -1517,6 +1551,10 @@ def _read( name, given, malformed = named_configuration(data) if name is None: return Refused(json=data, name=None, read_as=kind), () + if not well_named(name): + # The envelope rule every reader runs first reports it; no + # definition is asked to claim it. + return Refused(json=data, name=name, read_as=kind), () definition = context.claimant(kind, name) if len(malformed) != 0: # A configuration that is not an object, which the envelope's @@ -1544,7 +1582,9 @@ def _read( inside.extend(_envelope(field)) inner, found_inside = _read(field.json, field.kind, context, field.loc) within[field.loc[len(at) :]] = inner - written[field.loc] = document_json(inner) + # Refined JSON came in, so what was read of it is JSON, or a field + # refused for what it holds, never for not being JSON. + written[field.loc] = cast("JSONValue", document_json(inner)) inside.extend(found_inside) inside.extend(_sized(field, inner)) # The rules may read a field the configuration holds by its name, so @@ -1558,7 +1598,9 @@ def _read( ) own = list(found) if configuration is not None: - own.extend(ruled(definition, lambda: definition.rules(configuration, within), at)) + own.extend( + ruled(definition, lambda: definition.rules(read_only(configuration), within), at) + ) if configuration is None or not _usable(own): refused = Refused(json=data, name=name, read_as=kind, definition=definition, nested=within) return refused, (*own, *inside) @@ -1689,7 +1731,14 @@ def _canonical_field(resolved: Read[Any]) -> JSONValue | None: if simplest is None: return None configuration = _replaced(configuration, loc, simplest) - simplified = cast("Mapping[str, JSONValue]", definition.canonical(configuration)) + # The view every function of a definition is handed; what `canonical` + # gives, the view itself when nothing is folded, is taken as a dict. + view = read_only(cast("Mapping[str, JSONValue]", configuration)) + simplified = asked( + definition, + "canonical", + lambda: dict(cast("Mapping[str, JSONValue]", definition.canonical(view))), + ) _, refused = definition.judge(simplified) if len(refused) != 0: msg = ( @@ -1760,6 +1809,7 @@ def _replaced(value: JSONValue, path: Loc, new: JSONValue) -> JSONValue: "named_configuration", "no_pipelines", "no_rules", + "read_only", "resolve", "ruled", "single_byte", @@ -1771,6 +1821,7 @@ def _replaced(value: JSONValue, path: Loc, new: JSONValue) -> JSONValue: "unknown_lengths", "unknown_storage", "variable_length", + "well_named", "with_problems", "written_name", ] diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/_pipeline.py b/packages/zarr-metadata/src/zarr_metadata/v3/_pipeline.py index 6a65295c01..e80a50d1a2 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/_pipeline.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/_pipeline.py @@ -38,6 +38,7 @@ Read, Resolved, asked, + read_only, ruled, ) @@ -185,7 +186,9 @@ def _chunk_problems( at: Loc, ) -> Problems: """What `definition`'s chunk rules find in a codec handed `chunk`, located under `at`.""" - return ruled(definition, lambda: definition.chunk_rules(configuration, nested, chunk), at) + return ruled( + definition, lambda: definition.chunk_rules(read_only(configuration), nested, chunk), at + ) def _inner_pipelines( @@ -205,7 +208,7 @@ def _inner_pipelines( given = asked( definition, "pipelines", - lambda: cast("object", definition.pipelines(configuration, nested, chunk)), + lambda: cast("object", definition.pipelines(read_only(configuration), nested, chunk)), at, ) if not _is_pipelines(given): @@ -268,7 +271,7 @@ def _handed_on( given = asked( definition, "transition", - lambda: cast("object", definition.transition(configuration, nested, chunk)), + lambda: cast("object", definition.transition(read_only(configuration), nested, chunk)), at, ) if not isinstance(given, Chunk): diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/codec/sharding_indexed.py b/packages/zarr-metadata/src/zarr_metadata/v3/codec/sharding_indexed.py index 8a56d50a16..7402857221 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/codec/sharding_indexed.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/codec/sharding_indexed.py @@ -74,7 +74,8 @@ class ShardingIndexedCodecObject(TypedDict, closed=True): The configuration has multiple required keys (`chunk_shape`, `codecs`, `index_codecs`), so only the object form is valid; the short-hand-name form is not permitted by the spec for this codec. - https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v3/codecs/sharding-indexed/index.rst#L141-L155 (required members) + https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v3/codecs/sharding-indexed/index.rst#L129-L138 (`chunk_shape`) + https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v3/codecs/sharding-indexed/index.rst#L141-L155 (`codecs` and `index_codecs`, the members the spec marks required) https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v3/core/index.rst#L1562-L1564 (short-hand names only "if no configuration metadata is required") """ diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/_byte.py b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/_byte.py index 0d4ed5c98e..43a2aaed66 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/_byte.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/_byte.py @@ -1,7 +1,8 @@ """A byte value, as the fill values of raw bits and of `bytes` hold them. Each is an integer in `[0, 255]` -(https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v3/data-types/index.rst#L59-L61). +(https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v3/data-types/index.rst#L97-L99; +the `bytes` type's is https://github.com/zarr-developers/zarr-extensions/blob/4da7b37a84f76e660902f6d3de3eaef0e0febae6/data-types/bytes/README.md?plain=1#L8). """ from typing import Annotated diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/_float.py b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/_float.py index f30b842699..bcdb24cb8e 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/_float.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/_float.py @@ -100,8 +100,15 @@ def float_bits(value: float | str, width: FloatWidth) -> int: A number is read as a float64, as a JSON parser reads one -- an integer of more digits than a float64 holds is rounded to one -- and rounded to the nearest value the type represents, ties to even, and to an infinity - past the largest, as numpy casts a float64. So an integer and a number - with a fraction that read as one float64 spell one value. + past the largest: as zarrs reads one, `as_f64` then `as f32` + (https://github.com/zarrs/zarrs/blob/8d68f8522b382d050b768f84bce64c2935de4523/zarrs_metadata/src/v3/array/fill_value.rs#L160), + as tensorstore does, `static_cast` of `get()` + (https://github.com/google/tensorstore/blob/692d2798c51a76d2eed0b4aee85cad5fd4be950a/tensorstore/driver/zarr3/metadata.cc#L165), + and as numpy casts a float64. So an integer and a number with a + fraction that read as one float64 spell one value; and an integer past + 2**53 whose float64 sits halfway between two values of the type -- + 2**60 + 2**36 + 1, for float32 -- is the even one, 2**60, as those + readers store it, not the value nearest the integer itself. """ code, fraction = _FORMATS[width] exponent = width - 1 - fraction diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/struct.py b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/struct.py index 0aa0f55257..a5ded5ad8a 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/struct.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/struct.py @@ -115,6 +115,7 @@ def _fill_value_rules( not hold, leaves its fill value unjudged. """ names = [member["name"] for member in configuration["fields"]] + declared = set(names) for index, name in enumerate(names): if name not in value: yield ValidationProblem( @@ -125,7 +126,7 @@ def _fill_value_rules( if field_type is not None: yield from fill_value_problems(field_type, value[name], (name,)) for key in value: - if key not in names: + if key not in declared: yield ValidationProblem((key,), f"no struct field is named {key!r}", "unknown_key") diff --git a/packages/zarr-metadata/tests/model/conftest.py b/packages/zarr-metadata/tests/model/conftest.py new file mode 100644 index 0000000000..ea7c08e22a --- /dev/null +++ b/packages/zarr-metadata/tests/model/conftest.py @@ -0,0 +1,22 @@ +"""Fixtures for the model tests.""" + +from __future__ import annotations + +import sys +from typing import TYPE_CHECKING + +import pytest + +if TYPE_CHECKING: + from collections.abc import Iterator + + +@pytest.fixture +def interpreter_writes_4300_digits() -> Iterator[None]: + """The interpreter's default limit on writing an integer, pinned: an environment may lift it (`PYTHONINTMAXSTRDIGITS=0`), and a test of what happens past it needs it.""" + limit = sys.get_int_max_str_digits() + sys.set_int_max_str_digits(4300) + try: + yield + finally: + sys.set_int_max_str_digits(limit) diff --git a/packages/zarr-metadata/tests/model/test_array.py b/packages/zarr-metadata/tests/model/test_array.py index 34fd225c69..25e3f38a5f 100644 --- a/packages/zarr-metadata/tests/model/test_array.py +++ b/packages/zarr-metadata/tests/model/test_array.py @@ -7,13 +7,13 @@ import pickle from collections import UserDict from collections.abc import Callable -from typing import TYPE_CHECKING, TypeGuard, get_args, get_origin, get_type_hints +from typing import TYPE_CHECKING, Any, TypeGuard, cast, get_args, get_origin, get_type_hints import pytest from typing_extensions import Unpack from tests.model._cases import Expect, ExpectFail, mutate_nested_containers -from zarr_metadata._json import arrays_to_tuples, prefixed +from zarr_metadata._json import JSON_DEPTH, arrays_to_tuples, json_text, prefixed from zarr_metadata.model import ( ARRAY_METADATA_OPTIONAL_KEYS_V3, ARRAY_METADATA_REQUIRED_KEYS_V3, @@ -43,7 +43,16 @@ ) from zarr_metadata.v3.array import ZarrV3ArrayMetadataJSONPartial from zarr_metadata.v3.codec.gzip import GZIP_CODEC -from zarr_metadata.v3.definition import CORE, CORE_AND_EXTENSIONS, Read, Unclaimed, configuration_of +from zarr_metadata.v3.definition import ( + CORE, + CORE_AND_EXTENSIONS, + Read, + Unclaimed, + canonical_of, + configuration_of, + fields_of, + with_problems, +) if TYPE_CHECKING: from zarr_metadata._common import JSONValue, ZarrV3NamedConfigJSON @@ -834,7 +843,7 @@ def test_from_key_value_missing_key_raises( shape=(10,), attributes={"a": 1}, dimension_names=("x",), - storage_transformers=({"name": "t"},), + storage_transformers=({"name": "acme.t"},), ext={"must_understand": False}, ), id="v3-full", @@ -951,16 +960,16 @@ def test_v3_parser_accepts_bare_string_data_type() -> None: assert model.to_json()["data_type"] == "int32" -@pytest.mark.parametrize("name", ["bytes", "ANY string", "urn:example:codec"]) -def test_metadata_field_accepts_any_string_name(name: str) -> None: - """The structural layer checks the name type, not syntax or registration.""" +@pytest.mark.parametrize("name", ["bytes", "acme.codec", "urn:example:codec"]) +def test_metadata_field_accepts_a_name_as_the_spec_names_one(name: str) -> None: + """The structural layer checks the name is one the spec gives an extension, not that anything registered it.""" assert validate_metadata_field_v3({"name": name}) == () @pytest.mark.parametrize("value", [0, 1, "false", None]) def test_metadata_field_must_understand_must_be_boolean(value: object) -> None: """must_understand is a JSON boolean, not a truthy scalar.""" - problems = validate_metadata_field_v3({"name": "x", "must_understand": value}) + problems = validate_metadata_field_v3({"name": "acme.x", "must_understand": value}) assert [(problem.loc, problem.kind) for problem in problems] == [ (("must_understand",), "invalid_type") ] @@ -968,7 +977,7 @@ def test_metadata_field_must_understand_must_be_boolean(value: object) -> None: def test_metadata_field_rejects_unknown_envelope_member() -> None: """Unknown envelope keys cannot be silently discarded during normalization.""" - problems = validate_metadata_field_v3({"name": "x", "typo": 1}) + problems = validate_metadata_field_v3({"name": "acme.x", "typo": 1}) assert [(problem.loc, problem.kind) for problem in problems] == [(("typo",), "unknown_key")] @@ -1219,7 +1228,7 @@ def test_validate_json_reports_json_in_message() -> None: METADATA_FIELD_VALIDATE_CASES: list[Expect[object, frozenset[tuple[str | int, ...]]]] = [ Expect("bytes", frozenset(), id="bare-string"), - Expect({"name": "x", "configuration": {"a": 1}}, frozenset(), id="named-config"), + Expect({"name": "acme.x", "configuration": {"a": 1}}, frozenset(), id="named-config"), Expect({"name": "bytes"}, frozenset(), id="name-only"), Expect(5, frozenset({()}), id="not-str-or-mapping"), Expect({"configuration": {}}, frozenset({("name",)}), id="missing-name"), @@ -1584,7 +1593,7 @@ def test_v2_dtype_must_be_string_or_records() -> None: ( validate_array_metadata_v2, {**ZarrV2ArrayMetadata.create_default().to_json(), "dtype": (("f0", b""),)}, - ("dtype",), + ("dtype", 0, 1), ), ( validate_array_metadata_v2, @@ -1863,25 +1872,52 @@ def test_a_codec_that_is_not_json_is_placed_by_the_definition_that_claims_its_na ] -def test_a_shard_nested_hundreds_deep_is_read() -> None: - """A field it holds is read one frame deeper than it, as deep as the interpreter goes.""" +def test_a_shard_nested_as_deep_as_a_reader_walks_is_read_and_written() -> None: + """Every walker of fields takes more than one frame per shard, so the deepest nesting the cap admits is where the interpreter's limit would show; one shard deeper is the depth problem.""" little = {"name": "bytes", "configuration": {"endian": "little"}} - codecs: list[object] = [little] - for _ in range(200): - codecs = [ - { - "name": "sharding_indexed", - "configuration": {"chunk_shape": [1], "codecs": codecs, "index_codecs": [little]}, - } - ] + + def nested(shards: int) -> list[object]: + codecs: list[object] = [little] + for _ in range(shards): + codecs = [ + { + "name": "sharding_indexed", + "configuration": { + "chunk_shape": [1], + "codecs": codecs, + "index_codecs": [little], + }, + } + ] + return codecs + + # Each shard is three levels -- its object, its configuration and the + # `codecs` in it -- and the innermost codec's configuration is the + # last container a reader walks. + deepest = (JSON_DEPTH - 4) // 3 + codecs = nested(deepest) document = {**ZarrV3ArrayMetadata.create_default(shape=(2,)).to_json(), "codecs": codecs} assert validate_array_metadata_v3(document) == () + model = ZarrV3ArrayMetadata.from_json(document) + assert json_text(model.to_json()) == json_text(cast("JSONValue", document)) + assert ZarrV3ArrayMetadata.from_key_value(model.to_key_value()) == model + assert hash(model) == hash(ZarrV3ArrayMetadata.from_json(document)) + assert pickle.loads(pickle.dumps(model)) == model + assert copy.deepcopy(model) == model + (shard,) = model.codecs + assert json_text(canonical_of(shard, ())) == json_text(cast("JSONValue", codecs[0])) + # A shard and its index codec at each level, and the innermost codec. + assert len(list(with_problems(fields_of(shard), ()))) == 2 * deepest + 1 + problems = validate_array_metadata_v3({**document, "codecs": nested(deepest + 1)}) + assert {(problem.kind, len(problem.loc), problem.message) for problem in problems} == { + ("invalid_value", JSON_DEPTH, f"nested deeper than the {JSON_DEPTH} levels a reader walks") + } def test_a_model_is_written_as_deep_as_it_is_read() -> None: - """A fill value hundreds deep, of a data type nothing in scope claims, which takes any JSON.""" + """A fill value as deep as a reader walks, of a data type nothing in scope claims, which takes any JSON; pickled and deep-copied too, which take two frames a level.""" fill_value: dict[str, object] = {} - for _ in range(600): + for _ in range(JSON_DEPTH - 2): fill_value = {"x": fill_value} document = { **ZarrV3ArrayMetadata.create_default().to_json(), @@ -1891,6 +1927,70 @@ def test_a_model_is_written_as_deep_as_it_is_read() -> None: model = ZarrV3ArrayMetadata.from_json(document) assert model.to_json()["fill_value"] == fill_value assert ZarrV3ArrayMetadata.from_key_value(model.to_key_value()) == model + assert pickle.loads(pickle.dumps(model)) == model + assert copy.deepcopy(model) == model + + +def test_a_field_s_configuration_member_is_counted_from_the_field_s_root() -> None: + # `validate_metadata_field_v3` judges a field alone, at its own root: a + # member of its configuration sits two levels down, and the cap counts + # from the root, not from the member. + nested: dict[str, object] = {} + for _ in range(JSON_DEPTH - 2): + nested = {"x": nested} + problems = validate_metadata_field_v3({"name": "acme.x", "configuration": {"y": nested}}) + assert [(len(p.loc), p.kind) for p in problems] == [(JSON_DEPTH, "invalid_value")] + assert problems[0].loc[:2] == ("configuration", "y") + shallower = {"name": "acme.x", "configuration": {"y": nested["x"]}} + assert validate_metadata_field_v3(shallower) == () + + +def test_a_v2_dtype_of_nested_records_is_read_to_the_levels_a_reader_walks() -> None: + """Field records nest a dtype two levels a record: the shape check recursed a record at a time with no cap, and a thousand records overflowed.""" + + def records(levels: int) -> object: + dtype: object = " None: + """The v2 models copied with `copy.deepcopy`, two frames a level, and overflowed on documents their validators accept.""" + fill_value: dict[str, object] = {} + for _ in range(JSON_DEPTH - 2): + fill_value = {"x": fill_value} + document = { + **ZarrV2ArrayMetadata.create_default(shape=(2,)).to_json(), + "fill_value": fill_value, + } + assert validate_array_metadata_v2(document) == () + model = ZarrV2ArrayMetadata.from_json(document) + assert json_text(model.to_json()) == json_text(cast("JSONValue", document)) + assert ZarrV2ArrayMetadata.from_key_value(model.to_key_value()) == model + assert hash(model) == hash(ZarrV2ArrayMetadata.from_json(document)) + assert pickle.loads(pickle.dumps(model)) == model + assert copy.deepcopy(model) == model + problems = validate_array_metadata_v2({**document, "fill_value": {"x": fill_value}}) + assert [(problem.kind, len(problem.loc)) for problem in problems] == [ + ("invalid_value", JSON_DEPTH) + ] def test_v3_node_type_literal_enforced() -> None: @@ -2271,3 +2371,123 @@ def test_error_a_member_outside_the_values_it_takes( # type -- a string where a number belongs -- else of the wrong value. written = {**document.create_default().to_json(), member: value} assert [(p.loc, p.kind) for p in validate(written)] == [((member,), kind)] + + +def test_error_a_document_nested_deeper_than_a_reader_walks_is_a_problem() -> None: + # Not a `RecursionError`: a 10 KB document a stranger wrote, nested past + # the interpreter's limit, is refused at the level past the last a + # reader walks, by every validator, guard and reader. + deep: dict[str, object] = {} + for _ in range(2_000): + deep = {"a": deep} + document = {**ZarrV3ArrayMetadata.create_default().to_json(), "attributes": deep} + problems = validate_array_metadata_v3(document) + assert [(len(p.loc), p.kind) for p in problems] == [(JSON_DEPTH, "invalid_value")] + assert problems[0].loc[:2] == ("attributes", "a") + assert not is_json(document) + assert not is_array_metadata_v3(document) + assert not is_group_metadata_v3(document) + assert not is_array_metadata_v2(document) + assert not is_group_metadata_v2(document) + with pytest.raises(MetadataValidationError): + ZarrV3ArrayMetadata.from_json(document) + + +def test_a_field_built_by_hand_and_given_to_the_constructor_is_taken_as_read() -> None: + """As the class says: a field built by hand is taken as read, what it holds unjudged, so a model can write a document its validator refuses. `Read` judging its own configuration is the follow-up #379 named.""" + trusted = Read( + json="gzip", name="gzip", definition=GZIP_CODEC, configuration={"level": 99, "window": 1} + ) + default = ZarrV3ArrayMetadata.create_default(shape=(2,)) + model = dataclasses.replace(default, codecs=(*default.codecs, trusted)) + assert model.codecs[-1] is trusted + assert [ + (problem.loc, problem.kind) for problem in validate_array_metadata_v3(model.to_json()) + ] == [ + (("codecs", 1, "configuration", "window"), "unknown_key"), + (("codecs", 1, "configuration", "level"), "invalid_value"), + ] + + +def test_error_a_field_object_in_a_document_is_not_json() -> None: + # A `Read` built by hand, with a configuration its definition refuses, + # smuggled into a document: refused as what it is, so nothing built by + # hand passes as read. A model holds its own fields as read. + smuggled = Read( + json="gzip", name="gzip", definition=GZIP_CODEC, configuration={"level": 99, "window": 1} + ) + document = { + **ZarrV3ArrayMetadata.create_default(shape=(2,)).to_json(), + "codecs": [{"name": "bytes", "configuration": {"endian": "little"}}, smuggled], + } + assert [(p.loc, p.kind) for p in validate_array_metadata_v3(document)] == [ + (("codecs", 1), "invalid_type") + ] + assert not is_array_metadata_v3(document) + with pytest.raises(MetadataValidationError): + ZarrV3ArrayMetadata.from_json(document) + # Nor does `update` take one among the members it is given: only the + # fields the model holds are taken as read. + model = ZarrV3ArrayMetadata.create_default(shape=(2,)) + with pytest.raises(MetadataValidationError) as raised: + model.update(context=CORE, codecs=cast("Any", (model.codecs[0].to_json(), smuggled))) + assert [(p.loc, p.kind) for p in raised.value.problems] == [(("codecs", 1), "invalid_type")] + assert model.update(context=CORE, codecs=(cast("Any", model.codecs[0]),)) == model + + +def test_a_problem_shows_an_integer_too_long_to_write_by_its_size( + interpreter_writes_4300_digits: None, +) -> None: + # `int` refuses to write more than 4,300 digits; a message says the + # size instead of raising. + document = {**ZarrV3ArrayMetadata.create_default().to_json(), "zarr_format": 10**5000} + (problem,) = validate_array_metadata_v3(document) + assert problem.message == f"expected 3, got an integer of {(10**5000).bit_length()} bits" + (problem,) = validate_metadata_field_v3({"name": "gzip", 10**5000: 1}) + assert problem.message == ( + f"non-string metadata field key an integer of {(10**5000).bit_length()} bits" + ) + # Held by a container, JSON or not, or by a key, it is what the + # interpreter will not write. + for held in ([10**5000], [10**5000, object()]): + document = {**ZarrV3ArrayMetadata.create_default().to_json(), "zarr_format": held} + (problem,) = validate_array_metadata_v3(document) + assert ( + problem.message == "expected 3, got a value of type list the interpreter will not write" + ) + (problem,) = validate_metadata_field_v3({"name": "gzip", (10**5000,): 1}) + assert problem.message == ( + "non-string metadata field key a value of type tuple the interpreter will not write" + ) + # So is what `refine_json` reports as no JSON at all: a set holding + # one, or frozensets nested past what the interpreter's repr walks. + document = {**ZarrV3ArrayMetadata.create_default().to_json(), "attributes": {"a": {10**5000}}} + (problem,) = validate_array_metadata_v3(document) + assert (problem.loc, problem.message) == ( + ("attributes", "a"), + "not a JSON-serializable value: a value of type set the interpreter will not write", + ) + frozen: frozenset[object] = frozenset() + for _ in range(100_000): + frozen = frozenset([frozen]) + document = {**ZarrV3ArrayMetadata.create_default().to_json(), "attributes": {"a": frozen}} + (problem,) = validate_array_metadata_v3(document) + assert problem.message == "not a JSON-serializable value: a value nested too deep to show" + + +def test_a_problem_shows_a_value_nested_too_deep_to_write_by_saying_so() -> None: + # `_refine` stops at `JSON_DEPTH`; the repr that would show a value it + # refused walks as deep as the value nests, and overflows. + deep: list[object] = [] + innermost = deep + for _ in range(100_000): + nested: list[object] = [] + innermost.append(nested) + innermost = nested + document = {**ZarrV3ArrayMetadata.create_default().to_json(), "zarr_format": deep} + # One problem: a value the literal check refuses is not walked, so the + # depth rule does not judge it too. + assert [ + (problem.loc, problem.kind, problem.message) + for problem in validate_array_metadata_v3(document) + ] == [(("zarr_format",), "invalid_type", "expected 3, got a value nested too deep to show")] diff --git a/packages/zarr-metadata/tests/model/test_construction.py b/packages/zarr-metadata/tests/model/test_construction.py index 29a29402ec..c64450c0cc 100644 --- a/packages/zarr-metadata/tests/model/test_construction.py +++ b/packages/zarr-metadata/tests/model/test_construction.py @@ -26,9 +26,9 @@ ZarrV3ConsolidatedMetadata, ZarrV3GroupMetadata, ) -from zarr_metadata.model._validation import construct +from zarr_metadata.model._validation import ZarrV3ArrayMetadataReading, construct from zarr_metadata.v3.data_type.int8 import INT8_DATA_TYPE -from zarr_metadata.v3.definition import CORE_AND_EXTENSIONS +from zarr_metadata.v3.definition import CORE_AND_EXTENSIONS, Chunk if TYPE_CHECKING: from collections.abc import Iterator @@ -261,3 +261,9 @@ def test_error_extra_fields_are_a_mapping(model: object) -> None: def test_error_consolidated_metadata_paths_are_strings() -> None: with pytest.raises(TypeError, match="a document's path is a string, got 1"): ZarrV3ConsolidatedMetadata(metadata=cast("Any", {1: ARRAY})) + + +def test_construct_fills_a_member_from_its_default_factory() -> None: + reading = construct(ZarrV3ArrayMetadataReading, problems=()) + assert reading.chunk == Chunk() + assert reading.pipeline == () diff --git a/packages/zarr-metadata/tests/model/test_group.py b/packages/zarr-metadata/tests/model/test_group.py index 3542a3c24e..055eba86f7 100644 --- a/packages/zarr-metadata/tests/model/test_group.py +++ b/packages/zarr-metadata/tests/model/test_group.py @@ -3,18 +3,21 @@ import copy import dataclasses import json +import pickle from collections import UserDict from collections.abc import Callable, Iterator -from typing import cast +from typing import Any, cast import pytest from tests.model._cases import mutate_nested_containers from zarr_metadata._common import JSONValue, ZarrV3NamedConfigJSON from zarr_metadata._json import ( + JSON_DEPTH, MetadataValidationError, ValidationProblem, arrays_to_tuples, + json_text, ) from zarr_metadata.model import UNSET from zarr_metadata.model._array import ZarrV3ArrayMetadata, ZarrV3ArrayMetadataUpdate @@ -284,6 +287,211 @@ def test_group_v3_update() -> None: # --- ZarrV2GroupMetadata --------------------------------------------------- +def test_a_v2_group_nested_as_deep_as_a_reader_walks_is_read_and_written() -> None: + """The v2 group copied with `copy.deepcopy`, two frames a level, and overflowed on documents its validator accepts.""" + attributes: dict[str, object] = {} + for _ in range(JSON_DEPTH - 2): + attributes = {"x": attributes} + document = {"zarr_format": 2, "attributes": attributes} + assert validate_group_metadata_v2(document) == () + model = ZarrV2GroupMetadata.from_json(document) + assert json_text(model.to_json()) == json_text(cast("JSONValue", document)) + assert ZarrV2GroupMetadata.from_key_value(model.to_key_value()) == model + assert pickle.loads(pickle.dumps(model)) == model + assert copy.deepcopy(model) == model + problems = validate_group_metadata_v2({**document, "attributes": {"x": attributes}}) + assert [(problem.kind, len(problem.loc)) for problem in problems] == [ + ("invalid_value", JSON_DEPTH) + ] + + +def _chain_of_groups(documents: int) -> dict[str, object]: + """A group holding, in its consolidated metadata, a group holding a group..., `documents` of them below the root, each listing only the next: a document sits three levels below the one holding it.""" + document: dict[str, object] = {"zarr_format": 3, "node_type": "group"} + for _ in range(documents): + document = { + "zarr_format": 3, + "node_type": "group", + "consolidated_metadata": { + "kind": "inline", + "must_understand": False, + "metadata": {"a": document}, + }, + } + return document + + +def test_a_chain_of_consolidated_groups_is_bounded_by_the_levels_a_reader_walks() -> None: + """Each document is read from where it sits in the one handed in, so a chain is bounded as any nesting is: the reader took three frames a document, unbounded, and a 20 KB chain overflowed.""" + depth = f"nested deeper than the {JSON_DEPTH} levels a reader walks" + deepest = (JSON_DEPTH - 1) // 3 + problems = validate_group_metadata_v3(_chain_of_groups(deepest)) + assert all(len(problem.loc) < JSON_DEPTH for problem in problems) + assert {problem.message for problem in problems} == { + 'expected a node the group lists, got "/a/a", which "/a" lists alone' + } + for documents in (deepest + 1, 350): + problems = validate_group_metadata_v3(_chain_of_groups(documents)) + assert [ + (len(problem.loc), problem.message) + for problem in problems + if len(problem.loc) >= JSON_DEPTH + ] == [(JSON_DEPTH, depth)] + with pytest.raises(MetadataValidationError): + ZarrV3GroupMetadata.from_json(_chain_of_groups(documents)) + + +def test_a_consolidated_document_is_read_from_where_it_sits() -> None: + """Its levels are counted from the root of the document handed in, so what `read_node_metadata_v3` admits alone can sit too deep inside; its own reading locates the problem in it.""" + depth = f"nested deeper than the {JSON_DEPTH} levels a reader walks" + + def nested(levels: int) -> dict[str, object]: + value: dict[str, object] = {} + for _ in range(levels): + value = {"x": value} + return value + + # A document sits three levels down, its attribute, or a member of an + # extra field, two more. + group = { + "zarr_format": 3, + "node_type": "group", + "attributes": {"a": nested(JSON_DEPTH - 4)}, + "acme_extra": {"must_understand": False, "a": nested(JSON_DEPTH - 4)}, + } + array = { + **ZarrV3ArrayMetadata.create_default(shape=(2,)).to_json(), + "data_type": "acme.deep", + "fill_value": {"a": nested(JSON_DEPTH - 4)}, + "codecs": [ + {"name": "bytes", "configuration": {"endian": "little"}}, + {"name": "acme.x", "configuration": {"y": nested(JSON_DEPTH - 6)}}, + ], + } + assert validate_node_metadata_v3(group) == () + assert validate_node_metadata_v3(array) == () + document = { + "zarr_format": 3, + "node_type": "group", + "consolidated_metadata": { + "kind": "inline", + "must_understand": False, + "metadata": {"g": group, "h": array}, + }, + } + reading = read_group_metadata_v3(document) + assert [(len(problem.loc), problem.message) for problem in reading.problems] == [ + (JSON_DEPTH, depth) + ] * 4 + assert [ + (problem.loc[:2], len(problem.loc)) for problem in reading.consolidated["g"].problems + ] == [(("acme_extra", "a"), JSON_DEPTH - 3), (("attributes", "a"), JSON_DEPTH - 3)] + assert [ + (problem.loc[:2], len(problem.loc)) for problem in reading.consolidated["h"].problems + ] == [ + (("fill_value", "a"), JSON_DEPTH - 3), + (("codecs", 1), JSON_DEPTH - 3), + ] + + +def test_a_document_a_level_before_the_cap_holds_its_scalars_and_nothing_else() -> None: + """Each container member sits past the cap, and is judged where it sits before it is walked, as `refine_json` of the whole would judge it; the scalars are read.""" + depth = f"nested deeper than the {JSON_DEPTH} levels a reader walks" + + def a_level_before_the_cap(document: dict[str, object]) -> dict[str, object]: + for _ in range((JSON_DEPTH - 1) // 3): + document = { + "zarr_format": 3, + "node_type": "group", + "consolidated_metadata": { + "kind": "inline", + "must_understand": False, + "metadata": {"a": document}, + }, + } + return document + + scalars: dict[str, object] = {"zarr_format": 3, "node_type": "group"} + problems = validate_group_metadata_v3(a_level_before_the_cap(scalars)) + assert all(len(problem.loc) < JSON_DEPTH for problem in problems) + array = ZarrV3ArrayMetadata.create_default(shape=(2,)).to_json() + for document in ({**scalars, "attributes": {"a": 1}}, dict(array)): + containers = sorted( + key for key, item in document.items() if isinstance(item, (dict, list, tuple)) + ) + assert len(containers) != 0 + problems = validate_group_metadata_v3(a_level_before_the_cap(document)) + assert sorted( + (problem.loc[-1], len(problem.loc), problem.message) + for problem in problems + if len(problem.loc) >= JSON_DEPTH + ) == [(key, JSON_DEPTH, depth) for key in containers] + + +def test_a_listing_key_too_long_to_write_is_shown_by_its_size( + interpreter_writes_4300_digits: None, +) -> None: + document = { + "zarr_format": 3, + "node_type": "group", + "consolidated_metadata": { + "kind": "inline", + "must_understand": False, + "metadata": {10**5000: 1}, + }, + } + (problem,) = validate_group_metadata_v3(document) + assert problem.message == f"non-string key an integer of {(10**5000).bit_length()} bits" + + +def test_a_model_built_of_models_trusts_them_as_it_trusts_its_fields() -> None: + """Each held model checked itself at its own root, so a child valid alone can sit too deep in a group built by hand, whose document its validator refuses at the cap, as one holding a hand-built `Read` can; a reader builds only within the cap, the consolidated member's from where it sits.""" + # The innermost object sits at the last level a reader walks, alone; + # three deeper as a document a group holds. + nested: dict[str, object] = {} + for _ in range(JSON_DEPTH - 3): + nested = {"x": nested} + child = ZarrV3GroupMetadata.from_json( + {"zarr_format": 3, "node_type": "group", "attributes": {"a": nested}} + ) + group = dataclasses.replace( + ZarrV3GroupMetadata.create_default(), + consolidated_metadata=ZarrV3ConsolidatedMetadata(metadata={"a": child}), + ) + depth = f"nested deeper than the {JSON_DEPTH} levels a reader walks" + written = group.to_json() + assert [(len(p.loc), p.message) for p in validate_group_metadata_v3(written)] == [ + (JSON_DEPTH, depth) + ] + with pytest.raises(MetadataValidationError): + ZarrV3GroupMetadata.from_json(written) + with pytest.raises(MetadataValidationError): + ZarrV3ConsolidatedMetadata.from_json(written["consolidated_metadata"]) + + +def test_a_v2_consolidated_document_is_read_to_the_levels_a_reader_walks() -> None: + """An entry past the cap is the depth problem, not `RecursionError`: the document was normalized whole, a frame a level without bound, before any entry was refined.""" + + def nested(levels: int) -> dict[str, object]: + value: dict[str, object] = {} + for _ in range(levels): + value = {"x": value} + return value + + # An entry sits two levels down; its innermost object at the last + # level a reader walks. + document = {"zarr_consolidated_format": 1, "metadata": {"a/.zattrs": nested(JSON_DEPTH - 3)}} + model = ZarrV2ConsolidatedMetadata.from_json(document) + assert json_text(model.to_json()) == json_text(cast("JSONValue", document)) + for levels in (JSON_DEPTH - 2, 2000): + deeper = {**document, "metadata": {"a/.zattrs": nested(levels)}} + with pytest.raises(MetadataValidationError) as raised: + ZarrV2ConsolidatedMetadata.from_json(deeper) + assert [(len(p.loc), p.kind) for p in raised.value.problems] == [ + (JSON_DEPTH, "invalid_value") + ] + + def test_group_v2_key_value_split() -> None: """v2 to_key_value writes .zgroup and .zattrs; from_key_value merges them.""" model = ZarrV2GroupMetadata.create_default(attributes={"a": 1}) @@ -391,10 +599,10 @@ def test_consolidated_v3_roundtrip() -> None: assert model.to_json() == doc -def test_consolidated_v3_must_understand_true_rejected() -> None: - """ZarrV3ConsolidatedMetadata enforces must_understand=False at runtime.""" - with pytest.raises(ValueError, match="must_understand"): - ZarrV3ConsolidatedMetadata(must_understand=True, metadata={}) +def test_error_consolidated_v3_must_understand_is_not_a_member_to_set() -> None: + """`must_understand` is `False` by declaration, as `kind` is `"inline"`: not a member the constructor takes.""" + with pytest.raises(TypeError, match="must_understand"): + cast("Any", ZarrV3ConsolidatedMetadata)(must_understand=True, metadata={}) def test_consolidated_v3_from_json_must_understand_true_rejected() -> None: @@ -494,8 +702,6 @@ def _fields_of_an_array(*at: str | int) -> list[tuple[str | int, ...]]: ("document", "paths", "locs"), [ (_group(), [], []), - # A null, which a historical zarr-python bug wrote, holds nothing. - (_group(consolidated_metadata=None), [], []), ( _group(consolidated_metadata=_inline(a=_array(), g=_group())), ["a", "g"], @@ -529,7 +735,7 @@ def _fields_of_an_array(*at: str | int) -> list[tuple[str | int, ...]]: ], ), ], - ids=["no-consolidated-metadata", "null", "an-array-and-a-group", "paths", "nested"], + ids=["no-consolidated-metadata", "an-array-and-a-group", "paths", "nested"], ) def test_a_group_reads_each_document_its_consolidated_metadata_holds( document: dict[str, object], paths: list[str], locs: list[tuple[str | int, ...]] @@ -1136,17 +1342,16 @@ def test_group_must_understand_fields_partition() -> None: assert set(model.must_understand_fields) == {"implicit"} -def test_group_v3_null_consolidated_metadata_repaired_to_absence() -> None: - """consolidated_metadata: null was written by a historical zarr-python bug. - Those stores must remain readable, but the bug spelling is not honored: - it is read as absence (UNSET) and never written back — the round-trip - deliberately repairs the document rather than preserving the bug.""" +def test_error_a_null_consolidated_metadata_is_a_value_the_document_wrote() -> None: + """A zarr-python 3.0.x bug wrote `consolidated_metadata: null`; the spec says an object, and the package models nothing else as right: a reader of those stores strips the key first.""" null_doc = {"zarr_format": 3, "node_type": "group", "consolidated_metadata": None} - assert validate_group_metadata_v3(null_doc) == () - model = ZarrV3GroupMetadata.from_json(null_doc) - assert model.consolidated_metadata is UNSET - assert "consolidated_metadata" not in model.to_json() - assert model == ZarrV3GroupMetadata.from_json({"zarr_format": 3, "node_type": "group"}) + assert [(p.loc, p.kind) for p in validate_group_metadata_v3(null_doc)] == [ + (("consolidated_metadata",), "invalid_type") + ] + with pytest.raises(MetadataValidationError): + ZarrV3GroupMetadata.from_json(null_doc) + del null_doc["consolidated_metadata"] + assert ZarrV3GroupMetadata.from_json(null_doc).consolidated_metadata is UNSET # --- to_json shares no mutable state with the model ------------------------ @@ -1210,3 +1415,10 @@ def test_from_json_shares_no_mutable_state_with_its_input( baseline = copy.deepcopy(read.to_json()) mutate_nested_containers(document) assert read.to_json() == baseline + + +def test_a_group_writes_no_empty_attributes() -> None: + model = ZarrV3GroupMetadata.create_default() + assert model.attributes == {} + assert "attributes" not in model.to_json() + assert json.loads(model.to_key_value()["zarr.json"]) == {"zarr_format": 3, "node_type": "group"} diff --git a/packages/zarr-metadata/tests/model/test_pydantic_module.py b/packages/zarr-metadata/tests/model/test_pydantic_module.py index 70a3abc8b4..11da93cb1c 100644 --- a/packages/zarr-metadata/tests/model/test_pydantic_module.py +++ b/packages/zarr-metadata/tests/model/test_pydantic_module.py @@ -9,6 +9,7 @@ import math import warnings from collections.abc import Mapping +from typing import Any import pytest from jsonschema import Draft202012Validator @@ -17,6 +18,7 @@ import zarr_metadata.pydantic as zmp from zarr_metadata._common import JSONValue from zarr_metadata.model import ( + ValidationProblem, ZarrV2ArrayMetadata, ZarrV2ConsolidatedMetadata, ZarrV2GroupMetadata, @@ -24,6 +26,7 @@ ZarrV3ConsolidatedMetadata, ZarrV3GroupMetadata, ) +from zarr_metadata.v3.codec.gzip import GZIP_CODEC from zarr_metadata.v3.definition import CORE, Read V3_ARRAY_DOC = dict(ZarrV3ArrayMetadata.create_default(shape=(4,)).to_json()) @@ -95,8 +98,16 @@ class Manifest(BaseModel): doc = dict(V3_ARRAY_DOC) del doc["chunk_key_encoding"] - with pytest.raises(ValidationError, match="chunk_key_encoding: missing required key"): + with pytest.raises(ValidationError) as raised: Manifest.model_validate({"metadata": doc}) + (error,) = raised.value.errors() + assert (error["type"], error["loc"], error["msg"]) == ( + "missing_key", + ("metadata", "chunk_key_encoding"), + "missing required key", + ) + # A missing key's input is the object missing it, as pydantic's own is. + assert error["input"] == doc def test_json_schema_generation() -> None: @@ -313,3 +324,114 @@ def test_error_a_validation_context_holds_a_scope_that_is_not_one() -> None: TypeAdapter(zmp.ZarrV3ArrayMetadata).validate_python( _WITH_ZSTD, context={zmp.CONTEXT_KEY: "CORE"} ) + + +def test_each_problem_is_a_line_error_of_pydantic_s() -> None: + # As pydantic reports its own: one per problem, its type the problem's + # kind, at the problem's loc under the field's, with the input found + # there and what was expected; a message holding JSON is kept as it is. + class Manifest(BaseModel): + metadata: zmp.ZarrV3ArrayMetadata + + doc = { + **V3_ARRAY_DOC, + "fill_value": {"a": 1}, + "codecs": [ + {"name": "bytes", "configuration": {"endian": "little"}}, + {"name": "gzip", "configuration": {"level": 12}}, + ], + } + with pytest.raises(ValidationError) as raised: + Manifest.model_validate({"metadata": doc}) + errors = raised.value.errors() + assert [(e["type"], e["loc"], e["input"]) for e in errors] == [ + ("invalid_type", ("metadata", "fill_value"), {"a": 1}), + ("invalid_value", ("metadata", "codecs", 1, "configuration", "level"), 12), + ] + assert errors[0]["msg"] == 'expected an integer, got {"a": 1}' + assert errors[1].get("ctx") == {"ge": 0, "le": 9} + + +def test_a_message_holding_a_ctx_placeholder_is_reported_as_it_is() -> None: + # pydantic renders a message as a template of its ctx, with no escape: + # a document value written `{expected}` would be shown as the ctx's + # `expected`. Such a message rides whole in the ctx instead. + doc = {**V3_ARRAY_DOC, "zarr_format": "{expected}"} + with pytest.raises(ValidationError) as raised: + TypeAdapter(zmp.ZarrV3ArrayMetadata).validate_python(doc) + (error,) = raised.value.errors() + assert (error["type"], error["loc"], error["msg"]) == ( + "invalid_type", + ("zarr_format",), + 'expected 3, got "{expected}"', + ) + assert error.get("ctx") == {"expected": (3,), "message": 'expected 3, got "{expected}"'} + + +def test_a_ctx_member_named_message_yields_to_a_message_holding_a_placeholder() -> None: + # The carrier goes last, so what it carries is scanned for no key + # after it; a member of its name is kept when nothing rides. + ctx = {"message": "theirs", "expected": 3} + problem = ValidationProblem(("a",), 'got "{expected}"', "invalid_value", ctx=ctx) + error = ValidationError.from_exception_data("T", [zmp._line_error(problem, {"a": 1})]).errors()[ + 0 + ] + assert error["msg"] == 'got "{expected}"' + assert error.get("ctx") == {"expected": 3, "message": 'got "{expected}"'} + assert list(error.get("ctx", {})) == ["expected", "message"] + plain = ValidationProblem(("a",), "got 1", "invalid_value", ctx=ctx) + error = ValidationError.from_exception_data("T", [zmp._line_error(plain, {"a": 1})]).errors()[0] + assert (error["msg"], error.get("ctx")) == ("got 1", ctx) + + +def test_a_line_error_for_what_a_problem_could_not_hold_reports_what_sits_there() -> None: + # A problem holds no input for what is not JSON a reader walks -- a + # `Read` built by hand among the codecs -- and the line error reports + # that object, where a missing key's reports the object missing it. + smuggled = Read(json="gzip", name="gzip", definition=GZIP_CODEC, configuration={"level": 1}) + doc = { + **V3_ARRAY_DOC, + "codecs": [{"name": "bytes", "configuration": {"endian": "little"}}, smuggled], + } + with pytest.raises(ValidationError) as raised: + TypeAdapter(zmp.ZarrV3ArrayMetadata).validate_python(doc) + (error,) = raised.value.errors() + assert (error["type"], error["loc"]) == ("invalid_type", ("codecs", 1)) + assert error["input"] is smuggled + + +def test_the_pydantic_schema_refuses_a_null_consolidated_metadata_as_the_reader_does() -> None: + # The package publishes one verdict on `null` there: the reader's. + schema = TypeAdapter(zmp.ZarrV3GroupMetadata).json_schema() + document = {"zarr_format": 3, "node_type": "group", "consolidated_metadata": None} + assert not Draft202012Validator(schema).is_valid(document) + with pytest.raises(ValidationError): + TypeAdapter(zmp.ZarrV3GroupMetadata).validate_python(document) + + +def test_a_line_error_for_a_document_past_the_cap_renders() -> None: + # The depth problem holds no input; the line error reports the subtree + # at its loc, which pydantic renders itself, truncated in `str` and + # whole in `json`, as it renders any input. + deep: dict[str, object] = {} + for _ in range(2_000): + deep = {"a": deep} + with pytest.raises(ValidationError) as raised: + TypeAdapter(zmp.ZarrV3ArrayMetadata).validate_python({**V3_ARRAY_DOC, "attributes": deep}) + (error,) = raised.value.errors() + assert (error["type"], len(error["loc"])) == ("invalid_value", 256) + assert "nested deeper" in str(raised.value) + assert json.loads(raised.value.json())[0]["type"] == "invalid_value" + + +@pytest.mark.parametrize("field", ["Foo/bar", "", "r*", {"name": "Int8"}]) +def test_the_pydantic_schema_names_an_extension_as_the_reader_does(field: object) -> None: + # One verdict on a name, the reader's, in the schema pydantic generates too. + schema = TypeAdapter(zmp.ZarrV3ArrayMetadata).json_schema() + # As JSON has it, arrays as lists: `jsonschema` takes no tuple for one. + document: dict[str, Any] = json.loads(json.dumps(V3_ARRAY_DOC)) + assert Draft202012Validator(schema).is_valid(document) + named: dict[str, Any] = {**document, "data_type": field} + assert not Draft202012Validator(schema).is_valid(named) + with pytest.raises(ValidationError): + TypeAdapter(zmp.ZarrV3ArrayMetadata).validate_python(named) diff --git a/packages/zarr-metadata/tests/model/test_read_array_metadata.py b/packages/zarr-metadata/tests/model/test_read_array_metadata.py index c3433572b8..9beb44b35c 100644 --- a/packages/zarr-metadata/tests/model/test_read_array_metadata.py +++ b/packages/zarr-metadata/tests/model/test_read_array_metadata.py @@ -14,12 +14,12 @@ from zarr_metadata._json import arrays_to_tuples from zarr_metadata.model import ( + UNSET, ValidationProblem, ZarrV3ArrayMetadata, ZarrV3ArrayMetadataReading, read_array_metadata_v3, read_group_metadata_v3, - validate_array_metadata_v3, ) from zarr_metadata.v3.definition import ( CORE, @@ -225,11 +225,15 @@ def test_a_document_reads_as_each_field_where_it_sits_and_its_codecs_as_a_pipeli handed: list[Lengths | None], ) -> None: reading = read_array_metadata_v3(document, context=context) - assert reading.problems == validate_array_metadata_v3(document, context=context) # A model only of a document with no problem. assert (reading.metadata is None) is (reading.problems != ()) assert [(loc, field.read_as, type(field)) for loc, field in reading.fields()] == fields - assert reading.chunk.data_type is reading.data_type + # The chunk's vocabulary for what nothing says is None, the reading's + # for a key the document lacks is `UNSET`. + if reading.data_type is UNSET: + assert reading.chunk.data_type is None + else: + assert reading.chunk.data_type is reading.data_type assert reading.pipeline[0].incoming == reading.chunk assert [None if s.incoming is None else s.incoming.lengths for s in reading.pipeline] == handed @@ -369,11 +373,11 @@ def test_error_a_list_of_fields_that_is_not_a_list_reads_as_empty( @pytest.mark.parametrize("member", ["data_type", "chunk_grid", "chunk_key_encoding"]) -def test_error_a_document_without_a_field_reads_it_as_none(member: str) -> None: +def test_error_a_document_without_a_field_reads_it_as_unset(member: str) -> None: document = _document() del document[member] reading = read_array_metadata_v3(document) - assert getattr(reading, member) is None + assert getattr(reading, member) is UNSET assert (member,) not in [loc for loc, _ in reading.fields()] assert ValidationProblem((member,), "missing required key", "missing_key") in reading.problems diff --git a/packages/zarr-metadata/tests/model/test_refine_json.py b/packages/zarr-metadata/tests/model/test_refine_json.py index 7f2d514c96..35a54aaee0 100644 --- a/packages/zarr-metadata/tests/model/test_refine_json.py +++ b/packages/zarr-metadata/tests/model/test_refine_json.py @@ -8,7 +8,7 @@ import pytest -from zarr_metadata._json import refine_json, refine_user_data, shown +from zarr_metadata._json import JSON_DEPTH, refine_json, refine_user_data, shown if TYPE_CHECKING: from collections.abc import Callable @@ -93,16 +93,29 @@ def test_error_every_leaf_that_is_not_json_is_reported() -> None: def test_a_value_nested_hundreds_deep_is_read() -> None: - # One frame per level of nesting: as deep as the interpreter goes, less - # what the test runner's own frames take. + # One frame per level of nesting, up to `JSON_DEPTH` of them. deep: dict[str, object] = {} - for _ in range(600): + for _ in range(JSON_DEPTH - 1): deep = {"k": deep} refined, problems = refine_json(deep) assert problems == () assert refined is not None +def test_error_a_value_nested_deeper_than_a_reader_walks() -> None: + # The level past the last is the problem, wherever it sits, so no + # document takes a reader past what the interpreter allows. + deep: dict[str, object] = {} + for _ in range(JSON_DEPTH + 40): + deep = {"k": deep} + refined, problems = refine_json({"a": [deep]}) + assert refined is None + assert [(len(p.loc), p.kind, p.message) for p in problems] == [ + (JSON_DEPTH, "invalid_value", f"nested deeper than the {JSON_DEPTH} levels a reader walks") + ] + assert problems[0].loc[:2] == ("a", 0) + + @pytest.mark.parametrize( ("value", "text"), [ @@ -120,3 +133,15 @@ def test_a_value_is_shown_as_the_json_a_document_writes(value: object, text: str # What a problem's message says a document holds: its JSON, and the # value's repr only when it is not JSON. assert shown(value) == text + + +def test_bytes_past_the_levels_a_reader_walks_are_not_json_rather_than_nested() -> None: + # `bytes` is a sequence to Python and no container to JSON, wherever + # it sits: past the cap it is still what it is, not nesting. + deep: list[object] = [b""] + for _ in range(JSON_DEPTH - 1): + deep = [deep] + _, problems = refine_json(deep) + assert [(len(p.loc), p.kind, p.message[:31]) for p in problems] == [ + (JSON_DEPTH, "invalid_type", "not a JSON-serializable value: ") + ] diff --git a/packages/zarr-metadata/tests/model/test_store_json.py b/packages/zarr-metadata/tests/model/test_store_json.py index fd50eaa236..3192254fb2 100644 --- a/packages/zarr-metadata/tests/model/test_store_json.py +++ b/packages/zarr-metadata/tests/model/test_store_json.py @@ -328,3 +328,11 @@ def test_error_a_document_the_reader_refuses_is_not_written( with pytest.raises(MetadataValidationError) as raised: build() assert [(problem.loc, problem.kind) for problem in raised.value.problems] == problems + + +def test_error_store_bytes_nested_deeper_than_python_reads_are_invalid_json() -> None: + # `json.loads` gives up on a hundred thousand `[` with a `RecursionError`, + # which is an ingestion failure like any other. + with pytest.raises(MetadataValidationError) as raised: + ZarrV3ArrayMetadata.from_key_value({"zarr.json": b"[" * 100_000}) + assert [(p.loc, p.kind) for p in raised.value.problems] == [(("zarr.json",), "invalid_json")] diff --git a/packages/zarr-metadata/tests/test_json_schema.py b/packages/zarr-metadata/tests/test_json_schema.py index d2baa4b225..1f1f136727 100644 --- a/packages/zarr-metadata/tests/test_json_schema.py +++ b/packages/zarr-metadata/tests/test_json_schema.py @@ -622,17 +622,19 @@ def _field_validator(kind: type[Definition[Any]], scope: int) -> Draft202012Vali False, ), ("raw-bits-of-a-size-the-spec-refuses", "r12", DataTypeDefinition, CORE, True, False), - ("raw-bits-notation", "r*", DataTypeDefinition, CORE, True, True), + # `r*` is the table's notation, and no name at all: `*` is no character + # of one. + ("raw-bits-notation", "r*", DataTypeDefinition, CORE, False, False), # Matched to the end of the name, as the package matches it, so a - # validator that matches as Python does takes no final newline for it: - # a name nothing claims, which any configuration goes with. + # validator that matches as Python does takes no final newline for it; + # nor is it a name at all, a newline being no character of one. ( "raw-bits-and-a-newline", {"name": "r16\n", "configuration": {"x": 1}}, DataTypeDefinition, CORE, - True, - True, + False, + False, ), ( "struct-field-out-of-bounds", @@ -641,7 +643,7 @@ def _field_validator(kind: type[Definition[Any]], scope: int) -> Draft202012Vali "configuration": { "fields": [ { - "name": "t", + "name": "acme.t", "data_type": { "name": "numpy.datetime64", "configuration": {"unit": "s", "scale_factor": 0}, @@ -759,8 +761,8 @@ def _consolidated(**metadata: object) -> dict[str, Any]: ( "a-name-raw-bits-and-a-newline", {**ARRAY, "data_type": "r16\n", "fill_value": "any"}, - True, - True, + False, + False, ), ("raw-bits", {**ARRAY, "data_type": "r16", "fill_value": [0, 255]}, True, True), ( @@ -799,7 +801,7 @@ def _consolidated(**metadata: object) -> dict[str, Any]: False, ), ("consolidated", _consolidated(a=ARRAY, b=GROUP), True, True), - ("consolidated-null", {**GROUP, "consolidated_metadata": None}, True, True), + ("consolidated-null", {**GROUP, "consolidated_metadata": None}, False, False), ("consolidated-bad-array", _consolidated(a={**ARRAY, "fill_value": 300}), False, False), ( "consolidated-within-consolidated", @@ -897,3 +899,99 @@ def test_every_schema_is_a_json_schema_written_the_same_each_time(scope: Context StorageTransformerDefinition, ): Draft202012Validator.check_schema(field_json_schema(kind, scope)) + + +@pytest.mark.parametrize( + ("inner", "outer", "expected"), + [ + (Gt(1), Gt(3), {"type": "integer", "exclusiveMinimum": 3}), + (Gt(3), Gt(1), {"type": "integer", "exclusiveMinimum": 3}), + (Lt(9), Lt(5), {"type": "integer", "exclusiveMaximum": 5}), + (Lt(5), Lt(9), {"type": "integer", "exclusiveMaximum": 5}), + (Le(5), Le(9), {"type": "integer", "maximum": 5}), + (Le(9), Le(5), {"type": "integer", "maximum": 5}), + (Ge(5), Ge(9), {"type": "integer", "minimum": 9}), + (Ge(9), Ge(5), {"type": "integer", "minimum": 9}), + ], + ids=[ + "gt-looser-inside", + "gt-stricter-inside", + "lt", + "lt-stricter-inside", + "le", + "le-stricter-inside", + "ge", + "ge-stricter-inside", + ], +) +def test_the_stricter_of_two_bounds_of_one_keyword_is_kept( + inner: object, outer: object, expected: dict[str, object] +) -> None: + # A bounded NewType bounded again: each keyword keeps the bound that + # admits less, whichever layer carries it. + # `NewType` takes no `Annotated` for pyright; it does at runtime. + Bounded = cast("Callable[[str, object], type]", NewType)("Bounded", _annotated[int, inner]) + assert json_schema(_holding(_annotated[Bounded, outer])) == _held(expected) + + +def test_a_stray_member_beside_an_unclaimed_name_is_refused_by_the_schema() -> None: + # As the reader refuses it: the envelope of a name nothing claims holds + # nothing but the name and its configuration. + field = {"name": "zfpy", "version": 2} + assert [(p.loc, p.kind) for p in resolve(field, CodecDefinition, CORE)[1]] == [ + (("version",), "unknown_key") + ] + assert not Draft202012Validator(field_json_schema(CodecDefinition, CORE)).is_valid(field) + + +@pytest.mark.parametrize( + "name", + [ + "zstd", + "numcodecs.adler32", + "vlen-utf8", + "acme_x", + "r16", + "https://example.com/c", + "urn:acme:c", + "urn:acme:%C3%BC", + "", + " ", + "Int8", + "9x", + "int8 ", + "int8\n", + "foo/bar", + "-int8", + "a", + "r*", + "x:", + "a: b", + "urn:acme:c\n", + "urn:acme:ü", + "urn:a\x1c", + "urn:a", + ], +) +def test_the_schema_names_an_extension_as_the_reader_does(name: str) -> None: + # The spec's regex, or a URI: what `well_named` accepts, the schema's + # `pattern` accepts, and what it refuses, the schema refuses. + field = {"name": name} + _, problems = resolve(field, CodecDefinition, CORE) + accepted = len(problems) == 0 + assert ( + Draft202012Validator(field_json_schema(CodecDefinition, CORE)).is_valid(field) is accepted + ) + assert accepted is ( + name + in ( + "zstd", + "numcodecs.adler32", + "vlen-utf8", + "acme_x", + "https://example.com/c", + "urn:acme:c", + "urn:acme:%C3%BC", + ) + or name == "r16" + ) diff --git a/packages/zarr-metadata/tests/test_problem_data.py b/packages/zarr-metadata/tests/test_problem_data.py index 9541756ed0..29e59ea7b5 100644 --- a/packages/zarr-metadata/tests/test_problem_data.py +++ b/packages/zarr-metadata/tests/test_problem_data.py @@ -10,15 +10,17 @@ from __future__ import annotations import copy +import dataclasses import math import pickle -from typing import TYPE_CHECKING, Annotated, Literal +from types import MappingProxyType +from typing import TYPE_CHECKING, Annotated, Any, Literal, cast import pytest from annotated_types import Le from typing_extensions import TypedDict -from zarr_metadata._json import arrays_to_tuples, is_canonical_json, value_at, with_input +from zarr_metadata._json import arrays_to_tuples, is_canonical_json, value_at, with_input, within from zarr_metadata._sentinel import UNSET from zarr_metadata.model import ( MetadataValidationError, @@ -355,3 +357,44 @@ def test_a_value_outside_a_closed_set_is_told_the_set( # In the order the message lists them, as JSON writes them. (problem,) = [problem for problem in problems if problem.loc == loc] assert dict(problem.ctx) == {"expected": expected} + + +def test_error_a_ctx_of_another_mapping_type_is_checked() -> None: + # Only a problem's own `ctx`, checked when it was made, is taken as + # checked: any other mapping is, whatever its type. + with pytest.raises(TypeError, match="ctx is an object of JSON values"): + ValidationProblem( + (), "m", "invalid_value", ctx=cast("Any", MappingProxyType({"x": object()})) + ) + first = ValidationProblem((), "m", "invalid_value", ctx={"ge": 0}) + again = dataclasses.replace(first, loc=("a",)) + assert again.ctx == {"ge": 0} + assert again.ctx is first.ctx + + +def test_a_problem_at_the_first_element_holds_it() -> None: + # Checked against the literal, not `value_at`, which the reader uses. + (problem,) = with_input([ValidationProblem(("a", 0), "bad", "invalid_value")], {"a": [7]}) + assert problem.input == 7 + assert value_at([5, 6], (0,)) == 5 + read = resolve("bytes", DataTypeDefinition, CORE_AND_EXTENSIONS)[0] + (found,) = fill_value_problems(read, [-1]) + assert (found.loc, found.input) == ((0,), -1) + + +def test_a_problem_s_ctx_is_its_own_at_every_level() -> None: + # Copied when the problem is made, at every level, so nothing the + # caller does to what it handed in reaches the problem. + inner: dict[str, int] = {"a": 1} + problem = ValidationProblem((), "m", "invalid_value", ctx={"nested": inner}) + inner["a"] = 2 + assert problem.ctx["nested"] == {"a": 1} + + +def test_error_a_problem_not_below_where_a_reader_counts_from_is_refused() -> None: + # `within` relocates a problem from the document handed in to the one + # read; one located from another root would land in the wrong place. + problem = ValidationProblem(("a", "b"), "m", "invalid_value") + assert within((problem,), ("a",))[0].loc == ("b",) + with pytest.raises(TypeError, match="does not sit below"): + within((problem,), ("c",)) diff --git a/packages/zarr-metadata/tests/test_typed_json.py b/packages/zarr-metadata/tests/test_typed_json.py index 8e0fd54959..73b50f66d2 100644 --- a/packages/zarr-metadata/tests/test_typed_json.py +++ b/packages/zarr-metadata/tests/test_typed_json.py @@ -23,6 +23,7 @@ Literal, NewType, NotRequired, + Required, TypeVar, cast, ) @@ -1224,3 +1225,19 @@ def test_every_typeddict_the_package_declares_compiles(typeddict: type) -> None: # A declaration no parser reads would be a document type `check` could # not be asked about. assert parser_for(typeddict, no_leaf) is not None + + +def test_a_union_of_a_string_and_another_shape_names_both() -> None: + # "a string" takes in only the Literals of strings, not every other branch. + assert describe(str | int) == "a string or an integer" + assert describe(str | None) == "a string or null" + (problem,) = parser(str | int, no_leaf)(None, ())[1] + assert problem.message == "expected a string or an integer, got null" + + +class _Three(TypedDict, closed=True): + x: Annotated[Required[Annotated[ReadOnly[Annotated[int, "inner"]], "middle"]], "outer"] + + +def test_metadata_of_three_annotated_layers_is_kept_in_order() -> None: + assert typeddict_keys(_Three).members["x"] == (Annotated[int, "inner", "middle", "outer"], True) diff --git a/packages/zarr-metadata/tests/v3/test_definitions.py b/packages/zarr-metadata/tests/v3/test_definitions.py index 3477e66be8..35c33517d6 100644 --- a/packages/zarr-metadata/tests/v3/test_definitions.py +++ b/packages/zarr-metadata/tests/v3/test_definitions.py @@ -3,6 +3,7 @@ from __future__ import annotations import copy +import dataclasses import math import pickle from collections.abc import ( @@ -15,7 +16,12 @@ from annotated_types import Ge, Predicate from typing_extensions import TypedDict -from zarr_metadata.model import validate_array_metadata_v3 +from zarr_metadata._sentinel import UNSET +from zarr_metadata.model import ( + is_metadata_field_v3, + validate_array_metadata_v3, + validate_metadata_field_v3, +) from zarr_metadata.model._array import ZarrV3ArrayMetadata from zarr_metadata.v3.array import ZarrV3ArrayMetadataJSON from zarr_metadata.v3.chunk_grid.regular import REGULAR_CHUNK_GRID @@ -732,12 +738,13 @@ def test_error_null_is_not_a_field() -> None: def test_error_a_value_that_is_not_json() -> None: - # Held as None, not JSON; its name still says what claims it. + # Held as `UNSET`, not JSON, which no document holds; its name still + # says what claims it. resolved, found = resolve( {"name": "gzip", "configuration": {"level": math.nan}}, CodecDefinition, SCOPE ) assert resolved == Refused( - json=None, name="gzip", read_as=CodecDefinition, definition=GZIP_CODEC + json=UNSET, name="gzip", read_as=CodecDefinition, definition=GZIP_CODEC ) assert _locs(found) == [(("configuration", "level"), "invalid_value")] @@ -811,7 +818,7 @@ def test_error_a_container_rule_is_not_asked_of_a_malformed_nested_field() -> No field = {"name": "acme.stack", "configuration": {"codecs": [{"configuration": {}}]}} resolved, found = resolve(field, CodecDefinition, SCOPE) assert isinstance(resolved, Refused) - assert _locs(found) == [(("configuration", "codecs", 0, "name"), "invalid_type")] + assert _locs(found) == [(("configuration", "codecs", 0, "name"), "missing_key")] assert ACME_STACK.judge(field["configuration"])[0] is None @@ -1262,3 +1269,145 @@ def test_error_a_field_is_read_as_a_kind_of_metadata(kind: type[Definition[Any]] def test_error_a_scope_refuses_a_definition_of_no_kind() -> None: with pytest.raises(TypeError, match="a definition of no kind"): Context.of(Definition(name="acme.kindless", configuration=Empty)) + + +@pytest.mark.parametrize( + ("name", "valid"), + [ + ("zstd", True), + ("numcodecs.adler32", True), + ("vlen-utf8", True), + ("acme_x", True), + ("r16", True), + ("https://example.com/codec", True), + ("urn:acme:codec", True), + ("urn:acme:%C3%BC", True), + ("x:", False), + ("Int8:", False), + ("a: b", False), + ("urn:acme:codec\n", False), + ("urn:acme:ü", False), + # Whitespace to Python's `re` but not ECMA-262, and the other way + # round: neither a URI character, so both dialects refuse them. + ("urn:a\x1c", False), + ("urn:a", False), + ("", False), + (" ", False), + ("Int8", False), + ("9x", False), + ("int8 ", False), + ("int8\n", False), + ("foo/bar", False), + ("-int8", False), + ("a", False), + ("r*", False), + ], +) +def test_an_extension_is_named_as_the_spec_names_one(name: str, valid: bool) -> None: + """The spec's regex `^[a-z][a-z0-9-_.]+$`, or a URI, which earlier versions of the spec required; anything else is refused before any definition is asked, by the field validator as by the reader.""" + for field, at in ((name, ()), ({"name": name}, ("name",))): + resolved, problems = resolve(field, CodecDefinition, CORE) + if valid: + assert not isinstance(resolved, Refused) + assert problems == () + else: + assert isinstance(resolved, Refused) + assert [(p.loc, p.kind) for p in problems] == [(at, "invalid_value")] + assert "expected an extension name" in problems[0].message + assert [(p.loc, p.kind) for p in validate_metadata_field_v3(field)] == ( + [] if valid else [(at, "invalid_value")] + ) + assert is_metadata_field_v3(field) is valid + + +def test_error_a_bad_name_is_a_problem_when_the_field_is_not_json_too() -> None: + # Refused before a definition is asked on either path: a field whose + # configuration is not JSON reports its name as the JSON path does, + # and no definition is asked to claim it. + field = {"name": "Acme", "configuration": {"x": float("nan")}} + resolved, problems = resolve(field, CodecDefinition, CORE) + assert isinstance(resolved, Refused) + assert resolved.definition is None + assert [(p.loc, p.kind) for p in problems] == [ + (("name",), "invalid_value"), + (("configuration", "x"), "invalid_value"), + ] + + +@pytest.mark.parametrize("name", ["Acme", "acme/x", "", "x"]) +def test_error_a_definition_is_named_as_the_spec_names_an_extension(name: str) -> None: + # No document could name it, so nothing would ever read with it. + with pytest.raises(TypeError, match="is not a name the spec gives an extension"): + CodecDefinition( + name=name, configuration=EmptyConfiguration, kind="bytes_bytes", size="dynamic" + ) + + +def test_error_a_field_built_by_hand_is_named_as_the_spec_names_one() -> None: + with pytest.raises(TypeError, match="named as the spec names an extension"): + Unclaimed(json="Int8", name="Int8", read_as=CodecDefinition) + + +def test_error_a_field_without_a_name_is_missing_one() -> None: + resolved, problems = resolve({"configuration": {}}, CodecDefinition, CORE) + assert isinstance(resolved, Refused) + assert [(p.loc, p.kind, p.message) for p in problems] == [ + (("name",), "missing_key", "missing required key") + ] + + +@pytest.mark.parametrize("digits", [101, 4301]) +def test_raw_bits_of_more_than_a_hundred_digits_are_no_size_but_a_name(digits: int) -> None: + # `int` refuses to convert more than 4,300 digits, and no size has a + # hundred: such a name is an extension's, which nothing in scope claims. + resolved, problems = resolve("r" + "1" * digits, DataTypeDefinition, CORE_AND_EXTENSIONS) + assert type(resolved) is Unclaimed + assert problems == () + + +def test_error_a_rule_that_writes_to_the_configuration_fails_there() -> None: + # A definition's functions are handed a read-only view of the field's + # own configuration, so no field holds what a function wrote. + def writes( + configuration: GzipCodecConfiguration, nested: Nested + ) -> Iterator[ValidationProblem]: + cast("dict[str, object]", configuration)["level"] = -1 + yield from () + + scope = CORE.extended_with(dataclasses.replace(GZIP_CODEC, rules=writes)) + with pytest.raises(TypeError, match="does not support item assignment") as raised: + resolve({"name": "gzip", "configuration": {"level": 1}}, CodecDefinition, scope) + assert raised.value.__notes__ == ["raised by the rules of 'gzip', reading ('configuration',)"] + + +def test_error_a_function_that_writes_to_the_configuration_fails_on_every_path() -> None: + # `judge` hands the rules the same view `resolve` does, and `==` and + # `hash` hand `canonical` one, as `canonical_of` does. + def writes( + configuration: GzipCodecConfiguration, nested: Nested + ) -> Iterator[ValidationProblem]: + cast("dict[str, object]", configuration)["level"] = -1 + yield from () + + with pytest.raises(TypeError, match="does not support item assignment"): + dataclasses.replace(GZIP_CODEC, rules=writes).judge({"level": 1}) + + def folds_in_place(configuration: GzipCodecConfiguration) -> GzipCodecConfiguration: + cast("dict[str, object]", configuration)["level"] = 0 + return configuration + + scope = CORE.extended_with(dataclasses.replace(GZIP_CODEC, canonical=folds_in_place)) + read, _ = resolve({"name": "gzip", "configuration": {"level": 1}}, CodecDefinition, scope) + with pytest.raises(TypeError, match="does not support item assignment"): + hash(read) + + +def test_error_a_canonical_that_raises_says_which_definition_raised_it() -> None: + def refuses(configuration: GzipCodecConfiguration) -> GzipCodecConfiguration: + raise ValueError("no") + + scope = CORE.extended_with(dataclasses.replace(GZIP_CODEC, canonical=refuses)) + read, _ = resolve({"name": "gzip", "configuration": {"level": 1}}, CodecDefinition, scope) + with pytest.raises(ValueError, match="no") as raised: + canonical_of(read, ()) + assert raised.value.__notes__ == ["raised by the canonical of 'gzip'"] diff --git a/packages/zarr-metadata/tests/v3/test_every_definition.py b/packages/zarr-metadata/tests/v3/test_every_definition.py index e7138ba75c..19661e355b 100644 --- a/packages/zarr-metadata/tests/v3/test_every_definition.py +++ b/packages/zarr-metadata/tests/v3/test_every_definition.py @@ -26,7 +26,6 @@ Definition, Read, Refused, - Unclaimed, ValidationProblem, canonical_of, canonicalize, @@ -357,12 +356,17 @@ def test_a_reader_reads_raw_bits_its_own_way_by_defining_r_star() -> None: assert again.definition is RAW_BYTES_DATA_TYPE -@pytest.mark.parametrize("field", ["r*", {"name": "r*", "configuration": {"bits": 16}}]) -def test_r_star_is_notation_that_names_nothing(field: object) -> None: +@pytest.mark.parametrize( + ("field", "at"), + [("r*", ()), ({"name": "r*", "configuration": {"bits": 16}}, ("name",))], + ids=["bare", "object"], +) +def test_error_r_star_is_notation_and_no_name(field: object, at: tuple[str, ...]) -> None: # How the specification's table writes raw bits, and no document's name - # for them: read as any name nothing in scope claims. + # for them: `*` is no character of an extension name. resolved, problems = resolve(field, DataTypeDefinition, CORE_AND_EXTENSIONS) - assert (type(resolved), problems) == (Unclaimed, ()) + assert type(resolved) is Refused + assert [(p.loc, p.kind) for p in problems] == [(at, "invalid_value")] @pytest.mark.parametrize( diff --git a/packages/zarr-metadata/tests/v3/test_fill_values.py b/packages/zarr-metadata/tests/v3/test_fill_values.py index 7334bbcc82..cabd1a8a28 100644 --- a/packages/zarr-metadata/tests/v3/test_fill_values.py +++ b/packages/zarr-metadata/tests/v3/test_fill_values.py @@ -18,7 +18,7 @@ from hypothesis import strategies as st from typing_extensions import TypedDict -from zarr_metadata._json import value_at +from zarr_metadata._json import JSON_DEPTH, value_at from zarr_metadata._sentinel import UNSET from zarr_metadata.model import validate_array_metadata_v3, validate_group_metadata_v3 from zarr_metadata.model._array import ZarrV3ArrayMetadata @@ -298,7 +298,7 @@ def test_error_a_key_the_fill_value_shape_does_not_declare_hides_no_rule() -> No ] -def test_a_fill_value_nested_hundreds_deep_is_read() -> None: +def test_a_fill_value_nested_as_deep_as_a_reader_walks_is_read() -> None: def deep(levels: int) -> dict[str, object]: value: dict[str, object] = {} for _ in range(levels): @@ -308,7 +308,8 @@ def deep(levels: int) -> dict[str, object]: document = dict(ZarrV3ArrayMetadata.create_default().to_json()) | { "data_type": STRUCT, "codecs": [{"name": "bytes", "configuration": {"endian": "little"}}], - "fill_value": {"a": 1, "b": 0.5, "c": deep(600)}, + # `c` sits two levels down, and holds the deepest value the cap admits. + "fill_value": {"a": 1, "b": 0.5, "c": deep(JSON_DEPTH - 3)}, } assert [(p.loc, p.kind) for p in validate_array_metadata_v3(document)] == [ (("fill_value", "c"), "unknown_key") @@ -355,9 +356,12 @@ def _read(data_type: JSONValue) -> Read[DataTypeDefinition[Any]]: # The number of the fewest digits that rounds to the value. ("float32", [0.1, 0.10000000149011612, "0x3dcccccd"], 0.1), # A number is read as a float64, as a JSON parser reads one, and - # then rounded to the type, as numpy rounds it: an integer and a - # number with a fraction that read as one float64 are one value. + # then rounded to the type, as zarrs, tensorstore and numpy round + # it: an integer and a number with a fraction that read as one + # float64 are one value, and an integer whose float64 is halfway + # between two float32 values is the even one, as they store it. ("float32", [2**53 + 2**29 + 1, 9007199791611905.0, 2**53], 9007199000000000.0), + ("float32", [2**60 + 2**36 + 1, 2**60, 1.1529215e18], 1.1529215e18), # Of the numbers of the fewest digits, the one past the nearest: at # a power of two, those that round to it reach further above it. ("float16", [0.015625, "0x2400"], 0.01563), @@ -425,10 +429,16 @@ def test_a_float_s_canonical_spelling_spells_its_bits(data: st.DataObject) -> No # spelling is a fill value of the type, spelling the same bits, and # its own canonical spelling. width = data.draw(st.sampled_from(WIDTHS)) - bits = data.draw(st.integers(0, 2**width - 1)) + # Drawn by part, so normal values, infinities and NaNs each turn up: + # bits drawn whole are almost all subnormals. + fraction = {16: 10, 32: 23, 64: 52}[width] + sign = data.draw(st.integers(0, 1)) + exponent = data.draw(st.integers(0, 2 ** (width - 1 - fraction) - 1)) + mantissa = data.draw(st.integers(0, 2**fraction - 1)) + bits = (sign << (width - 1)) | (exponent << fraction) | mantissa read = _read(f"float{width}") spelled = canonical_fill_value(read, f"0x{bits:0{width // 4}x}") - assert spelled is not None + assert spelled is not UNSET assert fill_value_problems(read, spelled) == () assert float_bits(cast("float | str", spelled), width) == bits assert _alike(canonical_fill_value(read, spelled), spelled) @@ -467,3 +477,29 @@ def not_json(configuration: EmptyConfiguration, nested: Nested, value: object) - resolved, _ = resolve("acme.point", DataTypeDefinition, scope) with pytest.raises(TypeError, match="'acme.point': its fill_value_canonical gives JSON"): canonical_fill_value(resolved, {"x": 1}) + + +@pytest.mark.parametrize( + ("name", "low", "high"), + [ + ("int8", -(2**7), 2**7 - 1), + ("int16", -(2**15), 2**15 - 1), + ("int32", -(2**31), 2**31 - 1), + ("int64", -(2**63), 2**63 - 1), + ("uint8", 0, 2**8 - 1), + ("uint16", 0, 2**16 - 1), + ("uint32", 0, 2**32 - 1), + ("uint64", 0, 2**64 - 1), + ], +) +def test_an_integer_type_takes_exactly_its_range(name: str, low: int, high: int) -> None: + read = _read(name) + assert fill_value_problems(read, low) == () + assert fill_value_problems(read, high) == () + for outside in (low - 1, high + 1): + (problem,) = fill_value_problems(read, outside) + assert (problem.loc, problem.kind, dict(problem.ctx)) == ( + (), + "invalid_value", + {"ge": low, "le": high}, + ) From c7a38a683970eb357038b5d31258446949375468 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Mon, 5 Oct 2026 17:45:25 +0200 Subject: [PATCH 16/94] fix(zarr-metadata): report a value past JSON_DEPTH as too deep without calling repr `shown` relied on `repr` raising RecursionError for a deeply nested value. Python 3.14 limits recursion by the real C stack, so on Linux a list nested 100,000 levels deep printed fine and the problem message was the whole value. Now any value with a problem past `JSON_DEPTH` is reported as "a value nested too deep to show", on every interpreter and platform. The test nests one level past the limit, where repr always works. Assisted-by: ClaudeCode:claude-opus-5-5 --- packages/zarr-metadata/src/zarr_metadata/_json.py | 6 +++++- packages/zarr-metadata/tests/model/test_array.py | 7 ++++--- 2 files changed, 9 insertions(+), 4 deletions(-) diff --git a/packages/zarr-metadata/src/zarr_metadata/_json.py b/packages/zarr-metadata/src/zarr_metadata/_json.py index 4a97f7c9a7..ba985eeafc 100644 --- a/packages/zarr-metadata/src/zarr_metadata/_json.py +++ b/packages/zarr-metadata/src/zarr_metadata/_json.py @@ -427,8 +427,12 @@ def json_text(value: JSONValue) -> str: def shown(value: object) -> str: """`value` as a problem's message shows it: as the JSON a document writes, `null` and `[1, 2]`, or by its repr when it is not JSON; what the interpreter will not write, an integer of too many digits or a value nested too deep, by saying so.""" refined, problems = _refine(value, (), finite=False) + if any(len(problem.loc) >= JSON_DEPTH for problem in problems): + # Nested past the levels a reader walks: said so, not left to the + # repr, which overflows at a depth the interpreter and platform set. + return "a value nested too deep to show" if len(problems) != 0: - # Not JSON, or nested past the levels a reader walks. + # Not JSON. return shown_by_python(value) try: return json.dumps(refined, ensure_ascii=False) diff --git a/packages/zarr-metadata/tests/model/test_array.py b/packages/zarr-metadata/tests/model/test_array.py index 25e3f38a5f..befeff0019 100644 --- a/packages/zarr-metadata/tests/model/test_array.py +++ b/packages/zarr-metadata/tests/model/test_array.py @@ -2476,11 +2476,12 @@ def test_a_problem_shows_an_integer_too_long_to_write_by_its_size( def test_a_problem_shows_a_value_nested_too_deep_to_write_by_saying_so() -> None: - # `_refine` stops at `JSON_DEPTH`; the repr that would show a value it - # refused walks as deep as the value nests, and overflows. + # `_refine` stops at `JSON_DEPTH`, and a value nested past it is said to + # be too deep, not shown by a repr, whose own limit the interpreter and + # platform set: one level past, which any repr would write, is enough. deep: list[object] = [] innermost = deep - for _ in range(100_000): + for _ in range(JSON_DEPTH + 1): nested: list[object] = [] innermost.append(nested) innermost = nested From b55cafb120f721d795985ecc187c344389e2eeb4 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Mon, 5 Oct 2026 17:51:26 +0200 Subject: [PATCH 17/94] refactor(zarr-metadata): remove the RecursionError fallback for non-JSON values `shown_by_python` caught RecursionError so that a non-JSON value nested deeper than repr could handle was described instead of raising. No parsed document can contain such a value: sets and frozensets are not JSON. Its test only passed where the platform's stack happened to overflow at the chosen depth. Values a document can hold that nest past `JSON_DEPTH` are already reported as too deep by `shown`, without repr. Assisted-by: ClaudeCode:claude-opus-5-5 --- packages/zarr-metadata/src/zarr_metadata/_json.py | 4 +--- packages/zarr-metadata/tests/model/test_array.py | 9 +-------- 2 files changed, 2 insertions(+), 11 deletions(-) diff --git a/packages/zarr-metadata/src/zarr_metadata/_json.py b/packages/zarr-metadata/src/zarr_metadata/_json.py index ba985eeafc..636f5293cd 100644 --- a/packages/zarr-metadata/src/zarr_metadata/_json.py +++ b/packages/zarr-metadata/src/zarr_metadata/_json.py @@ -456,11 +456,9 @@ def shown_key(key: object) -> str: def shown_by_python(value: object) -> str: - """`value` as Python shows it, for what is not JSON a reader walks; what the interpreter will not write -- nested too deep for its repr, or holding an integer of more digits than it writes -- said so.""" + """`value` as Python shows it, for what is not JSON a reader walks; what the interpreter will not write -- holding an integer of more digits than it writes -- said so.""" try: return repr(value) - except RecursionError: - return "a value nested too deep to show" except ValueError: return f"a value of type {type(value).__name__} the interpreter will not write" diff --git a/packages/zarr-metadata/tests/model/test_array.py b/packages/zarr-metadata/tests/model/test_array.py index befeff0019..46f81e7283 100644 --- a/packages/zarr-metadata/tests/model/test_array.py +++ b/packages/zarr-metadata/tests/model/test_array.py @@ -2459,20 +2459,13 @@ def test_a_problem_shows_an_integer_too_long_to_write_by_its_size( assert problem.message == ( "non-string metadata field key a value of type tuple the interpreter will not write" ) - # So is what `refine_json` reports as no JSON at all: a set holding - # one, or frozensets nested past what the interpreter's repr walks. + # So is what `refine_json` reports as no JSON at all: a set holding one. document = {**ZarrV3ArrayMetadata.create_default().to_json(), "attributes": {"a": {10**5000}}} (problem,) = validate_array_metadata_v3(document) assert (problem.loc, problem.message) == ( ("attributes", "a"), "not a JSON-serializable value: a value of type set the interpreter will not write", ) - frozen: frozenset[object] = frozenset() - for _ in range(100_000): - frozen = frozenset([frozen]) - document = {**ZarrV3ArrayMetadata.create_default().to_json(), "attributes": {"a": frozen}} - (problem,) = validate_array_metadata_v3(document) - assert problem.message == "not a JSON-serializable value: a value nested too deep to show" def test_a_problem_shows_a_value_nested_too_deep_to_write_by_saying_so() -> None: From 96f17e3dcf5f57fbf664cb227a27c15d8397b7ee Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Mon, 5 Oct 2026 22:18:26 +0200 Subject: [PATCH 18/94] fix(zarr-metadata): fix review findings on the reader stack - `shown` reports "too deep" only for the depth problem itself. A non-JSON value sitting exactly on the last allowed level is now printed instead of being mislabeled. - Complex fill-value rules keep each component problem's `input` and `ctx`; only the `loc` is moved under the component index. - `Schemas.defined` restores every `$defs` entry, name and use count a failed write touched, not only its own name. - The node JSON Schema docstring no longer says `consolidated_metadata` may be `null`. Both the schema and the validator refuse that. - The 382 bugfix changelog fragment is rewritten for users without naming internals. - "modelled" is spelled "modeled" in new text. Assisted-by: ClaudeCode:claude-opus-5-5 --- packages/zarr-metadata/README.md | 2 +- packages/zarr-metadata/changes/381.feature.md | 2 +- packages/zarr-metadata/changes/382.bugfix.md | 29 +------------------ packages/zarr-metadata/docs/index.md | 2 +- .../zarr-metadata/src/zarr_metadata/_json.py | 8 +++-- .../src/zarr_metadata/_typed_json.py | 11 +++++-- .../src/zarr_metadata/model/_json_schema.py | 3 +- .../src/zarr_metadata/v3/_hierarchy.py | 2 +- .../src/zarr_metadata/v3/data_type/_float.py | 3 +- .../tests/model/test_refine_json.py | 27 ++++++++++++++--- .../zarr-metadata/tests/test_json_schema.py | 10 +++++++ .../tests/v3/test_fill_values.py | 19 +++++++++++- 12 files changed, 73 insertions(+), 45 deletions(-) diff --git a/packages/zarr-metadata/README.md b/packages/zarr-metadata/README.md index 02d132fd81..b9f66d75da 100644 --- a/packages/zarr-metadata/README.md +++ b/packages/zarr-metadata/README.md @@ -153,7 +153,7 @@ root: the document of the node at `/a/b` sits at the key `a/b`, and the documents and the group make a tree in which only groups hold nodes and each node's parent is held. `NodeName` and `NodePath`, in `zarr_metadata.v3`, are the strings the spec's rules for node names and -paths hold of, modelled on zarrs' types of those names, and +paths hold of, modeled on zarrs' types of those names, and `validate_node_name_v3`, `is_node_name_v3` and `parse_node_name_v3`, and their `node_path` twins, judge a string by them. diff --git a/packages/zarr-metadata/changes/381.feature.md b/packages/zarr-metadata/changes/381.feature.md index a526471296..03dd5fd425 100644 --- a/packages/zarr-metadata/changes/381.feature.md +++ b/packages/zarr-metadata/changes/381.feature.md @@ -1,5 +1,5 @@ `NodeName` and `NodePath`, in `zarr_metadata.v3`, are the names and paths -of the nodes of a v3 hierarchy, modelled on zarrs' types of those names, +of the nodes of a v3 hierarchy, modeled on zarrs' types of those names, and `zarr_metadata.model` judges a string by the spec's rules for them: `validate_node_name_v3`, `is_node_name_v3` and `parse_node_name_v3`, and their `node_path` twins. diff --git a/packages/zarr-metadata/changes/382.bugfix.md b/packages/zarr-metadata/changes/382.bugfix.md index 1837af2952..646cd330b3 100644 --- a/packages/zarr-metadata/changes/382.bugfix.md +++ b/packages/zarr-metadata/changes/382.bugfix.md @@ -1,28 +1 @@ -A reader walks 256 levels of nesting, and a value nested deeper is an -`invalid_value` at the level past the last, wherever it sits. A 2 KB -document nested a thousand deep took every validator, reader, parser and -`from_key_value` past the interpreter's recursion limit, which escaped as -`RecursionError`; so did `json.loads` on store bytes of a hundred thousand -`[`, now an `invalid_json` problem like any other undecodable bytes; and -so did a chain of groups each holding the next in its consolidated -metadata, which a reader descended without bound: each document is now -read from where it sits in the one handed in, its levels counted from -that one's root, and the reader judges each container it walks -- a -document's members, the consolidated metadata it descends into -- where -it sits, as `refine_json` of the whole would, and -`ZarrV3ConsolidatedMetadata.from_json` reads the member from where it -sits, under a group's key; `is_json` and the `is_*` guards walk no deeper -than a reader either, and a v2 `dtype` of field records is read as JSON -first, so no path recurses past the cap. Every reader, writer and -comparison takes one frame for each level -- the v2 models copied with -`copy.deepcopy`, which takes two, and overflowed on documents their -validators accept -- and `copy.deepcopy` of a model, and `pickle` before -Python 3.12, take two, so a document at the cap takes about half of the -interpreter's default limit, and the rest is the caller's. -`validate_json` takes the `loc` its value sits at, and the v2 validator -hands it one for a compressor, a filter and a fill value, so their -levels are counted from the document's root too, as every v3 reader -counts them. `arrays_to_tuples` walks one frame per level, as `copied` -does, so `parse_*` reads as deep as `validate_*`. A problem copied by -`with_input` or `prefixed` is not checked again, and a struct's fill -value is judged in time linear in its fields. +Validators, readers and `from_key_value` no longer crash with `RecursionError` on deeply nested documents, including a chain of groups each nested in the last one's consolidated metadata. Nesting is now capped at 256 levels: a value nested deeper is reported as an `invalid_value` problem where the nesting passes the limit. diff --git a/packages/zarr-metadata/docs/index.md b/packages/zarr-metadata/docs/index.md index 027800cb77..232ccd6a84 100644 --- a/packages/zarr-metadata/docs/index.md +++ b/packages/zarr-metadata/docs/index.md @@ -168,7 +168,7 @@ root: the document of the node at `/a/b` sits at the key `a/b`, and the documents and the group make a tree in which only groups hold nodes and each node's parent is held. `NodeName` and `NodePath`, in `zarr_metadata.v3`, are the strings the spec's rules for node names and -paths hold of, modelled on zarrs' types of those names, and +paths hold of, modeled on zarrs' types of those names, and `validate_node_name_v3`, `is_node_name_v3` and `parse_node_name_v3`, and their `node_path` twins, judge a string by them. diff --git a/packages/zarr-metadata/src/zarr_metadata/_json.py b/packages/zarr-metadata/src/zarr_metadata/_json.py index 636f5293cd..f3e00fb541 100644 --- a/packages/zarr-metadata/src/zarr_metadata/_json.py +++ b/packages/zarr-metadata/src/zarr_metadata/_json.py @@ -50,6 +50,9 @@ frames, and the rest is the caller's. """ +_PAST_THE_LEVELS: Final = f"nested deeper than the {JSON_DEPTH} levels a reader walks" +"""The message of the problem a container past the levels a reader walks is.""" + class _Ctx(Mapping[str, JSONValue]): """What a problem holds as its `ctx`: a copy of its own, arrays as tuples, checked to be JSON when it was made, which nothing edits after. @@ -349,8 +352,7 @@ def nested_past_the_levels(value: object, loc: tuple[str | int, ...]) -> Validat return None if not isinstance(value, (Mapping, Sequence)) or isinstance(value, (bytes, bytearray)): return None - message = f"nested deeper than the {JSON_DEPTH} levels a reader walks" - return ValidationProblem(loc, message, "invalid_value") + return ValidationProblem(loc, _PAST_THE_LEVELS, "invalid_value") def within( @@ -427,7 +429,7 @@ def json_text(value: JSONValue) -> str: def shown(value: object) -> str: """`value` as a problem's message shows it: as the JSON a document writes, `null` and `[1, 2]`, or by its repr when it is not JSON; what the interpreter will not write, an integer of too many digits or a value nested too deep, by saying so.""" refined, problems = _refine(value, (), finite=False) - if any(len(problem.loc) >= JSON_DEPTH for problem in problems): + if any(problem.message == _PAST_THE_LEVELS for problem in problems): # Nested past the levels a reader walks: said so, not left to the # repr, which overflows at a depth the interpreter and platform set. return "a value nested too deep to show" diff --git a/packages/zarr-metadata/src/zarr_metadata/_typed_json.py b/packages/zarr-metadata/src/zarr_metadata/_typed_json.py index d2dca722e7..468755e27d 100644 --- a/packages/zarr-metadata/src/zarr_metadata/_typed_json.py +++ b/packages/zarr-metadata/src/zarr_metadata/_typed_json.py @@ -1250,8 +1250,9 @@ def defined(self, key: object, name: str, write: Callable[[], JSONSchema]) -> JS The entry is named `name`, or `name` and a number when another key holds that name, and it is reserved before it is written, so a schema that holds itself refers to itself. One `write` fails to - write is not left reserved: a later reference to it would be to an - empty schema, which takes anything. + write is not left reserved -- a later reference to it would be to an + empty schema, which takes anything -- and nor is anything it wrote + or counted before it failed. """ name_held = self._names.get(key) if name_held is None: @@ -1259,12 +1260,16 @@ def defined(self, key: object, name: str, write: Callable[[], JSONSchema]) -> JS while name_held in self._defs: number += 1 name_held = f"{name}{number}" + # What a failed write leaves is put back as it was: the entries + # it wrote of what it holds, and the uses it counted, as well as + # its own name. + before = (dict(self._defs), dict(self._names), dict(self._uses)) self._names[key] = name_held self._defs[name_held] = {} try: self._defs[name_held] = write() except BaseException: - del self._names[key], self._defs[name_held] + self._defs, self._names, self._uses = before raise self._uses[name_held] = self._uses.get(name_held, 0) + 1 return {"$ref": _pointer(name_held)} diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_json_schema.py b/packages/zarr-metadata/src/zarr_metadata/model/_json_schema.py index b8b07e0a90..05dd0aef64 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_json_schema.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/_json_schema.py @@ -30,7 +30,8 @@ def node_metadata_json_schema_v3(*, context: Context = CORE_AND_EXTENSIONS) -> J the data type's definition declares for one -- an `int8`'s an integer in [-128, 127] -- when the document names a data type in scope. A group's `consolidated_metadata` holds array and group documents, by - path, or is `null`. Each document is in `$defs` under the name of its + path; a `null` one, which a zarr-python 3.0.x bug wrote, is refused, as + the validator refuses it. Each document is in `$defs` under the name of its TypedDict: `ZarrV3ArrayMetadataJSON` is an array's alone. A JSON Schema says what each member is, and what the rules say of diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/_hierarchy.py b/packages/zarr-metadata/src/zarr_metadata/v3/_hierarchy.py index 7b6cac919c..ac8efbd897 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/_hierarchy.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/_hierarchy.py @@ -1,6 +1,6 @@ """A Zarr v3 hierarchy: the names and paths of its nodes, and the tree they make. -`NodeName` and `NodePath` are modelled on zarrs' types of those names, and +`NodeName` and `NodePath` are modeled on zarrs' types of those names, and hold the spec's rules, which reserve `zarr.json` too. `hierarchy_problems` judges the node type of each node of a hierarchy, by its path, as a tree. diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/_float.py b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/_float.py index bcdb24cb8e..595ee23bf4 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/_float.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/_float.py @@ -9,6 +9,7 @@ (https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v3/data-types/index.rst#L88-L91). """ +import dataclasses import struct from collections.abc import Callable, Iterable, Iterator from dataclasses import dataclass @@ -202,7 +203,7 @@ def __call__( ) -> Iterator[ValidationProblem]: for index, part in enumerate(value): for found in self.component(configuration, nested, part): - yield ValidationProblem((index, *found.loc), found.message, found.kind) + yield dataclasses.replace(found, loc=(index, *found.loc)) def complex_fill_value_canonical( diff --git a/packages/zarr-metadata/tests/model/test_refine_json.py b/packages/zarr-metadata/tests/model/test_refine_json.py index 35a54aaee0..aad8edd2cb 100644 --- a/packages/zarr-metadata/tests/model/test_refine_json.py +++ b/packages/zarr-metadata/tests/model/test_refine_json.py @@ -4,7 +4,7 @@ import math from collections import OrderedDict -from typing import TYPE_CHECKING +from typing import TYPE_CHECKING, cast import pytest @@ -116,6 +116,14 @@ def test_error_a_value_nested_deeper_than_a_reader_walks() -> None: assert problems[0].loc[:2] == ("a", 0) +def _nested(depth: int, innermost: object) -> list[object]: + """`innermost` inside `depth` arrays, each holding the next: `innermost` sits `depth` levels down.""" + value: object = innermost + for _ in range(depth): + value = [value] + return cast("list[object]", value) + + @pytest.mark.parametrize( ("value", "text"), [ @@ -126,12 +134,23 @@ def test_error_a_value_nested_deeper_than_a_reader_walks() -> None: ({"a": None}, '{"a": null}'), (float("nan"), "NaN"), ({1: 2}, "{1: 2}"), + (_nested(JSON_DEPTH, []), "a value nested too deep to show"), + (_nested(JSON_DEPTH, {1}), "[" * JSON_DEPTH + "{1}" + "]" * JSON_DEPTH), + ], + ids=[ + "null", + "true", + "string", + "array", + "object", + "non-finite", + "not-json", + "past-the-levels", + "not-json-at-the-last-level", ], - ids=["null", "true", "string", "array", "object", "non-finite", "not-json"], ) def test_a_value_is_shown_as_the_json_a_document_writes(value: object, text: str) -> None: - # What a problem's message says a document holds: its JSON, and the - # value's repr only when it is not JSON. + """A problem's message shows a value as its JSON, by its repr when it is not JSON, and says so when it nests past the levels a reader walks: a container there, not a non-JSON value sitting on the last level, which is shown.""" assert shown(value) == text diff --git a/packages/zarr-metadata/tests/test_json_schema.py b/packages/zarr-metadata/tests/test_json_schema.py index 1f1f136727..960f32ddac 100644 --- a/packages/zarr-metadata/tests/test_json_schema.py +++ b/packages/zarr-metadata/tests/test_json_schema.py @@ -310,6 +310,16 @@ def test_error_a_schema_that_fails_to_write_leaves_nothing_behind() -> None: assert schemas.document({}) == {"$schema": DIALECT} with pytest.raises(TypeError, match="is not a shape JSON takes"): schemas.of(unread) + # Nor is what it wrote of a shape it holds before it failed, nor a use + # it counted: a later root of that shape, used once, is written in place. + held = _closed("Held", {"x": int}) + with pytest.raises(TypeError, match="is not a shape JSON takes"): + schemas.of(_closed("Partly", {"held": held, "value": bytes})) + assert schemas.document({}) == {"$schema": DIALECT} + assert schemas.document(schemas.of(held)) == { + "$schema": DIALECT, + **Schemas().object_of(held), + } def test_json_schema_refuses_metadata_check_does_not_hold_a_value_to() -> None: diff --git a/packages/zarr-metadata/tests/v3/test_fill_values.py b/packages/zarr-metadata/tests/v3/test_fill_values.py index cabd1a8a28..d871b106e9 100644 --- a/packages/zarr-metadata/tests/v3/test_fill_values.py +++ b/packages/zarr-metadata/tests/v3/test_fill_values.py @@ -22,7 +22,7 @@ from zarr_metadata._sentinel import UNSET from zarr_metadata.model import validate_array_metadata_v3, validate_group_metadata_v3 from zarr_metadata.model._array import ZarrV3ArrayMetadata -from zarr_metadata.v3.data_type._float import FloatWidth, float_bits +from zarr_metadata.v3.data_type._float import FloatWidth, complex_fill_value_rules, float_bits from zarr_metadata.v3.data_type.struct import STRUCT_DATA_TYPE from zarr_metadata.v3.definition import ( CORE_AND_EXTENSIONS, @@ -503,3 +503,20 @@ def test_an_integer_type_takes_exactly_its_range(name: str, low: int, high: int) "invalid_value", {"ge": low, "le": high}, ) + + +def test_a_complex_fill_value_keeps_what_its_component_rules_found() -> None: + """A complex fill value's problems are its component rules' own, each moved under its component's index, with the `input` and `ctx` they carry.""" + + def at_least_one( + configuration: EmptyConfiguration, nested: Nested, part: object + ) -> Iterator[ValidationProblem]: + if part == 0: + yield ValidationProblem( + ("x",), "expected >= 1", "invalid_value", input=0, ctx={"ge": 1} + ) + + rules = complex_fill_value_rules(at_least_one) + (found,) = rules({}, {}, (1.0, 0)) + assert (found.loc, found.message, found.kind) == ((1, "x"), "expected >= 1", "invalid_value") + assert (found.input, dict(found.ctx)) == (0, {"ge": 1}) From 022267413cda0e4b9ddce1aeea6ea8d734c8d2d7 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Tue, 6 Oct 2026 18:25:50 +0200 Subject: [PATCH 19/94] docs(zarr-metadata): rewrite the 382 changelog fragments in plain English Assisted-by: ClaudeCode:claude-fable-5-1 --- packages/zarr-metadata/changes/382.bugfix.1.md | 6 +----- packages/zarr-metadata/changes/382.bugfix.2.md | 9 +-------- packages/zarr-metadata/changes/382.bugfix.3.md | 7 +------ packages/zarr-metadata/changes/382.bugfix.4.md | 13 +------------ packages/zarr-metadata/changes/382.bugfix.md | 2 +- packages/zarr-metadata/changes/382.feature.1.md | 13 +------------ packages/zarr-metadata/changes/382.feature.md | 10 +--------- 7 files changed, 7 insertions(+), 53 deletions(-) diff --git a/packages/zarr-metadata/changes/382.bugfix.1.md b/packages/zarr-metadata/changes/382.bugfix.1.md index c4ea1234a5..e44aa01c12 100644 --- a/packages/zarr-metadata/changes/382.bugfix.1.md +++ b/packages/zarr-metadata/changes/382.bugfix.1.md @@ -1,5 +1 @@ -A field object -- a `Read` or `Unclaimed` built by hand -- inside a -document a caller hands in is not JSON, and is refused as such, where it -passed the validators as read and let a model write a document the same -validator refuses. Only a model's own fields are held as read: its -constructor and `update` pass them so. +A `Read` or `Unclaimed` object placed inside a document passed to a reader is now refused as not JSON. Before, it passed validation as if it had been read, and a model could then write a document that the same validator refused. diff --git a/packages/zarr-metadata/changes/382.bugfix.2.md b/packages/zarr-metadata/changes/382.bugfix.2.md index 7394646791..b4ef29356f 100644 --- a/packages/zarr-metadata/changes/382.bugfix.2.md +++ b/packages/zarr-metadata/changes/382.bugfix.2.md @@ -1,8 +1 @@ -What is absent is `UNSET`, and `None` is a JSON null the document wrote: -a reading's `data_type`, `chunk_grid` and `chunk_key_encoding` are `UNSET` -for a key the document does not hold, where they were `None`, which a -`data_type: null` reads as too; and `Refused.json` is `UNSET` for a field -that was not JSON. A metadata field without a `name` is a `missing_key` -at `name`, where it was an `invalid_type`. `ZarrV3ConsolidatedMetadata` -declares `must_understand` as `False`, as it declares `kind`, rather than -taking and refusing a value. +Absent members are now `UNSET` instead of `None`, so they can be told apart from a JSON `null`: a reading's `data_type`, `chunk_grid` and `chunk_key_encoding` when the key is missing, and `Refused.json` when the field was not JSON. A metadata field with no `name` is a `missing_key` problem instead of `invalid_type`. `ZarrV3ConsolidatedMetadata` declares `must_understand` as `False` instead of accepting and then refusing a value. diff --git a/packages/zarr-metadata/changes/382.bugfix.3.md b/packages/zarr-metadata/changes/382.bugfix.3.md index 5c022a5071..8ccd49e534 100644 --- a/packages/zarr-metadata/changes/382.bugfix.3.md +++ b/packages/zarr-metadata/changes/382.bugfix.3.md @@ -1,6 +1 @@ -A group's `consolidated_metadata` of `null` is a problem, `invalid_type` -at the key, and no longer read as absent, nor admitted by either JSON -Schema, the package's or pydantic's: -the spec says an object, and the package models nothing else as right. A -zarr-python 3.0.x bug wrote it; a reader of those stores strips the key -before reading. +A group's `consolidated_metadata` of `null` is now an `invalid_type` problem. It is no longer read as absent, and neither the package's JSON Schema nor pydantic's admits it. The spec says the member is an object. zarr-python 3.0.x wrote `null` by mistake; a reader of those stores should remove the key before reading. diff --git a/packages/zarr-metadata/changes/382.bugfix.4.md b/packages/zarr-metadata/changes/382.bugfix.4.md index 8196eeaefa..c0d3d91e27 100644 --- a/packages/zarr-metadata/changes/382.bugfix.4.md +++ b/packages/zarr-metadata/changes/382.bugfix.4.md @@ -1,12 +1 @@ -A definition's functions are handed a read-only view of a field's -configuration on every path that asks them -- `resolve` and `judge` the -rules, `canonical_of`, `==` and `hash` the `canonical` -- so a rule that -assigns one of its members fails there, with the note saying whose rule; -an error a `canonical` raises says which definition's it is, as every -other function's does. A raw-bits name of more than a hundred digits is -an extension's name, not a size, where `r` and 4,301 digits raised -`ValueError` from every reader; an integer too long for the interpreter -to write is shown by its size in a message, and a value nested too deep -for it to write by saying so, where each raised; a key no JSON object -holds -- a tuple, `None` -- is shown as Python shows it, not as the JSON -it is not. +Definition functions (`rules`, `canonical`, and the rest) always receive a read-only view of a field's configuration, so a rule that assigns to it fails with an error naming the rule. A raw-bits name with more than 100 digits is treated as an extension name, not a size; before, it raised `ValueError`. Problem messages now show an over-long integer by its bit length, a too-deeply-nested value as "too deep to show", and a non-string key with Python's repr; before, each of these raised. diff --git a/packages/zarr-metadata/changes/382.bugfix.md b/packages/zarr-metadata/changes/382.bugfix.md index 646cd330b3..9dbc638adb 100644 --- a/packages/zarr-metadata/changes/382.bugfix.md +++ b/packages/zarr-metadata/changes/382.bugfix.md @@ -1 +1 @@ -Validators, readers and `from_key_value` no longer crash with `RecursionError` on deeply nested documents, including a chain of groups each nested in the last one's consolidated metadata. Nesting is now capped at 256 levels: a value nested deeper is reported as an `invalid_value` problem where the nesting passes the limit. +Validators, readers and `from_key_value` no longer raise `RecursionError` on deeply nested documents, including a chain of groups each nested in the previous one's consolidated metadata. Nesting is capped at 256 levels. A value nested deeper is reported as an `invalid_value` problem at the level where the limit is passed. diff --git a/packages/zarr-metadata/changes/382.feature.1.md b/packages/zarr-metadata/changes/382.feature.1.md index 21e65f1424..eb59aa833c 100644 --- a/packages/zarr-metadata/changes/382.feature.1.md +++ b/packages/zarr-metadata/changes/382.feature.1.md @@ -1,12 +1 @@ -The pydantic types raise a `ValidationError` with one line error per -problem, as pydantic reports its own: its `type` the problem's `kind`, at -the problem's `loc` under the field's, with the `input` found there -- -for a key that is missing, the object that lacks it, as pydantic's own -`missing` reports; for what a problem could not hold, not being JSON a -reader walks, what sits at its loc -- and the problem's `ctx`, where -every problem was concatenated into one `value_error` at the field. -pydantic renders a message as a template of its ctx, key by key, with no -escape, so a message that holds a placeholder -- a document value -written `{expected}`, shown in it -- rides whole as the ctx's `message`, -last, and `msg` is the problem's message whatever it holds; a ctx member -of that name yields to it. +The pydantic types raise `ValidationError` with one line error per problem, each carrying `type` (the problem's kind), `loc`, `input` and `ctx`. Before, every problem was concatenated into one `value_error`. For a missing key, `input` is the object that lacks it, as pydantic's own `missing` error does. A message that contains a placeholder such as `{expected}` is passed through unchanged. diff --git a/packages/zarr-metadata/changes/382.feature.md b/packages/zarr-metadata/changes/382.feature.md index 9bfb35ebe3..3f7e74d7d7 100644 --- a/packages/zarr-metadata/changes/382.feature.md +++ b/packages/zarr-metadata/changes/382.feature.md @@ -1,9 +1 @@ -An extension is named as the spec names one -- `^[a-z][a-z0-9-_.]+$`, or -a URI, which earlier versions of the spec required, in the characters -RFC 3986 writes one in -- and any other name, `""` or `"foo/bar"`, is -refused before a definition is asked, an `invalid_value` at `name`, -where every string read as an unknown extension. `well_named` says so of -a string, `validate_metadata_field_v3` refuses it as the envelope rule it -is, the JSON Schema's unclaimed branch carries the same pattern, -and a definition given another name is a `TypeError` when it is built, -since no field could name it. +Extension names must match `^[a-z][a-z0-9-_.]+$` or be a URI in RFC 3986 characters. Any other name (`""`, `"foo/bar"`) is refused with an `invalid_value` problem at `name` before any definition is looked up; before, every string was read as an unknown extension. `well_named` checks a string, `validate_metadata_field_v3` enforces the rule, the JSON Schema carries the same pattern, and a definition built with an invalid name raises `TypeError`. From 635508df7636e680925965fb9da81dbd749e8f7f Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Tue, 6 Oct 2026 10:20:35 +0200 Subject: [PATCH 20/94] feat(zarr-metadata): add a separate read that repairs known writer bugs The strict readers refuse documents that some writers produced. Repair is now a separate step that a caller asks for by name: `read_repaired_node_metadata_v3` applies `repair_node_metadata_v3` to undo each known writer bug, then reads the result with `read_node_metadata_v3`, and returns the reading together with the list of repairs made. The set of repairable documents is closed. Each repair applies only when the document's members match that repair's TypedDict: `ZarrV3ZeroChunkArrayMetadataJSON` for the chunk length of 0 along an empty dimension that zarr-python 3.0 and 3.1 wrote, and `ZarrV3NullConsolidatedGroupMetadataJSON` for the `consolidated_metadata: null` that zarr-python 3.0.x wrote. Documents inside consolidated metadata are repaired too. Anything else is left for the strict read to report. Assisted-by: ClaudeCode:claude-opus-5-5 --- .../src/zarr_metadata/model/__init__.py | 25 +- .../src/zarr_metadata/model/_repair.py | 233 ++++++++++++++++++ .../zarr-metadata/tests/model/test_repair.py | 131 ++++++++++ .../zarr-metadata/tests/test_public_api.py | 4 + 4 files changed, 392 insertions(+), 1 deletion(-) create mode 100644 packages/zarr-metadata/src/zarr_metadata/model/_repair.py create mode 100644 packages/zarr-metadata/tests/model/test_repair.py diff --git a/packages/zarr-metadata/src/zarr_metadata/model/__init__.py b/packages/zarr-metadata/src/zarr_metadata/model/__init__.py index f967996b82..e1b2e6a151 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/__init__.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/__init__.py @@ -13,7 +13,10 @@ that narrows or raises `MetadataValidationError`; a v3 array or group document also gets `read_array_metadata_v3` or `read_group_metadata_v3`, one read that returns what it read, the problems, and the model when -there are none. `node_metadata_json_schema_v3` writes what the v3 +there are none. A store another writer made, holding a known writer +bug, is read by `read_repaired_node_metadata_v3`, which undoes each one +with `repair_node_metadata_v3` before the strict read and says what it +changed. `node_metadata_json_schema_v3` writes what the v3 validators read as a JSON Schema, but for the rules. Model `from_json` / `from_key_value` constructors raise `MetadataValidationError` for every ingestion failure, including missing @@ -61,6 +64,17 @@ validate_node_metadata_v3, ) from zarr_metadata.model._json_schema import node_metadata_json_schema_v3 +from zarr_metadata.model._repair import ( + Repair, + RepairKind, + ZarrV3NullConsolidatedGroupMetadataJSON, + ZarrV3RepairedNodeMetadataReading, + ZarrV3ZeroChunkArrayMetadataJSON, + ZarrV3ZeroChunkRegularGridConfigurationJSON, + ZarrV3ZeroChunkRegularGridJSON, + read_repaired_node_metadata_v3, + repair_node_metadata_v3, +) from zarr_metadata.model._validation import ( ARRAY_METADATA_OPTIONAL_KEYS_V3, ARRAY_METADATA_REQUIRED_KEYS_V2, @@ -143,6 +157,8 @@ "ZARR_V3_GROUP_METADATA_STORE_KEY", "MetadataValidationError", "ProblemKind", + "Repair", + "RepairKind", "ValidationProblem", "ZarrV2ArrayMetadata", "ZarrV2ArrayMetadataPartial", @@ -164,7 +180,12 @@ "ZarrV3GroupMetadataUpdate", "ZarrV3NodeMetadata", "ZarrV3NodeMetadataReading", + "ZarrV3NullConsolidatedGroupMetadataJSON", + "ZarrV3RepairedNodeMetadataReading", "ZarrV3UnknownNodeReading", + "ZarrV3ZeroChunkArrayMetadataJSON", + "ZarrV3ZeroChunkRegularGridConfigurationJSON", + "ZarrV3ZeroChunkRegularGridJSON", "is_array_metadata_v2", "is_array_metadata_v3", "is_group_metadata_v2", @@ -187,6 +208,8 @@ "read_array_metadata_v3", "read_group_metadata_v3", "read_node_metadata_v3", + "read_repaired_node_metadata_v3", + "repair_node_metadata_v3", "validate_array_metadata_v2", "validate_array_metadata_v3", "validate_group_metadata_v2", diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_repair.py b/packages/zarr-metadata/src/zarr_metadata/model/_repair.py new file mode 100644 index 0000000000..22936388cf --- /dev/null +++ b/packages/zarr-metadata/src/zarr_metadata/model/_repair.py @@ -0,0 +1,233 @@ +"""Repairing v3 metadata that a known writer bug made invalid. + +The readers are strict: a document the spec does not allow has problems, +whoever wrote it. Some writers have written documents the spec does not +allow, in ways that say plainly what they meant. Repairing one is a step +of its own, before a strict read, which a caller asks for by name: +`read_repaired_node_metadata_v3` rather than `read_node_metadata_v3`. + +What is repaired is a closed set, each member of it a writer's bug: the +shape of the JSON it wrote is a TypedDict, which a document must have for +the repair to apply, and the repair makes it correct JSON. Anything else +-- a missing `data_type`, a chunk length of 0 along a dimension that is +not empty -- is left as it is, for the strict read to report. +""" + +from __future__ import annotations + +from collections.abc import Mapping +from dataclasses import dataclass +from typing import Annotated, Literal, TypeAlias, TypedDict, cast + +from annotated_types import Ge + +from zarr_metadata._json import JSON_DEPTH +from zarr_metadata._typed_json import Loc, check +from zarr_metadata.model._group import ZarrV3NodeMetadataReading, read_node_metadata_v3 +from zarr_metadata.v3._registry import CORE_AND_EXTENSIONS, Context +from zarr_metadata.v3.consolidated import ZARR_V3_CONSOLIDATED_METADATA_KEY + +RepairKind: TypeAlias = Literal["zero_chunk_length", "null_consolidated_metadata"] +"""Each writer bug a repair undoes, by name.""" + + +@dataclass(frozen=True, slots=True) +class Repair: + """One change a repair made to a document: where, which bug it undid, and what it did.""" + + loc: Loc + """Where the change is, in the document: the value changed, or the key removed.""" + kind: RepairKind + """The writer bug the change undoes.""" + message: str + """What was written, by which writer, and what it became.""" + + +_Length = Annotated[int, Ge(0)] + + +class ZarrV3ZeroChunkRegularGridConfigurationJSON(TypedDict): + """A `regular` grid's configuration as zarr-python 3.0 and 3.1 wrote it: chunk lengths that may be 0.""" + + chunk_shape: tuple[_Length, ...] + + +class ZarrV3ZeroChunkRegularGridJSON(TypedDict): + """A `regular` chunk grid whose chunk lengths may be 0.""" + + name: Literal["regular"] + configuration: ZarrV3ZeroChunkRegularGridConfigurationJSON + + +class ZarrV3ZeroChunkArrayMetadataJSON(TypedDict): + """The members of an array document the `zero_chunk_length` repair reads. + + zarr-python 3.0 and 3.1 wrote a chunk length of 0 along a dimension of + length 0, which the regular grid refuses: "Chunk sizes must be greater + than zero". A chunk along an empty dimension holds nothing whatever its + length, so the repair writes 1 there, as zarr-python has since + (https://github.com/zarr-developers/zarr-python/pull/4328). + """ + + node_type: Literal["array"] + shape: tuple[_Length, ...] + chunk_grid: ZarrV3ZeroChunkRegularGridJSON + + +class ZarrV3NullConsolidatedGroupMetadataJSON(TypedDict): + """The members of a group document the `null_consolidated_metadata` repair reads. + + zarr-python 3.0.x wrote `"consolidated_metadata": null` on a group it had + not consolidated, which the convention does not allow: the member is + an object, or absent. The repair removes it, which is what the writer + meant. + """ + + node_type: Literal["group"] + consolidated_metadata: None + + +def repair_node_metadata_v3(value: object) -> tuple[object, tuple[Repair, ...]]: + """`value`, a v3 `zarr.json`, with each known writer bug in it undone, and what was changed. + + Each document consolidated metadata holds is repaired too. What no + repair applies to is left as it is, and `value` is not changed: a + repaired document is a new one, sharing what it did not change with + `value`. Repairing a document with none of the bugs gives it back, + and no repairs. + """ + return _repaired(value, ()) + + +def _repaired(value: object, at: Loc) -> tuple[object, tuple[Repair, ...]]: + if not isinstance(value, Mapping) or len(at) >= JSON_DEPTH: + # Not a document, or past the levels a reader walks, which the + # strict read reports. + return cast("object", value), () + original = cast("Mapping[str, object]", value) + repairs: list[Repair] = [] + document = _zero_chunk_length(original, at, repairs) + document = _null_consolidated_metadata(document, at, repairs) + document = _consolidated(document, at, repairs) + return (original if len(repairs) == 0 else dict(document)), tuple(repairs) + + +def _members(document: Mapping[str, object], keys: tuple[str, ...]) -> dict[str, object]: + """The members of `document` a repair reads: only those, so the rest is not walked.""" + return {key: document[key] for key in keys if key in document} + + +def _zero_chunk_length( + document: Mapping[str, object], at: Loc, repairs: list[Repair] +) -> Mapping[str, object]: + array, problems = check( + _members(document, ("node_type", "shape", "chunk_grid")), ZarrV3ZeroChunkArrayMetadataJSON + ) + if array is None or len(problems) != 0: + return document + shape = array["shape"] + chunk_shape = array["chunk_grid"]["configuration"]["chunk_shape"] + zeros = [axis for axis, length in enumerate(chunk_shape) if length == 0] + if len(zeros) == 0 or len(chunk_shape) != len(shape) or any(shape[a] != 0 for a in zeros): + # Nothing to repair, or a 0 that is no writer's bug: the strict + # read reports it. + return document + repairs.extend( + Repair( + (*at, "chunk_grid", "configuration", "chunk_shape", axis), + "zero_chunk_length", + "a chunk length of 0 along a dimension of length 0, as zarr-python 3.0 and 3.1 " + "wrote it, written as 1", + ) + for axis in zeros + ) + grid = cast("Mapping[str, object]", document["chunk_grid"]) + configuration = cast("Mapping[str, object]", grid["configuration"]) + repaired = tuple(max(length, 1) for length in chunk_shape) + return { + **document, + "chunk_grid": {**grid, "configuration": {**configuration, "chunk_shape": repaired}}, + } + + +def _null_consolidated_metadata( + document: Mapping[str, object], at: Loc, repairs: list[Repair] +) -> Mapping[str, object]: + group, problems = check( + _members(document, ("node_type", ZARR_V3_CONSOLIDATED_METADATA_KEY)), + ZarrV3NullConsolidatedGroupMetadataJSON, + ) + if group is None or len(problems) != 0: + return document + repairs.append( + Repair( + (*at, ZARR_V3_CONSOLIDATED_METADATA_KEY), + "null_consolidated_metadata", + "a consolidated_metadata of null, as zarr-python 3.0.x wrote it, removed", + ) + ) + return {key: item for key, item in document.items() if key != ZARR_V3_CONSOLIDATED_METADATA_KEY} + + +def _consolidated( + document: Mapping[str, object], at: Loc, repairs: list[Repair] +) -> Mapping[str, object]: + """`document` with each document its consolidated metadata holds repaired, where it holds any.""" + member = document.get(ZARR_V3_CONSOLIDATED_METADATA_KEY) + if not isinstance(member, Mapping): + return document + envelope = cast("Mapping[str, object]", member) + entries = envelope.get("metadata") + if not isinstance(entries, Mapping): + return document + held: dict[object, object] = {} + found = len(repairs) + for key, entry in cast("Mapping[object, object]", entries).items(): + if isinstance(key, str): + entry, inside = _repaired( + entry, (*at, ZARR_V3_CONSOLIDATED_METADATA_KEY, "metadata", key) + ) + repairs.extend(inside) + held[key] = entry + if len(repairs) == found: + return document + return {**document, ZARR_V3_CONSOLIDATED_METADATA_KEY: {**envelope, "metadata": held}} + + +@dataclass(frozen=True, slots=True) +class ZarrV3RepairedNodeMetadataReading: + """A v3 `zarr.json` read after its known writer bugs were undone: the strict reading of the repaired document, and the repairs.""" + + reading: ZarrV3NodeMetadataReading + """The repaired document, as `read_node_metadata_v3` reads it: its problems are the repaired document's, and its model when there are none.""" + repairs: tuple[Repair, ...] + """What was changed to make the document that was read.""" + + +def read_repaired_node_metadata_v3( + value: object, *, context: Context = CORE_AND_EXTENSIONS +) -> ZarrV3RepairedNodeMetadataReading: + """`value`, a v3 `zarr.json`, read in `context` as `read_node_metadata_v3` reads it, once `repair_node_metadata_v3` has undone each known writer bug in it. + + For a reader of stores other writers made, which asks for repairs by + calling this rather than `read_node_metadata_v3`. Whatever no repair + applies to is read as it is, and reported as `read_node_metadata_v3` + reports it. + """ + repaired, repairs = repair_node_metadata_v3(value) + return ZarrV3RepairedNodeMetadataReading( + read_node_metadata_v3(repaired, context=context), repairs + ) + + +__all__ = [ + "Repair", + "RepairKind", + "ZarrV3NullConsolidatedGroupMetadataJSON", + "ZarrV3RepairedNodeMetadataReading", + "ZarrV3ZeroChunkArrayMetadataJSON", + "ZarrV3ZeroChunkRegularGridConfigurationJSON", + "ZarrV3ZeroChunkRegularGridJSON", + "read_repaired_node_metadata_v3", + "repair_node_metadata_v3", +] diff --git a/packages/zarr-metadata/tests/model/test_repair.py b/packages/zarr-metadata/tests/model/test_repair.py new file mode 100644 index 0000000000..7eeff9ad86 --- /dev/null +++ b/packages/zarr-metadata/tests/model/test_repair.py @@ -0,0 +1,131 @@ +"""Repairing v3 metadata a known writer bug made invalid, before a strict read.""" + +from __future__ import annotations + +import copy +from typing import Any + +import pytest + +from zarr_metadata.model import ( + ZarrV3ArrayMetadata, + ZarrV3GroupMetadata, + read_node_metadata_v3, + read_repaired_node_metadata_v3, + repair_node_metadata_v3, +) + +ARRAY: dict[str, Any] = dict(ZarrV3ArrayMetadata.create_default(shape=(0, 3)).to_json()) +GROUP: dict[str, Any] = {"zarr_format": 3, "node_type": "group"} + + +def _grid(*chunk_shape: int) -> dict[str, Any]: + return {"name": "regular", "configuration": {"chunk_shape": chunk_shape}} + + +def _consolidated(**metadata: object) -> dict[str, Any]: + envelope = {"kind": "inline", "must_understand": False, "metadata": metadata} + return {**GROUP, "consolidated_metadata": envelope} + + +ZERO_CHUNK = {**ARRAY, "chunk_grid": _grid(0, 3)} +NULL_CONSOLIDATED = {**GROUP, "consolidated_metadata": None} + + +@pytest.mark.parametrize( + ("value", "repaired", "repairs"), + [ + ( + ZERO_CHUNK, + {**ARRAY, "chunk_grid": _grid(1, 3)}, + [(("chunk_grid", "configuration", "chunk_shape", 0), "zero_chunk_length")], + ), + ( + {**ARRAY, "shape": (0, 0), "chunk_grid": _grid(0, 0)}, + {**ARRAY, "shape": (0, 0), "chunk_grid": _grid(1, 1)}, + [ + (("chunk_grid", "configuration", "chunk_shape", 0), "zero_chunk_length"), + (("chunk_grid", "configuration", "chunk_shape", 1), "zero_chunk_length"), + ], + ), + (NULL_CONSOLIDATED, GROUP, [(("consolidated_metadata",), "null_consolidated_metadata")]), + ( + _consolidated(a=ZERO_CHUNK, b=NULL_CONSOLIDATED), + _consolidated(a={**ARRAY, "chunk_grid": _grid(1, 3)}, b=GROUP), + [ + ( + ("consolidated_metadata", "metadata", "a", "chunk_grid", "configuration") + + ("chunk_shape", 0), + "zero_chunk_length", + ), + ( + ("consolidated_metadata", "metadata", "b", "consolidated_metadata"), + "null_consolidated_metadata", + ), + ], + ), + # No writer's bug: left for the strict read to judge. + (ARRAY, ARRAY, []), + ({**ARRAY, "chunk_grid": _grid(3, 0)}, {**ARRAY, "chunk_grid": _grid(3, 0)}, []), + ({**ARRAY, "chunk_grid": _grid(0)}, {**ARRAY, "chunk_grid": _grid(0)}, []), + ( + {key: item for key, item in ZERO_CHUNK.items() if key != "data_type"}, + {key: item for key, item in ARRAY.items() if key != "data_type"} + | {"chunk_grid": _grid(1, 3)}, + [(("chunk_grid", "configuration", "chunk_shape", 0), "zero_chunk_length")], + ), + ({**GROUP, "consolidated_metadata": 0}, {**GROUP, "consolidated_metadata": 0}, []), + ([ZERO_CHUNK], [ZERO_CHUNK], []), + ], + ids=[ + "zero-chunk", + "zero-chunks", + "null-consolidated", + "inside-consolidated", + "valid", + "zero-chunk-on-a-full-dimension", + "zero-chunk-of-another-rank", + "zero-chunk-without-a-data-type", + "consolidated-of-another-type", + "not-a-document", + ], +) +def test_repair_undoes_each_known_writer_bug( + value: object, repaired: object, repairs: list[tuple[tuple[str | int, ...], str]] +) -> None: + """Each known writer bug in a document, its consolidated documents' too, is undone and said where; anything else is left as it is, for the strict read to report, and the document handed in is not changed.""" + before = copy.deepcopy(value) + document, made = repair_node_metadata_v3(value) + assert document == repaired + assert [(repair.loc, repair.kind) for repair in made] == repairs + assert value == before + if len(repairs) == 0: + assert document is value + + +@pytest.mark.parametrize( + ("value", "model"), + [ + (ZERO_CHUNK, ZarrV3ArrayMetadata.from_json({**ARRAY, "chunk_grid": _grid(1, 3)})), + (NULL_CONSOLIDATED, ZarrV3GroupMetadata.from_json(GROUP)), + (ARRAY, ZarrV3ArrayMetadata.from_json(ARRAY)), + ], + ids=["zero-chunk", "null-consolidated", "valid"], +) +def test_read_repaired_reads_what_the_strict_read_refuses(value: object, model: object) -> None: + """A document a writer bug made invalid is refused by `read_node_metadata_v3` and read by `read_repaired_node_metadata_v3`, as the strict read reads the repaired document; a valid one reads the same either way.""" + repaired = read_repaired_node_metadata_v3(value) + assert repaired.reading.problems == () + assert repaired.reading.metadata == model + assert (len(read_node_metadata_v3(value).problems) == 0) is (len(repaired.repairs) == 0) + + +def test_read_repaired_reports_what_no_repair_applies_to() -> None: + """A problem no repair undoes is reported as the strict read reports it, beside the repairs that were made.""" + value = {key: item for key, item in ZERO_CHUNK.items() if key != "data_type"} + repaired = read_repaired_node_metadata_v3(value) + assert [repair.kind for repair in repaired.repairs] == ["zero_chunk_length"] + assert [(problem.loc, problem.kind) for problem in repaired.reading.problems] == [ + (("data_type",), "missing_key") + ] + assert repaired.reading.metadata is None diff --git a/packages/zarr-metadata/tests/test_public_api.py b/packages/zarr-metadata/tests/test_public_api.py index 03e98d268f..b35c518dd7 100644 --- a/packages/zarr-metadata/tests/test_public_api.py +++ b/packages/zarr-metadata/tests/test_public_api.py @@ -315,6 +315,10 @@ def test_all_is_grouped_and_unique() -> None: "NumpyTimedelta64", "ProblemKind", "RectilinearDimSpec", + # What a repair of a writer's bug changed, as `ValidationProblem` and + # `ProblemKind` are what a read found. + "Repair", + "RepairKind", "ScalarMap", "ScalarMapEntry", "ShardingIndexLocation", From 0158308cf70ae6479f109813ca8e581438099784 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Tue, 6 Oct 2026 10:20:54 +0200 Subject: [PATCH 21/94] docs(zarr-metadata): add a changelog fragment for the repair read Assisted-by: ClaudeCode:claude-opus-5-5 --- packages/zarr-metadata/changes/391.feature.md | 1 + 1 file changed, 1 insertion(+) create mode 100644 packages/zarr-metadata/changes/391.feature.md diff --git a/packages/zarr-metadata/changes/391.feature.md b/packages/zarr-metadata/changes/391.feature.md new file mode 100644 index 0000000000..c079c8edde --- /dev/null +++ b/packages/zarr-metadata/changes/391.feature.md @@ -0,0 +1 @@ +`read_repaired_node_metadata_v3` reads a v3 `zarr.json` after undoing known writer bugs, and reports each change it made. It covers a chunk length of 0 along an empty dimension, as zarr-python 3.0 and 3.1 wrote it, and `consolidated_metadata: null`, as zarr-python 3.0.x wrote it. `repair_node_metadata_v3` does the repair alone, and the strict readers still refuse these documents. From 9e0bb82658bc9961d635abacef47a46bee0133e7 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Tue, 6 Oct 2026 18:25:52 +0200 Subject: [PATCH 22/94] docs(zarr-metadata): rewrite the 391 changelog fragments in plain English Assisted-by: ClaudeCode:claude-fable-5-1 --- packages/zarr-metadata/changes/391.feature.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/packages/zarr-metadata/changes/391.feature.md b/packages/zarr-metadata/changes/391.feature.md index c079c8edde..12e826e910 100644 --- a/packages/zarr-metadata/changes/391.feature.md +++ b/packages/zarr-metadata/changes/391.feature.md @@ -1 +1 @@ -`read_repaired_node_metadata_v3` reads a v3 `zarr.json` after undoing known writer bugs, and reports each change it made. It covers a chunk length of 0 along an empty dimension, as zarr-python 3.0 and 3.1 wrote it, and `consolidated_metadata: null`, as zarr-python 3.0.x wrote it. `repair_node_metadata_v3` does the repair alone, and the strict readers still refuse these documents. +`read_repaired_node_metadata_v3` reads a v3 `zarr.json` after fixing known writer bugs and returns the list of repairs it made. It covers a chunk length of 0 along an empty dimension (zarr-python 3.0 and 3.1) and `consolidated_metadata: null` (zarr-python 3.0.x). `repair_node_metadata_v3` does only the repair; the strict readers still refuse these documents. From 9bf5211a5fdb9f0439e5718708546da2d1e8877a Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Tue, 6 Oct 2026 13:58:40 +0200 Subject: [PATCH 23/94] feat(zarr-metadata): make Context compare and hash by its definitions Two scopes are equal when they hold the same definitions under the same kinds and names, regardless of construction order. Equal scopes hash alike. Before this, `Context` was not hashable at all because its tables are mapping proxies. Assisted-by: ClaudeCode:claude-fable-5-1 --- .../src/zarr_metadata/v3/_registry.py | 18 ++++++++- packages/zarr-metadata/tests/v3/test_scope.py | 37 +++++++++++++++++++ 2 files changed, 54 insertions(+), 1 deletion(-) create mode 100644 packages/zarr-metadata/tests/v3/test_scope.py diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/_registry.py b/packages/zarr-metadata/src/zarr_metadata/v3/_registry.py index d7475cd1c7..a1535f3174 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/_registry.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/_registry.py @@ -65,7 +65,7 @@ """By kind, then by the name each definition is filed under.""" -@dataclass(frozen=True, slots=True) +@dataclass(frozen=True, slots=True, eq=False) class Context: """The definitions in scope while metadata is read. @@ -132,6 +132,22 @@ def __deepcopy__(self, memo: dict[int, object]) -> Context: # A scope never changes, so a copy of it is itself. return self + def __eq__(self, other: object) -> bool: + if not isinstance(other, Context): + return NotImplemented + return self._filed() == other._filed() + + def __hash__(self) -> int: + return hash(self._filed()) + + def _filed(self) -> frozenset[tuple[type[Definition[Any]], str, Definition[Any]]]: + """Every definition in scope with the kind and name it is filed under: what two scopes are compared by.""" + return frozenset( + (kind, name, definition) + for kind, table in self.tables.items() + for name, definition in table.items() + ) + def claimant(self, kind: type[D], name: str) -> D | None: """The definition of `kind` in scope that reads `name`, a name a document writes; None if none does. diff --git a/packages/zarr-metadata/tests/v3/test_scope.py b/packages/zarr-metadata/tests/v3/test_scope.py new file mode 100644 index 0000000000..9811ca40b7 --- /dev/null +++ b/packages/zarr-metadata/tests/v3/test_scope.py @@ -0,0 +1,37 @@ +"""The algebra of scopes: value semantics for `Context`, claims, refinement, disagreements and joins.""" + +from __future__ import annotations + +import pickle + +import pytest + +from zarr_metadata.v3.codec.bytes import BYTES_CODEC +from zarr_metadata.v3.codec.gzip import GZIP_CODEC +from zarr_metadata.v3.codec.zstd import ZSTD_CODEC +from zarr_metadata.v3.definition import CORE, CORE_AND_EXTENSIONS, Context + + +@pytest.mark.parametrize( + ("left", "right", "equal"), + [ + (Context.of(GZIP_CODEC, BYTES_CODEC), Context.of(BYTES_CODEC, GZIP_CODEC), True), + (CORE, pickle.loads(pickle.dumps(CORE)), True), + (CORE, CORE_AND_EXTENSIONS, False), + (Context.of(GZIP_CODEC), Context.of(GZIP_CODEC, ZSTD_CODEC), False), + (Context.of(), Context.of(), True), + ], + ids=["order", "pickle", "core-vs-extensions", "subset", "empty"], +) +def test_a_scope_is_equal_to_another_by_the_definitions_it_files( + left: Context, right: Context, equal: bool +) -> None: + """Two scopes are one when they file the same definitions under the same names, however they were built; equal scopes hash alike.""" + assert (left == right) is equal + if equal: + assert hash(left) == hash(right) + + +def test_a_scope_is_a_set_member() -> None: + """A scope hashes, so it can key a dict or sit in a set, which it could not when its tables were mapping proxies.""" + assert len({CORE, CORE_AND_EXTENSIONS, pickle.loads(pickle.dumps(CORE))}) == 2 From 5df30473ae246d35a46fca13a5e1d7d00a87eea6 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Tue, 6 Oct 2026 13:59:37 +0200 Subject: [PATCH 24/94] feat(zarr-metadata): add the Claims type, Conflict, and ScopeConflictError `Claims` maps `(kind, filed name)` to the definition that read it, or None when nothing claimed it. `Conflict` records one disagreement: the key, what was claimed, what was found, and where. `ScopeConflictError` carries every conflict, as `MetadataValidationError` carries every problem. Assisted-by: ClaudeCode:claude-fable-5-1 --- .../src/zarr_metadata/v3/_scope.py | 75 +++++++++++++++++++ .../src/zarr_metadata/v3/definition.py | 5 ++ .../zarr-metadata/tests/test_public_api.py | 6 ++ packages/zarr-metadata/tests/v3/test_scope.py | 24 +++++- 4 files changed, 109 insertions(+), 1 deletion(-) create mode 100644 packages/zarr-metadata/src/zarr_metadata/v3/_scope.py diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/_scope.py b/packages/zarr-metadata/src/zarr_metadata/v3/_scope.py new file mode 100644 index 0000000000..256296476d --- /dev/null +++ b/packages/zarr-metadata/src/zarr_metadata/v3/_scope.py @@ -0,0 +1,75 @@ +"""The algebra of scopes: what a reading claims of each name, the order one reading refines another in, and where two scopes disagree. + +A model is a document read in a scope. What the document means depends +only on what the scope said of the names it writes -- its *claims* -- +not on everything the scope holds. Two readings of one document are +ordered by information: a name nothing claimed, read by a definition, +gains meaning and loses none; a name read by one definition and then +another is a conflict. `Context.disagreements` and `Context.joined` are +built on these. +""" + +from __future__ import annotations + +from dataclasses import dataclass +from typing import TYPE_CHECKING, Any, TypeAlias + +from zarr_metadata.v3._definition import ( + ChunkGridDefinition, + ChunkKeyEncodingDefinition, + CodecDefinition, + DataTypeDefinition, + Definition, + StorageTransformerDefinition, +) + +if TYPE_CHECKING: + from collections.abc import Mapping, Sequence + + from zarr_metadata._typed_json import Loc + +ClaimKey: TypeAlias = tuple[type[Definition[Any]], str] +"""A kind and the name a definition is filed under: what a scope answers `claimant` for.""" + +Claims: TypeAlias = "Mapping[ClaimKey, Definition[Any] | None]" +"""What a reading claims of each name a document writes: the definition that read it, or None where nothing claimed it.""" + +_KIND_NAMES: dict[type[Definition[Any]], str] = { + CodecDefinition: "codec", + DataTypeDefinition: "data type", + ChunkGridDefinition: "chunk grid", + ChunkKeyEncodingDefinition: "chunk key encoding", + StorageTransformerDefinition: "storage transformer", +} + + +@dataclass(frozen=True, slots=True) +class Conflict: + """One place two readings of a name disagree: the key, what one claimed, what the other found, and where in a document when known.""" + + key: ClaimKey + claimed: Definition[Any] | None + found: Definition[Any] | None + loc: Loc | None = None + + def __str__(self) -> str: + kind, name = self.key + where = "" if self.loc is None else f" at {self.loc!r}" + return ( + f"{_KIND_NAMES[kind]} {name!r}{where}: claimed {self.claimed!r}, found {self.found!r}" + ) + + +class ScopeConflictError(ValueError): + """Raised where two scopes, or a scope and a reading, give one name two meanings, or one would lose a meaning the other has. + + Carries every conflict in `.conflicts`, as `MetadataValidationError` + carries every problem. + """ + + def __init__(self, conflicts: Sequence[Conflict]) -> None: + self.conflicts: tuple[Conflict, ...] = tuple(conflicts) + super().__init__("; ".join(str(conflict) for conflict in self.conflicts)) + + +__all__ = ["ClaimKey", "Claims", "Conflict", "ScopeConflictError"] diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/definition.py b/packages/zarr-metadata/src/zarr_metadata/v3/definition.py index 5f112e6b18..2e81d64438 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/definition.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/definition.py @@ -353,6 +353,7 @@ def acme_lz4_rules( ) from zarr_metadata.v3._pipeline import Stage, read_pipeline from zarr_metadata.v3._registry import CORE, CORE_AND_EXTENSIONS, Context +from zarr_metadata.v3._scope import ClaimKey, Claims, Conflict, ScopeConflictError __all__ = [ "CORE", @@ -362,10 +363,13 @@ def acme_lz4_rules( "ChunkGridField", "ChunkKeyEncodingDefinition", "ChunkKeyEncodingField", + "ClaimKey", + "Claims", "CodecDefinition", "CodecField", "CodecKind", "CodecSize", + "Conflict", "Context", "DataTypeDefinition", "DataTypeField", @@ -380,6 +384,7 @@ def acme_lz4_rules( "Read", "Refused", "Resolved", + "ScopeConflictError", "Stage", "StaticCodecField", "StorageClass", diff --git a/packages/zarr-metadata/tests/test_public_api.py b/packages/zarr-metadata/tests/test_public_api.py index b35c518dd7..45649f28a1 100644 --- a/packages/zarr-metadata/tests/test_public_api.py +++ b/packages/zarr-metadata/tests/test_public_api.py @@ -284,6 +284,12 @@ def test_all_is_grouped_and_unique() -> None: "BloscCName", "BloscShuffle", "Chunk", + # The algebra of scopes: what a reading claims, and where two + # scopes disagree. + "ClaimKey", + "Claims", + "Conflict", + "ScopeConflictError", "CodecKind", "CodecSize", "Context", diff --git a/packages/zarr-metadata/tests/v3/test_scope.py b/packages/zarr-metadata/tests/v3/test_scope.py index 9811ca40b7..58fc9d014e 100644 --- a/packages/zarr-metadata/tests/v3/test_scope.py +++ b/packages/zarr-metadata/tests/v3/test_scope.py @@ -9,7 +9,14 @@ from zarr_metadata.v3.codec.bytes import BYTES_CODEC from zarr_metadata.v3.codec.gzip import GZIP_CODEC from zarr_metadata.v3.codec.zstd import ZSTD_CODEC -from zarr_metadata.v3.definition import CORE, CORE_AND_EXTENSIONS, Context +from zarr_metadata.v3.definition import ( + CORE, + CORE_AND_EXTENSIONS, + CodecDefinition, + Conflict, + Context, + ScopeConflictError, +) @pytest.mark.parametrize( @@ -35,3 +42,18 @@ def test_a_scope_is_equal_to_another_by_the_definitions_it_files( def test_a_scope_is_a_set_member() -> None: """A scope hashes, so it can key a dict or sit in a set, which it could not when its tables were mapping proxies.""" assert len({CORE, CORE_AND_EXTENSIONS, pickle.loads(pickle.dumps(CORE))}) == 2 + + +def test_error_a_scope_conflict_says_each_disagreement() -> None: + """A scope conflict lists each `(kind, name)` with what was claimed and what was found, and where, so a caller can see every disagreement at once.""" + error = ScopeConflictError( + ( + Conflict((CodecDefinition, "bytes"), BYTES_CODEC, None, ("codecs", 0)), + Conflict((CodecDefinition, "gzip"), None, GZIP_CODEC), + ) + ) + assert error.conflicts[0].key == (CodecDefinition, "bytes") + assert str(error) == ( + "codec 'bytes' at ('codecs', 0): claimed CodecDefinition(name='bytes'), found None; " + "codec 'gzip': claimed None, found CodecDefinition(name='gzip')" + ) From 0f16749e3df80eba8227e0dd5be71cbcda79cf4f Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Tue, 6 Oct 2026 14:01:09 +0200 Subject: [PATCH 25/94] feat(zarr-metadata): add claims_of, which lists what a reading claimed for each name `claims_of` takes the fields of one reading, nested fields included, and returns the definition that read each name, keyed as the scope files it (raw bits under `r*`). A name claimed two different ways raises `ScopeConflictError`. Assisted-by: ClaudeCode:claude-fable-5-1 --- .../src/zarr_metadata/v3/_scope.py | 39 ++++++++- .../src/zarr_metadata/v3/definition.py | 11 ++- packages/zarr-metadata/tests/v3/test_scope.py | 81 +++++++++++++++++++ 3 files changed, 128 insertions(+), 3 deletions(-) diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/_scope.py b/packages/zarr-metadata/src/zarr_metadata/v3/_scope.py index 256296476d..328219c8c1 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/_scope.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/_scope.py @@ -21,12 +21,14 @@ DataTypeDefinition, Definition, StorageTransformerDefinition, + spelled, ) if TYPE_CHECKING: - from collections.abc import Mapping, Sequence + from collections.abc import Iterable, Mapping, Sequence from zarr_metadata._typed_json import Loc + from zarr_metadata.v3._definition import Resolved ClaimKey: TypeAlias = tuple[type[Definition[Any]], str] """A kind and the name a definition is filed under: what a scope answers `claimant` for.""" @@ -72,4 +74,37 @@ def __init__(self, conflicts: Sequence[Conflict]) -> None: super().__init__("; ".join(str(conflict) for conflict in self.conflicts)) -__all__ = ["ClaimKey", "Claims", "Conflict", "ScopeConflictError"] +def claim_key(field: Resolved[Any]) -> ClaimKey | None: + """The key `field` is claimed under: its kind and the name its definition is filed under, `r*` for `r16`; None for a field that names nothing.""" + if field.name is None: + return None + filed, _ = spelled(field.read_as, field.name) + return None if filed is None else (field.read_as, filed) + + +def claims_of( + fields: Iterable[tuple[Loc, Resolved[Any]]], +) -> dict[ClaimKey, Definition[Any] | None]: + """What `fields`, each with where it sits, claim of each name: the definition that read it, None where nothing claimed it. + + `fields` are one reading's, as `fields_of` or a reading's `fields()` + gives them. A name claimed two ways among them is a + `ScopeConflictError`: no one scope read them. + """ + claims: dict[ClaimKey, Definition[Any] | None] = {} + conflicts: list[Conflict] = [] + for loc, field in fields: + key = claim_key(field) + if key is None: + continue + definition = field.definition + if key in claims and claims[key] != definition: + conflicts.append(Conflict(key, claims[key], definition, loc)) + continue + claims[key] = definition + if len(conflicts) != 0: + raise ScopeConflictError(conflicts) + return claims + + +__all__ = ["ClaimKey", "Claims", "Conflict", "ScopeConflictError", "claim_key", "claims_of"] diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/definition.py b/packages/zarr-metadata/src/zarr_metadata/v3/definition.py index 2e81d64438..0554238f41 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/definition.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/definition.py @@ -353,7 +353,14 @@ def acme_lz4_rules( ) from zarr_metadata.v3._pipeline import Stage, read_pipeline from zarr_metadata.v3._registry import CORE, CORE_AND_EXTENSIONS, Context -from zarr_metadata.v3._scope import ClaimKey, Claims, Conflict, ScopeConflictError +from zarr_metadata.v3._scope import ( + ClaimKey, + Claims, + Conflict, + ScopeConflictError, + claim_key, + claims_of, +) __all__ = [ "CORE", @@ -398,6 +405,8 @@ def acme_lz4_rules( "canonicalize", "check", "chunk_grid_lengths", + "claim_key", + "claims_of", "configuration_of", "field_json_schema", "fields_of", diff --git a/packages/zarr-metadata/tests/v3/test_scope.py b/packages/zarr-metadata/tests/v3/test_scope.py index 58fc9d014e..f3808e6495 100644 --- a/packages/zarr-metadata/tests/v3/test_scope.py +++ b/packages/zarr-metadata/tests/v3/test_scope.py @@ -3,21 +3,53 @@ from __future__ import annotations import pickle +from typing import Any import pytest from zarr_metadata.v3.codec.bytes import BYTES_CODEC +from zarr_metadata.v3.codec.crc32c import CRC32C_CODEC, Empty from zarr_metadata.v3.codec.gzip import GZIP_CODEC +from zarr_metadata.v3.codec.sharding_indexed import SHARDING_INDEXED_CODEC from zarr_metadata.v3.codec.zstd import ZSTD_CODEC +from zarr_metadata.v3.data_type.raw import RAW_BYTES_DATA_TYPE from zarr_metadata.v3.definition import ( CORE, CORE_AND_EXTENSIONS, CodecDefinition, Conflict, Context, + DataTypeDefinition, + Definition, + Refused, + Resolved, ScopeConflictError, + claims_of, + fields_of, + resolve, ) +SHARD = { + "name": "sharding_indexed", + "configuration": { + "chunk_shape": [1], + "codecs": ["bytes"], + "index_codecs": [{"name": "bytes", "configuration": {"endian": "little"}}, "crc32c"], + }, +} + +MY_GZIP = CodecDefinition(name="gzip", configuration=Empty, kind="bytes_bytes", size="dynamic") +"""A private reading of the name `gzip`: another definition under one name, which takes no configuration. + +Definitions compare by what they hold, so one rebuilt from the core +TypedDict with the core rules would be the core definition; this one +reads `gzip` otherwise. +""" + + +def _read(data: object, kind: type[Definition[Any]], scope: Context) -> Resolved[Any]: + return resolve(data, kind, scope)[0] + @pytest.mark.parametrize( ("left", "right", "equal"), @@ -57,3 +89,52 @@ def test_error_a_scope_conflict_says_each_disagreement() -> None: "codec 'bytes' at ('codecs', 0): claimed CodecDefinition(name='bytes'), found None; " "codec 'gzip': claimed None, found CodecDefinition(name='gzip')" ) + + +@pytest.mark.parametrize( + ("field", "claims"), + [ + (_read("gzip", CodecDefinition, CORE), {(CodecDefinition, "gzip"): GZIP_CODEC}), + (_read("zstd", CodecDefinition, CORE), {(CodecDefinition, "zstd"): None}), + ( + _read("r16", DataTypeDefinition, CORE), + {(DataTypeDefinition, "r*"): RAW_BYTES_DATA_TYPE}, + ), + ( + _read(SHARD, CodecDefinition, CORE), + { + (CodecDefinition, "sharding_indexed"): SHARDING_INDEXED_CODEC, + (CodecDefinition, "bytes"): BYTES_CODEC, + (CodecDefinition, "crc32c"): CRC32C_CODEC, + }, + ), + ( + _read({"name": "gzip", "configuration": {"level": 12}}, CodecDefinition, CORE), + {(CodecDefinition, "gzip"): GZIP_CODEC}, + ), + (Refused(json=3, name=None, read_as=CodecDefinition), {}), + ], + ids=["read", "unclaimed", "raw-bits", "nested", "refused-claimed", "refused-nameless"], +) +def test_claims_of_says_what_a_reading_claimed_of_each_name( + field: Resolved[Any], claims: dict[object, object] +) -> None: + """A reading's claims name the definition that read each name the field and the fields it holds write, keyed as the scope files it -- raw bits under `r*` -- and None where nothing claimed one; a field refused by a definition still claims it, and one that names nothing claims nothing.""" + assert claims_of(fields_of(field)) == claims + + +def test_error_claims_of_refuses_one_name_read_two_ways() -> None: + """Fields read in two scopes that give one name two definitions have no single set of claims: a `ScopeConflictError` naming the key.""" + fields = [ + *fields_of(_read("gzip", CodecDefinition, CORE), ("codecs", 0)), + *fields_of(_read("gzip", CodecDefinition, Context.of(MY_GZIP)), ("codecs", 1)), + ] + with pytest.raises(ScopeConflictError) as raised: + claims_of(fields) + (conflict,) = raised.value.conflicts + assert (conflict.key, conflict.claimed, conflict.found, conflict.loc) == ( + (CodecDefinition, "gzip"), + GZIP_CODEC, + MY_GZIP, + ("codecs", 1), + ) From 8723c6e51229fc55187fcbf8a269b5fbdc0a85c7 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Tue, 6 Oct 2026 14:02:42 +0200 Subject: [PATCH 26/94] feat(zarr-metadata): add refines, an information order on two readings of a field `refines(field, other)` is True when `field` holds everything `other` holds: the same definition and canonical configuration where both were read, and a read field over an unclaimed one with the same name and JSON (a gain). The reverse direction is a loss, two different definitions are a conflict, and both return False. Two fields that refine each other are equal. Assisted-by: ClaudeCode:claude-fable-5-1 --- .../src/zarr_metadata/v3/_definition.py | 6 ++ .../src/zarr_metadata/v3/_scope.py | 37 +++++++- .../src/zarr_metadata/v3/definition.py | 2 + packages/zarr-metadata/tests/v3/test_scope.py | 92 +++++++++++++++++++ 4 files changed, 136 insertions(+), 1 deletion(-) diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py b/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py index 3f76daa781..d4e98a2650 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py @@ -1191,6 +1191,11 @@ def field_key(field: Resolved[Any]) -> tuple[object, ...]: ) +def own_key(field: Read[Any]) -> tuple[object, ...]: + """What `field_key` compares of a read field without the fields it holds: the definition and the canonical spelling of its own members.""" + return field_key(field)[:3] + + def _nested_key(nested: Nested) -> tuple[tuple[Loc, tuple[object, ...]], ...]: """The fields a configuration holds, each by its key, where it sits.""" return tuple((loc, field_key(inner)) for loc, inner in nested.items()) @@ -1809,6 +1814,7 @@ def _replaced(value: JSONValue, path: Loc, new: JSONValue) -> JSONValue: "named_configuration", "no_pipelines", "no_rules", + "own_key", "read_only", "resolve", "ruled", diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/_scope.py b/packages/zarr-metadata/src/zarr_metadata/v3/_scope.py index 328219c8c1..e75eb8d1ad 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/_scope.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/_scope.py @@ -20,7 +20,11 @@ CodecDefinition, DataTypeDefinition, Definition, + Refused, StorageTransformerDefinition, + Unclaimed, + field_key, + own_key, spelled, ) @@ -107,4 +111,35 @@ def claims_of( return claims -__all__ = ["ClaimKey", "Claims", "Conflict", "ScopeConflictError", "claim_key", "claims_of"] +def refines(field: Resolved[Any], other: Resolved[Any]) -> bool: + """Whether `field` holds everything `other` holds: reads the same where both read, and reads what `other` left unclaimed. + + The order one reading of a document refines another in. A name nothing + claimed, read by a definition, is a gain; the reverse is a loss; one + name read by two definitions is a conflict; a refused field is in the + order with nothing. Two fields that refine each other are equal. + """ + if isinstance(field, Refused) or isinstance(other, Refused): + return False + if isinstance(other, Unclaimed): + if isinstance(field, Unclaimed): + return field_key(field) == field_key(other) + return claim_key(field) == claim_key(other) and field.json == other.json + if isinstance(field, Unclaimed): + return False + if field.definition != other.definition or own_key(field) != own_key(other): + return False + if set(field.nested) != set(other.nested): + return False + return all(refines(field.nested[loc], other.nested[loc]) for loc in field.nested) + + +__all__ = [ + "ClaimKey", + "Claims", + "Conflict", + "ScopeConflictError", + "claim_key", + "claims_of", + "refines", +] diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/definition.py b/packages/zarr-metadata/src/zarr_metadata/v3/definition.py index 0554238f41..d899fc5a1a 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/definition.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/definition.py @@ -360,6 +360,7 @@ def acme_lz4_rules( ScopeConflictError, claim_key, claims_of, + refines, ) __all__ = [ @@ -412,6 +413,7 @@ def acme_lz4_rules( "fields_of", "fill_value_problems", "read_pipeline", + "refines", "resolve", "shown", "storage_of", diff --git a/packages/zarr-metadata/tests/v3/test_scope.py b/packages/zarr-metadata/tests/v3/test_scope.py index f3808e6495..b7d8c2858b 100644 --- a/packages/zarr-metadata/tests/v3/test_scope.py +++ b/packages/zarr-metadata/tests/v3/test_scope.py @@ -6,6 +6,8 @@ from typing import Any import pytest +from hypothesis import given +from hypothesis import strategies as st from zarr_metadata.v3.codec.bytes import BYTES_CODEC from zarr_metadata.v3.codec.crc32c import CRC32C_CODEC, Empty @@ -26,6 +28,7 @@ ScopeConflictError, claims_of, fields_of, + refines, resolve, ) @@ -138,3 +141,92 @@ def test_error_claims_of_refuses_one_name_read_two_ways() -> None: MY_GZIP, ("codecs", 1), ) + + +GZIP_FIELD = {"name": "gzip", "configuration": {"level": 5}} +ZSTD_FIELD = {"name": "zstd", "configuration": {"level": 3}} +NESTED_ZSTD = { + "name": "sharding_indexed", + "configuration": { + "chunk_shape": [1], + "codecs": ["bytes", ZSTD_FIELD], + "index_codecs": [{"name": "bytes", "configuration": {"endian": "little"}}, "crc32c"], + }, +} + + +@pytest.mark.parametrize( + ("field", "other", "expected"), + [ + (_read(GZIP_FIELD, CodecDefinition, CORE), _read(GZIP_FIELD, CodecDefinition, CORE), True), + ( + _read({"name": "crc32c"}, CodecDefinition, CORE), + _read("crc32c", CodecDefinition, CORE), + True, + ), + ( + _read(ZSTD_FIELD, CodecDefinition, CORE_AND_EXTENSIONS), + _read(ZSTD_FIELD, CodecDefinition, CORE), + True, + ), + ( + _read(ZSTD_FIELD, CodecDefinition, CORE), + _read(ZSTD_FIELD, CodecDefinition, CORE_AND_EXTENSIONS), + False, + ), + ( + _read(GZIP_FIELD, CodecDefinition, CORE), + _read(GZIP_FIELD, CodecDefinition, Context.of(MY_GZIP)), + False, + ), + ( + _read(NESTED_ZSTD, CodecDefinition, CORE_AND_EXTENSIONS), + _read(NESTED_ZSTD, CodecDefinition, CORE), + True, + ), + ( + _read(NESTED_ZSTD, CodecDefinition, CORE), + _read(NESTED_ZSTD, CodecDefinition, CORE_AND_EXTENSIONS), + False, + ), + ( + _read({"name": "gzip", "configuration": {"level": 12}}, CodecDefinition, CORE), + _read("gzip", CodecDefinition, CORE), + False, + ), + (_read("zstd", CodecDefinition, CORE), _read("zstd", CodecDefinition, CORE), True), + ( + _read({"name": "zstd", "configuration": {"level": 1}}, CodecDefinition, CORE), + _read({"name": "zstd", "configuration": {"level": 2}}, CodecDefinition, CORE), + False, + ), + ], + ids=[ + "same", + "same-spelled-otherwise", + "gain", + "loss", + "conflict", + "nested-gain", + "nested-loss", + "refused", + "both-unclaimed", + "unclaimed-differ", + ], +) +def test_refines_orders_readings_by_information( + field: Resolved[Any], other: Resolved[Any], expected: bool +) -> None: + """`field` refines `other` when it reads the same where both read and gains where `other` left a name unclaimed -- in the fields it holds too; a loss, a conflict, a refused field, or two unclaimed fields written differently do not.""" + assert refines(field, other) is expected + + +@given(st.sampled_from([GZIP_FIELD, "crc32c", {"name": "crc32c"}, ZSTD_FIELD, NESTED_ZSTD])) +def test_refines_is_reflexive_and_mutual_refinement_is_equality(data: object) -> None: + """Every field a scope reads refines itself, and two such fields that refine each other are equal: the order's bottom is equality. A refused field is outside the order, so it is not sampled.""" + for scope in (CORE, CORE_AND_EXTENSIONS): + field = _read(data, CodecDefinition, scope) + assert refines(field, field) + low = _read(data, CodecDefinition, CORE) + high = _read(data, CodecDefinition, CORE_AND_EXTENSIONS) + assert (refines(low, high) and refines(high, low)) is (low == high) From 989367cef9fb029d002d7d57cd6a6dc7d9eb75f1 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Tue, 6 Oct 2026 14:03:35 +0200 Subject: [PATCH 27/94] feat(zarr-metadata): add Context.disagreements and Context.joined `Context.disagreements(claims)` reports, for each claim, whether this scope reads it the same way, would gain a definition, or conflicts with it (a different definition, or none where one was claimed). `Context.joined` returns the union of several scopes and raises `ScopeConflictError` when two of them file different definitions under one name. Assisted-by: ClaudeCode:claude-fable-5-1 --- .../src/zarr_metadata/v3/_registry.py | 33 +++++++++ .../src/zarr_metadata/v3/_scope.py | 35 ++++++++- .../src/zarr_metadata/v3/definition.py | 2 + .../zarr-metadata/tests/test_public_api.py | 1 + packages/zarr-metadata/tests/v3/test_scope.py | 73 +++++++++++++++++++ 5 files changed, 143 insertions(+), 1 deletion(-) diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/_registry.py b/packages/zarr-metadata/src/zarr_metadata/v3/_registry.py index a1535f3174..7af55d3dc3 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/_registry.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/_registry.py @@ -22,6 +22,7 @@ from typing import TYPE_CHECKING, Any, Final, cast from zarr_metadata.v3._definition import KINDS, Definition, as_kind, kind_of, spelled +from zarr_metadata.v3._scope import Conflict, ScopeConflictError, disagreements_of from zarr_metadata.v3.chunk_grid.rectilinear import RECTILINEAR_CHUNK_GRID from zarr_metadata.v3.chunk_grid.regular import REGULAR_CHUNK_GRID from zarr_metadata.v3.chunk_key_encoding.default import DEFAULT_CHUNK_KEY_ENCODING @@ -60,6 +61,7 @@ from collections.abc import Callable from zarr_metadata.v3._definition import D + from zarr_metadata.v3._scope import Claims, Disagreements Tables = Mapping[type[Definition[Any]], Mapping[str, Definition[Any]]] """By kind, then by the name each definition is filed under.""" @@ -148,6 +150,37 @@ def _filed(self) -> frozenset[tuple[type[Definition[Any]], str, Definition[Any]] for name, definition in table.items() ) + def disagreements(self, claims: Claims) -> Disagreements: + """Where this scope reads `claims`, a reading's, otherwise: what it would gain, and what it conflicts with, as `Disagreements` says. + + A claim is keyed by the name its definition is filed under -- raw + bits under `r*` -- so it is looked up as filed, not as a document + writes it. + """ + return disagreements_of(lambda kind, name: self.tables.get(kind, {}).get(name), claims) + + @classmethod + def joined(cls, *contexts: Context) -> Context: + """The least scope that files everything each of `contexts` files: their join. + + `ScopeConflictError` when two of them file different definitions + under one name of one kind; `extended_with` is for taking a name + over on purpose. + """ + filed: dict[tuple[type[Definition[Any]], str], Definition[Any]] = {} + conflicts: list[Conflict] = [] + for context in contexts: + for kind, table in context.tables.items(): + for name, definition in table.items(): + held = filed.get((kind, name)) + if held is not None and held != definition: + conflicts.append(Conflict((kind, name), held, definition)) + continue + filed[kind, name] = definition + if len(conflicts) != 0: + raise ScopeConflictError(conflicts) + return cls.of(*filed.values()) + def claimant(self, kind: type[D], name: str) -> D | None: """The definition of `kind` in scope that reads `name`, a name a document writes; None if none does. diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/_scope.py b/packages/zarr-metadata/src/zarr_metadata/v3/_scope.py index e75eb8d1ad..90255c07f3 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/_scope.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/_scope.py @@ -29,7 +29,7 @@ ) if TYPE_CHECKING: - from collections.abc import Iterable, Mapping, Sequence + from collections.abc import Callable, Iterable, Mapping, Sequence from zarr_metadata._typed_json import Loc from zarr_metadata.v3._definition import Resolved @@ -134,12 +134,45 @@ def refines(field: Resolved[Any], other: Resolved[Any]) -> bool: return all(refines(field.nested[loc], other.nested[loc]) for loc in field.nested) +@dataclass(frozen=True, slots=True) +class Disagreements: + """Where a scope reads a reading's claims otherwise: the names it would gain a meaning for, and those it conflicts with, a lost meaning among them.""" + + gains: tuple[ClaimKey, ...] + conflicts: tuple[Conflict, ...] + + @property + def agrees(self) -> bool: + """Whether the scope reads every claim identically.""" + return len(self.gains) == 0 and len(self.conflicts) == 0 + + +def disagreements_of( + claimant: Callable[[type[Definition[Any]], str], Definition[Any] | None], claims: Claims +) -> Disagreements: + """`Disagreements` between what `claimant`, asked by kind and filed name as a scope's tables answer, gives for each key and what `claims` records.""" + gains: list[ClaimKey] = [] + conflicts: list[Conflict] = [] + for key, claimed in claims.items(): + kind, name = key + found = claimant(kind, name) + if found == claimed: + continue + if claimed is None: + gains.append(key) + else: + conflicts.append(Conflict(key, claimed, found)) + return Disagreements(tuple(gains), tuple(conflicts)) + + __all__ = [ "ClaimKey", "Claims", "Conflict", + "Disagreements", "ScopeConflictError", "claim_key", "claims_of", + "disagreements_of", "refines", ] diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/definition.py b/packages/zarr-metadata/src/zarr_metadata/v3/definition.py index d899fc5a1a..3ded408e30 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/definition.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/definition.py @@ -357,6 +357,7 @@ def acme_lz4_rules( ClaimKey, Claims, Conflict, + Disagreements, ScopeConflictError, claim_key, claims_of, @@ -382,6 +383,7 @@ def acme_lz4_rules( "DataTypeDefinition", "DataTypeField", "Definition", + "Disagreements", "EmptyConfiguration", "JSONValue", "Lengths", diff --git a/packages/zarr-metadata/tests/test_public_api.py b/packages/zarr-metadata/tests/test_public_api.py index 45649f28a1..5a884c8e7e 100644 --- a/packages/zarr-metadata/tests/test_public_api.py +++ b/packages/zarr-metadata/tests/test_public_api.py @@ -289,6 +289,7 @@ def test_all_is_grouped_and_unique() -> None: "ClaimKey", "Claims", "Conflict", + "Disagreements", "ScopeConflictError", "CodecKind", "CodecSize", diff --git a/packages/zarr-metadata/tests/v3/test_scope.py b/packages/zarr-metadata/tests/v3/test_scope.py index b7d8c2858b..9d053c8bc2 100644 --- a/packages/zarr-metadata/tests/v3/test_scope.py +++ b/packages/zarr-metadata/tests/v3/test_scope.py @@ -14,6 +14,7 @@ from zarr_metadata.v3.codec.gzip import GZIP_CODEC from zarr_metadata.v3.codec.sharding_indexed import SHARDING_INDEXED_CODEC from zarr_metadata.v3.codec.zstd import ZSTD_CODEC +from zarr_metadata.v3.data_type.bytes import BYTES_DATA_TYPE from zarr_metadata.v3.data_type.raw import RAW_BYTES_DATA_TYPE from zarr_metadata.v3.definition import ( CORE, @@ -23,6 +24,7 @@ Context, DataTypeDefinition, Definition, + Disagreements, Refused, Resolved, ScopeConflictError, @@ -230,3 +232,74 @@ def test_refines_is_reflexive_and_mutual_refinement_is_equality(data: object) -> low = _read(data, CodecDefinition, CORE) high = _read(data, CodecDefinition, CORE_AND_EXTENSIONS) assert (refines(low, high) and refines(high, low)) is (low == high) + + +@pytest.mark.parametrize( + ("scope", "claims", "gains", "conflicts"), + [ + (CORE, {(CodecDefinition, "gzip"): GZIP_CODEC}, (), ()), + ( + CORE_AND_EXTENSIONS, + {(CodecDefinition, "zstd"): None}, + ((CodecDefinition, "zstd"),), + (), + ), + ( + CORE, + {(CodecDefinition, "zstd"): ZSTD_CODEC}, + (), + (Conflict((CodecDefinition, "zstd"), ZSTD_CODEC, None),), + ), + ( + Context.of(MY_GZIP), + {(CodecDefinition, "gzip"): GZIP_CODEC}, + (), + (Conflict((CodecDefinition, "gzip"), GZIP_CODEC, MY_GZIP),), + ), + (CORE, {(CodecDefinition, "acme.x"): None}, (), ()), + (CORE, {(DataTypeDefinition, "r*"): RAW_BYTES_DATA_TYPE}, (), ()), + ], + ids=["agrees", "gain", "loss", "conflict", "unclaimed-both", "raw-bits"], +) +def test_disagreements_says_where_a_scope_reads_claims_otherwise( + scope: Context, + claims: dict[Any, Any], + gains: tuple[Any, ...], + conflicts: tuple[Conflict, ...], +) -> None: + """A scope agrees with claims it reads identically, gains where it claims what the claims left unclaimed, and conflicts where it reads a name by another definition or by none.""" + found = scope.disagreements(claims) + assert isinstance(found, Disagreements) + assert (found.gains, found.conflicts) == (gains, conflicts) + assert found.agrees is (len(gains) == 0 and len(conflicts) == 0) + + +@pytest.mark.parametrize( + ("scopes", "joined"), + [ + ((CORE, CORE_AND_EXTENSIONS), CORE_AND_EXTENSIONS), + ((Context.of(GZIP_CODEC), Context.of(ZSTD_CODEC)), Context.of(GZIP_CODEC, ZSTD_CODEC)), + ((), Context.of()), + ((CORE, CORE), CORE), + ( + (Context.of(BYTES_CODEC), Context.of(BYTES_DATA_TYPE)), + Context.of(BYTES_CODEC, BYTES_DATA_TYPE), + ), + ], + ids=["subset", "disjoint", "none", "same", "same-name-two-kinds"], +) +def test_joined_is_the_least_scope_above_each(scopes: tuple[Context, ...], joined: Context) -> None: + """The join of scopes files every definition any of them files, once; one kind's name is not another's.""" + assert Context.joined(*scopes) == joined + + +def test_error_joined_refuses_one_name_filed_two_ways() -> None: + """Scopes that file different definitions under one name have no join: a `ScopeConflictError` naming the key and both definitions.""" + with pytest.raises(ScopeConflictError) as raised: + Context.joined(CORE, Context.of(MY_GZIP)) + (conflict,) = raised.value.conflicts + assert (conflict.key, conflict.claimed, conflict.found) == ( + (CodecDefinition, "gzip"), + GZIP_CODEC, + MY_GZIP, + ) From 4d85e806ff5518b918d1ab79d400c495f6640cd8 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Tue, 6 Oct 2026 14:04:51 +0200 Subject: [PATCH 28/94] docs(zarr-metadata): document the scope operations Assisted-by: ClaudeCode:claude-fable-5-1 --- .../zarr-metadata/changes/+scope-algebra.feature.md | 1 + .../zarr-metadata/src/zarr_metadata/v3/definition.py | 12 ++++++++++++ 2 files changed, 13 insertions(+) create mode 100644 packages/zarr-metadata/changes/+scope-algebra.feature.md diff --git a/packages/zarr-metadata/changes/+scope-algebra.feature.md b/packages/zarr-metadata/changes/+scope-algebra.feature.md new file mode 100644 index 0000000000..1f3dbfb922 --- /dev/null +++ b/packages/zarr-metadata/changes/+scope-algebra.feature.md @@ -0,0 +1 @@ +`Context` values are equal when they file the same definitions, and hash alike. `Context.joined` combines scopes, refusing one name filed two ways, and `Context.disagreements` says where a scope would read a document's fields otherwise. `claims_of` and `refines` give what a reading claimed of each name, and whether one reading of a field holds everything another does. diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/definition.py b/packages/zarr-metadata/src/zarr_metadata/v3/definition.py index 3ded408e30..8d041aa888 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/definition.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/definition.py @@ -292,6 +292,18 @@ def acme_lz4_rules( name. `node_metadata_json_schema_v3`, in `zarr_metadata.model`, writes a whole `zarr.json`, its fill value held to its data type's. +**Scopes as values.** Two scopes are equal when they file the same +definitions, and equal scopes hash alike. `claims_of(fields_of(field))` +says what a reading claimed of each name -- the definition that read it, +or None -- keyed as the scope files it, `r16` under `r*`. +`refines(field, other)` orders two readings of a field by information: a +name nothing claimed, read by a definition, is a gain; the reverse a +loss; one name read by two definitions a conflict. +`scope.disagreements(claims)` says where a scope would read a reading +otherwise, and `Context.joined(*scopes)` is the least scope above each, +or a `ScopeConflictError` naming each name filed two ways; `extended_with` +remains the way to take a name over on purpose. + A definition checks itself when it is built, and each of these is a `TypeError` saying what is wrong: a `configuration` that is not a TypedDict, says nothing of the keys it does not declare, or has a member From 990ad76bb3aa5887e02d11127085bfaa6fbb830b Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Tue, 6 Oct 2026 14:10:47 +0200 Subject: [PATCH 29/94] fix(zarr-metadata): compare a gain in refines the way Unclaimed compares `refines` compared a read field's written JSON to the unclaimed field's with Python `==`. That treated `"crc32c"` and `{"name": "crc32c"}` as different fields and `true` as equal to `1`, so the order was not transitive across spellings. A gain is now compared the way `Unclaimed` equality works: by name and by the configuration as JSON text. A refused field refines only itself, so a read field that holds one still refines itself. `Claims` is a real type at run time, not a string, so it works in signatures. The property test now checks reflexivity, transitivity, mutual refinement equals equality, and substitutivity, over readings of several documents in several scopes. Assisted-by: ClaudeCode:claude-fable-5-1 --- .../src/zarr_metadata/v3/_registry.py | 3 +- .../src/zarr_metadata/v3/_scope.py | 24 ++++-- packages/zarr-metadata/tests/v3/test_scope.py | 77 ++++++++++++++++--- 3 files changed, 86 insertions(+), 18 deletions(-) diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/_registry.py b/packages/zarr-metadata/src/zarr_metadata/v3/_registry.py index 7af55d3dc3..b2e134af89 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/_registry.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/_registry.py @@ -74,7 +74,8 @@ class Context: A value with no reading of its own: `resolve` reads a field in it, and `claimant` is the one question it answers, which definition a name belongs to. Built from definitions with `Context.of`, extended - with more by `extended_with`. + with more by `extended_with`; two scopes are equal when they file the + same definitions, and equal scopes hash alike. """ tables: Tables diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/_scope.py b/packages/zarr-metadata/src/zarr_metadata/v3/_scope.py index 90255c07f3..d00a02ca9a 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/_scope.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/_scope.py @@ -11,6 +11,7 @@ from __future__ import annotations +from collections.abc import Mapping from dataclasses import dataclass from typing import TYPE_CHECKING, Any, TypeAlias @@ -20,6 +21,7 @@ CodecDefinition, DataTypeDefinition, Definition, + Read, Refused, StorageTransformerDefinition, Unclaimed, @@ -29,7 +31,7 @@ ) if TYPE_CHECKING: - from collections.abc import Callable, Iterable, Mapping, Sequence + from collections.abc import Callable, Iterable, Sequence from zarr_metadata._typed_json import Loc from zarr_metadata.v3._definition import Resolved @@ -37,7 +39,7 @@ ClaimKey: TypeAlias = tuple[type[Definition[Any]], str] """A kind and the name a definition is filed under: what a scope answers `claimant` for.""" -Claims: TypeAlias = "Mapping[ClaimKey, Definition[Any] | None]" +Claims: TypeAlias = Mapping[ClaimKey, Definition[Any] | None] """What a reading claims of each name a document writes: the definition that read it, or None where nothing claimed it.""" _KIND_NAMES: dict[type[Definition[Any]], str] = { @@ -115,16 +117,19 @@ def refines(field: Resolved[Any], other: Resolved[Any]) -> bool: """Whether `field` holds everything `other` holds: reads the same where both read, and reads what `other` left unclaimed. The order one reading of a document refines another in. A name nothing - claimed, read by a definition, is a gain; the reverse is a loss; one - name read by two definitions is a conflict; a refused field is in the - order with nothing. Two fields that refine each other are equal. + claimed, read by a definition, is a gain: the read field is compared + as the unclaimed one would be, by its name and the configuration as + written, as JSON text, so two spellings of one field are one, and + `true` is not `1`. The reverse is a loss; one name read by two + definitions is a conflict; a refused field refines itself alone. Two + fields that refine each other are equal. """ if isinstance(field, Refused) or isinstance(other, Refused): - return False + return field == other if isinstance(other, Unclaimed): if isinstance(field, Unclaimed): return field_key(field) == field_key(other) - return claim_key(field) == claim_key(other) and field.json == other.json + return claim_key(field) == claim_key(other) and _as_unclaimed(field) == other if isinstance(field, Unclaimed): return False if field.definition != other.definition or own_key(field) != own_key(other): @@ -134,6 +139,11 @@ def refines(field: Resolved[Any], other: Resolved[Any]) -> bool: return all(refines(field.nested[loc], other.nested[loc]) for loc in field.nested) +def _as_unclaimed(field: Read[Any]) -> Unclaimed: + """`field` as it would have been read had nothing claimed its name: what a gain is compared against.""" + return Unclaimed(json=field.json, name=field.name, read_as=field.read_as) + + @dataclass(frozen=True, slots=True) class Disagreements: """Where a scope reads a reading's claims otherwise: the names it would gain a meaning for, and those it conflicts with, a lost meaning among them.""" diff --git a/packages/zarr-metadata/tests/v3/test_scope.py b/packages/zarr-metadata/tests/v3/test_scope.py index 9d053c8bc2..95f673b7d6 100644 --- a/packages/zarr-metadata/tests/v3/test_scope.py +++ b/packages/zarr-metadata/tests/v3/test_scope.py @@ -3,7 +3,7 @@ from __future__ import annotations import pickle -from typing import Any +from typing import Any, get_type_hints import pytest from hypothesis import given @@ -19,6 +19,7 @@ from zarr_metadata.v3.definition import ( CORE, CORE_AND_EXTENSIONS, + Claims, CodecDefinition, Conflict, Context, @@ -147,6 +148,19 @@ def test_error_claims_of_refuses_one_name_read_two_ways() -> None: GZIP_FIELD = {"name": "gzip", "configuration": {"level": 5}} ZSTD_FIELD = {"name": "zstd", "configuration": {"level": 3}} +BLOSC = {"cname": "zstd", "shuffle": "noshuffle", "blocksize": 0} +BLOSC_LEVEL_ONE = {"name": "blosc", "configuration": {**BLOSC, "clevel": 1}} +BLOSC_LEVEL_TRUE = {"name": "blosc", "configuration": {**BLOSC, "clevel": True}} +"""Two documents Python's `==` takes for one, which `json_text` tells apart.""" +SHARD_HOLDING_REFUSED = { + "name": "sharding_indexed", + "configuration": { + "chunk_shape": [1], + "codecs": ["bytes", {"name": "gzip", "configuration": {"level": 12}}], + "index_codecs": [{"name": "bytes", "configuration": {"endian": "little"}}, "crc32c"], + }, +} +"""A shard read, holding a gzip its definition refuses.""" NESTED_ZSTD = { "name": "sharding_indexed", "configuration": { @@ -202,6 +216,21 @@ def test_error_claims_of_refuses_one_name_read_two_ways() -> None: _read({"name": "zstd", "configuration": {"level": 2}}, CodecDefinition, CORE), False, ), + ( + _read("crc32c", CodecDefinition, CORE), + _read({"name": "crc32c"}, CodecDefinition, Context.of()), + True, + ), + ( + _read(BLOSC_LEVEL_ONE, CodecDefinition, CORE), + _read(BLOSC_LEVEL_TRUE, CodecDefinition, Context.of()), + False, + ), + ( + _read(SHARD_HOLDING_REFUSED, CodecDefinition, CORE), + _read(SHARD_HOLDING_REFUSED, CodecDefinition, CORE), + True, + ), ], ids=[ "same", @@ -214,6 +243,9 @@ def test_error_claims_of_refuses_one_name_read_two_ways() -> None: "refused", "both-unclaimed", "unclaimed-differ", + "gain-over-another-spelling", + "no-gain-over-true-for-one", + "holding-a-refused-field", ], ) def test_refines_orders_readings_by_information( @@ -223,15 +255,31 @@ def test_refines_orders_readings_by_information( assert refines(field, other) is expected -@given(st.sampled_from([GZIP_FIELD, "crc32c", {"name": "crc32c"}, ZSTD_FIELD, NESTED_ZSTD])) -def test_refines_is_reflexive_and_mutual_refinement_is_equality(data: object) -> None: - """Every field a scope reads refines itself, and two such fields that refine each other are equal: the order's bottom is equality. A refused field is outside the order, so it is not sampled.""" - for scope in (CORE, CORE_AND_EXTENSIONS): - field = _read(data, CodecDefinition, scope) - assert refines(field, field) - low = _read(data, CodecDefinition, CORE) - high = _read(data, CodecDefinition, CORE_AND_EXTENSIONS) - assert (refines(low, high) and refines(high, low)) is (low == high) +DOCUMENTS = [ + GZIP_FIELD, + "crc32c", + {"name": "crc32c"}, + {"name": "crc32c", "configuration": {}}, + ZSTD_FIELD, + NESTED_ZSTD, + BLOSC_LEVEL_ONE, + SHARD_HOLDING_REFUSED, +] +SCOPES = [Context.of(), CORE, CORE_AND_EXTENSIONS] +READINGS = [_read(document, CodecDefinition, scope) for document in DOCUMENTS for scope in SCOPES] + + +@given(st.sampled_from(READINGS), st.sampled_from(READINGS), st.sampled_from(READINGS)) +def test_refines_is_a_partial_order_whose_bottom_is_equality( + a: Resolved[Any], b: Resolved[Any], c: Resolved[Any] +) -> None: + """Over readings of documents in several scopes and spellings, `refines` is reflexive and transitive, two fields that refine each other are equal, and equal fields refine the same fields.""" + assert refines(a, a) + if refines(a, b) and refines(b, c): + assert refines(a, c) + assert (refines(a, b) and refines(b, a)) is (a == b) + if a == b: + assert refines(c, a) is refines(c, b) @pytest.mark.parametrize( @@ -303,3 +351,12 @@ def test_error_joined_refuses_one_name_filed_two_ways() -> None: GZIP_CODEC, MY_GZIP, ) + + +def test_claims_is_a_type_a_signature_can_hold() -> None: + """`Claims` resolves as an annotation at run time, as a caller's `get_type_hints` reads one: it is a type, not a string.""" + + def read(claims: Claims) -> None: + pass + + assert "claims" in get_type_hints(read) From c93da50af3058d63e7a3efbbcc19954a483ab90d Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Tue, 6 Oct 2026 14:11:08 +0200 Subject: [PATCH 30/94] docs(zarr-metadata): rename the scope fragment with its PR number Assisted-by: ClaudeCode:claude-fable-5-1 --- .../changes/{+scope-algebra.feature.md => 392.feature.md} | 0 1 file changed, 0 insertions(+), 0 deletions(-) rename packages/zarr-metadata/changes/{+scope-algebra.feature.md => 392.feature.md} (100%) diff --git a/packages/zarr-metadata/changes/+scope-algebra.feature.md b/packages/zarr-metadata/changes/392.feature.md similarity index 100% rename from packages/zarr-metadata/changes/+scope-algebra.feature.md rename to packages/zarr-metadata/changes/392.feature.md From de78ec134ef4100be45ff768472b2ef888dc1537 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Tue, 6 Oct 2026 18:25:53 +0200 Subject: [PATCH 31/94] docs(zarr-metadata): rewrite the 392 changelog fragments in plain English Assisted-by: ClaudeCode:claude-fable-5-1 --- packages/zarr-metadata/changes/392.feature.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/packages/zarr-metadata/changes/392.feature.md b/packages/zarr-metadata/changes/392.feature.md index 1f3dbfb922..e89f440e76 100644 --- a/packages/zarr-metadata/changes/392.feature.md +++ b/packages/zarr-metadata/changes/392.feature.md @@ -1 +1 @@ -`Context` values are equal when they file the same definitions, and hash alike. `Context.joined` combines scopes, refusing one name filed two ways, and `Context.disagreements` says where a scope would read a document's fields otherwise. `claims_of` and `refines` give what a reading claimed of each name, and whether one reading of a field holds everything another does. +`Context` values are equal when they hold the same definitions, and equal values hash alike. `Context.joined` combines scopes and raises `ScopeConflictError` if two of them define one name differently. `Context.disagreements` reports where a scope would read a document's fields differently. `claims_of` lists the definition a reading used for each name, and `refines` says whether one reading of a field contains everything another does. From 10ae2087e64b9387b548eb00f73fb021ac8c0ed3 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Tue, 6 Oct 2026 14:35:31 +0200 Subject: [PATCH 32/94] feat(zarr-metadata)!: make ZarrV3ArrayMetadata a (document, context) pair `ZarrV3ArrayMetadata(document, context=None)` reads the document in the given scope (`CORE_AND_EXTENSIONS` when None) and raises `MetadataValidationError` with every problem. The model stores the refined document, the scope and the reading. `to_json` returns the document as written, so spelling is preserved. The typed members (`data_type`, `codecs`, `shape`, `attributes`, ...) are read-only views of the reading. A private `_of` builds a model from an existing reading so readers never read twice. The typed-field constructor is removed. Assisted-by: ClaudeCode:claude-fable-5-1 --- .../src/zarr_metadata/model/_array.py | 476 ++++++++---------- .../zarr-metadata/tests/model/test_pair.py | 105 ++++ 2 files changed, 327 insertions(+), 254 deletions(-) create mode 100644 packages/zarr-metadata/tests/model/test_pair.py diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_array.py b/packages/zarr-metadata/src/zarr_metadata/model/_array.py index f78ebe0316..9bfabc2e95 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_array.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/_array.py @@ -3,9 +3,10 @@ from __future__ import annotations import dataclasses -from collections.abc import Callable, Mapping +from collections.abc import Mapping from dataclasses import dataclass, field -from typing import TYPE_CHECKING, Any, Literal, cast +from types import MappingProxyType +from typing import TYPE_CHECKING, Any, Final, Literal, cast from typing_extensions import TypedDict, Unpack @@ -14,12 +15,11 @@ ValidationProblem, copied, json_text, + refine_json, with_input, ) from zarr_metadata._sentinel import UNSET from zarr_metadata.model._validation import ( - ARRAY_METADATA_STANDARD_KEYS_V3, - NO_SCOPE, ArrayMembersV3, StoreKey, ZarrV3ArrayMetadataReading, @@ -27,7 +27,6 @@ dimension_lengths, dump_store_json, load_store_json, - overlapping, parse_array_metadata_v2, read_array_v3, ) @@ -41,11 +40,11 @@ Read, StorageTransformerDefinition, Unclaimed, - document_json, field_key, spelled_canonically, ) from zarr_metadata.v3._registry import CORE_AND_EXTENSIONS, Context +from zarr_metadata.v3._scope import Claims, claims_of from zarr_metadata.v3.array import ZARR_V3_ARRAY_METADATA_STORE_KEY, ZarrV3ExtensionField if TYPE_CHECKING: @@ -105,54 +104,190 @@ class ZarrV3ArrayMetadataUpdate(TypedDict, total=False, extra_items=ZarrV3Extens dimension_names: tuple[str | None, ...] | UNSET -@dataclass(frozen=True, slots=True, kw_only=True) class ZarrV3ArrayMetadata: - """In-memory model of a v3 array metadata document. + """A v3 array document, and the scope it was read in. - A canonical, semantically lossless representation of the `zarr.json` - content for an array. Each extension point -- `data_type`, + The model is the pair: `to_json` is the document as written, refined + -- arrays as tuples, string keys -- and `context` the scope. Every + typed member is a view of the reading the pair gives: `data_type`, `chunk_grid`, `chunk_key_encoding`, each codec and storage transformer - -- is held as a scope read it: `Read`, holding the definition that - read it, or `Unclaimed`, an extension that scope left unjudged. - `fill_value` is held verbatim in its JSON form. - - A model holds no scope: each field keeps the definition that read it, - and a scope is asked only to read new JSON -- by `from_json`, - `create_default`, and `update`, which each take one. A model checks - itself when it is built, as pydantic's `__init__` does: its document, - as its own fields read it, has no problem, or the constructor raises - `MetadataValidationError` with every one. So a model built by hand, or - changed as `dataclasses.replace` changes one, is refused at the - change, and none is built invalid. It holds its members as that read - refines them, in containers of its own -- a list given for an array - as a tuple -- as pydantic holds what its `__init__` coerced, and each - field as the scope read it: a field built by hand is taken as read. A - model a read builds is not read a second time. Change a model by - building another: a container it holds, changed in place, is not - checked again. `to_json` writes each extension point as its - readers take it, as `Read.to_json` says. A model pickles when the - definitions its fields hold do: ones whose functions are defined at a - module's top level. + as the scope read it, `Read` by the definition that claims its name or + `Unclaimed`; `shape`, `fill_value`, `dimension_names`, `attributes` and + `extra_fields` as the read refined them. Built only by reading: the + constructor reads `document` in `context` and raises + `MetadataValidationError` with every problem, so no model is invalid. + Two models are equal when their documents mean the same in their + scopes, as `array_key` says; the scope itself takes no part. `update` + reads new members in the model's own scope; `with_context` and + `refined_in` read the document in another. A model pickles as its + pair, when the definitions its scope holds do: ones whose functions + are defined at a module's top level. """ - zarr_format: Literal[3] = field(default=3, init=False) - node_type: Literal["array"] = field(default="array", init=False) - shape: tuple[int, ...] - fill_value: JSONValue - data_type: Read[DataTypeDefinition[Any]] | Unclaimed - chunk_grid: Read[ChunkGridDefinition[Any]] | Unclaimed - codecs: tuple[Read[CodecDefinition[Any]] | Unclaimed, ...] - chunk_key_encoding: Read[ChunkKeyEncodingDefinition[Any]] | Unclaimed - dimension_names: tuple[str | None, ...] | UNSET - attributes: dict[str, JSONValue] - storage_transformers: tuple[Read[StorageTransformerDefinition[Any]] | Unclaimed, ...] - extra_fields: dict[str, ZarrV3ExtensionField] + __slots__ = ("_claims", "_context", "_document", "_key", "_members", "_reading") + + zarr_format: Final = 3 + node_type: Final = "array" + + def __init__(self, document: object, context: Context | None = None) -> None: + scope = CORE_AND_EXTENSIONS if context is None else context + reading, members = read_array_v3(document, scope) + if members is None: + raise MetadataValidationError(reading.problems) + refined, _ = refine_json(document) + self._adopt(cast("dict[str, JSONValue]", refined), scope, reading, members) + + @classmethod + def _of( + cls, + document: dict[str, JSONValue], + context: Context, + reading: ZarrV3ArrayMetadataReading, + members: ArrayMembersV3, + ) -> ZarrV3ArrayMetadata: + """A model of a document a read found nothing wrong with, holding that reading: no second read.""" + model = object.__new__(cls) + model._adopt(document, context, reading, members) + return model + + def _adopt( + self, + document: dict[str, JSONValue], + context: Context, + reading: ZarrV3ArrayMetadataReading, + members: ArrayMembersV3, + ) -> None: + self._document = document + self._context = context + self._reading = reading + self._members = members + self._key = array_key(self) + self._claims = MappingProxyType(claims_of(reading.fields())) + + # --- the pair --------------------------------------------------------- + + @property + def context(self) -> Context: + """The scope the document was read in, which `update` reads new members in.""" + return self._context + + @property + def reading(self) -> ZarrV3ArrayMetadataReading: + """The document as the scope read it: each field, the pipeline, the chunk each codec is handed.""" + return self._reading + + @property + def claims(self) -> Claims: + """What the reading claimed of each name the document writes, keyed as the scope files it.""" + return self._claims + + def to_json(self) -> ZarrV3ArrayMetadataJSON: + """The document as written, refined, sharing nothing with the model.""" + return cast("ZarrV3ArrayMetadataJSON", copied(self._document)) + + def to_key_value( + self, *, indent: int | str | None = None + ) -> Mapping[ZarrV3ArrayMetadataStoreKey, bytes]: + """The document as a store holds it: JSON bytes at `zarr.json`, indented by `indent`. + + `NaN`, `Infinity` and `-Infinity` in `attributes` are written as + those bare tokens, as zarr-python writes them, which a strict JSON + parser refuses. + """ + return {ZARR_V3_ARRAY_METADATA_STORE_KEY: dump_store_json(self._document, indent=indent)} + + def __repr__(self) -> str: + return f"{type(self).__name__}({self._document!r}, context={self._context!r})" + + def __eq__(self, other: object) -> bool: + if type(other) is not type(self): + return NotImplemented + return self._key == cast("ZarrV3ArrayMetadata", other)._key + + def __hash__(self) -> int: + return hash(self._key) + + # --- typed views ------------------------------------------------------ + + @property + def shape(self) -> tuple[int, ...]: + """The array's shape.""" + return self._members.shape + + @property + def fill_value(self) -> JSONValue: + """The fill value as written.""" + return self._members.fill_value + + @property + def dimension_names(self) -> tuple[str | None, ...] | UNSET: + """The dimension names; `UNSET` when the document writes none.""" + return self._members.dimension_names + + @property + def attributes(self) -> Mapping[str, JSONValue]: + """The attributes, a read-only view; empty when the document writes none.""" + return MappingProxyType(self._members.attributes) + + @property + def extra_fields(self) -> Mapping[str, ZarrV3ExtensionField]: + """Each member the spec does not define, by name, a read-only view.""" + return MappingProxyType(self._members.extra_fields) + + @property + def must_understand_fields(self) -> dict[str, ZarrV3ExtensionField]: + """Extra fields the reader is obligated to understand. + + Everything in `extra_fields` not explicitly waived with + `must_understand: false` (the spec's implicit-true rule, https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v3/core/index.rst#L1571-L1578). A compliant + reader MUST fail to open the array if this contains any field it does + not recognize; the model layer only partitions by obligation, since + recognition is reader-specific. + """ + return must_understand_subset(self.extra_fields) + + @property + def data_type(self) -> Read[DataTypeDefinition[Any]] | Unclaimed: + """The data type, as the scope read it.""" + return cast("Read[DataTypeDefinition[Any]] | Unclaimed", self._reading.data_type) + + @property + def chunk_grid(self) -> Read[ChunkGridDefinition[Any]] | Unclaimed: + """The chunk grid, as the scope read it.""" + return cast("Read[ChunkGridDefinition[Any]] | Unclaimed", self._reading.chunk_grid) + + @property + def chunk_key_encoding(self) -> Read[ChunkKeyEncodingDefinition[Any]] | Unclaimed: + """The chunk key encoding, as the scope read it.""" + return cast( + "Read[ChunkKeyEncodingDefinition[Any]] | Unclaimed", self._reading.chunk_key_encoding + ) + + @property + def codecs(self) -> tuple[Read[CodecDefinition[Any]] | Unclaimed, ...]: + """The codecs, each as the scope read it, in pipeline order.""" + return tuple( + cast("Read[CodecDefinition[Any]] | Unclaimed", stage.codec) + for stage in self._reading.pipeline + ) + + @property + def storage_transformers( + self, + ) -> tuple[Read[StorageTransformerDefinition[Any]] | Unclaimed, ...]: + """The storage transformers, each as the scope read it.""" + return cast( + "tuple[Read[StorageTransformerDefinition[Any]] | Unclaimed, ...]", + self._reading.storage_transformers, + ) + + # --- constructors ----------------------------------------------------- @classmethod def create_default( cls, *, - context: Context = CORE_AND_EXTENSIONS, + context: Context | None = None, **overrides: Unpack[ZarrV3ArrayMetadataJSONPartial], ) -> ZarrV3ArrayMetadata: """A scalar `uint8` array, or the one `overrides`, members of its document, make of it, read in `context`. @@ -184,141 +319,28 @@ def create_default( "codecs": ({"name": "bytes", "configuration": {"endian": "little"}},), "chunk_key_encoding": {"name": "default"}, } - return cls.from_json({**document, **overrides}, context=context) - - def update( - self, *, context: Context, **members: Unpack[ZarrV3ArrayMetadataUpdate] - ) -> ZarrV3ArrayMetadata: - """This model with `members` in their place, each read in `context`; `UNSET` leaves an optional member out. - - Only the members given are read in `context`: each field the model - holds is kept as it was read, whatever scope read it. The document - they make is then read as a whole, so `MetadataValidationError` - when it has a problem, and members that go together are passed - together: a `shape` with a grid that fits it. - """ - document = {**held_document(self), **members} - for key, value in members.items(): - if value is UNSET: - del document[key] - # The model's own fields are held, read already; the members given - # are JSON, read in `context`, a field object among them refused. - reading, refined = read_array_v3(document, context, held=own_fields(self)) - if refined is None: - raise MetadataValidationError(reading.problems) - return array_model(reading, refined) - - def __post_init__(self) -> None: - # The runtime half of the annotations: extra fields by name, and each - # extension point a field read as its kind, read or unclaimed, as a - # read gives it. - extra = cast("object", self.extra_fields) - if not isinstance(extra, Mapping): - msg = f"extra_fields: expected a mapping of names to JSON, got {extra!r}" - raise TypeError(msg) - for key, kind, nodes in ( - ("data_type", DataTypeDefinition, (self.data_type,)), - ("chunk_grid", ChunkGridDefinition, (self.chunk_grid,)), - ("chunk_key_encoding", ChunkKeyEncodingDefinition, (self.chunk_key_encoding,)), - ("codecs", CodecDefinition, self.codecs), - ("storage_transformers", StorageTransformerDefinition, self.storage_transformers), - ): - for node in cast("tuple[object, ...]", nodes): - if not isinstance(node, (Read, Unclaimed)) or node.read_as is not kind: - msg = f"{key}: expected a field read as a {kind.__name__}, got {node!r}" - raise TypeError(msg) - # The rest held as the read of its document refines it, in containers - # of its own, as pydantic holds what its `__init__` coerced. - members = _members(self) - object.__setattr__(self, "shape", members.shape) - object.__setattr__(self, "fill_value", members.fill_value) - object.__setattr__(self, "dimension_names", members.dimension_names) - object.__setattr__(self, "attributes", members.attributes) - object.__setattr__(self, "extra_fields", members.extra_fields) - object.__setattr__(self, "codecs", tuple(self.codecs)) - object.__setattr__(self, "storage_transformers", tuple(self.storage_transformers)) - - def __eq__(self, other: object) -> bool: - """Whether `other` models the same array: the same document, however each is spelled. - - Compared as `array_key` says: what the package interprets -- each - field, and the fill value against the data type -- in its canonical - spelling, so `"NaN"` and `"0x7fc00000"` are one `float32` fill value - and a blosc with and without the `typesize` that `noshuffle` ignores - one codec; and what it does not interpret -- the attributes, the - extra fields, the fill value of a data type nothing in scope claims - -- as JSON text, which tells `true` from `1` and `-0.0` from `0.0`, - and takes `NaN` for itself. So two equal models may write two - documents: `to_json` writes each as it was given. Equal models hash - alike. - """ - if type(other) is not type(self): - return NotImplemented - return array_key(self) == array_key(cast("ZarrV3ArrayMetadata", other)) - - def __hash__(self) -> int: - return hash(array_key(self)) - - def to_json(self) -> ZarrV3ArrayMetadataJSON: - """The document as JSON, arrays as tuples, sharing no mutable state with the model. - - Each extension point as its readers take it, as `Read.to_json` - writes one; `dimension_names` when set, and `attributes` and - `storage_transformers` when not empty. - """ - return cast("ZarrV3ArrayMetadataJSON", copied(cast("JSONValue", array_json(self)))) + return cls({**document, **overrides}, context=context) @classmethod - def from_json( - cls, data: object, *, context: Context = CORE_AND_EXTENSIONS - ) -> ZarrV3ArrayMetadata: + def from_json(cls, data: object, *, context: Context | None = None) -> ZarrV3ArrayMetadata: """The model of `data`, a v3 array document read in `context`. `MetadataValidationError` with every problem the read finds. `read_array_metadata_v3` gives the reading this model is built from, and the problems of a document with some. """ - reading = read_array_metadata_v3(data, context=context) - if reading.metadata is None: - raise MetadataValidationError(reading.problems) - return reading.metadata - - @property - def must_understand_fields(self) -> dict[str, ZarrV3ExtensionField]: - """Extra fields the reader is obligated to understand. - - Everything in `extra_fields` not explicitly waived with - `must_understand: false` (the spec's implicit-true rule, https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v3/core/index.rst#L1571-L1578). A compliant - reader MUST fail to open the array if this contains any field it does - not recognize; the model layer only partitions by obligation, since - recognition is reader-specific. - """ - return must_understand_subset(self.extra_fields) + return cls(data, context=context) @classmethod def from_key_value( - cls, mapping: Mapping[StoreKey, bytes], *, context: Context = CORE_AND_EXTENSIONS + cls, mapping: Mapping[StoreKey, bytes], *, context: Context | None = None ) -> ZarrV3ArrayMetadata: """The model of the array document at `zarr.json` in `mapping`, read in `context`. `MetadataValidationError` when the key is missing, its bytes are not JSON, or the document is not valid. """ - return cls.from_json( - load_store_json(mapping, ZARR_V3_ARRAY_METADATA_STORE_KEY), context=context - ) - - def to_key_value( - self, *, indent: int | str | None = None - ) -> Mapping[ZarrV3ArrayMetadataStoreKey, bytes]: - """The document as a store holds it: JSON bytes at `zarr.json`, indented by `indent`. - - A model was checked when it was built, so its document is written - as it is. `NaN`, `Infinity` and `-Infinity` in `attributes` are - written as those bare tokens, as zarr-python writes them, which a - strict JSON parser refuses. - """ - return {ZARR_V3_ARRAY_METADATA_STORE_KEY: dump_store_json(array_json(self), indent=indent)} + return cls(load_store_json(mapping, ZARR_V3_ARRAY_METADATA_STORE_KEY), context=context) def array_key(model: ZarrV3ArrayMetadata) -> tuple[object, ...]: @@ -336,9 +358,9 @@ def array_key(model: ZarrV3ArrayMetadata) -> tuple[object, ...]: tuple(field_key(codec) for codec in model.codecs), field_key(model.chunk_key_encoding), model.dimension_names, - json_text(model.attributes), + json_text(dict(model.attributes)), tuple(field_key(transformer) for transformer in model.storage_transformers), - json_text(model.extra_fields), + json_text(dict(model.extra_fields)), ) @@ -349,75 +371,45 @@ def _fill_value_key(model: ZarrV3ArrayMetadata) -> str: return json_text(model.fill_value) -def _members(model: ZarrV3ArrayMetadata) -> ArrayMembersV3: - """`model`'s members other than its fields, as the read of its document by its own fields refines them; `MetadataValidationError` with every problem that document has.""" - reading, members = read_array_v3(held_document(model), NO_SCOPE, held=own_fields(model)) - extra = overlapping(model.extra_fields, ARRAY_METADATA_STANDARD_KEYS_V3, "array") - problems = (*extra, *reading.problems) - if len(problems) != 0: - raise MetadataValidationError(problems) - return cast("ArrayMembersV3", members) - - def array_json(model: ZarrV3ArrayMetadata) -> ZarrV3ArrayMetadataJSON: - """`model`'s document as JSON, holding the model's own values: what `to_key_value` serializes, which changes nothing, and `to_json` copies.""" - return cast("ZarrV3ArrayMetadataJSON", _document(model, document_json, whole=False)) - + """`model`'s document, holding the model's own values: what a writer serializes without copying.""" + return cast("ZarrV3ArrayMetadataJSON", model._document) # pyright: ignore[reportPrivateUsage] -def own_fields(model: ZarrV3ArrayMetadata) -> tuple[Read[Any] | Unclaimed, ...]: - """The fields `model` holds, each as a scope read it: what a read of the model's own document takes as read.""" - return ( - model.data_type, - model.chunk_grid, - model.chunk_key_encoding, - *model.codecs, - *model.storage_transformers, - ) +def array_model( + reading: ZarrV3ArrayMetadataReading, members: ArrayMembersV3 +) -> ZarrV3ArrayMetadata: + """The model of a document its reading found nothing wrong with, its document rebuilt from the reading. -def held_document(model: ZarrV3ArrayMetadata) -> dict[str, object]: - """`model`'s document with each field as it was read, which a read takes as it is: what `update` and the constructor read, reading no field again. - - Every member the model holds, an empty one too, so the read judges - each whatever it holds. + A shim for the group reader until it hands the nested documents down + with their readings: the document is rebuilt from what was read, so a + member written empty is written back as the read holds it. """ - return _document(model, _as_read, whole=True) - - -def _as_read(field: Read[Any] | Unclaimed) -> object: - return field - - -def _document( - model: ZarrV3ArrayMetadata, write: Callable[[Read[Any] | Unclaimed], object], *, whole: bool -) -> dict[str, object]: - """`model`'s document, each field as `write` gives it, the rest as the model holds it; empty `attributes` and `storage_transformers` too when `whole`, where a writer leaves them out.""" - out: dict[str, object] = { - "zarr_format": model.zarr_format, - "node_type": model.node_type, - "shape": model.shape, - "fill_value": model.fill_value, - "data_type": write(model.data_type), - "chunk_grid": write(model.chunk_grid), - "codecs": tuple(write(codec) for codec in model.codecs), - "chunk_key_encoding": write(model.chunk_key_encoding), + document: dict[str, JSONValue] = { + "zarr_format": 3, + "node_type": "array", + "shape": members.shape, + "data_type": cast("Read[Any] | Unclaimed", reading.data_type).json, + "chunk_grid": cast("Read[Any] | Unclaimed", reading.chunk_grid).json, + "chunk_key_encoding": cast("Read[Any] | Unclaimed", reading.chunk_key_encoding).json, + "fill_value": members.fill_value, + "codecs": tuple( + cast("Read[Any] | Unclaimed", stage.codec).json for stage in reading.pipeline + ), } - if model.dimension_names is not UNSET: - out["dimension_names"] = model.dimension_names - if whole or len(model.attributes) > 0: - out["attributes"] = model.attributes - if whole or len(model.storage_transformers) > 0: - out["storage_transformers"] = tuple( - write(transformer) for transformer in model.storage_transformers + if members.dimension_names is not UNSET: + document["dimension_names"] = members.dimension_names + if len(members.attributes) != 0: + document["attributes"] = members.attributes + if len(reading.storage_transformers) != 0: + document["storage_transformers"] = tuple( + cast("Read[Any] | Unclaimed", transformer).json + for transformer in reading.storage_transformers ) - # An extra field named as a member the document declares is no member - # of it, which the constructor reports. - out.update( - (key, value) - for key, value in model.extra_fields.items() - if key not in ARRAY_METADATA_STANDARD_KEYS_V3 + document.update(members.extra_fields) + return ZarrV3ArrayMetadata._of( # pyright: ignore[reportPrivateUsage] + document, CORE_AND_EXTENSIONS, reading, members ) - return out def read_array_metadata_v3( @@ -430,41 +422,17 @@ def read_array_metadata_v3( or `Refused` -- the chunks the codecs are handed, each codec with the chunk it is handed, every problem `validate_array_metadata_v3` finds, and, when there is none, the document's model, holding the same - fields. A policy over the fields, the core spec's alone, say, is a + reading. A policy over the fields, the core spec's alone, say, is a walk over its `fields()`. A value that is not an object holds no field. """ reading, members = read_array_v3(value, context) if members is None: return reading - return dataclasses.replace(reading, metadata=array_model(reading, members)) - - -def array_model( - reading: ZarrV3ArrayMetadataReading, members: ArrayMembersV3 -) -> ZarrV3ArrayMetadata: - """The model of a document its reading found nothing wrong with: its fields as read, and its other members as the read refined them, not read again.""" - return construct( - ZarrV3ArrayMetadata, - shape=members.shape, - fill_value=members.fill_value, - data_type=cast("Read[DataTypeDefinition[Any]] | Unclaimed", reading.data_type), - chunk_grid=cast("Read[ChunkGridDefinition[Any]] | Unclaimed", reading.chunk_grid), - codecs=tuple( - cast("Read[CodecDefinition[Any]] | Unclaimed", stage.codec) - for stage in reading.pipeline - ), - chunk_key_encoding=cast( - "Read[ChunkKeyEncodingDefinition[Any]] | Unclaimed", reading.chunk_key_encoding - ), - dimension_names=members.dimension_names, - attributes=members.attributes, - storage_transformers=cast( - "tuple[Read[StorageTransformerDefinition[Any]] | Unclaimed, ...]", - reading.storage_transformers, - ), - extra_fields=members.extra_fields, - ) + refined, _ = refine_json(value) + document = cast("dict[str, JSONValue]", refined) + model = ZarrV3ArrayMetadata._of(document, context, reading, members) # pyright: ignore[reportPrivateUsage] + return dataclasses.replace(reading, metadata=model) class ZarrV2ArrayMetadataPartial(TypedDict, total=False): diff --git a/packages/zarr-metadata/tests/model/test_pair.py b/packages/zarr-metadata/tests/model/test_pair.py new file mode 100644 index 0000000000..caaa0511be --- /dev/null +++ b/packages/zarr-metadata/tests/model/test_pair.py @@ -0,0 +1,105 @@ +"""A v3 model is a document and the scope it was read in: the pair, and what follows from it.""" + +from __future__ import annotations + +from collections.abc import Mapping +from typing import Any + +import pytest + +from zarr_metadata._json import refine_json +from zarr_metadata.model import ( + UNSET, + MetadataValidationError, + ZarrV3ArrayMetadata, + read_array_metadata_v3, +) +from zarr_metadata.v3.definition import CORE, CORE_AND_EXTENSIONS, Context, Read, Unclaimed + +ARRAY: dict[str, Any] = { + "zarr_format": 3, + "node_type": "array", + "shape": [4], + "data_type": "uint8", + "fill_value": 0, + "chunk_grid": {"name": "regular", "configuration": {"chunk_shape": [2]}}, + "chunk_key_encoding": {"name": "default"}, + "codecs": ["bytes"], +} +SPELLED_OUT: dict[str, Any] = { + **ARRAY, + "data_type": {"name": "uint8"}, + "codecs": [{"name": "bytes", "configuration": {}}], + "chunk_key_encoding": {"name": "default", "configuration": {"separator": "/"}}, +} + + +@pytest.mark.parametrize( + ("document", "context", "expected_context"), + [ + (ARRAY, None, CORE_AND_EXTENSIONS), + (ARRAY, CORE, CORE), + (SPELLED_OUT, Context.of(), Context.of()), + ], + ids=["default-scope", "core", "empty-scope"], +) +def test_a_model_is_its_document_read_in_its_scope( + document: dict[str, Any], context: Context | None, expected_context: Context +) -> None: + """A model built from a document holds that document, as written and refined, and the scope it was read in, `CORE_AND_EXTENSIONS` when none is given.""" + model = ZarrV3ArrayMetadata(document, context=context) + assert model.context == expected_context + assert model.to_json() == refine_json(document)[0] + + +@pytest.mark.parametrize( + ("document", "key", "written"), + [ + (ARRAY, "data_type", "uint8"), + (SPELLED_OUT, "data_type", {"name": "uint8"}), + (ARRAY, "codecs", ("bytes",)), + (SPELLED_OUT, "codecs", ({"name": "bytes", "configuration": {}},)), + ], + ids=["bare-data-type", "object-data-type", "bare-codec", "object-codec"], +) +def test_to_json_keeps_the_spelling_the_document_was_written_in( + document: dict[str, Any], key: str, written: object +) -> None: + """`to_json` writes each field as the document wrote it, not as a reader would respell it: the model is the document.""" + assert ZarrV3ArrayMetadata(document).to_json()[key] == written + + +def test_properties_are_what_the_reading_holds() -> None: + """The typed members -- fields as the scope read them, shape, fill value, attributes -- are views of the reading, read-only.""" + model = ZarrV3ArrayMetadata({**ARRAY, "attributes": {"a": [1]}, "acme": 1}) + assert isinstance(model.data_type, Read) + assert isinstance(model.codecs[0], Read) + assert model.shape == (4,) + assert model.fill_value == 0 + assert model.dimension_names is UNSET + assert isinstance(model.attributes, Mapping) + assert model.attributes == {"a": (1,)} + assert model.extra_fields == {"acme": 1} + assert (model.zarr_format, model.node_type) == (3, "array") + with pytest.raises(TypeError): + model.attributes["b"] = 1 # type: ignore[index] + with pytest.raises(AttributeError): + model.shape = (5,) # type: ignore[misc] + assert isinstance(ZarrV3ArrayMetadata(ARRAY, context=Context.of()).data_type, Unclaimed) + + +def test_a_reading_without_problems_builds_the_model_without_reading_again() -> None: + """`read_array_metadata_v3` hands its reading to the model it builds, so the model's `reading` is that reading, not a second one.""" + reading = read_array_metadata_v3(ARRAY) + assert reading.metadata is not None + assert reading.metadata.reading.pipeline is reading.pipeline + + +def test_error_a_document_with_a_problem_is_refused_at_construction() -> None: + """The constructor raises `MetadataValidationError` with every problem, as `from_json` does.""" + with pytest.raises(MetadataValidationError) as raised: + ZarrV3ArrayMetadata({**ARRAY, "fill_value": 300, "shape": [-1]}) + assert sorted(problem.loc for problem in raised.value.problems) == [ + ("fill_value",), + ("shape",), + ] From 21ce3f89cd798a2d64a055de08a6700480187136 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Tue, 6 Oct 2026 14:36:12 +0200 Subject: [PATCH 33/94] feat(zarr-metadata): cache the array model's equality key and pickle it as its pair Equality and hash use a key computed once at construction. The key compares what the document means (each field by definition and canonical configuration, the rest as JSON text); the scope itself is not part of it. Pickle stores `(document, context)` and reads again on load. Assisted-by: ClaudeCode:claude-fable-5-1 --- .../src/zarr_metadata/model/_array.py | 5 ++ .../zarr-metadata/tests/model/test_pair.py | 63 ++++++++++++++++++- 2 files changed, 67 insertions(+), 1 deletion(-) diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_array.py b/packages/zarr-metadata/src/zarr_metadata/model/_array.py index 9bfabc2e95..cfa60b9814 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_array.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/_array.py @@ -207,6 +207,11 @@ def __eq__(self, other: object) -> bool: def __hash__(self) -> int: return hash(self._key) + def __reduce__(self) -> tuple[type[ZarrV3ArrayMetadata], tuple[object, Context]]: + # The pair, read again on load: a model's reading never disagrees + # with its document. + return type(self), (self._document, self._context) + # --- typed views ------------------------------------------------------ @property diff --git a/packages/zarr-metadata/tests/model/test_pair.py b/packages/zarr-metadata/tests/model/test_pair.py index caaa0511be..0ff8b68354 100644 --- a/packages/zarr-metadata/tests/model/test_pair.py +++ b/packages/zarr-metadata/tests/model/test_pair.py @@ -2,6 +2,7 @@ from __future__ import annotations +import pickle from collections.abc import Mapping from typing import Any @@ -14,7 +15,15 @@ ZarrV3ArrayMetadata, read_array_metadata_v3, ) -from zarr_metadata.v3.definition import CORE, CORE_AND_EXTENSIONS, Context, Read, Unclaimed +from zarr_metadata.v3.codec.crc32c import Empty +from zarr_metadata.v3.definition import ( + CORE, + CORE_AND_EXTENSIONS, + CodecDefinition, + Context, + Read, + Unclaimed, +) ARRAY: dict[str, Any] = { "zarr_format": 3, @@ -26,6 +35,9 @@ "chunk_key_encoding": {"name": "default"}, "codecs": ["bytes"], } +MY_BYTES = CodecDefinition(name="bytes", configuration=Empty, kind="array_bytes", size="static") +"""A private `bytes`: another meaning under the core name, which takes no configuration.""" +PRIVATE = CORE.extended_with(MY_BYTES) SPELLED_OUT: dict[str, Any] = { **ARRAY, "data_type": {"name": "uint8"}, @@ -103,3 +115,52 @@ def test_error_a_document_with_a_problem_is_refused_at_construction() -> None: ("fill_value",), ("shape",), ] + + +@pytest.mark.parametrize( + ("left", "right", "equal"), + [ + (ZarrV3ArrayMetadata(ARRAY), ZarrV3ArrayMetadata(SPELLED_OUT), True), + (ZarrV3ArrayMetadata(ARRAY, context=CORE), ZarrV3ArrayMetadata(ARRAY), True), + (ZarrV3ArrayMetadata(ARRAY), ZarrV3ArrayMetadata(ARRAY, context=PRIVATE), False), + (ZarrV3ArrayMetadata(ARRAY), ZarrV3ArrayMetadata(ARRAY, context=Context.of()), False), + (ZarrV3ArrayMetadata(ARRAY), ZarrV3ArrayMetadata({**ARRAY, "attributes": {}}), True), + ], + ids=["spellings", "unused-definitions", "private-bytes", "unclaimed", "empty-attributes"], +) +def test_models_are_equal_by_what_their_documents_mean( + left: ZarrV3ArrayMetadata, right: ZarrV3ArrayMetadata, equal: bool +) -> None: + """Two models are equal when every field reads the same by the same definition and the rest is the same JSON: spelling and unused definitions do not matter, a private definition under a core name does, and so does a name read against one left unclaimed.""" + assert (left == right) is equal + if equal: + assert hash(left) == hash(right) + + +@pytest.mark.parametrize( + "context", [None, CORE, PRIVATE, Context.of()], ids=["default", "core", "private", "empty"] +) +def test_a_model_round_trips_through_its_document_in_its_scope(context: Context | None) -> None: + """`from_json(m.to_json(), context=m.context) == m`: the document and the scope determine the model, and pickle carries both.""" + model = ZarrV3ArrayMetadata(ARRAY, context=context) + assert ZarrV3ArrayMetadata(model.to_json(), context=model.context) == model + loaded = pickle.loads(pickle.dumps(model)) + assert loaded == model + assert loaded.context == model.context + assert loaded.to_json() == model.to_json() + + +def test_error_a_model_whose_scope_does_not_pickle_says_so() -> None: + """A scope holding a definition with a local function does not pickle, and the model raises the error pickling a `Context` raises.""" + local = CodecDefinition( + name="acme.c", + configuration=Empty, + kind="bytes_bytes", + size="dynamic", + rules=lambda configuration, nested: iter(()), + ) + model = ZarrV3ArrayMetadata( + {**ARRAY, "codecs": ["bytes", "acme.c"]}, context=CORE.extended_with(local) + ) + with pytest.raises((pickle.PicklingError, AttributeError)): + pickle.dumps(model) From 8464b1f0a7e1fd53db07cbb55de14be6d3125d44 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Tue, 6 Oct 2026 14:37:22 +0200 Subject: [PATCH 34/94] feat(zarr-metadata): add update, with_context, refined_in and refines to the array model `update(**members)` merges JSON members into the document (UNSET removes one) and reads the result in the model's own scope, so no scope is passed in. `with_context(scope)` reads the same document in another scope. `refined_in(scope)` does the same but raises `ScopeConflictError` if the new scope reads any claimed name differently or not at all. `refines(other)` returns True when this model holds everything `other` holds. Assisted-by: ClaudeCode:claude-fable-5-1 --- .../src/zarr_metadata/model/_array.py | 68 +++++++- .../zarr-metadata/tests/model/test_pair.py | 148 ++++++++++++++++++ 2 files changed, 215 insertions(+), 1 deletion(-) diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_array.py b/packages/zarr-metadata/src/zarr_metadata/model/_array.py index cfa60b9814..7718b18049 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_array.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/_array.py @@ -44,7 +44,8 @@ spelled_canonically, ) from zarr_metadata.v3._registry import CORE_AND_EXTENSIONS, Context -from zarr_metadata.v3._scope import Claims, claims_of +from zarr_metadata.v3._scope import Claims, ScopeConflictError, claims_of +from zarr_metadata.v3._scope import refines as refines_field from zarr_metadata.v3.array import ZARR_V3_ARRAY_METADATA_STORE_KEY, ZarrV3ExtensionField if TYPE_CHECKING: @@ -286,6 +287,58 @@ def storage_transformers( self._reading.storage_transformers, ) + # --- changing --------------------------------------------------------- + + def update(self, **members: Unpack[ZarrV3ArrayMetadataUpdate]) -> ZarrV3ArrayMetadata: + """This model with `members`, JSON, in place of the document's, `UNSET` leaving one out, read in this model's own scope. + + `MetadataValidationError` when the document they make has a + problem, so members that go together are passed together: a + `shape` with a grid that fits it. + """ + document: dict[str, object] = {**self._document, **members} + for key, value in members.items(): + if value is UNSET: + del document[key] + return type(self)(document, context=self._context) + + def with_context(self, context: Context | None = None) -> ZarrV3ArrayMetadata: + """This document read in `context`, whatever that changes: a gain, a loss, a conflict. + + `MetadataValidationError` when the document has a problem there. + The reading is kept when `context` reads every claim identically. + """ + scope = CORE_AND_EXTENSIONS if context is None else context + if scope.disagreements(self._claims).agrees: + return self._of(self._document, scope, self._reading, self._members) + return type(self)(self._document, context=scope) + + def refined_in(self, context: Context | None = None) -> ZarrV3ArrayMetadata: + """This document read in `context`, which may claim what this scope left unclaimed and contradict nothing. + + `ScopeConflictError` naming each name `context` reads by another + definition, or by none, where this scope read it by one: a loss + of meaning is refused as a conflict is. `with_context` reads the + document in any scope. + """ + scope = CORE_AND_EXTENSIONS if context is None else context + found = scope.disagreements(self._claims) + if len(found.conflicts) != 0: + raise ScopeConflictError(found.conflicts) + return self.with_context(scope) + + def refines(self, other: ZarrV3ArrayMetadata) -> bool: + """Whether this model holds everything `other` holds: each field refines its counterpart, as `refines` orders fields, and every other member is the same, the fill value as the more informed data type spells it.""" + if type(other) is not type(self): + return False + mine = dict(self._reading.fields()) + theirs = dict(other._reading.fields()) + if mine.keys() != theirs.keys(): + return False + if not all(refines_field(mine[loc], theirs[loc]) for loc in mine): + return False + return _plain_key(self, self.data_type) == _plain_key(other, self.data_type) + # --- constructors ----------------------------------------------------- @classmethod @@ -369,6 +422,19 @@ def array_key(model: ZarrV3ArrayMetadata) -> tuple[object, ...]: ) +def _plain_key( + model: ZarrV3ArrayMetadata, data_type: Read[DataTypeDefinition[Any]] | Unclaimed +) -> tuple[object, ...]: + """What `refines` compares of a model other than its fields, the fill value spelled as `data_type` -- the more informed side's -- spells it.""" + return ( + model.shape, + json_text(spelled_canonically(data_type, model.fill_value)), + model.dimension_names, + json_text(dict(model.attributes)), + json_text(dict(model.extra_fields)), + ) + + def _fill_value_key(model: ZarrV3ArrayMetadata) -> str: """What `==` compares of `model`'s fill value: its canonical spelling as JSON text when a definition in scope read the data type, and the fill value as written when none did.""" if isinstance(model.data_type, Read): diff --git a/packages/zarr-metadata/tests/model/test_pair.py b/packages/zarr-metadata/tests/model/test_pair.py index 0ff8b68354..496c64e915 100644 --- a/packages/zarr-metadata/tests/model/test_pair.py +++ b/packages/zarr-metadata/tests/model/test_pair.py @@ -22,6 +22,7 @@ CodecDefinition, Context, Read, + ScopeConflictError, Unclaimed, ) @@ -164,3 +165,150 @@ def test_error_a_model_whose_scope_does_not_pickle_says_so() -> None: ) with pytest.raises((pickle.PicklingError, AttributeError)): pickle.dumps(model) + + +FLOAT: dict[str, Any] = { + **ARRAY, + "data_type": "float32", + "fill_value": "0x7fc00000", + "codecs": [{"name": "bytes", "configuration": {"endian": "little"}}], +} + + +@pytest.mark.parametrize( + ("model", "members", "expected"), + [ + ( + ZarrV3ArrayMetadata(ARRAY), + {"attributes": {"a": 1}}, + ZarrV3ArrayMetadata({**ARRAY, "attributes": {"a": 1}}), + ), + ( + ZarrV3ArrayMetadata({**ARRAY, "attributes": {"a": 1}}), + {"attributes": UNSET}, + ZarrV3ArrayMetadata(ARRAY), + ), + ( + ZarrV3ArrayMetadata(ARRAY, context=PRIVATE), + {"attributes": {"a": 1}}, + ZarrV3ArrayMetadata({**ARRAY, "attributes": {"a": 1}}, context=PRIVATE), + ), + ( + ZarrV3ArrayMetadata(ARRAY), + { + "shape": (6,), + "chunk_grid": {"name": "regular", "configuration": {"chunk_shape": (3,)}}, + }, + ZarrV3ArrayMetadata( + { + **ARRAY, + "shape": [6], + "chunk_grid": {"name": "regular", "configuration": {"chunk_shape": [3]}}, + } + ), + ), + ], + ids=["set", "unset", "update-keeps-scope", "members-together"], +) +def test_update_reads_new_members_in_the_models_own_scope( + model: ZarrV3ArrayMetadata, members: dict[str, Any], expected: ZarrV3ArrayMetadata +) -> None: + """`update` puts JSON members in place of the document's, `UNSET` removing one, and reads the result in the model's own scope, so a model read privately stays private.""" + updated = model.update(**members) + assert updated == expected + assert updated.context == model.context + + +def test_error_update_refuses_a_document_with_a_problem() -> None: + """`update` raises `MetadataValidationError` when the members make an invalid document, as the constructor does: two dimension names for one dimension.""" + with pytest.raises(MetadataValidationError): + ZarrV3ArrayMetadata(ARRAY).update(dimension_names=("x", "y")) + + +@pytest.mark.parametrize( + ("model", "context", "gained"), + [ + (ZarrV3ArrayMetadata(ARRAY, context=Context.of()), CORE, True), + (ZarrV3ArrayMetadata(ARRAY, context=CORE), CORE_AND_EXTENSIONS, False), + (ZarrV3ArrayMetadata(FLOAT, context=Context.of()), CORE, True), + ], + ids=["gain", "nothing-to-gain", "gain-data-type-with-fill-value"], +) +def test_refined_in_moves_a_model_up_the_order( + model: ZarrV3ArrayMetadata, context: Context, gained: bool +) -> None: + """`refined_in` reads the document in a scope that claims what this one left unclaimed and contradicts nothing; the result refines the model, keeps its document, and is the model itself when the scope reads nothing otherwise.""" + refined = model.refined_in(context) + assert refined.context == context + assert refined.to_json() == model.to_json() + assert refined.refines(model) + assert (refined == model) is not gained + if not gained: + assert refined.reading is model.reading + + +@pytest.mark.parametrize( + ("model", "context"), + [ + (ZarrV3ArrayMetadata(ARRAY), PRIVATE), + (ZarrV3ArrayMetadata(ARRAY, context=CORE), Context.of()), + ], + ids=["conflict", "loss"], +) +def test_error_refined_in_refuses_a_conflict_or_a_loss( + model: ZarrV3ArrayMetadata, context: Context +) -> None: + """`refined_in` raises `ScopeConflictError` naming each name the scope reads by another definition, or by none.""" + with pytest.raises(ScopeConflictError) as raised: + model.refined_in(context) + assert "bytes" in [conflict.key[1] for conflict in raised.value.conflicts] + + +def test_with_context_reads_the_document_in_any_scope() -> None: + """`with_context` reads the same document in another scope, whatever that changes: a private `bytes` is read as such, and a scope that claims nothing leaves every name unclaimed.""" + model = ZarrV3ArrayMetadata(ARRAY) + private = model.with_context(PRIVATE) + assert private.context == PRIVATE + assert private.codecs[0].definition == MY_BYTES + assert isinstance(model.with_context(Context.of()).codecs[0], Unclaimed) + assert model.with_context(None).context == CORE_AND_EXTENSIONS + + +def test_error_with_context_refuses_a_document_the_scope_reads_with_a_problem() -> None: + """`with_context` raises `MetadataValidationError` when the document has a problem in the new scope: a gzip `level` the core definition refuses.""" + loose = ZarrV3ArrayMetadata( + {**ARRAY, "codecs": ["bytes", {"name": "gzip", "configuration": {"level": 12}}]}, + context=Context.of(), + ) + with pytest.raises(MetadataValidationError): + loose.with_context(CORE) + + +@pytest.mark.parametrize( + ("upper", "lower", "expected"), + [ + (ZarrV3ArrayMetadata(ARRAY), ZarrV3ArrayMetadata(ARRAY, context=Context.of()), True), + (ZarrV3ArrayMetadata(ARRAY, context=Context.of()), ZarrV3ArrayMetadata(ARRAY), False), + (ZarrV3ArrayMetadata(ARRAY), ZarrV3ArrayMetadata(SPELLED_OUT), True), + (ZarrV3ArrayMetadata(ARRAY, context=PRIVATE), ZarrV3ArrayMetadata(ARRAY), False), + (ZarrV3ArrayMetadata({**ARRAY, "attributes": {"a": 1}}), ZarrV3ArrayMetadata(ARRAY), False), + ( + ZarrV3ArrayMetadata(FLOAT, context=CORE), + ZarrV3ArrayMetadata(FLOAT, context=Context.of()), + True, + ), + ], + ids=[ + "gain", + "loss", + "equal", + "conflict", + "other-members-differ", + "fill-value-spelled-by-the-informed-side", + ], +) +def test_refines_orders_models_by_information( + upper: ZarrV3ArrayMetadata, lower: ZarrV3ArrayMetadata, expected: bool +) -> None: + """A model refines another when every field refines its counterpart and every other member is the same, the fill value compared as the more informed data type spells it.""" + assert upper.refines(lower) is expected From a0de8a4f87b32295e3925ae80b45f3686baf8377 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Tue, 6 Oct 2026 14:41:27 +0200 Subject: [PATCH 35/94] feat(zarr-metadata)!: make ZarrV3GroupMetadata and ZarrV3ConsolidatedMetadata (document, context) pairs The group model follows the array model. Its consolidated metadata is a view of the same pair: each nested document becomes a model in the group's scope, built from the group's single read, not read again. `ZarrV3ConsolidatedMetadata(member, context=None)` still reads a member on its own. Assisted-by: ClaudeCode:claude-fable-5-1 --- .../src/zarr_metadata/model/_array.py | 41 -- .../src/zarr_metadata/model/_group.py | 611 ++++++++++-------- .../zarr-metadata/tests/model/test_pair.py | 120 +++- 3 files changed, 448 insertions(+), 324 deletions(-) diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_array.py b/packages/zarr-metadata/src/zarr_metadata/model/_array.py index 7718b18049..d3dd4988c2 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_array.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/_array.py @@ -442,47 +442,6 @@ def _fill_value_key(model: ZarrV3ArrayMetadata) -> str: return json_text(model.fill_value) -def array_json(model: ZarrV3ArrayMetadata) -> ZarrV3ArrayMetadataJSON: - """`model`'s document, holding the model's own values: what a writer serializes without copying.""" - return cast("ZarrV3ArrayMetadataJSON", model._document) # pyright: ignore[reportPrivateUsage] - - -def array_model( - reading: ZarrV3ArrayMetadataReading, members: ArrayMembersV3 -) -> ZarrV3ArrayMetadata: - """The model of a document its reading found nothing wrong with, its document rebuilt from the reading. - - A shim for the group reader until it hands the nested documents down - with their readings: the document is rebuilt from what was read, so a - member written empty is written back as the read holds it. - """ - document: dict[str, JSONValue] = { - "zarr_format": 3, - "node_type": "array", - "shape": members.shape, - "data_type": cast("Read[Any] | Unclaimed", reading.data_type).json, - "chunk_grid": cast("Read[Any] | Unclaimed", reading.chunk_grid).json, - "chunk_key_encoding": cast("Read[Any] | Unclaimed", reading.chunk_key_encoding).json, - "fill_value": members.fill_value, - "codecs": tuple( - cast("Read[Any] | Unclaimed", stage.codec).json for stage in reading.pipeline - ), - } - if members.dimension_names is not UNSET: - document["dimension_names"] = members.dimension_names - if len(members.attributes) != 0: - document["attributes"] = members.attributes - if len(reading.storage_transformers) != 0: - document["storage_transformers"] = tuple( - cast("Read[Any] | Unclaimed", transformer).json - for transformer in reading.storage_transformers - ) - document.update(members.extra_fields) - return ZarrV3ArrayMetadata._of( # pyright: ignore[reportPrivateUsage] - document, CORE_AND_EXTENSIONS, reading, members - ) - - def read_array_metadata_v3( value: object, *, context: Context = CORE_AND_EXTENSIONS ) -> ZarrV3ArrayMetadataReading: diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_group.py b/packages/zarr-metadata/src/zarr_metadata/model/_group.py index 32ae4b9b4f..39e1f8cc04 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_group.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/_group.py @@ -5,6 +5,7 @@ import dataclasses from collections.abc import Callable, Mapping from dataclasses import dataclass, field +from types import MappingProxyType from typing import TYPE_CHECKING, Any, Final, Literal, TypeGuard, TypeVar, cast from typing_extensions import TypeAliasType, TypedDict, Unpack @@ -30,16 +31,12 @@ from zarr_metadata._sentinel import UNSET from zarr_metadata.model._array import ( ZarrV3ArrayMetadata, - array_json, - array_key, - array_model, must_understand_subset, read_array_metadata_v3, ) from zarr_metadata.model._validation import ( GROUP_METADATA_REQUIRED_KEYS_V3, GROUP_METADATA_STANDARD_KEYS_V3, - NO_SCOPE, ArrayMembersV3, StoreKey, ZarrV3ArrayMetadataReading, @@ -51,7 +48,6 @@ members_past_the_levels, missing_keys, other_members, - overlapping, parse_group_metadata_v2, read_array_v3, unexpected_keys, @@ -61,6 +57,7 @@ from zarr_metadata.v2.group import ZARR_V2_GROUP_METADATA_STORE_KEY from zarr_metadata.v3._hierarchy import NodeType, hierarchy_problems, path_faults, said from zarr_metadata.v3._registry import CORE_AND_EXTENSIONS, Context +from zarr_metadata.v3._scope import Claims, ScopeConflictError, claims_of from zarr_metadata.v3.array import ZarrV3ExtensionField from zarr_metadata.v3.consolidated import ZARR_V3_CONSOLIDATED_METADATA_KEY from zarr_metadata.v3.group import ZARR_V3_GROUP_METADATA_STORE_KEY, ZarrV3GroupMetadataJSON @@ -89,113 +86,141 @@ class ZarrV3GroupMetadataUpdate(TypedDict, total=False, extra_items=ZarrV3Extens attributes: Mapping[str, JSONValue] | UNSET -@dataclass(frozen=True, slots=True, kw_only=True) class ZarrV3GroupMetadata: - """In-memory model of a v3 group metadata document. - - A canonical, semantically lossless representation of the `zarr.json` - content for a group. The `consolidated_metadata` reference-implementation - convention is modeled as a typed field holding the model of each - document it holds; every other unknown top-level key lands in - `extra_fields` verbatim. A model holds no scope, as - `ZarrV3ArrayMetadata` holds none: `from_json`, `create_default` and - `update` each take the one they read new JSON in. It checks itself - when it is built, as `ZarrV3ArrayMetadata` does, and holds its members - as the read refines them: its own members -- each document its - consolidated metadata holds is a model, which checked itself -- or the - constructor raises `MetadataValidationError` with every problem. + """A v3 group document, and the scope it was read in. + + The model is the pair, as `ZarrV3ArrayMetadata` is: `to_json` is the + document as written, refined, and `context` the scope. `attributes` + and `extra_fields` are views of what the read refined. The + `consolidated_metadata` reference-implementation convention is a + `ZarrV3ConsolidatedMetadata` view of the same pair: each document it + holds is a model of this scope, built from this one read. Built only + by reading: the constructor reads `document` in `context` and raises + `MetadataValidationError` with every problem, a nested document's + located under `consolidated_metadata.metadata.`. """ - zarr_format: Literal[3] = field(default=3, init=False) - node_type: Literal["group"] = field(default="group", init=False) - attributes: dict[str, JSONValue] - consolidated_metadata: ZarrV3ConsolidatedMetadata | UNSET - extra_fields: dict[str, ZarrV3ExtensionField] + __slots__ = ( + "_claims", + "_consolidated", + "_context", + "_document", + "_key", + "_members", + "_reading", + ) - def __post_init__(self) -> None: - # The runtime half of the annotations. - extra = cast("object", self.extra_fields) - if not isinstance(extra, Mapping): - msg = f"extra_fields: expected a mapping of names to JSON, got {extra!r}" - raise TypeError(msg) - consolidated = cast("object", self.consolidated_metadata) - if consolidated is not UNSET and not isinstance(consolidated, ZarrV3ConsolidatedMetadata): - msg = f"consolidated_metadata: expected ZarrV3ConsolidatedMetadata or UNSET, got {consolidated!r}" - raise TypeError(msg) - # Its own members, each document its consolidated metadata holds - # being a model that checked itself, held as the read refines them. - reading, members = read_group_v3(_own_document(self, whole=True), NO_SCOPE) - problems = (*overlapping(self.extra_fields, _RESERVED_V3, "group"), *reading.problems) - if len(problems) != 0: - raise MetadataValidationError(problems) - own = cast("GroupMembersV3", members) - object.__setattr__(self, "attributes", own.attributes) - object.__setattr__(self, "extra_fields", own.extra_fields) + zarr_format: Final = 3 + node_type: Final = "group" + + def __init__(self, document: object, context: Context | None = None) -> None: + scope = CORE_AND_EXTENSIONS if context is None else context + reading, members = read_group_v3(document, scope) + if members is None or len(reading.problems) != 0: + raise MetadataValidationError(reading.problems) + refined, _ = refine_json(document) + self._adopt(cast("dict[str, JSONValue]", refined), scope, reading, members) @classmethod - def create_default( + def _of( cls, - *, - context: Context = CORE_AND_EXTENSIONS, - **members: Unpack[ZarrV3GroupMetadataJSONPartial], + document: dict[str, JSONValue], + context: Context, + reading: ZarrV3GroupMetadataReading, + members: GroupMembersV3, ) -> ZarrV3GroupMetadata: - """A group with no attributes, or the one `members` of its document make of it, read in `context`; `MetadataValidationError` when its document has a problem.""" - return cls.from_json({"zarr_format": 3, "node_type": "group", **members}, context=context) + """A model of a document a read found nothing wrong with, holding that reading: no second read.""" + model = object.__new__(cls) + model._adopt(document, context, reading, members) + return model + + def _adopt( + self, + document: dict[str, JSONValue], + context: Context, + reading: ZarrV3GroupMetadataReading, + members: GroupMembersV3, + ) -> None: + self._document = document + self._context = context + self._reading = reading + self._members = members + if members.consolidated is UNSET: + self._consolidated: ZarrV3ConsolidatedMetadata | UNSET = UNSET + else: + member = cast("dict[str, JSONValue]", document[ZARR_V3_CONSOLIDATED_METADATA_KEY]) + documents = cast("dict[str, JSONValue]", member["metadata"]) + self._consolidated = ZarrV3ConsolidatedMetadata._of( # pyright: ignore[reportPrivateUsage] + member, + context, + _models(documents, context, reading.consolidated, members.consolidated), + ) + self._key = group_key(self) + self._claims = MappingProxyType(claims_of(reading.fields())) - def update( - self, *, context: Context, **members: Unpack[ZarrV3GroupMetadataUpdate] - ) -> ZarrV3GroupMetadata: - """This model with `members` in their place, each read in `context`; `UNSET` leaves one out. + # --- the pair --------------------------------------------------------- + + @property + def context(self) -> Context: + """The scope the document was read in, which `update` reads new members in.""" + return self._context + + @property + def reading(self) -> ZarrV3GroupMetadataReading: + """The document as the scope read it: each document its consolidated metadata holds, as read.""" + return self._reading - Only the members given are read. A `consolidated_metadata` given - has each document it holds read in `context`; left out, the group - keeps the models it holds, however each was read, and reads none - of them again. `MetadataValidationError` when the document the - members make has a problem. + @property + def claims(self) -> Claims: + """What the reading claimed of each name the document writes, in the documents it holds, keyed as the scope files it.""" + return self._claims + + def to_json(self) -> ZarrV3GroupMetadataJSON: + """The document as written, refined, sharing nothing with the model.""" + return cast("ZarrV3GroupMetadataJSON", copied(self._document)) + + def to_key_value( + self, *, indent: int | str | None = None + ) -> Mapping[ZarrV3GroupMetadataStoreKey, bytes]: + """The document as a store holds it: JSON bytes at `zarr.json`, indented by `indent`. + + `NaN`, `Infinity` and `-Infinity` in `attributes` are written as + those bare tokens, as zarr-python writes them, which a strict JSON + parser refuses. """ - document = {**_own_document(self, whole=True), **members} - for key, value in members.items(): - if value is UNSET: - del document[key] - updated = type(self).from_json(document, context=context) - if ZARR_V3_CONSOLIDATED_METADATA_KEY in members: - return updated - return dataclasses.replace(updated, consolidated_metadata=self.consolidated_metadata) + return {ZARR_V3_GROUP_METADATA_STORE_KEY: dump_store_json(self._document, indent=indent)} + + def __repr__(self) -> str: + return f"{type(self).__name__}({self._document!r}, context={self._context!r})" def __eq__(self, other: object) -> bool: - """Whether `other` models the same group: the same document, however each is spelled, as `group_key` says; equal models hash alike.""" if type(other) is not type(self): return NotImplemented - return group_key(self) == group_key(cast("ZarrV3GroupMetadata", other)) + return self._key == cast("ZarrV3GroupMetadata", other)._key def __hash__(self) -> int: - return hash(group_key(self)) + return hash(self._key) - def to_json(self) -> ZarrV3GroupMetadataJSON: - """The document as JSON, sharing no mutable state with the model. + def __reduce__(self) -> tuple[type[ZarrV3GroupMetadata], tuple[object, Context]]: + # The pair, read again on load. + return type(self), (self._document, self._context) - `attributes` when not empty, `consolidated_metadata` when set, and - each extra field as held. - """ - return cast("ZarrV3GroupMetadataJSON", copied(cast("JSONValue", group_json(self)))) + # --- typed views ------------------------------------------------------ - @classmethod - def from_json( - cls, data: object, *, context: Context = CORE_AND_EXTENSIONS - ) -> ZarrV3GroupMetadata: - """The model of `data`, a v3 group document read in `context`, with each document its consolidated metadata holds. + @property + def attributes(self) -> Mapping[str, JSONValue]: + """The attributes, a read-only view; empty when the document writes none.""" + return MappingProxyType(cast("dict[str, JSONValue]", self._members.attributes)) - `MetadataValidationError` with every problem the read finds. A - `consolidated_metadata` of `null`, which a zarr-python 3.0.x bug - wrote, is a value the document wrote, and no object: a problem, - as the spec says an object; a reader of such a store strips the - key first. A member the spec does not define is held in - `extra_fields`. - """ - reading = read_group_metadata_v3(data, context=context) - if reading.metadata is None: - raise MetadataValidationError(reading.problems) - return reading.metadata + @property + def extra_fields(self) -> Mapping[str, ZarrV3ExtensionField]: + """Each member the spec does not define, `consolidated_metadata` apart, by name: a read-only view.""" + return MappingProxyType(self._members.extra_fields) + + @property + def consolidated_metadata(self) -> ZarrV3ConsolidatedMetadata | UNSET: + """The `consolidated_metadata` member as a model of this scope; `UNSET` when the document writes none.""" + return self._consolidated @property def must_understand_fields(self) -> dict[str, ZarrV3ExtensionField]: @@ -209,178 +234,191 @@ def must_understand_fields(self) -> dict[str, ZarrV3ExtensionField]: """ return must_understand_subset(self.extra_fields) + # --- changing --------------------------------------------------------- + + def update(self, **members: Unpack[ZarrV3GroupMetadataUpdate]) -> ZarrV3GroupMetadata: + """This model with `members`, JSON, in place of the document's, `UNSET` leaving one out, read in this model's own scope. + + A `consolidated_metadata` given is read; left out, the document's + is read again as part of the whole. `MetadataValidationError` when + the document they make has a problem. + """ + document: dict[str, object] = {**self._document, **members} + for key, value in members.items(): + if value is UNSET: + del document[key] + return type(self)(document, context=self._context) + + def with_context(self, context: Context | None = None) -> ZarrV3GroupMetadata: + """This document read in `context`, whatever that changes; `MetadataValidationError` when it has a problem there. The reading is kept when `context` reads every claim identically.""" + scope = CORE_AND_EXTENSIONS if context is None else context + if scope.disagreements(self._claims).agrees: + return self._of(self._document, scope, self._reading, self._members) + return type(self)(self._document, context=scope) + + def refined_in(self, context: Context | None = None) -> ZarrV3GroupMetadata: + """This document read in `context`, which may claim what this scope left unclaimed and contradict nothing; `ScopeConflictError` naming each name it reads by another definition, or by none.""" + scope = CORE_AND_EXTENSIONS if context is None else context + found = scope.disagreements(self._claims) + if len(found.conflicts) != 0: + raise ScopeConflictError(found.conflicts) + return self.with_context(scope) + + def refines(self, other: ZarrV3GroupMetadata) -> bool: + """Whether this model holds everything `other` holds: the same attributes and extra fields, and consolidated metadata whose every document refines its counterpart.""" + if type(other) is not type(self): + return False + if json_text(dict(self.attributes)) != json_text(dict(other.attributes)): + return False + if json_text(dict(self.extra_fields)) != json_text(dict(other.extra_fields)): + return False + mine, theirs = self._consolidated, other._consolidated + if mine is UNSET or theirs is UNSET: + return mine is UNSET and theirs is UNSET + return mine.refines(theirs) + + # --- constructors ----------------------------------------------------- + + @classmethod + def create_default( + cls, + *, + context: Context | None = None, + **members: Unpack[ZarrV3GroupMetadataJSONPartial], + ) -> ZarrV3GroupMetadata: + """A group with no attributes, or the one `members` of its document make of it, read in `context`; `MetadataValidationError` when its document has a problem.""" + return cls({"zarr_format": 3, "node_type": "group", **members}, context=context) + + @classmethod + def from_json(cls, data: object, *, context: Context | None = None) -> ZarrV3GroupMetadata: + """The model of `data`, a v3 group document read in `context`, with each document its consolidated metadata holds. + + `MetadataValidationError` with every problem the read finds. A + `consolidated_metadata` of `null`, which a zarr-python 3.0.x bug + wrote, is a value the document wrote, and no object: a problem, + as the spec says an object; `read_repaired_node_metadata_v3` reads + such a store. A member the spec does not define is held in + `extra_fields`. + """ + return cls(data, context=context) + @classmethod def from_key_value( - cls, mapping: Mapping[StoreKey, bytes], *, context: Context = CORE_AND_EXTENSIONS + cls, mapping: Mapping[StoreKey, bytes], *, context: Context | None = None ) -> ZarrV3GroupMetadata: """The model of the group document at `zarr.json` in `mapping`, read in `context`. `MetadataValidationError` when the key is missing, its bytes are not JSON, or the document is not valid. """ - return cls.from_json( - load_store_json(mapping, ZARR_V3_GROUP_METADATA_STORE_KEY), context=context - ) - - def to_key_value( - self, *, indent: int | str | None = None - ) -> Mapping[ZarrV3GroupMetadataStoreKey, bytes]: - """The document as a store holds it: JSON bytes at `zarr.json`, indented by `indent`. - - A model, and each it holds, was checked when it was built, so its - document is written as it is. `NaN`, `Infinity` and `-Infinity` in - `attributes` are written as those bare tokens, as zarr-python writes - them, which a strict JSON parser refuses. - """ - return {ZARR_V3_GROUP_METADATA_STORE_KEY: dump_store_json(group_json(self), indent=indent)} + return cls(load_store_json(mapping, ZARR_V3_GROUP_METADATA_STORE_KEY), context=context) -@dataclass(frozen=True, slots=True, kw_only=True) class ZarrV3ConsolidatedMetadata: - """In-memory model of v3 inline consolidated metadata. - - Models the reference-implementation convention where consolidated metadata - is embedded as an extension field on a group's `zarr.json`. Each entry in - `metadata` is the model of a complete child document, array or group. - `must_understand` is `False` by declaration, not given, as `kind` is - its literal. Each document it holds is a model, which checked itself - when it was built, at its own root: so a model built by hand of models - can hold one that sits, as a whole, deeper than a reader walks, and - write a document its validator refuses there, as one holding a - hand-built `Read` can, and a chain of them, each holding the last, is - bounded by nothing but the interpreter, which `to_json`, `==`, `hash` - and pickle walk a frame a level; `from_json` builds only what a reader - read. - The documents and the group make the hierarchy below the group, the - group its root, each at its node's path in it without the leading `/`: - the node at `/a/b` at `a/b`. + """A group's inline `consolidated_metadata` member, and the scope it was read in. + + Models the reference-implementation convention where consolidated + metadata is embedded as an extension field on a group's `zarr.json`. + `metadata` maps each path to the model of the complete document there, + array or group, of this scope: a view of the group's pair when a group + holds it, built from the group's one read; or of its own pair, when + the member is read on its own. `kind` is `inline` and `must_understand` + `False`, by declaration. The documents and the group make the + hierarchy below the group, the group its root, each at its node's + path without the leading `/`: the node at `/a/b` at `a/b`. """ - kind: Literal["inline"] = field(default="inline", init=False) - must_understand: Literal[False] = field(default=False, init=False) - metadata: dict[str, ZarrV3ArrayMetadata | ZarrV3GroupMetadata] + __slots__ = ("_context", "_document", "_key", "_metadata") - def __post_init__(self) -> None: - # The runtime half of the annotations. - for path, node in cast("dict[object, object]", self.metadata).items(): - if not isinstance(path, str): - msg = f"metadata: a document's path is a string, got {path!r}" - raise TypeError(msg) - if not isinstance(node, (ZarrV3ArrayMetadata, ZarrV3GroupMetadata)): - msg = f"metadata[{path!r}]: expected a v3 array or group model, got {node!r}" - raise TypeError(msg) - problems: list[ValidationProblem] = [] - node_types: dict[str, NodeType | None] = {} - for path, node in self.metadata.items(): - faults = _key_problems(path) - problems.extend(faults) - if len(faults) == 0: - node_types[path] = _model_node_type(node) - problems.extend(_hierarchy_problems(node_types)) - for path, node in self.metadata.items(): - if path in node_types: - problems.extend( - _nested_listing_problems( - path, _model_listing(node), node_types, _model_node_type - ) - ) + kind: Final = "inline" + must_understand: Final = False + + def __init__(self, member: object, context: Context | None = None) -> None: + scope = CORE_AND_EXTENSIONS if context is None else context + # The member sits under a group's key wherever it is read, so the + # levels a reader walks are counted from there, as in the group. + readings, members, problems = _read_consolidated_v3( + member, scope, (ZARR_V3_CONSOLIDATED_METADATA_KEY,) + ) if len(problems) != 0: raise MetadataValidationError(problems) - object.__setattr__(self, "metadata", dict(self.metadata)) + refined, _ = refine_json(member) + document = cast("dict[str, JSONValue]", refined) + documents = cast("dict[str, JSONValue]", document["metadata"]) + self._adopt(document, scope, _models(documents, scope, readings, members)) + + @classmethod + def _of( + cls, + document: dict[str, JSONValue], + context: Context, + metadata: dict[str, ZarrV3NodeMetadata], + ) -> ZarrV3ConsolidatedMetadata: + """The member of a group a read found nothing wrong with, holding the models that read built.""" + model = object.__new__(cls) + model._adopt(document, context, metadata) + return model + + def _adopt( + self, + document: dict[str, JSONValue], + context: Context, + metadata: dict[str, ZarrV3NodeMetadata], + ) -> None: + self._document = document + self._context = context + self._metadata = metadata + self._key = consolidated_key(self) + + @property + def context(self) -> Context: + """The scope the documents were read in.""" + return self._context + + @property + def metadata(self) -> Mapping[str, ZarrV3NodeMetadata]: + """The model of each document, by its path below the group: a read-only view.""" + return MappingProxyType(self._metadata) + + def to_json(self) -> ZarrV3ConsolidatedMetadataJSON: + """The member as written, refined, sharing nothing with the model.""" + return cast("ZarrV3ConsolidatedMetadataJSON", copied(self._document)) + + def __repr__(self) -> str: + return f"{type(self).__name__}({self._document!r}, context={self._context!r})" def __eq__(self, other: object) -> bool: - """Whether `other` holds the same documents at the same paths, each as its model compares; equal ones hash alike.""" if type(other) is not type(self): return NotImplemented - return consolidated_key(self) == consolidated_key(cast("ZarrV3ConsolidatedMetadata", other)) + return self._key == cast("ZarrV3ConsolidatedMetadata", other)._key def __hash__(self) -> int: - return hash(consolidated_key(self)) + return hash(self._key) - def to_json(self) -> ZarrV3ConsolidatedMetadataJSON: - """The `consolidated_metadata` member as JSON, sharing no mutable state with the model: its `kind`, `must_understand: false`, and each document by its path.""" - return cast( - "ZarrV3ConsolidatedMetadataJSON", copied(cast("JSONValue", consolidated_json(self))) + def __reduce__(self) -> tuple[type[ZarrV3ConsolidatedMetadata], tuple[object, Context]]: + return type(self), (self._document, self._context) + + def refines(self, other: ZarrV3ConsolidatedMetadata) -> bool: + """Whether every document this holds refines the one `other` holds at the same path, and neither holds a path the other does not.""" + if self._metadata.keys() != other._metadata.keys(): + return False + return all( + _node_refines(self._metadata[path], other._metadata[path]) for path in self._metadata ) @classmethod def from_json( - cls, data: object, *, context: Context = CORE_AND_EXTENSIONS + cls, data: object, *, context: Context | None = None ) -> ZarrV3ConsolidatedMetadata: - """The model of `data`, a group's `consolidated_metadata` member, each document read once in `context`, as the array or group its `node_type` says. + """The model of `data`, a group's `consolidated_metadata` member, each document read once in `context`, as the array or group its `node_type` says; `MetadataValidationError` with every problem found.""" + return cls(data, context=context) - `MetadataValidationError` with every problem found. - """ - # The member sits under a group's key wherever it is read, so the - # levels a reader walks are counted from there, as in the group. - readings, members, problems = _read_consolidated_v3( - data, context, (ZARR_V3_CONSOLIDATED_METADATA_KEY,) - ) - if len(problems) != 0: - raise MetadataValidationError(problems) - return construct(cls, metadata=_models(readings, members)[1]) - - -def group_json(model: ZarrV3GroupMetadata) -> ZarrV3GroupMetadataJSON: - """`model`'s document as JSON, holding the model's own values: what `to_key_value` serializes, which changes nothing, and `to_json` copies.""" - return cast("ZarrV3GroupMetadataJSON", _group_document(model, array_json, group_json)) - - -def consolidated_json(model: ZarrV3ConsolidatedMetadata) -> ZarrV3ConsolidatedMetadataJSON: - """`model`, a `consolidated_metadata` member, as JSON, holding the model's own values, as `group_json` holds them.""" - return cast( - "ZarrV3ConsolidatedMetadataJSON", _consolidated_document(model, array_json, group_json) - ) - -def _own_document(model: ZarrV3GroupMetadata, *, whole: bool) -> dict[str, object]: - """`model`'s document without its consolidated metadata: the members that are the group's own. - - Empty `attributes` too when `whole`, as the constructor reads them, - where a writer leaves them out. An extra field named as a member the - document declares is no member of it, which the constructor reports. - """ - out: dict[str, object] = {"zarr_format": model.zarr_format, "node_type": model.node_type} - if whole or len(model.attributes) > 0: - out["attributes"] = model.attributes - out.update((key, value) for key, value in model.extra_fields.items() if key not in _RESERVED_V3) - return out - - -_RESERVED_V3: Final = GROUP_METADATA_STANDARD_KEYS_V3 | {ZARR_V3_CONSOLIDATED_METADATA_KEY} -"""The members of a v3 group document the model holds apart from its extra fields.""" - - -def _group_document( - model: ZarrV3GroupMetadata, - array: Callable[[ZarrV3ArrayMetadata], object], - group: Callable[[ZarrV3GroupMetadata], object], -) -> dict[str, object]: - """`model`'s document, each document its consolidated metadata holds as `array` or `group` gives it.""" - out = _own_document(model, whole=False) - if model.consolidated_metadata is not UNSET: - # Consolidated metadata is a known non-core top-level JSON field. - out[ZARR_V3_CONSOLIDATED_METADATA_KEY] = _consolidated_document( - model.consolidated_metadata, array, group - ) - return out - - -def _consolidated_document( - model: ZarrV3ConsolidatedMetadata, - array: Callable[[ZarrV3ArrayMetadata], object], - group: Callable[[ZarrV3GroupMetadata], object], -) -> dict[str, object]: - """The member, each document it holds as `array` or `group` gives it.""" - # `must_understand` is False by declaration, as `kind` is its literal. - return { - "kind": model.kind, - "must_understand": False, - "metadata": { - path: array(node) if isinstance(node, ZarrV3ArrayMetadata) else group(node) - for path, node in model.metadata.items() - }, - } +def _node_refines(node: ZarrV3NodeMetadata, other: ZarrV3NodeMetadata) -> bool: + """Whether `node` refines `other`, as the models of one kind refine each other; models of two kinds do not.""" + if isinstance(node, ZarrV3ArrayMetadata): + return isinstance(other, ZarrV3ArrayMetadata) and node.refines(other) + return isinstance(other, ZarrV3GroupMetadata) and node.refines(other) def _no_documents() -> Mapping[str, ZarrV3NodeMetadataReading]: @@ -574,7 +612,8 @@ def read_group_metadata_v3( reading, members = read_group_v3(value, context) if members is None: return reading - return _with_models(reading, members) + refined, _ = refine_json(value) + return _with_models(reading, members, cast("dict[str, JSONValue]", refined), context) def read_group_v3( @@ -697,16 +736,16 @@ def group_key(model: ZarrV3GroupMetadata) -> tuple[object, ...]: """What `==` and `hash` compare of a v3 group model: its attributes and extra fields as JSON text, and what its consolidated metadata holds, by `consolidated_key`.""" consolidated = model.consolidated_metadata return ( - json_text(model.attributes), - UNSET if consolidated is UNSET else consolidated_key(consolidated), - json_text(model.extra_fields), + json_text(dict(model.attributes)), + UNSET if consolidated is UNSET else consolidated._key, # pyright: ignore[reportPrivateUsage] + json_text(dict(model.extra_fields)), ) def consolidated_key(model: ZarrV3ConsolidatedMetadata) -> tuple[object, ...]: """What `==` and `hash` compare of consolidated metadata: each document's key, by its path, in path order.""" return tuple( - (path, array_key(node) if isinstance(node, ZarrV3ArrayMetadata) else group_key(node)) + (path, node._key) # pyright: ignore[reportPrivateUsage] for path, node in sorted(model.metadata.items(), key=lambda item: item[0]) ) @@ -772,20 +811,6 @@ def _an(node_type: NodeType) -> str: return "an array" if node_type == "array" else "a group" -def _model_node_type(node: ZarrV3ArrayMetadata | ZarrV3GroupMetadata) -> NodeType: - """The node type a model is.""" - return "array" if isinstance(node, ZarrV3ArrayMetadata) else "group" - - -def _model_listing( - node: ZarrV3ArrayMetadata | ZarrV3GroupMetadata, -) -> Mapping[str, ZarrV3ArrayMetadata | ZarrV3GroupMetadata]: - """What a model lists in its own consolidated metadata: nothing, for an array or a group with none.""" - if isinstance(node, ZarrV3GroupMetadata) and node.consolidated_metadata is not UNSET: - return node.consolidated_metadata.metadata - return {} - - def _reading_listing(reading: ZarrV3NodeMetadataReading) -> Mapping[str, ZarrV3NodeMetadataReading]: """What a reading lists in its own consolidated metadata: nothing, for an array's or one of no node type.""" if isinstance(reading, ZarrV3GroupMetadataReading): @@ -827,50 +852,72 @@ def _below_faults(path: str) -> list[str]: def _with_models( - reading: ZarrV3GroupMetadataReading, members: GroupMembersV3 + reading: ZarrV3GroupMetadataReading, + members: GroupMembersV3, + document: dict[str, JSONValue], + context: Context, ) -> ZarrV3GroupMetadataReading: - """`reading`, holding the model of each document its consolidated metadata holds that has no problem, and its own when it has none.""" - readings, models = ( - (reading.consolidated, {}) - if members.consolidated is UNSET - else _models(reading.consolidated, members.consolidated) - ) + """`reading`, holding the model of each document its consolidated metadata holds that has no problem, and its own when it has none: each built from `document`, this read's, and `context`.""" + # A document with problems may hold a member that is no object, or + # whose `metadata` is none: then no document in it was read. + member = document.get(ZARR_V3_CONSOLIDATED_METADATA_KEY) + entries = member.get("metadata") if isinstance(member, Mapping) else None + if members.consolidated is UNSET or not isinstance(entries, Mapping): + readings: dict[str, ZarrV3NodeMetadataReading] = dict(reading.consolidated) + else: + readings = _readings_with_models( + cast("Mapping[str, JSONValue]", entries), + context, + reading.consolidated, + members.consolidated, + ) if len(reading.problems) != 0: return dataclasses.replace(reading, consolidated=readings) - model = construct( - ZarrV3GroupMetadata, - attributes=cast("dict[str, JSONValue]", members.attributes), - consolidated_metadata=( - UNSET - if members.consolidated is UNSET - else construct(ZarrV3ConsolidatedMetadata, metadata=models) - ), - extra_fields=members.extra_fields, - ) + model = ZarrV3GroupMetadata._of(document, context, reading, members) # pyright: ignore[reportPrivateUsage] return dataclasses.replace(reading, consolidated=readings, metadata=model) def _models( + documents: Mapping[str, JSONValue], + context: Context, readings: Mapping[str, ZarrV3NodeMetadataReading], members: Mapping[str, ArrayMembersV3 | GroupMembersV3], -) -> tuple[ - dict[str, ZarrV3NodeMetadataReading], - dict[str, ZarrV3ArrayMetadata | ZarrV3GroupMetadata], -]: - """Each reading, holding its model when a model can be built of its document, and those models, by path.""" +) -> dict[str, ZarrV3NodeMetadata]: + """The model of each document in `documents`, a `metadata` member's, a model can be built of: built from its reading and members, in `context`, not read again.""" + models: dict[str, ZarrV3NodeMetadata] = {} + for path, child in members.items(): + reading = readings[path] + document = cast("dict[str, JSONValue]", documents[path]) + if isinstance(reading, ZarrV3ArrayMetadataReading): + models[path] = ZarrV3ArrayMetadata._of( # pyright: ignore[reportPrivateUsage] + document, context, reading, cast("ArrayMembersV3", child) + ) + elif isinstance(reading, ZarrV3GroupMetadataReading) and len(reading.problems) == 0: + models[path] = ZarrV3GroupMetadata._of( # pyright: ignore[reportPrivateUsage] + document, context, reading, cast("GroupMembersV3", child) + ) + return models + + +def _readings_with_models( + documents: Mapping[str, JSONValue], + context: Context, + readings: Mapping[str, ZarrV3NodeMetadataReading], + members: Mapping[str, ArrayMembersV3 | GroupMembersV3], +) -> dict[str, ZarrV3NodeMetadataReading]: + """Each reading, holding its model when one can be built of its document, as `read_group_metadata_v3` hands them back.""" held: dict[str, ZarrV3NodeMetadataReading] = dict(readings) - models: dict[str, ZarrV3ArrayMetadata | ZarrV3GroupMetadata] = {} for path, child in members.items(): reading = readings[path] + document = cast("dict[str, JSONValue]", documents[path]) if isinstance(reading, ZarrV3ArrayMetadataReading): - array = array_model(reading, cast("ArrayMembersV3", child)) - held[path], models[path] = dataclasses.replace(reading, metadata=array), array + array = ZarrV3ArrayMetadata._of( # pyright: ignore[reportPrivateUsage] + document, context, reading, cast("ArrayMembersV3", child) + ) + held[path] = dataclasses.replace(reading, metadata=array) elif isinstance(reading, ZarrV3GroupMetadataReading): - group = _with_models(reading, cast("GroupMembersV3", child)) - held[path] = group - if group.metadata is not None: - models[path] = group.metadata - return held, models + held[path] = _with_models(reading, cast("GroupMembersV3", child), document, context) + return held def validate_group_metadata_v3( diff --git a/packages/zarr-metadata/tests/model/test_pair.py b/packages/zarr-metadata/tests/model/test_pair.py index 496c64e915..68fb14ee56 100644 --- a/packages/zarr-metadata/tests/model/test_pair.py +++ b/packages/zarr-metadata/tests/model/test_pair.py @@ -2,8 +2,9 @@ from __future__ import annotations +import dataclasses import pickle -from collections.abc import Mapping +from collections.abc import Iterator, Mapping from typing import Any import pytest @@ -13,17 +14,22 @@ UNSET, MetadataValidationError, ZarrV3ArrayMetadata, + ZarrV3ConsolidatedMetadata, + ZarrV3GroupMetadata, read_array_metadata_v3, ) +from zarr_metadata.v3.codec.bytes import BYTES_CODEC from zarr_metadata.v3.codec.crc32c import Empty from zarr_metadata.v3.definition import ( CORE, CORE_AND_EXTENSIONS, CodecDefinition, Context, + Nested, Read, ScopeConflictError, Unclaimed, + ValidationProblem, ) ARRAY: dict[str, Any] = { @@ -312,3 +318,115 @@ def test_refines_orders_models_by_information( ) -> None: """A model refines another when every field refines its counterpart and every other member is the same, the fill value compared as the more informed data type spells it.""" assert upper.refines(lower) is expected + + +GROUP: dict[str, Any] = {"zarr_format": 3, "node_type": "group", "attributes": {"g": 1}} +CONSOLIDATED: dict[str, Any] = { + **GROUP, + "consolidated_metadata": { + "kind": "inline", + "must_understand": False, + "metadata": {"a": ARRAY, "b": GROUP, "b/c": SPELLED_OUT}, + }, +} + + +def test_a_group_is_its_document_and_scope_and_its_nested_models_share_them() -> None: + """A group model is the pair, and each document its consolidated metadata holds is a model of the same scope, built from the group's one read, writing its document as written.""" + group = ZarrV3GroupMetadata(CONSOLIDATED, context=CORE) + held = group.consolidated_metadata + assert isinstance(held, ZarrV3ConsolidatedMetadata) + assert set(held.metadata) == {"a", "b", "b/c"} + for node in held.metadata.values(): + assert node.context is group.context + assert held.metadata["b/c"].to_json()["data_type"] == {"name": "uint8"} + assert group.to_json() == refine_json(CONSOLIDATED)[0] + assert ZarrV3GroupMetadata(GROUP).consolidated_metadata is UNSET + + +def test_a_group_reads_each_nested_field_once() -> None: + """Building a group with consolidated metadata asks a definition's rules once per nested field: the models are built from the read, not read again.""" + calls: list[int] = [] + + def counted(configuration: object, nested: Nested) -> Iterator[ValidationProblem]: + calls.append(1) + return iter(()) + + scope = CORE.extended_with(dataclasses.replace(BYTES_CODEC, rules=counted)) + ZarrV3GroupMetadata(CONSOLIDATED, context=scope) + assert len(calls) == 2 # `a` and `b/c` each hold one bytes codec + + +@pytest.mark.parametrize( + ("left", "right", "equal"), + [ + (ZarrV3GroupMetadata(CONSOLIDATED), ZarrV3GroupMetadata(CONSOLIDATED, context=CORE), True), + ( + ZarrV3GroupMetadata(CONSOLIDATED), + ZarrV3GroupMetadata(CONSOLIDATED, context=PRIVATE), + False, + ), + (ZarrV3GroupMetadata(GROUP), ZarrV3GroupMetadata({**GROUP, "attributes": {"g": 2}}), False), + ], + ids=["unused-definitions", "private-bytes-inside", "attributes"], +) +def test_groups_are_equal_by_what_their_documents_mean( + left: ZarrV3GroupMetadata, right: ZarrV3GroupMetadata, equal: bool +) -> None: + """A group compares by its attributes, extra fields and each nested model's meaning; equal groups hash alike.""" + assert (left == right) is equal + if equal: + assert hash(left) == hash(right) + + +@pytest.mark.parametrize("document", [GROUP, CONSOLIDATED], ids=["group", "consolidated"]) +def test_a_group_round_trips_through_its_document_in_its_scope(document: dict[str, Any]) -> None: + """`from_json(g.to_json(), context=g.context) == g`, and pickle carries the pair.""" + group = ZarrV3GroupMetadata(document, context=CORE) + assert ZarrV3GroupMetadata(group.to_json(), context=group.context) == group + assert pickle.loads(pickle.dumps(group)) == group + + +def test_group_update_with_context_and_refined_in_behave_as_the_arrays_do() -> None: + """`update` reads in the group's scope and keeps its consolidated metadata unless given; `refined_in` moves every nested model up the order; `with_context` reads all of it in another scope.""" + group = ZarrV3GroupMetadata(CONSOLIDATED, context=Context.of()) + assert group.update(attributes={"g": 2}).consolidated_metadata == group.consolidated_metadata + refined = group.refined_in(CORE) + assert refined.refines(group) + assert refined.context == CORE + held = refined.consolidated_metadata + assert isinstance(held, ZarrV3ConsolidatedMetadata) + array = held.metadata["a"] + assert isinstance(array, ZarrV3ArrayMetadata) + assert isinstance(array.codecs[0], Read) + with pytest.raises(ScopeConflictError): + ZarrV3GroupMetadata(CONSOLIDATED).refined_in(PRIVATE) + private = group.with_context(PRIVATE).consolidated_metadata + assert isinstance(private, ZarrV3ConsolidatedMetadata) + private_array = private.metadata["a"] + assert isinstance(private_array, ZarrV3ArrayMetadata) + assert private_array.codecs[0].definition == MY_BYTES + + +def test_consolidated_metadata_reads_on_its_own() -> None: + """`ZarrV3ConsolidatedMetadata(member, context)` reads the member as a group's read reads it, each document a model of that scope.""" + member = CONSOLIDATED["consolidated_metadata"] + held = ZarrV3ConsolidatedMetadata(member, context=CORE) + assert held == ZarrV3GroupMetadata(CONSOLIDATED, context=CORE).consolidated_metadata + assert held.to_json() == refine_json(member)[0] + assert ZarrV3ConsolidatedMetadata.from_json(member) == ZarrV3ConsolidatedMetadata(member) + + +def test_error_a_group_with_a_nested_problem_is_refused_at_the_nested_path() -> None: + """A nested document's problem is the group's, located under `consolidated_metadata.metadata.`.""" + with pytest.raises(MetadataValidationError) as raised: + ZarrV3GroupMetadata( + { + **CONSOLIDATED, + "consolidated_metadata": { + **CONSOLIDATED["consolidated_metadata"], + "metadata": {"a": {**ARRAY, "shape": [-1]}}, + }, + } + ) + assert raised.value.problems[0].loc == ("consolidated_metadata", "metadata", "a", "shape") From 0613866c027831942177c534031b4cd277d7d5d7 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Tue, 6 Oct 2026 14:42:15 +0200 Subject: [PATCH 36/94] refactor(zarr-metadata)!: remove the held-field machinery Models are now built only by reading a document, so `NO_SCOPE`, `overlapping`, `read_field`, `_read_as` and the `held=` parameter of `read_array_v3` are removed. `construct` stays because the v2 dataclass models still use it. Assisted-by: ClaudeCode:claude-fable-5-1 --- .../src/zarr_metadata/model/_validation.py | 58 +++---------------- .../zarr-metadata/tests/model/test_pair.py | 11 ++++ 2 files changed, 18 insertions(+), 51 deletions(-) diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_validation.py b/packages/zarr-metadata/src/zarr_metadata/model/_validation.py index 989d6097ae..9988959e62 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_validation.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/_validation.py @@ -26,7 +26,7 @@ import dataclasses import json -from collections.abc import Collection, Mapping, Sequence +from collections.abc import Mapping, Sequence from dataclasses import dataclass from typing import TYPE_CHECKING, Any, Final, TypeGuard, TypeVar, cast, get_args, get_origin @@ -56,10 +56,8 @@ DataTypeDefinition, Definition, Lengths, - Read, Resolved, StorageTransformerDefinition, - Unclaimed, chunk_grid_lengths, field_kind, fields_of, @@ -463,22 +461,9 @@ def fields(self) -> Iterator[tuple[Loc, Resolved[Any]]]: yield from fields_of(transformer, ("storage_transformers", index)) -NO_SCOPE: Final = Context.of() -"""A scope of no definitions, which a model's own document is read in: its fields are read already, and nothing else in it is a field.""" - M = TypeVar("M") -def overlapping( - extra_fields: Mapping[str, object], reserved: frozenset[str], node: str -) -> tuple[ValidationProblem, ...]: - """The problem of a model's extra fields named as a member its document declares, which the model holds apart; none when they are not.""" - if set(extra_fields).isdisjoint(reserved): - return () - message = f"Extra fields cannot overlap with standard Zarr V3 {node} metadata fields" - return (ValidationProblem(("extra_fields",), message, "invalid_value"),) - - def construct(model: type[M], /, **members: object) -> M: """A `model` of `members` a read found nothing wrong with: as its constructor builds one, without checking them again. @@ -518,43 +503,14 @@ class ArrayMembersV3: extra_fields: dict[str, JSONValue] -def read_field( - value: object, - kind: type[Definition[Any]], - context: Context, - loc: Loc, - *, - held: Collection[object], -) -> tuple[Resolved[Any], tuple[ValidationProblem, ...]]: - """`value`, one of a document's fields, as `context` reads it -- or as it is, when it is one of `held`: the fields a model holds, read already. - - A model's own fields come back this way, so a model checks itself - without reading them again, and a field read in one scope keeps the - definition that read it, however the scope it is handed on in differs. - Any other field object -- in a document a caller hands in, or among - the members `update` is given -- is not JSON, and is refused as such, - so nothing built by hand passes as read but what a model holds. - """ - if _read_as(value, kind) and any(value is field for field in held): - return cast("Resolved[Any]", value), () - return resolve(value, kind, context, loc) - - -def _read_as(value: object, kind: type[Definition[Any]]) -> bool: - """Whether `value` is a field a scope already read, as a `kind`, which a model holds.""" - return isinstance(value, (Read, Unclaimed)) and value.read_as is kind - - def read_array_v3( - value: object, context: Context, *, held: Collection[object] = (), at: Loc = () + value: object, context: Context, *, at: Loc = () ) -> tuple[ZarrV3ArrayMetadataReading, ArrayMembersV3 | None]: """`value`, a v3 array document, as `context` read it, without its model, and its other members refined; None when it has a problem. - `read_array_metadata_v3` builds the model from the two. `held` are the - fields a model holds, read already, which are taken as they are where - the document holds them; any other field object in the document is - refused as not JSON. `at` is where the document sits in the one handed - in -- a document consolidated metadata holds sits three levels below + `read_array_metadata_v3` builds the model from the two. A field object + in the document -- a `Read` built by hand -- is not JSON, and is refused + as such. `at` is where the document sits in the one handed in -- a document consolidated metadata holds sits three levels below its group's -- so the levels a reader walks are counted from that one's root; the problems are located in this document. """ @@ -584,7 +540,7 @@ def read_array_v3( read: dict[str, Resolved[Any]] = {} for key, kind in _EXTENSION_POINTS_V3: if key in doc: - read[key], found = read_field(doc[key], kind, context, (*at, key), held=held) + read[key], found = resolve(doc[key], kind, context, (*at, key)) problems.extend(within(found, at)) # The fill value is JSON, and judged by the data type the scope read, # when there is one: a data type nothing in scope claims leaves it @@ -611,7 +567,7 @@ def read_array_v3( else: listed[key] = [] for index, entry in enumerate(entries): - resolved, found = read_field(entry, kind, context, (*at, key, index), held=held) + resolved, found = resolve(entry, kind, context, (*at, key, index)) listed[key].append(resolved) problems.extend(within(found, at)) # The codecs are read as a pipeline, the first handed the grid's chunks diff --git a/packages/zarr-metadata/tests/model/test_pair.py b/packages/zarr-metadata/tests/model/test_pair.py index 68fb14ee56..3e27ad6cbe 100644 --- a/packages/zarr-metadata/tests/model/test_pair.py +++ b/packages/zarr-metadata/tests/model/test_pair.py @@ -430,3 +430,14 @@ def test_error_a_group_with_a_nested_problem_is_refused_at_the_nested_path() -> } ) assert raised.value.problems[0].loc == ("consolidated_metadata", "metadata", "a", "shape") + + +def test_the_held_field_machinery_is_gone() -> None: + """A v3 model is built only by reading its document, so nothing in the package takes fields read already, or re-reads a model in an empty scope: `NO_SCOPE`, `overlapping` and `held` are gone.""" + import inspect + + import zarr_metadata.model._validation as validation + + assert not hasattr(validation, "NO_SCOPE") + assert not hasattr(validation, "overlapping") + assert "held" not in inspect.signature(validation.read_array_v3).parameters From 13c3b5c1f6d5eab0717d800b23735e6ae126da12 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Tue, 6 Oct 2026 14:50:35 +0200 Subject: [PATCH 37/94] test(zarr-metadata): migrate tests to document construction and update Tests that built v3 models from typed fields, used `dataclasses.replace`, or passed a scope to `update` now build from documents and change with `update`. Two tests about extra fields overlapping a standard member are removed; a document cannot express that case. The migration found two bugs in the new models: a document with `NaN` in attributes refined to None (the whole document is now refined as user data), and a group built its nested models twice (the readings now hold the same models the group holds). Assisted-by: ClaudeCode:claude-fable-5-1 --- .../src/zarr_metadata/model/_array.py | 6 +- .../src/zarr_metadata/model/_group.py | 27 ++- .../zarr-metadata/tests/model/test_array.py | 108 +++------ .../tests/model/test_construction.py | 208 +++++++++++------- .../zarr-metadata/tests/model/test_group.py | 92 ++++---- .../zarr-metadata/tests/model/test_pair.py | 4 +- .../tests/model/test_sentinel.py | 23 +- 7 files changed, 236 insertions(+), 232 deletions(-) diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_array.py b/packages/zarr-metadata/src/zarr_metadata/model/_array.py index d3dd4988c2..d1d9c93841 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_array.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/_array.py @@ -15,7 +15,7 @@ ValidationProblem, copied, json_text, - refine_json, + refine_user_data, with_input, ) from zarr_metadata._sentinel import UNSET @@ -135,7 +135,7 @@ def __init__(self, document: object, context: Context | None = None) -> None: reading, members = read_array_v3(document, scope) if members is None: raise MetadataValidationError(reading.problems) - refined, _ = refine_json(document) + refined, _ = refine_user_data(document) self._adopt(cast("dict[str, JSONValue]", refined), scope, reading, members) @classmethod @@ -459,7 +459,7 @@ def read_array_metadata_v3( reading, members = read_array_v3(value, context) if members is None: return reading - refined, _ = refine_json(value) + refined, _ = refine_user_data(value) document = cast("dict[str, JSONValue]", refined) model = ZarrV3ArrayMetadata._of(document, context, reading, members) # pyright: ignore[reportPrivateUsage] return dataclasses.replace(reading, metadata=model) diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_group.py b/packages/zarr-metadata/src/zarr_metadata/model/_group.py index 39e1f8cc04..3c7cfd0258 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_group.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/_group.py @@ -118,7 +118,7 @@ def __init__(self, document: object, context: Context | None = None) -> None: reading, members = read_group_v3(document, scope) if members is None or len(reading.problems) != 0: raise MetadataValidationError(reading.problems) - refined, _ = refine_json(document) + refined, _ = refine_user_data(document) self._adopt(cast("dict[str, JSONValue]", refined), scope, reading, members) @classmethod @@ -155,6 +155,8 @@ def _adopt( context, _models(documents, context, reading.consolidated, members.consolidated), ) + # One model per document: the readings hold the models a read + # built, which the group holds too. self._key = group_key(self) self._claims = MappingProxyType(claims_of(reading.fields())) @@ -342,7 +344,7 @@ def __init__(self, member: object, context: Context | None = None) -> None: ) if len(problems) != 0: raise MetadataValidationError(problems) - refined, _ = refine_json(member) + refined, _ = refine_user_data(member) document = cast("dict[str, JSONValue]", refined) documents = cast("dict[str, JSONValue]", document["metadata"]) self._adopt(document, scope, _models(documents, scope, readings, members)) @@ -612,8 +614,13 @@ def read_group_metadata_v3( reading, members = read_group_v3(value, context) if members is None: return reading - refined, _ = refine_json(value) - return _with_models(reading, members, cast("dict[str, JSONValue]", refined), context) + # A document nested past the levels a reader walks refines to nothing: + # it has problems, and no model is built of it or of what it holds. + refined, _ = refine_user_data(value) + document: dict[str, JSONValue] = ( + cast("dict[str, JSONValue]", refined) if isinstance(refined, Mapping) else {} + ) + return _with_models(reading, members, document, context) def read_group_v3( @@ -871,10 +878,11 @@ def _with_models( reading.consolidated, members.consolidated, ) + held = dataclasses.replace(reading, consolidated=readings) if len(reading.problems) != 0: - return dataclasses.replace(reading, consolidated=readings) - model = ZarrV3GroupMetadata._of(document, context, reading, members) # pyright: ignore[reportPrivateUsage] - return dataclasses.replace(reading, consolidated=readings, metadata=model) + return held + model = ZarrV3GroupMetadata._of(document, context, held, members) # pyright: ignore[reportPrivateUsage] + return dataclasses.replace(held, metadata=model) def _models( @@ -887,6 +895,11 @@ def _models( models: dict[str, ZarrV3NodeMetadata] = {} for path, child in members.items(): reading = readings[path] + if reading.metadata is not None: + # The reading holds the model a read built already: one model + # per document, which `read_group_metadata_v3` hands back too. + models[path] = reading.metadata + continue document = cast("dict[str, JSONValue]", documents[path]) if isinstance(reading, ZarrV3ArrayMetadataReading): models[path] = ZarrV3ArrayMetadata._of( # pyright: ignore[reportPrivateUsage] diff --git a/packages/zarr-metadata/tests/model/test_array.py b/packages/zarr-metadata/tests/model/test_array.py index 46f81e7283..95dd77b6f8 100644 --- a/packages/zarr-metadata/tests/model/test_array.py +++ b/packages/zarr-metadata/tests/model/test_array.py @@ -24,7 +24,6 @@ ZarrV2ArrayMetadata, ZarrV2ArrayMetadataPartial, ZarrV3ArrayMetadata, - ZarrV3ConsolidatedMetadata, ZarrV3GroupMetadata, is_array_metadata_v2, is_array_metadata_v3, @@ -239,9 +238,10 @@ def test_v3_a_data_type_with_nothing_to_configure_is_written_by_its_bare_name( "codecs": [{"name": "bytes", "configuration": {"endian": "little"}}], } model = ZarrV3ArrayMetadata.from_json(document) - # The field alone is spelled as the document spells it, so a reader - # handed either takes it. - assert model.to_json()["data_type"] == model.data_type.to_json() == "int32" + # The document is written as it was written; the field alone, by its + # bare name, which every reader takes. + assert model.to_json()["data_type"] == data_type + assert model.data_type.to_json() == "int32" def test_v3_dimension_names_included_when_present() -> None: @@ -292,9 +292,10 @@ def test_v3_single_storage_transformer_included() -> None: def test_v3_no_storage_transformers_omitted() -> None: - """V3 to_json omits storage_transformers when empty.""" - out = ZarrV3ArrayMetadata.create_default(storage_transformers=()).to_json() - assert "storage_transformers" not in out + """V3 to_json omits storage_transformers when the document wrote none, and writes an empty one as written.""" + assert "storage_transformers" not in ZarrV3ArrayMetadata.create_default().to_json() + written = dict(ZarrV3ArrayMetadata.create_default(storage_transformers=()).to_json()) + assert written["storage_transformers"] == () # --- V3 extra fields ------------------------------------------------------- @@ -307,15 +308,6 @@ def test_v3_extra_fields_merged() -> None: assert model.to_json()["my_ext"] == {"must_understand": False} -def test_v3_extra_fields_overlapping_standard_field_rejected() -> None: - """Constructing a V3 model with an extra field that collides with a standard key is rejected.""" - with pytest.raises(ValueError): - dataclasses.replace( - ZarrV3ArrayMetadata.create_default(), - extra_fields={"shape": {"must_understand": False}}, - ) - - # --- V3 key/value ---------------------------------------------------------- @@ -412,11 +404,7 @@ def test_update_returns_new_instance( ) -> None: """update returns a new instance with the field replaced, leaving the original unchanged.""" base = model_cls.create_default(shape=(10,)) - updated = ( - base.update(context=CORE_AND_EXTENSIONS, shape=(20,)) - if isinstance(base, ZarrV3ArrayMetadata) - else base.update(shape=(20,)) - ) + updated = base.update(shape=(20,)) assert updated.shape == (20,) assert base.shape == (10,) # original unchanged assert isinstance(updated, model_cls) @@ -434,11 +422,7 @@ def test_update_no_args_returns_equal_model( ) -> None: """update with no arguments returns a model equal to the original.""" base = model_cls.create_default() - updated = ( - base.update(context=CORE_AND_EXTENSIONS) - if isinstance(base, ZarrV3ArrayMetadata) - else base.update() - ) + updated = base.update() assert updated == base @@ -448,12 +432,12 @@ def test_update_no_args_returns_equal_model( def test_update_can_add_an_extension_member() -> None: """update can add a member the spec does not define, which the model holds in extra_fields.""" base = ZarrV3ArrayMetadata.create_default() - updated = base.update(context=CORE_AND_EXTENSIONS, my_ext={"must_understand": False}) + updated = base.update(my_ext={"must_understand": False}) assert updated.extra_fields == {"my_ext": {"must_understand": False}} -def test_update_reads_only_the_members_it_is_given_in_its_scope() -> None: - """Each field the model holds is kept as it was read; the members given are read in `context`. +def test_update_reads_every_member_in_the_models_own_scope() -> None: + """`update` reads the document it makes in the scope the model was read in, so a name that scope leaves unclaimed stays unclaimed; `with_context` is how another scope reads it. `zstd` is an extension, which `CORE` leaves unclaimed and unjudged. """ @@ -461,9 +445,10 @@ def test_update_reads_only_the_members_it_is_given_in_its_scope() -> None: zstd: ZarrV3NamedConfigJSON = {"name": "zstd", "configuration": {"level": 3, "checksum": False}} base = ZarrV3ArrayMetadata.create_default(context=CORE, codecs=(little, zstd)) assert isinstance(base.codecs[1], Unclaimed) - kept = base.update(context=CORE_AND_EXTENSIONS, attributes={"k": 1}) - assert kept.codecs[1] is base.codecs[1] - given = base.update(context=CORE_AND_EXTENSIONS, codecs=(little, zstd)) + kept = base.update(attributes={"k": 1}) + assert kept.codecs[1] == base.codecs[1] + assert kept.context == CORE + given = base.with_context(CORE_AND_EXTENSIONS).update(codecs=(little, zstd)) assert isinstance(given.codecs[1], Read) @@ -473,9 +458,9 @@ def test_update_leaves_out_a_member_given_as_unset() -> None: shape=(2,), dimension_names=("x",), attributes={"a": 1}, my_ext={"must_understand": False} ) for updated, member in ( - (base.update(context=CORE_AND_EXTENSIONS, dimension_names=UNSET), "dimension_names"), - (base.update(context=CORE_AND_EXTENSIONS, attributes=UNSET), "attributes"), - (base.update(context=CORE_AND_EXTENSIONS, my_ext=UNSET), "my_ext"), + (base.update(dimension_names=UNSET), "dimension_names"), + (base.update(attributes=UNSET), "attributes"), + (base.update(my_ext=UNSET), "my_ext"), ): written = updated.to_json() assert member not in written @@ -566,16 +551,15 @@ def test_two_models_are_one_array_when_their_fill_values_are_one_value( assert hash(models[0]) == hash(models[1]) # A group holding the arrays says so too. groups = [ - ZarrV3GroupMetadata( - attributes={}, - consolidated_metadata=ZarrV3ConsolidatedMetadata(metadata={"a": model}), - extra_fields={}, + ZarrV3GroupMetadata.create_default( + consolidated_metadata={**_INLINE, "metadata": {"a": model.to_json()}} ) for model in models ] assert (groups[0] == groups[1]) is same +_INLINE: dict[str, Any] = {"kind": "inline", "must_understand": False} _NOSHUFFLE = {"cname": "lz4", "clevel": 5, "shuffle": "noshuffle", "blocksize": 0} @@ -627,10 +611,7 @@ def test_user_json_compares_as_text( ) -> None: """Attributes, which nothing interprets, compare as a document writes them: `NaN` is itself, `true` is not `1`, `-0.0` is not `0.0`.""" arrays = [ZarrV3ArrayMetadata.create_default(attributes=held) for held in (left, right)] - groups = [ - ZarrV3GroupMetadata(attributes=held, consolidated_metadata=UNSET, extra_fields={}) - for held in (left, right) - ] + groups = [ZarrV3GroupMetadata.create_default(attributes=held) for held in (left, right)] v2 = [ZarrV2ArrayMetadata.create_default(attributes=held) for held in (left, right)] for models in (arrays, groups, v2): assert (models[0] == models[1]) is same @@ -685,7 +666,7 @@ def test_update_replaces_a_member_rather_than_merging_into_it() -> None: base = ZarrV3ArrayMetadata.create_default( a={"must_understand": False, "x": 1}, b={"must_understand": False} ) - updated = base.update(context=CORE_AND_EXTENSIONS, a={"must_understand": False}) + updated = base.update(a={"must_understand": False}) assert updated.extra_fields == { "a": {"must_understand": False}, "b": {"must_understand": False}, @@ -1832,14 +1813,14 @@ def test_error_update_refuses_a_document_with_a_problem( ) -> None: base = ZarrV3ArrayMetadata.create_default(shape=(4,)) with pytest.raises(MetadataValidationError) as raised: - base.update(context=CORE_AND_EXTENSIONS, **members) + base.update(**members) assert [(p.loc, p.kind) for p in raised.value.problems] == problems def test_error_update_refuses_to_leave_out_a_member_a_document_holds() -> None: base = ZarrV3ArrayMetadata.create_default(shape=(4,)) with pytest.raises(MetadataValidationError) as raised: - base.update(context=CORE_AND_EXTENSIONS, shape=UNSET) # pyright: ignore[reportArgumentType] + base.update(shape=UNSET) # pyright: ignore[reportArgumentType] assert [(p.loc, p.kind) for p in raised.value.problems] == [(("shape",), "missing_key")] @@ -2045,16 +2026,6 @@ def test_from_key_value_missing_key_kind() -> None: ) -def test_extra_fields_overlap_raises_metadata_error() -> None: - """The extra-fields overlap invariant raises MetadataValidationError (a ValueError).""" - with pytest.raises(MetadataValidationError, match="Extra fields") as exc_info: - dataclasses.replace( - ZarrV3ArrayMetadata.create_default(), - extra_fields={"shape": {"must_understand": False}}, - ) - assert [p.kind for p in exc_info.value.problems] == ["invalid_value"] - - # --- Adversarial-probe fixes: documents that used to pass validation --------- @@ -2393,22 +2364,6 @@ def test_error_a_document_nested_deeper_than_a_reader_walks_is_a_problem() -> No ZarrV3ArrayMetadata.from_json(document) -def test_a_field_built_by_hand_and_given_to_the_constructor_is_taken_as_read() -> None: - """As the class says: a field built by hand is taken as read, what it holds unjudged, so a model can write a document its validator refuses. `Read` judging its own configuration is the follow-up #379 named.""" - trusted = Read( - json="gzip", name="gzip", definition=GZIP_CODEC, configuration={"level": 99, "window": 1} - ) - default = ZarrV3ArrayMetadata.create_default(shape=(2,)) - model = dataclasses.replace(default, codecs=(*default.codecs, trusted)) - assert model.codecs[-1] is trusted - assert [ - (problem.loc, problem.kind) for problem in validate_array_metadata_v3(model.to_json()) - ] == [ - (("codecs", 1, "configuration", "window"), "unknown_key"), - (("codecs", 1, "configuration", "level"), "invalid_value"), - ] - - def test_error_a_field_object_in_a_document_is_not_json() -> None: # A `Read` built by hand, with a configuration its definition refuses, # smuggled into a document: refused as what it is, so nothing built by @@ -2426,13 +2381,14 @@ def test_error_a_field_object_in_a_document_is_not_json() -> None: assert not is_array_metadata_v3(document) with pytest.raises(MetadataValidationError): ZarrV3ArrayMetadata.from_json(document) - # Nor does `update` take one among the members it is given: only the - # fields the model holds are taken as read. + # Nor does `update` take one among the members it is given: a model's + # own fields are no exception, since `update` reads JSON. model = ZarrV3ArrayMetadata.create_default(shape=(2,)) with pytest.raises(MetadataValidationError) as raised: - model.update(context=CORE, codecs=cast("Any", (model.codecs[0].to_json(), smuggled))) + model.update(codecs=cast("Any", (model.codecs[0].to_json(), smuggled))) assert [(p.loc, p.kind) for p in raised.value.problems] == [(("codecs", 1), "invalid_type")] - assert model.update(context=CORE, codecs=(cast("Any", model.codecs[0]),)) == model + with pytest.raises(MetadataValidationError): + model.update(codecs=(cast("Any", model.codecs[0]),)) def test_a_problem_shows_an_integer_too_long_to_write_by_its_size( diff --git a/packages/zarr-metadata/tests/model/test_construction.py b/packages/zarr-metadata/tests/model/test_construction.py index c64450c0cc..d164767e75 100644 --- a/packages/zarr-metadata/tests/model/test_construction.py +++ b/packages/zarr-metadata/tests/model/test_construction.py @@ -1,9 +1,10 @@ """A model checks itself when it is built, as pydantic's `__init__` does. -Built by hand, or changed as `dataclasses.replace` changes one, a model -whose document has a problem is refused at the change, with every problem, -so none is ever invalid. A model a read builds is not read a second time, -and `to_key_value` writes a model as it is. +A v3 model is built only by reading its document, and changed only by +`update`, which reads the document it makes: a document with a problem is +refused, with every problem, so no model is ever invalid. A v2 model, a +dataclass still, is refused at a `dataclasses.replace` into an invalid +one. `to_key_value` writes a model as it is, reading nothing. """ from __future__ import annotations @@ -64,21 +65,27 @@ ) +_INLINE: dict[str, Any] = {"kind": "inline", "must_understand": False} + + @pytest.mark.parametrize( "model", - [ - ARRAY, - GROUP, - GROUP.consolidated_metadata, - V2_ARRAY, - V2_GROUP, - V2_CONSOLIDATED, - ], - ids=["array", "group", "consolidated", "v2-array", "v2-group", "v2-consolidated"], + [ARRAY, GROUP, GROUP.consolidated_metadata], + ids=["array", "group", "consolidated"], ) -def test_a_model_is_built_as_a_read_builds_it(model: object) -> None: - # The constructor checks what the read checked, and of the same members - # builds the same model. +def test_a_v3_model_is_built_of_its_document_in_its_scope(model: object) -> None: + """A v3 model is its document read in its scope: built again of the two, it is the same model.""" + held = cast("Any", model) + assert type(held)(held.to_json(), context=held.context) == model + + +@pytest.mark.parametrize( + "model", + [V2_ARRAY, V2_GROUP, V2_CONSOLIDATED], + ids=["v2-array", "v2-group", "v2-consolidated"], +) +def test_a_v2_model_is_built_as_a_read_builds_it(model: object) -> None: + """A v2 model's constructor checks what the read checked, and of the same members builds the same model.""" held = cast("Any", model) members = { member.name: getattr(held, member.name) @@ -94,22 +101,30 @@ def test_a_model_is_built_as_a_read_builds_it(model: object) -> None: (ARRAY, {"shape": [4], "dimension_names": ["x"]}), (ARRAY, {"attributes": UserDict({"a": [1, None]})}), (GROUP, {"attributes": UserDict({"a": 1})}), + ], + ids=["array-lists", "array-mapping", "group-mapping"], +) +def test_a_v3_model_holds_its_members_as_a_read_refines_them( + model: object, changes: dict[str, object] +) -> None: + """Arrays as tuples and objects as dicts, as the read holds them, so a model updated with other containers is the model a read builds, and round-trips through its store.""" + changed = cast("Any", model).update(**changes) + assert changed == model + assert type(changed).from_key_value(changed.to_key_value()) == changed + + +@pytest.mark.parametrize( + ("model", "changes"), + [ (V2_ARRAY, {"shape": range(4, 5), "chunks": [4], "attributes": {"a": 1}}), (V2_CONSOLIDATED, {"metadata": {"a/.zattrs": UserDict({"x": 1})}}), ], - ids=[ - "array-lists", - "array-mapping", - "group-mapping", - "v2-sequences", - "v2-consolidated-mapping", - ], + ids=["v2-sequences", "v2-consolidated-mapping"], ) -def test_a_model_holds_its_members_as_a_read_refines_them( +def test_a_v2_model_holds_its_members_as_a_read_refines_them( model: object, changes: dict[str, object] ) -> None: - # Arrays as tuples and objects as dicts, as the read holds them, so a - # model built of other containers is the model a read builds. + """Arrays as tuples and objects as dicts, as the read holds them, so a v2 model built of other containers is the model a read builds.""" changed = dataclasses.replace(cast("Any", model), **changes) assert changed == model assert type(changed).from_key_value(changed.to_key_value()) == changed @@ -117,15 +132,34 @@ def test_a_model_holds_its_members_as_a_read_refines_them( @pytest.mark.parametrize( ("model", "member"), - [ - (ARRAY, "extra_fields"), - (GROUP, "attributes"), - (V2_GROUP, "attributes"), - (V2_CONSOLIDATED, "metadata"), - ], - ids=["array-extra-fields", "group-attributes", "v2-group-attributes", "v2-consolidated"], + [(ARRAY, "acme.x"), (GROUP, "attributes")], + ids=["array-extra-field", "group-attributes"], +) +def test_a_v3_model_shares_no_container_with_what_it_was_built_of( + model: object, member: str +) -> None: + """A model holds copies of the containers `update` is given: changing them afterwards changes nothing it holds.""" + held: dict[str, object] = {"must_understand": False, "z": {"y": 1}} + built = cast("Any", model).update(**{member: held}) + held["w"] = math.nan + cast("dict[str, object]", held["z"])["y"] = math.nan + expected = {"must_understand": False, "z": {"y": 1}} + if member == "attributes": + assert built.attributes == expected + else: + assert built.extra_fields[member] == expected + assert type(built).from_key_value(built.to_key_value()) == built + + +@pytest.mark.parametrize( + ("model", "member"), + [(V2_GROUP, "attributes"), (V2_CONSOLIDATED, "metadata")], + ids=["v2-group-attributes", "v2-consolidated"], ) -def test_a_model_shares_no_container_with_what_it_was_built_of(model: object, member: str) -> None: +def test_a_v2_model_shares_no_container_with_what_it_was_built_of( + model: object, member: str +) -> None: + """A v2 model holds copies of the containers it is built of.""" held: dict[str, object] = {"acme.x": {"must_understand": False}} built = dataclasses.replace(cast("Any", model), **{member: held}) held["acme.y"] = math.nan @@ -134,7 +168,8 @@ def test_a_model_shares_no_container_with_what_it_was_built_of(model: object, me assert type(built).from_key_value(built.to_key_value()) == built -def test_a_model_is_read_once_and_written_as_it_is() -> None: +def test_a_model_is_read_when_built_and_written_as_it_is() -> None: + """A model is read once, when it is built; `to_key_value` reads nothing; `update` reads the document it makes; a group reads the documents it holds once, as part of its own read.""" values: list[object] = [] def counted( @@ -146,18 +181,17 @@ def counted( counting = dataclasses.replace(INT8_DATA_TYPE, fill_value_rules=counted) scope = CORE_AND_EXTENSIONS.extended_with(counting) model = ZarrV3ArrayMetadata.create_default(context=scope, data_type="int8", fill_value=3) + assert values == [3] model.to_key_value() - # Built by hand, a model is read once, as its own fields read it. - dataclasses.replace(model, fill_value=4) + assert values == [3] + changed = model.update(fill_value=4) assert values == [3, 4] - # A group checks its own members: each document it holds checked itself. - group = ZarrV3GroupMetadata( - attributes={}, - consolidated_metadata=ZarrV3ConsolidatedMetadata(metadata={"a": model}), - extra_fields={}, + group = ZarrV3GroupMetadata.create_default( + context=scope, consolidated_metadata={**_INLINE, "metadata": {"a": changed.to_json()}} ) - dataclasses.replace(group, attributes={"b": 1}).to_key_value() - assert values == [3, 4] + assert values == [3, 4, 4] + group.to_key_value() + assert values == [3, 4, 4] @pytest.mark.parametrize( @@ -178,32 +212,9 @@ def counted( # Empty or not, a value that is no object is judged as one. (ARRAY, {"attributes": []}, [(("attributes",), "invalid_type")]), (ARRAY, {"attributes": None}, [(("attributes",), "invalid_type")]), - # Every problem: an extra field named as a member, and the rest. - ( - ARRAY, - {"extra_fields": {"shape": [1]}, "fill_value": math.nan}, - [(("extra_fields",), "invalid_value"), (("fill_value",), "invalid_value")], - ), (GROUP, {"attributes": {1: "a"}}, [(("attributes",), "invalid_type")]), (GROUP, {"attributes": ()}, [(("attributes",), "invalid_type")]), - (GROUP, {"extra_fields": {"acme": math.nan}}, [(("acme",), "invalid_value")]), - ( - GROUP.consolidated_metadata, - {"metadata": {"x": ARRAY, "x/a": ARRAY, "__b": ARRAY, "c/d": ARRAY}}, - [ - (("metadata", "__b"), "invalid_value"), - (("metadata", "x/a"), "invalid_value"), - (("metadata", "c"), "missing_key"), - ], - ), - (V2_ARRAY, {"order": "Q"}, [(("order",), "invalid_value")]), - (V2_ARRAY, {"chunks": (4, 4)}, [(("chunks",), "invalid_value")]), - (V2_GROUP, {"attributes": {1: "a"}}, [(("attributes",), "invalid_type")]), - ( - V2_CONSOLIDATED, - {"metadata": {"a/.zarray": {"x": math.nan}}}, - [(("metadata", "a/.zarray", "x"), "invalid_value")], - ), + (GROUP, {"acme": math.nan}, [(("acme",), "invalid_value")]), ], ids=[ "fill-value-not-json", @@ -212,21 +223,60 @@ def counted( "attribute-key", "attributes-empty-and-no-object", "attributes-null", - "extra-field-named-as-a-member-and-more", "group-attribute-key", "group-attributes-empty-and-no-object", "group-extension-not-json", - "consolidated-paths", + ], +) +def test_error_a_v3_model_updated_into_an_invalid_one_is_refused_at_the_change( + model: object, changes: dict[str, object], problems: list[tuple[tuple[str | int, ...], str]] +) -> None: + """`update` reads the document it makes, so a change that makes an invalid one is refused with every problem: no model is invalid, however it came to be.""" + with pytest.raises(MetadataValidationError) as raised: + cast("Any", model).update(**changes) + assert [(found.loc, found.kind) for found in raised.value.problems] == problems + + +def test_error_consolidated_metadata_of_documents_at_bad_paths_is_refused() -> None: + """The consolidated member read on its own refuses documents at paths no node has, and reports a group missing above one, as the group's read does.""" + documents = { + "x": ARRAY.to_json(), + "x/a": ARRAY.to_json(), + "__b": ARRAY.to_json(), + "c/d": ARRAY.to_json(), + } + with pytest.raises(MetadataValidationError) as raised: + ZarrV3ConsolidatedMetadata({**_INLINE, "metadata": documents}) + assert [(found.loc, found.kind) for found in raised.value.problems] == [ + (("metadata", "__b"), "invalid_value"), + (("metadata", "x/a"), "invalid_value"), + (("metadata", "c"), "missing_key"), + ] + + +@pytest.mark.parametrize( + ("model", "changes", "problems"), + [ + (V2_ARRAY, {"order": "Q"}, [(("order",), "invalid_value")]), + (V2_ARRAY, {"chunks": (4, 4)}, [(("chunks",), "invalid_value")]), + (V2_GROUP, {"attributes": {1: "a"}}, [(("attributes",), "invalid_type")]), + ( + V2_CONSOLIDATED, + {"metadata": {"a/.zarray": {"x": math.nan}}}, + [(("metadata", "a/.zarray", "x"), "invalid_value")], + ), + ], + ids=[ "v2-order", "v2-chunks-for-another-rank", "v2-group-attribute-key", "v2-consolidated-entry-not-json", ], ) -def test_error_a_model_changed_by_hand_into_an_invalid_one_is_refused_at_the_change( +def test_error_a_v2_model_changed_by_hand_into_an_invalid_one_is_refused_at_the_change( model: object, changes: dict[str, object], problems: list[tuple[tuple[str | int, ...], str]] ) -> None: - """As a read reads its document: no model is invalid, however it came to be.""" + """As a read reads its document: no v2 model is invalid, however it came to be.""" with pytest.raises(MetadataValidationError) as raised: dataclasses.replace(cast("Any", model), **changes) assert [(found.loc, found.kind) for found in raised.value.problems] == problems @@ -252,15 +302,13 @@ def test_error_construct_refuses_a_model_missing_a_member() -> None: construct(ZarrV2GroupMetadata) -@pytest.mark.parametrize("model", [ARRAY, GROUP], ids=["array", "group"]) -def test_error_extra_fields_are_a_mapping(model: object) -> None: - with pytest.raises(TypeError, match="extra_fields: expected a mapping of names to JSON"): - dataclasses.replace(cast("Any", model), extra_fields=[("acme", 1)]) - - def test_error_consolidated_metadata_paths_are_strings() -> None: - with pytest.raises(TypeError, match="a document's path is a string, got 1"): - ZarrV3ConsolidatedMetadata(metadata=cast("Any", {1: ARRAY})) + """A `metadata` member keyed by what is no string is a problem of the member, as the read reports it, not a `TypeError`.""" + with pytest.raises(MetadataValidationError) as raised: + ZarrV3ConsolidatedMetadata({**_INLINE, "metadata": {1: ARRAY.to_json()}}) + assert [(found.loc, found.kind) for found in raised.value.problems] == [ + (("metadata",), "invalid_type") + ] def test_construct_fills_a_member_from_its_default_factory() -> None: diff --git a/packages/zarr-metadata/tests/model/test_group.py b/packages/zarr-metadata/tests/model/test_group.py index 055eba86f7..fc8ce933bb 100644 --- a/packages/zarr-metadata/tests/model/test_group.py +++ b/packages/zarr-metadata/tests/model/test_group.py @@ -64,6 +64,8 @@ ) from zarr_metadata.v3.group import ZarrV3GroupMetadataJSONPartial +_INLINE_ENVELOPE: dict[str, Any] = {"kind": "inline", "must_understand": False} + # --- ZarrV3GroupMetadata --------------------------------------------------- @@ -107,26 +109,6 @@ def test_group_v3_json_extra_field_roundtrips_as_must_understand() -> None: assert model.must_understand_fields == {"ext": (1, 2)} -def test_group_v3_extra_fields_overlap_rejected() -> None: - """Constructing a v3 group model with extra_fields shadowing a standard key raises.""" - with pytest.raises(ValueError, match="Extra fields"): - ZarrV3GroupMetadata( - attributes={}, - consolidated_metadata=UNSET, - extra_fields={"node_type": {"name": "x", "must_understand": False}}, - ) - - -def test_group_v3_consolidated_extra_field_rejected() -> None: - """extra_fields may not shadow the consolidated_metadata convention key.""" - with pytest.raises(ValueError, match="Extra fields"): - ZarrV3GroupMetadata( - attributes={}, - consolidated_metadata=UNSET, - extra_fields={"consolidated_metadata": {"name": "x", "must_understand": False}}, - ) - - def test_group_v3_missing_required_key() -> None: """parse_group_metadata_v3 reports each missing required key.""" with pytest.raises(MetadataValidationError, match="node_type"): @@ -279,7 +261,7 @@ def test_group_v3_key_value_roundtrip() -> None: def test_group_v3_update() -> None: """update replaces the given fields and returns a new instance.""" base = ZarrV3GroupMetadata.create_default() - updated = base.update(context=CORE_AND_EXTENSIONS, attributes={"a": 1}) + updated = base.update(attributes={"a": 1}) assert updated.attributes == {"a": 1} assert base.attributes == {} @@ -444,8 +426,8 @@ def test_a_listing_key_too_long_to_write_is_shown_by_its_size( assert problem.message == f"non-string key an integer of {(10**5000).bit_length()} bits" -def test_a_model_built_of_models_trusts_them_as_it_trusts_its_fields() -> None: - """Each held model checked itself at its own root, so a child valid alone can sit too deep in a group built by hand, whose document its validator refuses at the cap, as one holding a hand-built `Read` can; a reader builds only within the cap, the consolidated member's from where it sits.""" +def test_a_group_reads_the_documents_it_holds_from_where_they_sit() -> None: + """A child valid alone, nested to the last level a reader walks, sits three levels deeper as a document a group holds: the group's read refuses it at the cap, as the validator does, and so does the consolidated member read on its own, from where it sits under the group's key.""" # The innermost object sits at the last level a reader walks, alone; # three deeper as a document a group holds. nested: dict[str, object] = {} @@ -454,12 +436,11 @@ def test_a_model_built_of_models_trusts_them_as_it_trusts_its_fields() -> None: child = ZarrV3GroupMetadata.from_json( {"zarr_format": 3, "node_type": "group", "attributes": {"a": nested}} ) - group = dataclasses.replace( - ZarrV3GroupMetadata.create_default(), - consolidated_metadata=ZarrV3ConsolidatedMetadata(metadata={"a": child}), - ) + written = { + **ZarrV3GroupMetadata.create_default().to_json(), + "consolidated_metadata": {**_INLINE_ENVELOPE, "metadata": {"a": child.to_json()}}, + } depth = f"nested deeper than the {JSON_DEPTH} levels a reader walks" - written = group.to_json() assert [(len(p.loc), p.message) for p in validate_group_metadata_v3(written)] == [ (JSON_DEPTH, depth) ] @@ -667,28 +648,26 @@ def test_group_update_keeps_the_documents_it_holds() -> None: child = ZarrV3ArrayMetadata.create_default( context=scope, shape=(4,), codecs=(LITTLE, {"name": "acme.x"}) ) - group = ZarrV3GroupMetadata( - attributes={}, - consolidated_metadata=ZarrV3ConsolidatedMetadata(metadata={"a": child}), - extra_fields={}, + group = ZarrV3GroupMetadata.create_default( + context=scope, + consolidated_metadata={**_INLINE_ENVELOPE, "metadata": {"a": child.to_json()}}, ) - updated = group.update(context=CORE_AND_EXTENSIONS, attributes={"k": 1}) + updated = group.update(attributes={"k": 1}) assert updated.attributes == {"k": 1} - assert updated.consolidated_metadata is group.consolidated_metadata + assert updated.context == scope + assert updated.consolidated_metadata == group.consolidated_metadata def test_group_update_reads_the_documents_it_is_given_in_its_scope() -> None: """And `UNSET` leaves them out. `zstd` is an extension, which `CORE` leaves unclaimed.""" zstd = {"name": "zstd", "configuration": {"level": 3, "checksum": False}} member = cast("JSONValue", _inline(a=_array(codecs=[LITTLE, zstd]))) - updated = ZarrV3GroupMetadata.create_default().update( - context=CORE, consolidated_metadata=member - ) + updated = ZarrV3GroupMetadata.create_default(context=CORE).update(consolidated_metadata=member) assert updated.consolidated_metadata is not UNSET child = updated.consolidated_metadata.metadata["a"] assert isinstance(child, ZarrV3ArrayMetadata) assert isinstance(child.codecs[1], Unclaimed) - removed = updated.update(context=CORE, consolidated_metadata=UNSET) + removed = updated.update(consolidated_metadata=UNSET) assert removed.consolidated_metadata is UNSET @@ -891,8 +870,8 @@ def test_a_listed_group_s_own_listing_lists_what_the_group_lists( ) problems = validate_group_metadata_v3(document) assert [(p.loc, p.message, p.kind) for p in problems] == expected - # The constructor refuses what the reader reports, built of models: a - # listed group whose own listing is wrong is refused as it is built. + # The consolidated member read on its own refuses what the group's read + # reports: a listed group whose own listing is wrong is refused. listing = {"g": _group(consolidated_metadata=_inline(**listed)), **flat} deeper = [(loc, message, kind) for loc, message, kind in expected if len(loc) > len(NESTED) + 1] if len(deeper) != 0: @@ -902,12 +881,11 @@ def test_a_listed_group_s_own_listing_lists_what_the_group_lists( (loc[3:], message, kind) for loc, message, kind in deeper ] return - members = {key: node_metadata_from_json_v3(value) for key, value in listing.items()} if expected == []: - assert ZarrV3ConsolidatedMetadata(metadata=members).metadata.keys() == members.keys() + assert ZarrV3ConsolidatedMetadata(_inline(**listing)).metadata.keys() == listing.keys() return with pytest.raises(MetadataValidationError) as raised: - ZarrV3ConsolidatedMetadata(metadata=members) + ZarrV3ConsolidatedMetadata(_inline(**listing)) assert [(p.loc[1:], p.message, p.kind) for p in raised.value.problems] == [ (loc[2:], message, kind) for loc, message, kind in expected ] @@ -1358,15 +1336,18 @@ def test_error_a_null_consolidated_metadata_is_a_value_the_document_wrote() -> N TO_JSON_NO_ALIASING_PARAMS = [ pytest.param( - ZarrV3GroupMetadata( + ZarrV3GroupMetadata.create_default( attributes={"a": {"b": [1]}}, - consolidated_metadata=ZarrV3ConsolidatedMetadata( - metadata={ - "child": ZarrV3ArrayMetadata.create_default(attributes={"x": {"y": 1}}), - "grp": ZarrV3GroupMetadata.create_default(attributes={"x": {"y": 1}}), - } - ), - extra_fields={"ext": {"must_understand": False, "cfg": {"x": [1]}}}, + consolidated_metadata={ + **_INLINE_ENVELOPE, + "metadata": { + "child": ZarrV3ArrayMetadata.create_default( + attributes={"x": {"y": 1}} + ).to_json(), + "grp": ZarrV3GroupMetadata.create_default(attributes={"x": {"y": 1}}).to_json(), + }, + }, + ext={"must_understand": False, "cfg": {"x": [1]}}, ), id="v3-group", ), @@ -1376,7 +1357,14 @@ def test_error_a_null_consolidated_metadata_is_a_value_the_document_wrote() -> N ), pytest.param( ZarrV3ConsolidatedMetadata( - metadata={"child": ZarrV3ArrayMetadata.create_default(attributes={"x": {"y": 1}})} + { + **_INLINE_ENVELOPE, + "metadata": { + "child": ZarrV3ArrayMetadata.create_default( + attributes={"x": {"y": 1}} + ).to_json() + }, + } ), id="v3-consolidated", ), diff --git a/packages/zarr-metadata/tests/model/test_pair.py b/packages/zarr-metadata/tests/model/test_pair.py index 3e27ad6cbe..2c9183939e 100644 --- a/packages/zarr-metadata/tests/model/test_pair.py +++ b/packages/zarr-metadata/tests/model/test_pair.py @@ -101,9 +101,9 @@ def test_properties_are_what_the_reading_holds() -> None: assert model.extra_fields == {"acme": 1} assert (model.zarr_format, model.node_type) == (3, "array") with pytest.raises(TypeError): - model.attributes["b"] = 1 # type: ignore[index] + model.attributes["b"] = 1 # pyright: ignore[reportIndexIssue] with pytest.raises(AttributeError): - model.shape = (5,) # type: ignore[misc] + model.shape = (5,) # pyright: ignore[reportAttributeAccessIssue] assert isinstance(ZarrV3ArrayMetadata(ARRAY, context=Context.of()).data_type, Unclaimed) diff --git a/packages/zarr-metadata/tests/model/test_sentinel.py b/packages/zarr-metadata/tests/model/test_sentinel.py index 17385d46c2..1ec219d5b6 100644 --- a/packages/zarr-metadata/tests/model/test_sentinel.py +++ b/packages/zarr-metadata/tests/model/test_sentinel.py @@ -6,8 +6,8 @@ across process boundaries; these tests pin that behavior, since models hold `UNSET` as field values and must survive pickling and deep-copying. -The model round-trip tests compare whole structures: dataclass equality -compares every field, and `UNSET` compares by identity, so an impostor +The model round-trip tests compare whole structures: a model's equality +compares every member, and `UNSET` compares by identity, so an impostor sentinel produced by state-based pickling would fail the equality check. """ @@ -24,7 +24,6 @@ ZarrV2ArrayMetadata, ZarrV2GroupMetadata, ZarrV3ArrayMetadata, - ZarrV3ConsolidatedMetadata, ZarrV3GroupMetadata, ) @@ -42,15 +41,15 @@ "group-v2-attributes-unset": ZarrV2GroupMetadata.create_default(), "group-v2-attributes-set": ZarrV2GroupMetadata.create_default(attributes={"a": 1}), "group-v3-consolidated-unset": ZarrV3GroupMetadata.create_default(), - "group-v3-consolidated-with-unset-inside": ZarrV3GroupMetadata( - attributes={}, - consolidated_metadata=ZarrV3ConsolidatedMetadata( - metadata={ - "child": ZarrV3ArrayMetadata.create_default(shape=(4,)), - "subgroup": ZarrV3GroupMetadata.create_default(), - } - ), - extra_fields={}, + "group-v3-consolidated-with-unset-inside": ZarrV3GroupMetadata.create_default( + consolidated_metadata={ + "kind": "inline", + "must_understand": False, + "metadata": { + "child": ZarrV3ArrayMetadata.create_default(shape=(4,)).to_json(), + "subgroup": ZarrV3GroupMetadata.create_default().to_json(), + }, + }, ), } From f9ebea3523643de62ee6167307269caa13dc50e8 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Tue, 6 Oct 2026 14:51:25 +0200 Subject: [PATCH 38/94] docs(zarr-metadata): describe the (document, context) model Assisted-by: ClaudeCode:claude-fable-5-1 --- packages/zarr-metadata/README.md | 27 +++++++++++-------- .../changes/+document-context.feature.md | 1 + .../changes/+document-context.removal.md | 1 + packages/zarr-metadata/docs/index.md | 27 +++++++++++-------- .../src/zarr_metadata/model/__init__.py | 11 +++++--- 5 files changed, 41 insertions(+), 26 deletions(-) create mode 100644 packages/zarr-metadata/changes/+document-context.feature.md create mode 100644 packages/zarr-metadata/changes/+document-context.removal.md diff --git a/packages/zarr-metadata/README.md b/packages/zarr-metadata/README.md index b9f66d75da..a4eaffaaca 100644 --- a/packages/zarr-metadata/README.md +++ b/packages/zarr-metadata/README.md @@ -160,17 +160,22 @@ their `node_path` twins, judge a string by them. A member the spec does not define is not a field; the model's `must_understand_fields` names those a reader must understand. -A model checks itself when it is built, as a pydantic model does in -`__init__`: one built by hand, or changed by `dataclasses.replace`, whose -document has a problem raises `MetadataValidationError` with every -problem at the change, so no model is built invalid, and `to_key_value` -writes each as it is. It holds its members as that read refines them, in -containers of its own -- a list given for an array as a tuple -- as -pydantic holds what its `__init__` coerced, and each field, a `Read` or -an `Unclaimed`, as the scope read it: one built by hand is taken as -read. A model a read builds is not read a second time. Change a model by -building another: a container it holds, changed in place, is not -checked again. +A v3 model is its document and the scope it was read in: +`ZarrV3ArrayMetadata(document, context=None)` reads the document in the +scope, `CORE_AND_EXTENSIONS` when none is given, and raises +`MetadataValidationError` with every problem, so no model is built +invalid; `to_json` writes the document as it was written, and +`to_key_value` writes it as it is. Every typed member is a view of that +read: each field as the scope read it, a `Read` or an `Unclaimed`, and +`shape`, `attributes` and the rest as the read refined them, a list +given for an array as a tuple. A model is changed by `update`, which puts +JSON members in place of the document's and reads the result in the +model's own scope, so no scope is passed back in; `with_context` reads +the document in another scope, and `refined_in` only in one that claims +what this one left unclaimed and contradicts nothing, raising +`ScopeConflictError` otherwise. The documents a group's +`consolidated_metadata` holds are models of the group's scope, built from +the group's one read. Two models are equal when they mean the same document, however each is spelled. What the package interprets -- each field, and the fill value diff --git a/packages/zarr-metadata/changes/+document-context.feature.md b/packages/zarr-metadata/changes/+document-context.feature.md new file mode 100644 index 0000000000..b76e73f7be --- /dev/null +++ b/packages/zarr-metadata/changes/+document-context.feature.md @@ -0,0 +1 @@ +A v3 model is its document and the scope it was read in: `ZarrV3ArrayMetadata(document, context=None)` reads the document in the scope, `to_json` writes it as it was written, and `update` reads new members in the model's own scope, so no scope is passed back in. `with_context` reads the document in another scope; `refined_in` does so only when that scope claims what this one left unclaimed and contradicts nothing, raising `ScopeConflictError` otherwise; `refines` says whether one model holds everything another does. The documents a group's consolidated metadata holds are models of the group's scope, built from one read. diff --git a/packages/zarr-metadata/changes/+document-context.removal.md b/packages/zarr-metadata/changes/+document-context.removal.md new file mode 100644 index 0000000000..510bc000fb --- /dev/null +++ b/packages/zarr-metadata/changes/+document-context.removal.md @@ -0,0 +1 @@ +The v3 models are no longer dataclasses built from typed fields: `ZarrV3ArrayMetadata(shape=..., data_type=, ...)`, `dataclasses.replace` and `dataclasses.fields` on them are gone; build a model from a document and change it with `update`. `update` no longer takes `context`. `to_json` no longer respells fields: `"bytes"` stays `"bytes"`. diff --git a/packages/zarr-metadata/docs/index.md b/packages/zarr-metadata/docs/index.md index 232ccd6a84..da8ceff8f3 100644 --- a/packages/zarr-metadata/docs/index.md +++ b/packages/zarr-metadata/docs/index.md @@ -175,17 +175,22 @@ their `node_path` twins, judge a string by them. A member the spec does not define is not a field; the model's `must_understand_fields` names those a reader must understand. -A model checks itself when it is built, as a pydantic model does in -`__init__`: one built by hand, or changed by `dataclasses.replace`, whose -document has a problem raises `MetadataValidationError` with every -problem at the change, so no model is built invalid, and `to_key_value` -writes each as it is. It holds its members as that read refines them, in -containers of its own -- a list given for an array as a tuple -- as -pydantic holds what its `__init__` coerced, and each field, a `Read` or -an `Unclaimed`, as the scope read it: one built by hand is taken as -read. A model a read builds is not read a second time. Change a model by -building another: a container it holds, changed in place, is not -checked again. +A v3 model is its document and the scope it was read in: +`ZarrV3ArrayMetadata(document, context=None)` reads the document in the +scope, `CORE_AND_EXTENSIONS` when none is given, and raises +`MetadataValidationError` with every problem, so no model is built +invalid; `to_json` writes the document as it was written, and +`to_key_value` writes it as it is. Every typed member is a view of that +read: each field as the scope read it, a `Read` or an `Unclaimed`, and +`shape`, `attributes` and the rest as the read refined them, a list +given for an array as a tuple. A model is changed by `update`, which puts +JSON members in place of the document's and reads the result in the +model's own scope, so no scope is passed back in; `with_context` reads +the document in another scope, and `refined_in` only in one that claims +what this one left unclaimed and contradicts nothing, raising +`ScopeConflictError` otherwise. The documents a group's +`consolidated_metadata` holds are models of the group's scope, built from +the group's one read. Two models are equal when they mean the same document, however each is spelled. What the package interprets -- each field, and the fill value diff --git a/packages/zarr-metadata/src/zarr_metadata/model/__init__.py b/packages/zarr-metadata/src/zarr_metadata/model/__init__.py index e1b2e6a151..a75b9904dc 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/__init__.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/__init__.py @@ -21,10 +21,13 @@ `from_key_value` constructors raise `MetadataValidationError` for every ingestion failure, including missing store keys and undecodable bytes, and the v3 ones take the same -`context`. A model checks itself when it is built, as a pydantic model -does in `__init__`, so one built by hand, or changed by -`dataclasses.replace`, is refused at the change when its document has a -problem, and `to_key_value` writes a model as it is. +`context`. A v3 model is its document and the scope it was read in: +`ZarrV3ArrayMetadata(document, context=None)` reads the document in the +scope and raises `MetadataValidationError` with every problem, so no +model is built invalid; `to_json` writes the document as written; +`update` reads new members in the model's own scope; `with_context` and +`refined_in` read the document in another; `to_key_value` writes a model +as it is. """ from zarr_metadata._json import ( From 581f2ab8b55553b4fe90fd38b903878eb40c4449 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Tue, 6 Oct 2026 15:01:11 +0200 Subject: [PATCH 39/94] fix(zarr-metadata): fix refines on nested fields, scope of nested models, and conflict locations Review round 1 findings: - `refines` compared every field in the reading, nested ones included, so a shard read in a scope never refined the same shard left unclaimed. It now compares top-level fields pairwise and lets the field-level `refines` recurse. - `refines` raised when the lower model's fill value was invalid for the upper model's data type. It now returns False. - `with_context` on a reader-built group kept nested models in the old scope. Nested models are now rebuilt when the scope changes. - `model.reading.metadata` is now always the model itself, however the model was built. - `ScopeConflictError` from `refined_in` now includes each conflict's location. - `refined_in` can raise `MetadataValidationError` on a gain; this is documented and tested. - Fixed a stale docstring and added a type guard to `ZarrV3ConsolidatedMetadata.refines`. Assisted-by: ClaudeCode:claude-fable-5-1 --- .../src/zarr_metadata/model/_array.py | 56 +++++++-- .../src/zarr_metadata/model/_group.py | 100 ++++++++-------- .../zarr-metadata/tests/model/test_pair.py | 109 +++++++++++++++++- 3 files changed, 196 insertions(+), 69 deletions(-) diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_array.py b/packages/zarr-metadata/src/zarr_metadata/model/_array.py index d1d9c93841..8efba5a255 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_array.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/_array.py @@ -41,15 +41,19 @@ StorageTransformerDefinition, Unclaimed, field_key, + fill_value_problems, spelled_canonically, ) from zarr_metadata.v3._registry import CORE_AND_EXTENSIONS, Context -from zarr_metadata.v3._scope import Claims, ScopeConflictError, claims_of +from zarr_metadata.v3._scope import Claims, Conflict, ScopeConflictError, claim_key, claims_of from zarr_metadata.v3._scope import refines as refines_field from zarr_metadata.v3.array import ZARR_V3_ARRAY_METADATA_STORE_KEY, ZarrV3ExtensionField if TYPE_CHECKING: + from collections.abc import Iterable, Sequence + from zarr_metadata._common import JSONValue + from zarr_metadata._typed_json import Loc from zarr_metadata.v2.array import ( ZarrV2ArrayDimensionSeparator, ZarrV2ArrayMetadataJSON, @@ -60,6 +64,7 @@ from zarr_metadata.v2.attributes import ZarrV2AttributesStoreKey from zarr_metadata.v2.codec import ZarrV2CodecMetadata from zarr_metadata.v3._common import ZarrV3MetadataFieldJSON + from zarr_metadata.v3._definition import Resolved from zarr_metadata.v3.array import ( ZarrV3ArrayMetadataJSON, ZarrV3ArrayMetadataJSONPartial, @@ -160,7 +165,9 @@ def _adopt( ) -> None: self._document = document self._context = context - self._reading = reading + # The reading holds the model it built, as `read_array_metadata_v3` + # hands it back, however the model was built. + self._reading = dataclasses.replace(reading, metadata=self) self._members = members self._key = array_key(self) self._claims = MappingProxyType(claims_of(reading.fields())) @@ -317,25 +324,36 @@ def refined_in(self, context: Context | None = None) -> ZarrV3ArrayMetadata: """This document read in `context`, which may claim what this scope left unclaimed and contradict nothing. `ScopeConflictError` naming each name `context` reads by another - definition, or by none, where this scope read it by one: a loss - of meaning is refused as a conflict is. `with_context` reads the - document in any scope. + definition, or by none, where this scope read it by one -- a loss + of meaning is refused as a conflict is -- and where each sits in + the document. `MetadataValidationError` when a name `context` + claims refuses what was written under it: a gain can surface a + problem. `with_context` reads the document in any scope. """ scope = CORE_AND_EXTENSIONS if context is None else context found = scope.disagreements(self._claims) if len(found.conflicts) != 0: - raise ScopeConflictError(found.conflicts) + raise ScopeConflictError(located_conflicts(self._reading.fields(), found.conflicts)) return self.with_context(scope) def refines(self, other: ZarrV3ArrayMetadata) -> bool: - """Whether this model holds everything `other` holds: each field refines its counterpart, as `refines` orders fields, and every other member is the same, the fill value as the more informed data type spells it.""" + """Whether this model holds everything `other` holds: each field refines its counterpart, as `refines` orders fields -- the fields a field holds with it -- and every other member is the same, the fill value as the more informed data type spells it; a fill value that data type refuses is no refinement.""" if type(other) is not type(self): return False - mine = dict(self._reading.fields()) - theirs = dict(other._reading.fields()) - if mine.keys() != theirs.keys(): + if len(self.codecs) != len(other.codecs) or len(self.storage_transformers) != len( + other.storage_transformers + ): + return False + pairs = ( + (self.data_type, other.data_type), + (self.chunk_grid, other.chunk_grid), + (self.chunk_key_encoding, other.chunk_key_encoding), + *zip(self.codecs, other.codecs, strict=True), + *zip(self.storage_transformers, other.storage_transformers, strict=True), + ) + if not all(refines_field(mine, theirs) for mine, theirs in pairs): return False - if not all(refines_field(mine[loc], theirs[loc]) for loc in mine): + if len(fill_value_problems(self.data_type, other.fill_value)) != 0: return False return _plain_key(self, self.data_type) == _plain_key(other, self.data_type) @@ -422,6 +440,20 @@ def array_key(model: ZarrV3ArrayMetadata) -> tuple[object, ...]: ) +def located_conflicts( + fields: Iterable[tuple[Loc, Resolved[Any]]], conflicts: Sequence[Conflict] +) -> tuple[Conflict, ...]: + """Each of `conflicts`, found against a reading's claims, once for each place among `fields` the name it is about sits: located, as a problem is.""" + located: list[Conflict] = [] + placed = list(fields) + for conflict in conflicts: + places = [loc for loc, field in placed if claim_key(field) == conflict.key] + if len(places) == 0: + located.append(conflict) + located.extend(dataclasses.replace(conflict, loc=loc) for loc in places) + return tuple(located) + + def _plain_key( model: ZarrV3ArrayMetadata, data_type: Read[DataTypeDefinition[Any]] | Unclaimed ) -> tuple[object, ...]: @@ -462,7 +494,7 @@ def read_array_metadata_v3( refined, _ = refine_user_data(value) document = cast("dict[str, JSONValue]", refined) model = ZarrV3ArrayMetadata._of(document, context, reading, members) # pyright: ignore[reportPrivateUsage] - return dataclasses.replace(reading, metadata=model) + return model.reading class ZarrV2ArrayMetadataPartial(TypedDict, total=False): diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_group.py b/packages/zarr-metadata/src/zarr_metadata/model/_group.py index 3c7cfd0258..2ab8b8fc77 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_group.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/_group.py @@ -31,6 +31,7 @@ from zarr_metadata._sentinel import UNSET from zarr_metadata.model._array import ( ZarrV3ArrayMetadata, + located_conflicts, must_understand_subset, read_array_metadata_v3, ) @@ -79,8 +80,9 @@ class ZarrV3GroupMetadataUpdate(TypedDict, total=False, extra_items=ZarrV3Extens """The members `ZarrV3GroupMetadata.update` puts in place: each as a document writes it, or `UNSET` to leave it out. `consolidated_metadata` is a member the spec does not define, so it is - one of the extra items: given, its documents are read; left out, the - group keeps the models it holds. + one of the extra items: given, its documents are read in place of the + document's; left out, the document's are read again as part of the + whole. """ attributes: Mapping[str, JSONValue] | UNSET @@ -143,20 +145,25 @@ def _adopt( ) -> None: self._document = document self._context = context - self._reading = reading self._members = members + held: Mapping[str, ZarrV3NodeMetadataReading] = reading.consolidated if members.consolidated is UNSET: self._consolidated: ZarrV3ConsolidatedMetadata | UNSET = UNSET else: member = cast("dict[str, JSONValue]", document[ZARR_V3_CONSOLIDATED_METADATA_KEY]) documents = cast("dict[str, JSONValue]", member["metadata"]) + models = _nested_models(documents, context, reading.consolidated, members.consolidated) + # One model per document, of this scope: each nested reading + # holds the model the group holds. + held = { + path: (models[path].reading if path in models else nested) + for path, nested in reading.consolidated.items() + } self._consolidated = ZarrV3ConsolidatedMetadata._of( # pyright: ignore[reportPrivateUsage] - member, - context, - _models(documents, context, reading.consolidated, members.consolidated), + member, context, models ) - # One model per document: the readings hold the models a read - # built, which the group holds too. + # The reading holds the model it built, however the model was built. + self._reading = dataclasses.replace(reading, consolidated=held, metadata=self) self._key = group_key(self) self._claims = MappingProxyType(claims_of(reading.fields())) @@ -259,11 +266,17 @@ def with_context(self, context: Context | None = None) -> ZarrV3GroupMetadata: return type(self)(self._document, context=scope) def refined_in(self, context: Context | None = None) -> ZarrV3GroupMetadata: - """This document read in `context`, which may claim what this scope left unclaimed and contradict nothing; `ScopeConflictError` naming each name it reads by another definition, or by none.""" + """This document read in `context`, which may claim what this scope left unclaimed and contradict nothing. + + `ScopeConflictError` naming each name `context` reads by another + definition, or by none, and where each sits, in the documents the + consolidated metadata holds too; `MetadataValidationError` when a + name `context` claims refuses what was written under it. + """ scope = CORE_AND_EXTENSIONS if context is None else context found = scope.disagreements(self._claims) if len(found.conflicts) != 0: - raise ScopeConflictError(found.conflicts) + raise ScopeConflictError(located_conflicts(self._reading.fields(), found.conflicts)) return self.with_context(scope) def refines(self, other: ZarrV3GroupMetadata) -> bool: @@ -347,7 +360,7 @@ def __init__(self, member: object, context: Context | None = None) -> None: refined, _ = refine_user_data(member) document = cast("dict[str, JSONValue]", refined) documents = cast("dict[str, JSONValue]", document["metadata"]) - self._adopt(document, scope, _models(documents, scope, readings, members)) + self._adopt(document, scope, _nested_models(documents, scope, readings, members)) @classmethod def _of( @@ -371,6 +384,7 @@ def _adopt( self._context = context self._metadata = metadata self._key = consolidated_key(self) + # Hidden from the readings, which hold their own models; see `metadata`. @property def context(self) -> Context: @@ -401,7 +415,9 @@ def __reduce__(self) -> tuple[type[ZarrV3ConsolidatedMetadata], tuple[object, Co return type(self), (self._document, self._context) def refines(self, other: ZarrV3ConsolidatedMetadata) -> bool: - """Whether every document this holds refines the one `other` holds at the same path, and neither holds a path the other does not.""" + """Whether every document this holds refines the one `other` holds at the same path, and neither holds a path the other does not; False of what is not consolidated metadata.""" + if type(other) is not type(self): + return False if self._metadata.keys() != other._metadata.keys(): return False return all( @@ -865,40 +881,41 @@ def _with_models( context: Context, ) -> ZarrV3GroupMetadataReading: """`reading`, holding the model of each document its consolidated metadata holds that has no problem, and its own when it has none: each built from `document`, this read's, and `context`.""" + if len(reading.problems) == 0: + model = ZarrV3GroupMetadata._of(document, context, reading, members) # pyright: ignore[reportPrivateUsage] + return model.reading # A document with problems may hold a member that is no object, or # whose `metadata` is none: then no document in it was read. member = document.get(ZARR_V3_CONSOLIDATED_METADATA_KEY) entries = member.get("metadata") if isinstance(member, Mapping) else None if members.consolidated is UNSET or not isinstance(entries, Mapping): - readings: dict[str, ZarrV3NodeMetadataReading] = dict(reading.consolidated) - else: - readings = _readings_with_models( - cast("Mapping[str, JSONValue]", entries), - context, - reading.consolidated, - members.consolidated, - ) - held = dataclasses.replace(reading, consolidated=readings) - if len(reading.problems) != 0: - return held - model = ZarrV3GroupMetadata._of(document, context, held, members) # pyright: ignore[reportPrivateUsage] - return dataclasses.replace(held, metadata=model) + return reading + models = _nested_models( + cast("Mapping[str, JSONValue]", entries), + context, + reading.consolidated, + members.consolidated, + ) + held = { + path: (models[path].reading if path in models else nested) + for path, nested in reading.consolidated.items() + } + return dataclasses.replace(reading, consolidated=held) -def _models( +def _nested_models( documents: Mapping[str, JSONValue], context: Context, readings: Mapping[str, ZarrV3NodeMetadataReading], members: Mapping[str, ArrayMembersV3 | GroupMembersV3], ) -> dict[str, ZarrV3NodeMetadata]: - """The model of each document in `documents`, a `metadata` member's, a model can be built of: built from its reading and members, in `context`, not read again.""" + """The model of each document in `documents`, a `metadata` member's, a model can be built of, in `context`: the one its reading holds already when that is of `context`, else built from its reading and members, not read again.""" models: dict[str, ZarrV3NodeMetadata] = {} for path, child in members.items(): reading = readings[path] - if reading.metadata is not None: - # The reading holds the model a read built already: one model - # per document, which `read_group_metadata_v3` hands back too. - models[path] = reading.metadata + held = reading.metadata + if held is not None and held.context is context: + models[path] = held continue document = cast("dict[str, JSONValue]", documents[path]) if isinstance(reading, ZarrV3ArrayMetadataReading): @@ -912,27 +929,6 @@ def _models( return models -def _readings_with_models( - documents: Mapping[str, JSONValue], - context: Context, - readings: Mapping[str, ZarrV3NodeMetadataReading], - members: Mapping[str, ArrayMembersV3 | GroupMembersV3], -) -> dict[str, ZarrV3NodeMetadataReading]: - """Each reading, holding its model when one can be built of its document, as `read_group_metadata_v3` hands them back.""" - held: dict[str, ZarrV3NodeMetadataReading] = dict(readings) - for path, child in members.items(): - reading = readings[path] - document = cast("dict[str, JSONValue]", documents[path]) - if isinstance(reading, ZarrV3ArrayMetadataReading): - array = ZarrV3ArrayMetadata._of( # pyright: ignore[reportPrivateUsage] - document, context, reading, cast("ArrayMembersV3", child) - ) - held[path] = dataclasses.replace(reading, metadata=array) - elif isinstance(reading, ZarrV3GroupMetadataReading): - held[path] = _with_models(reading, cast("GroupMembersV3", child), document, context) - return held - - def validate_group_metadata_v3( value: object, *, context: Context = CORE_AND_EXTENSIONS ) -> tuple[ValidationProblem, ...]: diff --git a/packages/zarr-metadata/tests/model/test_pair.py b/packages/zarr-metadata/tests/model/test_pair.py index 2c9183939e..fccbcf41c0 100644 --- a/packages/zarr-metadata/tests/model/test_pair.py +++ b/packages/zarr-metadata/tests/model/test_pair.py @@ -4,8 +4,8 @@ import dataclasses import pickle -from collections.abc import Iterator, Mapping -from typing import Any +from collections.abc import Callable, Iterator, Mapping +from typing import Any, cast import pytest @@ -17,9 +17,11 @@ ZarrV3ConsolidatedMetadata, ZarrV3GroupMetadata, read_array_metadata_v3, + read_group_metadata_v3, ) from zarr_metadata.v3.codec.bytes import BYTES_CODEC from zarr_metadata.v3.codec.crc32c import Empty +from zarr_metadata.v3.codec.zstd import ZSTD_CODEC from zarr_metadata.v3.definition import ( CORE, CORE_AND_EXTENSIONS, @@ -179,6 +181,22 @@ def test_error_a_model_whose_scope_does_not_pickle_says_so() -> None: "fill_value": "0x7fc00000", "codecs": [{"name": "bytes", "configuration": {"endian": "little"}}], } +SHARDED: dict[str, Any] = { + **ARRAY, + "codecs": [ + { + "name": "sharding_indexed", + "configuration": { + "chunk_shape": [1], + "codecs": ["bytes"], + "index_codecs": [ + {"name": "bytes", "configuration": {"endian": "little"}}, + "crc32c", + ], + }, + } + ], +} @pytest.mark.parametrize( @@ -237,8 +255,9 @@ def test_error_update_refuses_a_document_with_a_problem() -> None: (ZarrV3ArrayMetadata(ARRAY, context=Context.of()), CORE, True), (ZarrV3ArrayMetadata(ARRAY, context=CORE), CORE_AND_EXTENSIONS, False), (ZarrV3ArrayMetadata(FLOAT, context=Context.of()), CORE, True), + (ZarrV3ArrayMetadata(SHARDED, context=Context.of()), CORE, True), ], - ids=["gain", "nothing-to-gain", "gain-data-type-with-fill-value"], + ids=["gain", "nothing-to-gain", "gain-data-type-with-fill-value", "gain-of-a-shard"], ) def test_refined_in_moves_a_model_up_the_order( model: ZarrV3ArrayMetadata, context: Context, gained: bool @@ -250,7 +269,8 @@ def test_refined_in_moves_a_model_up_the_order( assert refined.refines(model) assert (refined == model) is not gained if not gained: - assert refined.reading is model.reading + # Not read again: the reading's pipeline is the one read. + assert refined.reading.pipeline is model.reading.pipeline @pytest.mark.parametrize( @@ -303,6 +323,16 @@ def test_error_with_context_refuses_a_document_the_scope_reads_with_a_problem() ZarrV3ArrayMetadata(FLOAT, context=Context.of()), True, ), + ( + ZarrV3ArrayMetadata(SHARDED, context=CORE), + ZarrV3ArrayMetadata(SHARDED, context=Context.of()), + True, + ), + ( + ZarrV3ArrayMetadata(FLOAT, context=CORE), + ZarrV3ArrayMetadata({**FLOAT, "fill_value": "banana"}, context=Context.of()), + False, + ), ], ids=[ "gain", @@ -311,12 +341,14 @@ def test_error_with_context_refuses_a_document_the_scope_reads_with_a_problem() "conflict", "other-members-differ", "fill-value-spelled-by-the-informed-side", + "gain-of-a-field-holding-fields", + "fill-value-the-gained-definition-refuses", ], ) def test_refines_orders_models_by_information( upper: ZarrV3ArrayMetadata, lower: ZarrV3ArrayMetadata, expected: bool ) -> None: - """A model refines another when every field refines its counterpart and every other member is the same, the fill value compared as the more informed data type spells it.""" + """A model refines another when every field refines its counterpart -- the fields a field holds too, so a shard gained is a gain -- and every other member is the same, the fill value compared as the more informed data type spells it; a fill value that definition refuses is no refinement, and no error.""" assert upper.refines(lower) is expected @@ -441,3 +473,70 @@ def test_the_held_field_machinery_is_gone() -> None: assert not hasattr(validation, "NO_SCOPE") assert not hasattr(validation, "overlapping") assert "held" not in inspect.signature(validation.read_array_v3).parameters + + +def test_error_refined_in_refuses_a_gain_that_surfaces_a_problem() -> None: + """A scope that claims a name this one left unclaimed may refuse what was written under it: `refined_in` raises `MetadataValidationError`, the document having a problem in that scope.""" + loose = ZarrV3ArrayMetadata({**ARRAY, "fill_value": "banana"}, context=Context.of()) + with pytest.raises(MetadataValidationError) as raised: + loose.refined_in(CORE) + assert [problem.loc for problem in raised.value.problems] == [("fill_value",)] + + +def test_a_scope_conflict_says_where_each_conflict_sits() -> None: + """`refined_in` names each conflict with where the field sits in the document, as a problem is located: a loss of `bytes` at `codecs.0`, in the shard too.""" + model = ZarrV3ArrayMetadata(SHARDED, context=CORE) + with pytest.raises(ScopeConflictError) as raised: + model.refined_in(Context.of()) + assert sorted((conflict.key[1], conflict.loc) for conflict in raised.value.conflicts) == [ + ("bytes", ("codecs", 0, "configuration", "codecs", 0)), + ("bytes", ("codecs", 0, "configuration", "index_codecs", 0)), + ("crc32c", ("codecs", 0, "configuration", "index_codecs", 1)), + ("default", ("chunk_key_encoding",)), + ("regular", ("chunk_grid",)), + ("sharding_indexed", ("codecs", 0)), + ("uint8", ("data_type",)), + ] + + +@pytest.mark.parametrize( + "build", + [ + lambda: ZarrV3GroupMetadata(CONSOLIDATED, context=CORE_AND_EXTENSIONS), + lambda: read_group_metadata_v3(CONSOLIDATED, context=CORE_AND_EXTENSIONS).metadata, + ], + ids=["constructor", "reader"], +) +def test_a_models_reading_holds_the_model_and_with_context_moves_the_whole_tree( + build: Callable[[], ZarrV3GroupMetadata | None], +) -> None: + """However a group was built, its reading holds it, each nested reading holds the nested model, and `with_context` into a scope that reads every claim identically moves every nested model to the new scope without reading again.""" + group = build() + assert group is not None + assert group.reading.metadata is group + nested = group.consolidated_metadata + assert isinstance(nested, ZarrV3ConsolidatedMetadata) + for path, node in nested.metadata.items(): + assert group.reading.consolidated[path].metadata is node + assert node.reading.metadata is node + scope = CORE.extended_with(ZSTD_CODEC) + moved = group.with_context(scope) + assert moved.reading is not group.reading + held = moved.consolidated_metadata + assert isinstance(held, ZarrV3ConsolidatedMetadata) + assert held.context is scope + for node in held.metadata.values(): + assert node.context is scope + assert node.reading.metadata is node + inner = held.metadata["b"] + assert isinstance(inner, ZarrV3GroupMetadata) + assert inner.consolidated_metadata is UNSET or all( + child.context is scope for child in inner.consolidated_metadata.metadata.values() + ) + + +def test_consolidated_metadata_refines_nothing_of_another_type() -> None: + """`refines` of consolidated metadata says False of what is not consolidated metadata, as the array's and group's do, rather than raising.""" + held = ZarrV3GroupMetadata(CONSOLIDATED).consolidated_metadata + assert isinstance(held, ZarrV3ConsolidatedMetadata) + assert held.refines(cast("Any", ZarrV3GroupMetadata(GROUP))) is False From 7dd18158d3d0fe846812f10e07325243a9ecfe0d Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Tue, 6 Oct 2026 15:12:34 +0200 Subject: [PATCH 40/94] fix(zarr-metadata): freeze model views at every level; build nested models despite sibling problems Review round 2 findings: - `attributes`, `extra_fields` and `fill_value` were read-only only at the top level. A nested dict could be changed in place, after which the document, the cached key and `refines` disagreed. They are now deeply frozen views (`frozen` in `_json`), and the model's key is computed from the raw members. - A group document with a refine-level problem anywhere (non-string key, non-JSON value, excess depth) refined to None as a whole, so no nested model was built for any valid child. Each nested document is now refined on its own, from its position, so valid children keep their models. Assisted-by: ClaudeCode:claude-fable-5-1 --- .../changes/+document-context.feature.md | 2 +- .../zarr-metadata/src/zarr_metadata/_json.py | 17 +++ .../src/zarr_metadata/model/_array.py | 48 ++++--- .../src/zarr_metadata/model/_group.py | 74 +++++++---- .../zarr-metadata/tests/model/test_pair.py | 124 +++++++++++++++++- 5 files changed, 215 insertions(+), 50 deletions(-) diff --git a/packages/zarr-metadata/changes/+document-context.feature.md b/packages/zarr-metadata/changes/+document-context.feature.md index b76e73f7be..170fce57fa 100644 --- a/packages/zarr-metadata/changes/+document-context.feature.md +++ b/packages/zarr-metadata/changes/+document-context.feature.md @@ -1 +1 @@ -A v3 model is its document and the scope it was read in: `ZarrV3ArrayMetadata(document, context=None)` reads the document in the scope, `to_json` writes it as it was written, and `update` reads new members in the model's own scope, so no scope is passed back in. `with_context` reads the document in another scope; `refined_in` does so only when that scope claims what this one left unclaimed and contradicts nothing, raising `ScopeConflictError` otherwise; `refines` says whether one model holds everything another does. The documents a group's consolidated metadata holds are models of the group's scope, built from one read. +A v3 model is now its document and the scope it was read in: `to_json` writes the document as it was written, and `update` reads changes in the model's own scope, so no scope is passed back in. `with_context`, `refined_in` and `refines` move a model between scopes and compare models by what they mean. diff --git a/packages/zarr-metadata/src/zarr_metadata/_json.py b/packages/zarr-metadata/src/zarr_metadata/_json.py index f3e00fb541..3152454f5e 100644 --- a/packages/zarr-metadata/src/zarr_metadata/_json.py +++ b/packages/zarr-metadata/src/zarr_metadata/_json.py @@ -14,6 +14,7 @@ import math from collections.abc import Callable, Iterator, Mapping, Sequence from dataclasses import dataclass +from types import MappingProxyType from typing import Final, Literal, TypeGuard, cast, get_args from zarr_metadata._common import JSONValue @@ -560,6 +561,21 @@ def parse_json(value: object) -> JSONValue: return refined +def frozen(value: JSONValue) -> JSONValue: + """`value` as a read-only view at every level: each object a mapping proxy, each array a tuple, so nothing handed out can be changed in place. + + Shares the scalars with `value`, and copies nothing else than the + containers a view needs. What a model shows of its document. + """ + if isinstance(value, Mapping): + return cast( + "JSONValue", MappingProxyType({key: frozen(item) for key, item in value.items()}) + ) + if isinstance(value, (tuple, list)): + return tuple(frozen(item) for item in value) + return value + + def copied(value: JSONValue) -> JSONValue: """`value` in containers of its own, sharing nothing with it: each object a new `dict`, each array a new one of its type. @@ -615,6 +631,7 @@ def arrays_to_tuples(obj: object) -> object: "arrays_to_tuples", "choices", "copied", + "frozen", "is_canonical_json", "is_json", "json_type", diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_array.py b/packages/zarr-metadata/src/zarr_metadata/model/_array.py index 8efba5a255..ed91a274c5 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_array.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/_array.py @@ -14,6 +14,7 @@ MetadataValidationError, ValidationProblem, copied, + frozen, json_text, refine_user_data, with_input, @@ -130,7 +131,7 @@ class ZarrV3ArrayMetadata: are defined at a module's top level. """ - __slots__ = ("_claims", "_context", "_document", "_key", "_members", "_reading") + __slots__ = ("_claims", "_context", "_document", "_key", "_members", "_reading", "_shown") zarr_format: Final = 3 node_type: Final = "array" @@ -169,6 +170,12 @@ def _adopt( # hands it back, however the model was built. self._reading = dataclasses.replace(reading, metadata=self) self._members = members + # What the model shows of its members, read-only at every level. + self._shown = ( + frozen(members.fill_value), + frozen(members.attributes), + frozen(cast("JSONValue", members.extra_fields)), + ) self._key = array_key(self) self._claims = MappingProxyType(claims_of(reading.fields())) @@ -229,8 +236,8 @@ def shape(self) -> tuple[int, ...]: @property def fill_value(self) -> JSONValue: - """The fill value as written.""" - return self._members.fill_value + """The fill value as written, read-only at every level.""" + return self._shown[0] @property def dimension_names(self) -> tuple[str | None, ...] | UNSET: @@ -239,13 +246,13 @@ def dimension_names(self) -> tuple[str | None, ...] | UNSET: @property def attributes(self) -> Mapping[str, JSONValue]: - """The attributes, a read-only view; empty when the document writes none.""" - return MappingProxyType(self._members.attributes) + """The attributes, read-only at every level; empty when the document writes none.""" + return cast("Mapping[str, JSONValue]", self._shown[1]) @property def extra_fields(self) -> Mapping[str, ZarrV3ExtensionField]: - """Each member the spec does not define, by name, a read-only view.""" - return MappingProxyType(self._members.extra_fields) + """Each member the spec does not define, by name, read-only at every level.""" + return cast("Mapping[str, ZarrV3ExtensionField]", self._shown[2]) @property def must_understand_fields(self) -> dict[str, ZarrV3ExtensionField]: @@ -353,7 +360,7 @@ def refines(self, other: ZarrV3ArrayMetadata) -> bool: ) if not all(refines_field(mine, theirs) for mine, theirs in pairs): return False - if len(fill_value_problems(self.data_type, other.fill_value)) != 0: + if len(fill_value_problems(self.data_type, other._members.fill_value)) != 0: return False return _plain_key(self, self.data_type) == _plain_key(other, self.data_type) @@ -426,17 +433,18 @@ def array_key(model: ZarrV3ArrayMetadata) -> tuple[object, ...]: as JSON text when a definition in scope read the data type, and every other member as it is, the JSON ones as text. """ + members = model._members # pyright: ignore[reportPrivateUsage] return ( - model.shape, + members.shape, _fill_value_key(model), field_key(model.data_type), field_key(model.chunk_grid), tuple(field_key(codec) for codec in model.codecs), field_key(model.chunk_key_encoding), - model.dimension_names, - json_text(dict(model.attributes)), + members.dimension_names, + json_text(members.attributes), tuple(field_key(transformer) for transformer in model.storage_transformers), - json_text(dict(model.extra_fields)), + json_text(members.extra_fields), ) @@ -458,20 +466,22 @@ def _plain_key( model: ZarrV3ArrayMetadata, data_type: Read[DataTypeDefinition[Any]] | Unclaimed ) -> tuple[object, ...]: """What `refines` compares of a model other than its fields, the fill value spelled as `data_type` -- the more informed side's -- spells it.""" + members = model._members # pyright: ignore[reportPrivateUsage] return ( - model.shape, - json_text(spelled_canonically(data_type, model.fill_value)), - model.dimension_names, - json_text(dict(model.attributes)), - json_text(dict(model.extra_fields)), + members.shape, + json_text(spelled_canonically(data_type, members.fill_value)), + members.dimension_names, + json_text(members.attributes), + json_text(members.extra_fields), ) def _fill_value_key(model: ZarrV3ArrayMetadata) -> str: """What `==` compares of `model`'s fill value: its canonical spelling as JSON text when a definition in scope read the data type, and the fill value as written when none did.""" + fill_value = model._members.fill_value # pyright: ignore[reportPrivateUsage] if isinstance(model.data_type, Read): - return json_text(spelled_canonically(model.data_type, model.fill_value)) - return json_text(model.fill_value) + return json_text(spelled_canonically(model.data_type, fill_value)) + return json_text(fill_value) def read_array_metadata_v3( diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_group.py b/packages/zarr-metadata/src/zarr_metadata/model/_group.py index 2ab8b8fc77..5f3a9796d3 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_group.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/_group.py @@ -15,6 +15,7 @@ ValidationProblem, arrays_to_tuples, copied, + frozen, is_canonical_json, json_text, nested_past_the_levels, @@ -110,6 +111,7 @@ class ZarrV3GroupMetadata: "_key", "_members", "_reading", + "_shown", ) zarr_format: Final = 3 @@ -146,6 +148,11 @@ def _adopt( self._document = document self._context = context self._members = members + # What the model shows of its members, read-only at every level. + self._shown = ( + frozen(cast("JSONValue", members.attributes)), + frozen(cast("JSONValue", members.extra_fields)), + ) held: Mapping[str, ZarrV3NodeMetadataReading] = reading.consolidated if members.consolidated is UNSET: self._consolidated: ZarrV3ConsolidatedMetadata | UNSET = UNSET @@ -218,13 +225,13 @@ def __reduce__(self) -> tuple[type[ZarrV3GroupMetadata], tuple[object, Context]] @property def attributes(self) -> Mapping[str, JSONValue]: - """The attributes, a read-only view; empty when the document writes none.""" - return MappingProxyType(cast("dict[str, JSONValue]", self._members.attributes)) + """The attributes, read-only at every level; empty when the document writes none.""" + return cast("Mapping[str, JSONValue]", self._shown[0]) @property def extra_fields(self) -> Mapping[str, ZarrV3ExtensionField]: - """Each member the spec does not define, `consolidated_metadata` apart, by name: a read-only view.""" - return MappingProxyType(self._members.extra_fields) + """Each member the spec does not define, `consolidated_metadata` apart, by name: read-only at every level.""" + return cast("Mapping[str, ZarrV3ExtensionField]", self._shown[1]) @property def consolidated_metadata(self) -> ZarrV3ConsolidatedMetadata | UNSET: @@ -283,9 +290,12 @@ def refines(self, other: ZarrV3GroupMetadata) -> bool: """Whether this model holds everything `other` holds: the same attributes and extra fields, and consolidated metadata whose every document refines its counterpart.""" if type(other) is not type(self): return False - if json_text(dict(self.attributes)) != json_text(dict(other.attributes)): + mine, theirs = self._members, other._members + if json_text(cast("JSONValue", mine.attributes)) != json_text( + cast("JSONValue", theirs.attributes) + ): return False - if json_text(dict(self.extra_fields)) != json_text(dict(other.extra_fields)): + if json_text(mine.extra_fields) != json_text(theirs.extra_fields): return False mine, theirs = self._consolidated, other._consolidated if mine is UNSET or theirs is UNSET: @@ -630,13 +640,7 @@ def read_group_metadata_v3( reading, members = read_group_v3(value, context) if members is None: return reading - # A document nested past the levels a reader walks refines to nothing: - # it has problems, and no model is built of it or of what it holds. - refined, _ = refine_user_data(value) - document: dict[str, JSONValue] = ( - cast("dict[str, JSONValue]", refined) if isinstance(refined, Mapping) else {} - ) - return _with_models(reading, members, document, context) + return _with_models(reading, members, value, context) def read_group_v3( @@ -758,10 +762,11 @@ def _read_consolidated_v3( def group_key(model: ZarrV3GroupMetadata) -> tuple[object, ...]: """What `==` and `hash` compare of a v3 group model: its attributes and extra fields as JSON text, and what its consolidated metadata holds, by `consolidated_key`.""" consolidated = model.consolidated_metadata + members = model._members # pyright: ignore[reportPrivateUsage] return ( - json_text(dict(model.attributes)), + json_text(cast("JSONValue", members.attributes)), UNSET if consolidated is UNSET else consolidated._key, # pyright: ignore[reportPrivateUsage] - json_text(dict(model.extra_fields)), + json_text(members.extra_fields), ) @@ -877,24 +882,45 @@ def _below_faults(path: str) -> list[str]: def _with_models( reading: ZarrV3GroupMetadataReading, members: GroupMembersV3, - document: dict[str, JSONValue], + value: object, context: Context, ) -> ZarrV3GroupMetadataReading: - """`reading`, holding the model of each document its consolidated metadata holds that has no problem, and its own when it has none: each built from `document`, this read's, and `context`.""" + """`reading`, holding the model of each document its consolidated metadata holds that has no problem, and its own when it has none: each built from `value`, the document this read read, in `context`. + + A document with a problem a reader walks past -- a key that is no + string, a value that is no JSON, a level past the cap -- refines to + nothing as a whole; each document its consolidated metadata holds is + then refined on its own, from where it sits, so the ones without a + problem still have their models. + """ + refined, _ = refine_user_data(value) if len(reading.problems) == 0: + document = cast("dict[str, JSONValue]", refined) model = ZarrV3GroupMetadata._of(document, context, reading, members) # pyright: ignore[reportPrivateUsage] return model.reading - # A document with problems may hold a member that is no object, or - # whose `metadata` is none: then no document in it was read. - member = document.get(ZARR_V3_CONSOLIDATED_METADATA_KEY) - entries = member.get("metadata") if isinstance(member, Mapping) else None - if members.consolidated is UNSET or not isinstance(entries, Mapping): + if members.consolidated is UNSET or not isinstance(value, Mapping): + return reading + member = cast("Mapping[object, object]", value).get(ZARR_V3_CONSOLIDATED_METADATA_KEY) + if not isinstance(member, Mapping): return reading + entries = cast("Mapping[object, object]", member).get("metadata") + if not isinstance(entries, Mapping): + return reading + held_entries = cast("Mapping[object, object]", entries) + documents: dict[str, JSONValue] = {} + for path in members.consolidated: + if path not in held_entries: + continue + entry, problems = refine_user_data( + held_entries[path], (ZARR_V3_CONSOLIDATED_METADATA_KEY, "metadata", path) + ) + if len(problems) == 0 and isinstance(entry, Mapping): + documents[path] = entry models = _nested_models( - cast("Mapping[str, JSONValue]", entries), + documents, context, reading.consolidated, - members.consolidated, + {path: child for path, child in members.consolidated.items() if path in documents}, ) held = { path: (models[path].reading if path in models else nested) diff --git a/packages/zarr-metadata/tests/model/test_pair.py b/packages/zarr-metadata/tests/model/test_pair.py index fccbcf41c0..73427ecde4 100644 --- a/packages/zarr-metadata/tests/model/test_pair.py +++ b/packages/zarr-metadata/tests/model/test_pair.py @@ -3,6 +3,7 @@ from __future__ import annotations import dataclasses +import operator import pickle from collections.abc import Callable, Iterator, Mapping from typing import Any, cast @@ -358,7 +359,18 @@ def test_refines_orders_models_by_information( "consolidated_metadata": { "kind": "inline", "must_understand": False, - "metadata": {"a": ARRAY, "b": GROUP, "b/c": SPELLED_OUT}, + "metadata": { + "a": ARRAY, + "b": { + **GROUP, + "consolidated_metadata": { + "kind": "inline", + "must_understand": False, + "metadata": {"c": SPELLED_OUT}, + }, + }, + "b/c": SPELLED_OUT, + }, }, } @@ -386,7 +398,7 @@ def counted(configuration: object, nested: Nested) -> Iterator[ValidationProblem scope = CORE.extended_with(dataclasses.replace(BYTES_CODEC, rules=counted)) ZarrV3GroupMetadata(CONSOLIDATED, context=scope) - assert len(calls) == 2 # `a` and `b/c` each hold one bytes codec + assert len(calls) == 3 # `a`, `b/c`, and `c` in `b`'s own listing, each one bytes codec @pytest.mark.parametrize( @@ -510,7 +522,7 @@ def test_a_scope_conflict_says_where_each_conflict_sits() -> None: def test_a_models_reading_holds_the_model_and_with_context_moves_the_whole_tree( build: Callable[[], ZarrV3GroupMetadata | None], ) -> None: - """However a group was built, its reading holds it, each nested reading holds the nested model, and `with_context` into a scope that reads every claim identically moves every nested model to the new scope without reading again.""" + """However a group was built, its reading holds it, each nested reading holds the nested model, and `with_context` into a scope that reads every claim identically moves every nested model, the nested ones' too, to the new scope without reading again.""" group = build() assert group is not None assert group.reading.metadata is group @@ -525,14 +537,21 @@ def test_a_models_reading_holds_the_model_and_with_context_moves_the_whole_tree( held = moved.consolidated_metadata assert isinstance(held, ZarrV3ConsolidatedMetadata) assert held.context is scope + before = nested.metadata["a"] + after = held.metadata["a"] + assert isinstance(before, ZarrV3ArrayMetadata) + assert isinstance(after, ZarrV3ArrayMetadata) + assert after.reading.pipeline is before.reading.pipeline # not read again for node in held.metadata.values(): assert node.context is scope assert node.reading.metadata is node inner = held.metadata["b"] assert isinstance(inner, ZarrV3GroupMetadata) - assert inner.consolidated_metadata is UNSET or all( - child.context is scope for child in inner.consolidated_metadata.metadata.values() - ) + innermost = inner.consolidated_metadata + assert isinstance(innermost, ZarrV3ConsolidatedMetadata) + assert innermost.context is scope + assert innermost.metadata["c"].context is scope + assert innermost.metadata["c"].reading.metadata is innermost.metadata["c"] def test_consolidated_metadata_refines_nothing_of_another_type() -> None: @@ -540,3 +559,96 @@ def test_consolidated_metadata_refines_nothing_of_another_type() -> None: held = ZarrV3GroupMetadata(CONSOLIDATED).consolidated_metadata assert isinstance(held, ZarrV3ConsolidatedMetadata) assert held.refines(cast("Any", ZarrV3GroupMetadata(GROUP))) is False + + +@pytest.mark.parametrize( + ("document", "change"), + [ + ( + {**ARRAY, "attributes": {"a": {"b": 1}}}, + lambda model: operator.setitem(model.attributes["a"], "b", 2), + ), + ( + {**ARRAY, "acme": {"x": 1, "must_understand": False}}, + lambda model: operator.setitem(model.extra_fields["acme"], "x", 2), + ), + ( + { + **ARRAY, + "data_type": { + "name": "struct", + "configuration": {"fields": [{"name": "a", "data_type": "uint8"}]}, + }, + "fill_value": {"a": 0}, + }, + lambda model: operator.setitem(model.fill_value, "a", 7), + ), + ], + ids=["attributes", "extra-fields", "fill-value"], +) +def test_a_models_views_cannot_be_changed_in_place( + document: dict[str, Any], change: Callable[[ZarrV3ArrayMetadata], None] +) -> None: + """What a model shows -- attributes, extra fields, a fill value -- is read-only at every level, so a model cannot be put in a state its document, its key and `refines` disagree about.""" + model = ZarrV3ArrayMetadata(document) + same = ZarrV3ArrayMetadata(document) + with pytest.raises(TypeError): + change(model) + assert model == same + assert model.to_json() == same.to_json() + assert model.refines(same) + + +@pytest.mark.parametrize( + "problem", + [ + {"attributes": {1: "x"}}, + {"attributes": 5}, + { + "consolidated_metadata": { + "kind": "inline", + "must_understand": False, + "metadata": {"a": ARRAY, "b": {**GROUP, "attributes": {"s": {1, 2}}}}, + } + }, + { + "consolidated_metadata": { + "kind": "inline", + "must_understand": False, + "metadata": {"a": ARRAY, "b": {**ARRAY, "shape": "x"}}, + } + }, + ], + ids=["non-string-key", "attributes-not-an-object", "sibling-not-json", "sibling-invalid"], +) +def test_a_reading_holds_a_model_of_each_nested_document_without_a_problem( + problem: dict[str, Any], +) -> None: + """A group document with a problem still holds, in its reading, a model of each document its consolidated metadata holds that has no problem, whatever the problem elsewhere is: one a reader walks past, or one it refuses.""" + member = {"kind": "inline", "must_understand": False, "metadata": {"a": ARRAY}} + document = {**GROUP, "consolidated_metadata": member, **problem} + reading = read_group_metadata_v3(document) + assert len(reading.problems) != 0 + assert reading.metadata is None + held = reading.consolidated["a"].metadata + assert isinstance(held, ZarrV3ArrayMetadata) + assert held == ZarrV3ArrayMetadata(ARRAY) + + +def test_a_scope_conflict_inside_consolidated_metadata_is_located_there() -> None: + """A group's `refined_in` locates a conflict in a document its consolidated metadata holds under that document's path, in a listing a listed group holds too.""" + group = ZarrV3GroupMetadata(CONSOLIDATED, context=CORE) + with pytest.raises(ScopeConflictError) as raised: + group.refined_in(Context.of()) + locs = {conflict.loc for conflict in raised.value.conflicts} + assert ("consolidated_metadata", "metadata", "a", "codecs", 0) in locs + assert ( + "consolidated_metadata", + "metadata", + "b", + "consolidated_metadata", + "metadata", + "c", + "codecs", + 0, + ) in locs From d3853d5e908aaabc385da019f28fe45acd369f8d Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Tue, 6 Oct 2026 15:20:29 +0200 Subject: [PATCH 41/94] fix(zarr-metadata): make reading.consolidated read-only; recurse into listed groups with problems Review round 3 findings: - `reading.consolidated` was a plain dict. A caller could put a reading there, and `with_context` would then adopt the model it held. It is now a read-only mapping wherever it is built. - A listed group with a problem of its own did not get models for the valid documents in its own listing. The reader now recurses into it. - The removal fragment and the docs now state that the views are read-only at every level and that `to_json()` returns plain containers. - The whole-document refine runs only when a model is actually built. Assisted-by: ClaudeCode:claude-fable-5-1 --- packages/zarr-metadata/README.md | 11 +++-- .../changes/+document-context.removal.md | 2 +- packages/zarr-metadata/docs/index.md | 11 +++-- .../src/zarr_metadata/model/_group.py | 41 +++++++++++++------ .../zarr-metadata/tests/model/test_pair.py | 38 +++++++++++++++++ 5 files changed, 81 insertions(+), 22 deletions(-) diff --git a/packages/zarr-metadata/README.md b/packages/zarr-metadata/README.md index a4eaffaaca..e22451b6dd 100644 --- a/packages/zarr-metadata/README.md +++ b/packages/zarr-metadata/README.md @@ -167,8 +167,10 @@ scope, `CORE_AND_EXTENSIONS` when none is given, and raises invalid; `to_json` writes the document as it was written, and `to_key_value` writes it as it is. Every typed member is a view of that read: each field as the scope read it, a `Read` or an `Unclaimed`, and -`shape`, `attributes` and the rest as the read refined them, a list -given for an array as a tuple. A model is changed by `update`, which puts +`shape`, `attributes` and the rest as the read refined them, read-only +at every level: a list given for an array as a tuple, an object as a +read-only mapping; `to_json` gives plain containers. A model is changed +by `update`, which puts JSON members in place of the document's and reads the result in the model's own scope, so no scope is passed back in; `with_context` reads the document in another scope, and `refined_in` only in one that claims @@ -188,8 +190,9 @@ configuration of a field nothing in scope claims, and every member of a v2 document -- compares as JSON text, which tells `true` from `1` and `-0.0` from `0.0`, and takes `NaN` for itself. Equal models hash alike, and may write two documents: `to_json` writes each as it was given. A -model's hash is of what its containers held when it was hashed, so a -model in a set, or a key of a dict, is not changed in place. +v3 model holds nothing that can be changed in place; a v2 model's hash +is of what its containers held when it was hashed, so one in a set, or a +key of a dict, is not changed in place. `node_metadata_json_schema_v3` writes what the validators read as a JSON Schema, draft 2020-12, for an editor that checks a `zarr.json` as it diff --git a/packages/zarr-metadata/changes/+document-context.removal.md b/packages/zarr-metadata/changes/+document-context.removal.md index 510bc000fb..8c0676d489 100644 --- a/packages/zarr-metadata/changes/+document-context.removal.md +++ b/packages/zarr-metadata/changes/+document-context.removal.md @@ -1 +1 @@ -The v3 models are no longer dataclasses built from typed fields: `ZarrV3ArrayMetadata(shape=..., data_type=, ...)`, `dataclasses.replace` and `dataclasses.fields` on them are gone; build a model from a document and change it with `update`. `update` no longer takes `context`. `to_json` no longer respells fields: `"bytes"` stays `"bytes"`. +The v3 models are no longer dataclasses built from typed fields: `ZarrV3ArrayMetadata(shape=..., data_type=, ...)`, `dataclasses.replace` and `dataclasses.fields` on them are gone; build a model from a document and change it with `update`. `update` no longer takes `context`, and `to_json` no longer respells fields: `"bytes"` stays `"bytes"`. `attributes`, `extra_fields` and a struct `fill_value` are read-only at every level, objects as read-only mappings and arrays as tuples, so they cannot be changed in place, serialized or pickled directly; `to_json()` gives plain JSON containers. diff --git a/packages/zarr-metadata/docs/index.md b/packages/zarr-metadata/docs/index.md index da8ceff8f3..a15c673d0d 100644 --- a/packages/zarr-metadata/docs/index.md +++ b/packages/zarr-metadata/docs/index.md @@ -182,8 +182,10 @@ scope, `CORE_AND_EXTENSIONS` when none is given, and raises invalid; `to_json` writes the document as it was written, and `to_key_value` writes it as it is. Every typed member is a view of that read: each field as the scope read it, a `Read` or an `Unclaimed`, and -`shape`, `attributes` and the rest as the read refined them, a list -given for an array as a tuple. A model is changed by `update`, which puts +`shape`, `attributes` and the rest as the read refined them, read-only +at every level: a list given for an array as a tuple, an object as a +read-only mapping; `to_json` gives plain containers. A model is changed +by `update`, which puts JSON members in place of the document's and reads the result in the model's own scope, so no scope is passed back in; `with_context` reads the document in another scope, and `refined_in` only in one that claims @@ -203,8 +205,9 @@ configuration of a field nothing in scope claims, and every member of a v2 document -- compares as JSON text, which tells `true` from `1` and `-0.0` from `0.0`, and takes `NaN` for itself. Equal models hash alike, and may write two documents: `to_json` writes each as it was given. A -model's hash is of what its containers held when it was hashed, so a -model in a set, or a key of a dict, is not changed in place. +v3 model holds nothing that can be changed in place; a v2 model's hash +is of what its containers held when it was hashed, so one in a set, or a +key of a dict, is not changed in place. `node_metadata_json_schema_v3` writes what the validators read as a JSON Schema, draft 2020-12, for an editor that checks a `zarr.json` as it diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_group.py b/packages/zarr-metadata/src/zarr_metadata/model/_group.py index 5f3a9796d3..4fe7f5e771 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_group.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/_group.py @@ -162,15 +162,20 @@ def _adopt( models = _nested_models(documents, context, reading.consolidated, members.consolidated) # One model per document, of this scope: each nested reading # holds the model the group holds. - held = { - path: (models[path].reading if path in models else nested) - for path, nested in reading.consolidated.items() - } + held = MappingProxyType( + { + path: (models[path].reading if path in models else nested) + for path, nested in reading.consolidated.items() + } + ) self._consolidated = ZarrV3ConsolidatedMetadata._of( # pyright: ignore[reportPrivateUsage] member, context, models ) - # The reading holds the model it built, however the model was built. - self._reading = dataclasses.replace(reading, consolidated=held, metadata=self) + # The reading holds the model it built, however the model was built; + # what it holds of the nested documents is read-only, as the model is. + self._reading = dataclasses.replace( + reading, consolidated=MappingProxyType(dict(held)), metadata=self + ) self._key = group_key(self) self._claims = MappingProxyType(claims_of(reading.fields())) @@ -684,7 +689,7 @@ def read_group_v3( raw, context, (*at, ZARR_V3_CONSOLIDATED_METADATA_KEY) ) found.extend(_prefix(ZARR_V3_CONSOLIDATED_METADATA_KEY, inside)) - reading = ZarrV3GroupMetadataReading(consolidated, with_input(found, whole)) + reading = ZarrV3GroupMetadataReading(MappingProxyType(consolidated), with_input(found, whole)) return reading, GroupMembersV3(attributes, extra_fields, held) @@ -893,8 +898,8 @@ def _with_models( then refined on its own, from where it sits, so the ones without a problem still have their models. """ - refined, _ = refine_user_data(value) if len(reading.problems) == 0: + refined, _ = refine_user_data(value) document = cast("dict[str, JSONValue]", refined) model = ZarrV3GroupMetadata._of(document, context, reading, members) # pyright: ignore[reportPrivateUsage] return model.reading @@ -922,11 +927,21 @@ def _with_models( reading.consolidated, {path: child for path, child in members.consolidated.items() if path in documents}, ) - held = { - path: (models[path].reading if path in models else nested) - for path, nested in reading.consolidated.items() - } - return dataclasses.replace(reading, consolidated=held) + held = dict(reading.consolidated) + for path, child in members.consolidated.items(): + nested = reading.consolidated[path] + if path in models: + held[path] = models[path].reading + elif ( + path in documents + and isinstance(nested, ZarrV3GroupMetadataReading) + and isinstance(child, GroupMembersV3) + ): + # A listed group with a problem of its own still holds, in its + # reading, a model of each document in its own listing that + # has none. + held[path] = _with_models(nested, child, held_entries[path], context) + return dataclasses.replace(reading, consolidated=MappingProxyType(held)) def _nested_models( diff --git a/packages/zarr-metadata/tests/model/test_pair.py b/packages/zarr-metadata/tests/model/test_pair.py index 73427ecde4..fb38f8ff21 100644 --- a/packages/zarr-metadata/tests/model/test_pair.py +++ b/packages/zarr-metadata/tests/model/test_pair.py @@ -17,6 +17,7 @@ ZarrV3ArrayMetadata, ZarrV3ConsolidatedMetadata, ZarrV3GroupMetadata, + ZarrV3GroupMetadataReading, read_array_metadata_v3, read_group_metadata_v3, ) @@ -652,3 +653,40 @@ def test_a_scope_conflict_inside_consolidated_metadata_is_located_there() -> Non "codecs", 0, ) in locs + + +def test_a_models_reading_cannot_be_changed_in_place() -> None: + """What a model's reading holds of the documents its consolidated metadata holds is read-only, so nothing planted there is taken up by `with_context`.""" + group = ZarrV3GroupMetadata(CONSOLIDATED) + planted = ZarrV3ArrayMetadata({**ARRAY, "shape": [9]}).reading + with pytest.raises(TypeError): + operator.setitem(cast("Any", group.reading.consolidated), "a", planted) + moved = group.with_context(CORE_AND_EXTENSIONS) + assert moved == ZarrV3GroupMetadata(CONSOLIDATED) + + +def test_a_reading_holds_a_model_of_each_healthy_document_in_a_listed_groups_own_listing() -> None: + """A listed group with a problem of its own still holds, in its reading, a model of each document in its own listing that has no problem, as the top group does.""" + inline = {"kind": "inline", "must_understand": False} + listed = { + **GROUP, + "consolidated_metadata": { + **inline, + "metadata": {"x": ARRAY, "y": {**ARRAY, "shape": [-1]}}, + }, + } + document = { + **GROUP, + "consolidated_metadata": { + **inline, + "metadata": {"a": listed, "a/x": ARRAY, "a/y": {**ARRAY, "shape": [-1]}}, + }, + } + reading = read_group_metadata_v3(document) + assert reading.metadata is None + nested = reading.consolidated["a"] + assert isinstance(nested, ZarrV3GroupMetadataReading) + assert nested.metadata is None + held = nested.consolidated["x"].metadata + assert isinstance(held, ZarrV3ArrayMetadata) + assert held == ZarrV3ArrayMetadata(ARRAY) From fa32f89db19c1cb782651c65464c1c05e4b9173b Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Tue, 6 Oct 2026 15:27:20 +0200 Subject: [PATCH 42/94] fix(zarr-metadata): restore pickling of group readings; recurse into listed groups with any problem kind Review round 4 findings: - Making `reading.consolidated` a mapping proxy broke pickle and deepcopy of group readings. The reading now has a `__reduce__` that rebuilds it from a plain dict. - The recursion into a listed group with problems only ran when that group's entry refined cleanly, so a listed group with a non-string key or a non-JSON value still hid its valid children. The reader now recurses into any listed group that is an object, and counts depth from the outer document root. Assisted-by: ClaudeCode:claude-fable-5-1 --- .../src/zarr_metadata/model/_group.py | 38 ++++++++++++++++--- .../zarr-metadata/tests/model/test_pair.py | 23 ++++++++++- 2 files changed, 53 insertions(+), 8 deletions(-) diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_group.py b/packages/zarr-metadata/src/zarr_metadata/model/_group.py index 4fe7f5e771..0f512b2f6e 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_group.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/_group.py @@ -478,6 +478,22 @@ def fields(self) -> Iterator[tuple[Loc, Resolved[Any]]]: for loc, node in reading.fields(): yield (ZARR_V3_CONSOLIDATED_METADATA_KEY, "metadata", path, *loc), node + def __reduce__( + self, + ) -> tuple[Callable[..., ZarrV3GroupMetadataReading], tuple[object, ...]]: + # Pickled and copied as built again: the read-only view the + # documents are held as does not pickle, and the dict it views does. + return (_group_reading, (dict(self.consolidated), self.problems, self.metadata)) + + +def _group_reading( + consolidated: dict[str, ZarrV3NodeMetadataReading], + problems: tuple[ValidationProblem, ...], + metadata: ZarrV3GroupMetadata | None, +) -> ZarrV3GroupMetadataReading: + """A group reading built again from what `ZarrV3GroupMetadataReading.__reduce__` gives, its documents held read-only.""" + return ZarrV3GroupMetadataReading(MappingProxyType(consolidated), problems, metadata) + @dataclass(frozen=True, slots=True) class ZarrV3UnknownNodeReading: @@ -889,14 +905,17 @@ def _with_models( members: GroupMembersV3, value: object, context: Context, + at: Loc = (), ) -> ZarrV3GroupMetadataReading: """`reading`, holding the model of each document its consolidated metadata holds that has no problem, and its own when it has none: each built from `value`, the document this read read, in `context`. A document with a problem a reader walks past -- a key that is no string, a value that is no JSON, a level past the cap -- refines to nothing as a whole; each document its consolidated metadata holds is - then refined on its own, from where it sits, so the ones without a - problem still have their models. + then refined on its own, from where it sits in the document handed in + -- `at` is where this one sits -- so the ones without a problem still + have their models, and a listed group with a problem of its own holds + those of its own listing. """ if len(reading.problems) == 0: refined, _ = refine_user_data(value) @@ -917,7 +936,7 @@ def _with_models( if path not in held_entries: continue entry, problems = refine_user_data( - held_entries[path], (ZARR_V3_CONSOLIDATED_METADATA_KEY, "metadata", path) + held_entries[path], (*at, ZARR_V3_CONSOLIDATED_METADATA_KEY, "metadata", path) ) if len(problems) == 0 and isinstance(entry, Mapping): documents[path] = entry @@ -933,14 +952,21 @@ def _with_models( if path in models: held[path] = models[path].reading elif ( - path in documents + path in held_entries and isinstance(nested, ZarrV3GroupMetadataReading) and isinstance(child, GroupMembersV3) ): - # A listed group with a problem of its own still holds, in its + # A listed group with a problem of its own -- one the reader + # refused, or one it walked past -- still holds, in its # reading, a model of each document in its own listing that # has none. - held[path] = _with_models(nested, child, held_entries[path], context) + held[path] = _with_models( + nested, + child, + held_entries[path], + context, + (*at, ZARR_V3_CONSOLIDATED_METADATA_KEY, "metadata", path), + ) return dataclasses.replace(reading, consolidated=MappingProxyType(held)) diff --git a/packages/zarr-metadata/tests/model/test_pair.py b/packages/zarr-metadata/tests/model/test_pair.py index fb38f8ff21..d2410b52f1 100644 --- a/packages/zarr-metadata/tests/model/test_pair.py +++ b/packages/zarr-metadata/tests/model/test_pair.py @@ -2,6 +2,7 @@ from __future__ import annotations +import copy import dataclasses import operator import pickle @@ -665,11 +666,19 @@ def test_a_models_reading_cannot_be_changed_in_place() -> None: assert moved == ZarrV3GroupMetadata(CONSOLIDATED) -def test_a_reading_holds_a_model_of_each_healthy_document_in_a_listed_groups_own_listing() -> None: - """A listed group with a problem of its own still holds, in its reading, a model of each document in its own listing that has no problem, as the top group does.""" +@pytest.mark.parametrize( + "fault", + [{"attributes": "bad"}, {"attributes": {1: "bad"}}, {"attributes": {"s": {1, 2}}}], + ids=["refused", "non-string-key", "not-json"], +) +def test_a_reading_holds_a_model_of_each_healthy_document_in_a_listed_groups_own_listing( + fault: dict[str, Any], +) -> None: + """A listed group with a problem of its own -- one the reader refuses, or one it walks past -- still holds, in its reading, a model of each document in its own listing that has no problem, as the top group does.""" inline = {"kind": "inline", "must_understand": False} listed = { **GROUP, + **fault, "consolidated_metadata": { **inline, "metadata": {"x": ARRAY, "y": {**ARRAY, "shape": [-1]}}, @@ -690,3 +699,13 @@ def test_a_reading_holds_a_model_of_each_healthy_document_in_a_listed_groups_own held = nested.consolidated["x"].metadata assert isinstance(held, ZarrV3ArrayMetadata) assert held == ZarrV3ArrayMetadata(ARRAY) + + +@pytest.mark.parametrize("document", [GROUP, CONSOLIDATED], ids=["group", "consolidated"]) +def test_a_group_reading_pickles_and_copies(document: dict[str, Any]) -> None: + """A group's reading pickles and deep-copies, with the documents its consolidated metadata holds and the model it built, and compares equal afterwards, as an array's does.""" + reading = read_group_metadata_v3(document) + for again in (pickle.loads(pickle.dumps(reading)), copy.deepcopy(reading)): + assert again == reading + assert again.metadata == reading.metadata + assert set(again.consolidated) == set(reading.consolidated) From c84a6470864187e1fd6f2a6116c8aeb08037e9bd Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Tue, 6 Oct 2026 15:46:01 +0200 Subject: [PATCH 43/94] fix(zarr-metadata): pickle a reading through its model Pickling or copying a reading rebuilt its model and every nested reading separately, so a document at depth k was re-read about k times, and the copy lost the one-model-per-document identity. A reading that holds a model now pickles as that model does (the pair, read once on load) and comes back as the model's own reading, nested readings included. Review round 5 found nothing critical or important; this fixes its one remaining minor. Assisted-by: ClaudeCode:claude-fable-5-1 --- .../src/zarr_metadata/model/_group.py | 15 +++++++++------ .../src/zarr_metadata/model/_validation.py | 13 +++++++++++++ packages/zarr-metadata/tests/model/test_pair.py | 17 +++++++++++++++++ 3 files changed, 39 insertions(+), 6 deletions(-) diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_group.py b/packages/zarr-metadata/src/zarr_metadata/model/_group.py index 0f512b2f6e..f1e339074d 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_group.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/_group.py @@ -52,6 +52,7 @@ other_members, parse_group_metadata_v2, read_array_v3, + reading_of, unexpected_keys, ) from zarr_metadata.v2.attributes import ZARR_V2_ATTRIBUTES_STORE_KEY @@ -478,12 +479,14 @@ def fields(self) -> Iterator[tuple[Loc, Resolved[Any]]]: for loc, node in reading.fields(): yield (ZARR_V3_CONSOLIDATED_METADATA_KEY, "metadata", path, *loc), node - def __reduce__( - self, - ) -> tuple[Callable[..., ZarrV3GroupMetadataReading], tuple[object, ...]]: - # Pickled and copied as built again: the read-only view the - # documents are held as does not pickle, and the dict it views does. - return (_group_reading, (dict(self.consolidated), self.problems, self.metadata)) + def __reduce__(self) -> tuple[Callable[..., object], tuple[object, ...]]: + # A reading that holds its model pickles and copies as the model + # does, and comes back as that model's own reading, so one model + # per document still; one without is built again from the dict its + # read-only view views, which pickles where the view does not. + if self.metadata is not None: + return (reading_of, (self.metadata,)) + return (_group_reading, (dict(self.consolidated), self.problems, None)) def _group_reading( diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_validation.py b/packages/zarr-metadata/src/zarr_metadata/model/_validation.py index 9988959e62..7e06c70735 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_validation.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/_validation.py @@ -440,6 +440,14 @@ class ZarrV3ArrayMetadataReading: metadata: ZarrV3ArrayMetadata | None = None """The document's model, holding these fields, when there is no problem; None otherwise.""" + def __reduce__(self) -> str | tuple[object, ...]: + # A reading that holds its model pickles and copies as the model + # does -- the pair, read once on load -- and comes back as that + # model's own reading, so one model per document still. + if self.metadata is not None: + return (reading_of, (self.metadata,)) + return object.__reduce__(self) + def fields(self) -> Iterator[tuple[Loc, Resolved[Any]]]: """Each field the document holds, as the scope read it, with where it sits in the document. @@ -464,6 +472,11 @@ def fields(self) -> Iterator[tuple[Loc, Resolved[Any]]]: M = TypeVar("M") +def reading_of(model: object) -> object: + """The reading `model`, a v3 model, holds: what a pickled reading that held a model is built again as.""" + return cast("Any", model).reading + + def construct(model: type[M], /, **members: object) -> M: """A `model` of `members` a read found nothing wrong with: as its constructor builds one, without checking them again. diff --git a/packages/zarr-metadata/tests/model/test_pair.py b/packages/zarr-metadata/tests/model/test_pair.py index d2410b52f1..6edfb19303 100644 --- a/packages/zarr-metadata/tests/model/test_pair.py +++ b/packages/zarr-metadata/tests/model/test_pair.py @@ -709,3 +709,20 @@ def test_a_group_reading_pickles_and_copies(document: dict[str, Any]) -> None: assert again == reading assert again.metadata == reading.metadata assert set(again.consolidated) == set(reading.consolidated) + # One model per document still: the reading comes back through + # the model it holds, which reads once. + assert again.metadata is not None + assert again.metadata.reading is again + for path, nested in again.consolidated.items(): + held = again.metadata.consolidated_metadata + assert isinstance(held, ZarrV3ConsolidatedMetadata) + assert nested.metadata is held.metadata[path] + + +def test_an_array_reading_pickles_through_its_model() -> None: + """An array's reading that holds a model pickles and deep-copies as that model does, and comes back as the model's own reading.""" + reading = read_array_metadata_v3(ARRAY) + for again in (pickle.loads(pickle.dumps(reading)), copy.deepcopy(reading)): + assert again == reading + assert again.metadata is not None + assert again.metadata.reading is again From baa96f03ed2709dd7aacb093b4c819c205904992 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Tue, 6 Oct 2026 15:46:15 +0200 Subject: [PATCH 44/94] docs(zarr-metadata): rename the document-context fragments with their PR number Assisted-by: ClaudeCode:claude-fable-5-1 --- .../changes/{+document-context.feature.md => 393.feature.md} | 0 .../changes/{+document-context.removal.md => 393.removal.md} | 0 2 files changed, 0 insertions(+), 0 deletions(-) rename packages/zarr-metadata/changes/{+document-context.feature.md => 393.feature.md} (100%) rename packages/zarr-metadata/changes/{+document-context.removal.md => 393.removal.md} (100%) diff --git a/packages/zarr-metadata/changes/+document-context.feature.md b/packages/zarr-metadata/changes/393.feature.md similarity index 100% rename from packages/zarr-metadata/changes/+document-context.feature.md rename to packages/zarr-metadata/changes/393.feature.md diff --git a/packages/zarr-metadata/changes/+document-context.removal.md b/packages/zarr-metadata/changes/393.removal.md similarity index 100% rename from packages/zarr-metadata/changes/+document-context.removal.md rename to packages/zarr-metadata/changes/393.removal.md From 6acbe60bcb69730f5975441ba31790534e1cae89 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Tue, 6 Oct 2026 18:25:55 +0200 Subject: [PATCH 45/94] docs(zarr-metadata): rewrite the 393 changelog fragments in plain English Assisted-by: ClaudeCode:claude-fable-5-1 --- packages/zarr-metadata/changes/393.feature.md | 2 +- packages/zarr-metadata/changes/393.removal.md | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/packages/zarr-metadata/changes/393.feature.md b/packages/zarr-metadata/changes/393.feature.md index 170fce57fa..df16213512 100644 --- a/packages/zarr-metadata/changes/393.feature.md +++ b/packages/zarr-metadata/changes/393.feature.md @@ -1 +1 @@ -A v3 model is now its document and the scope it was read in: `to_json` writes the document as it was written, and `update` reads changes in the model's own scope, so no scope is passed back in. `with_context`, `refined_in` and `refines` move a model between scopes and compare models by what they mean. +A v3 model is now a pair: its document and the scope it was read in. `to_json` returns the document as it was written, and `update` reads the changed document in the model's own scope, so no scope is passed in. `with_context` and `refined_in` re-read a model in another scope, and `refines` compares two models by what they mean. diff --git a/packages/zarr-metadata/changes/393.removal.md b/packages/zarr-metadata/changes/393.removal.md index 8c0676d489..2c58cf7ce8 100644 --- a/packages/zarr-metadata/changes/393.removal.md +++ b/packages/zarr-metadata/changes/393.removal.md @@ -1 +1 @@ -The v3 models are no longer dataclasses built from typed fields: `ZarrV3ArrayMetadata(shape=..., data_type=, ...)`, `dataclasses.replace` and `dataclasses.fields` on them are gone; build a model from a document and change it with `update`. `update` no longer takes `context`, and `to_json` no longer respells fields: `"bytes"` stays `"bytes"`. `attributes`, `extra_fields` and a struct `fill_value` are read-only at every level, objects as read-only mappings and arrays as tuples, so they cannot be changed in place, serialized or pickled directly; `to_json()` gives plain JSON containers. +The v3 models are no longer dataclasses built from typed fields. `ZarrV3ArrayMetadata(shape=..., data_type=, ...)`, `dataclasses.replace` and `dataclasses.fields` no longer work on them; build a model from a document and change it with `update`. `update` no longer takes `context`, and `to_json` no longer rewrites field spelling (`"bytes"` stays `"bytes"`). `attributes`, `extra_fields` and a struct `fill_value` are read-only at every level (objects as read-only mappings, arrays as tuples), so they cannot be modified in place, serialized or pickled directly; use `to_json()` for plain JSON containers. From 01001bc9ea4ccab50f3e5adc544448538f544760 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Tue, 6 Oct 2026 15:58:20 +0200 Subject: [PATCH 46/94] feat(zarr-metadata): accept context=None for the default scope in every reader Every free reader, validator and guard now takes `context: Context | None = None`, with None meaning `CORE_AND_EXTENSIONS`, matching the model constructors. Assisted-by: ClaudeCode:claude-fable-5-1 --- .../src/zarr_metadata/model/_array.py | 7 +- .../src/zarr_metadata/model/_group.py | 44 ++++--- .../src/zarr_metadata/model/_json_schema.py | 5 +- .../src/zarr_metadata/model/_repair.py | 5 +- .../src/zarr_metadata/model/_validation.py | 15 ++- .../tests/model/test_scope_threading.py | 123 ++++++++++++++++++ 6 files changed, 168 insertions(+), 31 deletions(-) create mode 100644 packages/zarr-metadata/tests/model/test_scope_threading.py diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_array.py b/packages/zarr-metadata/src/zarr_metadata/model/_array.py index ed91a274c5..8aead5e710 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_array.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/_array.py @@ -485,7 +485,7 @@ def _fill_value_key(model: ZarrV3ArrayMetadata) -> str: def read_array_metadata_v3( - value: object, *, context: Context = CORE_AND_EXTENSIONS + value: object, *, context: Context | None = None ) -> ZarrV3ArrayMetadataReading: """`value`, a v3 array document, as `context` read it, whatever it holds. @@ -498,12 +498,13 @@ def read_array_metadata_v3( walk over its `fields()`. A value that is not an object holds no field. """ - reading, members = read_array_v3(value, context) + scope = CORE_AND_EXTENSIONS if context is None else context + reading, members = read_array_v3(value, scope) if members is None: return reading refined, _ = refine_user_data(value) document = cast("dict[str, JSONValue]", refined) - model = ZarrV3ArrayMetadata._of(document, context, reading, members) # pyright: ignore[reportPrivateUsage] + model = ZarrV3ArrayMetadata._of(document, scope, reading, members) # pyright: ignore[reportPrivateUsage] return model.reading diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_group.py b/packages/zarr-metadata/src/zarr_metadata/model/_group.py index f1e339074d..8658c5ce65 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_group.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/_group.py @@ -528,7 +528,7 @@ def fields(self) -> Iterator[tuple[Loc, Resolved[Any]]]: def read_node_metadata_v3( - value: object, *, context: Context = CORE_AND_EXTENSIONS + value: object, *, context: Context | None = None ) -> ZarrV3NodeMetadataReading: """`value`, a v3 `zarr.json`, read in `context` as the node its `node_type` says it is. @@ -540,11 +540,12 @@ def read_node_metadata_v3( among them. So no caller reads `node_type` from JSON it has not read, and a document of another format says it is not v3. """ + scope = CORE_AND_EXTENSIONS if context is None else context node_type, problems = _node_type(value) if node_type == "array": - return read_array_metadata_v3(value, context=context) + return read_array_metadata_v3(value, context=scope) if node_type == "group": - return read_group_metadata_v3(value, context=context) + return read_group_metadata_v3(value, context=scope) return ZarrV3UnknownNodeReading(problems) @@ -555,7 +556,7 @@ def read_node_metadata_v3( def node_metadata_from_json_v3( - data: object, *, context: Context = CORE_AND_EXTENSIONS + data: object, *, context: Context | None = None ) -> ZarrV3NodeMetadata: """The model of `data`, a v3 `zarr.json` read in `context`, as the node its `node_type` says. @@ -564,30 +565,33 @@ def node_metadata_from_json_v3( `MetadataValidationError` with every problem `read_node_metadata_v3` finds, a `node_type` that says neither among them. """ - reading = read_node_metadata_v3(data, context=context) + scope = CORE_AND_EXTENSIONS if context is None else context + reading = read_node_metadata_v3(data, context=scope) if reading.metadata is None: raise MetadataValidationError(reading.problems) return reading.metadata def node_metadata_from_key_value_v3( - mapping: Mapping[StoreKey, bytes], *, context: Context = CORE_AND_EXTENSIONS + mapping: Mapping[StoreKey, bytes], *, context: Context | None = None ) -> ZarrV3NodeMetadata: """The model of the document at `zarr.json` in `mapping`, read in `context` as the node its `node_type` says, as `node_metadata_from_json_v3` reads one. `MetadataValidationError` when the key is missing, its bytes are not JSON, or the document is not a valid array or group. """ + scope = CORE_AND_EXTENSIONS if context is None else context # An array's document and a group's are both at `zarr.json`. document = load_store_json(mapping, ZARR_V3_GROUP_METADATA_STORE_KEY) - return node_metadata_from_json_v3(document, context=context) + return node_metadata_from_json_v3(document, context=scope) def validate_node_metadata_v3( - value: object, *, context: Context = CORE_AND_EXTENSIONS + value: object, *, context: Context | None = None ) -> tuple[ValidationProblem, ...]: """Every reason `value` is not a valid v3 `zarr.json`: those `validate_array_metadata_v3` or `validate_group_metadata_v3` finds in the node its `node_type` says it is, or why it says neither.""" - return _read_node_v3(value, context)[0].problems + scope = CORE_AND_EXTENSIONS if context is None else context + return _read_node_v3(value, scope)[0].problems def _read_node_v3( @@ -650,7 +654,7 @@ class GroupMembersV3: def read_group_metadata_v3( - value: object, *, context: Context = CORE_AND_EXTENSIONS + value: object, *, context: Context | None = None ) -> ZarrV3GroupMetadataReading: """`value`, a v3 group document, as `context` read it, whatever it holds. @@ -661,10 +665,11 @@ def read_group_metadata_v3( models of those documents, which their readings hold too. A value that is not an object holds nothing. """ - reading, members = read_group_v3(value, context) + scope = CORE_AND_EXTENSIONS if context is None else context + reading, members = read_group_v3(value, scope) if members is None: return reading - return _with_models(reading, members, value, context) + return _with_models(reading, members, value, scope) def read_group_v3( @@ -1000,7 +1005,7 @@ def _nested_models( def validate_group_metadata_v3( - value: object, *, context: Context = CORE_AND_EXTENSIONS + value: object, *, context: Context | None = None ) -> tuple[ValidationProblem, ...]: """Return every reason `value` is not a valid v3 group document. @@ -1013,23 +1018,26 @@ def validate_group_metadata_v3( `problems` of `read_group_metadata_v3`, which holds what was read to find them. """ - return read_group_v3(value, context)[0].problems + scope = CORE_AND_EXTENSIONS if context is None else context + return read_group_v3(value, scope)[0].problems def is_group_metadata_v3( - value: object, *, context: Context = CORE_AND_EXTENSIONS + value: object, *, context: Context | None = None ) -> TypeGuard[ZarrV3GroupMetadataJSON]: """Whether `value` is a v3 group document `validate_group_metadata_v3` finds nothing wrong with, written with tuples.""" + scope = CORE_AND_EXTENSIONS if context is None else context return is_canonical_json(value, finite=False) and not validate_group_metadata_v3( - value, context=context + value, context=scope ) def parse_group_metadata_v3( - value: object, *, context: Context = CORE_AND_EXTENSIONS + value: object, *, context: Context | None = None ) -> ZarrV3GroupMetadataJSON: """Return `value` narrowed to `ZarrV3GroupMetadataJSON`, or raise `MetadataValidationError`.""" - problems = validate_group_metadata_v3(value, context=context) + scope = CORE_AND_EXTENSIONS if context is None else context + problems = validate_group_metadata_v3(value, context=scope) if len(problems) != 0: raise MetadataValidationError(problems) return cast("ZarrV3GroupMetadataJSON", arrays_to_tuples(value)) diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_json_schema.py b/packages/zarr-metadata/src/zarr_metadata/model/_json_schema.py index 05dd0aef64..5dd53e0f3b 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_json_schema.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/_json_schema.py @@ -19,7 +19,7 @@ from zarr_metadata._typed_json import JSONSchema, SchemaLeaf -def node_metadata_json_schema_v3(*, context: Context = CORE_AND_EXTENSIONS) -> JSONSchema: +def node_metadata_json_schema_v3(*, context: Context | None = None) -> JSONSchema: """The JSON Schema of a v3 `zarr.json` read in `context`: an array document or a group document, as `validate_node_metadata_v3` reads one, but for the rules. For an editor that validates a `zarr.json` as it is written, or a @@ -45,7 +45,8 @@ def node_metadata_json_schema_v3(*, context: Context = CORE_AND_EXTENSIONS) -> J as lists: a model's `to_json` writes tuples, which a Python validator does not take for arrays. """ - schemas = Schemas(_documents(context)) + scope = CORE_AND_EXTENSIONS if context is None else context + schemas = Schemas(_documents(scope)) array = schemas.of(ZarrV3ArrayMetadataJSON) group = schemas.of(ZarrV3GroupMetadataJSON) return schemas.document({"anyOf": [array, group]}) diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_repair.py b/packages/zarr-metadata/src/zarr_metadata/model/_repair.py index 22936388cf..9b4bac4f81 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_repair.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/_repair.py @@ -205,7 +205,7 @@ class ZarrV3RepairedNodeMetadataReading: def read_repaired_node_metadata_v3( - value: object, *, context: Context = CORE_AND_EXTENSIONS + value: object, *, context: Context | None = None ) -> ZarrV3RepairedNodeMetadataReading: """`value`, a v3 `zarr.json`, read in `context` as `read_node_metadata_v3` reads it, once `repair_node_metadata_v3` has undone each known writer bug in it. @@ -214,9 +214,10 @@ def read_repaired_node_metadata_v3( applies to is read as it is, and reported as `read_node_metadata_v3` reports it. """ + scope = CORE_AND_EXTENSIONS if context is None else context repaired, repairs = repair_node_metadata_v3(value) return ZarrV3RepairedNodeMetadataReading( - read_node_metadata_v3(repaired, context=context), repairs + read_node_metadata_v3(repaired, context=scope), repairs ) diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_validation.py b/packages/zarr-metadata/src/zarr_metadata/model/_validation.py index 7e06c70735..3f78329592 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_validation.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/_validation.py @@ -641,7 +641,7 @@ def read_array_v3( def validate_array_metadata_v3( - value: object, *, context: Context = CORE_AND_EXTENSIONS + value: object, *, context: Context | None = None ) -> tuple[ValidationProblem, ...]: """Return every reason `value` is not a valid v3 array document. @@ -663,25 +663,28 @@ def validate_array_metadata_v3( reports as `must_understand_fields`. These are the `problems` of `read_array_metadata_v3`, which holds what was read to find them. """ - return read_array_v3(value, context)[0].problems + scope = CORE_AND_EXTENSIONS if context is None else context + return read_array_v3(value, scope)[0].problems def is_array_metadata_v3( - value: object, *, context: Context = CORE_AND_EXTENSIONS + value: object, *, context: Context | None = None ) -> TypeGuard[ZarrV3ArrayMetadataJSON]: """Whether `value` is a v3 array document `validate_array_metadata_v3` finds nothing wrong with, written with tuples.""" + scope = CORE_AND_EXTENSIONS if context is None else context return ( _is_canonical_json(value, finite=False) - and not validate_array_metadata_v3(value, context=context) + and not validate_array_metadata_v3(value, context=scope) and _is_canonical_array_metadata_v3(value) ) def parse_array_metadata_v3( - value: object, *, context: Context = CORE_AND_EXTENSIONS + value: object, *, context: Context | None = None ) -> ZarrV3ArrayMetadataJSON: """Return `value` as `ZarrV3ArrayMetadataJSON`, or raise `MetadataValidationError`.""" - problems = validate_array_metadata_v3(value, context=context) + scope = CORE_AND_EXTENSIONS if context is None else context + problems = validate_array_metadata_v3(value, context=scope) if len(problems) != 0: raise MetadataValidationError(problems) return cast("ZarrV3ArrayMetadataJSON", arrays_to_tuples(value)) diff --git a/packages/zarr-metadata/tests/model/test_scope_threading.py b/packages/zarr-metadata/tests/model/test_scope_threading.py new file mode 100644 index 0000000000..8051dadb9d --- /dev/null +++ b/packages/zarr-metadata/tests/model/test_scope_threading.py @@ -0,0 +1,123 @@ +"""Every entry point reads in the scope it is given, and in `CORE_AND_EXTENSIONS` when given none.""" + +from __future__ import annotations + +import json +from typing import TYPE_CHECKING, Any + +import pytest + +import zarr_metadata.model as zm +from zarr_metadata.model import ZarrV3ArrayMetadata +from zarr_metadata.v3.definition import CORE_AND_EXTENSIONS, Context + +if TYPE_CHECKING: + from collections.abc import Callable + +EMPTY = Context.of() +BASE: dict[str, Any] = dict(ZarrV3ArrayMetadata.create_default(shape=(4,)).to_json()) +BAD: dict[str, Any] = { + **BASE, + "codecs": (*BASE["codecs"], {"name": "gzip", "configuration": {"level": 12}}), +} +"""An array the default scope refuses -- a gzip level of 12 -- and an empty scope leaves unjudged.""" +GROUP: dict[str, Any] = { + "zarr_format": 3, + "node_type": "group", + "consolidated_metadata": {"kind": "inline", "must_understand": False, "metadata": {"a": BAD}}, +} +STORE_A = {"zarr.json": json.dumps(BAD).encode()} +STORE_G = {"zarr.json": json.dumps(GROUP).encode()} + + +def _accepts(read: Callable[..., object]) -> Callable[[Context | None], bool]: + """Whether `read`, given `context`, finds nothing wrong: True for a model, a reading with a model, an empty problem tuple, or a guard saying yes.""" + + def accepted(context: Context | None) -> bool: + try: + found = read(context=context) + except zm.MetadataValidationError: + return False + if isinstance(found, tuple): + return len(found) == 0 + if isinstance(found, bool): + return found + reading = getattr(found, "reading", found) + metadata = getattr(reading, "metadata", found) + return metadata is not None + + return accepted + + +ENTRY_POINTS: dict[str, Callable[..., object]] = { + "validate_array_metadata_v3": lambda context: zm.validate_array_metadata_v3( + BAD, context=context + ), + "is_array_metadata_v3": lambda context: zm.is_array_metadata_v3( + zm.parse_array_metadata_v3(BAD, context=EMPTY), context=context + ), + "parse_array_metadata_v3": lambda context: zm.parse_array_metadata_v3(BAD, context=context), + "read_array_metadata_v3": lambda context: zm.read_array_metadata_v3(BAD, context=context), + "ZarrV3ArrayMetadata": lambda context: zm.ZarrV3ArrayMetadata(BAD, context=context), + "ZarrV3ArrayMetadata.from_json": lambda context: zm.ZarrV3ArrayMetadata.from_json( + BAD, context=context + ), + "ZarrV3ArrayMetadata.from_key_value": lambda context: zm.ZarrV3ArrayMetadata.from_key_value( + STORE_A, context=context + ), + "ZarrV3ArrayMetadata.create_default": lambda context: zm.ZarrV3ArrayMetadata.create_default( + shape=(4,), codecs=BAD["codecs"], context=context + ), + "validate_group_metadata_v3": lambda context: zm.validate_group_metadata_v3( + GROUP, context=context + ), + "is_group_metadata_v3": lambda context: zm.is_group_metadata_v3( + zm.parse_group_metadata_v3(GROUP, context=EMPTY), context=context + ), + "parse_group_metadata_v3": lambda context: zm.parse_group_metadata_v3(GROUP, context=context), + "read_group_metadata_v3": lambda context: zm.read_group_metadata_v3(GROUP, context=context), + "ZarrV3GroupMetadata": lambda context: zm.ZarrV3GroupMetadata(GROUP, context=context), + "ZarrV3GroupMetadata.from_json": lambda context: zm.ZarrV3GroupMetadata.from_json( + GROUP, context=context + ), + "ZarrV3GroupMetadata.from_key_value": lambda context: zm.ZarrV3GroupMetadata.from_key_value( + STORE_G, context=context + ), + "ZarrV3GroupMetadata.create_default": lambda context: zm.ZarrV3GroupMetadata.create_default( + consolidated_metadata=GROUP["consolidated_metadata"], context=context + ), + "ZarrV3ConsolidatedMetadata": lambda context: zm.ZarrV3ConsolidatedMetadata( + GROUP["consolidated_metadata"], context=context + ), + "ZarrV3ConsolidatedMetadata.from_json": lambda context: zm.ZarrV3ConsolidatedMetadata.from_json( + GROUP["consolidated_metadata"], context=context + ), + "read_node_metadata_v3": lambda context: zm.read_node_metadata_v3(GROUP, context=context), + "validate_node_metadata_v3": lambda context: zm.validate_node_metadata_v3( + GROUP, context=context + ), + "node_metadata_from_json_v3": lambda context: zm.node_metadata_from_json_v3( + GROUP, context=context + ), + "node_metadata_from_key_value_v3": lambda context: zm.node_metadata_from_key_value_v3( + STORE_G, context=context + ), + "read_repaired_node_metadata_v3": lambda context: zm.read_repaired_node_metadata_v3( + GROUP, context=context + ), +} + + +@pytest.mark.parametrize("read", ENTRY_POINTS.values(), ids=ENTRY_POINTS.keys()) +def test_every_entry_point_reads_in_the_scope_it_is_given(read: Callable[..., object]) -> None: + """Each entry point refuses a gzip level of 12 in the default scope, which `None` names too, and accepts it in a scope that leaves gzip unclaimed: the scope it is given is the scope it reads in.""" + accepted = _accepts(read) + assert accepted(EMPTY) is True + assert accepted(CORE_AND_EXTENSIONS) is False + assert accepted(None) is False + + +def test_the_json_schema_is_written_in_the_scope_it_is_given() -> None: + """`node_metadata_json_schema_v3` takes `None` for the default scope, and writes a different schema for an empty one.""" + assert zm.node_metadata_json_schema_v3(context=None) == zm.node_metadata_json_schema_v3() + assert zm.node_metadata_json_schema_v3(context=EMPTY) != zm.node_metadata_json_schema_v3() From f3f14da87c56155ff39d82bdad87984c83bfd7aa Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Tue, 6 Oct 2026 16:02:46 +0200 Subject: [PATCH 47/94] feat(zarr-metadata): accept node models as consolidated-metadata entries A `ZarrV3ArrayMetadata` or `ZarrV3GroupMetadata` can be given where consolidated metadata lists a document. The model is accepted when the group's scope reads every one of its claims the same way (its reading is kept, no re-read), or claims names the model's scope left unclaimed (its document is re-read in the group's scope). If the two scopes read a name with different definitions, or the group's scope reads a claimed name by none, the model is refused with a problem at its path. A `ZarrV3ConsolidatedMetadata` given as the whole member contributes its models. The group's document stores each child's document as the child wrote it. Assisted-by: ClaudeCode:claude-fable-5-1 --- .../src/zarr_metadata/model/__init__.py | 2 + .../src/zarr_metadata/model/_group.py | 118 +++++++++++- .../tests/model/test_consolidating.py | 175 ++++++++++++++++++ .../zarr-metadata/tests/model/test_group.py | 14 +- 4 files changed, 293 insertions(+), 16 deletions(-) create mode 100644 packages/zarr-metadata/tests/model/test_consolidating.py diff --git a/packages/zarr-metadata/src/zarr_metadata/model/__init__.py b/packages/zarr-metadata/src/zarr_metadata/model/__init__.py index a75b9904dc..73ce61bf63 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/__init__.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/__init__.py @@ -51,6 +51,7 @@ ZarrV2GroupMetadata, ZarrV2GroupMetadataPartial, ZarrV3ConsolidatedMetadata, + ZarrV3ConsolidatedMetadataInput, ZarrV3GroupMetadata, ZarrV3GroupMetadataReading, ZarrV3GroupMetadataUpdate, @@ -177,6 +178,7 @@ "ZarrV3ArrayMetadataStoreKey", "ZarrV3ArrayMetadataUpdate", "ZarrV3ConsolidatedMetadata", + "ZarrV3ConsolidatedMetadataInput", "ZarrV3GroupMetadata", "ZarrV3GroupMetadataReading", "ZarrV3GroupMetadataStoreKey", diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_group.py b/packages/zarr-metadata/src/zarr_metadata/model/_group.py index 8658c5ce65..86f27615c6 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_group.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/_group.py @@ -6,7 +6,7 @@ from collections.abc import Callable, Mapping from dataclasses import dataclass, field from types import MappingProxyType -from typing import TYPE_CHECKING, Any, Final, Literal, TypeGuard, TypeVar, cast +from typing import TYPE_CHECKING, Any, Final, Literal, TypeAlias, TypeGuard, TypeVar, cast from typing_extensions import TypeAliasType, TypedDict, Unpack @@ -74,20 +74,43 @@ from zarr_metadata.v2.consolidated import ZarrV2ConsolidatedMetadataStoreKey from zarr_metadata.v2.group import ZarrV2GroupMetadataJSON, ZarrV2GroupMetadataStoreKey from zarr_metadata.v3._definition import Resolved + from zarr_metadata.v3.array import ZarrV3ArrayMetadataJSON from zarr_metadata.v3.consolidated import ZarrV3ConsolidatedMetadataJSON from zarr_metadata.v3.group import ZarrV3GroupMetadataJSONPartial, ZarrV3GroupMetadataStoreKey +ZarrV3NodeMetadataInput: TypeAlias = ( + "ZarrV3ArrayMetadataJSON | ZarrV3GroupMetadataJSON | ZarrV3ArrayMetadata | ZarrV3GroupMetadata" +) +"""What consolidated metadata lists at a path when given to a constructor or `update`: a document, or a model of it.""" + + +class ZarrV3ConsolidatedMetadataInput(TypedDict, closed=True): + """The `consolidated_metadata` member as a constructor or `update` takes it: as a document writes it, each entry a document or a node model. + + A node model is accepted when the group's scope reads every claim of + it identically, or claims what the model's scope left unclaimed -- it + is then read again there -- and refused, with a problem at its path, + where the two scopes read a name differently, or the group's reads it + by none. + """ + + kind: Literal["inline"] + must_understand: Literal[False] + metadata: Mapping[str, ZarrV3NodeMetadataInput] + + class ZarrV3GroupMetadataUpdate(TypedDict, total=False, extra_items=ZarrV3ExtensionField | UNSET): """The members `ZarrV3GroupMetadata.update` puts in place: each as a document writes it, or `UNSET` to leave it out. - `consolidated_metadata` is a member the spec does not define, so it is - one of the extra items: given, its documents are read in place of the - document's; left out, the document's are read again as part of the - whole. + `consolidated_metadata` is given as a document writes it, each entry a + document or a node model, as `ZarrV3ConsolidatedMetadataInput` says, + or as another group's `ZarrV3ConsolidatedMetadata`, whose models are + taken; left out, the document's are read again as part of the whole. """ attributes: Mapping[str, JSONValue] | UNSET + consolidated_metadata: ZarrV3ConsolidatedMetadataInput | ZarrV3ConsolidatedMetadata | UNSET class ZarrV3GroupMetadata: @@ -123,7 +146,7 @@ def __init__(self, document: object, context: Context | None = None) -> None: reading, members = read_group_v3(document, scope) if members is None or len(reading.problems) != 0: raise MetadataValidationError(reading.problems) - refined, _ = refine_user_data(document) + refined, _ = refine_user_data(documents_for(document)) self._adopt(cast("dict[str, JSONValue]", refined), scope, reading, members) @classmethod @@ -373,7 +396,7 @@ def __init__(self, member: object, context: Context | None = None) -> None: ) if len(problems) != 0: raise MetadataValidationError(problems) - refined, _ = refine_user_data(member) + refined, _ = refine_user_data(_member_documents_for(member)) document = cast("dict[str, JSONValue]", refined) documents = cast("dict[str, JSONValue]", document["metadata"]) self._adopt(document, scope, _nested_models(documents, scope, readings, members)) @@ -669,7 +692,7 @@ def read_group_metadata_v3( reading, members = read_group_v3(value, scope) if members is None: return reading - return _with_models(reading, members, value, scope) + return _with_models(reading, members, documents_for(value), scope) def read_group_v3( @@ -722,6 +745,77 @@ def read_group_v3( _CONSOLIDATED_MEMBERS: Final = ("kind", "must_understand", "metadata") """The members of an inline `consolidated_metadata`, in the order the convention declares them.""" +_CONSOLIDATED_ENVELOPE: Final[dict[str, object]] = {"kind": "inline", "must_understand": False} +"""What the member declares, by declaration.""" + + +def documents_for(value: object) -> object: + """`value`, a v3 group document, with each node model its consolidated metadata lists replaced by that model's document, and a `ZarrV3ConsolidatedMetadata` given as the member by the member it holds; `value` itself when it holds none. + + What a group's own document is built from, so a document built of + models is JSON as any other, each child written as the child wrote it. + """ + if not isinstance(value, Mapping): + return value + document = cast("Mapping[object, object]", value) + given = document.get(ZARR_V3_CONSOLIDATED_METADATA_KEY) + member = _member_documents_for(given) + if member is given: + return cast("object", value) + return {**document, ZARR_V3_CONSOLIDATED_METADATA_KEY: member} + + +def _member_documents_for(member: object) -> object: + """A `consolidated_metadata` member with each node model it lists replaced by its document; `member` itself when it lists none.""" + if isinstance(member, ZarrV3ConsolidatedMetadata): + return member._document # pyright: ignore[reportPrivateUsage] + if not isinstance(member, Mapping): + return member + entries = cast("Mapping[object, object]", member).get("metadata") + if not isinstance(entries, Mapping): + return cast("object", member) + replaced: dict[object, object] = {} + changed = False + for path, entry in cast("Mapping[object, object]", entries).items(): + if isinstance(entry, (ZarrV3ArrayMetadata, ZarrV3GroupMetadata)): + replaced[path] = entry._document # pyright: ignore[reportPrivateUsage] + changed = True + else: + replaced[path] = entry + if not changed: + return cast("object", member) + return {**cast("Mapping[object, object]", member), "metadata": replaced} + + +def _read_node_model( + entry: ZarrV3ArrayMetadata | ZarrV3GroupMetadata, context: Context, at: Loc +) -> tuple[ZarrV3NodeMetadataReading, ArrayMembersV3 | GroupMembersV3 | None]: + """`entry`, a node model given where a document is listed, as `context` takes it: its own reading when `context` reads every claim of it identically, a read of its document when `context` claims more, and problems where `context` reads a name otherwise, or by none. + + A model may join a group when its claims refine into the group's + scope. A gain reads the document again there, so a problem a newly + claimed definition finds is reported where it sits. + """ + found = context.disagreements(entry.claims) + if len(found.conflicts) != 0: + conflicts = located_conflicts(entry.reading.fields(), found.conflicts) + problems = tuple( + ValidationProblem( + conflict.loc if conflict.loc is not None else (), + f"expected a document read in the group's scope, got a model that reads " + f"{conflict.key[1]!r} by {conflict.claimed!r}, which the scope reads by " + f"{conflict.found!r}", + "invalid_value", + ) + for conflict in conflicts + ) + refused = dataclasses.replace(entry.reading, problems=problems, metadata=None) + return refused, None + if found.agrees: + moved = entry.with_context(context) + return moved.reading, moved._members # pyright: ignore[reportPrivateUsage] + return _read_node_v3(entry._document, context, at) # pyright: ignore[reportPrivateUsage] + def _read_consolidated_v3( value: object, context: Context, at: tuple[str | int, ...] = () @@ -737,6 +831,9 @@ def _read_consolidated_v3( -- is judged where it sits, as `_refine` judges one, so a chain of documents is bounded by the levels a reader walks. """ + if isinstance(value, ZarrV3ConsolidatedMetadata): + # Another group's member, given whole: its models at their paths. + value = {**_CONSOLIDATED_ENVELOPE, "metadata": dict(value.metadata)} past = nested_past_the_levels(value, at) if past is not None: return {}, {}, within((past,), at) @@ -772,7 +869,10 @@ def _read_consolidated_v3( continue faults = _key_problems(key) problems.extend(faults) - readings[key], child = _read_node_v3(entry, context, (*at, "metadata", key)) + if isinstance(entry, (ZarrV3ArrayMetadata, ZarrV3GroupMetadata)): + readings[key], child = _read_node_model(entry, context, (*at, "metadata", key)) + else: + readings[key], child = _read_node_v3(entry, context, (*at, "metadata", key)) if child is not None: members[key] = child problems.extend(_prefix("metadata", _prefix(key, readings[key].problems))) diff --git a/packages/zarr-metadata/tests/model/test_consolidating.py b/packages/zarr-metadata/tests/model/test_consolidating.py new file mode 100644 index 0000000000..593158d6fd --- /dev/null +++ b/packages/zarr-metadata/tests/model/test_consolidating.py @@ -0,0 +1,175 @@ +"""A group's consolidated metadata built from node models: each accepted when it reads the same in the group's scope, or gains there; refused when it conflicts or would lose.""" + +from __future__ import annotations + +import pickle +from typing import Any, cast + +import pytest + +from zarr_metadata.model import ( + MetadataValidationError, + ZarrV3ArrayMetadata, + ZarrV3ConsolidatedMetadata, + ZarrV3GroupMetadata, + read_group_metadata_v3, + validate_group_metadata_v3, +) +from zarr_metadata.v3.codec.crc32c import Empty +from zarr_metadata.v3.definition import CORE, CORE_AND_EXTENSIONS, CodecDefinition, Context + +ARRAY: dict[str, Any] = { + **ZarrV3ArrayMetadata.create_default(shape=(4,)).to_json(), + "codecs": ("bytes",), +} +"""A `uint8` array whose `bytes` takes no configuration, so a private `bytes` of none reads it too.""" +ZSTD = {"name": "zstd", "configuration": {"level": 3, "checksum": False}} +LOOSE_ZSTD = {"name": "zstd", "configuration": {"level": 30}} +WITH_ZSTD = {**ARRAY, "codecs": (*ARRAY["codecs"], ZSTD)} +WITH_LOOSE_ZSTD = {**ARRAY, "codecs": (*ARRAY["codecs"], LOOSE_ZSTD)} +MY_BYTES = CodecDefinition(name="bytes", configuration=Empty, kind="array_bytes", size="static") +PRIVATE = CORE.extended_with(MY_BYTES) +INLINE: dict[str, Any] = {"kind": "inline", "must_understand": False} + + +def _group(scope: Context | None = None, **entries: object) -> ZarrV3GroupMetadata: + member: Any = {**INLINE, "metadata": entries} + return ZarrV3GroupMetadata.create_default(context=scope, consolidated_metadata=member) + + +@pytest.mark.parametrize( + ("scope", "child", "beside", "gained"), + [ + (CORE_AND_EXTENSIONS, ZarrV3ArrayMetadata(ARRAY), {}, False), + (CORE_AND_EXTENSIONS, ZarrV3ArrayMetadata(ARRAY, context=CORE), {}, False), + (CORE_AND_EXTENSIONS, ZarrV3ArrayMetadata(WITH_ZSTD, context=CORE), {}, True), + ( + CORE, + ZarrV3GroupMetadata( + _group(CORE, x=ZarrV3ArrayMetadata(ARRAY, context=CORE)).to_json(), + context=Context.of(), + ), + {"a/x": ZarrV3ArrayMetadata(ARRAY, context=Context.of())}, + True, + ), + ], + ids=["same-scope", "unused-definitions", "gain", "nested-models"], +) +def test_a_model_is_accepted_as_a_consolidated_entry_when_it_refines_into_the_scope( + scope: Context, + child: ZarrV3ArrayMetadata | ZarrV3GroupMetadata, + beside: dict[str, ZarrV3ArrayMetadata], + gained: bool, +) -> None: + """A node model given where consolidated metadata lists a document is accepted when the group's scope reads every claim of it identically -- its reading is kept, not read again -- or claims what the model's scope left unclaimed, when its document is read again there; the group then holds it in its own scope, equal to the group built of the document. A listed group's own listing is listed flat beside it, as the convention asks.""" + group = _group(scope, a=child, **beside) + held = group.consolidated_metadata + assert isinstance(held, ZarrV3ConsolidatedMetadata) + node = held.metadata["a"] + assert node.context is group.context + assert cast("Any", node).refines(child) + assert (node == child) is not gained + if not gained and isinstance(child, ZarrV3ArrayMetadata): + assert isinstance(node, ZarrV3ArrayMetadata) + assert node.reading.pipeline is child.reading.pipeline + written = {path: node.to_json() for path, node in beside.items()} + assert group == _group(scope, a=child.to_json(), **written) + + +def test_documents_as_written_and_pickle() -> None: + """A group built of models writes each child's document as the child wrote it, and pickles as any group does.""" + child = ZarrV3ArrayMetadata({**ARRAY, "data_type": {"name": "uint8"}}) + group = _group(None, a=child) + written = cast("Any", group.to_json()["consolidated_metadata"]) + assert written["metadata"]["a"] == child.to_json() + assert pickle.loads(pickle.dumps(group)) == group + + +def test_error_a_model_that_conflicts_with_the_scope_is_refused_at_its_path() -> None: + """A model read with a private `bytes`, given to a group whose scope reads the core `bytes`, is a conflict: a problem at the entry's path, naming the field, not a silent re-read.""" + child = ZarrV3ArrayMetadata(ARRAY, context=PRIVATE) + with pytest.raises(MetadataValidationError) as raised: + _group(CORE_AND_EXTENSIONS, a=child) + (problem,) = raised.value.problems + assert problem.loc == ("consolidated_metadata", "metadata", "a", "codecs", 0) + assert problem.kind == "invalid_value" + assert "bytes" in problem.message + + +def test_error_a_model_that_would_lose_a_meaning_is_refused_at_its_path() -> None: + """A model read where `zstd` is claimed, given to a group whose scope leaves it unclaimed, would lose what it reads: refused as a conflict is.""" + child = ZarrV3ArrayMetadata(WITH_ZSTD, context=CORE_AND_EXTENSIONS) + with pytest.raises(MetadataValidationError) as raised: + _group(CORE, a=child) + assert [problem.loc for problem in raised.value.problems] == [ + ("consolidated_metadata", "metadata", "a", "codecs", 1) + ] + + +def test_error_a_gain_that_surfaces_a_problem_is_reported_at_the_problem() -> None: + """A model whose scope left `zstd` unclaimed, given to a group whose scope claims it, is read again there: a level that definition refuses is a problem where it sits.""" + child = ZarrV3ArrayMetadata(WITH_LOOSE_ZSTD, context=CORE) + with pytest.raises(MetadataValidationError) as raised: + _group(CORE_AND_EXTENSIONS, a=child) + assert [problem.loc for problem in raised.value.problems] == [ + ("consolidated_metadata", "metadata", "a", "codecs", 1, "configuration", "level") + ] + + +def test_readers_see_models_as_the_constructor_does() -> None: + """`read_group_metadata_v3` and `validate_group_metadata_v3` given a document holding models read them as the constructor does: the reading holds the adopted model, and the validator reports a conflict.""" + fine = {**INLINE, "metadata": {"a": ZarrV3ArrayMetadata(ARRAY)}} + document = {"zarr_format": 3, "node_type": "group", "consolidated_metadata": fine} + reading = read_group_metadata_v3(document) + assert reading.problems == () + assert isinstance(reading.consolidated["a"].metadata, ZarrV3ArrayMetadata) + assert validate_group_metadata_v3(document) == () + clashing = { + **document, + "consolidated_metadata": { + **INLINE, + "metadata": {"a": ZarrV3ArrayMetadata(ARRAY, context=PRIVATE)}, + }, + } + assert [problem.loc for problem in validate_group_metadata_v3(clashing)] == [ + ("consolidated_metadata", "metadata", "a", "codecs", 0) + ] + + +def test_consolidated_metadata_given_whole_is_taken_as_its_models() -> None: + """A `ZarrV3ConsolidatedMetadata` given as the member is its models at their paths: `update(consolidated_metadata=other.consolidated_metadata)` carries them over.""" + source = _group(CORE_AND_EXTENSIONS, a=ZarrV3ArrayMetadata(ARRAY)) + target = ZarrV3GroupMetadata.create_default(attributes={"t": 1}).update( + consolidated_metadata=source.consolidated_metadata + ) + assert target.consolidated_metadata == source.consolidated_metadata + assert target.attributes == {"t": 1} + + +def test_children_of_different_scopes_consolidate_in_their_join() -> None: + """Children read in different scopes are consolidated in `Context.joined` of them: each refines into the join, and the group reads in it.""" + a = ZarrV3ArrayMetadata(ARRAY, context=CORE) + b = ZarrV3ArrayMetadata(WITH_ZSTD, context=CORE_AND_EXTENSIONS) + scope = Context.joined(a.context, b.context) + group = _group(scope, a=a, b=b) + assert group.context == CORE_AND_EXTENSIONS + held = group.consolidated_metadata + assert isinstance(held, ZarrV3ConsolidatedMetadata) + assert held.metadata["a"] == a + assert held.metadata["b"] == b + + +def test_update_keeps_a_models_scope_apart_from_the_groups() -> None: + """A model given to `update` keeps nothing of its own scope in the group: the group's `context` is the group's, and the child's is the group's too.""" + group = ZarrV3GroupMetadata.create_default(context=CORE) + updated = group.update( + consolidated_metadata={ + "kind": "inline", + "must_understand": False, + "metadata": {"a": ZarrV3ArrayMetadata(ARRAY)}, + } + ) + assert updated.context == CORE + held = updated.consolidated_metadata + assert isinstance(held, ZarrV3ConsolidatedMetadata) + assert held.metadata["a"].context == CORE diff --git a/packages/zarr-metadata/tests/model/test_group.py b/packages/zarr-metadata/tests/model/test_group.py index fc8ce933bb..d276757e49 100644 --- a/packages/zarr-metadata/tests/model/test_group.py +++ b/packages/zarr-metadata/tests/model/test_group.py @@ -548,13 +548,13 @@ def test_group_partial_keys_match_settable_model_fields() -> None: def test_update_takes_every_member_of_the_document_it_may_change() -> None: - """Each v3 model's `update` takes each member of its document but `zarr_format` and `node_type`, which it cannot change.""" - for update, partial in ( - (ZarrV3ArrayMetadataUpdate, ZarrV3ArrayMetadataJSONPartial), - (ZarrV3GroupMetadataUpdate, ZarrV3GroupMetadataJSONPartial), + """Each v3 model's `update` takes each member of its document but `zarr_format` and `node_type`, which it cannot change; a group's declares `consolidated_metadata` too, a member the document leaves to its extra items, since `update` takes node models there.""" + fixed = {"zarr_format", "node_type"} + for update, partial, declared in ( + (ZarrV3ArrayMetadataUpdate, ZarrV3ArrayMetadataJSONPartial, set[str]()), + (ZarrV3GroupMetadataUpdate, ZarrV3GroupMetadataJSONPartial, {"consolidated_metadata"}), ): - fixed = {"zarr_format", "node_type"} - assert set(update.__annotations__) == set(partial.__annotations__) - fixed + assert set(update.__annotations__) == (set(partial.__annotations__) - fixed) | declared # --- ZarrV3ConsolidatedMetadata -------------------------------------------- @@ -661,7 +661,7 @@ def test_group_update_keeps_the_documents_it_holds() -> None: def test_group_update_reads_the_documents_it_is_given_in_its_scope() -> None: """And `UNSET` leaves them out. `zstd` is an extension, which `CORE` leaves unclaimed.""" zstd = {"name": "zstd", "configuration": {"level": 3, "checksum": False}} - member = cast("JSONValue", _inline(a=_array(codecs=[LITTLE, zstd]))) + member = cast("Any", _inline(a=_array(codecs=[LITTLE, zstd]))) updated = ZarrV3GroupMetadata.create_default(context=CORE).update(consolidated_metadata=member) assert updated.consolidated_metadata is not UNSET child = updated.consolidated_metadata.metadata["a"] From 513f6989c25f73830069f17dd00a7873d783ab5b Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Tue, 6 Oct 2026 16:03:22 +0200 Subject: [PATCH 48/94] chore(zarr-metadata): run pyright unpinned The pin held pyright at the first release that typed typing_extensions.Sentinel in every position. That fix has shipped, so the pin is no longer needed. Assisted-by: ClaudeCode:claude-fable-5-1 --- .github/workflows/zarr-metadata.yml | 2 +- packages/zarr-metadata/justfile | 9 +++------ packages/zarr-metadata/pyproject.toml | 3 +-- 3 files changed, 5 insertions(+), 9 deletions(-) diff --git a/.github/workflows/zarr-metadata.yml b/.github/workflows/zarr-metadata.yml index a4a06670db..16f77839ea 100644 --- a/.github/workflows/zarr-metadata.yml +++ b/.github/workflows/zarr-metadata.yml @@ -102,7 +102,7 @@ jobs: repo: casey/just version: 1.58.0 - name: Run pyright - # The pyright version and interpreter pins live in the justfile. + # Pyright runs unpinned; the interpreter pin lives in the justfile. run: just typecheck docs: diff --git a/packages/zarr-metadata/justfile b/packages/zarr-metadata/justfile index b7734c885a..20b9e8fbeb 100644 --- a/packages/zarr-metadata/justfile +++ b/packages/zarr-metadata/justfile @@ -45,15 +45,12 @@ test-versions *args: lint: uvx ruff check . -# Pinned so a pyright release cannot turn CI red on its own schedule. Bump it -# deliberately: run the new version, read what it found, then move the pin. -pyright_version := "1.1.414" - # Run under the interpreter CI runs pyright under, so a version-dependent -# stdlib or stub difference shows up here rather than only in CI. +# stdlib or stub difference shows up here rather than only in CI. The latest +# pyright: the sentinel typing it once lacked is fixed, so nothing holds it back. # Type-check the package, sources and tests alike typecheck: - uv run --python 3.11 --group test --with 'pyright=={{ pyright_version }}' pyright + uv run --python 3.11 --group test --with pyright pyright # Run everything CI runs for this package, on both ends of the version range check: lint typecheck test-versions docs-check diff --git a/packages/zarr-metadata/pyproject.toml b/packages/zarr-metadata/pyproject.toml index 8d37031881..588455b660 100644 --- a/packages/zarr-metadata/pyproject.toml +++ b/packages/zarr-metadata/pyproject.toml @@ -139,8 +139,7 @@ checks = [ "PR06", ] -# The pyright version lives in the justfile, which is what CI runs; pinning it -# keeps a pyright release from turning CI red on its own schedule. +# Pyright runs unpinned, from the justfile, which is what CI runs. [tool.pyright] include = ["src", "tests"] # Pyright is the only checker that runs on this package (the root mypy From db5e3d1c822188646332ec5bdaa1a15fb4827db8 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Tue, 6 Oct 2026 16:05:20 +0200 Subject: [PATCH 49/94] docs(zarr-metadata): document model entries in consolidated metadata and context=None Assisted-by: ClaudeCode:claude-fable-5-1 --- packages/zarr-metadata/README.md | 6 +++++- .../changes/+consolidating-models.feature.1.md | 1 + .../zarr-metadata/changes/+consolidating-models.feature.md | 1 + packages/zarr-metadata/docs/index.md | 6 +++++- packages/zarr-metadata/src/zarr_metadata/model/__init__.py | 5 ++++- 5 files changed, 16 insertions(+), 3 deletions(-) create mode 100644 packages/zarr-metadata/changes/+consolidating-models.feature.1.md create mode 100644 packages/zarr-metadata/changes/+consolidating-models.feature.md diff --git a/packages/zarr-metadata/README.md b/packages/zarr-metadata/README.md index e22451b6dd..17edb0b2c1 100644 --- a/packages/zarr-metadata/README.md +++ b/packages/zarr-metadata/README.md @@ -177,7 +177,11 @@ the document in another scope, and `refined_in` only in one that claims what this one left unclaimed and contradicts nothing, raising `ScopeConflictError` otherwise. The documents a group's `consolidated_metadata` holds are models of the group's scope, built from -the group's one read. +the group's one read; it takes node models as entries too, each accepted +when its claims refine into the group's scope and refused at its path +otherwise, and `Context.joined` is the scope to consolidate children of +several scopes in. Every reader takes `context=None` for the default +scope, `CORE_AND_EXTENSIONS`. Two models are equal when they mean the same document, however each is spelled. What the package interprets -- each field, and the fill value diff --git a/packages/zarr-metadata/changes/+consolidating-models.feature.1.md b/packages/zarr-metadata/changes/+consolidating-models.feature.1.md new file mode 100644 index 0000000000..879ad81b41 --- /dev/null +++ b/packages/zarr-metadata/changes/+consolidating-models.feature.1.md @@ -0,0 +1 @@ +Every reader, validator and guard takes `context=None` for the default scope, `CORE_AND_EXTENSIONS`, as the models do. diff --git a/packages/zarr-metadata/changes/+consolidating-models.feature.md b/packages/zarr-metadata/changes/+consolidating-models.feature.md new file mode 100644 index 0000000000..99cea9e08d --- /dev/null +++ b/packages/zarr-metadata/changes/+consolidating-models.feature.md @@ -0,0 +1 @@ +A group's `consolidated_metadata` can be built from node models, in a constructor or `update`: a model is accepted when the group's scope reads it as its own scope did, or claims what that scope left unclaimed, and refused with a problem at its path when the two scopes read a name differently. `Context.joined` gives the scope to consolidate children read in different scopes in. diff --git a/packages/zarr-metadata/docs/index.md b/packages/zarr-metadata/docs/index.md index a15c673d0d..7ee50fb213 100644 --- a/packages/zarr-metadata/docs/index.md +++ b/packages/zarr-metadata/docs/index.md @@ -192,7 +192,11 @@ the document in another scope, and `refined_in` only in one that claims what this one left unclaimed and contradicts nothing, raising `ScopeConflictError` otherwise. The documents a group's `consolidated_metadata` holds are models of the group's scope, built from -the group's one read. +the group's one read; it takes node models as entries too, each accepted +when its claims refine into the group's scope and refused at its path +otherwise, and `Context.joined` is the scope to consolidate children of +several scopes in. Every reader takes `context=None` for the default +scope, `CORE_AND_EXTENSIONS`. Two models are equal when they mean the same document, however each is spelled. What the package interprets -- each field, and the fill value diff --git a/packages/zarr-metadata/src/zarr_metadata/model/__init__.py b/packages/zarr-metadata/src/zarr_metadata/model/__init__.py index 73ce61bf63..ed0226348a 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/__init__.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/__init__.py @@ -27,7 +27,10 @@ model is built invalid; `to_json` writes the document as written; `update` reads new members in the model's own scope; `with_context` and `refined_in` read the document in another; `to_key_value` writes a model -as it is. +as it is. A group's `consolidated_metadata` takes node models as entries, +each accepted when its claims refine into the group's scope and refused +at its path otherwise. Every reader takes `context=None` for the default +scope. """ from zarr_metadata._json import ( From bfe5c642728d5eb7794dd3f97dbfcca53e176da4 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Tue, 6 Oct 2026 16:12:17 +0200 Subject: [PATCH 50/94] fix(zarr-metadata): replace models nested inside listed documents; fix parse and conflict messages Review round 1 findings: - `documents_for` only replaced models at the top of a group's listing. A model inside a listed document crashed every constructor and reader while the validator accepted it. It now recurses into each listed document. - `parse_group_metadata_v3` returned model objects for a document holding models. It now returns JSON, which `is_group_metadata_v3` accepts. - The conflict message names the kind, distinguishes the two definitions by their configuration TypedDict, and says the group's scope "leaves unclaimed" for a loss instead of "reads by None". - A refused model entry's reading no longer holds models from another scope. - The input types are exported from the top-level package. Assisted-by: ClaudeCode:claude-fable-5-1 --- .../changes/+consolidating-models.feature.md | 2 +- .../src/zarr_metadata/__init__.py | 4 + .../src/zarr_metadata/model/__init__.py | 2 + .../src/zarr_metadata/model/_group.py | 54 +++++++++++-- .../tests/model/test_consolidating.py | 81 ++++++++++++++++--- .../zarr-metadata/tests/test_public_api.py | 2 + 6 files changed, 126 insertions(+), 19 deletions(-) diff --git a/packages/zarr-metadata/changes/+consolidating-models.feature.md b/packages/zarr-metadata/changes/+consolidating-models.feature.md index 99cea9e08d..d9abd7949f 100644 --- a/packages/zarr-metadata/changes/+consolidating-models.feature.md +++ b/packages/zarr-metadata/changes/+consolidating-models.feature.md @@ -1 +1 @@ -A group's `consolidated_metadata` can be built from node models, in a constructor or `update`: a model is accepted when the group's scope reads it as its own scope did, or claims what that scope left unclaimed, and refused with a problem at its path when the two scopes read a name differently. `Context.joined` gives the scope to consolidate children read in different scopes in. +A group's `consolidated_metadata` can be built from node models, in a constructor or `update`. A model is accepted when the group's scope reads it as its own scope did, or claims what that scope left unclaimed, and refused with a problem at its path when the two scopes read a name differently. diff --git a/packages/zarr-metadata/src/zarr_metadata/__init__.py b/packages/zarr-metadata/src/zarr_metadata/__init__.py index 5b7e154449..ba041a3b70 100644 --- a/packages/zarr-metadata/src/zarr_metadata/__init__.py +++ b/packages/zarr-metadata/src/zarr_metadata/__init__.py @@ -26,9 +26,11 @@ ZarrV3ArrayMetadataStoreKey, ZarrV3ArrayMetadataUpdate, ZarrV3ConsolidatedMetadata, + ZarrV3ConsolidatedMetadataInput, ZarrV3GroupMetadata, ZarrV3GroupMetadataStoreKey, ZarrV3GroupMetadataUpdate, + ZarrV3NodeMetadataInput, ) from zarr_metadata.v2.array import ( ZARR_V2_ARRAY_DIMENSION_SEPARATOR, @@ -383,6 +385,7 @@ "ZarrV3ArrayMetadataStoreKey", "ZarrV3ArrayMetadataUpdate", "ZarrV3ConsolidatedMetadata", + "ZarrV3ConsolidatedMetadataInput", "ZarrV3ConsolidatedMetadataJSON", "ZarrV3ExtensionField", "ZarrV3GroupMetadata", @@ -392,6 +395,7 @@ "ZarrV3GroupMetadataUpdate", "ZarrV3MetadataFieldJSON", "ZarrV3NamedConfigJSON", + "ZarrV3NodeMetadataInput", "ZstdCodecMetadata", "ZstdCodecName", "__version__", diff --git a/packages/zarr-metadata/src/zarr_metadata/model/__init__.py b/packages/zarr-metadata/src/zarr_metadata/model/__init__.py index ed0226348a..78b852dbbc 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/__init__.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/__init__.py @@ -59,6 +59,7 @@ ZarrV3GroupMetadataReading, ZarrV3GroupMetadataUpdate, ZarrV3NodeMetadata, + ZarrV3NodeMetadataInput, ZarrV3NodeMetadataReading, ZarrV3UnknownNodeReading, is_group_metadata_v3, @@ -187,6 +188,7 @@ "ZarrV3GroupMetadataStoreKey", "ZarrV3GroupMetadataUpdate", "ZarrV3NodeMetadata", + "ZarrV3NodeMetadataInput", "ZarrV3NodeMetadataReading", "ZarrV3NullConsolidatedGroupMetadataJSON", "ZarrV3RepairedNodeMetadataReading", diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_group.py b/packages/zarr-metadata/src/zarr_metadata/model/_group.py index 86f27615c6..e86f599387 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_group.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/_group.py @@ -73,7 +73,7 @@ from zarr_metadata.v2.attributes import ZarrV2AttributesStoreKey from zarr_metadata.v2.consolidated import ZarrV2ConsolidatedMetadataStoreKey from zarr_metadata.v2.group import ZarrV2GroupMetadataJSON, ZarrV2GroupMetadataStoreKey - from zarr_metadata.v3._definition import Resolved + from zarr_metadata.v3._definition import Definition, Resolved from zarr_metadata.v3.array import ZarrV3ArrayMetadataJSON from zarr_metadata.v3.consolidated import ZarrV3ConsolidatedMetadataJSON from zarr_metadata.v3.group import ZarrV3GroupMetadataJSONPartial, ZarrV3GroupMetadataStoreKey @@ -781,7 +781,9 @@ def _member_documents_for(member: object) -> object: replaced[path] = entry._document # pyright: ignore[reportPrivateUsage] changed = True else: - replaced[path] = entry + # A document listed here may list models of its own. + replaced[path] = documents_for(entry) + changed = changed or replaced[path] is not entry if not changed: return cast("object", member) return {**cast("Mapping[object, object]", member), "metadata": replaced} @@ -802,21 +804,57 @@ def _read_node_model( problems = tuple( ValidationProblem( conflict.loc if conflict.loc is not None else (), - f"expected a document read in the group's scope, got a model that reads " - f"{conflict.key[1]!r} by {conflict.claimed!r}, which the scope reads by " - f"{conflict.found!r}", + f"expected a document read in the group's scope, got a model that reads the " + f"{_kind_said(conflict.key[0])} {conflict.key[1]!r} by " + f"{_definition_said(conflict.claimed)}, which the group's scope " + f"{_reads_said(conflict.found)}", "invalid_value", ) for conflict in conflicts ) - refused = dataclasses.replace(entry.reading, problems=problems, metadata=None) - return refused, None + return _without_models(entry.reading, problems), None if found.agrees: moved = entry.with_context(context) return moved.reading, moved._members # pyright: ignore[reportPrivateUsage] return _read_node_v3(entry._document, context, at) # pyright: ignore[reportPrivateUsage] +def _kind_said(kind: type[Definition[Any]]) -> str: + """A kind of definition as a message names it: `CodecDefinition` is "codec".""" + return kind.__name__.removesuffix("Definition").replace("Type", " type").lower() + + +def _definition_said(definition: Definition[Any] | None) -> str: + """A definition as a message tells it from another of the same name: by the TypedDict its configuration is.""" + if definition is None: + return "no definition" + return f"{definition!r} of {definition.configuration.__qualname__}" + + +def _reads_said(definition: Definition[Any] | None) -> str: + """How the group's scope reads a name, as a message says it: by a definition, or by none.""" + if definition is None: + return "leaves unclaimed" + return f"reads by {_definition_said(definition)}" + + +def _without_models( + reading: ZarrV3NodeMetadataReading, problems: tuple[ValidationProblem, ...] +) -> ZarrV3NodeMetadataReading: + """`reading`, a model's, as the reading of an entry refused with `problems`: no model, in it or in any document it holds, since a reading of the group's scope holds models of that scope alone.""" + if isinstance(reading, ZarrV3GroupMetadataReading): + nested = MappingProxyType( + { + path: _without_models(held, held.problems) + for path, held in reading.consolidated.items() + } + ) + return dataclasses.replace(reading, consolidated=nested, problems=problems, metadata=None) + if isinstance(reading, ZarrV3ArrayMetadataReading): + return dataclasses.replace(reading, problems=problems, metadata=None) + return reading + + def _read_consolidated_v3( value: object, context: Context, at: tuple[str | int, ...] = () ) -> tuple[ @@ -1140,7 +1178,7 @@ def parse_group_metadata_v3( problems = validate_group_metadata_v3(value, context=scope) if len(problems) != 0: raise MetadataValidationError(problems) - return cast("ZarrV3GroupMetadataJSON", arrays_to_tuples(value)) + return cast("ZarrV3GroupMetadataJSON", arrays_to_tuples(documents_for(value))) class ZarrV2GroupMetadataPartial(TypedDict, total=False): diff --git a/packages/zarr-metadata/tests/model/test_consolidating.py b/packages/zarr-metadata/tests/model/test_consolidating.py index 593158d6fd..a241f7cd1b 100644 --- a/packages/zarr-metadata/tests/model/test_consolidating.py +++ b/packages/zarr-metadata/tests/model/test_consolidating.py @@ -12,6 +12,9 @@ ZarrV3ArrayMetadata, ZarrV3ConsolidatedMetadata, ZarrV3GroupMetadata, + ZarrV3GroupMetadataReading, + is_group_metadata_v3, + parse_group_metadata_v3, read_group_metadata_v3, validate_group_metadata_v3, ) @@ -52,28 +55,50 @@ def _group(scope: Context | None = None, **entries: object) -> ZarrV3GroupMetada {"a/x": ZarrV3ArrayMetadata(ARRAY, context=Context.of())}, True, ), + ( + CORE, + { + "zarr_format": 3, + "node_type": "group", + "consolidated_metadata": {**INLINE, "metadata": {"x": ZarrV3ArrayMetadata(ARRAY)}}, + }, + {"a/x": ARRAY}, + False, + ), ], - ids=["same-scope", "unused-definitions", "gain", "nested-models"], + ids=["same-scope", "unused-definitions", "gain", "nested-models", "model-in-a-document"], ) def test_a_model_is_accepted_as_a_consolidated_entry_when_it_refines_into_the_scope( scope: Context, - child: ZarrV3ArrayMetadata | ZarrV3GroupMetadata, - beside: dict[str, ZarrV3ArrayMetadata], + child: ZarrV3ArrayMetadata | ZarrV3GroupMetadata | dict[str, Any], + beside: dict[str, Any], gained: bool, ) -> None: - """A node model given where consolidated metadata lists a document is accepted when the group's scope reads every claim of it identically -- its reading is kept, not read again -- or claims what the model's scope left unclaimed, when its document is read again there; the group then holds it in its own scope, equal to the group built of the document. A listed group's own listing is listed flat beside it, as the convention asks.""" + """A node model given where consolidated metadata lists a document -- at the top, or inside a document listed there -- is accepted when the group's scope reads every claim of it identically -- its reading is kept, not read again -- or claims what the model's scope left unclaimed, when its document is read again there; the group then holds it in its own scope, equal to the group built of the documents, and writes plain JSON. A listed group's own listing is listed flat beside it, as the convention asks.""" group = _group(scope, a=child, **beside) held = group.consolidated_metadata assert isinstance(held, ZarrV3ConsolidatedMetadata) node = held.metadata["a"] assert node.context is group.context + if isinstance(child, dict): + child = ZarrV3GroupMetadata(_documents(child), context=scope) assert cast("Any", node).refines(child) assert (node == child) is not gained if not gained and isinstance(child, ZarrV3ArrayMetadata): assert isinstance(node, ZarrV3ArrayMetadata) assert node.reading.pipeline is child.reading.pipeline - written = {path: node.to_json() for path, node in beside.items()} + written = {path: _documents(entry) for path, entry in beside.items()} assert group == _group(scope, a=child.to_json(), **written) + assert is_group_metadata_v3(group.to_json()) + + +def _documents(value: object) -> object: + """`value` with every node model in it replaced by its document, however deep.""" + if isinstance(value, (ZarrV3ArrayMetadata, ZarrV3GroupMetadata)): + return value.to_json() + if isinstance(value, dict): + return {key: _documents(item) for key, item in cast("dict[str, object]", value).items()} + return value def test_documents_as_written_and_pickle() -> None: @@ -86,14 +111,18 @@ def test_documents_as_written_and_pickle() -> None: def test_error_a_model_that_conflicts_with_the_scope_is_refused_at_its_path() -> None: - """A model read with a private `bytes`, given to a group whose scope reads the core `bytes`, is a conflict: a problem at the entry's path, naming the field, not a silent re-read.""" + """A model read with a private `bytes`, given to a group whose scope reads the core `bytes`, is a conflict: a problem at the entry's path, naming the kind and the name, and telling the two definitions apart, not a silent re-read.""" child = ZarrV3ArrayMetadata(ARRAY, context=PRIVATE) with pytest.raises(MetadataValidationError) as raised: _group(CORE_AND_EXTENSIONS, a=child) (problem,) = raised.value.problems assert problem.loc == ("consolidated_metadata", "metadata", "a", "codecs", 0) assert problem.kind == "invalid_value" - assert "bytes" in problem.message + assert problem.message == ( + "expected a document read in the group's scope, got a model that reads the codec " + "'bytes' by CodecDefinition(name='bytes') of Empty, which the group's scope reads by " + "CodecDefinition(name='bytes') of BytesCodecConfiguration" + ) def test_error_a_model_that_would_lose_a_meaning_is_refused_at_its_path() -> None: @@ -101,9 +130,9 @@ def test_error_a_model_that_would_lose_a_meaning_is_refused_at_its_path() -> Non child = ZarrV3ArrayMetadata(WITH_ZSTD, context=CORE_AND_EXTENSIONS) with pytest.raises(MetadataValidationError) as raised: _group(CORE, a=child) - assert [problem.loc for problem in raised.value.problems] == [ - ("consolidated_metadata", "metadata", "a", "codecs", 1) - ] + (problem,) = raised.value.problems + assert problem.loc == ("consolidated_metadata", "metadata", "a", "codecs", 1) + assert problem.message.endswith("which the group's scope leaves unclaimed") def test_error_a_gain_that_surfaces_a_problem_is_reported_at_the_problem() -> None: @@ -173,3 +202,35 @@ def test_update_keeps_a_models_scope_apart_from_the_groups() -> None: held = updated.consolidated_metadata assert isinstance(held, ZarrV3ConsolidatedMetadata) assert held.metadata["a"].context == CORE + + +def test_parse_gives_json_for_a_document_holding_models() -> None: + """`parse_group_metadata_v3` of a document holding models gives JSON, each model as its document, which the guard then says yes to: the trio agree.""" + document = { + "zarr_format": 3, + "node_type": "group", + "consolidated_metadata": {**INLINE, "metadata": {"a": ZarrV3ArrayMetadata(ARRAY)}}, + } + parsed = parse_group_metadata_v3(document) + member = cast("Any", parsed["consolidated_metadata"]) + assert member["metadata"]["a"] == ZarrV3ArrayMetadata(ARRAY).to_json() + assert is_group_metadata_v3(parsed) + + +def test_a_refused_model_entrys_reading_holds_no_model_of_another_scope() -> None: + """A group model refused as an entry leaves, in its reading, no model of the scope it was read in: a reading of the group's scope holds models of that scope alone.""" + child = ZarrV3GroupMetadata( + _group(CORE_AND_EXTENSIONS, x=ZarrV3ArrayMetadata(WITH_ZSTD)).to_json(), + context=CORE_AND_EXTENSIONS, + ) + document = { + "zarr_format": 3, + "node_type": "group", + "consolidated_metadata": {**INLINE, "metadata": {"a": child, "a/x": WITH_ZSTD}}, + } + reading = read_group_metadata_v3(document, context=CORE) + nested = reading.consolidated["a"] + assert isinstance(nested, ZarrV3GroupMetadataReading) + assert nested.metadata is None + assert len(nested.problems) != 0 + assert nested.consolidated["x"].metadata is None diff --git a/packages/zarr-metadata/tests/test_public_api.py b/packages/zarr-metadata/tests/test_public_api.py index 5a884c8e7e..c59948cb0d 100644 --- a/packages/zarr-metadata/tests/test_public_api.py +++ b/packages/zarr-metadata/tests/test_public_api.py @@ -50,6 +50,8 @@ def _group_rank(s: str) -> int: "ZarrV2GroupMetadataPartial", "ZarrV3GroupMetadata", "ZarrV3GroupMetadataUpdate", + "ZarrV3ConsolidatedMetadataInput", + "ZarrV3NodeMetadataInput", "ZarrV2ConsolidatedMetadata", "ZarrV3ConsolidatedMetadata", "ValidationProblem", From f076ff3a5647ec0f94a67edc6561924da23ed33c Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Tue, 6 Oct 2026 16:21:47 +0200 Subject: [PATCH 51/94] fix(zarr-metadata): check depth for adopted model entries; re-read refused entries in the group's scope Review round 2 findings: - When a model's scope agreed with the group's, the model was adopted without checking its depth at the new position. A model that is valid on its own but nested near the cap crashed every constructor and reader as a listed document while the validator accepted it. Its document is now checked from where it is listed. - A model refused for a conflict kept its own scope's reading inside the group's reading, so `claims_of` on the group's fields found two scopes. Its document is now re-read in the group's scope, and the conflicts are added as problems. - The conflict message uses `kind_name` for the kind, the name as the document writes it, and says plainly when two definitions print alike. A loss says the group's scope leaves the name unclaimed. Assisted-by: ClaudeCode:claude-fable-5-1 --- .../src/zarr_metadata/model/_group.py | 76 ++++++++-------- .../src/zarr_metadata/v3/_scope.py | 10 ++- .../tests/model/test_consolidating.py | 86 ++++++++++++++++++- 3 files changed, 130 insertions(+), 42 deletions(-) diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_group.py b/packages/zarr-metadata/src/zarr_metadata/model/_group.py index e86f599387..a9b6cd207d 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_group.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/_group.py @@ -60,7 +60,7 @@ from zarr_metadata.v2.group import ZARR_V2_GROUP_METADATA_STORE_KEY from zarr_metadata.v3._hierarchy import NodeType, hierarchy_problems, path_faults, said from zarr_metadata.v3._registry import CORE_AND_EXTENSIONS, Context -from zarr_metadata.v3._scope import Claims, ScopeConflictError, claims_of +from zarr_metadata.v3._scope import Claims, Conflict, ScopeConflictError, claims_of, kind_name from zarr_metadata.v3.array import ZarrV3ExtensionField from zarr_metadata.v3.consolidated import ZARR_V3_CONSOLIDATED_METADATA_KEY from zarr_metadata.v3.group import ZARR_V3_GROUP_METADATA_STORE_KEY, ZarrV3GroupMetadataJSON @@ -91,8 +91,8 @@ class ZarrV3ConsolidatedMetadataInput(TypedDict, closed=True): A node model is accepted when the group's scope reads every claim of it identically, or claims what the model's scope left unclaimed -- it is then read again there -- and refused, with a problem at its path, - where the two scopes read a name differently, or the group's reads it - by none. + where the two scopes read a name differently, or the group's scope + leaves it unclaimed. """ kind: Literal["inline"] @@ -798,30 +798,51 @@ def _read_node_model( scope. A gain reads the document again there, so a problem a newly claimed definition finds is reported where it sits. """ + document = entry._document # pyright: ignore[reportPrivateUsage] found = context.disagreements(entry.claims) if len(found.conflicts) != 0: - conflicts = located_conflicts(entry.reading.fields(), found.conflicts) + # Read again in the group's scope, so the reading is of that scope, + # as every reading the group holds is; the conflicts are its + # problems, each where the field sits. + reading, _ = _read_node_v3(document, context, at) + written = {loc: field.name for loc, field in entry.reading.fields()} problems = tuple( ValidationProblem( conflict.loc if conflict.loc is not None else (), - f"expected a document read in the group's scope, got a model that reads the " - f"{_kind_said(conflict.key[0])} {conflict.key[1]!r} by " - f"{_definition_said(conflict.claimed)}, which the group's scope " - f"{_reads_said(conflict.found)}", + _conflict_said( + conflict, None if conflict.loc is None else written.get(conflict.loc) + ), "invalid_value", ) - for conflict in conflicts + for conflict in located_conflicts(entry.reading.fields(), found.conflicts) ) - return _without_models(entry.reading, problems), None + return _with_problems(reading, (*reading.problems, *problems)), None if found.agrees: + # The model's own reading, unless the document sits too deep + # where it is listed, which a read from there reports. + _, past = refine_user_data(document, at) + if len(past) != 0: + return _read_node_v3(document, context, at) moved = entry.with_context(context) return moved.reading, moved._members # pyright: ignore[reportPrivateUsage] - return _read_node_v3(entry._document, context, at) # pyright: ignore[reportPrivateUsage] - - -def _kind_said(kind: type[Definition[Any]]) -> str: - """A kind of definition as a message names it: `CodecDefinition` is "codec".""" - return kind.__name__.removesuffix("Definition").replace("Type", " type").lower() + return _read_node_v3(document, context, at) + + +def _conflict_said(conflict: Conflict, written: str | None) -> str: + """What a conflict between a model entry's scope and the group's says: the kind and the name as the document writes it, what read it, and how the group's scope reads it -- by another definition, told apart from the model's where the two print alike, or by none.""" + kind, filed = conflict.key + name = filed if written is None else written + claimed = _definition_said(conflict.claimed) + head = f"expected a document read in the group's scope, got a model that reads the {kind_name(kind)} {name!r}" + if conflict.found is None: + return f"{head} by {claimed}, which the group's scope leaves unclaimed" + found = _definition_said(conflict.found) + if claimed == found: + return ( + f"{head} by a definition of that name other than the one the group's scope " + f"reads it by, {found}" + ) + return f"{head} by {claimed}, which the group's scope reads by {found}" def _definition_said(definition: Definition[Any] | None) -> str: @@ -831,28 +852,13 @@ def _definition_said(definition: Definition[Any] | None) -> str: return f"{definition!r} of {definition.configuration.__qualname__}" -def _reads_said(definition: Definition[Any] | None) -> str: - """How the group's scope reads a name, as a message says it: by a definition, or by none.""" - if definition is None: - return "leaves unclaimed" - return f"reads by {_definition_said(definition)}" - - -def _without_models( +def _with_problems( reading: ZarrV3NodeMetadataReading, problems: tuple[ValidationProblem, ...] ) -> ZarrV3NodeMetadataReading: - """`reading`, a model's, as the reading of an entry refused with `problems`: no model, in it or in any document it holds, since a reading of the group's scope holds models of that scope alone.""" - if isinstance(reading, ZarrV3GroupMetadataReading): - nested = MappingProxyType( - { - path: _without_models(held, held.problems) - for path, held in reading.consolidated.items() - } - ) - return dataclasses.replace(reading, consolidated=nested, problems=problems, metadata=None) - if isinstance(reading, ZarrV3ArrayMetadataReading): + """`reading`, a read of a document in the group's scope, with `problems` as its problems and no model.""" + if isinstance(reading, (ZarrV3GroupMetadataReading, ZarrV3ArrayMetadataReading)): return dataclasses.replace(reading, problems=problems, metadata=None) - return reading + return dataclasses.replace(reading, problems=problems) def _read_consolidated_v3( diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/_scope.py b/packages/zarr-metadata/src/zarr_metadata/v3/_scope.py index d00a02ca9a..bbabefcbf8 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/_scope.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/_scope.py @@ -63,9 +63,12 @@ class Conflict: def __str__(self) -> str: kind, name = self.key where = "" if self.loc is None else f" at {self.loc!r}" - return ( - f"{_KIND_NAMES[kind]} {name!r}{where}: claimed {self.claimed!r}, found {self.found!r}" - ) + return f"{kind_name(kind)} {name!r}{where}: claimed {self.claimed!r}, found {self.found!r}" + + +def kind_name(kind: type[Definition[Any]]) -> str: + """A kind of definition as a message names it: `CodecDefinition` is "codec", `ChunkKeyEncodingDefinition` "chunk key encoding"; a kind the table does not know, by its class name.""" + return _KIND_NAMES.get(kind, kind.__name__) class ScopeConflictError(ValueError): @@ -184,5 +187,6 @@ def disagreements_of( "claim_key", "claims_of", "disagreements_of", + "kind_name", "refines", ] diff --git a/packages/zarr-metadata/tests/model/test_consolidating.py b/packages/zarr-metadata/tests/model/test_consolidating.py index a241f7cd1b..e83a7a875b 100644 --- a/packages/zarr-metadata/tests/model/test_consolidating.py +++ b/packages/zarr-metadata/tests/model/test_consolidating.py @@ -2,11 +2,13 @@ from __future__ import annotations +import dataclasses import pickle from typing import Any, cast import pytest +from zarr_metadata._json import JSON_DEPTH from zarr_metadata.model import ( MetadataValidationError, ZarrV3ArrayMetadata, @@ -19,7 +21,18 @@ validate_group_metadata_v3, ) from zarr_metadata.v3.codec.crc32c import Empty -from zarr_metadata.v3.definition import CORE, CORE_AND_EXTENSIONS, CodecDefinition, Context +from zarr_metadata.v3.codec.gzip import GZIP_CODEC +from zarr_metadata.v3.definition import ( + CORE, + CORE_AND_EXTENSIONS, + ChunkGridDefinition, + ChunkKeyEncodingDefinition, + CodecDefinition, + Context, + DataTypeDefinition, + StorageTransformerDefinition, + claims_of, +) ARRAY: dict[str, Any] = { **ZarrV3ArrayMetadata.create_default(shape=(4,)).to_json(), @@ -153,8 +166,13 @@ def test_readers_see_models_as_the_constructor_does() -> None: assert reading.problems == () assert isinstance(reading.consolidated["a"].metadata, ZarrV3ArrayMetadata) assert validate_group_metadata_v3(document) == () + + +def test_error_the_validator_reports_a_conflicting_model_as_the_constructor_does() -> None: + """`validate_group_metadata_v3` given a document holding a model the scope conflicts with reports the conflict at the entry's path, as the constructor refuses it.""" clashing = { - **document, + "zarr_format": 3, + "node_type": "group", "consolidated_metadata": { **INLINE, "metadata": {"a": ZarrV3ArrayMetadata(ARRAY, context=PRIVATE)}, @@ -217,8 +235,8 @@ def test_parse_gives_json_for_a_document_holding_models() -> None: assert is_group_metadata_v3(parsed) -def test_a_refused_model_entrys_reading_holds_no_model_of_another_scope() -> None: - """A group model refused as an entry leaves, in its reading, no model of the scope it was read in: a reading of the group's scope holds models of that scope alone.""" +def test_error_a_refused_model_entrys_reading_is_of_the_groups_scope() -> None: + """A group model refused as an entry is read again in the group's scope for its reading, so the group's reading holds no field, and no model, of another scope: `claims_of` its fields finds one scope.""" child = ZarrV3GroupMetadata( _group(CORE_AND_EXTENSIONS, x=ZarrV3ArrayMetadata(WITH_ZSTD)).to_json(), context=CORE_AND_EXTENSIONS, @@ -234,3 +252,63 @@ def test_a_refused_model_entrys_reading_holds_no_model_of_another_scope() -> Non assert nested.metadata is None assert len(nested.problems) != 0 assert nested.consolidated["x"].metadata is None + assert claims_of(reading.fields()) == claims_of( + read_group_metadata_v3(_documents(document), context=CORE).fields() + ) + for _, field in reading.fields(): + assert field.definition is None or field.definition in CORE.definitions() + + +def test_error_a_model_listed_too_deep_is_refused_where_it_sits() -> None: + """A model valid on its own, nested to the last level a reader walks, sits past the cap as a listed document: refused there, as its document would be, by the validator, the constructor and the reader alike.""" + deep: list[object] = [] + for _ in range(JSON_DEPTH - 3): + deep = [deep] + child = ZarrV3ArrayMetadata({**ARRAY, "attributes": {"x": deep}}) + document = { + "zarr_format": 3, + "node_type": "group", + "consolidated_metadata": {**INLINE, "metadata": {"a": child}}, + } + problems = validate_group_metadata_v3(document) + assert [problem.loc[:4] for problem in problems] == [ + ("consolidated_metadata", "metadata", "a", "attributes") + ] + with pytest.raises(MetadataValidationError): + ZarrV3GroupMetadata(document) + assert read_group_metadata_v3(document).metadata is None + + +@pytest.mark.parametrize( + ("kind", "said"), + [ + (CodecDefinition, "codec"), + (DataTypeDefinition, "data type"), + (ChunkGridDefinition, "chunk grid"), + (ChunkKeyEncodingDefinition, "chunk key encoding"), + (StorageTransformerDefinition, "storage transformer"), + ], + ids=["codec", "data-type", "chunk-grid", "chunk-key-encoding", "storage-transformer"], +) +def test_a_conflict_names_each_kind_as_the_spec_does(kind: type, said: str) -> None: + """A conflict's problem names the kind of the field in words: `ChunkKeyEncodingDefinition` is "chunk key encoding", as a scope conflict names it.""" + from zarr_metadata.v3._scope import kind_name + + assert kind_name(kind) == said + + +def test_error_a_conflict_between_definitions_alike_says_they_differ() -> None: + """Two definitions of one name that read the same TypedDict, differing in their rules, are told apart in the message by saying so, since nothing else shows it.""" + strict = dataclasses.replace(GZIP_CODEC, rules=lambda configuration, nested: iter(())) + child = ZarrV3ArrayMetadata( + {**ARRAY, "codecs": ("bytes", {"name": "gzip", "configuration": {"level": 1}})}, + context=CORE.extended_with(strict), + ) + with pytest.raises(MetadataValidationError) as raised: + _group(CORE, a=child) + (problem,) = raised.value.problems + assert problem.message == ( + "expected a document read in the group's scope, got a model that reads the codec " + "'gzip' by a definition of that name other than the one the group's scope reads it " + "by, CodecDefinition(name='gzip') of GzipCodecConfiguration" + ) From ddfeba7baf1ed98a2b60f330f5700d36a67455ee Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Tue, 6 Oct 2026 16:27:40 +0200 Subject: [PATCH 52/94] fix(zarr-metadata): make ZarrV3NodeMetadataInput a runtime type; report conflicts first Review round 3 findings: - `ZarrV3NodeMetadataInput` was a string at run time, so `get_type_hints` failed on a signature that used it. It is now a `TypeAliasType`. - A conflict problem is now listed before the problems the re-read finds in the same field, so the cause comes before its symptom. - Two definitions that print alike are described as "another definition than the one the group's scope reads it by", which also fits wildcard names such as `r*`. - The changelog fragment mentions that a model is refused when the group's scope leaves a claimed name unclaimed. Assisted-by: ClaudeCode:claude-fable-5-1 --- .../changes/+consolidating-models.feature.md | 2 +- .../src/zarr_metadata/model/_group.py | 15 ++-- .../tests/model/test_consolidating.py | 71 +++++++++++++------ packages/zarr-metadata/tests/v3/test_scope.py | 21 ++++++ 4 files changed, 77 insertions(+), 32 deletions(-) diff --git a/packages/zarr-metadata/changes/+consolidating-models.feature.md b/packages/zarr-metadata/changes/+consolidating-models.feature.md index d9abd7949f..483db8d485 100644 --- a/packages/zarr-metadata/changes/+consolidating-models.feature.md +++ b/packages/zarr-metadata/changes/+consolidating-models.feature.md @@ -1 +1 @@ -A group's `consolidated_metadata` can be built from node models, in a constructor or `update`. A model is accepted when the group's scope reads it as its own scope did, or claims what that scope left unclaimed, and refused with a problem at its path when the two scopes read a name differently. +A group's `consolidated_metadata` can be built from node models, in a constructor or `update`. A model is accepted when the group's scope reads it as its own scope did, or claims what that scope left unclaimed, and refused with a problem at its path when the two scopes read a name differently or the group's scope leaves it unclaimed. diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_group.py b/packages/zarr-metadata/src/zarr_metadata/model/_group.py index a9b6cd207d..add64ca80d 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_group.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/_group.py @@ -6,7 +6,7 @@ from collections.abc import Callable, Mapping from dataclasses import dataclass, field from types import MappingProxyType -from typing import TYPE_CHECKING, Any, Final, Literal, TypeAlias, TypeGuard, TypeVar, cast +from typing import TYPE_CHECKING, Any, Final, Literal, TypeGuard, TypeVar, cast from typing_extensions import TypeAliasType, TypedDict, Unpack @@ -79,8 +79,9 @@ from zarr_metadata.v3.group import ZarrV3GroupMetadataJSONPartial, ZarrV3GroupMetadataStoreKey -ZarrV3NodeMetadataInput: TypeAlias = ( - "ZarrV3ArrayMetadataJSON | ZarrV3GroupMetadataJSON | ZarrV3ArrayMetadata | ZarrV3GroupMetadata" +ZarrV3NodeMetadataInput = TypeAliasType( + "ZarrV3NodeMetadataInput", + "ZarrV3ArrayMetadataJSON | ZarrV3GroupMetadataJSON | ZarrV3ArrayMetadata | ZarrV3GroupMetadata", ) """What consolidated metadata lists at a path when given to a constructor or `update`: a document, or a model of it.""" @@ -816,7 +817,8 @@ def _read_node_model( ) for conflict in located_conflicts(entry.reading.fields(), found.conflicts) ) - return _with_problems(reading, (*reading.problems, *problems)), None + # The conflicts first: the cause, before what the re-read finds of it. + return _with_problems(reading, (*problems, *reading.problems)), None if found.agrees: # The model's own reading, unless the document sits too deep # where it is listed, which a read from there reports. @@ -838,10 +840,7 @@ def _conflict_said(conflict: Conflict, written: str | None) -> str: return f"{head} by {claimed}, which the group's scope leaves unclaimed" found = _definition_said(conflict.found) if claimed == found: - return ( - f"{head} by a definition of that name other than the one the group's scope " - f"reads it by, {found}" - ) + return f"{head} by another definition than the one the group's scope reads it by, {found}" return f"{head} by {claimed}, which the group's scope reads by {found}" diff --git a/packages/zarr-metadata/tests/model/test_consolidating.py b/packages/zarr-metadata/tests/model/test_consolidating.py index e83a7a875b..09b012ddb5 100644 --- a/packages/zarr-metadata/tests/model/test_consolidating.py +++ b/packages/zarr-metadata/tests/model/test_consolidating.py @@ -4,7 +4,7 @@ import dataclasses import pickle -from typing import Any, cast +from typing import Any, cast, get_type_hints import pytest @@ -13,8 +13,10 @@ MetadataValidationError, ZarrV3ArrayMetadata, ZarrV3ConsolidatedMetadata, + ZarrV3ConsolidatedMetadataInput, ZarrV3GroupMetadata, ZarrV3GroupMetadataReading, + ZarrV3NodeMetadataInput, is_group_metadata_v3, parse_group_metadata_v3, read_group_metadata_v3, @@ -22,15 +24,12 @@ ) from zarr_metadata.v3.codec.crc32c import Empty from zarr_metadata.v3.codec.gzip import GZIP_CODEC +from zarr_metadata.v3.data_type.raw import RAW_BYTES_DATA_TYPE from zarr_metadata.v3.definition import ( CORE, CORE_AND_EXTENSIONS, - ChunkGridDefinition, - ChunkKeyEncodingDefinition, CodecDefinition, Context, - DataTypeDefinition, - StorageTransformerDefinition, claims_of, ) @@ -279,22 +278,24 @@ def test_error_a_model_listed_too_deep_is_refused_where_it_sits() -> None: assert read_group_metadata_v3(document).metadata is None -@pytest.mark.parametrize( - ("kind", "said"), - [ - (CodecDefinition, "codec"), - (DataTypeDefinition, "data type"), - (ChunkGridDefinition, "chunk grid"), - (ChunkKeyEncodingDefinition, "chunk key encoding"), - (StorageTransformerDefinition, "storage transformer"), - ], - ids=["codec", "data-type", "chunk-grid", "chunk-key-encoding", "storage-transformer"], -) -def test_a_conflict_names_each_kind_as_the_spec_does(kind: type, said: str) -> None: - """A conflict's problem names the kind of the field in words: `ChunkKeyEncodingDefinition` is "chunk key encoding", as a scope conflict names it.""" - from zarr_metadata.v3._scope import kind_name - - assert kind_name(kind) == said +def test_error_a_data_type_conflict_names_the_kind_and_the_written_name() -> None: + """A conflict over a data type names the kind in words and the name as the document writes it -- `r16`, filed under `r*` -- and says the two definitions differ when they print alike.""" + strict = dataclasses.replace( + RAW_BYTES_DATA_TYPE, fill_value_rules=lambda configuration, nested, value: iter(()) + ) + child = ZarrV3ArrayMetadata( + {**ARRAY, "data_type": "r16", "fill_value": [0, 0], "codecs": ("bytes",)}, + context=CORE.extended_with(strict), + ) + with pytest.raises(MetadataValidationError) as raised: + _group(CORE, a=child) + (problem,) = raised.value.problems + assert problem.loc == ("consolidated_metadata", "metadata", "a", "data_type") + assert problem.message == ( + "expected a document read in the group's scope, got a model that reads the data type " + "'r16' by another definition than the one the group's scope reads it by, " + "DataTypeDefinition(name='r*') of RawBytesConfiguration" + ) def test_error_a_conflict_between_definitions_alike_says_they_differ() -> None: @@ -309,6 +310,30 @@ def test_error_a_conflict_between_definitions_alike_says_they_differ() -> None: (problem,) = raised.value.problems assert problem.message == ( "expected a document read in the group's scope, got a model that reads the codec " - "'gzip' by a definition of that name other than the one the group's scope reads it " - "by, CodecDefinition(name='gzip') of GzipCodecConfiguration" + "'gzip' by another definition than the one the group's scope reads it by, " + "CodecDefinition(name='gzip') of GzipCodecConfiguration" + ) + + +def test_the_input_types_are_types_a_signature_can_hold() -> None: + """`ZarrV3NodeMetadataInput` and `ZarrV3ConsolidatedMetadataInput` resolve as annotations at run time, as a caller's `get_type_hints` reads them: types, not strings.""" + + def take(entry: ZarrV3NodeMetadataInput, member: ZarrV3ConsolidatedMetadataInput) -> None: + pass + + hints = get_type_hints(take) + assert {"entry", "member"} <= set(hints) + assert not isinstance(hints["entry"], str) + + +def test_error_a_conflict_is_reported_before_what_the_re_read_finds() -> None: + """A model read with the core `bytes` and an `endian`, given to a group whose private `bytes` takes none, is refused for the conflict first, and for the key that `bytes` does not take after it: the cause before its symptom.""" + child = ZarrV3ArrayMetadata( + {**ARRAY, "codecs": ({"name": "bytes", "configuration": {"endian": "little"}},)} ) + with pytest.raises(MetadataValidationError) as raised: + _group(PRIVATE, a=child) + assert [problem.loc for problem in raised.value.problems] == [ + ("consolidated_metadata", "metadata", "a", "codecs", 0), + ("consolidated_metadata", "metadata", "a", "codecs", 0, "configuration", "endian"), + ] diff --git a/packages/zarr-metadata/tests/v3/test_scope.py b/packages/zarr-metadata/tests/v3/test_scope.py index 95f673b7d6..00fc877a85 100644 --- a/packages/zarr-metadata/tests/v3/test_scope.py +++ b/packages/zarr-metadata/tests/v3/test_scope.py @@ -9,6 +9,7 @@ from hypothesis import given from hypothesis import strategies as st +from zarr_metadata.v3._scope import kind_name from zarr_metadata.v3.codec.bytes import BYTES_CODEC from zarr_metadata.v3.codec.crc32c import CRC32C_CODEC, Empty from zarr_metadata.v3.codec.gzip import GZIP_CODEC @@ -19,6 +20,8 @@ from zarr_metadata.v3.definition import ( CORE, CORE_AND_EXTENSIONS, + ChunkGridDefinition, + ChunkKeyEncodingDefinition, Claims, CodecDefinition, Conflict, @@ -29,6 +32,7 @@ Refused, Resolved, ScopeConflictError, + StorageTransformerDefinition, claims_of, fields_of, refines, @@ -360,3 +364,20 @@ def read(claims: Claims) -> None: pass assert "claims" in get_type_hints(read) + + +@pytest.mark.parametrize( + ("kind", "said"), + [ + (CodecDefinition, "codec"), + (DataTypeDefinition, "data type"), + (ChunkGridDefinition, "chunk grid"), + (ChunkKeyEncodingDefinition, "chunk key encoding"), + (StorageTransformerDefinition, "storage transformer"), + ], + ids=["codec", "data-type", "chunk-grid", "chunk-key-encoding", "storage-transformer"], +) +def test_a_kind_is_named_in_words(kind: type[Definition[Any]], said: str) -> None: + """`kind_name` names each kind of definition as a message does: `ChunkKeyEncodingDefinition` is "chunk key encoding".""" + assert kind_name(kind) == said + assert said in str(Conflict((kind, "x"), None, None)) From c54522c0de64de46e0539fa876aa840e64c775d7 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Tue, 6 Oct 2026 16:27:41 +0200 Subject: [PATCH 53/94] docs(zarr-metadata): rename the consolidating-models fragments with their PR number Assisted-by: ClaudeCode:claude-fable-5-1 --- .../{+consolidating-models.feature.1.md => 395.feature.1.md} | 0 .../changes/{+consolidating-models.feature.md => 395.feature.md} | 0 2 files changed, 0 insertions(+), 0 deletions(-) rename packages/zarr-metadata/changes/{+consolidating-models.feature.1.md => 395.feature.1.md} (100%) rename packages/zarr-metadata/changes/{+consolidating-models.feature.md => 395.feature.md} (100%) diff --git a/packages/zarr-metadata/changes/+consolidating-models.feature.1.md b/packages/zarr-metadata/changes/395.feature.1.md similarity index 100% rename from packages/zarr-metadata/changes/+consolidating-models.feature.1.md rename to packages/zarr-metadata/changes/395.feature.1.md diff --git a/packages/zarr-metadata/changes/+consolidating-models.feature.md b/packages/zarr-metadata/changes/395.feature.md similarity index 100% rename from packages/zarr-metadata/changes/+consolidating-models.feature.md rename to packages/zarr-metadata/changes/395.feature.md From 6f1a3663bec3b44a67cbed4e1e8a259ea40d4d44 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Tue, 6 Oct 2026 18:25:57 +0200 Subject: [PATCH 54/94] docs(zarr-metadata): rewrite the 395 changelog fragments in plain English Assisted-by: ClaudeCode:claude-fable-5-1 --- packages/zarr-metadata/changes/395.feature.1.md | 2 +- packages/zarr-metadata/changes/395.feature.md | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/packages/zarr-metadata/changes/395.feature.1.md b/packages/zarr-metadata/changes/395.feature.1.md index 879ad81b41..f01fc9f35b 100644 --- a/packages/zarr-metadata/changes/395.feature.1.md +++ b/packages/zarr-metadata/changes/395.feature.1.md @@ -1 +1 @@ -Every reader, validator and guard takes `context=None` for the default scope, `CORE_AND_EXTENSIONS`, as the models do. +Every reader, validator and guard accepts `context=None` for the default scope, `CORE_AND_EXTENSIONS`, like the model constructors. diff --git a/packages/zarr-metadata/changes/395.feature.md b/packages/zarr-metadata/changes/395.feature.md index 483db8d485..9b7be745ee 100644 --- a/packages/zarr-metadata/changes/395.feature.md +++ b/packages/zarr-metadata/changes/395.feature.md @@ -1 +1 @@ -A group's `consolidated_metadata` can be built from node models, in a constructor or `update`. A model is accepted when the group's scope reads it as its own scope did, or claims what that scope left unclaimed, and refused with a problem at its path when the two scopes read a name differently or the group's scope leaves it unclaimed. +A group's `consolidated_metadata` can be built from node models, in a constructor or in `update`. A model is accepted when the group's scope reads it the same way its own scope did, or claims names its scope left unclaimed. It is refused, with a problem at its path, when the two scopes read a name with different definitions, or when the group's scope does not claim a name the model's scope did. From 7db638397946fbf43d0c7e21432d9d17c2ea31d0 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Tue, 6 Oct 2026 21:42:19 +0200 Subject: [PATCH 55/94] refactor(zarr-metadata): a kind is the class that declares is_kind A definition's kind is the nearest class in its MRO that sets is_kind in its own body, so a kind of another format needs no registry. Field aliases are filed by each kind as its class is built. kind_name reads the kind's label. Assisted-by: ClaudeCode:claude-fable-5-1 Co-Authored-By: Claude Fable 5.1 --- .../src/zarr_metadata/v3/_definition.py | 101 +++++++++++++----- .../src/zarr_metadata/v3/_registry.py | 11 +- .../src/zarr_metadata/v3/_scope.py | 17 +-- packages/zarr-metadata/tests/v3/test_kinds.py | 74 +++++++++++++ 4 files changed, 156 insertions(+), 47 deletions(-) create mode 100644 packages/zarr-metadata/tests/v3/test_kinds.py diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py b/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py index d4e98a2650..9c9fccdfa6 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py @@ -38,6 +38,7 @@ from typing import ( TYPE_CHECKING, Any, + ClassVar, Final, Generic, Literal, @@ -170,6 +171,10 @@ class EmptyConfiguration(TypedDict, closed=True): """The configuration of a definition with nothing to configure: its field is written with its name alone.""" +_FIELD_KINDS: Final[dict[object, type[Definition[Any]]]] = {} +"""Each field alias, and the kind a member annotated with it is read as; filed by each kind as its class is built.""" + + @dataclass(frozen=True, kw_only=True, slots=True) class Definition(Generic[C]): """One extension's metadata, as JSON: its name, the TypedDict its configuration is, its rules. @@ -203,6 +208,17 @@ class Definition(Generic[C]): and a `name` or rules that are not what they say. """ + is_kind: ClassVar[bool] = False + """Whether this class is a kind: what a scope files definitions by. + + Set in a kind's own body and read from it, never inherited: a + subclass of a kind is a definition of that kind. + """ + label: ClassVar[str] = "definition" + """The kind as a message names it: "codec".""" + field_aliases: ClassVar[tuple[object, ...]] = () + """The field aliases a configuration member holding a field of this kind is annotated with: `CodecField` and `StaticCodecField` for a codec.""" + name: str """The name the metadata carries, which a scope files the definition under.""" configuration: type[C] @@ -216,6 +232,15 @@ class Definition(Generic[C]): canonical form by `canonicalize`, which knows where each one sits. """ + def __init_subclass__(cls, **kwargs: object) -> None: + # Named, not `super()`: a dataclass with slots is rebuilt, and the + # cell a bare `super()` reads names the class that was thrown away. + super(Definition, cls).__init_subclass__(**kwargs) + # A dataclass with slots is built twice, and the class built last + # is the one a document is read with: it files its aliases last. + for alias in cls.__dict__.get("field_aliases", ()): + _FIELD_KINDS[alias] = cls + def __post_init__(self) -> None: refusal = _malformed(self) or self._refusal() if refusal is not None: @@ -381,6 +406,10 @@ class DataTypeDefinition(Definition[C]): this definition. """ + is_kind: ClassVar[bool] = True + label: ClassVar[str] = "data type" + field_aliases: ClassVar[tuple[object, ...]] = (DataTypeField,) + fill_value: object = JSONValue """The JSON shape of a fill value, as an annotation: `Int8FillValue`.""" fill_value_rules: Callable[[C, Nested, Any], Iterable[ValidationProblem]] = no_rules @@ -428,6 +457,10 @@ class ChunkGridDefinition(Definition[C]): every axis unknown. """ + is_kind: ClassVar[bool] = True + label: ClassVar[str] = "chunk grid" + field_aliases: ClassVar[tuple[object, ...]] = (ChunkGridField,) + shape_rules: Callable[[C, Nested, tuple[int, ...]], Iterable[ValidationProblem]] = no_rules """What the spec disallows in this grid over an array of a shape, located in the configuration.""" chunk_lengths: Callable[[C, Nested, tuple[int, ...]], Lengths] = unknown_lengths @@ -438,6 +471,10 @@ class ChunkGridDefinition(Definition[C]): class ChunkKeyEncodingDefinition(Definition[C]): """A chunk key encoding.""" + is_kind: ClassVar[bool] = True + label: ClassVar[str] = "chunk key encoding" + field_aliases: ClassVar[tuple[object, ...]] = (ChunkKeyEncodingField,) + CodecKind = Literal["array_array", "array_bytes", "bytes_bytes"] """What a codec does to what it is handed: the three positions a pipeline orders.""" @@ -489,6 +526,10 @@ class CodecDefinition(Definition[C]): codec, which is handed bytes -- is refused. """ + is_kind: ClassVar[bool] = True + label: ClassVar[str] = "codec" + field_aliases: ClassVar[tuple[object, ...]] = (CodecField, StaticCodecField) + kind: CodecKind size: CodecSize chunk_rules: Callable[[C, Nested, Chunk], Iterable[ValidationProblem]] = no_rules @@ -516,6 +557,10 @@ def _refusal(self) -> str | None: class StorageTransformerDefinition(Definition[C]): """A storage transformer.""" + is_kind: ClassVar[bool] = True + label: ClassVar[str] = "storage transformer" + field_aliases: ClassVar[tuple[object, ...]] = (StorageTransformerField,) + KINDS: Final[tuple[type[Definition[Any]], ...]] = ( DataTypeDefinition, @@ -528,35 +573,39 @@ class StorageTransformerDefinition(Definition[C]): def kind_of(definition: Definition[Any]) -> type[Definition[Any]] | None: - """The kind `definition` is; None for a definition of no kind, which no scope files.""" - return next((kind for kind in KINDS if isinstance(definition, kind)), None) + """The kind `definition` is: the nearest class in its MRO that declares `is_kind`; None for a definition of no kind, which no scope files.""" + return _kind_in(type(definition)) + + +def _kind_in(cls: type) -> type[Definition[Any]] | None: + return next( + ( + cast("type[Definition[Any]]", base) + for base in cls.__mro__ + if vars(base).get("is_kind") is True + ), + None, + ) def as_kind(kind: object) -> type[Definition[Any]]: """The kind of metadata `kind` names, type arguments dropped; `TypeError` if it names none. - A scope files definitions by kind, so a field is read as one of - `KINDS` -- `CodecDefinition`, or `CodecDefinition[Any]` -- and never - as the base `Definition` or a class of the caller's own, under which - nothing is filed: a field read as one would go unjudged. + A scope files definitions by kind, so a field is read as a kind -- + `CodecDefinition`, or `CodecDefinition[Any]` -- and never as the base + `Definition` or a class that declares no kind, under which nothing is + filed: a field read as one would go unjudged. """ origin = get_origin(kind) or kind - found = next((known for known in KINDS if origin is known), None) - if found is None: - names = ", ".join(known.__name__ for known in KINDS) - msg = f"{kind!r} is not a kind of metadata; read a field as one of {names}" - raise TypeError(msg) - return found - + if isinstance(origin, type) and vars(origin).get("is_kind") is True: + return cast("type[Definition[Any]]", origin) + names = ", ".join(known.__name__ for known in KINDS) + msg = ( + f"{kind!r} is not a kind of metadata; read a field as one of {names}, or as a " + "subclass of Definition that sets is_kind in its own body" + ) + raise TypeError(msg) -_FIELD_KINDS: Final[Mapping[object, type[Definition[Any]]]] = { - DataTypeField: DataTypeDefinition, - ChunkGridField: ChunkGridDefinition, - ChunkKeyEncodingField: ChunkKeyEncodingDefinition, - CodecField: CodecDefinition, - StaticCodecField: CodecDefinition, - StorageTransformerField: StorageTransformerDefinition, -} _STATIC_SIZE: Final[frozenset[object]] = frozenset({StaticCodecField}) """The field aliases whose codec must be of static size.""" @@ -663,10 +712,10 @@ def _vet(configuration: type) -> None: raise TypeError(msg) -_KIND_FIELDS: Final[Mapping[type[Definition[Any]], object]] = { - kind: alias for alias, kind in _FIELD_KINDS.items() if alias not in _STATIC_SIZE -} -"""The field alias of each kind: the one a member holding any field of the kind is annotated with.""" +def kind_field(kind: type[Definition[Any]]) -> object: + """The field alias a member holding any field of `kind` is annotated with: the first the kind declares that does not narrow the field, `CodecField` rather than `StaticCodecField`.""" + return next(alias for alias in kind.field_aliases if alias not in _STATIC_SIZE) + _RAW_BYTES_SCHEMA_PATTERN: Final = f"^{RAW_BYTES_NAME_PATTERN.pattern}(?![\\s\\S])" """`RAW_BYTES_NAME_PATTERN`, matched whole, as a JSON Schema writes a pattern. @@ -694,7 +743,7 @@ def field_json_schema(kind: type[Definition[Any]], context: Context) -> JSONSche so a field it accepts may still have a problem. """ schemas = Schemas(field_schemas(context)) - return schemas.document(schemas.of(_KIND_FIELDS[as_kind(kind)])) + return schemas.document(schemas.of(kind_field(as_kind(kind)))) def field_schemas(context: Context) -> SchemaLeaf: diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/_registry.py b/packages/zarr-metadata/src/zarr_metadata/v3/_registry.py index b2e134af89..16341ea6a6 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/_registry.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/_registry.py @@ -21,7 +21,7 @@ from types import MappingProxyType from typing import TYPE_CHECKING, Any, Final, cast -from zarr_metadata.v3._definition import KINDS, Definition, as_kind, kind_of, spelled +from zarr_metadata.v3._definition import Definition, as_kind, kind_of, spelled from zarr_metadata.v3._scope import Conflict, ScopeConflictError, disagreements_of from zarr_metadata.v3.chunk_grid.rectilinear import RECTILINEAR_CHUNK_GRID from zarr_metadata.v3.chunk_grid.regular import REGULAR_CHUNK_GRID @@ -87,19 +87,18 @@ def of(cls, *definitions: Definition[Any]) -> Context: `TypeError` for a definition of no kind, which no position in a document could hold. """ - tables: dict[type[Definition[Any]], dict[str, Definition[Any]]] = { - kind: {} for kind in KINDS - } + tables: dict[type[Definition[Any]], dict[str, Definition[Any]]] = {} for definition in definitions: kind = kind_of(definition) if kind is None: msg = ( f"{definition.name!r} is a definition of no kind; build it as a " "CodecDefinition, DataTypeDefinition, ChunkGridDefinition, " - "ChunkKeyEncodingDefinition or StorageTransformerDefinition" + "ChunkKeyEncodingDefinition or StorageTransformerDefinition, or as a " + "kind of your own" ) raise TypeError(msg) - tables[kind][definition.name] = definition + tables.setdefault(kind, {})[definition.name] = definition return cls( MappingProxyType({kind: MappingProxyType(table) for kind, table in tables.items()}) ) diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/_scope.py b/packages/zarr-metadata/src/zarr_metadata/v3/_scope.py index bbabefcbf8..1488cb0f51 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/_scope.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/_scope.py @@ -16,14 +16,9 @@ from typing import TYPE_CHECKING, Any, TypeAlias from zarr_metadata.v3._definition import ( - ChunkGridDefinition, - ChunkKeyEncodingDefinition, - CodecDefinition, - DataTypeDefinition, Definition, Read, Refused, - StorageTransformerDefinition, Unclaimed, field_key, own_key, @@ -42,14 +37,6 @@ Claims: TypeAlias = Mapping[ClaimKey, Definition[Any] | None] """What a reading claims of each name a document writes: the definition that read it, or None where nothing claimed it.""" -_KIND_NAMES: dict[type[Definition[Any]], str] = { - CodecDefinition: "codec", - DataTypeDefinition: "data type", - ChunkGridDefinition: "chunk grid", - ChunkKeyEncodingDefinition: "chunk key encoding", - StorageTransformerDefinition: "storage transformer", -} - @dataclass(frozen=True, slots=True) class Conflict: @@ -67,8 +54,8 @@ def __str__(self) -> str: def kind_name(kind: type[Definition[Any]]) -> str: - """A kind of definition as a message names it: `CodecDefinition` is "codec", `ChunkKeyEncodingDefinition` "chunk key encoding"; a kind the table does not know, by its class name.""" - return _KIND_NAMES.get(kind, kind.__name__) + """A kind of definition as a message names it: its label, "codec" for `CodecDefinition`.""" + return kind.label class ScopeConflictError(ValueError): diff --git a/packages/zarr-metadata/tests/v3/test_kinds.py b/packages/zarr-metadata/tests/v3/test_kinds.py new file mode 100644 index 0000000000..1d0c642595 --- /dev/null +++ b/packages/zarr-metadata/tests/v3/test_kinds.py @@ -0,0 +1,74 @@ +"""Kinds of definition: the class that declares `is_kind`, open to kinds of another format.""" + +from __future__ import annotations + +from dataclasses import dataclass +from typing import Any, ClassVar, TypeVar + +import pytest + +from zarr_metadata.v3._definition import as_kind, kind_of +from zarr_metadata.v3._scope import kind_name +from zarr_metadata.v3.codec.gzip import GZIP_CODEC +from zarr_metadata.v3.definition import ( + CodecDefinition, + Context, + Definition, + EmptyConfiguration, + Read, + resolve, +) + +C = TypeVar("C") + + +@dataclass(frozen=True, kw_only=True, slots=True, repr=False) +class MyCodec(CodecDefinition[Any]): + """A codec definition with a member of its own: still a codec.""" + + note: str = "" + + +@dataclass(frozen=True, kw_only=True, slots=True, repr=False) +class Tag(Definition[C]): + """A kind of its own, filed apart from every v3 kind.""" + + is_kind: ClassVar[bool] = True + label: ClassVar[str] = "tag" + + +@dataclass(frozen=True, kw_only=True, slots=True, repr=False) +class NoKind(Definition[Any]): + """A definition subclass that declares no kind.""" + + +def test_a_subclass_of_a_kind_is_a_definition_of_that_kind() -> None: + """A definition built as a subclass of `CodecDefinition` is filed, read and named as a codec: the kind is the nearest class in the MRO that declares `is_kind`, not the class of the definition.""" + mine = MyCodec(name="mine", configuration=EmptyConfiguration, kind="bytes_bytes", size="static") + assert kind_of(mine) is CodecDefinition + scope = Context.of(mine) + assert scope.claimant(CodecDefinition, "mine") is mine + assert isinstance(resolve({"name": "mine"}, CodecDefinition, scope)[0], Read) + + +def test_a_kind_of_its_own_is_filed_apart() -> None: + """A class that sets `is_kind` in its body is a kind: `as_kind` accepts it with or without type arguments, a scope files its definitions apart from every other kind's, scopes compare by what each files, and messages name the kind by its label.""" + tag = Tag(name="tag1", configuration=EmptyConfiguration) + scope = Context.of(tag, GZIP_CODEC) + assert as_kind(Tag) is Tag + assert as_kind(Tag[Any]) is Tag + assert scope.claimant(Tag, "tag1") is tag + assert scope.claimant(CodecDefinition, "tag1") is None + assert scope == Context.of(GZIP_CODEC, tag) + assert kind_name(Tag) == "tag" + assert kind_name(CodecDefinition) == "codec" + + +def test_error_a_class_that_declares_no_kind_is_of_none() -> None: + """A `Definition` subclass that does not set `is_kind` is of no kind: `kind_of` is None, `Context.of` refuses a definition of it, and `as_kind` refuses the class.""" + none = NoKind(name="nokind", configuration=EmptyConfiguration) + assert kind_of(none) is None + with pytest.raises(TypeError, match="a definition of no kind"): + Context.of(none) + with pytest.raises(TypeError, match="is not a kind of metadata"): + as_kind(NoKind) From 13b1ffb3fbc5bbc34d58c53c24c97ae2b07f2e73 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Tue, 6 Oct 2026 21:46:13 +0200 Subject: [PATCH 56/94] refactor(zarr-metadata): let each kind say how its format writes a field Definition gains classmethods for the envelope: name_problem, spelled, named_configuration, envelope_problems, envelope_json, configuration_loc and name_loc, and the instance method carrying_name. The v3 kinds keep today's behavior as the defaults and the data type's overrides. resolve, document_json and the canonical spelling go through them. Assisted-by: ClaudeCode:claude-fable-5-1 Co-Authored-By: Claude Fable 5.1 --- .../src/zarr_metadata/v3/_definition.py | 158 ++++++++++++------ .../tests/v3/test_definitions.py | 4 +- packages/zarr-metadata/tests/v3/test_kinds.py | 83 ++++++++- 3 files changed, 192 insertions(+), 53 deletions(-) diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py b/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py index 9c9fccdfa6..fe9653f4d3 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py @@ -232,6 +232,60 @@ class Definition(Generic[C]): canonical form by `canonicalize`, which knows where each one sits. """ + @classmethod + def name_problem(cls, name: str, at: Loc) -> ValidationProblem | None: + """The problem `name`, at `at`, is when no document of the format writes it for a field of this kind; None when one may: for Zarr v3, when the spec gives an extension such a name.""" + return name_problem(name, at) + + @classmethod + def well_named(cls, name: str) -> bool: + """Whether a document of the format may write `name` for a field of this kind, as `name_problem` says.""" + return cls.name_problem(name, ()) is None + + @classmethod + def spelled(cls, name: str) -> tuple[str | None, dict[str, JSONValue] | None]: + """How a name a document writes reads: the name its definition is filed under, and the configuration the name carries. + + A name is filed as itself and carries nothing, `(name, None)`. A + kind whose names carry configuration says otherwise: a v3 data + type `r16` is filed under `r*` with `{"bits": 16}`. A name no + document writes, which only files a definition, is `(None, None)`. + """ + return name, None + + def carrying_name(self, configuration: Mapping[str, JSONValue]) -> str | None: + """The name that carries `configuration` for this definition, the inverse of `spelled`: `r16` for `r*` with `{"bits": 16}`; None when its names carry nothing.""" + return None + + @classmethod + def named_configuration( + cls, value: object + ) -> tuple[str | None, Mapping[str, object] | None, Problems]: + """`value`, a field as a document of the format writes it, split into `(name, configuration, problems)`, as the module's `named_configuration` splits a v3 field.""" + return named_configuration(value) + + @classmethod + def envelope_problems(cls, value: object) -> Problems: + """Every reason `value` is not a field's envelope as the format writes one for this kind, what the configuration holds left unjudged.""" + return envelope_problems(value, allow_must_understand_false=False) + + @classmethod + def envelope_json(cls, name: str, configuration: Mapping[str, JSONValue]) -> JSONValue: + """A field of `name` and `configuration` as a document of the format writes it, in the fewest words every reader takes: for v3, an object.""" + if len(configuration) != 0: + return {"name": name, "configuration": configuration} + return {"name": name} + + @classmethod + def configuration_loc(cls, loc: Loc) -> Loc: + """Where the configuration of a field at `loc` sits: under `configuration` for v3; at the field for a format that writes the parameters beside the name.""" + return (*loc, "configuration") + + @classmethod + def name_loc(cls, loc: Loc) -> Loc: + """Where the name of a field at `loc`, written as an object, sits: under `name` for v3.""" + return (*loc, "name") + def __init_subclass__(cls, **kwargs: object) -> None: # Named, not `super()`: a dataclass with slots is rebuilt, and the # cell a bare `super()` reads names the class that was thrown away. @@ -305,11 +359,13 @@ def _malformed(definition: Definition[Any]) -> str | None: value = getattr(definition, member) if not callable(value): return f"{name!r}: {member} is a function, got {value!r}" - if definition.name != RAW_BYTES_NAME and not well_named(definition.name): + kind = type(definition) + filed, _ = kind.spelled(name) + bad = None if filed is None else kind.name_problem(name, ()) + if bad is not None: return ( - f"{definition.name!r} is not a name the spec gives an extension -- lower-case " - "letters, digits, '-', '_' and '.', starting with a letter, or a URI -- so no " - "document names it, and nothing would ever read with this definition" + f"{name!r}: {bad.message}, so no document names it, and nothing would ever read " + "with this definition" ) return None @@ -355,22 +411,7 @@ def spelled( itself is filed under nothing, `(None, None)`, since no document writes it. """ - if kind is DataTypeDefinition: - if name == RAW_BYTES_NAME: - return None, None - written = RAW_BYTES_NAME_PATTERN.fullmatch(name) - if written is not None: - return RAW_BYTES_NAME, {"bits": int(written.group(1))} - return name, None - - -def _carrying_name( - definition: Definition[Any], configuration: Mapping[str, JSONValue] -) -> str | None: - """The name that carries `configuration` for `definition`: `r16` for `r*` with `{"bits": 16}`; None when its name carries nothing.""" - if not isinstance(definition, DataTypeDefinition) or definition.name != RAW_BYTES_NAME: - return None - return f"r{configuration['bits']}" + return kind.spelled(name) @dataclass(frozen=True, kw_only=True, slots=True, repr=False) @@ -410,6 +451,28 @@ class DataTypeDefinition(Definition[C]): label: ClassVar[str] = "data type" field_aliases: ClassVar[tuple[object, ...]] = (DataTypeField,) + @classmethod + def spelled(cls, name: str) -> tuple[str | None, dict[str, JSONValue] | None]: + if name == RAW_BYTES_NAME: + return None, None + written = RAW_BYTES_NAME_PATTERN.fullmatch(name) + if written is not None: + return RAW_BYTES_NAME, {"bits": int(written.group(1))} + return name, None + + def carrying_name(self, configuration: Mapping[str, JSONValue]) -> str | None: + if self.name != RAW_BYTES_NAME: + return None + return f"r{configuration['bits']}" + + @classmethod + def envelope_json(cls, name: str, configuration: Mapping[str, JSONValue]) -> JSONValue: + # A data type with nothing to configure is its bare name, as core + # data types have been written since Zarr v3.0. + if len(configuration) != 0: + return {"name": name, "configuration": configuration} + return name + fill_value: object = JSONValue """The JSON shape of a fill value, as an annotation: `Int8FillValue`.""" fill_value_rules: Callable[[C, Nested, Any], Iterable[ValidationProblem]] = no_rules @@ -649,9 +712,8 @@ def _field(annotation: object) -> Parser | None: static = annotation in _STATIC_SIZE def parse(value: object, loc: Loc) -> Parsed: - if not isinstance(value, (str, Mapping)): - return value, problem(loc, f"expected a metadata field, got {shown(value)}") - # Refined JSON, which the checker only knows as `object`. + # Refined JSON, which the checker only knows as `object`; what is + # not a field at all, the kind's envelope says. return _NestedField(loc, kind, cast("JSONValue", value), static), () return parse @@ -959,7 +1021,7 @@ def _located(prefix: Loc, problems: Iterable[ValidationProblem]) -> Problems: def _envelope(field: _NestedField) -> Problems: """What is wrong with a nested field's envelope, at the field.""" - return _located(field.loc, envelope_problems(field.json, allow_must_understand_false=False)) + return _located(field.loc, field.kind.envelope_problems(field.json)) def _configuration_checked( @@ -1127,10 +1189,13 @@ def __post_init__(self) -> None: # arguments dropped, as `resolve` drops them. object.__setattr__(self, "read_as", as_kind(self.read_as)) name = cast("object", self.name) - if not isinstance(name, str) or not well_named(name): - msg = f"a field nothing in scope claims is named as the spec names an extension, got {name!r}" + if not isinstance(name, str) or not self.read_as.well_named(name): + msg = ( + "a field nothing in scope claims is named as a document names a " + f"{self.read_as.label}, got {name!r}" + ) raise TypeError(msg) - _, written, _ = named_configuration(self.json) + _, written, _ = self.read_as.named_configuration(self.json) configuration: Mapping[str, object] = {} if written is None else written object.__setattr__(self, "configuration", cast("Mapping[str, JSONValue]", configuration)) @@ -1259,18 +1324,9 @@ def document_json(field: Resolved[Any]) -> JSONValue | UNSET: if isinstance(field, Refused): return field.json kind = field.read_as - if isinstance(field, Read) and spelled(kind, field.name)[1] is not None: - return _envelope_json(kind, field.name, {}) - return _envelope_json(kind, field.name, field.configuration) - - -def _envelope_json( - kind: type[Definition[Any]], name: str, configuration: Mapping[str, JSONValue] -) -> JSONValue: - """A field's envelope in the fewest words every reader takes: a data type with nothing to configure by its bare name, any other field an object.""" - if len(configuration) != 0: - return {"name": name, "configuration": configuration} - return name if kind is DataTypeDefinition else {"name": name} + if isinstance(field, Read) and kind.spelled(field.name)[1] is not None: + return kind.envelope_json(field.name, {}) + return kind.envelope_json(field.name, field.configuration) Nested: TypeAlias = Mapping[Loc, Resolved[Any]] @@ -1574,8 +1630,8 @@ def resolve( # Not JSON, so not read; its name, if it has one, still says what # claims it, and one the spec does not give an extension is a # problem here as on the other path, asked of no definition. - name = named_configuration(data)[0] - bad = None if name is None else name_problem(name, (*loc, "name")) + name = asked.named_configuration(data)[0] + bad = None if name is None else asked.name_problem(name, asked.name_loc(loc)) claimant = None if name is None or bad is not None else context.claimant(asked, name) refused = Refused(json=UNSET, name=name, read_as=asked, definition=claimant) found = problems if bad is None else (bad, *problems) @@ -1594,7 +1650,7 @@ def _resolve_field( so it is reported beside the field that was read, which later layers can still judge. """ - envelope = _located(loc, envelope_problems(data, allow_must_understand_false=False)) + envelope = _located(loc, kind.envelope_problems(data)) resolved, found = _read(data, kind, context, loc) return resolved, (*envelope, *found) @@ -1602,10 +1658,10 @@ def _resolve_field( def _read( data: JSONValue, kind: type[Definition[Any]], context: Context, loc: Loc ) -> tuple[Resolved[Definition[Any]], Problems]: - name, given, malformed = named_configuration(data) + name, given, malformed = kind.named_configuration(data) if name is None: return Refused(json=data, name=None, read_as=kind), () - if not well_named(name): + if not kind.well_named(name): # The envelope rule every reader runs first reports it; no # definition is asked to claim it. return Refused(json=data, name=name, read_as=kind), () @@ -1616,10 +1672,10 @@ def _read( return Refused(json=data, name=name, read_as=kind, definition=definition), () if definition is None: return Unclaimed(json=data, name=name, read_as=kind), () - _, carried = spelled(kind, name) + _, carried = kind.spelled(name) if carried is not None: return _read_carried(data, name, definition, given, carried, loc) - at = (*loc, "configuration") + at = kind.configuration_loc(loc) if given is None and definition.requires_configuration: missing = problem(at, f"{name!r} requires a configuration", "missing_key") return Refused(json=data, name=name, read_as=kind, definition=definition), missing @@ -1666,7 +1722,7 @@ def _read( def _named(field: _NestedField) -> bool: """Whether a field a configuration holds is named, with an object for its configuration if it has one.""" - name, _, malformed = named_configuration(field.json) + name, _, malformed = field.kind.named_configuration(field.json) return name is not None and len(malformed) == 0 @@ -1686,7 +1742,9 @@ def _read_carried( member of one is a key nothing declares. """ _, beside, _ = _checked( - EmptyConfiguration, {} if given is None else given, (*loc, "configuration") + EmptyConfiguration, + {} if given is None else given, + type(definition).configuration_loc(loc), ) configuration, judged = definition.judge(carried) # What is wrong with what the name carries is the field's: found at @@ -1800,10 +1858,10 @@ def _canonical_field(resolved: Read[Any]) -> JSONValue | None: f"{list(refused)!r}" ) raise ValueError(msg) - carrying = _carrying_name(definition, simplified) + carrying = definition.carrying_name(simplified) if carrying is not None: return carrying - return _envelope_json(resolved.read_as, name, simplified) + return resolved.read_as.envelope_json(name, simplified) def _replaced(value: JSONValue, path: Loc, new: JSONValue) -> JSONValue: diff --git a/packages/zarr-metadata/tests/v3/test_definitions.py b/packages/zarr-metadata/tests/v3/test_definitions.py index 35c33517d6..d961a81ce6 100644 --- a/packages/zarr-metadata/tests/v3/test_definitions.py +++ b/packages/zarr-metadata/tests/v3/test_definitions.py @@ -1337,14 +1337,14 @@ def test_error_a_bad_name_is_a_problem_when_the_field_is_not_json_too() -> None: @pytest.mark.parametrize("name", ["Acme", "acme/x", "", "x"]) def test_error_a_definition_is_named_as_the_spec_names_an_extension(name: str) -> None: # No document could name it, so nothing would ever read with it. - with pytest.raises(TypeError, match="is not a name the spec gives an extension"): + with pytest.raises(TypeError, match="so no document names it"): CodecDefinition( name=name, configuration=EmptyConfiguration, kind="bytes_bytes", size="dynamic" ) def test_error_a_field_built_by_hand_is_named_as_the_spec_names_one() -> None: - with pytest.raises(TypeError, match="named as the spec names an extension"): + with pytest.raises(TypeError, match="named as a document names a"): Unclaimed(json="Int8", name="Int8", read_as=CodecDefinition) diff --git a/packages/zarr-metadata/tests/v3/test_kinds.py b/packages/zarr-metadata/tests/v3/test_kinds.py index 1d0c642595..e44f0906bb 100644 --- a/packages/zarr-metadata/tests/v3/test_kinds.py +++ b/packages/zarr-metadata/tests/v3/test_kinds.py @@ -2,11 +2,15 @@ from __future__ import annotations +from collections.abc import Mapping from dataclasses import dataclass -from typing import Any, ClassVar, TypeVar +from typing import TYPE_CHECKING, Annotated, Any, ClassVar, TypeVar, cast import pytest +from annotated_types import Ge +from typing_extensions import TypedDict +from zarr_metadata._json import ValidationProblem from zarr_metadata.v3._definition import as_kind, kind_of from zarr_metadata.v3._scope import kind_name from zarr_metadata.v3.codec.gzip import GZIP_CODEC @@ -16,9 +20,14 @@ Definition, EmptyConfiguration, Read, + Refused, + Unclaimed, resolve, ) +if TYPE_CHECKING: + from zarr_metadata._common import JSONValue + C = TypeVar("C") @@ -72,3 +81,75 @@ def test_error_a_class_that_declares_no_kind_is_of_none() -> None: Context.of(none) with pytest.raises(TypeError, match="is not a kind of metadata"): as_kind(NoKind) + + +class Params(TypedDict, closed=True): + level: Annotated[int, Ge(0)] + + +Loc = tuple[str | int, ...] + + +@dataclass(frozen=True, kw_only=True, slots=True, repr=False) +class Flat(Definition[C]): + """A kind whose format writes the parameters beside the name: `{"id": name, **parameters}`.""" + + is_kind: ClassVar[bool] = True + label: ClassVar[str] = "flat" + + @classmethod + def named_configuration( + cls, value: object + ) -> tuple[str | None, Mapping[str, object] | None, tuple[ValidationProblem, ...]]: + if not isinstance(value, Mapping): + return None, None, () + entry = cast("Mapping[str, object]", value) + name = entry.get("id") + if not isinstance(name, str): + return None, None, () + return name, {key: item for key, item in entry.items() if key != "id"}, () + + @classmethod + def envelope_problems(cls, value: object) -> tuple[ValidationProblem, ...]: + if cls.named_configuration(value)[0] is None: + return (ValidationProblem((), "expected an object with a string 'id'", "invalid_type"),) + return () + + @classmethod + def envelope_json(cls, name: str, configuration: Mapping[str, JSONValue]) -> JSONValue: + return {"id": name, **configuration} + + @classmethod + def configuration_loc(cls, loc: Loc) -> Loc: + return loc + + @classmethod + def name_loc(cls, loc: Loc) -> Loc: + return (*loc, "id") + + +FLAT = Flat(name="flat", configuration=Params) +FLAT_SCOPE = Context.of(FLAT) + + +@pytest.mark.parametrize( + ("field", "kind", "problems", "written"), + [ + ({"id": "flat", "level": 1}, Read, [], {"id": "flat", "level": 1}), + ({"id": "other", "x": 1}, Unclaimed, [], {"id": "other", "x": 1}), + ({"id": "flat", "level": -1}, Refused, [(("c", "level"), "invalid_value")], None), + ({"id": "flat", "payload": object()}, Refused, [(("c", "payload"), "invalid_type")], None), + ("flat", Refused, [(("c",), "invalid_type")], None), + ], + ids=["read", "unclaimed", "out-of-range", "not-json", "not-an-object"], +) +def test_a_kind_reads_the_envelope_its_format_writes( + field: object, kind: type, problems: list[tuple[Loc, str]], written: object +) -> None: + """`resolve` reads a field as its kind's classmethods say the format writes one: the name and parameters are split as the kind splits them, problems sit where the kind puts the configuration, and a field read or unclaimed is written back in the kind's envelope.""" + resolved, found = resolve(field, Flat, FLAT_SCOPE, ("c",)) + assert type(resolved) is kind + assert [(problem.loc, problem.kind) for problem in found] == problems + if written is not None: + assert not isinstance(resolved, Refused) + assert resolved.to_json() == written From 0359b4029e76dcca7f11b1275fe6aa3ac3cb2d60 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Tue, 6 Oct 2026 21:49:08 +0200 Subject: [PATCH 57/94] refactor(zarr-metadata): move the fill value members to a WithFillValue base DataTypeDefinition derives from WithFillValue, which holds fill_value, fill_value_rules and fill_value_canonical. fill_value_problems and canonical_fill_value accept any definition of a kind that derives from it, so a v2 data type kind can be judged the same way. Assisted-by: ClaudeCode:claude-fable-5-1 Co-Authored-By: Claude Fable 5.1 --- .../src/zarr_metadata/v3/_definition.py | 60 ++++++++++++------- .../src/zarr_metadata/v3/definition.py | 2 + packages/zarr-metadata/tests/v3/test_kinds.py | 29 ++++++++- 3 files changed, 69 insertions(+), 22 deletions(-) diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py b/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py index fe9653f4d3..d1d9d026bb 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py @@ -104,6 +104,7 @@ T = TypeVar("T") D = TypeVar("D", bound="Definition[Any]") +F = TypeVar("F", bound="WithFillValue[Any]") Problems: TypeAlias = tuple[ValidationProblem, ...] @@ -415,7 +416,39 @@ def spelled( @dataclass(frozen=True, kw_only=True, slots=True, repr=False) -class DataTypeDefinition(Definition[C]): +class WithFillValue(Definition[C]): + """A definition of a type whose arrays take a fill value: the JSON shape of one, what the rules disallow in one, and its canonical spelling. + + Not a kind: each format's data type kind derives from it, so + `fill_value_problems` judges a fill value of either. + + `fill_value` is the JSON shape of a fill value, as an annotation the + checker reads as it reads a configuration's members, its range among + it. `fill_value_rules` is what the spec disallows in a fill value of + that shape that the type cannot say; it is handed the configuration, + the fields it holds as the scope read them, and the typed fill value. + `fill_value_canonical` spells a fill value that has no problem in the + one spelling its value has, so two fill values are one value of the + type exactly when their canonical spellings are written alike. + """ + + fill_value: object = JSONValue + """The JSON shape of a fill value, as an annotation: `Int8FillValue`.""" + fill_value_rules: Callable[[C, Nested, Any], Iterable[ValidationProblem]] = no_rules + """What the spec disallows in a fill value of that shape, located in it.""" + fill_value_canonical: Callable[[C, Nested, Any], JSONValue] = fill_value_as_written + """A fill value that has no problem, in the one spelling its value has.""" + + def _refusal(self) -> str | None: + try: + _fill_value_parser(self.fill_value) + except TypeError as error: + return f"{self.name!r}: fill_value: {error}" + return None + + +@dataclass(frozen=True, kw_only=True, slots=True, repr=False) +class DataTypeDefinition(WithFillValue[C]): """A data type, and the fill value an array of it takes. `fill_value` is the JSON shape of a fill value -- `Int8FillValue`, an @@ -473,12 +506,6 @@ def envelope_json(cls, name: str, configuration: Mapping[str, JSONValue]) -> JSO return {"name": name, "configuration": configuration} return name - fill_value: object = JSONValue - """The JSON shape of a fill value, as an annotation: `Int8FillValue`.""" - fill_value_rules: Callable[[C, Nested, Any], Iterable[ValidationProblem]] = no_rules - """What the spec disallows in a fill value of that shape, located in it.""" - fill_value_canonical: Callable[[C, Nested, Any], JSONValue] = fill_value_as_written - """A fill value that has no problem, in the one spelling its value has.""" storage: Callable[[C, Nested], StorageClass | None] = unknown_storage """How its values are stored, given the configuration and the fields it holds; None when unknown.""" @@ -488,11 +515,8 @@ def _refusal(self) -> str | None: f"{self.name!r} is how a document writes raw bits of one size, which read as " f"{RAW_BYTES_NAME!r}; to read raw bits your own way, define {RAW_BYTES_NAME!r}" ) - try: - _fill_value_parser(self.fill_value) - except TypeError as error: - return f"{self.name!r}: fill_value: {error}" - return None + # Named, not a bare `super()`: a dataclass with slots is rebuilt. + return super(DataTypeDefinition, self)._refusal() @functools.cache @@ -1445,9 +1469,7 @@ def configuration_of(resolved: Resolved[Any], definition: Definition[C]) -> C | return cast("C", resolved.configuration) -def fill_value_problems( - data_type: Resolved[DataTypeDefinition[Any]], value: object, loc: Loc = () -) -> Problems: +def fill_value_problems(data_type: Resolved[F], value: object, loc: Loc = ()) -> Problems: """What is wrong with `value` as a fill value of `data_type`, a data type field a scope read. `value` is refined to JSON first: not JSON is the first verdict, @@ -1475,9 +1497,7 @@ def fill_value_problems( return with_input((*problems, *refused), value, loc) -def canonical_fill_value( - data_type: Resolved[DataTypeDefinition[Any]], value: object -) -> JSONValue | UNSET: +def canonical_fill_value(data_type: Resolved[F], value: object) -> JSONValue | UNSET: """`value`, a fill value of `data_type`, a data type field a scope read, in the one spelling its value has; `UNSET` when it has a problem. As the data type's `fill_value_canonical` spells it, so two fill @@ -1495,9 +1515,7 @@ def canonical_fill_value( return spelled_canonically(data_type, refined) -def spelled_canonically( - data_type: Resolved[DataTypeDefinition[Any]], value: JSONValue -) -> JSONValue: +def spelled_canonically(data_type: Resolved[F], value: JSONValue) -> JSONValue: """`value`, a fill value of `data_type` with no problem, in its canonical spelling, as `canonical_fill_value` gives it, without judging it again. Its `fill_value_canonical` is the extension author's code: what it diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/definition.py b/packages/zarr-metadata/src/zarr_metadata/v3/definition.py index 8d041aa888..290579951f 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/definition.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/definition.py @@ -351,6 +351,7 @@ def acme_lz4_rules( StorageClass, StorageTransformerDefinition, Unclaimed, + WithFillValue, canonical_fill_value, canonical_of, canonicalize, @@ -414,6 +415,7 @@ def acme_lz4_rules( "StorageTransformerField", "Unclaimed", "ValidationProblem", + "WithFillValue", "ZarrV3MetadataFieldJSON", "canonical_fill_value", "canonical_of", diff --git a/packages/zarr-metadata/tests/v3/test_kinds.py b/packages/zarr-metadata/tests/v3/test_kinds.py index e44f0906bb..9a33a421e7 100644 --- a/packages/zarr-metadata/tests/v3/test_kinds.py +++ b/packages/zarr-metadata/tests/v3/test_kinds.py @@ -4,7 +4,7 @@ from collections.abc import Mapping from dataclasses import dataclass -from typing import TYPE_CHECKING, Annotated, Any, ClassVar, TypeVar, cast +from typing import TYPE_CHECKING, Annotated, Any, ClassVar, TypeAlias, TypeVar, cast import pytest from annotated_types import Ge @@ -22,6 +22,9 @@ Read, Refused, Unclaimed, + WithFillValue, + canonical_fill_value, + fill_value_problems, resolve, ) @@ -153,3 +156,27 @@ def test_a_kind_reads_the_envelope_its_format_writes( if written is not None: assert not isinstance(resolved, Refused) assert resolved.to_json() == written + + +@dataclass(frozen=True, kw_only=True, slots=True, repr=False) +class Typed(WithFillValue[C]): + """A kind of another format whose definitions take a fill value.""" + + is_kind: ClassVar[bool] = True + label: ClassVar[str] = "typed" + + +Count: TypeAlias = Annotated[int, Ge(0)] | None + +TYPED = Typed(name="typed", configuration=EmptyConfiguration, fill_value=Count) + + +def test_a_fill_value_is_judged_by_any_kind_with_one() -> None: + """`fill_value_problems` and `canonical_fill_value` judge a fill value by the definition's `fill_value` members whatever kind it is, since the members live on `WithFillValue`, which every data type kind derives from.""" + resolved, _ = resolve("typed", Typed, Context.of(TYPED)) + assert fill_value_problems(resolved, 3) == () + assert fill_value_problems(resolved, None) == () + assert [(p.loc, p.kind) for p in fill_value_problems(resolved, -1, ("fill_value",))] == [ + (("fill_value",), "invalid_value") + ] + assert canonical_fill_value(resolved, 3) == 3 From 9d73abf76ecd3900a786eb4a4951bdf19bff8373 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Tue, 6 Oct 2026 21:51:28 +0200 Subject: [PATCH 58/94] feat(zarr-metadata): add the v2 data type and codec kinds ZarrV2DataTypeDefinition reads a NumPy typestr as its family with the byte order, size and unit carried, and a records array as struct. ZarrV2CodecDefinition reads a numcodecs configuration, its id the name and its other members the parameters. Assisted-by: ClaudeCode:claude-fable-5-1 Co-Authored-By: Claude Fable 5.1 --- .../src/zarr_metadata/v2/_definition.py | 264 ++++++++++++++++++ packages/zarr-metadata/tests/v2/test_kinds.py | 158 +++++++++++ 2 files changed, 422 insertions(+) create mode 100644 packages/zarr-metadata/src/zarr_metadata/v2/_definition.py create mode 100644 packages/zarr-metadata/tests/v2/test_kinds.py diff --git a/packages/zarr-metadata/src/zarr_metadata/v2/_definition.py b/packages/zarr-metadata/src/zarr_metadata/v2/_definition.py new file mode 100644 index 0000000000..04297ff11a --- /dev/null +++ b/packages/zarr-metadata/src/zarr_metadata/v2/_definition.py @@ -0,0 +1,264 @@ +"""The kinds of field a Zarr v2 array document holds: a dtype, and a codec in `compressor` or `filters`. + +A v2 dtype is written as NumPy writes a typestr -- a byte order, a type +code and a size, `|])([A-Za-z])(\d*)(?:\[(\d*)([A-Za-zμ]+)\])?") +"""A typestr: a byte order, a type code, a size in bytes if any, and a bracketed time unit with its multiplier if any.""" + + +def parse_typestr(name: str) -> tuple[str, dict[str, JSONValue]] | None: + """`name` split into its type code and what it carries; None when it is no typestr. + + What is carried is the byte order; the item size, which only `|O` + leaves out; and, when a unit is bracketed, the unit + and its multiplier, 1 when none is written. + """ + match = TYPESTR_PATTERN.fullmatch(name) + if match is None: + return None + byteorder, code, size, factor, unit = match.groups() + if size == "" and code != "O": + # "An integer specifying the number of bytes": only `|O` writes none. + return None + carried: dict[str, JSONValue] = {"byteorder": byteorder} + if size != "": + carried["itemsize"] = int(size) + if unit is not None: + carried["unit"] = unit + carried["scale_factor"] = 1 if factor == "" else int(factor) + return code, carried + + +def typestr_problem(name: str, at: Loc) -> ValidationProblem | None: + """The problem `name`, at `at`, is when it is no typestr; None when it is one, or is `struct`.""" + if name == STRUCT_NAME or TYPESTR_PATTERN.fullmatch(name) is not None: + return None + return ValidationProblem( + at, + "expected a NumPy typestr -- a byte order '<', '>' or '|', a type code and a size in " + f"bytes, ' str | None: + if parse_typestr(self.name) is not None: + return ( + f"{self.name!r} is how a document writes one type of a family, which reads as " + "the family; define the family" + ) + # Named, not a bare `super()`: a dataclass with slots is rebuilt. + return super(ZarrV2DataTypeDefinition, self)._refusal() + + @classmethod + def name_problem(cls, name: str, at: Loc) -> ValidationProblem | None: + return typestr_problem(name, at) + + @classmethod + def spelled(cls, name: str) -> tuple[str | None, dict[str, JSONValue] | None]: + if name == STRUCT_NAME: + return name, None + if name in FAMILIES.values(): + return None, None + parsed = parse_typestr(name) + if parsed is None: + return name, None + code, carried = parsed + family = FAMILIES.get(code) + if family is None: + return name, None + return family, carried + + def carrying_name(self, configuration: Mapping[str, JSONValue]) -> str | None: + if self.name == STRUCT_NAME: + return None + code = next(code for code, family in FAMILIES.items() if family == self.name) + name = f"{configuration['byteorder']}{code}{configuration.get('itemsize', '')}" + if "unit" in configuration: + factor = configuration.get("scale_factor", 1) + name += f"[{'' if factor == 1 else factor}{configuration['unit']}]" + return name + + @classmethod + def named_configuration( + cls, value: object + ) -> tuple[str | None, Mapping[str, object] | None, Problems]: + if isinstance(value, str): + return value, None, () + if isinstance(value, (list, tuple)): + return STRUCT_NAME, {"fields": cast("JSONValue", value)}, () + return None, None, () + + @classmethod + def envelope_problems(cls, value: object) -> Problems: + if isinstance(value, str): + bad = typestr_problem(value, ()) + return () if bad is None else (bad,) + if isinstance(value, (list, tuple)): + return () + return ( + ValidationProblem( + (), + "expected a v2 dtype -- a NumPy typestr, or an array of field records -- got " + f"{shown(value)}", + "invalid_type", + ), + ) + + @classmethod + def envelope_json(cls, name: str, configuration: Mapping[str, JSONValue]) -> JSONValue: + if name == STRUCT_NAME and "fields" in configuration: + return configuration["fields"] + return name + + @classmethod + def configuration_loc(cls, loc: Loc) -> Loc: + return loc + + @classmethod + def name_loc(cls, loc: Loc) -> Loc: + return loc + + +@dataclass(frozen=True, kw_only=True, slots=True, repr=False) +class ZarrV2CodecDefinition(Definition[C]): + """A v2 codec: a numcodecs id, and the TypedDict its parameters are. + + A document writes `{"id": name, **parameters}`; the definition's + configuration is the parameters, read at the field itself. + """ + + is_kind: ClassVar[bool] = True + label: ClassVar[str] = "v2 codec" + + @classmethod + def name_problem(cls, name: str, at: Loc) -> ValidationProblem | None: + if len(name) != 0: + return None + return ValidationProblem(at, "expected a codec id, got ''", "invalid_value") + + @classmethod + def named_configuration( + cls, value: object + ) -> tuple[str | None, Mapping[str, object] | None, Problems]: + if not isinstance(value, Mapping): + return None, None, () + entry = cast("Mapping[str, object]", value) + name = entry.get("id") + if not isinstance(name, str): + return None, None, () + return name, {key: item for key, item in entry.items() if key != "id"}, () + + @classmethod + def envelope_problems(cls, value: object) -> Problems: + if not isinstance(value, Mapping): + return ( + ValidationProblem( + (), "expected a codec configuration with a string 'id'", "invalid_type" + ), + ) + entry = cast("Mapping[str, object]", value) + if "id" not in entry: + return (ValidationProblem(("id",), "missing required key", "missing_key"),) + if not isinstance(entry["id"], str): + return ( + ValidationProblem( + ("id",), + f"expected a string codec id, got {shown(entry['id'])}", + "invalid_type", + ), + ) + return () + + @classmethod + def envelope_json(cls, name: str, configuration: Mapping[str, JSONValue]) -> JSONValue: + return {"id": name, **configuration} + + @classmethod + def configuration_loc(cls, loc: Loc) -> Loc: + return loc + + @classmethod + def name_loc(cls, loc: Loc) -> Loc: + return (*loc, "id") + + +__all__ = [ + "FAMILIES", + "STRUCT_NAME", + "TYPESTR_PATTERN", + "ZarrV2CodecDefinition", + "ZarrV2DataTypeDefinition", + "ZarrV2DataTypeField", + "parse_typestr", + "typestr_problem", +] diff --git a/packages/zarr-metadata/tests/v2/test_kinds.py b/packages/zarr-metadata/tests/v2/test_kinds.py new file mode 100644 index 0000000000..024bbacef3 --- /dev/null +++ b/packages/zarr-metadata/tests/v2/test_kinds.py @@ -0,0 +1,158 @@ +"""The two kinds of a v2 array document's fields: a dtype as NumPy spells one, and a numcodecs configuration.""" + +from __future__ import annotations + +from typing import Any + +import pytest + +from zarr_metadata.v2._definition import ( + ZarrV2CodecDefinition, + ZarrV2DataTypeDefinition, + parse_typestr, +) +from zarr_metadata.v3.definition import EmptyConfiguration + +Loc = tuple[str | int, ...] + + +@pytest.mark.parametrize( + ("name", "code", "carried"), + [ + ("i8", "i", {"byteorder": ">", "itemsize": 8}), + ("|b1", "b", {"byteorder": "|", "itemsize": 1}), + ("|S12", "S", {"byteorder": "|", "itemsize": 12}), + (" None: + """A NumPy typestr splits into its type code and what the name carries: the byte order, the item size when written, and a time unit with its multiplier when bracketed.""" + assert parse_typestr(name) == (code, carried) + + +@pytest.mark.parametrize( + "name", ["float32", "f4", " None: + """A string without a byte order, a type code and a size in that order is not a typestr.""" + assert parse_typestr(name) is None + + +@pytest.mark.parametrize( + ("name", "filed", "carried"), + [ + (" None: + """A typestr is filed under its family and carries its byte order, size and unit; `struct` is filed as itself; a typestr of a type code the spec does not list is filed under itself, so a scope leaves it unclaimed; a family name is filed by nothing, since no document writes it.""" + assert ZarrV2DataTypeDefinition.spelled(name) == (filed, carried) + + +@pytest.mark.parametrize( + ("value", "name", "configuration"), + [ + (" None: + """A dtype field is a string, which is its name, or an array of field records, which is a `struct` whose configuration is the records; anything else names nothing.""" + assert ZarrV2DataTypeDefinition.named_configuration(value) == (name, configuration, ()) + + +@pytest.mark.parametrize( + ("value", "problems"), + [ + (" None: + """The envelope of a dtype field holds when it is a typestr or an array of records; a string that is neither is reported as an invalid value, naming the typestr form, and anything else as an invalid type.""" + found = ZarrV2DataTypeDefinition.envelope_problems(value) + assert [(p.loc, p.kind) for p in found] == problems + if value == "float32": + assert "typestr" in found[0].message + + +def test_a_v2_dtype_writes_back_as_a_string_or_records() -> None: + """`envelope_json` writes a dtype as the string that carries it, and a struct as its records; its name and configuration both sit at the field.""" + assert ZarrV2DataTypeDefinition.envelope_json(" None: + """A codec field is an object whose `id` names it and whose other members are its parameters; one without a string `id`, or not an object, names nothing, and the envelope says why.""" + assert ZarrV2CodecDefinition.named_configuration(value) == (name, configuration, ()) + assert [(p.loc, p.kind) for p in ZarrV2CodecDefinition.envelope_problems(value)] == problems + + +def test_a_v2_codec_writes_back_with_its_id_beside_its_parameters() -> None: + """`envelope_json` writes a codec as `{"id": name, **parameters}`: its parameters at the field and its name under `id`.""" + assert ZarrV2CodecDefinition.envelope_json("zlib", {"level": 1}) == {"id": "zlib", "level": 1} + assert ZarrV2CodecDefinition.configuration_loc(("compressor",)) == ("compressor",) + assert ZarrV2CodecDefinition.name_loc(("compressor",)) == ("compressor", "id") + assert ZarrV2CodecDefinition.spelled("zlib") == ("zlib", None) + + +def test_error_a_v2_definition_is_named_as_its_format_names_one() -> None: + """A v2 codec definition with an empty id, and a v2 data type definition named by a typestr a document would write rather than a family, are refused when built.""" + with pytest.raises(TypeError, match="so no document names it"): + ZarrV2CodecDefinition(name="", configuration=EmptyConfiguration) + with pytest.raises(TypeError, match="is how a document writes"): + ZarrV2DataTypeDefinition(name=" Date: Tue, 6 Oct 2026 21:57:29 +0200 Subject: [PATCH 59/94] feat(zarr-metadata): define the v2 data types One definition per NumPy family: bool, int, uint, float, complex, bytes, str, void, datetime64, timedelta64, object and struct. Each says which sizes, byte orders and units it takes, the JSON shape of its fill value, and its canonical spelling. A name that carries what its family does not take is refused. Assisted-by: ClaudeCode:claude-fable-5-1 Co-Authored-By: Claude Fable 5.1 --- .../zarr_metadata/v2/data_type/__init__.py | 42 ++++ .../zarr_metadata/v2/data_type/fixed_width.py | 63 ++++++ .../src/zarr_metadata/v2/data_type/object.py | 31 +++ .../src/zarr_metadata/v2/data_type/scalar.py | 191 ++++++++++++++++ .../src/zarr_metadata/v2/data_type/struct.py | 64 ++++++ .../src/zarr_metadata/v2/data_type/time.py | 117 ++++++++++ .../src/zarr_metadata/v3/_definition.py | 12 +- .../zarr-metadata/tests/v2/test_data_types.py | 214 ++++++++++++++++++ 8 files changed, 730 insertions(+), 4 deletions(-) create mode 100644 packages/zarr-metadata/src/zarr_metadata/v2/data_type/__init__.py create mode 100644 packages/zarr-metadata/src/zarr_metadata/v2/data_type/fixed_width.py create mode 100644 packages/zarr-metadata/src/zarr_metadata/v2/data_type/object.py create mode 100644 packages/zarr-metadata/src/zarr_metadata/v2/data_type/scalar.py create mode 100644 packages/zarr-metadata/src/zarr_metadata/v2/data_type/struct.py create mode 100644 packages/zarr-metadata/src/zarr_metadata/v2/data_type/time.py create mode 100644 packages/zarr-metadata/tests/v2/test_data_types.py diff --git a/packages/zarr-metadata/src/zarr_metadata/v2/data_type/__init__.py b/packages/zarr-metadata/src/zarr_metadata/v2/data_type/__init__.py new file mode 100644 index 0000000000..85441a2828 --- /dev/null +++ b/packages/zarr-metadata/src/zarr_metadata/v2/data_type/__init__.py @@ -0,0 +1,42 @@ +"""The data types zarr-python 2.x writes, one definition per NumPy family.""" + +from typing import Any, Final + +from zarr_metadata.v2._definition import ZarrV2DataTypeDefinition +from zarr_metadata.v2.data_type.fixed_width import BYTES_V2, STR_V2, VOID_V2 +from zarr_metadata.v2.data_type.object import OBJECT_V2 +from zarr_metadata.v2.data_type.scalar import BOOL_V2, COMPLEX_V2, FLOAT_V2, INT_V2, UINT_V2 +from zarr_metadata.v2.data_type.struct import STRUCT_V2 +from zarr_metadata.v2.data_type.time import DATETIME64_V2, TIMEDELTA64_V2 + +V2_DATA_TYPES: Final[tuple[ZarrV2DataTypeDefinition[Any], ...]] = ( + BOOL_V2, + INT_V2, + UINT_V2, + FLOAT_V2, + COMPLEX_V2, + BYTES_V2, + STR_V2, + VOID_V2, + DATETIME64_V2, + TIMEDELTA64_V2, + OBJECT_V2, + STRUCT_V2, +) +"""Every v2 data type, by family.""" + +__all__ = [ + "BOOL_V2", + "BYTES_V2", + "COMPLEX_V2", + "DATETIME64_V2", + "FLOAT_V2", + "INT_V2", + "OBJECT_V2", + "STRUCT_V2", + "STR_V2", + "TIMEDELTA64_V2", + "UINT_V2", + "V2_DATA_TYPES", + "VOID_V2", +] diff --git a/packages/zarr-metadata/src/zarr_metadata/v2/data_type/fixed_width.py b/packages/zarr-metadata/src/zarr_metadata/v2/data_type/fixed_width.py new file mode 100644 index 0000000000..e870bde388 --- /dev/null +++ b/packages/zarr-metadata/src/zarr_metadata/v2/data_type/fixed_width.py @@ -0,0 +1,63 @@ +"""The v2 families of a fixed width per item: `bytes` (`S`), `str` (`U`) and `void` (`V`), of any size.""" + +from __future__ import annotations + +from typing import TYPE_CHECKING, Final + +from zarr_metadata._json import ValidationProblem, shown +from zarr_metadata.v2._definition import ZarrV2DataTypeDefinition +from zarr_metadata.v2.data_type.scalar import ZarrV2ScalarConfiguration, orderless_at, sized +from zarr_metadata.v3.data_type.bytes import base64_bytes + +if TYPE_CHECKING: + from collections.abc import Iterator + + from zarr_metadata.v3._definition import Nested + +ZarrV2Base64FillValue = str | None +"""A fill value the v2 spec encodes as base64: for a fixed-length byte string, void, or a structured type (https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v2/v2.0.rst#L191-L193); or null.""" + + +def base64_fill_value_rules( + configuration: object, nested: Nested, value: ZarrV2Base64FillValue +) -> Iterator[ValidationProblem]: + """A string of standard-alphabet base64, when it is not null.""" + if value is None: + return + try: + base64_bytes(value) + except ValueError: + yield ValidationProblem( + (), f"expected standard-alphabet base64, got {shown(value)}", "invalid_value" + ) + + +BYTES_V2: Final = ZarrV2DataTypeDefinition( + name="bytes", + configuration=ZarrV2ScalarConfiguration, + rules=sized(None, None), + canonical=orderless_at(None), + fill_value=ZarrV2Base64FillValue, + fill_value_rules=base64_fill_value_rules, +) +"""`|S`: byte strings of `n` bytes; the fill value base64 of one.""" + +STR_V2: Final = ZarrV2DataTypeDefinition( + name="str", + configuration=ZarrV2ScalarConfiguration, + rules=sized(None, frozenset()), + fill_value=str | None, +) +"""``: strings of `n` code points, each four bytes in the byte order written; the fill value a string.""" + +VOID_V2: Final = ZarrV2DataTypeDefinition( + name="void", + configuration=ZarrV2ScalarConfiguration, + rules=sized(None, None), + canonical=orderless_at(None), + fill_value=ZarrV2Base64FillValue, + fill_value_rules=base64_fill_value_rules, +) +"""`|V`: `n` bytes of no type; the fill value base64 of them.""" + +__all__ = ["BYTES_V2", "STR_V2", "VOID_V2", "ZarrV2Base64FillValue", "base64_fill_value_rules"] diff --git a/packages/zarr-metadata/src/zarr_metadata/v2/data_type/object.py b/packages/zarr-metadata/src/zarr_metadata/v2/data_type/object.py new file mode 100644 index 0000000000..8840b7d325 --- /dev/null +++ b/packages/zarr-metadata/src/zarr_metadata/v2/data_type/object.py @@ -0,0 +1,31 @@ +"""The v2 `object` family, `|O`: Python objects, whose fill value is any JSON.""" + +from __future__ import annotations + +from typing import Final, cast + +from typing_extensions import ReadOnly, TypedDict + +from zarr_metadata.v2._definition import ZarrV2DataTypeDefinition +from zarr_metadata.v2.data_type.scalar import ( + ZarrV2ByteOrder, # noqa: TC001 - a TypedDict's annotations are evaluated at run time +) + + +class ZarrV2ObjectConfiguration(TypedDict, closed=True): + """What `|O` carries: a byte order, which the type ignores.""" + + byteorder: ReadOnly[ZarrV2ByteOrder] + + +def _canonical(configuration: ZarrV2ObjectConfiguration) -> ZarrV2ObjectConfiguration: + """No byte order, `|`, as NumPy writes it.""" + return cast("ZarrV2ObjectConfiguration", {**configuration, "byteorder": "|"}) + + +OBJECT_V2: Final = ZarrV2DataTypeDefinition( + name="object", configuration=ZarrV2ObjectConfiguration, canonical=_canonical +) +"""`|O`: Python objects, each encoded by a filter; the fill value any JSON.""" + +__all__ = ["OBJECT_V2", "ZarrV2ObjectConfiguration"] diff --git a/packages/zarr-metadata/src/zarr_metadata/v2/data_type/scalar.py b/packages/zarr-metadata/src/zarr_metadata/v2/data_type/scalar.py new file mode 100644 index 0000000000..0bf4cbcab0 --- /dev/null +++ b/packages/zarr-metadata/src/zarr_metadata/v2/data_type/scalar.py @@ -0,0 +1,191 @@ +"""The v2 scalar families: `bool`, `int`, `uint`, `float` and `complex`, filed by family and read for every typestr of the family.""" + +from __future__ import annotations + +from collections.abc import Callable, Iterator +from typing import Annotated, Final, Literal, cast + +from annotated_types import Ge +from typing_extensions import ReadOnly, TypedDict + +from zarr_metadata._json import ValidationProblem +from zarr_metadata.v2._definition import ZarrV2DataTypeDefinition +from zarr_metadata.v3._definition import Nested + +ZarrV2ByteOrder = Literal["<", ">", "|"] +"""The byte orders a typestr writes: little-endian, big-endian, and not relevant.""" + + +class ZarrV2ScalarConfiguration(TypedDict, closed=True): + """What a scalar typestr carries: its byte order and its size in bytes, `{"byteorder": "<", "itemsize": 4}` for ` Rules: + """The rules of a family whose types are `sizes` bytes wide, any width when None. + + `orderless` is the sizes at which the type has no byte order, which + NumPy writes as `|`: one byte for an integer, every size for bytes + and void, none for a float; None is every size. At any other size + the typestr says which end comes first, `<` or `>`. + """ + + def rules( + configuration: ZarrV2ScalarConfiguration, nested: Nested + ) -> Iterator[ValidationProblem]: + size = configuration["itemsize"] + if sizes is not None and size not in sizes: + yield ValidationProblem( + ("itemsize",), + f"expected a size of {', '.join(map(str, sizes))} bytes, got {size}", + "invalid_value", + ) + return + if configuration["byteorder"] == "|" and orderless is not None and size not in orderless: + yield ValidationProblem( + ("byteorder",), + f"expected a byte order '<' or '>' for a type of {size} bytes, got '|'", + "invalid_value", + ) + + return rules + + +def orderless_at( + sizes: frozenset[int] | None, +) -> Callable[[ZarrV2ScalarConfiguration], ZarrV2ScalarConfiguration]: + """The canonical spelling of a family whose types of `sizes` bytes have no byte order: `|`, as NumPy writes it; None is every size.""" + + def canonical(configuration: ZarrV2ScalarConfiguration) -> ZarrV2ScalarConfiguration: + if (sizes is None or configuration["itemsize"] in sizes) and configuration[ + "byteorder" + ] != "|": + return cast("ZarrV2ScalarConfiguration", {**configuration, "byteorder": "|"}) + return configuration + + return canonical + + +_ONE_BYTE: Final = frozenset({1}) + + +def _in_range( + signed: bool, +) -> Callable[[ZarrV2ScalarConfiguration, Nested, int | None], Iterator[ValidationProblem]]: + """The rule that an integer fill value lies in the range of the type's size.""" + + def rules( + configuration: ZarrV2ScalarConfiguration, nested: Nested, value: int | None + ) -> Iterator[ValidationProblem]: + if value is None: + return + bits = 8 * configuration["itemsize"] + low, high = (-(2 ** (bits - 1)), 2 ** (bits - 1) - 1) if signed else (0, 2**bits - 1) + if not low <= value <= high: + yield ValidationProblem( + (), f"expected an integer in [{low}, {high}], got {value}", "invalid_value" + ) + + return rules + + +ZarrV2FloatSpecial = Literal["NaN", "Infinity", "-Infinity"] +"""The non-finite values the v2 spec spells by name (https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v2/v2.0.rst#L178-L190).""" + +ZarrV2FloatFillValue = float | int | ZarrV2FloatSpecial | None +"""A v2 float fill value: a number, a named non-finite value, or null.""" + +ZarrV2ComplexComponent = float | int | ZarrV2FloatSpecial +"""One component of a complex fill value: a float fill value that is not null.""" + +ZarrV2ComplexFillValue = tuple[ZarrV2ComplexComponent, ZarrV2ComplexComponent] | None +"""A v2 complex fill value: `[real, imag]`, each a float fill value, or null.""" + + +def _float_canonical( + configuration: ZarrV2ScalarConfiguration, nested: Nested, value: ZarrV2FloatFillValue +) -> ZarrV2FloatFillValue: + """An integer written for a float is the float: `0` and `0.0` are one value.""" + return float(value) if isinstance(value, int) and not isinstance(value, bool) else value + + +def _complex_canonical( + configuration: ZarrV2ScalarConfiguration, nested: Nested, value: ZarrV2ComplexFillValue +) -> ZarrV2ComplexFillValue: + """Each component in the float's canonical spelling.""" + if value is None: + return None + real, imag = value + return ( + cast("ZarrV2ComplexComponent", _float_canonical(configuration, nested, real)), + cast("ZarrV2ComplexComponent", _float_canonical(configuration, nested, imag)), + ) + + +BOOL_V2: Final = ZarrV2DataTypeDefinition( + name="bool", + configuration=ZarrV2ScalarConfiguration, + rules=sized((1,), None), + canonical=orderless_at(None), + fill_value=bool | None, +) +"""`|b1`: one byte, true or false.""" + +INT_V2: Final = ZarrV2DataTypeDefinition( + name="int", + configuration=ZarrV2ScalarConfiguration, + rules=sized((1, 2, 4, 8), _ONE_BYTE), + canonical=orderless_at(_ONE_BYTE), + fill_value=int | None, + fill_value_rules=_in_range(True), +) +"""`i1` to `i8`: signed integers, the fill value in the type's range.""" + +UINT_V2: Final = ZarrV2DataTypeDefinition( + name="uint", + configuration=ZarrV2ScalarConfiguration, + rules=sized((1, 2, 4, 8), _ONE_BYTE), + canonical=orderless_at(_ONE_BYTE), + fill_value=int | None, + fill_value_rules=_in_range(False), +) +"""`u1` to `u8`: unsigned integers, the fill value in the type's range.""" + +FLOAT_V2: Final = ZarrV2DataTypeDefinition( + name="float", + configuration=ZarrV2ScalarConfiguration, + rules=sized((2, 4, 8), frozenset()), + fill_value=ZarrV2FloatFillValue, + fill_value_canonical=_float_canonical, +) +"""`f2`, `f4`, `f8`: IEEE 754 floats; the fill value a number or a named non-finite value.""" + +COMPLEX_V2: Final = ZarrV2DataTypeDefinition( + name="complex", + configuration=ZarrV2ScalarConfiguration, + rules=sized((8, 16), frozenset()), + fill_value=ZarrV2ComplexFillValue, + fill_value_canonical=_complex_canonical, +) +"""`c8`, `c16`: complex floats; the fill value `[real, imag]`.""" + +__all__ = [ + "BOOL_V2", + "COMPLEX_V2", + "FLOAT_V2", + "INT_V2", + "UINT_V2", + "ZarrV2ByteOrder", + "ZarrV2ComplexComponent", + "ZarrV2ComplexFillValue", + "ZarrV2FloatFillValue", + "ZarrV2FloatSpecial", + "ZarrV2ScalarConfiguration", + "orderless_at", + "sized", +] diff --git a/packages/zarr-metadata/src/zarr_metadata/v2/data_type/struct.py b/packages/zarr-metadata/src/zarr_metadata/v2/data_type/struct.py new file mode 100644 index 0000000000..a96686d460 --- /dev/null +++ b/packages/zarr-metadata/src/zarr_metadata/v2/data_type/struct.py @@ -0,0 +1,64 @@ +"""The v2 `struct` type: an array of field records, each record's type a field of its own.""" + +from __future__ import annotations + +from typing import TYPE_CHECKING, Annotated, Final + +from annotated_types import Ge +from typing_extensions import ReadOnly, TypedDict + +from zarr_metadata._json import ValidationProblem +from zarr_metadata.v2._definition import STRUCT_NAME, ZarrV2DataTypeDefinition, ZarrV2DataTypeField +from zarr_metadata.v2.data_type.fixed_width import ZarrV2Base64FillValue, base64_fill_value_rules + +if TYPE_CHECKING: + from collections.abc import Iterator + + from zarr_metadata.v3._definition import Nested + +# The longer record first: a record that fits neither is reported by the +# first branch, and a shape that is wrong is deeper than a length that is. +ZarrV2StructRecord = ( + tuple[str, ZarrV2DataTypeField, tuple[Annotated[int, Ge(0)], ...]] + | tuple[str, ZarrV2DataTypeField] +) +"""A field record: `[name, dtype]` or `[name, dtype, shape]`, the dtype a nested field (https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v2/v2.0.rst#L152-L174).""" + + +class ZarrV2StructConfiguration(TypedDict, closed=True): + """The records of a structured dtype, which the document writes as the dtype itself.""" + + fields: ReadOnly[tuple[ZarrV2StructRecord, ...]] + + +def _rules(configuration: ZarrV2StructConfiguration, nested: Nested) -> Iterator[ValidationProblem]: + """At least one record, each named, the names distinct.""" + fields = configuration["fields"] + if len(fields) == 0: + yield ValidationProblem(("fields",), "expected at least one field record", "invalid_value") + seen: dict[str, int] = {} + for index, record in enumerate(fields): + name = record[0] + if name == "": + yield ValidationProblem( + ("fields", index, 0), "expected a non-empty field name", "invalid_value" + ) + first = seen.setdefault(name, index) + if first != index: + yield ValidationProblem( + ("fields", index, 0), + f"duplicate field name {name!r}, already used by record {first}", + "invalid_value", + ) + + +STRUCT_V2: Final = ZarrV2DataTypeDefinition( + name=STRUCT_NAME, + configuration=ZarrV2StructConfiguration, + rules=_rules, + fill_value=ZarrV2Base64FillValue, + fill_value_rules=base64_fill_value_rules, +) +"""A structured type: an array of field records; the fill value base64 of one record (https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v2/v2.0.rst#L191-L193).""" + +__all__ = ["STRUCT_V2", "ZarrV2StructConfiguration", "ZarrV2StructRecord"] diff --git a/packages/zarr-metadata/src/zarr_metadata/v2/data_type/time.py b/packages/zarr-metadata/src/zarr_metadata/v2/data_type/time.py new file mode 100644 index 0000000000..f769104a61 --- /dev/null +++ b/packages/zarr-metadata/src/zarr_metadata/v2/data_type/time.py @@ -0,0 +1,117 @@ +"""The v2 time families, `datetime64` (`M`) and `timedelta64` (`m`): eight bytes, in a unit the typestr brackets.""" + +from __future__ import annotations + +from typing import TYPE_CHECKING, Annotated, Final, Literal, NotRequired, cast + +from annotated_types import Ge +from typing_extensions import ReadOnly, TypedDict + +from zarr_metadata._json import ValidationProblem +from zarr_metadata.v2._definition import ZarrV2DataTypeDefinition +from zarr_metadata.v2.data_type.scalar import ( + ZarrV2ByteOrder, # noqa: TC001 - a TypedDict's annotations are evaluated at run time +) +from zarr_metadata.v3.data_type._numpy_time import ( + NumpyTimeScaleFactor, + NumpyTimeTicks, + numpy_time_fill_value_canonical, +) + +if TYPE_CHECKING: + from collections.abc import Iterator + + from zarr_metadata.v3._definition import Nested + + +class ZarrV2TimeConfiguration(TypedDict, closed=True): + """What a time typestr carries: ` Iterator[ValidationProblem]: + """Eight bytes, in an order; and a unit, which the v2 spec requires (https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v2/v2.0.rst#L147-L150).""" + size = configuration["itemsize"] + if size != 8: + yield ValidationProblem( + ("itemsize",), f"expected a size of 8 bytes, got {size}", "invalid_value" + ) + return + if configuration["byteorder"] == "|": + yield ValidationProblem( + ("byteorder",), + "expected a byte order '<' or '>' for a type of 8 bytes, got '|'", + "invalid_value", + ) + if "unit" not in configuration: + yield ValidationProblem( + ("unit",), + "expected a time unit in brackets, ' ZarrV2TimeConfiguration: + """`μs` is written `us`, as NumPy writes it.""" + if configuration.get("unit") == "μs": + return cast("ZarrV2TimeConfiguration", {**configuration, "unit": "us"}) + return configuration + + +ZarrV2TimeFillValue = NumpyTimeTicks | Literal["NaT"] | None +"""A v2 time fill value: a count of ticks, `"NaT"`, or null.""" + + +def _fill_value_canonical( + configuration: ZarrV2TimeConfiguration, nested: Nested, value: ZarrV2TimeFillValue +) -> ZarrV2TimeFillValue: + """`NaT` for not-a-time however it is written; any other value as written.""" + if value is None: + return None + return cast( + "ZarrV2TimeFillValue", numpy_time_fill_value_canonical(configuration, nested, value) + ) + + +DATETIME64_V2: Final = ZarrV2DataTypeDefinition( + name="datetime64", + configuration=ZarrV2TimeConfiguration, + rules=_rules, + canonical=_canonical, + fill_value=ZarrV2TimeFillValue, + fill_value_canonical=_fill_value_canonical, +) +"""` bool: def _read_carried( data: JSONValue, name: str, + kind: type[Definition[Any]], definition: Definition[Any], given: Mapping[str, object] | None, carried: Mapping[str, JSONValue], @@ -1762,7 +1763,7 @@ def _read_carried( _, beside, _ = _checked( EmptyConfiguration, {} if given is None else given, - type(definition).configuration_loc(loc), + kind.configuration_loc(loc), ) configuration, judged = definition.judge(carried) # What is wrong with what the name carries is the field's: found at @@ -1772,8 +1773,11 @@ def _read_carried( *beside, *(dataclasses.replace(found, loc=loc, input=UNSET, ctx={}) for found in judged), ) - if configuration is None or not _usable(problems): - refused = Refused(json=data, name=name, read_as=DataTypeDefinition, definition=definition) + # A name is one word: what it carries that the definition does not + # declare cannot be left out, as a stray key beside the name can, so + # anything wrong with what the name carries refuses the field. + if configuration is None or not _usable(beside) or len(judged) != 0: + refused = Refused(json=data, name=name, read_as=kind, definition=definition) return refused, problems return Read(json=data, name=name, definition=definition, configuration=configuration), problems diff --git a/packages/zarr-metadata/tests/v2/test_data_types.py b/packages/zarr-metadata/tests/v2/test_data_types.py new file mode 100644 index 0000000000..eb00b7c478 --- /dev/null +++ b/packages/zarr-metadata/tests/v2/test_data_types.py @@ -0,0 +1,214 @@ +"""Every v2 data type, read as its family: which typestrs it takes, and which fill values.""" + +from __future__ import annotations + +import pytest + +from zarr_metadata.v2._definition import ZarrV2DataTypeDefinition +from zarr_metadata.v2.data_type import V2_DATA_TYPES +from zarr_metadata.v3.definition import ( + Context, + Read, + Refused, + Unclaimed, + canonical_fill_value, + canonical_of, + fill_value_problems, + resolve, +) + +SCOPE = Context.of(*V2_DATA_TYPES) + +Loc = tuple[str | int, ...] + + +def _read(value: object) -> tuple[type, list[tuple[Loc, str]]]: + resolved, problems = resolve(value, ZarrV2DataTypeDefinition, SCOPE, ("dtype",)) + return type(resolved), [(p.loc, p.kind) for p in problems] + + +@pytest.mark.parametrize( + ("value", "family", "canonical"), + [ + ("|b1", "bool", "|b1"), + ("i4", "int", ">i4"), + ("u8", "uint", ">u8"), + ("f8", "float", ">f8"), + ("c16", "complex", ">c16"), + ("|S0", "bytes", "|S0"), + ("|S12", "bytes", "|S12"), + ("U0", "str", ">U0"), + ("|V8", "void", "|V8"), + ("|O", "object", "|O"), + ("M8[10s]", "datetime64", ">M8[10s]"), + (" None: + """Each typestr of a family reads by the family's definition, and its simplest spelling is the typestr NumPy writes: `|` for a type of one byte or no byte order, `us` for a microsecond unit; a records array reads as `struct`, each record's type read too.""" + resolved, problems = resolve(value, ZarrV2DataTypeDefinition, SCOPE) + assert isinstance(resolved, Read) + assert resolved.definition.name == family + assert problems == () + assert canonical_of(resolved, problems) == canonical + + +@pytest.mark.parametrize("value", [" None: + """A typestr whose type code the v2 spec does not list is filed under itself, which nothing in scope claims: read as `Unclaimed`, with no problem, and written back as it was.""" + resolved, problems = resolve(value, ZarrV2DataTypeDefinition, SCOPE) + assert isinstance(resolved, Unclaimed) + assert problems == () + assert resolved.to_json() == value + + +@pytest.mark.parametrize( + ("value", "at"), + [ + ("float32", ("dtype",)), + (" None: + """A typestr of a size, byte order or unit its family does not take is refused, the problem at the field; a struct's problems sit under `fields`, at the record and its position, and a record's type that is refused is reported there while the struct still reads.""" + kind, problems = _read(value) + assert kind is Refused or "fields" in at + assert next(loc for loc, _ in problems) == at + + +@pytest.mark.parametrize( + ("dtype", "value", "canonical"), + [ + ("|b1", True, True), + ("|b1", None, None), + (" None: + """Each family takes the fill values the v2 spec gives it, and `null` always: booleans, integers in the type's range, numbers or the named non-finite strings for floats and each complex component, base64 for bytes, void and struct, a string for str, any JSON for object, ticks or `NaT` for times. The canonical spelling makes an integer written for a float a float, and `NaT` of its ticks.""" + resolved, _ = resolve(dtype, ZarrV2DataTypeDefinition, SCOPE) + assert fill_value_problems(resolved, value) == () + assert canonical_fill_value(resolved, value) == canonical + + +@pytest.mark.parametrize( + ("dtype", "value", "at"), + [ + ("|b1", 1, ()), + (" None: + """A fill value outside the shape or range its family takes is reported, at the value or at the component that is wrong: a v2 float takes no hex string, and a struct's fill value is base64 of the record, not an object of fields.""" + resolved, _ = resolve(dtype, ZarrV2DataTypeDefinition, SCOPE) + problems = fill_value_problems(resolved, value, ("fill_value",)) + assert len(problems) != 0 + assert problems[0].loc == ("fill_value", *at) + + +def test_every_family_has_an_example() -> None: + """Every definition in `V2_DATA_TYPES` is exercised above, so none is filed untested.""" + names = {definition.name for definition in V2_DATA_TYPES} + assert names == { + "bool", + "int", + "uint", + "float", + "complex", + "bytes", + "str", + "void", + "datetime64", + "timedelta64", + "object", + "struct", + } From 49c9a9f059029c07d76b39c9feb28bb8b18f0e83 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Tue, 6 Oct 2026 22:02:30 +0200 Subject: [PATCH 60/94] feat(zarr-metadata): define the v2 codecs numcodecs 0.16 writes zarr_metadata.v2.codec becomes a package. It defines the 21 numcodecs configurations the package models, each a TypedDict of its parameters with the ranges numcodecs takes. A parameter numcodecs defaults is optional. A dtype parameter must be a NumPy typestr. The codec JSON shape moves to a leaf module so the array document and the definitions can both import it. Assisted-by: ClaudeCode:claude-fable-5-1 Co-Authored-By: Claude Fable 5.1 --- .../v2/{codec.py => _codec_json.py} | 11 +- .../src/zarr_metadata/v2/array.py | 2 +- .../src/zarr_metadata/v2/codec/__init__.py | 81 ++++++++++++ .../src/zarr_metadata/v2/codec/_dtype.py | 42 +++++++ .../src/zarr_metadata/v2/codec/checksum.py | 28 +++++ .../src/zarr_metadata/v2/codec/compression.py | 109 +++++++++++++++++ .../src/zarr_metadata/v2/codec/filters.py | 105 ++++++++++++++++ .../src/zarr_metadata/v2/codec/vlen.py | 29 +++++ .../zarr-metadata/tests/v2/test_codecs.py | 115 ++++++++++++++++++ 9 files changed, 512 insertions(+), 10 deletions(-) rename packages/zarr-metadata/src/zarr_metadata/v2/{codec.py => _codec_json.py} (67%) create mode 100644 packages/zarr-metadata/src/zarr_metadata/v2/codec/__init__.py create mode 100644 packages/zarr-metadata/src/zarr_metadata/v2/codec/_dtype.py create mode 100644 packages/zarr-metadata/src/zarr_metadata/v2/codec/checksum.py create mode 100644 packages/zarr-metadata/src/zarr_metadata/v2/codec/compression.py create mode 100644 packages/zarr-metadata/src/zarr_metadata/v2/codec/filters.py create mode 100644 packages/zarr-metadata/src/zarr_metadata/v2/codec/vlen.py create mode 100644 packages/zarr-metadata/tests/v2/test_codecs.py diff --git a/packages/zarr-metadata/src/zarr_metadata/v2/codec.py b/packages/zarr-metadata/src/zarr_metadata/v2/_codec_json.py similarity index 67% rename from packages/zarr-metadata/src/zarr_metadata/v2/codec.py rename to packages/zarr-metadata/src/zarr_metadata/v2/_codec_json.py index 69125544e6..68d17a59c0 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v2/codec.py +++ b/packages/zarr-metadata/src/zarr_metadata/v2/_codec_json.py @@ -1,9 +1,4 @@ -""" -Zarr v2 codec configuration shape. - -In v2, compressors and filters are numcodecs configuration dicts: a required -`id` field naming the codec, plus arbitrary codec-specific extra fields. -""" +"""The JSON shape of a v2 codec, in a module of its own so the array document and the codec definitions can both import it.""" from typing_extensions import TypedDict @@ -24,6 +19,4 @@ class ZarrV2CodecMetadata(TypedDict, extra_items=JSONValue): id: str -__all__ = [ - "ZarrV2CodecMetadata", -] +__all__ = ["ZarrV2CodecMetadata"] diff --git a/packages/zarr-metadata/src/zarr_metadata/v2/array.py b/packages/zarr-metadata/src/zarr_metadata/v2/array.py index 1b02e1ffc7..5c0f75dce6 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v2/array.py +++ b/packages/zarr-metadata/src/zarr_metadata/v2/array.py @@ -6,7 +6,7 @@ from typing_extensions import TypeAliasType, TypedDict from zarr_metadata._common import JSONValue -from zarr_metadata.v2.codec import ZarrV2CodecMetadata +from zarr_metadata.v2._codec_json import ZarrV2CodecMetadata ZarrV2DataTypeMetadata = TypeAliasType( "ZarrV2DataTypeMetadata", diff --git a/packages/zarr-metadata/src/zarr_metadata/v2/codec/__init__.py b/packages/zarr-metadata/src/zarr_metadata/v2/codec/__init__.py new file mode 100644 index 0000000000..71bab561a7 --- /dev/null +++ b/packages/zarr-metadata/src/zarr_metadata/v2/codec/__init__.py @@ -0,0 +1,81 @@ +"""Zarr v2 codecs: the configuration shape, and one definition per numcodecs id this package models. + +In v2, compressors and filters are numcodecs configuration dicts: a required +`id` field naming the codec, plus codec-specific parameters. +""" + +from typing import Any, Final + +from zarr_metadata.v2._codec_json import ZarrV2CodecMetadata +from zarr_metadata.v2._definition import ZarrV2CodecDefinition +from zarr_metadata.v2.codec.checksum import ADLER32_V2, CRC32_V2, CRC32C_V2, FLETCHER32_V2 +from zarr_metadata.v2.codec.compression import ( + BLOSC_V2, + BZ2_V2, + GZIP_V2, + LZ4_V2, + LZMA_V2, + ZLIB_V2, + ZSTD_V2, +) +from zarr_metadata.v2.codec.filters import ( + ASTYPE_V2, + BITROUND_V2, + DELTA_V2, + FIXEDSCALEOFFSET_V2, + PACKBITS_V2, + QUANTIZE_V2, + SHUFFLE_V2, +) +from zarr_metadata.v2.codec.vlen import VLEN_ARRAY_V2, VLEN_BYTES_V2, VLEN_UTF8_V2 + +V2_CODECS: Final[tuple[ZarrV2CodecDefinition[Any], ...]] = ( + ZLIB_V2, + GZIP_V2, + BZ2_V2, + LZMA_V2, + BLOSC_V2, + ZSTD_V2, + LZ4_V2, + SHUFFLE_V2, + DELTA_V2, + FIXEDSCALEOFFSET_V2, + QUANTIZE_V2, + BITROUND_V2, + ASTYPE_V2, + PACKBITS_V2, + VLEN_UTF8_V2, + VLEN_BYTES_V2, + VLEN_ARRAY_V2, + CRC32_V2, + CRC32C_V2, + ADLER32_V2, + FLETCHER32_V2, +) +"""Every codec numcodecs 0.16 configures that this package models.""" + +__all__ = [ + "ADLER32_V2", + "ASTYPE_V2", + "BITROUND_V2", + "BLOSC_V2", + "BZ2_V2", + "CRC32C_V2", + "CRC32_V2", + "DELTA_V2", + "FIXEDSCALEOFFSET_V2", + "FLETCHER32_V2", + "GZIP_V2", + "LZ4_V2", + "LZMA_V2", + "PACKBITS_V2", + "QUANTIZE_V2", + "SHUFFLE_V2", + "V2_CODECS", + "VLEN_ARRAY_V2", + "VLEN_BYTES_V2", + "VLEN_UTF8_V2", + "ZLIB_V2", + "ZSTD_V2", + "ZarrV2CodecMetadata", +] diff --git a/packages/zarr-metadata/src/zarr_metadata/v2/codec/_dtype.py b/packages/zarr-metadata/src/zarr_metadata/v2/codec/_dtype.py new file mode 100644 index 0000000000..7ced0b5580 --- /dev/null +++ b/packages/zarr-metadata/src/zarr_metadata/v2/codec/_dtype.py @@ -0,0 +1,42 @@ +"""A rule for the dtype parameters of a v2 codec.""" + +from __future__ import annotations + +from typing import TYPE_CHECKING, cast + +from zarr_metadata._json import ValidationProblem +from zarr_metadata.v2._definition import parse_typestr, typestr_problem + +if TYPE_CHECKING: + from collections.abc import Callable, Iterator, Mapping + + from zarr_metadata.v3._definition import Nested + + +def dtype_parameter( + *keys: str, float_only: bool = False +) -> Callable[[Mapping[str, object], Nested], Iterator[ValidationProblem]]: + """The rule that each of `keys`, when written, is a typestr: numcodecs writes `np.dtype(x).str` there, and a struct is no element type. + + With `float_only`, each must name a float, as `quantize` requires. + """ + + def rules(configuration: Mapping[str, object], nested: Nested) -> Iterator[ValidationProblem]: + for key in keys: + if key not in configuration: + continue + written = cast("str", configuration[key]) + bad = typestr_problem(written, (key,)) + if bad is not None: + yield bad + continue + parsed = parse_typestr(written) + if float_only and (parsed is None or parsed[0] != "f"): + yield ValidationProblem( + (key,), f"expected a float typestr, ' None: + """Every definition in `V2_CODECS` is exercised below.""" + assert {definition.name for definition in V2_CODECS} == set(EXAMPLES) + + +@pytest.mark.parametrize( + ("name", "field"), CASES, ids=[f"{n}:{i}" for i, (n, _) in enumerate(CASES)] +) +def test_every_example_reads_and_is_written_back_as_it_was( + name: str, field: ZarrV2CodecMetadata +) -> None: + """A numcodecs configuration reads by the definition its id names, with no problem, and is written back as it was: the id beside the parameters, which have one spelling.""" + resolved, problems = resolve(field, ZarrV2CodecDefinition, SCOPE) + assert isinstance(resolved, Read) + assert resolved.definition.name == name + assert problems == () + assert resolved.to_json() == field + assert canonical_of(resolved, problems) == field + + +@pytest.mark.parametrize( + "field", + [{"id": "categorize", "labels": ["a"]}, {"id": "pickle"}, {"id": "n5_wrapper", "inner": 1}], +) +def test_an_id_the_package_does_not_model_is_unclaimed(field: dict[str, Any]) -> None: + """A codec id nothing in scope claims reads as `Unclaimed`, its parameters kept and unjudged, as a v3 extension nothing claims is.""" + resolved, problems = resolve(field, ZarrV2CodecDefinition, SCOPE) + assert isinstance(resolved, Unclaimed) + assert problems == () + assert json.dumps(resolved.to_json()) == json.dumps(field) + + +@pytest.mark.parametrize( + ("field", "at", "kind"), + [ + ({"id": "zlib", "level": 10}, ("c", "level"), "invalid_value"), + ({"id": "gzip", "level": -1}, ("c", "level"), "invalid_value"), + ({"id": "bz2", "level": 0}, ("c", "level"), "invalid_value"), + ({"id": "blosc", "cname": "brotli"}, ("c", "cname"), "invalid_value"), + ({"id": "blosc", "shuffle": 3}, ("c", "shuffle"), "invalid_value"), + ({"id": "blosc", "blocksize": -1}, ("c", "blocksize"), "invalid_value"), + ({"id": "zstd", "level": 23}, ("c", "level"), "invalid_value"), + ({"id": "shuffle", "elementsize": 0}, ("c", "elementsize"), "invalid_value"), + ({"id": "delta"}, ("c", "dtype"), "missing_key"), + ({"id": "delta", "dtype": "float32"}, ("c", "dtype"), "invalid_value"), + ( + {"id": "astype", "encode_dtype": " None: + """A parameter out of its range, of the wrong type, missing when numcodecs has no default, a dtype parameter that is no typestr, or a key no codec declares is reported at the parameter, beside the field.""" + resolved, problems = resolve(field, ZarrV2CodecDefinition, SCOPE, ("c",)) + assert [(p.loc, p.kind) for p in problems] == [(at, kind)] + assert isinstance(resolved, Read if kind == "unknown_key" else Refused) From 79eb1964be99f2cfaa2738cc3d83db22d5c0f397 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Tue, 6 Oct 2026 22:05:43 +0200 Subject: [PATCH 61/94] feat(zarr-metadata): add CORE_V2 and the v2 field readers zarr_metadata.v2.definition is the public door: CORE_V2 files the v2 data types and codecs, and resolve_dtype_v2 and resolve_codec_v2 read one field in a scope, CORE_V2 by default. The v2 rules are values, not closures, so a scope that holds them pickles. Assisted-by: ClaudeCode:claude-fable-5-1 Co-Authored-By: Claude Fable 5.1 --- packages/zarr-metadata/docs/api/index.md | 4 +- packages/zarr-metadata/docs/api/v2.md | 7 ++ .../src/zarr_metadata/v2/codec/_dtype.py | 30 +++--- .../src/zarr_metadata/v2/data_type/scalar.py | 93 +++++++++++-------- .../src/zarr_metadata/v2/definition.py | 83 +++++++++++++++++ .../zarr-metadata/tests/v2/test_definition.py | 45 +++++++++ 6 files changed, 212 insertions(+), 50 deletions(-) create mode 100644 packages/zarr-metadata/src/zarr_metadata/v2/definition.py create mode 100644 packages/zarr-metadata/tests/v2/test_definition.py diff --git a/packages/zarr-metadata/docs/api/index.md b/packages/zarr-metadata/docs/api/index.md index f7ec11a57f..574cbcb9af 100644 --- a/packages/zarr-metadata/docs/api/index.md +++ b/packages/zarr-metadata/docs/api/index.md @@ -16,7 +16,9 @@ The package is organized to mirror the structure of the Zarr specifications: typing spec defines them, with every problem located, and `json_schema`, which writes what `check` reads as a JSON Schema - [`zarr_metadata.v2`](v2.md) — `TypedDict` shapes for Zarr v2 documents - (`.zarray`, `.zgroup`, `.zattrs`, `.zmetadata`) + (`.zarray`, `.zgroup`, `.zattrs`, `.zmetadata`), and + `zarr_metadata.v2.definition`: the v2 dtypes and numcodecs codecs as + definitions, `CORE_V2`, and the readers of one `dtype` or codec - [`zarr_metadata.v3`](v3/index.md) — `TypedDict` shapes for Zarr v3 documents, with subpackages for [chunk grids](v3/chunk_grid.md), [chunk key encodings](v3/chunk_key_encoding.md), [codecs](v3/codec.md), diff --git a/packages/zarr-metadata/docs/api/v2.md b/packages/zarr-metadata/docs/api/v2.md index 2fe5b6ec56..6fdc61ec84 100644 --- a/packages/zarr-metadata/docs/api/v2.md +++ b/packages/zarr-metadata/docs/api/v2.md @@ -15,3 +15,10 @@ title: v2 ::: zarr_metadata.v2.codec ::: zarr_metadata.v2.consolidated + +::: zarr_metadata.v2.definition + options: + inherited_members: false + show_if_no_docstring: false + +::: zarr_metadata.v2.data_type diff --git a/packages/zarr-metadata/src/zarr_metadata/v2/codec/_dtype.py b/packages/zarr-metadata/src/zarr_metadata/v2/codec/_dtype.py index 7ced0b5580..9f75076529 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v2/codec/_dtype.py +++ b/packages/zarr-metadata/src/zarr_metadata/v2/codec/_dtype.py @@ -2,27 +2,29 @@ from __future__ import annotations +from dataclasses import dataclass from typing import TYPE_CHECKING, cast from zarr_metadata._json import ValidationProblem from zarr_metadata.v2._definition import parse_typestr, typestr_problem if TYPE_CHECKING: - from collections.abc import Callable, Iterator, Mapping + from collections.abc import Iterator, Mapping from zarr_metadata.v3._definition import Nested -def dtype_parameter( - *keys: str, float_only: bool = False -) -> Callable[[Mapping[str, object], Nested], Iterator[ValidationProblem]]: - """The rule that each of `keys`, when written, is a typestr: numcodecs writes `np.dtype(x).str` there, and a struct is no element type. +@dataclass(frozen=True, slots=True) +class _DtypeParameter: + """The rule `dtype_parameter` gives: a value rather than a closure, so a scope holding it pickles.""" - With `float_only`, each must name a float, as `quantize` requires. - """ + keys: tuple[str, ...] + float_only: bool - def rules(configuration: Mapping[str, object], nested: Nested) -> Iterator[ValidationProblem]: - for key in keys: + def __call__( + self, configuration: Mapping[str, object], nested: Nested + ) -> Iterator[ValidationProblem]: + for key in self.keys: if key not in configuration: continue written = cast("str", configuration[key]) @@ -31,12 +33,18 @@ def rules(configuration: Mapping[str, object], nested: Nested) -> Iterator[Valid yield bad continue parsed = parse_typestr(written) - if float_only and (parsed is None or parsed[0] != "f"): + if self.float_only and (parsed is None or parsed[0] != "f"): yield ValidationProblem( (key,), f"expected a float typestr, ' _DtypeParameter: + """The rule that each of `keys`, when written, is a typestr: numcodecs writes `np.dtype(x).str` there, and a struct is no element type. + + With `float_only`, each must name a float, as `quantize` requires. + """ + return _DtypeParameter(keys, float_only) __all__ = ["dtype_parameter"] diff --git a/packages/zarr-metadata/src/zarr_metadata/v2/data_type/scalar.py b/packages/zarr-metadata/src/zarr_metadata/v2/data_type/scalar.py index 0bf4cbcab0..82baf7cccc 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v2/data_type/scalar.py +++ b/packages/zarr-metadata/src/zarr_metadata/v2/data_type/scalar.py @@ -2,15 +2,19 @@ from __future__ import annotations -from collections.abc import Callable, Iterator -from typing import Annotated, Final, Literal, cast +from dataclasses import dataclass +from typing import TYPE_CHECKING, Annotated, Final, Literal, cast from annotated_types import Ge from typing_extensions import ReadOnly, TypedDict from zarr_metadata._json import ValidationProblem from zarr_metadata.v2._definition import ZarrV2DataTypeDefinition -from zarr_metadata.v3._definition import Nested + +if TYPE_CHECKING: + from collections.abc import Iterator + + from zarr_metadata.v3._definition import Nested ZarrV2ByteOrder = Literal["<", ">", "|"] """The byte orders a typestr writes: little-endian, big-endian, and not relevant.""" @@ -23,76 +27,89 @@ class ZarrV2ScalarConfiguration(TypedDict, closed=True): itemsize: ReadOnly[Annotated[int, Ge(0)]] -Rules = Callable[[ZarrV2ScalarConfiguration, Nested], Iterator[ValidationProblem]] - +@dataclass(frozen=True, slots=True) +class _Sized: + """The rules of a family whose types are `sizes` bytes wide: a value rather than a closure, so a scope holding it pickles.""" -def sized(sizes: tuple[int, ...] | None, orderless: frozenset[int] | None) -> Rules: - """The rules of a family whose types are `sizes` bytes wide, any width when None. - - `orderless` is the sizes at which the type has no byte order, which - NumPy writes as `|`: one byte for an integer, every size for bytes - and void, none for a float; None is every size. At any other size - the typestr says which end comes first, `<` or `>`. - """ + sizes: tuple[int, ...] | None + orderless: frozenset[int] | None - def rules( - configuration: ZarrV2ScalarConfiguration, nested: Nested + def __call__( + self, configuration: ZarrV2ScalarConfiguration, nested: Nested ) -> Iterator[ValidationProblem]: size = configuration["itemsize"] - if sizes is not None and size not in sizes: + if self.sizes is not None and size not in self.sizes: yield ValidationProblem( ("itemsize",), - f"expected a size of {', '.join(map(str, sizes))} bytes, got {size}", + f"expected a size of {', '.join(map(str, self.sizes))} bytes, got {size}", "invalid_value", ) return - if configuration["byteorder"] == "|" and orderless is not None and size not in orderless: + if ( + configuration["byteorder"] == "|" + and self.orderless is not None + and size not in self.orderless + ): yield ValidationProblem( ("byteorder",), f"expected a byte order '<' or '>' for a type of {size} bytes, got '|'", "invalid_value", ) - return rules +def sized(sizes: tuple[int, ...] | None, orderless: frozenset[int] | None) -> _Sized: + """The rules of a family whose types are `sizes` bytes wide, any width when None. + + `orderless` is the sizes at which the type has no byte order, which + NumPy writes as `|`: one byte for an integer, every size for bytes + and void, none for a float; None is every size. At any other size + the typestr says which end comes first, `<` or `>`. + """ + return _Sized(sizes, orderless) -def orderless_at( - sizes: frozenset[int] | None, -) -> Callable[[ZarrV2ScalarConfiguration], ZarrV2ScalarConfiguration]: - """The canonical spelling of a family whose types of `sizes` bytes have no byte order: `|`, as NumPy writes it; None is every size.""" - def canonical(configuration: ZarrV2ScalarConfiguration) -> ZarrV2ScalarConfiguration: - if (sizes is None or configuration["itemsize"] in sizes) and configuration[ - "byteorder" - ] != "|": +@dataclass(frozen=True, slots=True) +class _OrderlessAt: + """The canonical spelling of a family whose types of `sizes` bytes have no byte order.""" + + sizes: frozenset[int] | None + + def __call__(self, configuration: ZarrV2ScalarConfiguration) -> ZarrV2ScalarConfiguration: + orderless = self.sizes is None or configuration["itemsize"] in self.sizes + if orderless and configuration["byteorder"] != "|": return cast("ZarrV2ScalarConfiguration", {**configuration, "byteorder": "|"}) return configuration - return canonical + +def orderless_at(sizes: frozenset[int] | None) -> _OrderlessAt: + """The canonical spelling of a family whose types of `sizes` bytes have no byte order: `|`, as NumPy writes it; None is every size.""" + return _OrderlessAt(sizes) _ONE_BYTE: Final = frozenset({1}) -def _in_range( - signed: bool, -) -> Callable[[ZarrV2ScalarConfiguration, Nested, int | None], Iterator[ValidationProblem]]: +@dataclass(frozen=True, slots=True) +class _InRange: """The rule that an integer fill value lies in the range of the type's size.""" - def rules( - configuration: ZarrV2ScalarConfiguration, nested: Nested, value: int | None + signed: bool + + def __call__( + self, configuration: ZarrV2ScalarConfiguration, nested: Nested, value: int | None ) -> Iterator[ValidationProblem]: if value is None: return bits = 8 * configuration["itemsize"] - low, high = (-(2 ** (bits - 1)), 2 ** (bits - 1) - 1) if signed else (0, 2**bits - 1) + if self.signed: + low, high = -(2 ** (bits - 1)), 2 ** (bits - 1) - 1 + else: + low, high = 0, 2**bits - 1 if not low <= value <= high: yield ValidationProblem( (), f"expected an integer in [{low}, {high}], got {value}", "invalid_value" ) - return rules - ZarrV2FloatSpecial = Literal["NaN", "Infinity", "-Infinity"] """The non-finite values the v2 spec spells by name (https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v2/v2.0.rst#L178-L190).""" @@ -142,7 +159,7 @@ def _complex_canonical( rules=sized((1, 2, 4, 8), _ONE_BYTE), canonical=orderless_at(_ONE_BYTE), fill_value=int | None, - fill_value_rules=_in_range(True), + fill_value_rules=_InRange(True), ) """`i1` to `i8`: signed integers, the fill value in the type's range.""" @@ -152,7 +169,7 @@ def _complex_canonical( rules=sized((1, 2, 4, 8), _ONE_BYTE), canonical=orderless_at(_ONE_BYTE), fill_value=int | None, - fill_value_rules=_in_range(False), + fill_value_rules=_InRange(False), ) """`u1` to `u8`: unsigned integers, the fill value in the type's range.""" diff --git a/packages/zarr-metadata/src/zarr_metadata/v2/definition.py b/packages/zarr-metadata/src/zarr_metadata/v2/definition.py new file mode 100644 index 0000000000..56d27ae31a --- /dev/null +++ b/packages/zarr-metadata/src/zarr_metadata/v2/definition.py @@ -0,0 +1,83 @@ +"""Zarr v2 fields read against their definitions: the public door. + +A v2 array document has two kinds of field: its `dtype`, and the codecs +in `compressor` and `filters`. Zarr v2 has no extension registry, but it +has a scope all the same: the data types zarr-python 2.x writes, one +definition per NumPy family, and the codecs numcodecs 0.16 configures, +one per id this package models. `CORE_V2` is that scope. Read one field +in it with `resolve_dtype_v2` or `resolve_codec_v2`, which give `Read`, +`Unclaimed` or `Refused` as `zarr_metadata.v3.definition.resolve` does +for a v3 field; the scope algebra -- `Context.of`, `extended_with`, +`joined`, `claimant` -- is the same `Context`. + +A dtype reads as its family, the typestr's byte order, size and unit its +configuration: ` tuple[Resolved[ZarrV2DataTypeDefinition[Any]], Problems]: + """`value`, a v2 `dtype`, read in `context`, `CORE_V2` when none is given: what the scope made of it, and every problem, each prefixed with `loc`.""" + return resolve(value, ZarrV2DataTypeDefinition, CORE_V2 if context is None else context, loc) + + +def resolve_codec_v2( + value: object, context: Context | None = None, loc: Loc = () +) -> tuple[Resolved[ZarrV2CodecDefinition[Any]], Problems]: + """`value`, a v2 `compressor` or one of its `filters`, read in `context`, `CORE_V2` when none is given: what the scope made of it, and every problem, each prefixed with `loc`.""" + return resolve(value, ZarrV2CodecDefinition, CORE_V2 if context is None else context, loc) + + +__all__ = [ + "CORE_V2", + "V2_CODECS", + "V2_DATA_TYPES", + "Context", + "Read", + "Refused", + "Resolved", + "Unclaimed", + "ZarrV2CodecDefinition", + "ZarrV2DataTypeDefinition", + "ZarrV2DataTypeField", + "canonical_fill_value", + "canonical_of", + "fill_value_problems", + "resolve_codec_v2", + "resolve_dtype_v2", +] diff --git a/packages/zarr-metadata/tests/v2/test_definition.py b/packages/zarr-metadata/tests/v2/test_definition.py new file mode 100644 index 0000000000..8b06c655fb --- /dev/null +++ b/packages/zarr-metadata/tests/v2/test_definition.py @@ -0,0 +1,45 @@ +"""The public door to v2 definitions: the scope zarr-python 2.x reads in, and one field read in it.""" + +from __future__ import annotations + +import pickle + +from zarr_metadata.v2.definition import ( + CORE_V2, + V2_CODECS, + V2_DATA_TYPES, + Context, + Read, + Unclaimed, + ZarrV2CodecDefinition, + ZarrV2DataTypeDefinition, + resolve_codec_v2, + resolve_dtype_v2, +) +from zarr_metadata.v3.codec.gzip import GZIP_CODEC +from zarr_metadata.v3.definition import CORE_AND_EXTENSIONS, CodecDefinition, DataTypeDefinition + + +def test_core_v2_files_every_v2_definition_apart_from_v3() -> None: + """`CORE_V2` files the 12 data types and 21 codecs by their v2 kinds; a scope joined with the v3 scope files all of them apart, since the kinds differ: a v2 `gzip` and a v3 `gzip` are two definitions.""" + assert set(CORE_V2.definitions()) == {*V2_DATA_TYPES, *V2_CODECS} + both = Context.joined(CORE_V2, CORE_AND_EXTENSIONS) + assert both.claimant(ZarrV2CodecDefinition, "gzip") is not GZIP_CODEC + assert both.claimant(ZarrV2CodecDefinition, "gzip") is not None + assert both.claimant(CodecDefinition, "gzip") is GZIP_CODEC + assert both.claimant(ZarrV2DataTypeDefinition, " None: + """`resolve_dtype_v2` and `resolve_codec_v2` read one field in `CORE_V2` when no scope is given, and in the scope given otherwise, prefixing every problem with `loc`.""" + dtype, problems = resolve_dtype_v2(" Date: Tue, 6 Oct 2026 22:08:51 +0200 Subject: [PATCH 62/94] feat(zarr-metadata): read v2 dtype, codecs and fill value in CORE_V2 validate_array_metadata_v2 reads dtype, compressor and filters through resolve_dtype_v2 and resolve_codec_v2, and judges fill_value by the dtype the scope read. A string that is no typestr, a size or unit the family does not take, a codec parameter outside numcodecs' range, or a fill value the dtype refuses is now a problem. Assisted-by: ClaudeCode:claude-fable-5-1 Co-Authored-By: Claude Fable 5.1 --- .../changes/v2-scopes.feature.md | 1 + .../src/zarr_metadata/model/_validation.py | 102 +++++------------- .../zarr-metadata/tests/model/test_array.py | 39 ++++--- .../tests/model/test_pydantic_module.py | 1 + .../tests/model/test_v2_scope_reads.py | 96 +++++++++++++++++ 5 files changed, 150 insertions(+), 89 deletions(-) create mode 100644 packages/zarr-metadata/changes/v2-scopes.feature.md create mode 100644 packages/zarr-metadata/tests/model/test_v2_scope_reads.py diff --git a/packages/zarr-metadata/changes/v2-scopes.feature.md b/packages/zarr-metadata/changes/v2-scopes.feature.md new file mode 100644 index 0000000000..dbdfd719b2 --- /dev/null +++ b/packages/zarr-metadata/changes/v2-scopes.feature.md @@ -0,0 +1 @@ +Zarr v2 array metadata is now read against a scope, as v3 metadata is. `zarr_metadata.v2.definition` defines the data types zarr-python 2.x writes (one definition per NumPy family) and the codecs numcodecs 0.16 configures (21 ids), files them in `CORE_V2`, and reads one `dtype` or codec with `resolve_dtype_v2` and `resolve_codec_v2`. `validate_array_metadata_v2` now reports a dtype string that is not a NumPy typestr or that its family does not take, a codec parameter outside what numcodecs accepts, and a fill value its dtype does not take; a dtype or codec id the package does not model is left unjudged as before. diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_validation.py b/packages/zarr-metadata/src/zarr_metadata/model/_validation.py index 3f78329592..60d44cdfd3 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_validation.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/_validation.py @@ -40,7 +40,6 @@ refine_json, refine_user_data, shown_key, - validate_json, with_input, within, ) @@ -48,6 +47,7 @@ from zarr_metadata._sentinel import UNSET from zarr_metadata._typed_json import typeddict_keys from zarr_metadata.v2.array import ZarrV2ArrayMetadataJSON +from zarr_metadata.v2.definition import resolve_codec_v2, resolve_dtype_v2 from zarr_metadata.v2.group import ZarrV2GroupMetadataJSON from zarr_metadata.v3._definition import ( Chunk, @@ -236,30 +236,6 @@ def dimension_lengths( return tuple(value), () -def _is_dtype_v2(value: object) -> bool: - """Whether `value` is shaped like a v2 dtype: a string or field records. - - A field record is a `(name, dtype)` or `(name, dtype, shape)` sequence, - where `dtype` is itself a string or nested field records and `shape` is a - sequence of int. The string content is NOT interpreted — whether the - string names a real dtype is domain validity, not structure. - """ - if isinstance(value, str): - return True - if not _is_array(value): - return False - for record in value: - if not _is_array(record) or len(record) not in (2, 3): - return False - if not isinstance(record[0], str): - return False - if not _is_dtype_v2(record[1]): - return False - if len(record) == 3 and not _is_int_sequence(record[2]): - return False - return True - - def _is_canonical_dtype_v2(value: object) -> bool: """Whether a validated v2 dtype uses the tuple-backed public representation.""" if isinstance(value, str): @@ -324,24 +300,6 @@ def _is_canonical_array_metadata_v2(value: object) -> bool: ) -def _is_codec_v2(value: object) -> bool: - """Whether `value` is shaped like a v2 codec config: a mapping with a string `id`.""" - return isinstance(value, Mapping) and isinstance( - cast("Mapping[object, object]", value).get("id"), str - ) - - -def _validate_codec_v2(value: object, loc: tuple[str | int, ...]) -> tuple[ValidationProblem, ...]: - """Validate a v2 codec's required shape and JSON-valued configuration, `value` sitting at `loc`.""" - if not _is_codec_v2(value): - return ( - ValidationProblem( - loc, "expected a codec configuration with a string 'id'", "invalid_type" - ), - ) - return validate_json(value, loc) - - def validate_attributes(value: object) -> tuple[ValidationProblem, ...]: """Validate an `attributes` value: a mapping with string keys. @@ -693,10 +651,10 @@ def parse_array_metadata_v3( def validate_array_metadata_v2(value: object) -> tuple[ValidationProblem, ...]: """Return every reason `value` is not a structurally-valid v2 array doc. - Checks structure, not domain validity: `dtype` must be a string or field - records, but the string content is not interpreted; `compressor` and - `filters` are required keys that may be `None`, and otherwise must be - codec configurations (mappings with a string `id`). + `dtype`, `compressor` and `filters` are read in `CORE_V2`: a dtype or + codec the scope refuses is a problem, one it does not claim is not; + `fill_value` is judged by the dtype the scope read. `compressor` and + `filters` are required keys that may be `None`. """ if not isinstance(value, Mapping): return not_an_object(value) @@ -722,48 +680,44 @@ def validate_array_metadata_v2(value: object) -> tuple[ValidationProblem, ...]: "invalid_value", ) ) + dtype = None if "dtype" in doc: - # JSON first, nested no deeper than a reader walks, then its shape: - # field records nest a dtype two levels a record, without bound. - found = validate_json(doc["dtype"], ("dtype",)) + # Read in CORE_V2: a typestr by its family, field records as a struct. + dtype, found = resolve_dtype_v2(doc["dtype"], loc=("dtype",)) problems.extend(found) - if len(found) == 0 and not _is_dtype_v2(doc["dtype"]): - problems.append( - ValidationProblem( - ("dtype",), - "expected a v2 dtype string or an array of field records", - "invalid_type", - ) - ) if "order" in doc and doc["order"] not in ("C", "F"): problems.append(outside_of(("order",), doc["order"], ("C", "F"))) if "compressor" in doc: compressor = doc["compressor"] if compressor is not None: - problems.extend(_validate_codec_v2(compressor, ("compressor",))) + problems.extend(resolve_codec_v2(compressor, loc=("compressor",))[1]) if "filters" in doc: filters = doc["filters"] - if filters is not None and ( - not _is_array(filters) or not all(_is_codec_v2(item) for item in filters) - ): - problems.append( - ValidationProblem( - ("filters",), - "expected null or an array of codec configurations, each with a string 'id'", - "invalid_type", + if filters is not None: + if not _is_array(filters): + problems.append( + ValidationProblem( + ("filters",), + "expected null or an array of codec configurations", + "invalid_type", + ) ) - ) - elif _is_array(filters): - # "A list of JSON objects providing codec configurations, or - # null" (https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v2/v2.0.rst#L76-L79): an empty list is a list. - for index, item in enumerate(filters): - problems.extend(validate_json(item, ("filters", index))) + else: + # "A list of JSON objects providing codec configurations, or + # null" (https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v2/v2.0.rst#L76-L79): an empty list is a list. + for index, item in enumerate(filters): + problems.extend(resolve_codec_v2(item, loc=("filters", index))[1]) if "dimension_separator" in doc and doc["dimension_separator"] not in (".", "/"): problems.append( outside_of(("dimension_separator",), doc["dimension_separator"], (".", "/")) ) if "fill_value" in doc: - problems.extend(validate_json(doc["fill_value"], ("fill_value",))) + # JSON, judged by the dtype the scope read, when there is one: a + # dtype nothing in scope claims leaves it unjudged. + fill_value, found = refine_json(doc["fill_value"], ("fill_value",)) + problems.extend(found) + if len(found) == 0 and dtype is not None: + problems.extend(fill_value_problems(dtype, fill_value, ("fill_value",))) if "attributes" in doc: problems.extend(validate_attributes(doc["attributes"])) return with_input(problems, doc) diff --git a/packages/zarr-metadata/tests/model/test_array.py b/packages/zarr-metadata/tests/model/test_array.py index 95dd77b6f8..d924b3dbd0 100644 --- a/packages/zarr-metadata/tests/model/test_array.py +++ b/packages/zarr-metadata/tests/model/test_array.py @@ -634,13 +634,13 @@ def test_a_model_holding_nan_user_data_equals_its_copies() -> None: ("NaN", "NaN", True), (0.0, -0.0, False), (1, 1.0, False), - (0, False, False), + ("Infinity", "NaN", False), ], ) def test_two_v2_models_are_one_array_when_their_documents_are_written_alike( left: object, right: object, same: bool ) -> None: - """A v2 data type is not interpreted, so nor is its fill value: the document as text decides.""" + """A v2 model compares by its document as text: a fill value spelled two ways is two models.""" model = ZarrV2ArrayMetadata.create_default(shape=(2,), chunks=(2,), dtype=" None: pytest.param( ZarrV2ArrayMetadata.create_default( attributes={"a": {"b": [1]}}, - compressor={"id": "zstd", "opts": {"level": 1}}, - filters=({"id": "delta", "cfg": [1]},), + # Ids nothing in the scope claims, so their parameters can nest; + # a complex type, whose fill value is a pair. + dtype=" None: # Non-None compressor/filters must round-trip; extra assertion on .compressor. """A v2 model with non-None compressor and filters round-trips.""" compressor: ZarrV2CodecMetadata = {"id": "blosc", "clevel": 5} - filters: tuple[ZarrV2CodecMetadata, ...] = ({"id": "delta"},) + filters: tuple[ZarrV2CodecMetadata, ...] = ({"id": "delta", "dtype": " None: """A structured v2 dtype (field records, optionally nested/shaped) validates.""" dtype = (("a", " None: - """A field record with the wrong arity is rejected.""" + """A field record with the wrong arity is rejected, at the record.""" doc = dict(ZarrV2ArrayMetadata.create_default().to_json()) | {"dtype": (("a",),)} problems = validate_array_metadata_v2(doc) - assert [p.loc for p in problems] == [("dtype",)] + assert [p.loc for p in problems] == [("dtype", "fields", 0)] def test_v2_order_literal_enforced() -> None: @@ -1623,18 +1629,18 @@ def test_v2_compressor_must_be_codec_or_none() -> None: def test_v2_compressor_requires_string_id() -> None: - """A compressor mapping without a string id is rejected.""" + """A compressor mapping without a string id is rejected, at the id.""" doc = dict(ZarrV2ArrayMetadata.create_default().to_json()) | {"compressor": {"level": 3}} problems = validate_array_metadata_v2(doc) - assert [p.loc for p in problems] == [("compressor",)] + assert [p.loc for p in problems] == [("compressor", "id")] def test_v2_filters_must_be_codec_sequence_or_none() -> None: - """Filters that are not null or a sequence of codec configs are rejected.""" - for bad in (7, (5,), "gzip"): + """Filters that are not null or a sequence of codec configs are rejected: an item that is no codec at the item, anything else at the field.""" + for bad, at in ((7, ("filters",)), ((5,), ("filters", 0)), ("gzip", ("filters",))): doc = dict(ZarrV2ArrayMetadata.create_default().to_json()) | {"filters": bad} problems = validate_array_metadata_v2(doc) - assert [(p.loc, p.kind) for p in problems] == [(("filters",), "invalid_type")], bad + assert [(p.loc, p.kind) for p in problems] == [(at, "invalid_type")], bad def test_v2_shape_and_chunks_must_have_equal_rank() -> None: @@ -1941,6 +1947,7 @@ def records(levels: int) -> object: document = { **ZarrV2ArrayMetadata.create_default(shape=(2,)).to_json(), "dtype": records(deepest), + "fill_value": None, } assert validate_array_metadata_v2(document) == () model = ZarrV2ArrayMetadata.from_json(document) @@ -1957,8 +1964,9 @@ def test_a_v2_document_nested_as_deep_as_a_reader_walks_is_read_and_written() -> fill_value: dict[str, object] = {} for _ in range(JSON_DEPTH - 2): fill_value = {"x": fill_value} + # An object type, whose fill value is any JSON. document = { - **ZarrV2ArrayMetadata.create_default(shape=(2,)).to_json(), + **ZarrV2ArrayMetadata.create_default(shape=(2,), dtype="|O").to_json(), "fill_value": fill_value, } assert validate_array_metadata_v2(document) == () @@ -2157,7 +2165,8 @@ def test_array_guards_reject_noncanonical_nested_json() -> None: # Raw bits of 16, whose fill value is two byte values. v3 = dict(ZarrV3ArrayMetadata.create_default().to_json()) | {"data_type": "r16"} v3["fill_value"] = range(2) - v2 = dict(ZarrV2ArrayMetadata.create_default().to_json()) + # A complex type, whose fill value is a pair. + v2 = dict(ZarrV2ArrayMetadata.create_default(dtype=" None: """The schema accepts nested structured dtypes supported by the v2 specification.""" doc = json.loads(json.dumps(V2_ARRAY_DOC)) doc["dtype"] = [["outer", [["inner", " None: + """A dtype the scope reads with a fill value its family takes, a compressor and filters the scope reads or leaves unclaimed, and a null fill value, each validate with no problem.""" + assert validate_array_metadata_v2({**BASE, **changes}) == () + + +@pytest.mark.parametrize( + ("changes", "at", "kind"), + [ + ({"dtype": "float32"}, ("dtype",), "invalid_value"), + ({"dtype": " None: + """A dtype, fill value, compressor or filter the scope refuses is one problem of the document, at the field or the parameter that is wrong; before, the string content of a dtype and the parameters of a codec went unjudged.""" + problems = validate_array_metadata_v2({**BASE, **changes}) + assert [(p.loc, p.kind) for p in problems] == [(at, kind)] + + +def test_a_v2_model_refuses_a_dtype_the_scope_refuses() -> None: + """`ZarrV2ArrayMetadata` checks itself when built, so a dtype the scope refuses raises as any other problem does.""" + with pytest.raises(MetadataValidationError, match="typestr"): + ZarrV2ArrayMetadata.create_default(dtype="float32") From 4ac25eb2761a9a817fd8ad7c62fdffb5b8d1fd6c Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Tue, 6 Oct 2026 22:25:23 +0200 Subject: [PATCH 63/94] fix(zarr-metadata): accept what zarr-python 2.x writes that review round 1 found refused Aligned structs name their padding '' and may repeat it. zlib and gzip take level -1, which numcodecs writes. An empty codec id is reported, not accepted silently. create_default with a dtype and no fill value takes null. field_json_schema refuses a kind of another format with a clear error. A nested value that is no field is shown in its message. Assisted-by: ClaudeCode:claude-fable-5-1 Co-Authored-By: Claude Fable 5.1 --- .../src/zarr_metadata/model/_array.py | 4 ++++ .../src/zarr_metadata/v2/_definition.py | 3 ++- .../src/zarr_metadata/v2/codec/compression.py | 4 ++-- .../src/zarr_metadata/v2/data_type/struct.py | 6 ++--- .../src/zarr_metadata/v3/_common.py | 2 +- .../src/zarr_metadata/v3/_definition.py | 9 +++++++- .../tests/model/test_v2_scope_reads.py | 7 ++++++ .../zarr-metadata/tests/v2/test_codecs.py | 8 ++++--- .../zarr-metadata/tests/v2/test_data_types.py | 8 +++++-- packages/zarr-metadata/tests/v2/test_kinds.py | 3 ++- packages/zarr-metadata/tests/v3/test_kinds.py | 23 +++++++++++++++++++ 11 files changed, 62 insertions(+), 15 deletions(-) diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_array.py b/packages/zarr-metadata/src/zarr_metadata/model/_array.py index 8aead5e710..fdd35f4e95 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_array.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/_array.py @@ -612,6 +612,10 @@ def create_default(cls, **overrides: Unpack[ZarrV2ArrayMetadataPartial]) -> Zarr filters=None, attributes=UNSET, ) + # `0` is a fill value of the integer families only: a dtype given + # without a fill value takes `null`, which every family takes. + if "dtype" in overrides and "fill_value" not in overrides: + overrides["fill_value"] = None return default.update(**overrides) def __eq__(self, other: object) -> bool: diff --git a/packages/zarr-metadata/src/zarr_metadata/v2/_definition.py b/packages/zarr-metadata/src/zarr_metadata/v2/_definition.py index 04297ff11a..e40dd6c22e 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v2/_definition.py +++ b/packages/zarr-metadata/src/zarr_metadata/v2/_definition.py @@ -237,7 +237,8 @@ def envelope_problems(cls, value: object) -> Problems: "invalid_type", ), ) - return () + bad = cls.name_problem(entry["id"], ("id",)) + return () if bad is None else (bad,) @classmethod def envelope_json(cls, name: str, configuration: Mapping[str, JSONValue]) -> JSONValue: diff --git a/packages/zarr-metadata/src/zarr_metadata/v2/codec/compression.py b/packages/zarr-metadata/src/zarr_metadata/v2/codec/compression.py index cd7843c36f..dd43c41c2a 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v2/codec/compression.py +++ b/packages/zarr-metadata/src/zarr_metadata/v2/codec/compression.py @@ -16,8 +16,8 @@ ) from zarr_metadata.v2._definition import ZarrV2CodecDefinition -ZarrV2CompressionLevel = Annotated[int, Interval(ge=0, le=9)] -"""A zlib-style compression level, 0 to 9.""" +ZarrV2CompressionLevel = Annotated[int, Interval(ge=-1, le=9)] +"""A zlib-style compression level, 0 to 9, or -1 for zlib's default, which numcodecs writes as given.""" class ZarrV2ZlibParameters(TypedDict, closed=True): diff --git a/packages/zarr-metadata/src/zarr_metadata/v2/data_type/struct.py b/packages/zarr-metadata/src/zarr_metadata/v2/data_type/struct.py index a96686d460..08ff8e86d2 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v2/data_type/struct.py +++ b/packages/zarr-metadata/src/zarr_metadata/v2/data_type/struct.py @@ -32,7 +32,7 @@ class ZarrV2StructConfiguration(TypedDict, closed=True): def _rules(configuration: ZarrV2StructConfiguration, nested: Nested) -> Iterator[ValidationProblem]: - """At least one record, each named, the names distinct.""" + """At least one record, the names distinct; `""` is NumPy's name for the padding of an aligned struct, which zarr-python 2.x writes, and may repeat.""" fields = configuration["fields"] if len(fields) == 0: yield ValidationProblem(("fields",), "expected at least one field record", "invalid_value") @@ -40,9 +40,7 @@ def _rules(configuration: ZarrV2StructConfiguration, nested: Nested) -> Iterator for index, record in enumerate(fields): name = record[0] if name == "": - yield ValidationProblem( - ("fields", index, 0), "expected a non-empty field name", "invalid_value" - ) + continue first = seen.setdefault(name, index) if first != index: yield ValidationProblem( diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/_common.py b/packages/zarr-metadata/src/zarr_metadata/v3/_common.py index c55a5123b3..8411fa6224 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/_common.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/_common.py @@ -134,7 +134,7 @@ def envelope_problems( return ( ValidationProblem( (), - "expected a metadata field (string or extension object)", + f"expected a metadata field (string or extension object), got {shown(value)}", "invalid_type", ), ) diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py b/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py index 115e40cd7b..c6980f0678 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py @@ -828,8 +828,15 @@ def field_json_schema(kind: type[Definition[Any]], context: Context) -> JSONSche rule says -- a blosc `typesize` against its `shuffle` -- is not in it, so a field it accepts may still have a problem. """ + asked = as_kind(kind) + if not any(issubclass(asked, known) for known in KINDS): + msg = ( + f"{asked.__name__} is a kind of another format; the JSON Schema writer writes " + "Zarr v3 fields only" + ) + raise TypeError(msg) schemas = Schemas(field_schemas(context)) - return schemas.document(schemas.of(kind_field(as_kind(kind)))) + return schemas.document(schemas.of(kind_field(asked))) def field_schemas(context: Context) -> SchemaLeaf: diff --git a/packages/zarr-metadata/tests/model/test_v2_scope_reads.py b/packages/zarr-metadata/tests/model/test_v2_scope_reads.py index 9b835e81a0..2f0e3aac00 100644 --- a/packages/zarr-metadata/tests/model/test_v2_scope_reads.py +++ b/packages/zarr-metadata/tests/model/test_v2_scope_reads.py @@ -90,6 +90,13 @@ def test_error_what_the_scope_refuses_is_a_problem_of_the_document( assert [(p.loc, p.kind) for p in problems] == [(at, kind)] +def test_create_default_with_a_dtype_takes_no_fill_value() -> None: + """`create_default` given a dtype and no fill value builds a model with `null` for one, since `0` is no fill value of most families; a fill value given is kept.""" + assert ZarrV2ArrayMetadata.create_default(dtype="|b1").fill_value is None + assert ZarrV2ArrayMetadata.create_default(dtype=" None: """`ZarrV2ArrayMetadata` checks itself when built, so a dtype the scope refuses raises as any other problem does.""" with pytest.raises(MetadataValidationError, match="typestr"): diff --git a/packages/zarr-metadata/tests/v2/test_codecs.py b/packages/zarr-metadata/tests/v2/test_codecs.py index a79bf4c29d..9ad04cc485 100644 --- a/packages/zarr-metadata/tests/v2/test_codecs.py +++ b/packages/zarr-metadata/tests/v2/test_codecs.py @@ -16,8 +16,9 @@ # numcodecs 0.16.5 `get_config()` of a default instance, where one exists. EXAMPLES: dict[str, tuple[dict[str, Any], ...]] = { - "zlib": ({"id": "zlib", "level": 1}, {"id": "zlib"}), - "gzip": ({"id": "gzip", "level": 9},), + # -1 is zlib's default-level constant, which numcodecs writes as given. + "zlib": ({"id": "zlib", "level": 1}, {"id": "zlib"}, {"id": "zlib", "level": -1}), + "gzip": ({"id": "gzip", "level": 9}, {"id": "gzip", "level": -1}), "bz2": ({"id": "bz2", "level": 1},), "lzma": ({"id": "lzma", "format": 1, "check": -1, "preset": None, "filters": None},), "blosc": ( @@ -83,7 +84,8 @@ def test_an_id_the_package_does_not_model_is_unclaimed(field: dict[str, Any]) -> ("field", "at", "kind"), [ ({"id": "zlib", "level": 10}, ("c", "level"), "invalid_value"), - ({"id": "gzip", "level": -1}, ("c", "level"), "invalid_value"), + ({"id": "gzip", "level": -2}, ("c", "level"), "invalid_value"), + ({"id": ""}, ("c", "id"), "invalid_value"), ({"id": "bz2", "level": 0}, ("c", "level"), "invalid_value"), ({"id": "blosc", "cname": "brotli"}, ("c", "cname"), "invalid_value"), ({"id": "blosc", "shuffle": 3}, ("c", "shuffle"), "invalid_value"), diff --git a/packages/zarr-metadata/tests/v2/test_data_types.py b/packages/zarr-metadata/tests/v2/test_data_types.py index eb00b7c478..dc8c9df788 100644 --- a/packages/zarr-metadata/tests/v2/test_data_types.py +++ b/packages/zarr-metadata/tests/v2/test_data_types.py @@ -57,6 +57,12 @@ def _read(value: object) -> tuple[type, list[tuple[Loc, str]]]: (" None: @@ -95,7 +101,6 @@ def test_a_type_code_the_spec_does_not_list_is_unclaimed(value: str) -> None: ("|O4", ("dtype",)), ([], ("dtype", "fields")), ([["x", " None: "object-size", "no-records", "duplicate-name", - "empty-name", "record-type", "record-shape", "short-record", diff --git a/packages/zarr-metadata/tests/v2/test_kinds.py b/packages/zarr-metadata/tests/v2/test_kinds.py index 024bbacef3..1cd4642bef 100644 --- a/packages/zarr-metadata/tests/v2/test_kinds.py +++ b/packages/zarr-metadata/tests/v2/test_kinds.py @@ -127,9 +127,10 @@ def test_a_v2_dtype_writes_back_as_a_string_or_records() -> None: ({"id": "packbits"}, "packbits", {}, []), ({"level": 1}, None, None, [(("id",), "missing_key")]), ({"id": 3}, None, None, [(("id",), "invalid_type")]), + ({"id": ""}, "", {}, [(("id",), "invalid_value")]), ("zlib", None, None, [((), "invalid_type")]), ], - ids=["parameters", "bare", "no-id", "id-not-a-string", "not-an-object"], + ids=["parameters", "bare", "no-id", "id-not-a-string", "empty-id", "not-an-object"], ) def test_a_v2_codec_is_an_object_with_a_string_id( value: object, diff --git a/packages/zarr-metadata/tests/v3/test_kinds.py b/packages/zarr-metadata/tests/v3/test_kinds.py index 9a33a421e7..5be3d5dfb8 100644 --- a/packages/zarr-metadata/tests/v3/test_kinds.py +++ b/packages/zarr-metadata/tests/v3/test_kinds.py @@ -24,6 +24,7 @@ Unclaimed, WithFillValue, canonical_fill_value, + field_json_schema, fill_value_problems, resolve, ) @@ -180,3 +181,25 @@ def test_a_fill_value_is_judged_by_any_kind_with_one() -> None: (("fill_value",), "invalid_value") ] assert canonical_fill_value(resolved, 3) == 3 + + +def test_error_the_json_schema_writer_writes_v3_fields_only() -> None: + """`field_json_schema` writes the v3 envelope, so a kind of another format is refused with a `TypeError` naming the limit rather than a wrong schema.""" + from zarr_metadata.v2.definition import CORE_V2, ZarrV2DataTypeDefinition + + with pytest.raises(TypeError, match="Zarr v3"): + field_json_schema(ZarrV2DataTypeDefinition, CORE_V2) + with pytest.raises(TypeError, match="Zarr v3"): + field_json_schema(Tag, Context.of(Tag(name="tag1", configuration=EmptyConfiguration))) + + +def test_a_v3_value_that_is_no_field_is_shown_in_the_message() -> None: + """A nested value that is not a metadata field at all is reported with the value shown, as it was before the kind's envelope judged it.""" + _, problems = resolve( + {"name": "sharding_indexed", "configuration": {"chunk_shape": [1], "codecs": [3]}}, + CodecDefinition, + Context.of( + *__import__("zarr_metadata.v3.definition", fromlist=["CORE"]).CORE.definitions() + ), + ) + assert any(p.message.endswith("got 3") for p in problems), [p.message for p in problems] From 59b86d763a8471a21c8c16b304c1369728e3155f Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Wed, 7 Oct 2026 00:19:02 +0200 Subject: [PATCH 64/94] fix(zarr-metadata): keep a zero fill value where the v2 family takes it; narrow blosc clevel create_default with a dtype keeps fill_value 0 for the numeric families and takes null only where 0 is no fill value. blosc clevel is 0 to 9 again; only zlib and gzip take -1. A typestr size is ASCII digits only. fields_of places a v2 struct record's type at the struct's own location. The changelog says 21 of the numcodecs ids and the create_default change. Assisted-by: ClaudeCode:claude-fable-5-1 Co-Authored-By: Claude Fable 5.1 --- .../zarr-metadata/changes/v2-scopes.feature.md | 2 +- .../src/zarr_metadata/model/_array.py | 10 +++++++--- .../src/zarr_metadata/v2/_definition.py | 3 ++- .../src/zarr_metadata/v2/codec/compression.py | 2 +- .../src/zarr_metadata/v3/_definition.py | 4 ++-- .../tests/model/test_v2_scope_reads.py | 5 ++++- packages/zarr-metadata/tests/v2/test_codecs.py | 1 + .../zarr-metadata/tests/v2/test_data_types.py | 11 +++++++++++ packages/zarr-metadata/tests/v2/test_kinds.py | 15 ++++++++++++++- 9 files changed, 43 insertions(+), 10 deletions(-) diff --git a/packages/zarr-metadata/changes/v2-scopes.feature.md b/packages/zarr-metadata/changes/v2-scopes.feature.md index dbdfd719b2..d1b2b9c09d 100644 --- a/packages/zarr-metadata/changes/v2-scopes.feature.md +++ b/packages/zarr-metadata/changes/v2-scopes.feature.md @@ -1 +1 @@ -Zarr v2 array metadata is now read against a scope, as v3 metadata is. `zarr_metadata.v2.definition` defines the data types zarr-python 2.x writes (one definition per NumPy family) and the codecs numcodecs 0.16 configures (21 ids), files them in `CORE_V2`, and reads one `dtype` or codec with `resolve_dtype_v2` and `resolve_codec_v2`. `validate_array_metadata_v2` now reports a dtype string that is not a NumPy typestr or that its family does not take, a codec parameter outside what numcodecs accepts, and a fill value its dtype does not take; a dtype or codec id the package does not model is left unjudged as before. +Zarr v2 array metadata is now read against a scope, as v3 metadata is. `zarr_metadata.v2.definition` defines the data types zarr-python 2.x writes (one definition per NumPy family) and 21 of the codecs numcodecs 0.16 configures, files them in `CORE_V2`, and reads one `dtype` or codec with `resolve_dtype_v2` and `resolve_codec_v2`. `validate_array_metadata_v2` now reports a dtype string that is not a NumPy typestr or that its family does not take, a codec parameter outside what numcodecs accepts, and a fill value its dtype does not take; a dtype or codec id the package does not model is left unjudged as before. `ZarrV2ArrayMetadata.create_default` given a dtype whose family does not take `0` as a fill value, such as `|b1` or `|S3`, now sets `fill_value` to `null` unless one is given. diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_array.py b/packages/zarr-metadata/src/zarr_metadata/model/_array.py index fdd35f4e95..dcf37e3913 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_array.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/_array.py @@ -33,6 +33,7 @@ ) from zarr_metadata.v2.array import ZARR_V2_ARRAY_METADATA_STORE_KEY from zarr_metadata.v2.attributes import ZARR_V2_ATTRIBUTES_STORE_KEY +from zarr_metadata.v2.definition import resolve_dtype_v2 from zarr_metadata.v3._definition import ( ChunkGridDefinition, ChunkKeyEncodingDefinition, @@ -612,10 +613,13 @@ def create_default(cls, **overrides: Unpack[ZarrV2ArrayMetadataPartial]) -> Zarr filters=None, attributes=UNSET, ) - # `0` is a fill value of the integer families only: a dtype given - # without a fill value takes `null`, which every family takes. + # The default fill value, `0`, is a value of the numeric families + # only: a dtype of another family given without a fill value takes + # `null`, which every family takes. if "dtype" in overrides and "fill_value" not in overrides: - overrides["fill_value"] = None + dtype, _ = resolve_dtype_v2(overrides["dtype"]) + if len(fill_value_problems(dtype, 0)) != 0: + overrides["fill_value"] = None return default.update(**overrides) def __eq__(self, other: object) -> bool: diff --git a/packages/zarr-metadata/src/zarr_metadata/v2/_definition.py b/packages/zarr-metadata/src/zarr_metadata/v2/_definition.py index e40dd6c22e..e58d6d8563 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v2/_definition.py +++ b/packages/zarr-metadata/src/zarr_metadata/v2/_definition.py @@ -55,7 +55,8 @@ } """Each type code the v2 spec lists, and the family its definition is filed under.""" -TYPESTR_PATTERN: Final = re.compile(r"([<>|])([A-Za-z])(\d*)(?:\[(\d*)([A-Za-zμ]+)\])?") +# ASCII digits only: `\d` matches every Unicode digit, which NumPy does not read. +TYPESTR_PATTERN: Final = re.compile(r"([<>|])([A-Za-z])([0-9]*)(?:\[([0-9]*)([A-Za-z\u03bc]+)\])?") """A typestr: a byte order, a type code, a size in bytes if any, and a bracketed time unit with its multiplier if any.""" diff --git a/packages/zarr-metadata/src/zarr_metadata/v2/codec/compression.py b/packages/zarr-metadata/src/zarr_metadata/v2/codec/compression.py index dd43c41c2a..c873634ab1 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v2/codec/compression.py +++ b/packages/zarr-metadata/src/zarr_metadata/v2/codec/compression.py @@ -55,7 +55,7 @@ class ZarrV2BloscParameters(TypedDict, closed=True): """`numcodecs.Blosc(cname='lz4', clevel=5, shuffle=1, blocksize=0, typesize=None)`: shuffle -1 is automatic, 0 none, 1 byte, 2 bit.""" cname: NotRequired[ReadOnly[BloscCName]] - clevel: NotRequired[ReadOnly[ZarrV2CompressionLevel]] + clevel: NotRequired[ReadOnly[Annotated[int, Interval(ge=0, le=9)]]] shuffle: NotRequired[ReadOnly[Literal[-1, 0, 1, 2]]] blocksize: NotRequired[ReadOnly[Annotated[int, Ge(0)]]] typesize: NotRequired[ReadOnly[Annotated[int, Ge(1)] | None]] diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py b/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py index c6980f0678..af5b76d809 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py @@ -1425,13 +1425,13 @@ def fields_of(resolved: Resolved[Any], loc: Loc = ()) -> Iterator[tuple[Loc, Res """`resolved`, a field a scope read, where it sits, then each field it holds and theirs in turn, each where it sits. `loc` is where `resolved` sits; a field it holds sits in its - configuration, at `(*loc, "configuration", *place)`, as `resolve` + configuration, at the kind's `configuration_loc` and its place, as `resolve` locates its problems. What each holds is its `nested`: a field whose configuration was not checked holds none. """ yield loc, resolved for place, inner in resolved.nested.items(): - yield from fields_of(inner, (*loc, "configuration", *place)) + yield from fields_of(inner, (*resolved.read_as.configuration_loc(loc), *place)) def with_problems( diff --git a/packages/zarr-metadata/tests/model/test_v2_scope_reads.py b/packages/zarr-metadata/tests/model/test_v2_scope_reads.py index 2f0e3aac00..fbc0d74886 100644 --- a/packages/zarr-metadata/tests/model/test_v2_scope_reads.py +++ b/packages/zarr-metadata/tests/model/test_v2_scope_reads.py @@ -91,8 +91,11 @@ def test_error_what_the_scope_refuses_is_a_problem_of_the_document( def test_create_default_with_a_dtype_takes_no_fill_value() -> None: - """`create_default` given a dtype and no fill value builds a model with `null` for one, since `0` is no fill value of most families; a fill value given is kept.""" + """`create_default` given a dtype and no fill value keeps `0` when the family takes it, and takes `null` otherwise; a fill value given is kept.""" assert ZarrV2ArrayMetadata.create_default(dtype="|b1").fill_value is None + assert ZarrV2ArrayMetadata.create_default(dtype=" ({"id": ""}, ("c", "id"), "invalid_value"), ({"id": "bz2", "level": 0}, ("c", "level"), "invalid_value"), ({"id": "blosc", "cname": "brotli"}, ("c", "cname"), "invalid_value"), + ({"id": "blosc", "clevel": -1}, ("c", "clevel"), "invalid_value"), ({"id": "blosc", "shuffle": 3}, ("c", "shuffle"), "invalid_value"), ({"id": "blosc", "blocksize": -1}, ("c", "blocksize"), "invalid_value"), ({"id": "zstd", "level": 23}, ("c", "level"), "invalid_value"), diff --git a/packages/zarr-metadata/tests/v2/test_data_types.py b/packages/zarr-metadata/tests/v2/test_data_types.py index dc8c9df788..4deaa4d888 100644 --- a/packages/zarr-metadata/tests/v2/test_data_types.py +++ b/packages/zarr-metadata/tests/v2/test_data_types.py @@ -13,6 +13,7 @@ Unclaimed, canonical_fill_value, canonical_of, + fields_of, fill_value_problems, resolve, ) @@ -216,3 +217,13 @@ def test_every_family_has_an_example() -> None: "object", "struct", } + + +def test_a_struct_record_type_sits_where_the_document_writes_it() -> None: + """`fields_of` places a record's type under the struct's own configuration location, `("dtype", "fields", 0, 1)`, not under a `configuration` key a v2 document does not have.""" + resolved, _ = resolve([["a", " None: """A string without a byte order, a type code and a size in that order is not a typestr.""" From ef75bda15cf969e2fea33eea2bb8ce2b914c119f Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Thu, 8 Oct 2026 11:54:08 +0200 Subject: [PATCH 65/94] fix(zarr-metadata): refuse a v2 typestr that writes no size A typestr with a listed type code and no size, such as ' --- .../zarr-metadata/src/zarr_metadata/v2/_definition.py | 11 ++++++++++- .../zarr-metadata/tests/model/test_v2_scope_reads.py | 2 +- packages/zarr-metadata/tests/v2/test_data_types.py | 4 ++++ 3 files changed, 15 insertions(+), 2 deletions(-) diff --git a/packages/zarr-metadata/src/zarr_metadata/v2/_definition.py b/packages/zarr-metadata/src/zarr_metadata/v2/_definition.py index e58d6d8563..0e52bf6d65 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v2/_definition.py +++ b/packages/zarr-metadata/src/zarr_metadata/v2/_definition.py @@ -85,8 +85,17 @@ def parse_typestr(name: str) -> tuple[str, dict[str, JSONValue]] | None: def typestr_problem(name: str, at: Loc) -> ValidationProblem | None: """The problem `name`, at `at`, is when it is no typestr; None when it is one, or is `struct`.""" - if name == STRUCT_NAME or TYPESTR_PATTERN.fullmatch(name) is not None: + if name == STRUCT_NAME: return None + if TYPESTR_PATTERN.fullmatch(name) is not None: + if parse_typestr(name) is not None: + return None + # "An integer specifying the number of bytes": a listed code without one. + return ValidationProblem( + at, + f"expected a size in bytes after the type code, '' or '|', a type code and a size in " diff --git a/packages/zarr-metadata/tests/model/test_v2_scope_reads.py b/packages/zarr-metadata/tests/model/test_v2_scope_reads.py index fbc0d74886..58a708084c 100644 --- a/packages/zarr-metadata/tests/model/test_v2_scope_reads.py +++ b/packages/zarr-metadata/tests/model/test_v2_scope_reads.py @@ -90,7 +90,7 @@ def test_error_what_the_scope_refuses_is_a_problem_of_the_document( assert [(p.loc, p.kind) for p in problems] == [(at, kind)] -def test_create_default_with_a_dtype_takes_no_fill_value() -> None: +def test_create_default_keeps_zero_where_the_family_takes_it() -> None: """`create_default` given a dtype and no fill value keeps `0` when the family takes it, and takes `null` otherwise; a fill value given is kept.""" assert ZarrV2ArrayMetadata.create_default(dtype="|b1").fill_value is None assert ZarrV2ArrayMetadata.create_default(dtype=" None: ("value", "at"), [ ("float32", ("dtype",)), + (" None: ], ids=[ "no-typestr", + "no-size", + "no-size-bytes", "bool-size", "int-size", "int-order", From ec08ff6ae7fc91b0a5fd28117d51f357d877b24e Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Thu, 8 Oct 2026 11:54:53 +0200 Subject: [PATCH 66/94] docs(zarr-metadata): name the v2 scopes fragment after its PR Assisted-by: ClaudeCode:claude-fable-5-1 Co-Authored-By: Claude Fable 5.1 --- .../changes/{v2-scopes.feature.md => 396.feature.md} | 0 1 file changed, 0 insertions(+), 0 deletions(-) rename packages/zarr-metadata/changes/{v2-scopes.feature.md => 396.feature.md} (100%) diff --git a/packages/zarr-metadata/changes/v2-scopes.feature.md b/packages/zarr-metadata/changes/396.feature.md similarity index 100% rename from packages/zarr-metadata/changes/v2-scopes.feature.md rename to packages/zarr-metadata/changes/396.feature.md From a2ba363f6a67357039489beef4f8c3f4fa1ea2aa Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Thu, 8 Oct 2026 12:35:22 +0200 Subject: [PATCH 67/94] feat(zarr-metadata): read a v2 array document once, in a scope read_array_metadata_v2 returns a ZarrV2ArrayMetadataReading: dtype, compressor and filters as the scope read them, every problem, and the model when there is none. validate_array_metadata_v2, is_array_metadata_v2 and parse_array_metadata_v2 read in the scope given, CORE_V2 by default; the group readers take a scope too. Assisted-by: ClaudeCode:claude-fable-5-1 Co-Authored-By: Claude Fable 5.1 --- .../src/zarr_metadata/model/__init__.py | 4 + .../src/zarr_metadata/model/_array.py | 20 +- .../src/zarr_metadata/model/_validation.py | 187 ++++++++++++++---- .../model/test_read_array_metadata_v2.py | 123 ++++++++++++ 4 files changed, 297 insertions(+), 37 deletions(-) create mode 100644 packages/zarr-metadata/tests/model/test_read_array_metadata_v2.py diff --git a/packages/zarr-metadata/src/zarr_metadata/model/__init__.py b/packages/zarr-metadata/src/zarr_metadata/model/__init__.py index 78b852dbbc..8b6ac46978 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/__init__.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/__init__.py @@ -47,6 +47,7 @@ ZarrV2ArrayMetadataPartial, ZarrV3ArrayMetadata, ZarrV3ArrayMetadataUpdate, + read_array_metadata_v2, read_array_metadata_v3, ) from zarr_metadata.model._group import ( @@ -92,6 +93,7 @@ GROUP_METADATA_REQUIRED_KEYS_V2, GROUP_METADATA_REQUIRED_KEYS_V3, GROUP_METADATA_STANDARD_KEYS_V3, + ZarrV2ArrayMetadataReading, ZarrV3ArrayMetadataReading, is_array_metadata_v2, is_array_metadata_v3, @@ -170,6 +172,7 @@ "ValidationProblem", "ZarrV2ArrayMetadata", "ZarrV2ArrayMetadataPartial", + "ZarrV2ArrayMetadataReading", "ZarrV2ArrayMetadataStoreKey", "ZarrV2AttributesStoreKey", "ZarrV2ConsolidatedMetadata", @@ -215,6 +218,7 @@ "parse_metadata_field_v3", "parse_node_name_v3", "parse_node_path_v3", + "read_array_metadata_v2", "read_array_metadata_v3", "read_group_metadata_v3", "read_node_metadata_v3", diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_array.py b/packages/zarr-metadata/src/zarr_metadata/model/_array.py index dcf37e3913..d28dabfd9b 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_array.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/_array.py @@ -23,17 +23,19 @@ from zarr_metadata.model._validation import ( ArrayMembersV3, StoreKey, + ZarrV2ArrayMetadataReading, ZarrV3ArrayMetadataReading, construct, dimension_lengths, dump_store_json, load_store_json, parse_array_metadata_v2, + read_array_v2, read_array_v3, ) from zarr_metadata.v2.array import ZARR_V2_ARRAY_METADATA_STORE_KEY from zarr_metadata.v2.attributes import ZARR_V2_ATTRIBUTES_STORE_KEY -from zarr_metadata.v2.definition import resolve_dtype_v2 +from zarr_metadata.v2.definition import CORE_V2, resolve_dtype_v2 from zarr_metadata.v3._definition import ( ChunkGridDefinition, ChunkKeyEncodingDefinition, @@ -509,6 +511,22 @@ def read_array_metadata_v3( return model.reading +def read_array_metadata_v2( + value: object, *, context: Context | None = None +) -> ZarrV2ArrayMetadataReading: + """`value`, a v2 array document, as `context` read it, `CORE_V2` when none is given, whatever it holds. + + Everything a read finds, in one: the dtype, the compressor and each + filter as the scope read them -- `Read` by the definition that claims + the typestr or id, `Unclaimed`, or `Refused` -- every problem + `validate_array_metadata_v2` finds, and, when there is none, the + document's model. + """ + scope = CORE_V2 if context is None else context + reading, _ = read_array_v2(value, scope) + return reading + + class ZarrV2ArrayMetadataPartial(TypedDict, total=False): """ Partial form of the constructor-settable fields of `ZarrV2ArrayMetadata`. diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_validation.py b/packages/zarr-metadata/src/zarr_metadata/model/_validation.py index 60d44cdfd3..b39fd4c85e 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_validation.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/_validation.py @@ -47,7 +47,13 @@ from zarr_metadata._sentinel import UNSET from zarr_metadata._typed_json import typeddict_keys from zarr_metadata.v2.array import ZarrV2ArrayMetadataJSON -from zarr_metadata.v2.definition import resolve_codec_v2, resolve_dtype_v2 +from zarr_metadata.v2.definition import ( + CORE_V2, + ZarrV2CodecDefinition, + ZarrV2DataTypeDefinition, + resolve_codec_v2, + resolve_dtype_v2, +) from zarr_metadata.v2.group import ZarrV2GroupMetadataJSON from zarr_metadata.v3._definition import ( Chunk, @@ -74,7 +80,8 @@ from zarr_metadata._common import JSONValue from zarr_metadata._typed_json import Loc - from zarr_metadata.model._array import ZarrV3ArrayMetadata + from zarr_metadata.model._array import ZarrV2ArrayMetadata, ZarrV3ArrayMetadata + from zarr_metadata.v2.array import ZarrV2ArrayDimensionSeparator, ZarrV2ArrayOrder # The standard top-level keys of a v3 array metadata document. Anything outside # this set is an extension field. Built from the TypedDict's required/optional @@ -474,6 +481,55 @@ class ArrayMembersV3: extra_fields: dict[str, JSONValue] +@dataclass(frozen=True, slots=True) +class ArrayMembersV2: + """The members of a v2 array document a read found nothing wrong with, other than its fields, refined as the model holds them.""" + + shape: tuple[int, ...] + chunks: tuple[int, ...] + fill_value: JSONValue + order: ZarrV2ArrayOrder + dimension_separator: ZarrV2ArrayDimensionSeparator + attributes: dict[str, JSONValue] | UNSET + extra_fields: dict[str, JSONValue] + + +@dataclass(frozen=True, slots=True) +class ZarrV2ArrayMetadataReading: + """A v2 array document as a scope read it, whatever it holds: its dtype, compressor and filters, every problem, and the model when there is none. + + A field the document does not hold is `UNSET`; a `compressor` or + `filters` written as `null` is None. + """ + + dtype: Resolved[ZarrV2DataTypeDefinition[Any]] | UNSET = UNSET + """The dtype, as the scope read it.""" + compressor: Resolved[ZarrV2CodecDefinition[Any]] | UNSET | None = UNSET + """The compressor, as the scope read it; None when written as `null`.""" + filters: tuple[Resolved[ZarrV2CodecDefinition[Any]], ...] | UNSET | None = UNSET + """The filters, each as the scope read it; None when written as `null`.""" + problems: tuple[ValidationProblem, ...] = () + """Every reason the document is not a valid one.""" + metadata: ZarrV2ArrayMetadata | None = None + """The document's model, holding these fields, when there is no problem; None otherwise.""" + + def __reduce__(self) -> str | tuple[object, ...]: + # As the v3 reading: through its model, when it holds one. + if self.metadata is not None: + return (reading_of, (self.metadata,)) + return object.__reduce__(self) + + def fields(self) -> Iterator[tuple[Loc, Resolved[Any]]]: + """Each field the document holds, as the scope read it, where it sits: the dtype, a struct's record types after it, the compressor, each filter at its index.""" + if self.dtype is not UNSET: + yield from fields_of(cast("Resolved[Any]", self.dtype), ("dtype",)) + if self.compressor is not UNSET and self.compressor is not None: + yield from fields_of(self.compressor, ("compressor",)) + if self.filters is not UNSET and self.filters is not None: + for index, entry in enumerate(self.filters): + yield from fields_of(entry, ("filters", index)) + + def read_array_v3( value: object, context: Context, *, at: Loc = () ) -> tuple[ZarrV3ArrayMetadataReading, ArrayMembersV3 | None]: @@ -648,16 +704,19 @@ def parse_array_metadata_v3( return cast("ZarrV3ArrayMetadataJSON", arrays_to_tuples(value)) -def validate_array_metadata_v2(value: object) -> tuple[ValidationProblem, ...]: - """Return every reason `value` is not a structurally-valid v2 array doc. +def read_array_v2( + value: object, context: Context, *, at: Loc = () +) -> tuple[ZarrV2ArrayMetadataReading, ArrayMembersV2 | None]: + """`value`, a v2 array document, read once in `context`: the reading, and its other members when nothing is wrong. - `dtype`, `compressor` and `filters` are read in `CORE_V2`: a dtype or + `dtype`, `compressor` and `filters` are read in `context`: a dtype or codec the scope refuses is a problem, one it does not claim is not; `fill_value` is judged by the dtype the scope read. `compressor` and - `filters` are required keys that may be `None`. + `filters` are required keys that may be `None`. `at` is where the + document sits in the one handed in, which prefixes every problem. """ if not isinstance(value, Mapping): - return not_an_object(value) + return ZarrV2ArrayMetadataReading(problems=within(not_an_object(value), at)), None doc = cast("Mapping[object, object]", value) # Unlike the group document ("Other keys MUST NOT be present", # https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v2/v2.0.rst#L313), the v2 array document is open: other keys "SHOULD NOT be @@ -666,7 +725,8 @@ def validate_array_metadata_v2(value: object) -> tuple[ValidationProblem, ...]: # ARRAY_METADATA_STANDARD_KEYS_V2 is not a problem for being there. Ignored # is not unchecked: it is JSON, and its key a string, as in v3. problems: list[ValidationProblem] = list(missing_keys(ARRAY_METADATA_REQUIRED_KEYS_V2, doc)) - problems.extend(other_members_problems(doc, ARRAY_METADATA_STANDARD_KEYS_V2)) + extra_fields, found = other_members(doc, ARRAY_METADATA_STANDARD_KEYS_V2) + problems.extend(found) problems.extend(check_literal(doc, "zarr_format", 2)) shape, shape_problems = dimension_lengths(doc, "shape") chunks, chunks_problems = dimension_lengths(doc, "chunks") @@ -680,21 +740,25 @@ def validate_array_metadata_v2(value: object) -> tuple[ValidationProblem, ...]: "invalid_value", ) ) - dtype = None + dtype: Resolved[ZarrV2DataTypeDefinition[Any]] | UNSET = UNSET if "dtype" in doc: - # Read in CORE_V2: a typestr by its family, field records as a struct. - dtype, found = resolve_dtype_v2(doc["dtype"], loc=("dtype",)) + # A typestr by its family, field records as a struct. + dtype, found = resolve_dtype_v2(doc["dtype"], context, ("dtype",)) problems.extend(found) if "order" in doc and doc["order"] not in ("C", "F"): problems.append(outside_of(("order",), doc["order"], ("C", "F"))) + compressor: Resolved[ZarrV2CodecDefinition[Any]] | UNSET | None = UNSET if "compressor" in doc: - compressor = doc["compressor"] - if compressor is not None: - problems.extend(resolve_codec_v2(compressor, loc=("compressor",))[1]) + compressor = None + if doc["compressor"] is not None: + compressor, found = resolve_codec_v2(doc["compressor"], context, ("compressor",)) + problems.extend(found) + filters: tuple[Resolved[ZarrV2CodecDefinition[Any]], ...] | UNSET | None = UNSET if "filters" in doc: - filters = doc["filters"] - if filters is not None: - if not _is_array(filters): + filters = None + entries = doc["filters"] + if entries is not None: + if not _is_array(entries): problems.append( ValidationProblem( ("filters",), @@ -705,46 +769,91 @@ def validate_array_metadata_v2(value: object) -> tuple[ValidationProblem, ...]: else: # "A list of JSON objects providing codec configurations, or # null" (https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v2/v2.0.rst#L76-L79): an empty list is a list. - for index, item in enumerate(filters): - problems.extend(resolve_codec_v2(item, loc=("filters", index))[1]) + read: list[Resolved[ZarrV2CodecDefinition[Any]]] = [] + for index, item in enumerate(entries): + entry, found = resolve_codec_v2(item, context, ("filters", index)) + read.append(entry) + problems.extend(found) + filters = tuple(read) if "dimension_separator" in doc and doc["dimension_separator"] not in (".", "/"): problems.append( outside_of(("dimension_separator",), doc["dimension_separator"], (".", "/")) ) + fill_value: JSONValue = None if "fill_value" in doc: # JSON, judged by the dtype the scope read, when there is one: a # dtype nothing in scope claims leaves it unjudged. fill_value, found = refine_json(doc["fill_value"], ("fill_value",)) problems.extend(found) - if len(found) == 0 and dtype is not None: + if len(found) == 0 and dtype is not UNSET: problems.extend(fill_value_problems(dtype, fill_value, ("fill_value",))) + attributes: dict[str, JSONValue] | UNSET | None = UNSET if "attributes" in doc: - problems.extend(validate_attributes(doc["attributes"])) - return with_input(problems, doc) + attributes, found = attributes_of(doc["attributes"]) + problems.extend(found) + reading = ZarrV2ArrayMetadataReading( + dtype=dtype, + compressor=compressor, + filters=filters, + problems=within(with_input(problems, doc), at), + ) + if len(problems) != 0 or shape is None or chunks is None or attributes is None: + return reading, None + members = ArrayMembersV2( + shape=shape, + chunks=chunks, + fill_value=fill_value, + order=cast("ZarrV2ArrayOrder", doc["order"]), + dimension_separator=cast( + "ZarrV2ArrayDimensionSeparator", doc.get("dimension_separator", ".") + ), + attributes=attributes, + extra_fields=extra_fields, + ) + return reading, members -def is_array_metadata_v2(value: object) -> TypeGuard[ZarrV2ArrayMetadataJSON]: - """Whether `value` is a structurally-valid v2 array metadata document.""" +def validate_array_metadata_v2( + value: object, *, context: Context | None = None +) -> tuple[ValidationProblem, ...]: + """Every reason `value` is not a valid v2 array document, read in `context`, `CORE_V2` when none is given. + + `dtype`, `compressor` and `filters` are read in the scope: a dtype or + codec the scope refuses is a problem, one it does not claim is not; + `fill_value` is judged by the dtype the scope read. + """ + return read_array_v2(value, CORE_V2 if context is None else context)[0].problems + + +def is_array_metadata_v2( + value: object, *, context: Context | None = None +) -> TypeGuard[ZarrV2ArrayMetadataJSON]: + """Whether `value` is a valid v2 array metadata document, read in `context`, `CORE_V2` when none is given.""" return ( _is_canonical_json(value, finite=False) - and not validate_array_metadata_v2(value) + and not validate_array_metadata_v2(value, context=context) and _is_canonical_array_metadata_v2(value) ) -def parse_array_metadata_v2(value: object) -> ZarrV2ArrayMetadataJSON: - """Return `value` as `ZarrV2ArrayMetadataJSON`, or raise `MetadataValidationError`.""" - problems = validate_array_metadata_v2(value) +def parse_array_metadata_v2( + value: object, *, context: Context | None = None +) -> ZarrV2ArrayMetadataJSON: + """`value` as `ZarrV2ArrayMetadataJSON`, read in `context`, `CORE_V2` when none is given; `MetadataValidationError` with every problem.""" + problems = validate_array_metadata_v2(value, context=context) if len(problems) != 0: raise MetadataValidationError(problems) return cast("ZarrV2ArrayMetadataJSON", arrays_to_tuples(value)) -def validate_group_metadata_v2(value: object) -> tuple[ValidationProblem, ...]: +def validate_group_metadata_v2( + value: object, *, context: Context | None = None +) -> tuple[ValidationProblem, ...]: """Return every reason `value` is not a structurally-valid v2 group doc. Validates the in-memory merged form: the `.zgroup` fields plus an - optional `attributes` mapping folded in from `.zattrs`. + optional `attributes` mapping folded in from `.zattrs`. A group holds + no field a scope reads; `context` is taken as every v2 reader takes it. """ if not isinstance(value, Mapping): return not_an_object(value) @@ -757,14 +866,20 @@ def validate_group_metadata_v2(value: object) -> tuple[ValidationProblem, ...]: return with_input(problems, doc) -def is_group_metadata_v2(value: object) -> TypeGuard[ZarrV2GroupMetadataJSON]: - """Whether `value` is a structurally-valid v2 group metadata document.""" - return _is_canonical_json(value, finite=False) and not validate_group_metadata_v2(value) +def is_group_metadata_v2( + value: object, *, context: Context | None = None +) -> TypeGuard[ZarrV2GroupMetadataJSON]: + """Whether `value` is a structurally-valid v2 group metadata document; `context` is taken as every v2 reader takes it.""" + return _is_canonical_json(value, finite=False) and not validate_group_metadata_v2( + value, context=context + ) -def parse_group_metadata_v2(value: object) -> ZarrV2GroupMetadataJSON: - """Return `value` narrowed to `ZarrV2GroupMetadataJSON`, or raise `MetadataValidationError`.""" - problems = validate_group_metadata_v2(value) +def parse_group_metadata_v2( + value: object, *, context: Context | None = None +) -> ZarrV2GroupMetadataJSON: + """`value` narrowed to `ZarrV2GroupMetadataJSON`, or `MetadataValidationError`; `context` is taken as every v2 reader takes it.""" + problems = validate_group_metadata_v2(value, context=context) if len(problems) != 0: raise MetadataValidationError(problems) return cast(ZarrV2GroupMetadataJSON, arrays_to_tuples(value)) diff --git a/packages/zarr-metadata/tests/model/test_read_array_metadata_v2.py b/packages/zarr-metadata/tests/model/test_read_array_metadata_v2.py new file mode 100644 index 0000000000..1eac349756 --- /dev/null +++ b/packages/zarr-metadata/tests/model/test_read_array_metadata_v2.py @@ -0,0 +1,123 @@ +"""A v2 array document read once: each field as the scope read it, every problem, and the model.""" + +from __future__ import annotations + +from typing import Any + +import pytest + +from zarr_metadata._sentinel import UNSET +from zarr_metadata.model import ( + ZarrV2ArrayMetadata, + is_array_metadata_v2, + parse_array_metadata_v2, + read_array_metadata_v2, + validate_array_metadata_v2, +) +from zarr_metadata.v2.codec.compression import ZLIB_V2 +from zarr_metadata.v2.data_type.scalar import FLOAT_V2 +from zarr_metadata.v2.definition import ( + CORE_V2, + Context, + Read, + Refused, + Unclaimed, + ZarrV2CodecDefinition, +) +from zarr_metadata.v3.definition import EmptyConfiguration + +Loc = tuple[str | int, ...] +BASE: dict[str, Any] = dict(ZarrV2ArrayMetadata.create_default(shape=(4,)).to_json()) +PRIVATE = Context.of(FLOAT_V2, ZLIB_V2) + + +@pytest.mark.parametrize( + ("changes", "context", "dtype", "compressor", "filters", "locs"), + [ + ({}, None, Read, None, None, [("dtype",)]), + ( + { + "dtype": " None: + """`read_array_metadata_v2` reads the document once in the scope given (`CORE_V2` by default): `dtype`, `compressor` and `filters` are each `Read`, `Unclaimed` or None as written, `fields()` walks them in document order with a struct's record types after it, and a document with no problem has none.""" + reading = read_array_metadata_v2({**BASE, **changes}, context=context) + assert reading.problems == () + assert type(reading.dtype) is dtype + assert (None if reading.compressor is None else type(reading.compressor)) == compressor + assert reading.filters is not UNSET + found = None if reading.filters is None else tuple(type(f) for f in reading.filters) + assert found == filters + assert [loc for loc, _ in reading.fields()] == locs + + +@pytest.mark.parametrize( + ("value", "problems"), + [ + ({**BASE, "dtype": "float32"}, [(("dtype",), "invalid_value")]), + ( + {**BASE, "compressor": {"id": "zlib", "level": 10}}, + [(("compressor", "level"), "invalid_value")], + ), + ({**BASE, "fill_value": "NaN"}, [(("fill_value",), "invalid_type")]), + (3, [((), "invalid_type")]), + ], + ids=["dtype", "compressor", "fill", "not-an-object"], +) +def test_error_a_reading_reports_what_the_validator_reports( + value: object, problems: list[tuple[Loc, str]] +) -> None: + """A reading of a document with a problem holds every problem `validate_array_metadata_v2` finds, a `Refused` field where one was refused, and no model.""" + reading = read_array_metadata_v2(value) + assert [(p.loc, p.kind) for p in reading.problems] == problems + assert reading.metadata is None + assert [(p.loc, p.kind) for p in validate_array_metadata_v2(value)] == problems + if isinstance(value, dict) and "dtype" in problems[0][0]: + assert isinstance(reading.dtype, Refused) + + +def test_the_v2_readers_read_in_the_scope_given() -> None: + """`validate_array_metadata_v2`, `is_array_metadata_v2` and `parse_array_metadata_v2` take a scope: in one that does not claim `|u1`, the default document still validates (an unclaimed dtype is left unjudged), and in one where a private `zlib` takes no level, a written level is a problem.""" + assert validate_array_metadata_v2(BASE, context=PRIVATE) == () + assert is_array_metadata_v2(BASE, context=PRIVATE) + assert parse_array_metadata_v2(BASE, context=PRIVATE)["dtype"] == "|u1" + bare = ZarrV2CodecDefinition(name="zlib", configuration=EmptyConfiguration) + doc = {**BASE, "compressor": {"id": "zlib", "level": 1}} + assert validate_array_metadata_v2(doc) == () + found = validate_array_metadata_v2(doc, context=CORE_V2.extended_with(bare)) + assert [p.loc for p in found] == [("compressor", "level")] From 5c756670e2993807b60844cc7b33d52dccec0017 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Thu, 8 Oct 2026 12:44:38 +0200 Subject: [PATCH 68/94] feat(zarr-metadata)!: make the v2 array model a pair of document and scope ZarrV2ArrayMetadata(document, context=None) reads the document in a scope, CORE_V2 by default. dtype, compressor and filters are the fields as the scope read them. Equality is by what the document means. update reads new members in the model's own scope; with_context and refined_in read the document in another; refines orders models. ZarrV2ArrayMetadataUpdate replaces ZarrV2ArrayMetadataPartial. A member the spec does not define is kept. Assisted-by: ClaudeCode:claude-fable-5-1 Co-Authored-By: Claude Fable 5.1 --- .../src/zarr_metadata/__init__.py | 4 +- .../src/zarr_metadata/model/__init__.py | 4 +- .../src/zarr_metadata/model/_array.py | 514 ++++++++++++------ .../src/zarr_metadata/v2/definition.py | 18 + .../zarr-metadata/tests/model/test_array.py | 45 +- .../tests/model/test_construction.py | 42 +- .../zarr-metadata/tests/model/test_pair_v2.py | 241 ++++++++ .../tests/model/test_pydantic_module.py | 6 +- .../zarr-metadata/tests/test_public_api.py | 2 +- 9 files changed, 659 insertions(+), 217 deletions(-) create mode 100644 packages/zarr-metadata/tests/model/test_pair_v2.py diff --git a/packages/zarr-metadata/src/zarr_metadata/__init__.py b/packages/zarr-metadata/src/zarr_metadata/__init__.py index ba041a3b70..824911aadb 100644 --- a/packages/zarr-metadata/src/zarr_metadata/__init__.py +++ b/packages/zarr-metadata/src/zarr_metadata/__init__.py @@ -14,8 +14,8 @@ ProblemKind, ValidationProblem, ZarrV2ArrayMetadata, - ZarrV2ArrayMetadataPartial, ZarrV2ArrayMetadataStoreKey, + ZarrV2ArrayMetadataUpdate, ZarrV2AttributesStoreKey, ZarrV2ConsolidatedMetadata, ZarrV2ConsolidatedMetadataStoreKey, @@ -362,8 +362,8 @@ "ZarrV2ArrayMetadata", "ZarrV2ArrayMetadataJSON", "ZarrV2ArrayMetadataJSONPartial", - "ZarrV2ArrayMetadataPartial", "ZarrV2ArrayMetadataStoreKey", + "ZarrV2ArrayMetadataUpdate", "ZarrV2ArrayOrder", "ZarrV2AttributesStoreKey", "ZarrV2CodecMetadata", diff --git a/packages/zarr-metadata/src/zarr_metadata/model/__init__.py b/packages/zarr-metadata/src/zarr_metadata/model/__init__.py index 8b6ac46978..35e88001f4 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/__init__.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/__init__.py @@ -44,7 +44,7 @@ from zarr_metadata._sentinel import UNSET from zarr_metadata.model._array import ( ZarrV2ArrayMetadata, - ZarrV2ArrayMetadataPartial, + ZarrV2ArrayMetadataUpdate, ZarrV3ArrayMetadata, ZarrV3ArrayMetadataUpdate, read_array_metadata_v2, @@ -171,9 +171,9 @@ "RepairKind", "ValidationProblem", "ZarrV2ArrayMetadata", - "ZarrV2ArrayMetadataPartial", "ZarrV2ArrayMetadataReading", "ZarrV2ArrayMetadataStoreKey", + "ZarrV2ArrayMetadataUpdate", "ZarrV2AttributesStoreKey", "ZarrV2ConsolidatedMetadata", "ZarrV2ConsolidatedMetadataStoreKey", diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_array.py b/packages/zarr-metadata/src/zarr_metadata/model/_array.py index d28dabfd9b..01218ac22a 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_array.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/_array.py @@ -4,12 +4,14 @@ import dataclasses from collections.abc import Mapping -from dataclasses import dataclass, field from types import MappingProxyType -from typing import TYPE_CHECKING, Any, Final, Literal, cast +from typing import TYPE_CHECKING, Any, Final, cast from typing_extensions import TypedDict, Unpack +from zarr_metadata._common import ( + JSONValue, +) from zarr_metadata._json import ( MetadataValidationError, ValidationProblem, @@ -21,15 +23,14 @@ ) from zarr_metadata._sentinel import UNSET from zarr_metadata.model._validation import ( + ArrayMembersV2, ArrayMembersV3, StoreKey, ZarrV2ArrayMetadataReading, ZarrV3ArrayMetadataReading, - construct, dimension_lengths, dump_store_json, load_store_json, - parse_array_metadata_v2, read_array_v2, read_array_v3, ) @@ -56,8 +57,8 @@ if TYPE_CHECKING: from collections.abc import Iterable, Sequence - from zarr_metadata._common import JSONValue from zarr_metadata._typed_json import Loc + from zarr_metadata.v2._definition import ZarrV2CodecDefinition, ZarrV2DataTypeDefinition from zarr_metadata.v2.array import ( ZarrV2ArrayDimensionSeparator, ZarrV2ArrayMetadataJSON, @@ -523,21 +524,22 @@ def read_array_metadata_v2( document's model. """ scope = CORE_V2 if context is None else context - reading, _ = read_array_v2(value, scope) - return reading + reading, members = read_array_v2(value, scope) + if members is None: + return reading + refined, _ = refine_user_data(value) + document = cast("dict[str, JSONValue]", refined) + if "dimension_separator" not in document: + document = {**document, "dimension_separator": "."} + model = ZarrV2ArrayMetadata._of(document, scope, reading, members) # pyright: ignore[reportPrivateUsage] + return model.reading -class ZarrV2ArrayMetadataPartial(TypedDict, total=False): - """ - Partial form of the constructor-settable fields of `ZarrV2ArrayMetadata`. +class ZarrV2ArrayMetadataUpdate(TypedDict, total=False, extra_items=JSONValue | UNSET): + """The members `ZarrV2ArrayMetadata.update` puts in place: each as a document writes it, or `UNSET` to leave out one a document may leave out. - Every key is optional and typed with the model's own value types, so it - describes valid keyword arguments to `ZarrV2ArrayMetadata.update` and - `create_default`. The `init=False` field `zarr_format` is intentionally - excluded, since it cannot be passed to `dataclasses.replace`. - - Drift between this type and the model's settable fields is prevented by - `tests/model/test_array.py::test_v2_partial_keys_match_settable_model_fields`. + Those are `attributes` (no `.zattrs`), `dimension_separator` (read as + `"."`), and a member the spec does not define. """ shape: tuple[int, ...] @@ -547,163 +549,311 @@ class ZarrV2ArrayMetadataPartial(TypedDict, total=False): order: ZarrV2ArrayOrder compressor: ZarrV2CodecMetadata | None filters: tuple[ZarrV2CodecMetadata, ...] | None - dimension_separator: ZarrV2ArrayDimensionSeparator - attributes: dict[str, JSONValue] | UNSET + dimension_separator: ZarrV2ArrayDimensionSeparator | UNSET + attributes: Mapping[str, JSONValue] | UNSET -@dataclass(frozen=True, slots=True, kw_only=True) class ZarrV2ArrayMetadata: - """In-memory model of a v2 array metadata document. - - A canonical, lossless representation of the `.zarray` content plus the - sibling `.zattrs` attributes. `dtype`, `compressor`, and `filters` are - held in their raw JSON forms and are never interpreted; `fill_value` is - held verbatim in its JSON form. `attributes` is `UNSET` when no - `.zattrs` file (or merged `attributes` key) exists — distinct from an - explicit empty `.zattrs`, which is `{}` and round-trips as a file. One - spelling normalization: a `.zarray` that omits `dimension_separator` - means `"."` by the v2 convention, and the model holds and re-emits that - value explicitly. A model checks itself when it is built, as the v3 - models do: its document has no problem `validate_array_metadata_v2` - finds, or the constructor raises `MetadataValidationError`, so - `update` refuses a change that would make one. + """A v2 array document, and the scope it was read in. + + The pair, as the v3 models are: `to_json` is the merged document -- + the `.zarray` members, and `attributes` when a `.zattrs` holds them -- + as written, refined, with one spelling put in: a `.zarray` that omits + `dimension_separator` means `"."` by the v2 convention + (https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v2/v2.0.rst#L81-L86), + which the model holds and writes. `dtype`, `compressor` and each + filter are views of the reading: `Read` by the definition in scope + that claims the typestr or id, or `Unclaimed`. `attributes` is + `UNSET` when no `.zattrs` exists, distinct from an empty one. A + member the spec does not define is kept, in `extra_fields`. Built + only by reading: the constructor reads `document` in `context` and + raises `MetadataValidationError` with every problem, so no model is + invalid. Two models are equal when their documents mean the same in + their scopes, as `array_key_v2` says. `update` reads new members in + the model's own scope; `with_context` and `refined_in` read the + document in another. A model pickles as its pair. """ - zarr_format: Literal[2] = field(default=2, init=False) - shape: tuple[int, ...] - dtype: ZarrV2DataTypeMetadata - chunks: tuple[int, ...] - fill_value: JSONValue - order: ZarrV2ArrayOrder - compressor: ZarrV2CodecMetadata | None - filters: tuple[ZarrV2CodecMetadata, ...] | None - # "." is the v2 convention's default for an ABSENT dimension_separator key; - # from_json normalizes absence to it (a semantics-preserving spelling - # normalization, like the v3 bare-string metadata-field form). The value - # is never None: the document grammar has no null spelling for this field. - dimension_separator: ZarrV2ArrayDimensionSeparator = field(default=".") - attributes: dict[str, JSONValue] | UNSET - - def __post_init__(self) -> None: - # Held as a read refines them, in containers of its own. - members = _v2_array_members(parse_array_metadata_v2(self.to_json())) - for name, value in members.items(): - object.__setattr__(self, name, value) - - def update(self, **kwargs: Unpack[ZarrV2ArrayMetadataPartial]) -> ZarrV2ArrayMetadata: - """ - Return a new `ZarrV2ArrayMetadata` with the given fields updated. + __slots__ = ("_claims", "_context", "_document", "_key", "_members", "_reading", "_shown") - Only the constructor-settable fields listed in - `ZarrV2ArrayMetadataPartial` can be updated; the fixed `zarr_format` is - rejected at the type level. Each given field fully replaces its previous - value. `MetadataValidationError` when the document the change makes - has a problem, as the model checks itself when it is built. - """ - return dataclasses.replace(self, **kwargs) + zarr_format: Final = 2 + + def __init__(self, document: object, context: Context | None = None) -> None: + scope = CORE_V2 if context is None else context + reading, members = read_array_v2(document, scope) + if members is None: + raise MetadataValidationError(reading.problems) + refined, _ = refine_user_data(document) + held = cast("dict[str, JSONValue]", refined) + if "dimension_separator" not in held: + held = {**held, "dimension_separator": "."} + self._adopt(held, scope, reading, members) @classmethod - def create_default(cls, **overrides: Unpack[ZarrV2ArrayMetadataPartial]) -> ZarrV2ArrayMetadata: - """ - Create a default (empty) v2 array metadata model, with optional overrides. - - The default is a structurally-valid scalar `uint8` (`"|u1"`) array — the - array analog of `list()` returning `[]`. Any field can be overridden by - keyword (the same fields accepted by `update`). Overriding `shape` - without `chunks` derives `chunks` equal to `shape` (one chunk covering - the array). - - The derivation is deliberately one-way, matching the v3 model: - overriding `chunks` without `shape` keeps the scalar default - `shape=()`, which `chunks` of any other rank do not fit, so - `MetadataValidationError`, as the v3 model refuses a grid its - default shape does not take. - """ - if "shape" in overrides and "chunks" not in overrides: - overrides["chunks"] = tuple(overrides["shape"]) - default = cls( - shape=(), - dtype="|u1", - chunks=(), - fill_value=0, - order="C", - compressor=None, - filters=None, - attributes=UNSET, + def _of( + cls, + document: dict[str, JSONValue], + context: Context, + reading: ZarrV2ArrayMetadataReading, + members: ArrayMembersV2, + ) -> ZarrV2ArrayMetadata: + """A model of a document a read found nothing wrong with, holding that reading: no second read.""" + model = object.__new__(cls) + model._adopt(document, context, reading, members) + return model + + def _adopt( + self, + document: dict[str, JSONValue], + context: Context, + reading: ZarrV2ArrayMetadataReading, + members: ArrayMembersV2, + ) -> None: + self._document = document + self._context = context + self._reading = dataclasses.replace(reading, metadata=self) + self._members = members + # What the model shows of its members, read-only at every level. + self._shown = ( + frozen(members.fill_value), + UNSET if members.attributes is UNSET else frozen(members.attributes), + frozen(cast("JSONValue", members.extra_fields)), ) - # The default fill value, `0`, is a value of the numeric families - # only: a dtype of another family given without a fill value takes - # `null`, which every family takes. - if "dtype" in overrides and "fill_value" not in overrides: - dtype, _ = resolve_dtype_v2(overrides["dtype"]) - if len(fill_value_problems(dtype, 0)) != 0: - overrides["fill_value"] = None - return default.update(**overrides) + self._key = array_key_v2(self) + self._claims = MappingProxyType(claims_of(reading.fields())) - def __eq__(self, other: object) -> bool: - """Whether `other` models the same array: the same document, as JSON text. + # --- the pair --------------------------------------------------------- + + @property + def context(self) -> Context: + """The scope the document was read in, which `update` reads new members in.""" + return self._context - Nothing in a v2 document is interpreted, so two models are one when - their documents are written alike, which tells `0` from `0.0` and - `-0.0`, and takes `NaN` for itself. Equal models hash alike. + @property + def reading(self) -> ZarrV2ArrayMetadataReading: + """The document as the scope read it: the dtype, the compressor, each filter.""" + return self._reading + + @property + def claims(self) -> Claims: + """What the reading claimed of each typestr and codec id the document writes, keyed as the scope files them.""" + return self._claims + + def to_json(self) -> ZarrV2ArrayMetadataJSON: + """The merged document as written, refined, sharing nothing with the model. + + `attributes` is included when set, even empty. This is not the + on-disk `.zarray`, which excludes them: `to_key_value` splits the + document as a store holds it + (https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v2/v2.0.rst#L323-L330). """ + return cast("ZarrV2ArrayMetadataJSON", copied(self._document)) + + def to_key_value( + self, *, indent: int | str | None = None + ) -> Mapping[ZarrV2ArrayMetadataStoreKey | ZarrV2AttributesStoreKey, bytes]: + """The document as a store holds it: `.zarray` without the attributes, and `.zattrs` with them when they are set, even empty.""" + zarray = {key: value for key, value in self._document.items() if key != "attributes"} + out: dict[ZarrV2ArrayMetadataStoreKey | ZarrV2AttributesStoreKey, bytes] = { + ZARR_V2_ARRAY_METADATA_STORE_KEY: dump_store_json(zarray, indent=indent) + } + if "attributes" in self._document: + out[ZARR_V2_ATTRIBUTES_STORE_KEY] = dump_store_json( + self._document["attributes"], indent=indent + ) + return out + + def __repr__(self) -> str: + return f"{type(self).__name__}({self._document!r}, context={self._context!r})" + + def __eq__(self, other: object) -> bool: if type(other) is not type(self): return NotImplemented - return json_text(self.to_json()) == json_text(cast("ZarrV2ArrayMetadata", other).to_json()) + return self._key == cast("ZarrV2ArrayMetadata", other)._key def __hash__(self) -> int: - return hash(json_text(self.to_json())) + return hash(self._key) - def to_json(self) -> ZarrV2ArrayMetadataJSON: - """Return the merged in-memory document form. + def __reduce__(self) -> tuple[type[ZarrV2ArrayMetadata], tuple[object, Context]]: + return type(self), (self._document, self._context) - `attributes` is included when set (even empty). This is not the - on-disk `.zarray` content: a conforming `.zarray` must exclude - `attributes` (they live in the sibling `.zattrs` file). Use - `to_key_value` to produce the spec-conforming split for storage - (https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v2/v2.0.rst#L323-L330). + # --- typed views ------------------------------------------------------ + + @property + def shape(self) -> tuple[int, ...]: + """The array's shape.""" + return self._members.shape + + @property + def chunks(self) -> tuple[int, ...]: + """The shape of each chunk.""" + return self._members.chunks + + @property + def fill_value(self) -> JSONValue: + """The fill value as written, refined; read-only at every level.""" + return self._shown[0] + + @property + def order(self) -> ZarrV2ArrayOrder: + """The in-chunk layout, `"C"` or `"F"`.""" + return self._members.order + + @property + def dimension_separator(self) -> ZarrV2ArrayDimensionSeparator: + """What joins the chunk indices in a key: `"."` when the document writes none.""" + return self._members.dimension_separator + + @property + def attributes(self) -> Mapping[str, JSONValue] | UNSET: + """The user attributes a `.zattrs` holds, read-only at every level; `UNSET` when there is no `.zattrs`.""" + return cast("Mapping[str, JSONValue] | UNSET", self._shown[1]) + + @property + def extra_fields(self) -> Mapping[str, JSONValue]: + """Every member the spec does not define, as written, read-only at every level.""" + return cast("Mapping[str, JSONValue]", self._shown[2]) + + @property + def dtype(self) -> Read[ZarrV2DataTypeDefinition[Any]] | Unclaimed: + """The dtype as the scope read it: by its family's definition, or unclaimed.""" + return cast("Read[ZarrV2DataTypeDefinition[Any]] | Unclaimed", self._reading.dtype) + + @property + def compressor(self) -> Read[ZarrV2CodecDefinition[Any]] | Unclaimed | None: + """The compressor as the scope read it; None when written as `null`.""" + return cast("Read[ZarrV2CodecDefinition[Any]] | Unclaimed | None", self._reading.compressor) + + @property + def filters(self) -> tuple[Read[ZarrV2CodecDefinition[Any]] | Unclaimed, ...] | None: + """The filters, each as the scope read it; None when written as `null`.""" + return cast( + "tuple[Read[ZarrV2CodecDefinition[Any]] | Unclaimed, ...] | None", + self._reading.filters, + ) + + # --- changing --------------------------------------------------------- + + def update(self, **members: Unpack[ZarrV2ArrayMetadataUpdate]) -> ZarrV2ArrayMetadata: + """This model with `members`, JSON, in place of the document's, `UNSET` leaving one out, read in this model's own scope. + + `MetadataValidationError` when the document they make has a + problem, so members that go together are passed together: a + `dtype` with a fill value of it. + """ + document: dict[str, object] = {**self._document, **members} + for key, value in members.items(): + if value is UNSET: + del document[key] + return type(self)(document, context=self._context) + + def with_context(self, context: Context | None = None) -> ZarrV2ArrayMetadata: + """This document read in `context`, whatever that changes: a gain, a loss, a conflict. + + `MetadataValidationError` when the document has a problem there. + The reading is kept when `context` reads every claim identically. """ - # to_json output shares no mutable state with the model: the document - # is copied whole, one frame for each level of nesting. - out: ZarrV2ArrayMetadataJSON = { - "zarr_format": self.zarr_format, - "shape": self.shape, - "dtype": self.dtype, - "order": self.order, - "chunks": self.chunks, - "fill_value": self.fill_value, - "dimension_separator": self.dimension_separator, - "compressor": self.compressor, - "filters": self.filters, + scope = CORE_V2 if context is None else context + if scope.disagreements(self._claims).agrees: + return self._of(self._document, scope, self._reading, self._members) + return type(self)(self._document, context=scope) + + def refined_in(self, context: Context | None = None) -> ZarrV2ArrayMetadata: + """This document read in `context`, which may claim what this scope left unclaimed and contradict nothing. + + `ScopeConflictError` naming each typestr or id `context` reads by + another definition, or by none, where this scope read it by one, + and where each sits in the document. `MetadataValidationError` + when a definition `context` claims refuses what was written. + """ + scope = CORE_V2 if context is None else context + found = scope.disagreements(self._claims) + if len(found.conflicts) != 0: + raise ScopeConflictError(located_conflicts(self._reading.fields(), found.conflicts)) + return self.with_context(scope) + + def refines(self, other: ZarrV2ArrayMetadata) -> bool: + """Whether this model holds everything `other` holds: each field refines its counterpart, a `null` compressor or filters only a `null`, and every other member is the same, the fill value as the more informed dtype spells it; a fill value that dtype refuses is no refinement.""" + if type(other) is not type(self): + return False + if (self.compressor is None) != (other.compressor is None): + return False + if (self.filters is None) != (other.filters is None): + return False + mine = () if self.filters is None else self.filters + theirs = () if other.filters is None else other.filters + if len(mine) != len(theirs): + return False + pairs = [(self.dtype, other.dtype), *zip(mine, theirs, strict=True)] + if self.compressor is not None and other.compressor is not None: + pairs.append((self.compressor, other.compressor)) + if not all(refines_field(one, another) for one, another in pairs): + return False + if len(fill_value_problems(self.dtype, other._members.fill_value)) != 0: + return False + return _plain_key_v2(self, self.dtype) == _plain_key_v2(other, self.dtype) + + # --- constructors ----------------------------------------------------- + + @classmethod + def create_default( + cls, *, context: Context | None = None, **overrides: Unpack[ZarrV2ArrayMetadataUpdate] + ) -> ZarrV2ArrayMetadata: + """A scalar `|u1` array, or the one `overrides`, members of its document, make of it, read in `context`. + + `MetadataValidationError` when the document they make has a + problem. Overriding `shape` without `chunks` derives `chunks` + equal to `shape`, one chunk covering the array; overriding `chunks` + without `shape` keeps the scalar default shape, which chunks of + another rank do not fit. A dtype given without a fill value takes + `0` when its family takes it, and `null` otherwise, which every + family takes. + """ + document: dict[str, object] = { + "zarr_format": 2, + "shape": (), + "chunks": (), + "dtype": "|u1", + "fill_value": 0, + "order": "C", + "compressor": None, + "filters": None, + "dimension_separator": ".", } - if self.attributes is not UNSET: - out["attributes"] = self.attributes - return cast("ZarrV2ArrayMetadataJSON", copied(cast("JSONValue", out))) + given: dict[str, object] = dict(overrides) + if "shape" in given and "chunks" not in given: + lengths, _ = dimension_lengths(cast("Mapping[object, object]", given), "shape") + if lengths is not None: + given["chunks"] = lengths + if "dtype" in given and "fill_value" not in given: + dtype, _ = resolve_dtype_v2(given["dtype"], context) + if len(fill_value_problems(dtype, 0)) != 0: + given["fill_value"] = None + merged = {key: value for key, value in {**document, **given}.items() if value is not UNSET} + return cls(merged, context=context) @classmethod - def from_json(cls, data: object) -> ZarrV2ArrayMetadata: - """The model of `data`, a v2 array document with its attributes under `attributes`. + def from_json(cls, data: object, *, context: Context | None = None) -> ZarrV2ArrayMetadata: + """The model of `data`, a v2 array document with its attributes under `attributes`, read in `context`. - `MetadataValidationError` with every problem `validate_array_metadata_v2` - finds. A missing `dimension_separator` is read as `"."`, which is - written back. The model shares no mutable state with `data`. + `MetadataValidationError` with every problem the read finds. + `read_array_metadata_v2` gives the reading this model is built + from, and the problems of a document with some. """ - # A read model shares no mutable state with what it read. - parsed = cast( - "ZarrV2ArrayMetadataJSON", copied(cast("JSONValue", parse_array_metadata_v2(data))) - ) - return construct(cls, **_v2_array_members(parsed)) + return cls(data, context=context) @classmethod - def from_key_value(cls, mapping: Mapping[StoreKey, bytes]) -> ZarrV2ArrayMetadata: - """The model of the array at `.zarray` in `mapping`, with the attributes at `.zattrs` when there is one. + def from_key_value( + cls, mapping: Mapping[StoreKey, bytes], *, context: Context | None = None + ) -> ZarrV2ArrayMetadata: + """The model of the array at `.zarray` in `mapping`, with the attributes at `.zattrs` when there is one, read in `context`. `MetadataValidationError` when `.zarray` is missing, bytes are not JSON, `.zarray` holds `attributes`, or the document is not valid. """ zarray_raw = load_store_json(mapping, ZARR_V2_ARRAY_METADATA_STORE_KEY) if not isinstance(zarray_raw, Mapping): - return cls.from_json(zarray_raw) + return cls(zarray_raw, context=context) zarray = cast("Mapping[str, object]", zarray_raw) if "attributes" in zarray: refused = ValidationProblem( @@ -712,42 +862,52 @@ def from_key_value(cls, mapping: Mapping[StoreKey, bytes]) -> ZarrV2ArrayMetadat raise MetadataValidationError(with_input((refused,), zarray)) if ZARR_V2_ATTRIBUTES_STORE_KEY in mapping: zattrs = load_store_json(mapping, ZARR_V2_ATTRIBUTES_STORE_KEY) - return cls.from_json({**zarray, "attributes": zattrs}) - return cls.from_json(zarray) + return cls({**zarray, "attributes": zattrs}, context=context) + return cls(zarray, context=context) - def to_key_value( - self, *, indent: int | str | None = None - ) -> Mapping[ZarrV2ArrayMetadataStoreKey | ZarrV2AttributesStoreKey, bytes]: - """The document as a store holds it: `.zarray` without the attributes, and `.zattrs` with them when they are set, even empty. - A model was checked when it was built, so its document is written - as it is. - """ - # Attributes live only in the sibling `.zattrs` file; the `.zarray` - # document must exclude them. The `.zattrs` key is present exactly - # when attributes are set (even empty) — UNSET emits no file. - document = self.to_json() - zarray = {k: v for k, v in document.items() if k != "attributes"} - out: dict[ZarrV2ArrayMetadataStoreKey | ZarrV2AttributesStoreKey, bytes] = { - ZARR_V2_ARRAY_METADATA_STORE_KEY: dump_store_json(zarray, indent=indent) - } - if "attributes" in document: - out[ZARR_V2_ATTRIBUTES_STORE_KEY] = dump_store_json( - document["attributes"], indent=indent - ) - return out +def array_key_v2(model: ZarrV2ArrayMetadata) -> tuple[object, ...]: + """What `==` and `hash` compare of a v2 array model: what its document means. + Each field by its `field_key`, the fill value in its canonical spelling + as JSON text when a definition in scope read the dtype, and every other + member as it is, the JSON ones as text; `attributes` as `UNSET` when + there is no `.zattrs`. + """ + members = model._members # pyright: ignore[reportPrivateUsage] + return ( + members.shape, + members.chunks, + members.order, + members.dimension_separator, + _fill_value_key_v2(model), + field_key(model.dtype), + None if model.compressor is None else field_key(model.compressor), + None if model.filters is None else tuple(field_key(entry) for entry in model.filters), + UNSET if members.attributes is UNSET else json_text(members.attributes), + json_text(members.extra_fields), + ) -def _v2_array_members(document: ZarrV2ArrayMetadataJSON) -> dict[str, object]: - """The members of the v2 array model of `document`, which `parse_array_metadata_v2` gave: a missing `dimension_separator` is `"."`.""" - return { - "shape": document["shape"], - "dtype": document["dtype"], - "chunks": document["chunks"], - "fill_value": document["fill_value"], - "order": document["order"], - "compressor": document["compressor"], - "filters": document["filters"], - "dimension_separator": document.get("dimension_separator", "."), - "attributes": dict(document["attributes"]) if "attributes" in document else UNSET, - } + +def _plain_key_v2( + model: ZarrV2ArrayMetadata, dtype: Read[ZarrV2DataTypeDefinition[Any]] | Unclaimed +) -> tuple[object, ...]: + """What `refines` compares of a model other than its fields, the fill value spelled as `dtype` -- the more informed side's -- spells it.""" + members = model._members # pyright: ignore[reportPrivateUsage] + return ( + members.shape, + members.chunks, + members.order, + members.dimension_separator, + json_text(spelled_canonically(dtype, members.fill_value)), + UNSET if members.attributes is UNSET else json_text(members.attributes), + json_text(members.extra_fields), + ) + + +def _fill_value_key_v2(model: ZarrV2ArrayMetadata) -> str: + """What `==` compares of `model`'s fill value: its canonical spelling as JSON text when a definition in scope read the dtype, and the fill value as written when none did.""" + fill_value = model._members.fill_value # pyright: ignore[reportPrivateUsage] + if isinstance(model.dtype, Read): + return json_text(spelled_canonically(model.dtype, fill_value)) + return json_text(fill_value) diff --git a/packages/zarr-metadata/src/zarr_metadata/v2/definition.py b/packages/zarr-metadata/src/zarr_metadata/v2/definition.py index 56d27ae31a..8b4bddd0c2 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v2/definition.py +++ b/packages/zarr-metadata/src/zarr_metadata/v2/definition.py @@ -40,6 +40,16 @@ resolve, ) from zarr_metadata.v3._registry import Context +from zarr_metadata.v3._scope import ( + ClaimKey, + Claims, + Conflict, + Disagreements, + ScopeConflictError, + claim_key, + claims_of, + refines, +) if TYPE_CHECKING: from zarr_metadata._typed_json import Loc @@ -67,17 +77,25 @@ def resolve_codec_v2( "CORE_V2", "V2_CODECS", "V2_DATA_TYPES", + "ClaimKey", + "Claims", + "Conflict", "Context", + "Disagreements", "Read", "Refused", "Resolved", + "ScopeConflictError", "Unclaimed", "ZarrV2CodecDefinition", "ZarrV2DataTypeDefinition", "ZarrV2DataTypeField", "canonical_fill_value", "canonical_of", + "claim_key", + "claims_of", "fill_value_problems", + "refines", "resolve_codec_v2", "resolve_dtype_v2", ] diff --git a/packages/zarr-metadata/tests/model/test_array.py b/packages/zarr-metadata/tests/model/test_array.py index d924b3dbd0..3596b2e265 100644 --- a/packages/zarr-metadata/tests/model/test_array.py +++ b/packages/zarr-metadata/tests/model/test_array.py @@ -22,7 +22,7 @@ MetadataValidationError, ValidationProblem, ZarrV2ArrayMetadata, - ZarrV2ArrayMetadataPartial, + ZarrV2ArrayMetadataUpdate, ZarrV3ArrayMetadata, ZarrV3GroupMetadata, is_array_metadata_v2, @@ -385,7 +385,7 @@ def test_v2_create_default_applies_overrides() -> None: m = ZarrV2ArrayMetadata.create_default(shape=(8,), attributes={"k": "v"}) assert m.shape == (8,) assert m.attributes == {"k": "v"} - assert m.dtype == "|u1" # default dtype unchanged + assert m.dtype.to_json() == "|u1" # default dtype unchanged # --- V3 update ------------------------------------------------------------- @@ -633,16 +633,16 @@ def test_a_model_holding_nan_user_data_equals_its_copies() -> None: (0.0, 0.0, True), ("NaN", "NaN", True), (0.0, -0.0, False), - (1, 1.0, False), + (1, 1.0, True), ("Infinity", "NaN", False), ], ) def test_two_v2_models_are_one_array_when_their_documents_are_written_alike( left: object, right: object, same: bool ) -> None: - """A v2 model compares by its document as text: a fill value spelled two ways is two models.""" + """A v2 model compares by what its document means: a fill value an integer or a float is one value of a float type, `0.0` and `-0.0` two, and `NaN` is itself.""" model = ZarrV2ArrayMetadata.create_default(shape=(2,), chunks=(2,), dtype=" None: def test_v2_partial_keys_match_settable_model_fields() -> None: """The v2 partial TypedDict must list exactly the settable fields.""" - settable = {f.name for f in dataclasses.fields(ZarrV2ArrayMetadata) if f.init} - assert set(ZarrV2ArrayMetadataPartial.__annotations__) == settable + assert set(ZarrV2ArrayMetadataUpdate.__annotations__) == { + "shape", + "dtype", + "chunks", + "fill_value", + "order", + "compressor", + "filters", + "dimension_separator", + "attributes", + } def test_v2_to_key_value_splits_zarray_and_zattrs() -> None: @@ -1008,7 +1017,8 @@ def test_v2_roundtrip_with_compressor_and_filters() -> None: m = ZarrV2ArrayMetadata.create_default(compressor=compressor, filters=filters) restored = ZarrV2ArrayMetadata.from_json(m.to_json()) assert restored == m - assert restored.compressor == {"id": "blosc", "clevel": 5} + assert restored.compressor is not None + assert restored.compressor.to_json() == {"id": "blosc", "clevel": 5} # --- ZarrV2ArrayMetadata.from_json ---------------------------------------- @@ -1019,7 +1029,7 @@ def test_v2_from_json_reconstructs_fields() -> None: doc = ZarrV2ArrayMetadata.create_default(shape=(4,), attributes={"a": 1}, dtype=" None: ] -def test_v2_from_key_value_ignores_zarray_extra_members() -> None: - """Other raw `.zarray` members "SHOULD be ignored by implementations" (https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v2/v2.0.rst#L91-L92).""" +def test_v2_from_key_value_keeps_zarray_extra_members() -> None: + """Other raw `.zarray` members "SHOULD be ignored by implementations" (https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v2/v2.0.rst#L91-L92): left unjudged, and kept as written.""" doc: dict[str, object] = dict(ZarrV2ArrayMetadata.create_default().to_json()) doc.pop("attributes", None) doc["vendor_extension"] = {} model = ZarrV2ArrayMetadata.from_key_value({".zarray": json.dumps(doc).encode()}) - assert "vendor_extension" not in model.to_json() + assert model.to_json()["vendor_extension"] == {} + assert model.extra_fields == {"vendor_extension": {}} def test_v2_zattrs_presence_round_trips() -> None: @@ -1269,7 +1280,7 @@ def _build_v3(**overrides: Unpack[ZarrV3ArrayMetadataJSONPartial]) -> dict[str, return dict(ZarrV3ArrayMetadata.create_default(**overrides).to_json()) -def _build_v2(**overrides: Unpack[ZarrV2ArrayMetadataPartial]) -> dict[str, object]: +def _build_v2(**overrides: Unpack[ZarrV2ArrayMetadataUpdate]) -> dict[str, object]: return dict(ZarrV2ArrayMetadata.create_default(**overrides).to_json()) @@ -1710,15 +1721,17 @@ def test_array_zarr_format_rejects_float( assert [(p.loc, p.kind) for p in validate(document)] == [(("zarr_format",), "invalid_value")] -def test_array_v2_ignores_unknown_document_member() -> None: - """Other .zarray keys "SHOULD NOT be present ... and SHOULD be ignored": tolerated, dropped. +def test_array_v2_keeps_an_unknown_document_member() -> None: + """Other .zarray keys "SHOULD NOT be present ... and SHOULD be ignored": tolerated, left unjudged, and kept as written, in `extra_fields`. https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v2/v2.0.rst#L91-L92 """ doc = dict(ZarrV2ArrayMetadata.create_default().to_json()) | {"unexpected": 1} assert validate_array_metadata_v2(doc) == () - assert "unexpected" not in ZarrV2ArrayMetadata.from_json(doc).to_json() + model = ZarrV2ArrayMetadata.from_json(doc) + assert model.to_json()["unexpected"] == 1 + assert model.extra_fields == {"unexpected": 1} @pytest.mark.parametrize( diff --git a/packages/zarr-metadata/tests/model/test_construction.py b/packages/zarr-metadata/tests/model/test_construction.py index d164767e75..ad2addb4aa 100644 --- a/packages/zarr-metadata/tests/model/test_construction.py +++ b/packages/zarr-metadata/tests/model/test_construction.py @@ -81,18 +81,17 @@ def test_a_v3_model_is_built_of_its_document_in_its_scope(model: object) -> None @pytest.mark.parametrize( "model", - [V2_ARRAY, V2_GROUP, V2_CONSOLIDATED], + [ + V2_ARRAY, + pytest.param(V2_GROUP, marks=pytest.mark.xfail(strict=True, reason="Task 3")), + pytest.param(V2_CONSOLIDATED, marks=pytest.mark.xfail(strict=True, reason="Task 4")), + ], ids=["v2-array", "v2-group", "v2-consolidated"], ) def test_a_v2_model_is_built_as_a_read_builds_it(model: object) -> None: - """A v2 model's constructor checks what the read checked, and of the same members builds the same model.""" + """A v2 model is built of its document in its scope, as a read builds it: the constructor given the model's own document and scope builds an equal model.""" held = cast("Any", model) - members = { - member.name: getattr(held, member.name) - for member in dataclasses.fields(held) - if member.init - } - assert type(held)(**members) == model + assert type(held)(held.to_json(), context=held.context) == model @pytest.mark.parametrize( @@ -117,15 +116,19 @@ def test_a_v3_model_holds_its_members_as_a_read_refines_them( ("model", "changes"), [ (V2_ARRAY, {"shape": range(4, 5), "chunks": [4], "attributes": {"a": 1}}), - (V2_CONSOLIDATED, {"metadata": {"a/.zattrs": UserDict({"x": 1})}}), + pytest.param( + V2_CONSOLIDATED, + {"metadata": {"a/.zattrs": UserDict({"x": 1})}}, + marks=pytest.mark.xfail(strict=True, reason="Task 4"), + ), ], ids=["v2-sequences", "v2-consolidated-mapping"], ) def test_a_v2_model_holds_its_members_as_a_read_refines_them( model: object, changes: dict[str, object] ) -> None: - """Arrays as tuples and objects as dicts, as the read holds them, so a v2 model built of other containers is the model a read builds.""" - changed = dataclasses.replace(cast("Any", model), **changes) + """Arrays as tuples and objects as dicts, as the read holds them, so a v2 model updated with other containers is the model a read builds.""" + changed = cast("Any", model).update(**changes) assert changed == model assert type(changed).from_key_value(changed.to_key_value()) == changed @@ -153,15 +156,21 @@ def test_a_v3_model_shares_no_container_with_what_it_was_built_of( @pytest.mark.parametrize( ("model", "member"), - [(V2_GROUP, "attributes"), (V2_CONSOLIDATED, "metadata")], - ids=["v2-group-attributes", "v2-consolidated"], + [ + (V2_ARRAY, "attributes"), + (V2_GROUP, "attributes"), + pytest.param( + V2_CONSOLIDATED, "metadata", marks=pytest.mark.xfail(strict=True, reason="Task 4") + ), + ], + ids=["v2-array-attributes", "v2-group-attributes", "v2-consolidated"], ) def test_a_v2_model_shares_no_container_with_what_it_was_built_of( model: object, member: str ) -> None: """A v2 model holds copies of the containers it is built of.""" held: dict[str, object] = {"acme.x": {"must_understand": False}} - built = dataclasses.replace(cast("Any", model), **{member: held}) + built = cast("Any", model).update(**{member: held}) held["acme.y"] = math.nan cast("dict[str, object]", held["acme.x"])["z"] = math.nan assert getattr(built, member) == {"acme.x": {"must_understand": False}} @@ -260,10 +269,11 @@ def test_error_consolidated_metadata_of_documents_at_bad_paths_is_refused() -> N (V2_ARRAY, {"order": "Q"}, [(("order",), "invalid_value")]), (V2_ARRAY, {"chunks": (4, 4)}, [(("chunks",), "invalid_value")]), (V2_GROUP, {"attributes": {1: "a"}}, [(("attributes",), "invalid_type")]), - ( + pytest.param( V2_CONSOLIDATED, {"metadata": {"a/.zarray": {"x": math.nan}}}, [(("metadata", "a/.zarray", "x"), "invalid_value")], + marks=pytest.mark.xfail(strict=True, reason="Task 4"), ), ], ids=[ @@ -278,7 +288,7 @@ def test_error_a_v2_model_changed_by_hand_into_an_invalid_one_is_refused_at_the_ ) -> None: """As a read reads its document: no v2 model is invalid, however it came to be.""" with pytest.raises(MetadataValidationError) as raised: - dataclasses.replace(cast("Any", model), **changes) + cast("Any", model).update(**changes) assert [(found.loc, found.kind) for found in raised.value.problems] == problems diff --git a/packages/zarr-metadata/tests/model/test_pair_v2.py b/packages/zarr-metadata/tests/model/test_pair_v2.py new file mode 100644 index 0000000000..a317a5a88d --- /dev/null +++ b/packages/zarr-metadata/tests/model/test_pair_v2.py @@ -0,0 +1,241 @@ +"""A v2 model is its document and the scope it was read in, as the v3 models are.""" + +from __future__ import annotations + +import copy +import dataclasses +import pickle +from typing import Any, cast + +import pytest + +from zarr_metadata._sentinel import UNSET +from zarr_metadata.model import ( + MetadataValidationError, + ZarrV2ArrayMetadata, + read_array_metadata_v2, +) +from zarr_metadata.v2.codec.compression import ZLIB_V2 +from zarr_metadata.v2.data_type.scalar import FLOAT_V2, UINT_V2 +from zarr_metadata.v2.definition import ( + CORE_V2, + Context, + Read, + ScopeConflictError, + Unclaimed, + ZarrV2CodecDefinition, + ZarrV2DataTypeDefinition, +) +from zarr_metadata.v3.definition import EmptyConfiguration + +Loc = tuple[str | int, ...] +ARRAY: dict[str, Any] = { + "zarr_format": 2, + "shape": [4], + "chunks": [2], + "dtype": " dict[str, Any]: + """`ARRAY` with `changes`, a member given as `UNSET` left out.""" + return {key: value for key, value in {**ARRAY, **changes}.items() if value is not UNSET} + + +def test_a_model_is_its_document_read_in_its_scope() -> None: + """The constructor reads the document in the scope given, `CORE_V2` by default; `to_json` is the document refined (arrays as tuples, `dimension_separator` put in when missing), sharing nothing with the model; `context` is the scope; the reading holds the model.""" + model = ZarrV2ArrayMetadata(ARRAY) + assert model.context is CORE_V2 + document = model.to_json() + assert document["shape"] == (4,) + assert document.get("attributes") == {"a": (1, 2)} + without = {k: v for k, v in ARRAY.items() if k != "dimension_separator"} + assert ZarrV2ArrayMetadata(without).to_json().get("dimension_separator") == "." + assert model.reading.metadata is model + assert read_array_metadata_v2(ARRAY).metadata == model + assert ZarrV2ArrayMetadata(ARRAY, SMALL).context is SMALL + + +def test_properties_are_what_the_reading_holds() -> None: + """`dtype`, `compressor` and `filters` are the fields as the scope read them, None where `null` is written; `shape`, `chunks`, `fill_value`, `order`, `dimension_separator`, `attributes` and `extra_fields` are the members as the read refined them, read-only; `claims` is keyed as the scope files them.""" + model = ZarrV2ArrayMetadata({**ARRAY, "filters": [{"id": "x"}], "extra": [1]}) + assert isinstance(model.dtype, Read) + assert model.dtype.definition is FLOAT_V2 + assert isinstance(model.compressor, Read) + assert model.compressor.definition is ZLIB_V2 + assert model.filters is not None + assert isinstance(model.filters[0], Unclaimed) + members = (model.shape, model.chunks, model.fill_value, model.order, model.dimension_separator) + assert members == ((4,), (2,), 0, "C", ".") + assert model.attributes == {"a": (1, 2)} + assert model.extra_fields == {"extra": (1,)} + assert (ZarrV2DataTypeDefinition, "float") in model.claims + with pytest.raises(TypeError): + cast("dict[str, object]", model.attributes)["b"] = 1 + assert ZarrV2ArrayMetadata({**ARRAY, "compressor": None}).compressor is None + assert ZarrV2ArrayMetadata(_document({"attributes": UNSET})).attributes is UNSET + + +@pytest.mark.parametrize( + ("left", "right", "same"), + [ + ({}, {"shape": (4,), "attributes": {"a": (1, 2)}}, True), + ({"dtype": "f4"}, False), + ({"attributes": {"a": [1, 2]}}, {"attributes": {"a": [2, 1]}}, False), + ({"attributes": UNSET}, {"attributes": {}}, False), + ({"extra": 1}, {}, False), + ], + ids=[ + "same", + "int-for-float", + "nan", + "spelling", + "key-order", + "signed-zero", + "byte-order", + "attributes", + "zattrs-presence", + "extra", + ], +) +def test_models_are_equal_by_what_their_documents_mean( + left: dict[str, Any], right: dict[str, Any], same: bool +) -> None: + """Two models are one array when their documents mean the same in their scopes: a typestr spelled two ways or a fill value an integer or a float is one, a byte order or a `.zattrs` present or not is two; equal models hash alike.""" + one, other = ZarrV2ArrayMetadata(_document(left)), ZarrV2ArrayMetadata(_document(right)) + assert (one == other) is same + if same: + assert hash(one) == hash(other) + + +def test_a_model_read_in_two_scopes_that_read_it_alike_is_one_model() -> None: + """Equality is by interpretation: the same document read by a private `zlib` with no parameters and by the core one are two models, and read in two scopes that file the same `zlib` are one.""" + document = {**ARRAY, "compressor": {"id": "zlib"}} + assert ZarrV2ArrayMetadata(document) != ZarrV2ArrayMetadata(document, PRIVATE) + same = Context.of(*CORE_V2.definitions()) + assert ZarrV2ArrayMetadata(document) == ZarrV2ArrayMetadata(document, same) + + +@pytest.mark.parametrize("context", [None, SMALL, PRIVATE], ids=["core", "small", "private"]) +def test_a_model_round_trips_through_its_document_pickle_and_copy(context: Context | None) -> None: + """`from_json(m.to_json(), context=m.context)`, `from_key_value(m.to_key_value())`, `pickle` and `copy.deepcopy` give an equal model in the same scope, and a pickled reading comes back as its model's own reading.""" + model = ZarrV2ArrayMetadata(ARRAY, context) + assert ZarrV2ArrayMetadata.from_json(model.to_json(), context=model.context) == model + restored = ZarrV2ArrayMetadata.from_key_value(model.to_key_value(), context=model.context) + assert restored == model + loaded = pickle.loads(pickle.dumps(model)) + assert loaded == model + assert loaded.context == model.context + assert copy.deepcopy(model) == model + assert pickle.loads(pickle.dumps(model.reading)).metadata == model + + +@pytest.mark.parametrize( + ("changes", "expect"), + [ + ({"shape": [8], "chunks": [4]}, {"shape": (8,), "chunks": (4,)}), + ({"dtype": " None: + """`update` puts JSON members in place of the document's, leaves out one given as `UNSET`, and reads the result in the model's own scope; the model is unchanged.""" + model = ZarrV2ArrayMetadata({**ARRAY, "extra": 0}, PRIVATE) + changed = model.update(**changes) + assert changed.context is PRIVATE + document = changed.to_json() + for key, value in expect.items(): + assert document[key] == value + for key, value in changes.items(): + if value is UNSET: + assert key not in document + assert model.to_json()["extra"] == 0 + + +@pytest.mark.parametrize( + ("changes", "at"), + [ + ({"dtype": "float32"}, ("dtype",)), + ({"chunks": [1, 1]}, ("chunks",)), + ({"dtype": "|b1"}, ("fill_value",)), + ], + ids=["dtype", "rank", "fill-no-longer-fits"], +) +def test_error_update_refuses_a_document_with_a_problem(changes: dict[str, Any], at: Loc) -> None: + """A change that makes a document with a problem is refused at the change, with the problem where it is: a dtype the family's fill value no longer fits is reported at the fill value.""" + with pytest.raises(MetadataValidationError) as raised: + ZarrV2ArrayMetadata(ARRAY).update(**changes) + assert raised.value.problems[0].loc == at + + +def test_with_context_and_refined_in_read_the_document_in_another_scope() -> None: + """`with_context` reads the document in any scope (a loss is allowed), `refined_in` only up the order: a scope that claims what this one left unclaimed is a gain, one that reads a name by another definition or by none is a `ScopeConflictError` naming where the name sits, and a gain that surfaces a problem is a `MetadataValidationError`.""" + unclaimed = ZarrV2ArrayMetadata({**ARRAY, "compressor": {"id": "zlib"}}, SMALL) + assert isinstance(unclaimed.compressor, Unclaimed) + gained = unclaimed.refined_in(PRIVATE) + assert isinstance(gained.compressor, Read) + assert gained.compressor.definition is BARE_ZLIB + assert unclaimed.refines(unclaimed) + assert gained.refines(unclaimed) + assert not unclaimed.refines(gained) + with pytest.raises(ScopeConflictError) as conflict: + gained.refined_in(CORE_V2) + assert [c.loc for c in conflict.value.conflicts] == [("compressor",)] + lost = gained.with_context(SMALL) + assert isinstance(lost.compressor, Unclaimed) + assert lost == unclaimed + assert gained.with_context(PRIVATE) == gained + with pytest.raises(MetadataValidationError): + ZarrV2ArrayMetadata({**ARRAY, "compressor": {"id": "zlib", "level": 1}}, SMALL).refined_in( + PRIVATE + ) + + +def test_refines_orders_models_by_information() -> None: + """`refines` holds when each field refines its counterpart and every other member is the same, the fill value as the more informed dtype spells it; a fill value that dtype refuses is no refinement; a value of another type refines nothing.""" + core = ZarrV2ArrayMetadata(ARRAY) + assert core.refines(ZarrV2ArrayMetadata({**ARRAY, "fill_value": 0.0})) + assert not core.refines(ZarrV2ArrayMetadata({**ARRAY, "shape": [8], "chunks": [2]})) + small = ZarrV2ArrayMetadata({**ARRAY, "dtype": " None: + """No model is invalid: the constructor raises with every problem the read finds.""" + with pytest.raises(MetadataValidationError) as raised: + ZarrV2ArrayMetadata({**ARRAY, "dtype": "float32", "order": "Q"}) + assert [p.loc for p in raised.value.problems] == [("dtype",), ("order",)] + + +def test_the_dataclass_machinery_is_gone() -> None: + """A v2 model is built only from a document: `dataclasses.replace` and `dataclasses.fields` do not apply, and the old `...Partial` names are gone.""" + import zarr_metadata + + assert not dataclasses.is_dataclass(ZarrV2ArrayMetadata) + assert not hasattr(zarr_metadata, "ZarrV2ArrayMetadataPartial") + assert "ZarrV2ArrayMetadataUpdate" in zarr_metadata.__all__ diff --git a/packages/zarr-metadata/tests/model/test_pydantic_module.py b/packages/zarr-metadata/tests/model/test_pydantic_module.py index b0e444463e..9aae2c6948 100644 --- a/packages/zarr-metadata/tests/model/test_pydantic_module.py +++ b/packages/zarr-metadata/tests/model/test_pydantic_module.py @@ -160,7 +160,7 @@ def test_v2_recursive_structured_dtype_is_in_pydantic_schema() -> None: doc["fill_value"] = None adapter = TypeAdapter(zmp.ZarrV2ArrayMetadata) - assert adapter.validate_python(doc).dtype == (("outer", (("inner", " None: - """The v2 array document is open ("SHOULD be ignored", https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v2/v2.0.rst#L91-L92), in runtime and schema.""" + """The v2 array document is open ("SHOULD be ignored", https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v2/v2.0.rst#L91-L92), in runtime and schema; the member is kept as written.""" doc = json.loads(json.dumps(V2_ARRAY_DOC)) doc["unexpected"] = 1 adapter = TypeAdapter(zmp.ZarrV2ArrayMetadata) - assert "unexpected" not in adapter.validate_python(doc).to_json() + assert adapter.validate_python(doc).to_json()["unexpected"] == 1 assert Draft202012Validator(adapter.json_schema()).is_valid(doc) diff --git a/packages/zarr-metadata/tests/test_public_api.py b/packages/zarr-metadata/tests/test_public_api.py index c59948cb0d..48360e8e22 100644 --- a/packages/zarr-metadata/tests/test_public_api.py +++ b/packages/zarr-metadata/tests/test_public_api.py @@ -43,7 +43,7 @@ def _group_rank(s: str) -> int: "JSONValue", # Category A' — metadata models (in-memory dataclasses over the documents) "ZarrV2ArrayMetadata", - "ZarrV2ArrayMetadataPartial", + "ZarrV2ArrayMetadataUpdate", "ZarrV3ArrayMetadata", "ZarrV3ArrayMetadataUpdate", "ZarrV2GroupMetadata", From fd8873b241eda97874a03826aa019efb5d57c0ff Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Thu, 8 Oct 2026 12:48:12 +0200 Subject: [PATCH 69/94] feat(zarr-metadata)!: make the v2 group model a pair of document and scope ZarrV2GroupMetadata(document, context=None) holds its document and scope like the array model. ZarrV2GroupMetadataUpdate replaces ZarrV2GroupMetadataPartial. Assisted-by: ClaudeCode:claude-fable-5-1 Co-Authored-By: Claude Fable 5.1 --- .../src/zarr_metadata/__init__.py | 4 +- .../src/zarr_metadata/model/__init__.py | 4 +- .../src/zarr_metadata/model/_group.py | 238 ++++++++++-------- .../tests/model/test_construction.py | 13 +- .../zarr-metadata/tests/model/test_group.py | 5 +- .../zarr-metadata/tests/model/test_pair_v2.py | 51 ++++ .../zarr-metadata/tests/test_public_api.py | 2 +- 7 files changed, 194 insertions(+), 123 deletions(-) diff --git a/packages/zarr-metadata/src/zarr_metadata/__init__.py b/packages/zarr-metadata/src/zarr_metadata/__init__.py index 824911aadb..a227609c19 100644 --- a/packages/zarr-metadata/src/zarr_metadata/__init__.py +++ b/packages/zarr-metadata/src/zarr_metadata/__init__.py @@ -20,8 +20,8 @@ ZarrV2ConsolidatedMetadata, ZarrV2ConsolidatedMetadataStoreKey, ZarrV2GroupMetadata, - ZarrV2GroupMetadataPartial, ZarrV2GroupMetadataStoreKey, + ZarrV2GroupMetadataUpdate, ZarrV3ArrayMetadata, ZarrV3ArrayMetadataStoreKey, ZarrV3ArrayMetadataUpdate, @@ -374,8 +374,8 @@ "ZarrV2GroupMetadata", "ZarrV2GroupMetadataJSON", "ZarrV2GroupMetadataJSONPartial", - "ZarrV2GroupMetadataPartial", "ZarrV2GroupMetadataStoreKey", + "ZarrV2GroupMetadataUpdate", "ZarrV2ZArrayJSON", "ZarrV2ZAttrsJSON", "ZarrV2ZGroupJSON", diff --git a/packages/zarr-metadata/src/zarr_metadata/model/__init__.py b/packages/zarr-metadata/src/zarr_metadata/model/__init__.py index 35e88001f4..76e2c27fdd 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/__init__.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/__init__.py @@ -53,7 +53,7 @@ from zarr_metadata.model._group import ( ZarrV2ConsolidatedMetadata, ZarrV2GroupMetadata, - ZarrV2GroupMetadataPartial, + ZarrV2GroupMetadataUpdate, ZarrV3ConsolidatedMetadata, ZarrV3ConsolidatedMetadataInput, ZarrV3GroupMetadata, @@ -178,8 +178,8 @@ "ZarrV2ConsolidatedMetadata", "ZarrV2ConsolidatedMetadataStoreKey", "ZarrV2GroupMetadata", - "ZarrV2GroupMetadataPartial", "ZarrV2GroupMetadataStoreKey", + "ZarrV2GroupMetadataUpdate", "ZarrV3ArrayMetadata", "ZarrV3ArrayMetadataReading", "ZarrV3ArrayMetadataStoreKey", diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_group.py b/packages/zarr-metadata/src/zarr_metadata/model/_group.py index add64ca80d..773eb97352 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_group.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/_group.py @@ -57,6 +57,7 @@ ) from zarr_metadata.v2.attributes import ZARR_V2_ATTRIBUTES_STORE_KEY from zarr_metadata.v2.consolidated import ZARR_V2_CONSOLIDATED_METADATA_STORE_KEY +from zarr_metadata.v2.definition import CORE_V2 from zarr_metadata.v2.group import ZARR_V2_GROUP_METADATA_STORE_KEY from zarr_metadata.v3._hierarchy import NodeType, hierarchy_problems, path_faults, said from zarr_metadata.v3._registry import CORE_AND_EXTENSIONS, Context @@ -1186,118 +1187,165 @@ def parse_group_metadata_v3( return cast("ZarrV3GroupMetadataJSON", arrays_to_tuples(documents_for(value))) -class ZarrV2GroupMetadataPartial(TypedDict, total=False): - """ - Partial form of the constructor-settable fields of `ZarrV2GroupMetadata`. - - Every key is optional and typed with the model's own value types, so it - describes valid keyword arguments to `ZarrV2GroupMetadata.update` and - `create_default`. The `init=False` field `zarr_format` is intentionally - excluded, since it cannot be passed to `dataclasses.replace`. - - Drift between this type and the model's settable fields is prevented by - `tests/model/test_group.py::test_group_partial_keys_match_settable_model_fields`. - """ +class ZarrV2GroupMetadataUpdate(TypedDict, total=False): + """The members `ZarrV2GroupMetadata.update` puts in place: `attributes` as a `.zattrs` writes them, or `UNSET` for no `.zattrs`.""" - attributes: dict[str, JSONValue] | UNSET + attributes: Mapping[str, JSONValue] | UNSET -@dataclass(frozen=True, slots=True, kw_only=True) class ZarrV2GroupMetadata: - """In-memory model of a v2 group metadata document. - - A canonical, lossless representation of the `.zgroup` content plus the - sibling `.zattrs` attributes, folded into a single in-memory value - (mirroring the merged `ZarrV2GroupMetadataJSON` document form). `attributes` is - `UNSET` when no `.zattrs` file (or merged `attributes` key) exists — - distinct from an explicit empty `.zattrs`, which is `{}` and round-trips - as a file. A model checks itself when it is built: its document has no - problem `validate_group_metadata_v2` finds, or the constructor raises - `MetadataValidationError`. + """A v2 group document, and the scope it was read in. + + The pair, as the v3 models are: `to_json` is the merged document -- + `.zgroup`, and `attributes` when a `.zattrs` holds them -- as written, + refined. `attributes` is `UNSET` when no `.zattrs` exists, distinct + from an empty one. A group holds no field a scope reads, so the scope + is held for uniformity: `update` reads new attributes in it, and no + other scope conflicts with the reading. Built only by reading: the + constructor raises `MetadataValidationError` with every problem. Two + groups are equal when their attributes are written alike, as + `group_key_v2` says. """ - zarr_format: Literal[2] = field(default=2, init=False) - attributes: dict[str, JSONValue] | UNSET + __slots__ = ("_attributes", "_context", "_document", "_key", "_shown") - def __post_init__(self) -> None: - # Held as a read refines them, in containers of its own. - object.__setattr__( - self, "attributes", _v2_attributes(parse_group_metadata_v2(self.to_json())) - ) + _attributes: dict[str, JSONValue] | UNSET + _context: Context + _document: dict[str, JSONValue] + _key: tuple[object, ...] + _shown: object + + zarr_format: Final = 2 + + def __init__(self, document: object, context: Context | None = None) -> None: + scope = CORE_V2 if context is None else context + parsed = parse_group_metadata_v2(document, context=scope) + refined, _ = refine_user_data(document) + self._adopt(cast("dict[str, JSONValue]", refined), scope, _v2_attributes(parsed)) @classmethod - def create_default(cls, **overrides: Unpack[ZarrV2GroupMetadataPartial]) -> ZarrV2GroupMetadata: - """ - Create a default (empty) v2 group metadata model, with optional overrides. + def _of( + cls, + document: dict[str, JSONValue], + context: Context, + attributes: dict[str, JSONValue] | UNSET, + ) -> ZarrV2GroupMetadata: + """A model of a document a read found nothing wrong with: no second read.""" + model = object.__new__(cls) + model._adopt(document, context, attributes) + return model - The default is a structurally-valid group with no attributes — the group - analog of `list()` returning `[]`. Any field can be overridden by keyword - (the same fields accepted by `update`). - """ - default = cls(attributes=UNSET) - return default.update(**overrides) + def _adopt( + self, + document: dict[str, JSONValue], + context: Context, + attributes: dict[str, JSONValue] | UNSET, + ) -> None: + self._document = document + self._context = context + self._attributes = attributes + self._shown = UNSET if attributes is UNSET else frozen(attributes) + self._key = group_key_v2(self) - def update(self, **kwargs: Unpack[ZarrV2GroupMetadataPartial]) -> ZarrV2GroupMetadata: - """ - Return a new `ZarrV2GroupMetadata` with the given fields updated. - - Only the constructor-settable fields listed in - `ZarrV2GroupMetadataPartial` can be updated; the fixed `zarr_format` - is rejected at the type level. Each given field fully replaces its - previous value. `MetadataValidationError` when the document the - change makes has a problem, as the model checks itself when it is - built. + @property + def context(self) -> Context: + """The scope the document was read in, which `update` reads new attributes in.""" + return self._context + + @property + def claims(self) -> Claims: + """What the reading claimed: nothing, since a group holds no field.""" + return MappingProxyType({}) + + @property + def attributes(self) -> Mapping[str, JSONValue] | UNSET: + """The user attributes a `.zattrs` holds, read-only at every level; `UNSET` when there is no `.zattrs`.""" + return cast("Mapping[str, JSONValue] | UNSET", self._shown) + + def to_json(self) -> ZarrV2GroupMetadataJSON: + """The merged document as written, refined, sharing nothing with the model. + + `attributes` is included when set, even empty. This is not the + on-disk `.zgroup`, which excludes them: `to_key_value` splits the + document as a store holds it + (https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v2/v2.0.rst#L313; https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v2/v2.0.rst#L323-L330). """ - return dataclasses.replace(self, **kwargs) + return cast("ZarrV2GroupMetadataJSON", copied(self._document)) + + def to_key_value( + self, *, indent: int | str | None = None + ) -> Mapping[ZarrV2GroupMetadataStoreKey | ZarrV2AttributesStoreKey, bytes]: + """The document as a store holds it: `.zgroup` without the attributes, and `.zattrs` with them when they are set, even empty.""" + zgroup = {key: value for key, value in self._document.items() if key != "attributes"} + out: dict[ZarrV2GroupMetadataStoreKey | ZarrV2AttributesStoreKey, bytes] = { + ZARR_V2_GROUP_METADATA_STORE_KEY: dump_store_json(zgroup, indent=indent) + } + if "attributes" in self._document: + out[ZARR_V2_ATTRIBUTES_STORE_KEY] = dump_store_json( + self._document["attributes"], indent=indent + ) + return out + + def __repr__(self) -> str: + return f"{type(self).__name__}({self._document!r}, context={self._context!r})" def __eq__(self, other: object) -> bool: - """Whether `other` models the same group: the same document, as JSON text, which takes `NaN` for itself; equal models hash alike.""" if type(other) is not type(self): return NotImplemented - return json_text(self.to_json()) == json_text(cast("ZarrV2GroupMetadata", other).to_json()) + return self._key == cast("ZarrV2GroupMetadata", other)._key def __hash__(self) -> int: - return hash(json_text(self.to_json())) + return hash(self._key) - def to_json(self) -> ZarrV2GroupMetadataJSON: - """Return the merged in-memory document form. + def __reduce__(self) -> tuple[type[ZarrV2GroupMetadata], tuple[object, Context]]: + return type(self), (self._document, self._context) - `attributes` is included when set (even empty). This is not the - on-disk `.zgroup` content: a conforming `.zgroup` must exclude - `attributes` (they live in the sibling `.zattrs` file). Use - `to_key_value` to produce the spec-conforming split for storage - (https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v2/v2.0.rst#L313; https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v2/v2.0.rst#L323-L330). - """ - # to_json output shares no mutable state with the model: the document - # is copied whole, one frame for each level of nesting. - out: ZarrV2GroupMetadataJSON = {"zarr_format": self.zarr_format} - if self.attributes is not UNSET: - out["attributes"] = self.attributes - return cast("ZarrV2GroupMetadataJSON", copied(cast("JSONValue", out))) + def update(self, **members: Unpack[ZarrV2GroupMetadataUpdate]) -> ZarrV2GroupMetadata: + """This model with `attributes` in place of the document's, `UNSET` leaving them out, read in this model's own scope; `MetadataValidationError` when the document they make has a problem.""" + document: dict[str, object] = {**self._document, **members} + for key, value in members.items(): + if value is UNSET: + del document[key] + return type(self)(document, context=self._context) + + def with_context(self, context: Context | None = None) -> ZarrV2GroupMetadata: + """This document read in `context`: the same group, holding that scope.""" + scope = CORE_V2 if context is None else context + return self._of(self._document, scope, self._attributes) + + def refined_in(self, context: Context | None = None) -> ZarrV2GroupMetadata: + """This document read in `context`: a group holds no field, so no scope conflicts with its reading, and this is `with_context`.""" + return self.with_context(context) + + def refines(self, other: ZarrV2GroupMetadata) -> bool: + """Whether this group holds everything `other` holds: its attributes written alike; False of what is not a v2 group.""" + return type(other) is type(self) and self._key == other._key @classmethod - def from_json(cls, data: object) -> ZarrV2GroupMetadata: - """The model of `data`, a v2 group document with its attributes under `attributes`. + def create_default( + cls, *, context: Context | None = None, **overrides: Unpack[ZarrV2GroupMetadataUpdate] + ) -> ZarrV2GroupMetadata: + """A group with no `.zattrs`, or the one `overrides` make of it, read in `context`; `MetadataValidationError` when the document they make has a problem.""" + given = {key: value for key, value in overrides.items() if value is not UNSET} + return cls({"zarr_format": 2, **given}, context=context) - `MetadataValidationError` with every problem `validate_group_metadata_v2` - finds. The model shares no mutable state with `data`. - """ - # A read model shares no mutable state with what it read. - parsed = cast( - "ZarrV2GroupMetadataJSON", copied(cast("JSONValue", parse_group_metadata_v2(data))) - ) - return construct(cls, attributes=_v2_attributes(parsed)) + @classmethod + def from_json(cls, data: object, *, context: Context | None = None) -> ZarrV2GroupMetadata: + """The model of `data`, a v2 group document with its attributes under `attributes`, read in `context`; `MetadataValidationError` with every problem.""" + return cls(data, context=context) @classmethod - def from_key_value(cls, mapping: Mapping[StoreKey, bytes]) -> ZarrV2GroupMetadata: - """The model of the group at `.zgroup` in `mapping`, with the attributes at `.zattrs` when there is one. + def from_key_value( + cls, mapping: Mapping[StoreKey, bytes], *, context: Context | None = None + ) -> ZarrV2GroupMetadata: + """The model of the group at `.zgroup` in `mapping`, with the attributes at `.zattrs` when there is one, read in `context`. `MetadataValidationError` when `.zgroup` is missing, bytes are not JSON, `.zgroup` holds `attributes`, or the document is not valid. """ zgroup_raw = load_store_json(mapping, ZARR_V2_GROUP_METADATA_STORE_KEY) if not isinstance(zgroup_raw, Mapping): - return cls.from_json(zgroup_raw) + return cls(zgroup_raw, context=context) zgroup = cast("Mapping[str, object]", zgroup_raw) if "attributes" in zgroup: # A key `.zgroup` does not declare: its attributes are `.zattrs`. @@ -1307,30 +1355,14 @@ def from_key_value(cls, mapping: Mapping[StoreKey, bytes]) -> ZarrV2GroupMetadat raise MetadataValidationError(with_input((refused,), zgroup)) if ZARR_V2_ATTRIBUTES_STORE_KEY in mapping: zattrs = load_store_json(mapping, ZARR_V2_ATTRIBUTES_STORE_KEY) - return cls.from_json({**zgroup, "attributes": zattrs}) - return cls.from_json(zgroup) + return cls({**zgroup, "attributes": zattrs}, context=context) + return cls(zgroup, context=context) - def to_key_value( - self, *, indent: int | str | None = None - ) -> Mapping[ZarrV2GroupMetadataStoreKey | ZarrV2AttributesStoreKey, bytes]: - """The document as a store holds it: `.zgroup` without the attributes, and `.zattrs` with them when they are set, even empty. - A model was checked when it was built, so its document is written - as it is. - """ - # Attributes live only in the sibling `.zattrs` file; the `.zgroup` - # document must exclude them. The `.zattrs` key is present exactly - # when attributes are set (even empty) — UNSET emits no file. - document = self.to_json() - zgroup = {k: v for k, v in document.items() if k != "attributes"} - out: dict[ZarrV2GroupMetadataStoreKey | ZarrV2AttributesStoreKey, bytes] = { - ZARR_V2_GROUP_METADATA_STORE_KEY: dump_store_json(zgroup, indent=indent) - } - if "attributes" in document: - out[ZARR_V2_ATTRIBUTES_STORE_KEY] = dump_store_json( - document["attributes"], indent=indent - ) - return out +def group_key_v2(model: ZarrV2GroupMetadata) -> tuple[object, ...]: + """What `==` and `hash` compare of a v2 group model: its attributes as JSON text, or `UNSET` when there is no `.zattrs`.""" + attributes = model._attributes # pyright: ignore[reportPrivateUsage] + return (UNSET if attributes is UNSET else json_text(attributes),) @dataclass(frozen=True, slots=True, kw_only=True) diff --git a/packages/zarr-metadata/tests/model/test_construction.py b/packages/zarr-metadata/tests/model/test_construction.py index ad2addb4aa..cc30d2a107 100644 --- a/packages/zarr-metadata/tests/model/test_construction.py +++ b/packages/zarr-metadata/tests/model/test_construction.py @@ -17,7 +17,6 @@ import pytest from zarr_metadata.model import ( - UNSET, MetadataValidationError, ValidationProblem, ZarrV2ArrayMetadata, @@ -83,7 +82,7 @@ def test_a_v3_model_is_built_of_its_document_in_its_scope(model: object) -> None "model", [ V2_ARRAY, - pytest.param(V2_GROUP, marks=pytest.mark.xfail(strict=True, reason="Task 3")), + V2_GROUP, pytest.param(V2_CONSOLIDATED, marks=pytest.mark.xfail(strict=True, reason="Task 4")), ], ids=["v2-array", "v2-group", "v2-consolidated"], @@ -302,16 +301,6 @@ def test_error_v2_create_default_refuses_chunks_its_default_shape_does_not_take( ] -def test_error_construct_refuses_a_member_the_model_does_not_take() -> None: - with pytest.raises(TypeError, match="ZarrV2GroupMetadata has no member \\['bogus'\\] to build"): - construct(ZarrV2GroupMetadata, attributes=UNSET, bogus=1) - - -def test_error_construct_refuses_a_model_missing_a_member() -> None: - with pytest.raises(TypeError, match="ZarrV2GroupMetadata is built with 'attributes'"): - construct(ZarrV2GroupMetadata) - - def test_error_consolidated_metadata_paths_are_strings() -> None: """A `metadata` member keyed by what is no string is a problem of the member, as the read reports it, not a `TypeError`.""" with pytest.raises(MetadataValidationError) as raised: diff --git a/packages/zarr-metadata/tests/model/test_group.py b/packages/zarr-metadata/tests/model/test_group.py index d276757e49..c1db0e63fd 100644 --- a/packages/zarr-metadata/tests/model/test_group.py +++ b/packages/zarr-metadata/tests/model/test_group.py @@ -24,7 +24,7 @@ from zarr_metadata.model._group import ( ZarrV2ConsolidatedMetadata, ZarrV2GroupMetadata, - ZarrV2GroupMetadataPartial, + ZarrV2GroupMetadataUpdate, ZarrV3ConsolidatedMetadata, ZarrV3GroupMetadata, ZarrV3GroupMetadataReading, @@ -543,8 +543,7 @@ def test_group_partial_keys_match_settable_model_fields() -> None: Guards against drift: adding/removing a settable field on the model without updating its `*Partial` TypedDict fails here. """ - settable = {f.name for f in dataclasses.fields(ZarrV2GroupMetadata) if f.init} - assert set(ZarrV2GroupMetadataPartial.__annotations__) == settable + assert set(ZarrV2GroupMetadataUpdate.__annotations__) == {"attributes"} def test_update_takes_every_member_of_the_document_it_may_change() -> None: diff --git a/packages/zarr-metadata/tests/model/test_pair_v2.py b/packages/zarr-metadata/tests/model/test_pair_v2.py index a317a5a88d..5c4a2f394e 100644 --- a/packages/zarr-metadata/tests/model/test_pair_v2.py +++ b/packages/zarr-metadata/tests/model/test_pair_v2.py @@ -13,6 +13,7 @@ from zarr_metadata.model import ( MetadataValidationError, ZarrV2ArrayMetadata, + ZarrV2GroupMetadata, read_array_metadata_v2, ) from zarr_metadata.v2.codec.compression import ZLIB_V2 @@ -239,3 +240,53 @@ def test_the_dataclass_machinery_is_gone() -> None: assert not dataclasses.is_dataclass(ZarrV2ArrayMetadata) assert not hasattr(zarr_metadata, "ZarrV2ArrayMetadataPartial") assert "ZarrV2ArrayMetadataUpdate" in zarr_metadata.__all__ + + +GROUP: dict[str, Any] = {"zarr_format": 2, "attributes": {"g": [1]}} + + +@pytest.mark.parametrize( + ("document", "attributes"), + [ + (GROUP, {"g": (1,)}), + ({"zarr_format": 2}, UNSET), + ({"zarr_format": 2, "attributes": {}}, {}), + ], + ids=["attributes", "no-zattrs", "empty-zattrs"], +) +def test_a_group_is_its_document_read_in_its_scope( + document: dict[str, Any], attributes: object +) -> None: + """A v2 group model is its document and scope: `attributes` as the read refined them, `UNSET` when no `.zattrs` exists; it round-trips through its document, its store keys, pickle and copy, in any scope.""" + model = ZarrV2GroupMetadata(document, SMALL) + assert model.context is SMALL + assert model.attributes == attributes + assert model.to_json() == ZarrV2GroupMetadata.from_json(document).to_json() + assert ZarrV2GroupMetadata.from_key_value(model.to_key_value(), context=SMALL) == model + assert pickle.loads(pickle.dumps(model)) == model + assert copy.deepcopy(model) == model + assert dict(model.claims) == {} + + +def test_a_group_updates_moves_scope_and_refines_as_the_arrays_do() -> None: + """`update` reads the new attributes in the model's own scope and `UNSET` leaves them out; `with_context` and `refined_in` read the document in another scope and conflict with nothing, since a group holds no field; two groups are one when their attributes are written alike, and `refines` is that equality.""" + model = ZarrV2GroupMetadata(GROUP) + assert model.update(attributes={"h": 2}).attributes == {"h": 2} + assert model.update(attributes=UNSET).attributes is UNSET + assert model.refined_in(SMALL).context is SMALL + assert model.with_context(PRIVATE) == model + assert model == ZarrV2GroupMetadata({"zarr_format": 2, "attributes": {"g": (1,)}}) + assert model != ZarrV2GroupMetadata({"zarr_format": 2}) + assert model.refines(model) + assert not model.refines(ZarrV2GroupMetadata({"zarr_format": 2})) + assert not dataclasses.is_dataclass(ZarrV2GroupMetadata) + + +def test_error_a_group_document_with_a_problem_is_refused_at_construction() -> None: + """A group document with a problem -- a member the spec does not define, an attribute key that is not a string -- is refused with every problem.""" + with pytest.raises(MetadataValidationError) as raised: + ZarrV2GroupMetadata({"zarr_format": 2, "attributes": {1: "a"}, "extra": 1}) + assert sorted((p.loc, p.kind) for p in raised.value.problems) == [ + (("attributes",), "invalid_type"), + (("extra",), "unknown_key"), + ] diff --git a/packages/zarr-metadata/tests/test_public_api.py b/packages/zarr-metadata/tests/test_public_api.py index 48360e8e22..f8ef36cb04 100644 --- a/packages/zarr-metadata/tests/test_public_api.py +++ b/packages/zarr-metadata/tests/test_public_api.py @@ -47,7 +47,7 @@ def _group_rank(s: str) -> int: "ZarrV3ArrayMetadata", "ZarrV3ArrayMetadataUpdate", "ZarrV2GroupMetadata", - "ZarrV2GroupMetadataPartial", + "ZarrV2GroupMetadataUpdate", "ZarrV3GroupMetadata", "ZarrV3GroupMetadataUpdate", "ZarrV3ConsolidatedMetadataInput", From 042aaa6c27731df88ccf641ffa424b4db77972a3 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Thu, 8 Oct 2026 12:53:18 +0200 Subject: [PATCH 70/94] feat(zarr-metadata)!: make v2 consolidated metadata a pair that holds each node as a model ZarrV2ConsolidatedMetadata(document, context=None) keeps its entries as written and adds nodes: each .zarray or .zgroup entry, merged with its .zattrs, as a model of the consolidated scope, keyed by node path. Equality, refines, with_context and refined_in go through the nodes. A problem in an entry is reported under that entry. Assisted-by: ClaudeCode:claude-fable-5-1 Co-Authored-By: Claude Fable 5.1 --- .../src/zarr_metadata/model/_group.py | 333 ++++++++++++++---- .../tests/model/test_construction.py | 29 +- .../zarr-metadata/tests/model/test_group.py | 9 +- .../zarr-metadata/tests/model/test_pair_v2.py | 79 +++++ 4 files changed, 366 insertions(+), 84 deletions(-) diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_group.py b/packages/zarr-metadata/src/zarr_metadata/model/_group.py index 773eb97352..bf2b366953 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_group.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/_group.py @@ -4,9 +4,9 @@ import dataclasses from collections.abc import Callable, Mapping -from dataclasses import dataclass, field +from dataclasses import dataclass from types import MappingProxyType -from typing import TYPE_CHECKING, Any, Final, Literal, TypeGuard, TypeVar, cast +from typing import TYPE_CHECKING, Any, Final, Literal, TypeAlias, TypeGuard, TypeVar, cast from typing_extensions import TypeAliasType, TypedDict, Unpack @@ -31,6 +31,7 @@ from zarr_metadata._json import prefixed as _prefix from zarr_metadata._sentinel import UNSET from zarr_metadata.model._array import ( + ZarrV2ArrayMetadata, ZarrV3ArrayMetadata, located_conflicts, must_understand_subset, @@ -44,17 +45,19 @@ ZarrV3ArrayMetadataReading, attributes_of, check_literal, - construct, dump_store_json, load_store_json, members_past_the_levels, missing_keys, other_members, parse_group_metadata_v2, + read_array_v2, read_array_v3, reading_of, unexpected_keys, + validate_group_metadata_v2, ) +from zarr_metadata.v2.array import ZARR_V2_ARRAY_METADATA_STORE_KEY from zarr_metadata.v2.attributes import ZARR_V2_ATTRIBUTES_STORE_KEY from zarr_metadata.v2.consolidated import ZARR_V2_CONSOLIDATED_METADATA_STORE_KEY from zarr_metadata.v2.definition import CORE_V2 @@ -1365,97 +1368,222 @@ def group_key_v2(model: ZarrV2GroupMetadata) -> tuple[object, ...]: return (UNSET if attributes is UNSET else json_text(attributes),) -@dataclass(frozen=True, slots=True, kw_only=True) +ZarrV2NodeMetadata: TypeAlias = "ZarrV2ArrayMetadata | ZarrV2GroupMetadata" +"""The model of one node a v2 `.zmetadata` document holds: an array, or a group.""" + + class ZarrV2ConsolidatedMetadata: - """In-memory model of a v2 `.zmetadata` document. - - The `metadata` map holds the flat file-keyed entries (`"path/.zarray"`, - `"path/.zattrs"`, ...) verbatim, preserving the normalized JSON tree. - Entries are deliberately NOT merged into per-node models: which nodes had - a `.zattrs` file at all is information the canonical representation must - keep. Interpreting entries into node models is consumer work. A model - checks itself when it is built, as `from_json` checks a document, or - the constructor raises `MetadataValidationError`. + """A v2 `.zmetadata` document, and the scope its nodes were read in. + + `metadata` holds the flat file-keyed entries (`"path/.zarray"`, + `"path/.zattrs"`, ...) as written, refined: which nodes had a + `.zattrs` at all is kept. `nodes` is each `.zarray` or `.zgroup` + entry, merged with its sibling `.zattrs`, as a model of this scope, + keyed by the node's path, `""` for the root; a `.zattrs` with no + sibling is kept and makes no node, and any other entry is JSON, kept. + Built only by reading: the constructor raises `MetadataValidationError` + with every problem, each located under its entry. Two documents are + equal when each node means the same and the other entries are written + alike, as `consolidated_key_v2` says; `refines`, `with_context` and + `refined_in` go through the nodes. """ - zarr_consolidated_format: Literal[1] = field(default=1, init=False) - metadata: dict[str, JSONValue] + __slots__ = ("_context", "_document", "_key", "_nodes", "_shown") + + _context: Context + _document: dict[str, JSONValue] + _key: tuple[object, ...] + _nodes: dict[str, ZarrV2NodeMetadata] + _shown: object - def __post_init__(self) -> None: - # Held as a read refines them, in containers of its own. - document = {"zarr_consolidated_format": 1, "metadata": self.metadata} - refined, problems = _read_consolidated_v2(document) + zarr_consolidated_format: Final = 1 + + def __init__(self, document: object, context: Context | None = None) -> None: + scope = CORE_V2 if context is None else context + entries, nodes, problems = _read_consolidated_v2(document, scope) if len(problems) != 0: raise MetadataValidationError(problems) - object.__setattr__(self, "metadata", refined) + self._adopt({"zarr_consolidated_format": 1, "metadata": entries}, scope, nodes) + + @classmethod + def _of( + cls, document: dict[str, JSONValue], context: Context, nodes: dict[str, ZarrV2NodeMetadata] + ) -> ZarrV2ConsolidatedMetadata: + """A model of a document a read found nothing wrong with, holding the nodes that read built.""" + model = object.__new__(cls) + model._adopt(document, context, nodes) + return model + + def _adopt( + self, document: dict[str, JSONValue], context: Context, nodes: dict[str, ZarrV2NodeMetadata] + ) -> None: + self._document = document + self._context = context + self._nodes = nodes + self._shown = frozen(document["metadata"]) + self._key = consolidated_key_v2(self) + + @property + def context(self) -> Context: + """The scope the nodes were read in.""" + return self._context + + @property + def metadata(self) -> Mapping[str, JSONValue]: + """The entries as written, refined, by store key; read-only at every level.""" + return cast("Mapping[str, JSONValue]", self._shown) + + @property + def nodes(self) -> Mapping[str, ZarrV2NodeMetadata]: + """The model of each node, by its path below the root, `""` for the root: a read-only view.""" + return MappingProxyType(self._nodes) + + def to_json(self) -> dict[str, JSONValue]: + """The `.zmetadata` document as written, refined, sharing nothing with the model.""" + return cast("dict[str, JSONValue]", copied(self._document)) + + def to_key_value( + self, *, indent: int | str | None = None + ) -> Mapping[ZarrV2ConsolidatedMetadataStoreKey, bytes]: + """The document as a store holds it: JSON bytes at `.zmetadata`, indented by `indent`.""" + return { + ZARR_V2_CONSOLIDATED_METADATA_STORE_KEY: dump_store_json(self._document, indent=indent) + } + + def __repr__(self) -> str: + return f"{type(self).__name__}({self._document!r}, context={self._context!r})" def __eq__(self, other: object) -> bool: - """Whether `other` holds the same document: the same JSON text; equal ones hash alike.""" if type(other) is not type(self): return NotImplemented - return json_text(self.to_json()) == json_text( - cast("ZarrV2ConsolidatedMetadata", other).to_json() - ) + return self._key == cast("ZarrV2ConsolidatedMetadata", other)._key def __hash__(self) -> int: - return hash(json_text(self.to_json())) + return hash(self._key) - def to_json(self) -> dict[str, JSONValue]: - """The `.zmetadata` document as JSON, sharing no mutable state with the model.""" - # to_json output shares no mutable state with the model. - return { - "zarr_consolidated_format": self.zarr_consolidated_format, - "metadata": copied(self.metadata), - } + def __reduce__(self) -> tuple[type[ZarrV2ConsolidatedMetadata], tuple[object, Context]]: + return type(self), (self._document, self._context) - @classmethod - def from_json(cls, data: object) -> ZarrV2ConsolidatedMetadata: - """The model of `data`, a `.zmetadata` document, its entries held as written. + def with_context(self, context: Context | None = None) -> ZarrV2ConsolidatedMetadata: + """This document with every node read in `context`, whatever that changes; `MetadataValidationError` when a node has a problem there.""" + scope = CORE_V2 if context is None else context + return type(self)(self._document, context=scope) + + def refined_in(self, context: Context | None = None) -> ZarrV2ConsolidatedMetadata: + """This document with every node read in `context`, which may claim what this scope left unclaimed and contradict nothing. - `MetadataValidationError` with every problem: a member missing or - unexpected, a format other than 1, an entry that is not JSON. A - `.zattrs` entry is user data, and may hold `NaN`, `Infinity` and - `-Infinity`. + `ScopeConflictError` naming each conflict, located at the node's + entry; `MetadataValidationError` when a gain surfaces a problem. """ - refined, problems = _read_consolidated_v2(data) - if len(problems) != 0: - raise MetadataValidationError(problems) - return construct(cls, metadata=refined) + scope = CORE_V2 if context is None else context + conflicts: list[Conflict] = [] + for path, node in self._nodes.items(): + if not isinstance(node, ZarrV2ArrayMetadata): + continue + found = scope.disagreements(node.claims) + key = ( + f"{path}/{ZARR_V2_ARRAY_METADATA_STORE_KEY}" + if path != "" + else ZARR_V2_ARRAY_METADATA_STORE_KEY + ) + conflicts.extend( + dataclasses.replace( + conflict, + loc=("metadata", key, *(() if conflict.loc is None else conflict.loc)), + ) + for conflict in located_conflicts(node.reading.fields(), found.conflicts) + ) + if len(conflicts) != 0: + raise ScopeConflictError(conflicts) + return self.with_context(scope) + + def refines(self, other: ZarrV2ConsolidatedMetadata) -> bool: + """Whether every node this holds refines the one `other` holds at the same path, neither holds a path the other does not, and the other entries are written alike; False of what is not v2 consolidated metadata.""" + if type(other) is not type(self): + return False + if self._nodes.keys() != other._nodes.keys(): + return False + if _other_entries_text(self) != _other_entries_text(other): + return False + return all(_v2_node_refines(self._nodes[path], other._nodes[path]) for path in self._nodes) @classmethod - def from_key_value(cls, mapping: Mapping[StoreKey, bytes]) -> ZarrV2ConsolidatedMetadata: - """The model of the document at `.zmetadata` in `mapping`. + def from_json( + cls, data: object, *, context: Context | None = None + ) -> ZarrV2ConsolidatedMetadata: + """The model of `data`, a `.zmetadata` document, its nodes read in `context`; `MetadataValidationError` with every problem.""" + return cls(data, context=context) + + @classmethod + def from_key_value( + cls, mapping: Mapping[StoreKey, bytes], *, context: Context | None = None + ) -> ZarrV2ConsolidatedMetadata: + """The model of the document at `.zmetadata` in `mapping`, read in `context`. `MetadataValidationError` when the key is missing, its bytes are not JSON, or the document is not valid. """ - return cls.from_json(load_store_json(mapping, ZARR_V2_CONSOLIDATED_METADATA_STORE_KEY)) + return cls( + load_store_json(mapping, ZARR_V2_CONSOLIDATED_METADATA_STORE_KEY), context=context + ) - def to_key_value( - self, *, indent: int | str | None = None - ) -> Mapping[ZarrV2ConsolidatedMetadataStoreKey, bytes]: - """The document as a store holds it: JSON bytes at `.zmetadata`, indented by `indent`. - A model was checked when it was built, so its document is written - as it is. - """ - return { - ZARR_V2_CONSOLIDATED_METADATA_STORE_KEY: dump_store_json(self.to_json(), indent=indent) - } +def _v2_node_refines(node: ZarrV2NodeMetadata, other: ZarrV2NodeMetadata) -> bool: + """Whether `node` refines `other`, as models of one kind refine each other; models of two kinds do not.""" + if isinstance(node, ZarrV2ArrayMetadata): + return isinstance(other, ZarrV2ArrayMetadata) and node.refines(other) + return isinstance(other, ZarrV2GroupMetadata) and node.refines(other) -def _read_consolidated_v2( - data: object, -) -> tuple[dict[str, JSONValue], tuple[ValidationProblem, ...]]: - """`data`, a `.zmetadata` document: each entry as read, and every problem, located in the document. +_NODE_FILES: Final = ( + ZARR_V2_ARRAY_METADATA_STORE_KEY, + ZARR_V2_GROUP_METADATA_STORE_KEY, + ZARR_V2_ATTRIBUTES_STORE_KEY, +) + + +def _entries_by_path(entries: Mapping[str, JSONValue]) -> dict[str, dict[str, str]]: + """The `.zarray`, `.zgroup` and `.zattrs` entries, by node path, then by file: the key each sits under.""" + by_path: dict[str, dict[str, str]] = {} + for key in entries: + path, _, name = key.rpartition("/") + if name in _NODE_FILES: + by_path.setdefault(path, {})[name] = key + return by_path + + +def _other_entries_text(model: ZarrV2ConsolidatedMetadata) -> str: + """The entries no node is read from -- an orphan `.zattrs`, any other key -- as JSON text: what `==` compares of them.""" + entries = cast("Mapping[str, JSONValue]", model._document["metadata"]) # pyright: ignore[reportPrivateUsage] + consumed: set[str] = set() + for path, names in _entries_by_path(entries).items(): + if path in model._nodes: # pyright: ignore[reportPrivateUsage] + consumed.update(names.values()) + return json_text({key: value for key, value in entries.items() if key not in consumed}) + + +def consolidated_key_v2(model: ZarrV2ConsolidatedMetadata) -> tuple[object, ...]: + """What `==` and `hash` compare of v2 consolidated metadata: each node by its path and its own key, and every other entry as JSON text.""" + nodes = model._nodes # pyright: ignore[reportPrivateUsage] + return ( + tuple(sorted((path, node._key) for path, node in nodes.items())), # pyright: ignore[reportPrivateUsage] + _other_entries_text(model), + ) + - Each entry is the document its key names: a `.zattrs` is user data, and - any other is JSON by RFC 8259. +def _read_consolidated_v2( + data: object, context: Context +) -> tuple[dict[str, JSONValue], dict[str, ZarrV2NodeMetadata], tuple[ValidationProblem, ...]]: + """`data`, a `.zmetadata` document: each entry as read, the model of each node read in `context`, and every problem, located in the document. + + Each entry is the document its key names: a `.zattrs` is user data, + any other is JSON by RFC 8259; a `.zarray` or `.zgroup` is read as the + document it is, merged with its sibling `.zattrs`, its problems under + its entry and the attributes' under the `.zattrs` entry. The nodes are + empty when anything is wrong. """ - # Each entry is refined below, arrays as tuples, no deeper than a - # reader walks; the document around them is walked as it is. if not isinstance(data, Mapping): - return {}, not_an_object(data) + return {}, {}, not_an_object(data) doc = cast("Mapping[object, object]", data) problems: list[ValidationProblem] = [ ValidationProblem((key,), "missing required key", "missing_key") @@ -1485,7 +1613,80 @@ def _read_consolidated_v2( entry, found = refine(value, ("metadata", key)) problems.extend(found) refined[key] = entry - return refined, with_input(problems, doc) + nodes: dict[str, ZarrV2NodeMetadata] = {} + if len(problems) == 0: + for path, names in _entries_by_path(refined).items(): + node, found = _read_node_v2(path, names, refined, context) + problems.extend(found) + if node is not None: + nodes[path] = node + if len(problems) != 0: + nodes = {} + return refined, nodes, with_input(problems, doc) + + +def _read_node_v2( + path: str, names: Mapping[str, str], entries: Mapping[str, JSONValue], context: Context +) -> tuple[ZarrV2NodeMetadata | None, list[ValidationProblem]]: + """The node at `path`, read from its `.zarray` or `.zgroup` entry merged with its `.zattrs`, and every problem, located under the entries; None and no problem when there is only a `.zattrs`.""" + zarray = names.get(ZARR_V2_ARRAY_METADATA_STORE_KEY) + zgroup = names.get(ZARR_V2_GROUP_METADATA_STORE_KEY) + zattrs = names.get(ZARR_V2_ATTRIBUTES_STORE_KEY) + if zarray is not None and zgroup is not None: + return None, [ + ValidationProblem( + ("metadata", zgroup), + f"a node is an array or a group, not both: a {ZARR_V2_ARRAY_METADATA_STORE_KEY} is at " + f"{path!r} too", + "invalid_value", + ) + ] + key = zarray if zarray is not None else zgroup + if key is None: + return None, [] + document = entries[key] + if not isinstance(document, Mapping): + return None, [ + ValidationProblem( + ("metadata", key), f"expected an object, got {shown(document)}", "invalid_type" + ) + ] + merged = dict(cast("Mapping[str, JSONValue]", document)) + if "attributes" in merged: + return None, [ + ValidationProblem( + ("metadata", key, "attributes"), "unexpected document member", "invalid_value" + ) + ] + if zattrs is not None: + merged["attributes"] = entries[zattrs] + + def located(found: tuple[ValidationProblem, ...]) -> list[ValidationProblem]: + placed: list[ValidationProblem] = [] + for problem in found: + if zattrs is not None and problem.loc[:1] == ("attributes",): + placed.append( + dataclasses.replace(problem, loc=("metadata", zattrs, *problem.loc[1:])) + ) + else: + placed.append(dataclasses.replace(problem, loc=("metadata", key, *problem.loc))) + return placed + + if zarray is not None: + reading, members = read_array_v2(merged, context) + if members is None: + return None, located(reading.problems) + held = merged if "dimension_separator" in merged else {**merged, "dimension_separator": "."} + return ZarrV2ArrayMetadata._of(held, context, reading, members), [] # pyright: ignore[reportPrivateUsage] + found = validate_group_metadata_v2(merged, context=context) + if len(found) != 0: + return None, located(found) + attributes: dict[str, JSONValue] | UNSET = ( + dict(cast("Mapping[str, JSONValue]", merged["attributes"])) + if "attributes" in merged + else UNSET + ) + return ZarrV2GroupMetadata._of(merged, context, attributes), [] # pyright: ignore[reportPrivateUsage] def _v2_attributes(document: ZarrV2GroupMetadataJSON) -> dict[str, JSONValue] | UNSET: diff --git a/packages/zarr-metadata/tests/model/test_construction.py b/packages/zarr-metadata/tests/model/test_construction.py index cc30d2a107..3f62b56327 100644 --- a/packages/zarr-metadata/tests/model/test_construction.py +++ b/packages/zarr-metadata/tests/model/test_construction.py @@ -64,6 +64,12 @@ ) +def _rebuilt(model: object, changes: dict[str, object]) -> Any: # noqa: ANN401 - the tests give the model as object + """`model`, a v2 model, built again of its document with `changes` in place of members, in its own scope: what `update` does, for the consolidated model too.""" + held = cast("Any", model) + return type(held)({**held.to_json(), **changes}, context=held.context) + + _INLINE: dict[str, Any] = {"kind": "inline", "must_understand": False} @@ -83,7 +89,7 @@ def test_a_v3_model_is_built_of_its_document_in_its_scope(model: object) -> None [ V2_ARRAY, V2_GROUP, - pytest.param(V2_CONSOLIDATED, marks=pytest.mark.xfail(strict=True, reason="Task 4")), + V2_CONSOLIDATED, ], ids=["v2-array", "v2-group", "v2-consolidated"], ) @@ -115,19 +121,15 @@ def test_a_v3_model_holds_its_members_as_a_read_refines_them( ("model", "changes"), [ (V2_ARRAY, {"shape": range(4, 5), "chunks": [4], "attributes": {"a": 1}}), - pytest.param( - V2_CONSOLIDATED, - {"metadata": {"a/.zattrs": UserDict({"x": 1})}}, - marks=pytest.mark.xfail(strict=True, reason="Task 4"), - ), + (V2_CONSOLIDATED, {"metadata": {"a/.zattrs": UserDict({"x": 1})}}), ], ids=["v2-sequences", "v2-consolidated-mapping"], ) def test_a_v2_model_holds_its_members_as_a_read_refines_them( model: object, changes: dict[str, object] ) -> None: - """Arrays as tuples and objects as dicts, as the read holds them, so a v2 model updated with other containers is the model a read builds.""" - changed = cast("Any", model).update(**changes) + """Arrays as tuples and objects as dicts, as the read holds them, so a v2 model built again with other containers is the model a read builds.""" + changed = _rebuilt(model, changes) assert changed == model assert type(changed).from_key_value(changed.to_key_value()) == changed @@ -158,9 +160,7 @@ def test_a_v3_model_shares_no_container_with_what_it_was_built_of( [ (V2_ARRAY, "attributes"), (V2_GROUP, "attributes"), - pytest.param( - V2_CONSOLIDATED, "metadata", marks=pytest.mark.xfail(strict=True, reason="Task 4") - ), + (V2_CONSOLIDATED, "metadata"), ], ids=["v2-array-attributes", "v2-group-attributes", "v2-consolidated"], ) @@ -169,7 +169,7 @@ def test_a_v2_model_shares_no_container_with_what_it_was_built_of( ) -> None: """A v2 model holds copies of the containers it is built of.""" held: dict[str, object] = {"acme.x": {"must_understand": False}} - built = cast("Any", model).update(**{member: held}) + built = _rebuilt(model, {member: held}) held["acme.y"] = math.nan cast("dict[str, object]", held["acme.x"])["z"] = math.nan assert getattr(built, member) == {"acme.x": {"must_understand": False}} @@ -268,11 +268,10 @@ def test_error_consolidated_metadata_of_documents_at_bad_paths_is_refused() -> N (V2_ARRAY, {"order": "Q"}, [(("order",), "invalid_value")]), (V2_ARRAY, {"chunks": (4, 4)}, [(("chunks",), "invalid_value")]), (V2_GROUP, {"attributes": {1: "a"}}, [(("attributes",), "invalid_type")]), - pytest.param( + ( V2_CONSOLIDATED, {"metadata": {"a/.zarray": {"x": math.nan}}}, [(("metadata", "a/.zarray", "x"), "invalid_value")], - marks=pytest.mark.xfail(strict=True, reason="Task 4"), ), ], ids=[ @@ -287,7 +286,7 @@ def test_error_a_v2_model_changed_by_hand_into_an_invalid_one_is_refused_at_the_ ) -> None: """As a read reads its document: no v2 model is invalid, however it came to be.""" with pytest.raises(MetadataValidationError) as raised: - cast("Any", model).update(**changes) + _rebuilt(model, changes) assert [(found.loc, found.kind) for found in raised.value.problems] == problems diff --git a/packages/zarr-metadata/tests/model/test_group.py b/packages/zarr-metadata/tests/model/test_group.py index c1db0e63fd..f367fa3d68 100644 --- a/packages/zarr-metadata/tests/model/test_group.py +++ b/packages/zarr-metadata/tests/model/test_group.py @@ -1136,10 +1136,10 @@ def test_consolidated_v2_lists_become_tuples() -> None: """from_json converts JSON arrays inside entries to tuples.""" doc = { "zarr_consolidated_format": 1, - "metadata": {"a/.zarray": {"shape": [2, 3]}}, + "metadata": {"a/.zattrs": {"shape": [2, 3]}}, } model = ZarrV2ConsolidatedMetadata.from_json(doc) - assert model.metadata == {"a/.zarray": {"shape": (2, 3)}} + assert model.metadata == {"a/.zattrs": {"shape": (2, 3)}} def test_consolidated_v2_envelope_validation() -> None: @@ -1368,7 +1368,10 @@ def test_error_a_null_consolidated_metadata_is_a_value_the_document_wrote() -> N id="v3-consolidated", ), pytest.param( - ZarrV2ConsolidatedMetadata(metadata={"a/.zarray": {"nested": {"x": [1]}}}), + # A `.zattrs` entry: user data, whose containers can nest. + ZarrV2ConsolidatedMetadata( + {"zarr_consolidated_format": 1, "metadata": {"a/.zattrs": {"nested": {"x": [1]}}}} + ), id="v2-consolidated", ), ] diff --git a/packages/zarr-metadata/tests/model/test_pair_v2.py b/packages/zarr-metadata/tests/model/test_pair_v2.py index 5c4a2f394e..2e58c8fac5 100644 --- a/packages/zarr-metadata/tests/model/test_pair_v2.py +++ b/packages/zarr-metadata/tests/model/test_pair_v2.py @@ -13,6 +13,7 @@ from zarr_metadata.model import ( MetadataValidationError, ZarrV2ArrayMetadata, + ZarrV2ConsolidatedMetadata, ZarrV2GroupMetadata, read_array_metadata_v2, ) @@ -290,3 +291,81 @@ def test_error_a_group_document_with_a_problem_is_refused_at_construction() -> N (("attributes",), "invalid_type"), (("extra",), "unknown_key"), ] + + +ZARRAY = {k: v for k, v in ARRAY.items() if k != "attributes"} +CONSOLIDATED: dict[str, Any] = { + "zarr_consolidated_format": 1, + "metadata": { + ".zgroup": {"zarr_format": 2}, + ".zattrs": {"root": True}, + "a/.zarray": ZARRAY, + "a/.zattrs": {"a": [1, 2]}, + "b/.zgroup": {"zarr_format": 2}, + "orphan/.zattrs": {"o": 1}, + }, +} + + +def _consolidated(**entries: object) -> dict[str, Any]: + return {**CONSOLIDATED, "metadata": {**CONSOLIDATED["metadata"], **entries}} + + +def test_consolidated_metadata_holds_its_entries_verbatim_and_each_node_as_a_model() -> None: + """`metadata` is the flat file-keyed map as written, refined, read-only; `nodes` is each `.zarray`/`.zgroup` entry merged with its sibling `.zattrs` as a model of the consolidated scope, keyed by node path (`""` for the root); a `.zattrs` with no sibling is kept and makes no node; the whole round-trips through its document, store keys, pickle and copy.""" + model = ZarrV2ConsolidatedMetadata(CONSOLIDATED, PRIVATE) + assert model.context is PRIVATE + assert model.metadata["a/.zattrs"] == {"a": (1, 2)} + assert set(model.nodes) == {"", "a", "b"} + assert model.nodes["a"] == ZarrV2ArrayMetadata(ARRAY, PRIVATE) + root = ZarrV2GroupMetadata({"zarr_format": 2, "attributes": {"root": True}}, PRIVATE) + assert model.nodes[""] == root + assert model.nodes["b"].attributes is UNSET + entries = cast("dict[str, Any]", model.to_json()["metadata"]) + assert entries["orphan/.zattrs"] == {"o": 1} + assert ZarrV2ConsolidatedMetadata.from_key_value(model.to_key_value(), context=PRIVATE) == model + assert pickle.loads(pickle.dumps(model)) == model + assert copy.deepcopy(model) == model + with pytest.raises(TypeError): + cast("dict[str, object]", model.metadata)["x"] = 1 + assert not dataclasses.is_dataclass(ZarrV2ConsolidatedMetadata) + + +def test_consolidated_metadata_is_equal_by_its_nodes_and_moves_scope_with_them() -> None: + """Two consolidated documents are one when each node means the same and the other entries are written alike; `with_context`/`refined_in` read every node in the new scope, a conflict located at the node's entry; `refines` holds when each node refines its counterpart.""" + spelled = _consolidated(**{"a/.zarray": {**ZARRAY, "fill_value": 0.0}}) + assert ZarrV2ConsolidatedMetadata(CONSOLIDATED) == ZarrV2ConsolidatedMetadata(spelled) + other = _consolidated(**{"orphan/.zattrs": {"o": 2}}) + assert ZarrV2ConsolidatedMetadata(CONSOLIDATED) != ZarrV2ConsolidatedMetadata(other) + small = ZarrV2ConsolidatedMetadata(CONSOLIDATED, SMALL) + assert isinstance(small.nodes["a"], ZarrV2ArrayMetadata) + assert isinstance(small.nodes["a"].compressor, Unclaimed) + gained = small.refined_in(PRIVATE) + assert isinstance(gained.nodes["a"], ZarrV2ArrayMetadata) + assert isinstance(gained.nodes["a"].compressor, Read) + assert gained.refines(small) + assert not small.refines(gained) + with pytest.raises(ScopeConflictError) as conflict: + gained.refined_in(CORE_V2) + assert [c.loc for c in conflict.value.conflicts] == [("metadata", "a/.zarray", "compressor")] + assert gained.with_context(SMALL) == small + + +@pytest.mark.parametrize( + ("entries", "at"), + [ + ({"a/.zarray": {**ZARRAY, "dtype": "float32"}}, ("metadata", "a/.zarray", "dtype")), + ({"a/.zarray": ZARRAY, "a/.zattrs": {1: 2}}, ("metadata", "a/.zattrs")), + ({"a/.zarray": ZARRAY, "a/.zgroup": {"zarr_format": 2}}, ("metadata", "a/.zgroup")), + ({"a/.zgroup": {"zarr_format": 3}}, ("metadata", "a/.zgroup", "zarr_format")), + ({"a/.zarray": 3}, ("metadata", "a/.zarray")), + ], + ids=["array-dtype", "zattrs-key", "array-and-group", "group-format", "not-an-object"], +) +def test_error_a_node_entry_with_a_problem_is_refused_at_the_entry( + entries: dict[str, Any], at: Loc +) -> None: + """A `.zarray` or `.zgroup` entry is read as the document it is, in the consolidated scope; its problems sit under the entry, a `.zattrs`'s under its own entry; a path that is both an array and a group is a problem at the `.zgroup`.""" + with pytest.raises(MetadataValidationError) as raised: + ZarrV2ConsolidatedMetadata({"zarr_consolidated_format": 1, "metadata": entries}) + assert raised.value.problems[0].loc == at From d904570d904791b8dfc9231555a6e51e2cbf63a9 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Thu, 8 Oct 2026 12:57:41 +0200 Subject: [PATCH 71/94] feat(zarr-metadata)!: v2 pydantic field types read in a scope; remove construct The v2 pydantic field types read a document in the scope the validation context holds, CORE_V2 when it holds none. construct is gone: every model is built from its document. Assisted-by: ClaudeCode:claude-fable-5-1 Co-Authored-By: Claude Fable 5.1 --- .../changes/v2-models.feature.md | 1 + .../changes/v2-models.removal.md | 1 + .../src/zarr_metadata/model/__init__.py | 3 +- .../src/zarr_metadata/model/_validation.py | 33 +------------ .../src/zarr_metadata/pydantic.py | 46 +++++++++---------- .../tests/model/test_construction.py | 9 +--- .../tests/model/test_pydantic_module.py | 18 ++++++++ 7 files changed, 46 insertions(+), 65 deletions(-) create mode 100644 packages/zarr-metadata/changes/v2-models.feature.md create mode 100644 packages/zarr-metadata/changes/v2-models.removal.md diff --git a/packages/zarr-metadata/changes/v2-models.feature.md b/packages/zarr-metadata/changes/v2-models.feature.md new file mode 100644 index 0000000000..601a90feff --- /dev/null +++ b/packages/zarr-metadata/changes/v2-models.feature.md @@ -0,0 +1 @@ +The v2 models are now a document and the scope it was read in, as the v3 models are: `ZarrV2ArrayMetadata(document, context=None)` reads the document in `CORE_V2` or the scope given, holds `dtype`, `compressor` and each filter as the scope read them, and compares by what the document means, so ` Iterator[tuple[Loc, Resolved[Any]]]: yield from fields_of(transformer, ("storage_transformers", index)) -M = TypeVar("M") - - def reading_of(model: object) -> object: - """The reading `model`, a v3 model, holds: what a pickled reading that held a model is built again as.""" + """The reading `model`, a model, holds: what a pickled reading that held a model is built again as.""" return cast("Any", model).reading -def construct(model: type[M], /, **members: object) -> M: - """A `model` of `members` a read found nothing wrong with: as its constructor builds one, without checking them again. - - What pydantic's `model_construct` is to its `__init__`. A model checks - itself when it is built; a read has checked what it read already, so - the model it builds is not checked twice. A member not given, or one - the constructor does not take, is its default. - """ - built = object.__new__(model) - declared = dataclasses.fields(cast("Any", model)) - unknown = members.keys() - {member.name for member in declared if member.init} - if len(unknown) != 0: - msg = f"{model.__name__} has no member {sorted(unknown)!r} to build" - raise TypeError(msg) - for member in declared: - if member.init and member.name in members: - value = members[member.name] - elif member.default is not dataclasses.MISSING: - value = member.default - elif member.default_factory is not dataclasses.MISSING: - value = member.default_factory() - else: - msg = f"{model.__name__} is built with {member.name!r}" - raise TypeError(msg) - object.__setattr__(built, member.name, value) - return built - - @dataclass(frozen=True, slots=True) class ArrayMembersV3: """The members of a v3 array document a read found nothing wrong with, other than its fields, refined as the model holds them.""" diff --git a/packages/zarr-metadata/src/zarr_metadata/pydantic.py b/packages/zarr-metadata/src/zarr_metadata/pydantic.py index 9c436a2598..862e6cc468 100644 --- a/packages/zarr-metadata/src/zarr_metadata/pydantic.py +++ b/packages/zarr-metadata/src/zarr_metadata/pydantic.py @@ -74,6 +74,7 @@ class ArrayManifest(BaseModel): ZarrV3GroupMetadataJSON as _ZarrV3GroupMetadataSchema, ) from zarr_metadata._sentinel import UNSET +from zarr_metadata.v2.definition import CORE_V2 from zarr_metadata.v3._registry import CORE_AND_EXTENSIONS, Context if TYPE_CHECKING: @@ -82,17 +83,6 @@ class ArrayManifest(BaseModel): _M = TypeVar("_M") -def _coerce_to(cls: type[_M], parse: Callable[[object], _M]) -> Callable[[object], _M]: - """A validator that passes instances of `cls` through and parses anything else, each problem a line error of pydantic's.""" - - def coerce(value: object) -> _M: - if isinstance(value, cls): - return value - return _as_pydantic_raises(cls, value, lambda: parse(value)) - - return coerce - - def _as_pydantic_raises(cls: type[_M], value: object, read: Callable[[], _M]) -> _M: """What `read` gives of `value`, or the `MetadataValidationError` it raises as a `pydantic_core.ValidationError`: one line error per problem, its type the problem's kind, at the problem's loc, with its input and ctx. @@ -151,23 +141,27 @@ class _Reads(Protocol[_Read_co]): def __call__(self, data: object, /, *, context: Context) -> _Read_co: ... -def _read_in_scope(cls: type[_M], read: _Reads[_M]) -> Callable[[object, ValidationInfo], _M]: - """A validator that passes instances of `cls` through and reads anything else in the scope the validation context holds.""" +def _read_in_scope( + cls: type[_M], read: _Reads[_M], default: Context +) -> Callable[[object, ValidationInfo], _M]: + """A validator that passes instances of `cls` through and reads anything else in the scope the validation context holds, `default` when it holds none.""" def coerce(value: object, info: ValidationInfo) -> _M: if isinstance(value, cls): return value - return _as_pydantic_raises(cls, value, lambda: read(value, context=_scope(info.context))) + return _as_pydantic_raises( + cls, value, lambda: read(value, context=_scope(info.context, default)) + ) return coerce -def _scope(context: object) -> Context: - """The scope a validation context holds: itself, a `Context`; its `CONTEXT_KEY` item; or, holding none, `CORE_AND_EXTENSIONS`.""" +def _scope(context: object, default: Context) -> Context: + """The scope a validation context holds: itself, a `Context`; its `CONTEXT_KEY` item; or, holding none, `default`: the format's own scope.""" if isinstance(context, Context): return context if not isinstance(context, Mapping) or CONTEXT_KEY not in context: - return CORE_AND_EXTENSIONS + return default scope = cast("Mapping[object, object]", context)[CONTEXT_KEY] if not isinstance(scope, Context): msg = f"{CONTEXT_KEY}: the scope to read in is a Context, got {scope!r}" @@ -178,7 +172,9 @@ def _scope(context: object) -> Context: ZarrV3ArrayMetadata = Annotated[ InstanceOf[_model.ZarrV3ArrayMetadata], BeforeValidator( - _read_in_scope(_model.ZarrV3ArrayMetadata, _model.ZarrV3ArrayMetadata.from_json), + _read_in_scope( + _model.ZarrV3ArrayMetadata, _model.ZarrV3ArrayMetadata.from_json, CORE_AND_EXTENSIONS + ), json_schema_input_type=_ZarrV3ArrayMetadataSchema, ), PlainSerializer(_model.ZarrV3ArrayMetadata.to_json, return_type=_ZarrV3ArrayMetadataSchema), @@ -188,7 +184,7 @@ def _scope(context: object) -> Context: ZarrV2ArrayMetadata = Annotated[ InstanceOf[_model.ZarrV2ArrayMetadata], BeforeValidator( - _coerce_to(_model.ZarrV2ArrayMetadata, _model.ZarrV2ArrayMetadata.from_json), + _read_in_scope(_model.ZarrV2ArrayMetadata, _model.ZarrV2ArrayMetadata.from_json, CORE_V2), json_schema_input_type=_ZarrV2ArrayMetadataSchema, ), PlainSerializer(_model.ZarrV2ArrayMetadata.to_json, return_type=_ZarrV2ArrayMetadataSchema), @@ -198,7 +194,9 @@ def _scope(context: object) -> Context: ZarrV3GroupMetadata = Annotated[ InstanceOf[_model.ZarrV3GroupMetadata], BeforeValidator( - _read_in_scope(_model.ZarrV3GroupMetadata, _model.ZarrV3GroupMetadata.from_json), + _read_in_scope( + _model.ZarrV3GroupMetadata, _model.ZarrV3GroupMetadata.from_json, CORE_AND_EXTENSIONS + ), json_schema_input_type=_ZarrV3GroupMetadataSchema, ), PlainSerializer(_model.ZarrV3GroupMetadata.to_json, return_type=_ZarrV3GroupMetadataSchema), @@ -208,7 +206,7 @@ def _scope(context: object) -> Context: ZarrV2GroupMetadata = Annotated[ InstanceOf[_model.ZarrV2GroupMetadata], BeforeValidator( - _coerce_to(_model.ZarrV2GroupMetadata, _model.ZarrV2GroupMetadata.from_json), + _read_in_scope(_model.ZarrV2GroupMetadata, _model.ZarrV2GroupMetadata.from_json, CORE_V2), json_schema_input_type=_ZarrV2GroupMetadataSchema, ), PlainSerializer(_model.ZarrV2GroupMetadata.to_json, return_type=_ZarrV2GroupMetadataSchema), @@ -221,6 +219,7 @@ def _scope(context: object) -> Context: _read_in_scope( _model.ZarrV3ConsolidatedMetadata, _model.ZarrV3ConsolidatedMetadata.from_json, + CORE_AND_EXTENSIONS, ), json_schema_input_type=_ZarrV3ConsolidatedMetadataSchema, ), @@ -234,9 +233,8 @@ def _scope(context: object) -> Context: ZarrV2ConsolidatedMetadata = Annotated[ InstanceOf[_model.ZarrV2ConsolidatedMetadata], BeforeValidator( - _coerce_to( - _model.ZarrV2ConsolidatedMetadata, - _model.ZarrV2ConsolidatedMetadata.from_json, + _read_in_scope( + _model.ZarrV2ConsolidatedMetadata, _model.ZarrV2ConsolidatedMetadata.from_json, CORE_V2 ), json_schema_input_type=_ZarrV2ConsolidatedMetadataSchema, ), diff --git a/packages/zarr-metadata/tests/model/test_construction.py b/packages/zarr-metadata/tests/model/test_construction.py index 3f62b56327..e0ec182f07 100644 --- a/packages/zarr-metadata/tests/model/test_construction.py +++ b/packages/zarr-metadata/tests/model/test_construction.py @@ -26,9 +26,8 @@ ZarrV3ConsolidatedMetadata, ZarrV3GroupMetadata, ) -from zarr_metadata.model._validation import ZarrV3ArrayMetadataReading, construct from zarr_metadata.v3.data_type.int8 import INT8_DATA_TYPE -from zarr_metadata.v3.definition import CORE_AND_EXTENSIONS, Chunk +from zarr_metadata.v3.definition import CORE_AND_EXTENSIONS if TYPE_CHECKING: from collections.abc import Iterator @@ -307,9 +306,3 @@ def test_error_consolidated_metadata_paths_are_strings() -> None: assert [(found.loc, found.kind) for found in raised.value.problems] == [ (("metadata",), "invalid_type") ] - - -def test_construct_fills_a_member_from_its_default_factory() -> None: - reading = construct(ZarrV3ArrayMetadataReading, problems=()) - assert reading.chunk == Chunk() - assert reading.pipeline == () diff --git a/packages/zarr-metadata/tests/model/test_pydantic_module.py b/packages/zarr-metadata/tests/model/test_pydantic_module.py index 9aae2c6948..a77907f096 100644 --- a/packages/zarr-metadata/tests/model/test_pydantic_module.py +++ b/packages/zarr-metadata/tests/model/test_pydantic_module.py @@ -436,3 +436,21 @@ def test_the_pydantic_schema_names_an_extension_as_the_reader_does(field: object assert not Draft202012Validator(schema).is_valid(named) with pytest.raises(ValidationError): TypeAdapter(zmp.ZarrV3ArrayMetadata).validate_python(named) + + +def test_v2_field_types_read_in_the_validation_contexts_scope() -> None: + """The v2 field types read a document in the scope the validation context holds -- itself a `Context`, or its `zarr_metadata_context` item -- and in `CORE_V2` when it holds none, as the v3 field types read in theirs.""" + from zarr_metadata.v2.data_type.scalar import UINT_V2 + from zarr_metadata.v2.definition import CORE_V2, Context, Unclaimed + + adapter = TypeAdapter(zmp.ZarrV2ArrayMetadata) + doc = json.loads(json.dumps(V2_ARRAY_DOC)) + assert adapter.validate_python(doc).context is CORE_V2 + small = Context.of(UINT_V2) + read = adapter.validate_python({**doc, "compressor": {"id": "zlib"}}, context=small) + assert read.context is small + assert isinstance(read.compressor, Unclaimed) + held = adapter.validate_python(doc, context={"zarr_metadata_context": small}) + assert held.context is small + group = TypeAdapter(zmp.ZarrV2GroupMetadata).validate_python(V2_GROUP_DOC, context=small) + assert group.context is small From e207cd8a34d0b7d2772a676796ff1abd5acd3abd Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Thu, 8 Oct 2026 13:32:56 +0200 Subject: [PATCH 72/94] fix(zarr-metadata): repair the member zarr-python 3.x writes into .zgroup entries; a scope key per format zarr-python 3.x writes a consolidated_metadata member into each .zgroup entry below the root of a v2 .zmetadata, which the strict read refuses. repair_consolidated_metadata_v2 removes it and read_repaired_consolidated_metadata_v2 reads the repaired document. The pydantic field types read the v3 scope under zarr_metadata_context and the v2 scope under zarr_metadata_context_v2; a bare Context is the scope of every field type. Two keys naming one file of one consolidated node are a problem. ZarrV2NodeMetadata is exported. Assisted-by: ClaudeCode:claude-fable-5-1 Co-Authored-By: Claude Fable 5.1 --- .../src/zarr_metadata/model/__init__.py | 8 ++ .../src/zarr_metadata/model/_group.py | 40 +++++-- .../src/zarr_metadata/model/_repair.py | 108 +++++++++++++++++- .../src/zarr_metadata/pydantic.py | 64 ++++++++--- .../zarr-metadata/tests/model/test_pair_v2.py | 24 ++++ .../tests/model/test_pydantic_module.py | 25 +++- .../zarr-metadata/tests/model/test_repair.py | 92 +++++++++++++++ 7 files changed, 327 insertions(+), 34 deletions(-) diff --git a/packages/zarr-metadata/src/zarr_metadata/model/__init__.py b/packages/zarr-metadata/src/zarr_metadata/model/__init__.py index b85ca0c835..3a6db67f38 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/__init__.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/__init__.py @@ -55,6 +55,7 @@ ZarrV2ConsolidatedMetadata, ZarrV2GroupMetadata, ZarrV2GroupMetadataUpdate, + ZarrV2NodeMetadata, ZarrV3ConsolidatedMetadata, ZarrV3ConsolidatedMetadataInput, ZarrV3GroupMetadata, @@ -77,12 +78,15 @@ from zarr_metadata.model._repair import ( Repair, RepairKind, + ZarrV2RepairedConsolidatedMetadataReading, ZarrV3NullConsolidatedGroupMetadataJSON, ZarrV3RepairedNodeMetadataReading, ZarrV3ZeroChunkArrayMetadataJSON, ZarrV3ZeroChunkRegularGridConfigurationJSON, ZarrV3ZeroChunkRegularGridJSON, + read_repaired_consolidated_metadata_v2, read_repaired_node_metadata_v3, + repair_consolidated_metadata_v2, repair_node_metadata_v3, ) from zarr_metadata.model._validation import ( @@ -181,6 +185,8 @@ "ZarrV2GroupMetadata", "ZarrV2GroupMetadataStoreKey", "ZarrV2GroupMetadataUpdate", + "ZarrV2NodeMetadata", + "ZarrV2RepairedConsolidatedMetadataReading", "ZarrV3ArrayMetadata", "ZarrV3ArrayMetadataReading", "ZarrV3ArrayMetadataStoreKey", @@ -223,7 +229,9 @@ "read_array_metadata_v3", "read_group_metadata_v3", "read_node_metadata_v3", + "read_repaired_consolidated_metadata_v2", "read_repaired_node_metadata_v3", + "repair_consolidated_metadata_v2", "repair_node_metadata_v3", "validate_array_metadata_v2", "validate_array_metadata_v3", diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_group.py b/packages/zarr-metadata/src/zarr_metadata/model/_group.py index bf2b366953..095ea7e090 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_group.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/_group.py @@ -1477,15 +1477,13 @@ def refined_in(self, context: Context | None = None) -> ZarrV2ConsolidatedMetada """ scope = CORE_V2 if context is None else context conflicts: list[Conflict] = [] + entries = cast("Mapping[str, JSONValue]", self._document["metadata"]) + by_path, _ = _entries_by_path(entries) for path, node in self._nodes.items(): if not isinstance(node, ZarrV2ArrayMetadata): continue found = scope.disagreements(node.claims) - key = ( - f"{path}/{ZARR_V2_ARRAY_METADATA_STORE_KEY}" - if path != "" - else ZARR_V2_ARRAY_METADATA_STORE_KEY - ) + key = by_path[path][ZARR_V2_ARRAY_METADATA_STORE_KEY] conflicts.extend( dataclasses.replace( conflict, @@ -1542,21 +1540,35 @@ def _v2_node_refines(node: ZarrV2NodeMetadata, other: ZarrV2NodeMetadata) -> boo ) -def _entries_by_path(entries: Mapping[str, JSONValue]) -> dict[str, dict[str, str]]: - """The `.zarray`, `.zgroup` and `.zattrs` entries, by node path, then by file: the key each sits under.""" +def _entries_by_path( + entries: Mapping[str, JSONValue], +) -> tuple[dict[str, dict[str, str]], list[ValidationProblem]]: + """The `.zarray`, `.zgroup` and `.zattrs` entries, by node path, then by file: the key each sits under; and a problem for each second key naming one file of one node, which would otherwise go unread.""" by_path: dict[str, dict[str, str]] = {} + problems: list[ValidationProblem] = [] for key in entries: path, _, name = key.rpartition("/") - if name in _NODE_FILES: - by_path.setdefault(path, {})[name] = key - return by_path + if name not in _NODE_FILES: + continue + files = by_path.setdefault(path, {}) + if name in files: + problems.append( + ValidationProblem( + ("metadata", key), + f"a second {name} for the node at {path!r}, which {files[name]!r} is", + "invalid_value", + ) + ) + continue + files[name] = key + return by_path, problems def _other_entries_text(model: ZarrV2ConsolidatedMetadata) -> str: """The entries no node is read from -- an orphan `.zattrs`, any other key -- as JSON text: what `==` compares of them.""" entries = cast("Mapping[str, JSONValue]", model._document["metadata"]) # pyright: ignore[reportPrivateUsage] consumed: set[str] = set() - for path, names in _entries_by_path(entries).items(): + for path, names in _entries_by_path(entries)[0].items(): if path in model._nodes: # pyright: ignore[reportPrivateUsage] consumed.update(names.values()) return json_text({key: value for key, value in entries.items() if key not in consumed}) @@ -1614,8 +1626,12 @@ def _read_consolidated_v2( problems.extend(found) refined[key] = entry nodes: dict[str, ZarrV2NodeMetadata] = {} + by_path: dict[str, dict[str, str]] = {} + if len(problems) == 0: + by_path, doubled = _entries_by_path(refined) + problems.extend(doubled) if len(problems) == 0: - for path, names in _entries_by_path(refined).items(): + for path, names in by_path.items(): node, found = _read_node_v2(path, names, refined, context) problems.extend(found) if node is not None: diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_repair.py b/packages/zarr-metadata/src/zarr_metadata/model/_repair.py index 9b4bac4f81..09e8034c10 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_repair.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/_repair.py @@ -21,13 +21,24 @@ from annotated_types import Ge -from zarr_metadata._json import JSON_DEPTH +from zarr_metadata._common import ( + JSONValue, # noqa: TC001 - a TypedDict's annotations are evaluated at run time +) +from zarr_metadata._json import JSON_DEPTH, MetadataValidationError, ValidationProblem from zarr_metadata._typed_json import Loc, check -from zarr_metadata.model._group import ZarrV3NodeMetadataReading, read_node_metadata_v3 +from zarr_metadata.model._group import ( + ZarrV2ConsolidatedMetadata, + ZarrV3NodeMetadataReading, + read_node_metadata_v3, +) +from zarr_metadata.v2.definition import CORE_V2 +from zarr_metadata.v2.group import ZARR_V2_GROUP_METADATA_STORE_KEY from zarr_metadata.v3._registry import CORE_AND_EXTENSIONS, Context from zarr_metadata.v3.consolidated import ZARR_V3_CONSOLIDATED_METADATA_KEY -RepairKind: TypeAlias = Literal["zero_chunk_length", "null_consolidated_metadata"] +RepairKind: TypeAlias = Literal[ + "zero_chunk_length", "null_consolidated_metadata", "consolidated_metadata_in_zgroup_entry" +] """Each writer bug a repair undoes, by name.""" @@ -221,14 +232,105 @@ def read_repaired_node_metadata_v3( ) +class ZarrV2ZGroupWithConsolidatedMetadataJSON(TypedDict): + """A `.zgroup` entry of a `.zmetadata` as zarr-python 3.x writes one below the root: with a `consolidated_metadata` member, which a v2 group document does not take.""" + + zarr_format: Literal[2] + consolidated_metadata: Mapping[str, JSONValue] + + +def repair_consolidated_metadata_v2(value: object) -> tuple[object, tuple[Repair, ...]]: + """`value`, a v2 `.zmetadata`, with each known writer bug in it undone, and what was changed. + + zarr-python 3.x writes a `consolidated_metadata` member into each + `.zgroup` entry below the root, which is removed. What no repair + applies to is left as it is, and `value` is not changed; a document + with none of the bugs is given back, and no repairs. + """ + if not isinstance(value, Mapping): + return value, () + document = cast("Mapping[str, object]", value) + entries = document.get("metadata") + if not isinstance(entries, Mapping): + return cast("object", value), () + repairs: list[Repair] = [] + held: dict[object, object] = {} + for key, entry in cast("Mapping[object, object]", entries).items(): + if isinstance(key, str) and key.rsplit("/", 1)[-1] == ZARR_V2_GROUP_METADATA_STORE_KEY: + entry = _without_consolidated_metadata(entry, ("metadata", key), repairs) + held[key] = entry + if len(repairs) == 0: + return cast("object", value), () + return {**document, "metadata": held}, tuple(repairs) + + +def _without_consolidated_metadata(entry: object, at: Loc, repairs: list[Repair]) -> object: + """`entry`, a `.zgroup` entry, without the member zarr-python 3.x writes into it, when it is one such.""" + if not isinstance(entry, Mapping): + return entry + group = cast("Mapping[str, object]", entry) + shaped, problems = check( + _members(group, ("zarr_format", ZARR_V3_CONSOLIDATED_METADATA_KEY)), + ZarrV2ZGroupWithConsolidatedMetadataJSON, + ) + if shaped is None or len(problems) != 0: + return cast("object", entry) + repairs.append( + Repair( + (*at, ZARR_V3_CONSOLIDATED_METADATA_KEY), + "consolidated_metadata_in_zgroup_entry", + "a consolidated_metadata, as zarr-python 3.x writes into a .zgroup entry of a " + ".zmetadata, removed", + ) + ) + kept: dict[str, object] = { + key: item for key, item in group.items() if key != ZARR_V3_CONSOLIDATED_METADATA_KEY + } + return kept + + +@dataclass(frozen=True, slots=True) +class ZarrV2RepairedConsolidatedMetadataReading: + """A v2 `.zmetadata` read after its known writer bugs were undone: the repaired document's problems, its model when there are none, and the repairs.""" + + problems: tuple[ValidationProblem, ...] + """Every problem of the repaired document.""" + metadata: ZarrV2ConsolidatedMetadata | None + """The repaired document's model, when it has no problem; None otherwise.""" + repairs: tuple[Repair, ...] + """What was changed to make the document that was read.""" + + +def read_repaired_consolidated_metadata_v2( + value: object, *, context: Context | None = None +) -> ZarrV2RepairedConsolidatedMetadataReading: + """`value`, a v2 `.zmetadata`, read in `context` as `ZarrV2ConsolidatedMetadata` reads it, once `repair_consolidated_metadata_v2` has undone each known writer bug in it. + + For a reader of stores other writers made, which asks for repairs by + calling this rather than the strict model. Whatever no repair applies + to is read as it is, and reported as the strict read reports it. + """ + scope = CORE_V2 if context is None else context + repaired, repairs = repair_consolidated_metadata_v2(value) + try: + model: ZarrV2ConsolidatedMetadata | None = ZarrV2ConsolidatedMetadata(repaired, scope) + except MetadataValidationError as error: + return ZarrV2RepairedConsolidatedMetadataReading(error.problems, None, repairs) + return ZarrV2RepairedConsolidatedMetadataReading((), model, repairs) + + __all__ = [ "Repair", "RepairKind", + "ZarrV2RepairedConsolidatedMetadataReading", + "ZarrV2ZGroupWithConsolidatedMetadataJSON", "ZarrV3NullConsolidatedGroupMetadataJSON", "ZarrV3RepairedNodeMetadataReading", "ZarrV3ZeroChunkArrayMetadataJSON", "ZarrV3ZeroChunkRegularGridConfigurationJSON", "ZarrV3ZeroChunkRegularGridJSON", + "read_repaired_consolidated_metadata_v2", "read_repaired_node_metadata_v3", + "repair_consolidated_metadata_v2", "repair_node_metadata_v3", ] diff --git a/packages/zarr-metadata/src/zarr_metadata/pydantic.py b/packages/zarr-metadata/src/zarr_metadata/pydantic.py index 862e6cc468..3890753cc9 100644 --- a/packages/zarr-metadata/src/zarr_metadata/pydantic.py +++ b/packages/zarr-metadata/src/zarr_metadata/pydantic.py @@ -8,14 +8,18 @@ freely with non-pydantic code (equality, isinstance, nesting). Validation delegates to the library: a raw document routes through `from_json` (the single source of truth for validation and normalization, so pydantic's -field-level coercion can never bypass it). A v3 field type reads extension -points in the scope pydantic's validation context holds, as pydantic hands -any validator its context: the context itself, when it is a `Context`, or -its `"zarr_metadata_context"` item, when it is a mapping; otherwise -`CORE_AND_EXTENSIONS`: +field-level coercion can never bypass it). A field type reads the fields +of its document in the scope pydantic's validation context holds, as +pydantic hands any validator its context: the context itself, when it is +a `Context`, which every field type then reads in; or, when it is a +mapping, its `"zarr_metadata_context"` item for the v3 field types and +its `"zarr_metadata_context_v2"` item for the v2 ones, so a model holding +both kinds of field names each format's scope; a format whose scope is +not given reads in its own, `CORE_AND_EXTENSIONS` or `CORE_V2`: TypeAdapter(zmp.ZarrV3ArrayMetadata).validate_python(document, context=SCOPE) ArrayManifest.model_validate(data, context={"zarr_metadata_context": SCOPE}) + Mixed.model_validate(data, context={"zarr_metadata_context": V3, "zarr_metadata_context_v2": V2}) An existing model instance passes through unchanged, as pydantic's does, and serialization emits the canonical document via `to_json`. A failed parse @@ -131,6 +135,9 @@ def _input_of(problem: ValidationProblem, value: object) -> object: CONTEXT_KEY: Final = "zarr_metadata_context" +"""The key of a mapping validation context under which the scope the v3 field types read in sits.""" +CONTEXT_KEY_V2: Final = "zarr_metadata_context_v2" +"""The key of a mapping validation context under which the scope the v2 field types read in sits.""" """The item of a mapping pydantic's validation context is that holds the scope a v3 field type reads in.""" @@ -142,29 +149,29 @@ def __call__(self, data: object, /, *, context: Context) -> _Read_co: ... def _read_in_scope( - cls: type[_M], read: _Reads[_M], default: Context + cls: type[_M], read: _Reads[_M], default: Context, key: str ) -> Callable[[object, ValidationInfo], _M]: - """A validator that passes instances of `cls` through and reads anything else in the scope the validation context holds, `default` when it holds none.""" + """A validator that passes instances of `cls` through and reads anything else in the scope the validation context holds under `key`, `default` when it holds none.""" def coerce(value: object, info: ValidationInfo) -> _M: if isinstance(value, cls): return value return _as_pydantic_raises( - cls, value, lambda: read(value, context=_scope(info.context, default)) + cls, value, lambda: read(value, context=_scope(info.context, default, key)) ) return coerce -def _scope(context: object, default: Context) -> Context: - """The scope a validation context holds: itself, a `Context`; its `CONTEXT_KEY` item; or, holding none, `default`: the format's own scope.""" +def _scope(context: object, default: Context, key: str) -> Context: + """The scope a validation context holds for one format: itself, a `Context`, the scope of every field type; its `key` item, the format's own; or, holding none, `default`, the format's core scope.""" if isinstance(context, Context): return context - if not isinstance(context, Mapping) or CONTEXT_KEY not in context: + if not isinstance(context, Mapping) or key not in context: return default - scope = cast("Mapping[object, object]", context)[CONTEXT_KEY] + scope = cast("Mapping[object, object]", context)[key] if not isinstance(scope, Context): - msg = f"{CONTEXT_KEY}: the scope to read in is a Context, got {scope!r}" + msg = f"{key}: the scope to read in is a Context, got {scope!r}" raise TypeError(msg) return scope @@ -173,7 +180,10 @@ def _scope(context: object, default: Context) -> Context: InstanceOf[_model.ZarrV3ArrayMetadata], BeforeValidator( _read_in_scope( - _model.ZarrV3ArrayMetadata, _model.ZarrV3ArrayMetadata.from_json, CORE_AND_EXTENSIONS + _model.ZarrV3ArrayMetadata, + _model.ZarrV3ArrayMetadata.from_json, + CORE_AND_EXTENSIONS, + CONTEXT_KEY, ), json_schema_input_type=_ZarrV3ArrayMetadataSchema, ), @@ -184,7 +194,12 @@ def _scope(context: object, default: Context) -> Context: ZarrV2ArrayMetadata = Annotated[ InstanceOf[_model.ZarrV2ArrayMetadata], BeforeValidator( - _read_in_scope(_model.ZarrV2ArrayMetadata, _model.ZarrV2ArrayMetadata.from_json, CORE_V2), + _read_in_scope( + _model.ZarrV2ArrayMetadata, + _model.ZarrV2ArrayMetadata.from_json, + CORE_V2, + CONTEXT_KEY_V2, + ), json_schema_input_type=_ZarrV2ArrayMetadataSchema, ), PlainSerializer(_model.ZarrV2ArrayMetadata.to_json, return_type=_ZarrV2ArrayMetadataSchema), @@ -195,7 +210,10 @@ def _scope(context: object, default: Context) -> Context: InstanceOf[_model.ZarrV3GroupMetadata], BeforeValidator( _read_in_scope( - _model.ZarrV3GroupMetadata, _model.ZarrV3GroupMetadata.from_json, CORE_AND_EXTENSIONS + _model.ZarrV3GroupMetadata, + _model.ZarrV3GroupMetadata.from_json, + CORE_AND_EXTENSIONS, + CONTEXT_KEY, ), json_schema_input_type=_ZarrV3GroupMetadataSchema, ), @@ -206,7 +224,12 @@ def _scope(context: object, default: Context) -> Context: ZarrV2GroupMetadata = Annotated[ InstanceOf[_model.ZarrV2GroupMetadata], BeforeValidator( - _read_in_scope(_model.ZarrV2GroupMetadata, _model.ZarrV2GroupMetadata.from_json, CORE_V2), + _read_in_scope( + _model.ZarrV2GroupMetadata, + _model.ZarrV2GroupMetadata.from_json, + CORE_V2, + CONTEXT_KEY_V2, + ), json_schema_input_type=_ZarrV2GroupMetadataSchema, ), PlainSerializer(_model.ZarrV2GroupMetadata.to_json, return_type=_ZarrV2GroupMetadataSchema), @@ -220,6 +243,7 @@ def _scope(context: object, default: Context) -> Context: _model.ZarrV3ConsolidatedMetadata, _model.ZarrV3ConsolidatedMetadata.from_json, CORE_AND_EXTENSIONS, + CONTEXT_KEY, ), json_schema_input_type=_ZarrV3ConsolidatedMetadataSchema, ), @@ -234,7 +258,10 @@ def _scope(context: object, default: Context) -> Context: InstanceOf[_model.ZarrV2ConsolidatedMetadata], BeforeValidator( _read_in_scope( - _model.ZarrV2ConsolidatedMetadata, _model.ZarrV2ConsolidatedMetadata.from_json, CORE_V2 + _model.ZarrV2ConsolidatedMetadata, + _model.ZarrV2ConsolidatedMetadata.from_json, + CORE_V2, + CONTEXT_KEY_V2, ), json_schema_input_type=_ZarrV2ConsolidatedMetadataSchema, ), @@ -247,6 +274,7 @@ def _scope(context: object, default: Context) -> Context: __all__ = [ "CONTEXT_KEY", + "CONTEXT_KEY_V2", "ZarrV2ArrayMetadata", "ZarrV2ConsolidatedMetadata", "ZarrV2GroupMetadata", diff --git a/packages/zarr-metadata/tests/model/test_pair_v2.py b/packages/zarr-metadata/tests/model/test_pair_v2.py index 2e58c8fac5..fe67d4c942 100644 --- a/packages/zarr-metadata/tests/model/test_pair_v2.py +++ b/packages/zarr-metadata/tests/model/test_pair_v2.py @@ -369,3 +369,27 @@ def test_error_a_node_entry_with_a_problem_is_refused_at_the_entry( with pytest.raises(MetadataValidationError) as raised: ZarrV2ConsolidatedMetadata({"zarr_consolidated_format": 1, "metadata": entries}) assert raised.value.problems[0].loc == at + + +def test_error_two_entries_for_one_node_file_are_refused() -> None: + """Two keys that name one file of one node -- `.zarray` and `/.zarray` -- are a problem at the second, since one would otherwise go unread; and a conflict is located at the key the document writes.""" + with pytest.raises(MetadataValidationError) as raised: + ZarrV2ConsolidatedMetadata( + {"zarr_consolidated_format": 1, "metadata": {".zarray": ZARRAY, "/.zarray": ZARRAY}} + ) + assert [(p.loc, p.kind) for p in raised.value.problems] == [ + (("metadata", "/.zarray"), "invalid_value") + ] + gained = ZarrV2ConsolidatedMetadata( + {"zarr_consolidated_format": 1, "metadata": {"/.zarray": ZARRAY}}, PRIVATE + ) + with pytest.raises(ScopeConflictError) as conflict: + gained.refined_in(CORE_V2) + assert [c.loc for c in conflict.value.conflicts] == [("metadata", "/.zarray", "compressor")] + + +def test_the_node_type_is_exported() -> None: + """`ZarrV2NodeMetadata`, the type of each node `nodes` holds, is exported beside the models, as `ZarrV3NodeMetadata` is.""" + from zarr_metadata import model + + assert "ZarrV2NodeMetadata" in model.__all__ diff --git a/packages/zarr-metadata/tests/model/test_pydantic_module.py b/packages/zarr-metadata/tests/model/test_pydantic_module.py index a77907f096..c75f4a9833 100644 --- a/packages/zarr-metadata/tests/model/test_pydantic_module.py +++ b/packages/zarr-metadata/tests/model/test_pydantic_module.py @@ -450,7 +450,30 @@ def test_v2_field_types_read_in_the_validation_contexts_scope() -> None: read = adapter.validate_python({**doc, "compressor": {"id": "zlib"}}, context=small) assert read.context is small assert isinstance(read.compressor, Unclaimed) - held = adapter.validate_python(doc, context={"zarr_metadata_context": small}) + held = adapter.validate_python(doc, context={"zarr_metadata_context_v2": small}) assert held.context is small group = TypeAdapter(zmp.ZarrV2GroupMetadata).validate_python(V2_GROUP_DOC, context=small) assert group.context is small + + +def test_each_format_reads_in_its_own_context_key() -> None: + """A mapping context names each format's scope by its own key -- `zarr_metadata_context` for v3, `zarr_metadata_context_v2` for v2 -- so a v3 scope given for the v3 fields leaves the v2 fields in `CORE_V2`; a bare `Context` is the scope of every field type.""" + from zarr_metadata.v2.data_type.scalar import UINT_V2 + from zarr_metadata.v2.definition import CORE_V2, Context, Unclaimed + from zarr_metadata.v3.definition import CORE_AND_EXTENSIONS + + adapter = TypeAdapter(zmp.ZarrV2ArrayMetadata) + doc = json.loads(json.dumps(V2_ARRAY_DOC)) + small = Context.of(UINT_V2) + v3_only = adapter.validate_python(doc, context={"zarr_metadata_context": CORE_AND_EXTENSIONS}) + assert v3_only.context is CORE_V2 + with pytest.raises(ValidationError): + adapter.validate_python( + {**doc, "fill_value": "garbage"}, context={"zarr_metadata_context": CORE_AND_EXTENSIONS} + ) + own = adapter.validate_python(doc, context={"zarr_metadata_context_v2": small}) + assert own.context is small + bare = adapter.validate_python({**doc, "compressor": {"id": "zlib"}}, context=small) + assert bare.context is small + assert isinstance(bare.compressor, Unclaimed) + assert zmp.CONTEXT_KEY_V2 == "zarr_metadata_context_v2" diff --git a/packages/zarr-metadata/tests/model/test_repair.py b/packages/zarr-metadata/tests/model/test_repair.py index 7eeff9ad86..bf36a2fe39 100644 --- a/packages/zarr-metadata/tests/model/test_repair.py +++ b/packages/zarr-metadata/tests/model/test_repair.py @@ -8,10 +8,14 @@ import pytest from zarr_metadata.model import ( + MetadataValidationError, + ZarrV2ConsolidatedMetadata, ZarrV3ArrayMetadata, ZarrV3GroupMetadata, read_node_metadata_v3, + read_repaired_consolidated_metadata_v2, read_repaired_node_metadata_v3, + repair_consolidated_metadata_v2, repair_node_metadata_v3, ) @@ -129,3 +133,91 @@ def test_read_repaired_reports_what_no_repair_applies_to() -> None: (("data_type",), "missing_key") ] assert repaired.reading.metadata is None + + +# --- v2 --------------------------------------------------------------------- + +ZGROUP_FROM_ZARR3: dict[str, Any] = { + "zarr_format": 2, + "consolidated_metadata": {"metadata": {}, "must_understand": False, "kind": "inline"}, +} +ZMETADATA_FROM_ZARR3: dict[str, Any] = { + "zarr_consolidated_format": 1, + "metadata": { + ".zgroup": {"zarr_format": 2}, + "b/.zgroup": ZGROUP_FROM_ZARR3, + "b/.zattrs": {"x": 1}, + }, +} +ZMETADATA_CLEAN: dict[str, Any] = { + "zarr_consolidated_format": 1, + "metadata": { + ".zgroup": {"zarr_format": 2}, + "b/.zgroup": {"zarr_format": 2}, + "b/.zattrs": {"x": 1}, + }, +} + + +@pytest.mark.parametrize( + ("value", "repaired", "repairs"), + [ + ( + ZMETADATA_FROM_ZARR3, + ZMETADATA_CLEAN, + [ + ( + ("metadata", "b/.zgroup", "consolidated_metadata"), + "consolidated_metadata_in_zgroup_entry", + ) + ], + ), + (ZMETADATA_CLEAN, ZMETADATA_CLEAN, []), + ( + { + "zarr_consolidated_format": 1, + "metadata": {"b/.zgroup": {"zarr_format": 2, "consolidated_metadata": 3}}, + }, + { + "zarr_consolidated_format": 1, + "metadata": {"b/.zgroup": {"zarr_format": 2, "consolidated_metadata": 3}}, + }, + [], + ), + (3, 3, []), + ], + ids=["zarr-3-zgroup-entry", "clean", "not-the-bug", "not-a-document"], +) +def test_repair_v2_undoes_the_consolidated_metadata_zarr_3_writes_into_a_zgroup_entry( + value: object, repaired: object, repairs: list[tuple[tuple[str | int, ...], str]] +) -> None: + """zarr-python 3.x writes a `consolidated_metadata` member into each non-root `.zgroup` entry of a `.zmetadata`, which the v2 group document does not take; `repair_consolidated_metadata_v2` removes it and says where; anything else is left as it is, and the document handed in is not changed.""" + before = copy.deepcopy(value) + document, made = repair_consolidated_metadata_v2(value) + assert document == repaired + assert [(repair.loc, repair.kind) for repair in made] == repairs + assert value == before + if len(repairs) == 0: + assert document is value + + +def test_read_repaired_consolidated_v2_reads_what_the_strict_read_refuses() -> None: + """The strict `ZarrV2ConsolidatedMetadata` refuses the zarr-python 3.x `.zmetadata` at the entry's member; `read_repaired_consolidated_metadata_v2` reads the repaired document, holds its model, and lists the repairs.""" + with pytest.raises(MetadataValidationError) as raised: + ZarrV2ConsolidatedMetadata(ZMETADATA_FROM_ZARR3) + assert [p.loc for p in raised.value.problems] == [ + ("metadata", "b/.zgroup", "consolidated_metadata") + ] + repaired = read_repaired_consolidated_metadata_v2(ZMETADATA_FROM_ZARR3) + assert repaired.problems == () + assert repaired.metadata == ZarrV2ConsolidatedMetadata(ZMETADATA_CLEAN) + assert [r.kind for r in repaired.repairs] == ["consolidated_metadata_in_zgroup_entry"] + broken = read_repaired_consolidated_metadata_v2( + { + **ZMETADATA_FROM_ZARR3, + "metadata": {**ZMETADATA_FROM_ZARR3["metadata"], "b/.zattrs": {1: "x"}}, + } + ) + assert broken.metadata is None + assert [p.loc for p in broken.problems] == [("metadata", "b/.zattrs")] + assert [r.kind for r in broken.repairs] == ["consolidated_metadata_in_zgroup_entry"] From 874a46d6c28033c7c682aad7be5af2dcb1641aec Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Thu, 8 Oct 2026 13:40:02 +0200 Subject: [PATCH 73/94] fix(zarr-metadata): read every consolidated node beside a doubled key; leave the root .zgroup to the strict read A second key for one file of a v2 consolidated node no longer hides the other nodes' problems, and a leading slash names the same node. The repair leaves the root .zgroup alone, since no writer puts the member there. The fragments say how to read a zarr-python 3.x .zmetadata and which context key the v2 pydantic field types read. Assisted-by: ClaudeCode:claude-fable-5-1 Co-Authored-By: Claude Fable 5.1 --- .../changes/v2-models.feature.md | 2 +- .../src/zarr_metadata/model/__init__.py | 2 ++ .../src/zarr_metadata/model/_group.py | 5 ++- .../src/zarr_metadata/model/_repair.py | 10 ++++-- .../src/zarr_metadata/pydantic.py | 1 - .../zarr-metadata/tests/model/test_pair_v2.py | 33 +++++++++++++++++++ .../zarr-metadata/tests/model/test_repair.py | 8 ++++- 7 files changed, 54 insertions(+), 7 deletions(-) diff --git a/packages/zarr-metadata/changes/v2-models.feature.md b/packages/zarr-metadata/changes/v2-models.feature.md index 601a90feff..182c72003d 100644 --- a/packages/zarr-metadata/changes/v2-models.feature.md +++ b/packages/zarr-metadata/changes/v2-models.feature.md @@ -1 +1 @@ -The v2 models are now a document and the scope it was read in, as the v3 models are: `ZarrV2ArrayMetadata(document, context=None)` reads the document in `CORE_V2` or the scope given, holds `dtype`, `compressor` and each filter as the scope read them, and compares by what the document means, so ` tuple[object, tuple[Repair """`value`, a v2 `.zmetadata`, with each known writer bug in it undone, and what was changed. zarr-python 3.x writes a `consolidated_metadata` member into each - `.zgroup` entry below the root, which is removed. What no repair + `.zgroup` entry below the root, which is removed; the root's is left, + since no writer puts one there. What no repair applies to is left as it is, and `value` is not changed; a document with none of the bugs is given back, and no repairs. """ @@ -256,8 +257,11 @@ def repair_consolidated_metadata_v2(value: object) -> tuple[object, tuple[Repair repairs: list[Repair] = [] held: dict[object, object] = {} for key, entry in cast("Mapping[object, object]", entries).items(): - if isinstance(key, str) and key.rsplit("/", 1)[-1] == ZARR_V2_GROUP_METADATA_STORE_KEY: - entry = _without_consolidated_metadata(entry, ("metadata", key), repairs) + if isinstance(key, str): + path, _, name = key.rpartition("/") + # Below the root only: no writer puts the member in the root's .zgroup. + if name == ZARR_V2_GROUP_METADATA_STORE_KEY and path.strip("/") != "": + entry = _without_consolidated_metadata(entry, ("metadata", key), repairs) held[key] = entry if len(repairs) == 0: return cast("object", value), () diff --git a/packages/zarr-metadata/src/zarr_metadata/pydantic.py b/packages/zarr-metadata/src/zarr_metadata/pydantic.py index 3890753cc9..8c7bf064bf 100644 --- a/packages/zarr-metadata/src/zarr_metadata/pydantic.py +++ b/packages/zarr-metadata/src/zarr_metadata/pydantic.py @@ -138,7 +138,6 @@ def _input_of(problem: ValidationProblem, value: object) -> object: """The key of a mapping validation context under which the scope the v3 field types read in sits.""" CONTEXT_KEY_V2: Final = "zarr_metadata_context_v2" """The key of a mapping validation context under which the scope the v2 field types read in sits.""" -"""The item of a mapping pydantic's validation context is that holds the scope a v3 field type reads in.""" _Read_co = TypeVar("_Read_co", covariant=True) diff --git a/packages/zarr-metadata/tests/model/test_pair_v2.py b/packages/zarr-metadata/tests/model/test_pair_v2.py index fe67d4c942..d81d52259a 100644 --- a/packages/zarr-metadata/tests/model/test_pair_v2.py +++ b/packages/zarr-metadata/tests/model/test_pair_v2.py @@ -393,3 +393,36 @@ def test_the_node_type_is_exported() -> None: from zarr_metadata import model assert "ZarrV2NodeMetadata" in model.__all__ + + +def test_error_a_second_key_for_a_node_file_hides_no_other_problem() -> None: + """A second key naming one file of one node is one problem among the document's: every node is still read, and each node's problems reported, so a user sees everything at once; a leading `/` does not make a second node.""" + with pytest.raises(MetadataValidationError) as raised: + ZarrV2ConsolidatedMetadata( + { + "zarr_consolidated_format": 1, + "metadata": { + "a/.zarray": ZARRAY, + "/a/.zarray": ZARRAY, + "b/.zgroup": {"zarr_format": 3}, + }, + } + ) + assert [p.loc for p in raised.value.problems] == [ + ("metadata", "/a/.zarray"), + ("metadata", "b/.zgroup", "zarr_format"), + ] + one = ZarrV2ConsolidatedMetadata( + {"zarr_consolidated_format": 1, "metadata": {"/a/.zarray": ZARRAY}} + ) + assert set(one.nodes) == {"a"} + assert one == ZarrV2ConsolidatedMetadata( + {"zarr_consolidated_format": 1, "metadata": {"a/.zarray": ZARRAY}} + ) + + +def test_the_repair_shapes_are_exported() -> None: + """The JSON shape of each known v2 writer bug is exported beside the v3 ones.""" + from zarr_metadata import model + + assert "ZarrV2ZGroupWithConsolidatedMetadataJSON" in model.__all__ diff --git a/packages/zarr-metadata/tests/model/test_repair.py b/packages/zarr-metadata/tests/model/test_repair.py index bf36a2fe39..76432baa90 100644 --- a/packages/zarr-metadata/tests/model/test_repair.py +++ b/packages/zarr-metadata/tests/model/test_repair.py @@ -173,6 +173,12 @@ def test_read_repaired_reports_what_no_repair_applies_to() -> None: ], ), (ZMETADATA_CLEAN, ZMETADATA_CLEAN, []), + # The root's .zgroup: no zarr-python version writes the member there. + ( + {"zarr_consolidated_format": 1, "metadata": {".zgroup": ZGROUP_FROM_ZARR3}}, + {"zarr_consolidated_format": 1, "metadata": {".zgroup": ZGROUP_FROM_ZARR3}}, + [], + ), ( { "zarr_consolidated_format": 1, @@ -186,7 +192,7 @@ def test_read_repaired_reports_what_no_repair_applies_to() -> None: ), (3, 3, []), ], - ids=["zarr-3-zgroup-entry", "clean", "not-the-bug", "not-a-document"], + ids=["zarr-3-zgroup-entry", "clean", "root", "not-the-bug", "not-a-document"], ) def test_repair_v2_undoes_the_consolidated_metadata_zarr_3_writes_into_a_zgroup_entry( value: object, repaired: object, repairs: list[tuple[tuple[str | int, ...], str]] From e022f4aadba83abc62fe5ba9fb8e020f033b5f37 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Thu, 8 Oct 2026 13:44:01 +0200 Subject: [PATCH 74/94] fix(zarr-metadata): a repeated slash inside a consolidated node path names the same node Assisted-by: ClaudeCode:claude-fable-5-1 Co-Authored-By: Claude Fable 5.1 --- packages/zarr-metadata/src/zarr_metadata/model/_group.py | 4 ++-- packages/zarr-metadata/tests/model/test_pair_v2.py | 7 ++++++- 2 files changed, 8 insertions(+), 3 deletions(-) diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_group.py b/packages/zarr-metadata/src/zarr_metadata/model/_group.py index f2cb5b3913..4653db4ad1 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_group.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/_group.py @@ -1548,8 +1548,8 @@ def _entries_by_path( problems: list[ValidationProblem] = [] for key in entries: path, _, name = key.rpartition("/") - # A node's path, as a store names it: without a `/` at either end. - path = path.strip("/") + # A node's path, as a store names it: segments joined by one `/`. + path = "/".join(segment for segment in path.split("/") if segment != "") if name not in _NODE_FILES: continue files = by_path.setdefault(path, {}) diff --git a/packages/zarr-metadata/tests/model/test_pair_v2.py b/packages/zarr-metadata/tests/model/test_pair_v2.py index d81d52259a..b03ab20c8a 100644 --- a/packages/zarr-metadata/tests/model/test_pair_v2.py +++ b/packages/zarr-metadata/tests/model/test_pair_v2.py @@ -396,7 +396,7 @@ def test_the_node_type_is_exported() -> None: def test_error_a_second_key_for_a_node_file_hides_no_other_problem() -> None: - """A second key naming one file of one node is one problem among the document's: every node is still read, and each node's problems reported, so a user sees everything at once; a leading `/` does not make a second node.""" + """A second key naming one file of one node is one problem among the document's: every node is still read, and each node's problems reported, so a user sees everything at once; a leading or repeated `/` does not make a second node.""" with pytest.raises(MetadataValidationError) as raised: ZarrV2ConsolidatedMetadata( { @@ -415,6 +415,11 @@ def test_error_a_second_key_for_a_node_file_hides_no_other_problem() -> None: one = ZarrV2ConsolidatedMetadata( {"zarr_consolidated_format": 1, "metadata": {"/a/.zarray": ZARRAY}} ) + assert set( + ZarrV2ConsolidatedMetadata( + {"zarr_consolidated_format": 1, "metadata": {"x//y/.zarray": ZARRAY}} + ).nodes + ) == {"x/y"} assert set(one.nodes) == {"a"} assert one == ZarrV2ConsolidatedMetadata( {"zarr_consolidated_format": 1, "metadata": {"a/.zarray": ZARRAY}} From 79003858a53236fa8a6fe7670c80a01c8cd3a73c Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Thu, 8 Oct 2026 13:44:39 +0200 Subject: [PATCH 75/94] docs(zarr-metadata): name the v2 models fragments after their PR Assisted-by: ClaudeCode:claude-fable-5-1 Co-Authored-By: Claude Fable 5.1 --- .../changes/{v2-models.feature.md => 397.feature.md} | 0 .../changes/{v2-models.removal.md => 397.removal.md} | 0 2 files changed, 0 insertions(+), 0 deletions(-) rename packages/zarr-metadata/changes/{v2-models.feature.md => 397.feature.md} (100%) rename packages/zarr-metadata/changes/{v2-models.removal.md => 397.removal.md} (100%) diff --git a/packages/zarr-metadata/changes/v2-models.feature.md b/packages/zarr-metadata/changes/397.feature.md similarity index 100% rename from packages/zarr-metadata/changes/v2-models.feature.md rename to packages/zarr-metadata/changes/397.feature.md diff --git a/packages/zarr-metadata/changes/v2-models.removal.md b/packages/zarr-metadata/changes/397.removal.md similarity index 100% rename from packages/zarr-metadata/changes/v2-models.removal.md rename to packages/zarr-metadata/changes/397.removal.md From 44307163648d8d2fbba5a68884f7e9be92b4b661 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Thu, 8 Oct 2026 14:16:43 +0200 Subject: [PATCH 76/94] refactor(zarr-metadata)!: export less; say why the core has no validation framework The public doors keep what a user of a document, a model, a definition or a scope needs and no longer export the helpers behind them: the scope algebra's functions, the canonical-spelling and field-walking helpers, the writer-bug JSON shapes, the key sets and the JSON helpers. The README says the core depends on no validation framework and why. ruff knows a TypedDict's annotations are evaluated at run time, so the per-import noqa comments go. Assisted-by: ClaudeCode:claude-fable-5-1 Co-Authored-By: Claude Fable 5.1 --- packages/zarr-metadata/README.md | 13 +++ packages/zarr-metadata/changes/398.removal.md | 1 + packages/zarr-metadata/docs/api/index.md | 6 +- packages/zarr-metadata/pyproject.toml | 4 + .../src/zarr_metadata/_pydantic_schema.py | 4 +- .../src/zarr_metadata/model/__init__.py | 32 ------ .../src/zarr_metadata/model/_array.py | 39 ++++--- .../src/zarr_metadata/model/_group.py | 22 ++-- .../src/zarr_metadata/model/_repair.py | 2 +- .../src/zarr_metadata/typed_json.py | 4 - .../src/zarr_metadata/v2/codec/compression.py | 2 +- .../src/zarr_metadata/v2/data_type/object.py | 2 +- .../src/zarr_metadata/v2/data_type/time.py | 2 +- .../src/zarr_metadata/v2/definition.py | 8 -- .../src/zarr_metadata/v3/_common.py | 4 +- .../src/zarr_metadata/v3/definition.py | 103 ++++++------------ .../zarr-metadata/tests/model/test_array.py | 34 +++--- .../tests/model/test_consolidating.py | 4 +- .../tests/model/test_construction.py | 4 +- .../tests/model/test_extension_points.py | 7 +- .../zarr-metadata/tests/model/test_group.py | 4 +- .../zarr-metadata/tests/model/test_pair_v2.py | 11 +- .../tests/model/test_pydantic.py | 4 +- .../tests/model/test_pydantic_module.py | 5 +- .../tests/model/test_read_array_metadata.py | 6 +- .../model/test_read_array_metadata_v2.py | 4 +- .../tests/model/test_scope_threading.py | 9 +- .../zarr-metadata/tests/test_json_schema.py | 14 ++- .../zarr-metadata/tests/test_problem_data.py | 14 ++- .../zarr-metadata/tests/test_public_api.py | 1 - .../zarr-metadata/tests/test_typed_json.py | 9 +- .../zarr-metadata/tests/v2/test_codecs.py | 11 +- .../zarr-metadata/tests/v2/test_data_types.py | 6 +- .../zarr-metadata/tests/v2/test_definition.py | 6 +- packages/zarr-metadata/tests/v2/test_kinds.py | 4 +- .../tests/v3/test_definitions.py | 17 +-- .../tests/v3/test_every_definition.py | 8 +- .../tests/v3/test_fill_values.py | 5 +- .../tests/v3/test_grid_shapes.py | 9 +- packages/zarr-metadata/tests/v3/test_kinds.py | 9 +- .../zarr-metadata/tests/v3/test_pipelines.py | 9 +- packages/zarr-metadata/tests/v3/test_scope.py | 16 ++- .../zarr-metadata/tests/v3/test_sharding.py | 9 +- 43 files changed, 257 insertions(+), 230 deletions(-) create mode 100644 packages/zarr-metadata/changes/398.removal.md diff --git a/packages/zarr-metadata/README.md b/packages/zarr-metadata/README.md index 17edb0b2c1..782eed1972 100644 --- a/packages/zarr-metadata/README.md +++ b/packages/zarr-metadata/README.md @@ -53,6 +53,19 @@ A bare `TypeAdapter` over a public document `TypedDict` is a coercive shape adapter, not a Zarr conformance validator; it may coerce values or discard members that the strict model parser rejects. +## Dependencies + +The core of this package depends on `typing-extensions` and +`annotated-types` only. It does not depend on a validation framework, +and it will not: a dependency on pydantic, or any other framework with +its own release cadence and compiled parts, would pin that framework for +every consumer of zarr-metadata and collide with the pins consumers +already carry. Instead the package reads a `TypedDict` as the typing spec +defines it with its own checker (`zarr_metadata.typed_json.check`), and +spells bounds in the `annotated-types` vocabulary, which pydantic reads +too. `zarr_metadata.pydantic` is an optional integration over the models; +nothing in the core imports it. + ## Validation boundary The model validators enforce the declared document structure and a small set diff --git a/packages/zarr-metadata/changes/398.removal.md b/packages/zarr-metadata/changes/398.removal.md new file mode 100644 index 0000000000..30c8758beb --- /dev/null +++ b/packages/zarr-metadata/changes/398.removal.md @@ -0,0 +1 @@ +The public doors are smaller. `zarr_metadata.v3.definition` and `zarr_metadata.v2.definition` no longer export the scope algebra's helpers (`claims_of`, `claim_key`, `refines`, `Claims`, `ClaimKey`, `Disagreements`), the canonical-spelling and field-walking helpers (`canonical_of`, `canonicalize`, `configuration_of`, `fields_of`, `with_problems`, `chunk_grid_lengths`, `read_pipeline`, `field_json_schema`), `check` (still in `zarr_metadata.typed_json`) or `WithFillValue`; a model's `claims` and a scope's `disagreements` are no longer public. `zarr_metadata.model` no longer re-exports the `*_KEYS_*` sets, `is_json`/`parse_json`/`validate_json`, or the JSON shapes of known writer bugs. `zarr_metadata.typed_json` no longer exports `typeddict_keys`. What a document, a model, a definition, a scope and a reading are is unchanged. diff --git a/packages/zarr-metadata/docs/api/index.md b/packages/zarr-metadata/docs/api/index.md index 574cbcb9af..3874a814ac 100644 --- a/packages/zarr-metadata/docs/api/index.md +++ b/packages/zarr-metadata/docs/api/index.md @@ -25,9 +25,9 @@ The package is organized to mirror the structure of the Zarr specifications: and [data types](v3/data_type.md) - [`zarr_metadata.v3.definition`](v3/definition.md) — each extension's metadata as a definition: the TypedDict its configuration is, and the - rules on it; check JSON against a TypedDict, judge a configuration, - read a whole field in a scope, read a codec pipeline, or write a - scope's fields as a JSON Schema. Its module docstring is the guide + rules on it; judge a configuration, read a whole field in a scope, and + the scopes, `CORE` and `CORE_AND_EXTENSIONS`, fields are read in. Its + module docstring is the guide The document types, models, and spec vocabulary — including the store keys — are re-exported at the top level, so diff --git a/packages/zarr-metadata/pyproject.toml b/packages/zarr-metadata/pyproject.toml index 588455b660..486d754b7c 100644 --- a/packages/zarr-metadata/pyproject.toml +++ b/packages/zarr-metadata/pyproject.toml @@ -215,3 +215,7 @@ showcontent = true directory = "misc" name = "Misc" showcontent = true + +[tool.ruff.lint.flake8-type-checking] +# A TypedDict's annotations are evaluated at run time: the checker reads them. +runtime-evaluated-base-classes = ["typing_extensions.TypedDict", "typing.TypedDict"] diff --git a/packages/zarr-metadata/src/zarr_metadata/_pydantic_schema.py b/packages/zarr-metadata/src/zarr_metadata/_pydantic_schema.py index 810c9d69ae..64e4d01bd5 100644 --- a/packages/zarr-metadata/src/zarr_metadata/_pydantic_schema.py +++ b/packages/zarr-metadata/src/zarr_metadata/_pydantic_schema.py @@ -3,14 +3,14 @@ from __future__ import annotations import re -from collections.abc import Mapping # noqa: TC003 # resolved by Pydantic at runtime +from collections.abc import Mapping # resolved by Pydantic at runtime from typing import Annotated, Literal, NotRequired from pydantic import Field from typing_extensions import TypedDict from zarr_metadata._common import JSONValue -from zarr_metadata.v2.array import ( # noqa: TC001 # resolved by Pydantic at runtime +from zarr_metadata.v2.array import ( # resolved by Pydantic at runtime ZarrV2DataTypeMetadata, ) from zarr_metadata.v2.codec import ( # resolved by Pydantic at runtime diff --git a/packages/zarr-metadata/src/zarr_metadata/model/__init__.py b/packages/zarr-metadata/src/zarr_metadata/model/__init__.py index f8f2b975bc..a19f4cb72b 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/__init__.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/__init__.py @@ -38,9 +38,6 @@ MetadataValidationError, ProblemKind, ValidationProblem, - is_json, - parse_json, - validate_json, ) from zarr_metadata._sentinel import UNSET from zarr_metadata.model._array import ( @@ -79,26 +76,13 @@ Repair, RepairKind, ZarrV2RepairedConsolidatedMetadataReading, - ZarrV2ZGroupWithConsolidatedMetadataJSON, - ZarrV3NullConsolidatedGroupMetadataJSON, ZarrV3RepairedNodeMetadataReading, - ZarrV3ZeroChunkArrayMetadataJSON, - ZarrV3ZeroChunkRegularGridConfigurationJSON, - ZarrV3ZeroChunkRegularGridJSON, read_repaired_consolidated_metadata_v2, read_repaired_node_metadata_v3, repair_consolidated_metadata_v2, repair_node_metadata_v3, ) from zarr_metadata.model._validation import ( - ARRAY_METADATA_OPTIONAL_KEYS_V3, - ARRAY_METADATA_REQUIRED_KEYS_V2, - ARRAY_METADATA_REQUIRED_KEYS_V3, - ARRAY_METADATA_STANDARD_KEYS_V3, - GROUP_METADATA_OPTIONAL_KEYS_V3, - GROUP_METADATA_REQUIRED_KEYS_V2, - GROUP_METADATA_REQUIRED_KEYS_V3, - GROUP_METADATA_STANDARD_KEYS_V3, ZarrV2ArrayMetadataReading, ZarrV3ArrayMetadataReading, is_array_metadata_v2, @@ -155,14 +139,6 @@ ) __all__ = [ - "ARRAY_METADATA_OPTIONAL_KEYS_V3", - "ARRAY_METADATA_REQUIRED_KEYS_V2", - "ARRAY_METADATA_REQUIRED_KEYS_V3", - "ARRAY_METADATA_STANDARD_KEYS_V3", - "GROUP_METADATA_OPTIONAL_KEYS_V3", - "GROUP_METADATA_REQUIRED_KEYS_V2", - "GROUP_METADATA_REQUIRED_KEYS_V3", - "GROUP_METADATA_STANDARD_KEYS_V3", "UNSET", "ZARR_V2_ARRAY_METADATA_STORE_KEY", "ZARR_V2_ATTRIBUTES_STORE_KEY", @@ -188,7 +164,6 @@ "ZarrV2GroupMetadataUpdate", "ZarrV2NodeMetadata", "ZarrV2RepairedConsolidatedMetadataReading", - "ZarrV2ZGroupWithConsolidatedMetadataJSON", "ZarrV3ArrayMetadata", "ZarrV3ArrayMetadataReading", "ZarrV3ArrayMetadataStoreKey", @@ -202,17 +177,12 @@ "ZarrV3NodeMetadata", "ZarrV3NodeMetadataInput", "ZarrV3NodeMetadataReading", - "ZarrV3NullConsolidatedGroupMetadataJSON", "ZarrV3RepairedNodeMetadataReading", "ZarrV3UnknownNodeReading", - "ZarrV3ZeroChunkArrayMetadataJSON", - "ZarrV3ZeroChunkRegularGridConfigurationJSON", - "ZarrV3ZeroChunkRegularGridJSON", "is_array_metadata_v2", "is_array_metadata_v3", "is_group_metadata_v2", "is_group_metadata_v3", - "is_json", "is_metadata_field_v3", "is_node_name_v3", "is_node_path_v3", @@ -223,7 +193,6 @@ "parse_array_metadata_v3", "parse_group_metadata_v2", "parse_group_metadata_v3", - "parse_json", "parse_metadata_field_v3", "parse_node_name_v3", "parse_node_path_v3", @@ -239,7 +208,6 @@ "validate_array_metadata_v3", "validate_group_metadata_v2", "validate_group_metadata_v3", - "validate_json", "validate_metadata_field_v3", "validate_node_metadata_v3", "validate_node_name_v3", diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_array.py b/packages/zarr-metadata/src/zarr_metadata/model/_array.py index 01218ac22a..1155080b92 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_array.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/_array.py @@ -34,9 +34,16 @@ read_array_v2, read_array_v3, ) -from zarr_metadata.v2.array import ZARR_V2_ARRAY_METADATA_STORE_KEY +from zarr_metadata.v2.array import ( + ZARR_V2_ARRAY_METADATA_STORE_KEY, + ZarrV2ArrayDimensionSeparator, + ZarrV2ArrayOrder, + ZarrV2DataTypeMetadata, +) from zarr_metadata.v2.attributes import ZARR_V2_ATTRIBUTES_STORE_KEY +from zarr_metadata.v2.codec import ZarrV2CodecMetadata from zarr_metadata.v2.definition import CORE_V2, resolve_dtype_v2 +from zarr_metadata.v3._common import ZarrV3MetadataFieldJSON from zarr_metadata.v3._definition import ( ChunkGridDefinition, ChunkKeyEncodingDefinition, @@ -59,16 +66,8 @@ from zarr_metadata._typed_json import Loc from zarr_metadata.v2._definition import ZarrV2CodecDefinition, ZarrV2DataTypeDefinition - from zarr_metadata.v2.array import ( - ZarrV2ArrayDimensionSeparator, - ZarrV2ArrayMetadataJSON, - ZarrV2ArrayMetadataStoreKey, - ZarrV2ArrayOrder, - ZarrV2DataTypeMetadata, - ) + from zarr_metadata.v2.array import ZarrV2ArrayMetadataJSON, ZarrV2ArrayMetadataStoreKey from zarr_metadata.v2.attributes import ZarrV2AttributesStoreKey - from zarr_metadata.v2.codec import ZarrV2CodecMetadata - from zarr_metadata.v3._common import ZarrV3MetadataFieldJSON from zarr_metadata.v3._definition import Resolved from zarr_metadata.v3.array import ( ZarrV3ArrayMetadataJSON, @@ -140,6 +139,11 @@ class ZarrV3ArrayMetadata: zarr_format: Final = 3 node_type: Final = "array" + @property + def claims(self) -> Claims: + """What the reading claimed of each name the document writes, keyed as the scope files it.""" + return self._claims + def __init__(self, document: object, context: Context | None = None) -> None: scope = CORE_AND_EXTENSIONS if context is None else context reading, members = read_array_v3(document, scope) @@ -195,11 +199,6 @@ def reading(self) -> ZarrV3ArrayMetadataReading: """The document as the scope read it: each field, the pipeline, the chunk each codec is handed.""" return self._reading - @property - def claims(self) -> Claims: - """What the reading claimed of each name the document writes, keyed as the scope files it.""" - return self._claims - def to_json(self) -> ZarrV3ArrayMetadataJSON: """The document as written, refined, sharing nothing with the model.""" return cast("ZarrV3ArrayMetadataJSON", copied(self._document)) @@ -578,6 +577,11 @@ class ZarrV2ArrayMetadata: zarr_format: Final = 2 + @property + def claims(self) -> Claims: + """What the reading claimed of each typestr and codec id the document writes, keyed as the scope files them.""" + return self._claims + def __init__(self, document: object, context: Context | None = None) -> None: scope = CORE_V2 if context is None else context reading, members = read_array_v2(document, scope) @@ -634,11 +638,6 @@ def reading(self) -> ZarrV2ArrayMetadataReading: """The document as the scope read it: the dtype, the compressor, each filter.""" return self._reading - @property - def claims(self) -> Claims: - """What the reading claimed of each typestr and codec id the document writes, keyed as the scope files them.""" - return self._claims - def to_json(self) -> ZarrV2ArrayMetadataJSON: """The merged document as written, refined, sharing nothing with the model. diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_group.py b/packages/zarr-metadata/src/zarr_metadata/model/_group.py index 4653db4ad1..f9706d782c 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_group.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/_group.py @@ -10,6 +10,7 @@ from typing_extensions import TypeAliasType, TypedDict, Unpack +from zarr_metadata._common import JSONValue from zarr_metadata._json import ( MetadataValidationError, ValidationProblem, @@ -72,7 +73,6 @@ if TYPE_CHECKING: from collections.abc import Iterator - from zarr_metadata._common import JSONValue from zarr_metadata._typed_json import Loc from zarr_metadata.v2.attributes import ZarrV2AttributesStoreKey from zarr_metadata.v2.consolidated import ZarrV2ConsolidatedMetadataStoreKey @@ -146,6 +146,11 @@ class ZarrV3GroupMetadata: zarr_format: Final = 3 node_type: Final = "group" + @property + def claims(self) -> Claims: + """What the reading claimed of each name the document and its consolidated documents write, keyed as the scope files it.""" + return self._claims + def __init__(self, document: object, context: Context | None = None) -> None: scope = CORE_AND_EXTENSIONS if context is None else context reading, members = read_group_v3(document, scope) @@ -220,11 +225,6 @@ def reading(self) -> ZarrV3GroupMetadataReading: """The document as the scope read it: each document its consolidated metadata holds, as read.""" return self._reading - @property - def claims(self) -> Claims: - """What the reading claimed of each name the document writes, in the documents it holds, keyed as the scope files it.""" - return self._claims - def to_json(self) -> ZarrV3GroupMetadataJSON: """The document as written, refined, sharing nothing with the model.""" return cast("ZarrV3GroupMetadataJSON", copied(self._document)) @@ -1220,6 +1220,11 @@ class ZarrV2GroupMetadata: zarr_format: Final = 2 + @property + def claims(self) -> Claims: + """What the reading claimed: nothing, since a group holds no field.""" + return MappingProxyType({}) + def __init__(self, document: object, context: Context | None = None) -> None: scope = CORE_V2 if context is None else context parsed = parse_group_metadata_v2(document, context=scope) @@ -1255,11 +1260,6 @@ def context(self) -> Context: """The scope the document was read in, which `update` reads new attributes in.""" return self._context - @property - def claims(self) -> Claims: - """What the reading claimed: nothing, since a group holds no field.""" - return MappingProxyType({}) - @property def attributes(self) -> Mapping[str, JSONValue] | UNSET: """The user attributes a `.zattrs` holds, read-only at every level; `UNSET` when there is no `.zattrs`.""" diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_repair.py b/packages/zarr-metadata/src/zarr_metadata/model/_repair.py index 293f3ac42b..7e198497f1 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_repair.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/_repair.py @@ -22,7 +22,7 @@ from annotated_types import Ge from zarr_metadata._common import ( - JSONValue, # noqa: TC001 - a TypedDict's annotations are evaluated at run time + JSONValue, ) from zarr_metadata._json import JSON_DEPTH, MetadataValidationError, ValidationProblem from zarr_metadata._typed_json import Loc, check diff --git a/packages/zarr-metadata/src/zarr_metadata/typed_json.py b/packages/zarr-metadata/src/zarr_metadata/typed_json.py index 3ca22ad4f8..9ee3ef8688 100644 --- a/packages/zarr-metadata/src/zarr_metadata/typed_json.py +++ b/packages/zarr-metadata/src/zarr_metadata/typed_json.py @@ -96,10 +96,8 @@ from zarr_metadata._typed_json import ( JSONSchema, Loc, - TypedDictKeys, check, json_schema, - typeddict_keys, ) __all__ = [ @@ -107,9 +105,7 @@ "JSONValue", "Loc", "ProblemKind", - "TypedDictKeys", "ValidationProblem", "check", "json_schema", - "typeddict_keys", ] diff --git a/packages/zarr-metadata/src/zarr_metadata/v2/codec/compression.py b/packages/zarr-metadata/src/zarr_metadata/v2/codec/compression.py index c873634ab1..5f9092ea30 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v2/codec/compression.py +++ b/packages/zarr-metadata/src/zarr_metadata/v2/codec/compression.py @@ -12,7 +12,7 @@ from typing_extensions import ReadOnly, TypedDict from zarr_metadata._common import ( - JSONValue, # noqa: TC001 - a TypedDict's annotations are evaluated at run time + JSONValue, ) from zarr_metadata.v2._definition import ZarrV2CodecDefinition diff --git a/packages/zarr-metadata/src/zarr_metadata/v2/data_type/object.py b/packages/zarr-metadata/src/zarr_metadata/v2/data_type/object.py index 8840b7d325..2ea6547cf9 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v2/data_type/object.py +++ b/packages/zarr-metadata/src/zarr_metadata/v2/data_type/object.py @@ -8,7 +8,7 @@ from zarr_metadata.v2._definition import ZarrV2DataTypeDefinition from zarr_metadata.v2.data_type.scalar import ( - ZarrV2ByteOrder, # noqa: TC001 - a TypedDict's annotations are evaluated at run time + ZarrV2ByteOrder, ) diff --git a/packages/zarr-metadata/src/zarr_metadata/v2/data_type/time.py b/packages/zarr-metadata/src/zarr_metadata/v2/data_type/time.py index f769104a61..71410bfb47 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v2/data_type/time.py +++ b/packages/zarr-metadata/src/zarr_metadata/v2/data_type/time.py @@ -10,7 +10,7 @@ from zarr_metadata._json import ValidationProblem from zarr_metadata.v2._definition import ZarrV2DataTypeDefinition from zarr_metadata.v2.data_type.scalar import ( - ZarrV2ByteOrder, # noqa: TC001 - a TypedDict's annotations are evaluated at run time + ZarrV2ByteOrder, ) from zarr_metadata.v3.data_type._numpy_time import ( NumpyTimeScaleFactor, diff --git a/packages/zarr-metadata/src/zarr_metadata/v2/definition.py b/packages/zarr-metadata/src/zarr_metadata/v2/definition.py index 8b4bddd0c2..cd7862b5e2 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v2/definition.py +++ b/packages/zarr-metadata/src/zarr_metadata/v2/definition.py @@ -35,7 +35,6 @@ Resolved, Unclaimed, canonical_fill_value, - canonical_of, fill_value_problems, resolve, ) @@ -46,9 +45,6 @@ Conflict, Disagreements, ScopeConflictError, - claim_key, - claims_of, - refines, ) if TYPE_CHECKING: @@ -91,11 +87,7 @@ def resolve_codec_v2( "ZarrV2DataTypeDefinition", "ZarrV2DataTypeField", "canonical_fill_value", - "canonical_of", - "claim_key", - "claims_of", "fill_value_problems", - "refines", "resolve_codec_v2", "resolve_dtype_v2", ] diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/_common.py b/packages/zarr-metadata/src/zarr_metadata/v3/_common.py index 8411fa6224..d433cbb270 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/_common.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/_common.py @@ -9,7 +9,7 @@ import re from collections.abc import Mapping -from typing import Final, TypeGuard, cast +from typing import Final, TypeAlias, TypeGuard, cast from typing_extensions import TypeAliasType @@ -25,7 +25,7 @@ with_input, ) -ZarrV3MetadataFieldJSON = str | ZarrV3NamedConfigJSON +ZarrV3MetadataFieldJSON: TypeAlias = str | ZarrV3NamedConfigJSON """The JSON shape of any v3 metadata extension-point entry: either a bare short-hand name string or a `{name, configuration, must_understand}` envelope. diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/definition.py b/packages/zarr-metadata/src/zarr_metadata/v3/definition.py index 290579951f..a1bd7905d7 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/definition.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/definition.py @@ -28,8 +28,8 @@ **Reading JSON.** Three steps, each feeding the next, and each usable on its own by a caller that holds nothing but JSON: -1. `check(value, SomeTypedDict)`, from `zarr_metadata.typed_json` and - here too, type-checks JSON against a TypedDict and needs nothing else: +1. `check(value, SomeTypedDict)`, from `zarr_metadata.typed_json`, + type-checks JSON against a TypedDict and needs nothing else: a value of the TypedDict or None, and every problem, each located. The value holds what the TypedDict admits and nothing else. A member typed with a field alias is checked as the JSON a metadata field is. @@ -55,19 +55,12 @@ field is read as one of the five kinds, with or without type arguments; `resolve(field, Definition, scope)` is a `TypeError`, since nothing - is filed under it. `configuration_of(resolved, GZIP_CODEC)` is the - configuration typed as that definition's TypedDict, when it read it. - `fields_of(resolved)` gives the field and each field it holds, with - where each sits. A whole v3 array document is read by - `read_array_metadata_v3`, in `zarr_metadata.model`. + is filed under it. A whole v3 array document is read by + `read_array_metadata_v3`, in `zarr_metadata.model`, whose reading + gives each field with where it sits in the document. from zarr_metadata.v3.codec.gzip import GZIP_CODEC - from zarr_metadata.v3.definition import ( - CORE_AND_EXTENSIONS, - CodecDefinition, - configuration_of, - resolve, - ) + from zarr_metadata.v3.definition import CORE_AND_EXTENSIONS, CodecDefinition, resolve resolved, problems = resolve({"name": "gzip", "configuration": {"level": 12}}, CodecDefinition, CORE_AND_EXTENSIONS) @@ -78,7 +71,8 @@ resolved, problems = resolve({"name": "gzip", "configuration": {"level": 5}}, CodecDefinition, CORE_AND_EXTENSIONS) - configuration_of(resolved, GZIP_CODEC) # {'level': 5}, a GzipCodecConfiguration + resolved.definition is GZIP_CODEC # True + resolved.configuration # {'level': 5} Problems are values, not exceptions: `ValidationProblem(loc, message, kind)`, with `kind` one of `invalid_type`, `invalid_value`, @@ -227,9 +221,9 @@ def acme_lz4_rules( fall short of one -- located in the configuration. A grid that says nothing of the shape fits every one. It also says the lengths its chunks take along each axis of an array it fits, `chunk_lengths`: a set per -axis, since a rectilinear grid's chunks differ. `chunk_grid_lengths(grid, -shape)` gives both of a chunk grid field the scope read: an entry for -each dimension of the shape, None where nothing says the lengths. +axis, since a rectilinear grid's chunks differ. A reading holds both of +the grid it read: an entry for each dimension of the shape, None where +nothing says the lengths. A codec is judged against what it is handed. The array hands its first codec a `Chunk`: the lengths of its grid's chunks along each of the @@ -238,11 +232,10 @@ def acme_lz4_rules( a chunk: `chunk_rules`, located in its configuration -- a `transpose` whose `order` has another number of axes. An array -> array codec says what it hands the next, whatever its chunk rules found: `transition` -- -`transpose` permutes the axes. `read_pipeline(codecs, chunk)` reads codec -fields the scope read as a pipeline: their order -- array -> array -codecs, one array -> bytes codec, bytes -> bytes codecs -- and then each -against the chunk it is handed, giving each codec's `Stage` with that -chunk. A codec that holds pipelines of its own says what each is +`transpose` permutes the axes. A reading reads the codec fields as a +pipeline: their order -- array -> array codecs, one array -> bytes codec, +bytes -> bytes codecs -- and then each against the chunk it is handed, +giving each codec's `Stage` with that chunk. A codec that holds pipelines of its own says what each is handed: `pipelines`, by the member of its configuration that holds each -- a shard's inner codecs its inner chunks, its index codecs the shard index -- and each is read the same way, its stages kept as the codec's @@ -259,8 +252,8 @@ def acme_lz4_rules( refused, since that name reads as `r*`. `r*` itself is notation, and a document that writes it names nothing in any scope. -**The simplest spelling.** `canonicalize(field, kind, scope)` gives a -field without problems in its simplest equivalent spelling: each nested +**The simplest spelling.** A field without problems has a simplest +equivalent spelling, which is what two fields are compared by: each nested field in its own simplest spelling, then the definition's `canonical` -- blosc drops a `typesize` that `noshuffle` ignores, a rectilinear grid run-length encodes its chunk shapes -- and the envelope in the fewest @@ -271,15 +264,13 @@ def acme_lz4_rules( problem, an unknown key included, has none: a simpler spelling of it would erase what its author wrote. What `canonical` gives is judged again: one that does not hold is a `ValueError`, a fault in the -definition. `canonical_of(resolved, problems)` spells a field a scope -has read already, given its problems -- as `resolve` gives them, or -`with_problems` gives each field of a reading -- without reading it -again, and gives what `canonicalize` gives: None for a field with a -problem. - -**JSON Schema.** `field_json_schema(CodecDefinition, SCOPE)` writes the -fields of one kind a scope reads as a JSON Schema, draft 2020-12, for a -validator in another language or an editor: each definition's field -- +definition. `canonical_fill_value` spells a fill value the same way, as +the data type that read it spells one. + +**JSON Schema.** `node_metadata_json_schema_v3`, in `zarr_metadata.model`, +writes a whole `zarr.json` as a JSON Schema, draft 2020-12, for a +validator in another language or an editor, its fields as the scope +reads them: each definition's field -- its name, its configuration as its TypedDict says, bounds and all, a `must_understand` of `true`, and its bare name when it needs no configuration -- and a name nothing in scope claims, with any @@ -289,20 +280,16 @@ def acme_lz4_rules( one `resolve` reads without a problem, it accepts, as JSON: arrays as lists, as a parser gives them. Each configuration TypedDict, and each field alias, is written once, in `$defs`, under its -name. `node_metadata_json_schema_v3`, in `zarr_metadata.model`, writes a -whole `zarr.json`, its fill value held to its data type's. +name; the fill value is held to its data type's. **Scopes as values.** Two scopes are equal when they file the same -definitions, and equal scopes hash alike. `claims_of(fields_of(field))` -says what a reading claimed of each name -- the definition that read it, -or None -- keyed as the scope files it, `r16` under `r*`. -`refines(field, other)` orders two readings of a field by information: a +definitions, and equal scopes hash alike. `Context.joined(*scopes)` is +the least scope above each, or a `ScopeConflictError` naming each name +filed two ways; `extended_with` remains the way to take a name over on +purpose. A model's `refined_in` moves it to a scope that claims more and +contradicts nothing, and `refines` orders two models by information: a name nothing claimed, read by a definition, is a gain; the reverse a loss; one name read by two definitions a conflict. -`scope.disagreements(claims)` says where a scope would read a reading -otherwise, and `Context.joined(*scopes)` is the least scope above each, -or a `ScopeConflictError` naming each name filed two ways; `extended_with` -remains the way to take a name over on purpose. A definition checks itself when it is built, and each of these is a `TypeError` saying what is wrong: a `configuration` that is not a @@ -323,7 +310,7 @@ def acme_lz4_rules( from zarr_metadata._common import JSONValue from zarr_metadata._json import MetadataValidationError, ProblemKind, ValidationProblem, shown -from zarr_metadata._typed_json import Loc, check +from zarr_metadata._typed_json import Loc from zarr_metadata.v3._common import ( ChunkGridField, ChunkKeyEncodingField, @@ -331,7 +318,6 @@ def acme_lz4_rules( DataTypeField, StaticCodecField, StorageTransformerField, - ZarrV3MetadataFieldJSON, ) from zarr_metadata.v3._definition import ( Chunk, @@ -351,20 +337,12 @@ def acme_lz4_rules( StorageClass, StorageTransformerDefinition, Unclaimed, - WithFillValue, canonical_fill_value, - canonical_of, - canonicalize, - chunk_grid_lengths, - configuration_of, - field_json_schema, - fields_of, fill_value_problems, resolve, storage_of, - with_problems, ) -from zarr_metadata.v3._pipeline import Stage, read_pipeline +from zarr_metadata.v3._pipeline import Stage from zarr_metadata.v3._registry import CORE, CORE_AND_EXTENSIONS, Context from zarr_metadata.v3._scope import ( ClaimKey, @@ -372,9 +350,6 @@ def acme_lz4_rules( Conflict, Disagreements, ScopeConflictError, - claim_key, - claims_of, - refines, ) __all__ = [ @@ -415,23 +390,9 @@ def acme_lz4_rules( "StorageTransformerField", "Unclaimed", "ValidationProblem", - "WithFillValue", - "ZarrV3MetadataFieldJSON", "canonical_fill_value", - "canonical_of", - "canonicalize", - "check", - "chunk_grid_lengths", - "claim_key", - "claims_of", - "configuration_of", - "field_json_schema", - "fields_of", "fill_value_problems", - "read_pipeline", - "refines", "resolve", "shown", "storage_of", - "with_problems", ] diff --git a/packages/zarr-metadata/tests/model/test_array.py b/packages/zarr-metadata/tests/model/test_array.py index 3596b2e265..f9d667d471 100644 --- a/packages/zarr-metadata/tests/model/test_array.py +++ b/packages/zarr-metadata/tests/model/test_array.py @@ -13,11 +13,16 @@ from typing_extensions import Unpack from tests.model._cases import Expect, ExpectFail, mutate_nested_containers -from zarr_metadata._json import JSON_DEPTH, arrays_to_tuples, json_text, prefixed +from zarr_metadata._json import ( + JSON_DEPTH, + arrays_to_tuples, + is_json, + json_text, + parse_json, + prefixed, + validate_json, +) from zarr_metadata.model import ( - ARRAY_METADATA_OPTIONAL_KEYS_V3, - ARRAY_METADATA_REQUIRED_KEYS_V3, - ARRAY_METADATA_STANDARD_KEYS_V3, UNSET, MetadataValidationError, ValidationProblem, @@ -29,17 +34,25 @@ is_array_metadata_v3, is_group_metadata_v2, is_group_metadata_v3, - is_json, is_metadata_field_v3, parse_array_metadata_v2, parse_array_metadata_v3, - parse_json, parse_metadata_field_v3, validate_array_metadata_v2, validate_array_metadata_v3, - validate_json, validate_metadata_field_v3, ) +from zarr_metadata.model._validation import ( + ARRAY_METADATA_OPTIONAL_KEYS_V3, + ARRAY_METADATA_REQUIRED_KEYS_V3, + ARRAY_METADATA_STANDARD_KEYS_V3, +) +from zarr_metadata.v3._definition import ( + canonical_of, + configuration_of, + fields_of, + with_problems, +) from zarr_metadata.v3.array import ZarrV3ArrayMetadataJSONPartial from zarr_metadata.v3.codec.gzip import GZIP_CODEC from zarr_metadata.v3.definition import ( @@ -47,10 +60,6 @@ CORE_AND_EXTENSIONS, Read, Unclaimed, - canonical_of, - configuration_of, - fields_of, - with_problems, ) if TYPE_CHECKING: @@ -65,8 +74,6 @@ def test_guards_exported_from_package() -> None: import zarr_metadata.model for name in ( - "is_json", - "parse_json", "is_metadata_field_v3", "parse_metadata_field_v3", "is_array_metadata_v3", @@ -150,7 +157,6 @@ def test_validation_diagnostics_exported_from_package() -> None: for name in ( "ValidationProblem", "MetadataValidationError", - "validate_json", "validate_metadata_field_v3", "validate_array_metadata_v3", "validate_array_metadata_v2", diff --git a/packages/zarr-metadata/tests/model/test_consolidating.py b/packages/zarr-metadata/tests/model/test_consolidating.py index 09b012ddb5..eebbd3ad7c 100644 --- a/packages/zarr-metadata/tests/model/test_consolidating.py +++ b/packages/zarr-metadata/tests/model/test_consolidating.py @@ -22,6 +22,9 @@ read_group_metadata_v3, validate_group_metadata_v3, ) +from zarr_metadata.v3._scope import ( + claims_of, +) from zarr_metadata.v3.codec.crc32c import Empty from zarr_metadata.v3.codec.gzip import GZIP_CODEC from zarr_metadata.v3.data_type.raw import RAW_BYTES_DATA_TYPE @@ -30,7 +33,6 @@ CORE_AND_EXTENSIONS, CodecDefinition, Context, - claims_of, ) ARRAY: dict[str, Any] = { diff --git a/packages/zarr-metadata/tests/model/test_construction.py b/packages/zarr-metadata/tests/model/test_construction.py index e0ec182f07..185b6740cb 100644 --- a/packages/zarr-metadata/tests/model/test_construction.py +++ b/packages/zarr-metadata/tests/model/test_construction.py @@ -27,7 +27,9 @@ ZarrV3GroupMetadata, ) from zarr_metadata.v3.data_type.int8 import INT8_DATA_TYPE -from zarr_metadata.v3.definition import CORE_AND_EXTENSIONS +from zarr_metadata.v3.definition import ( + CORE_AND_EXTENSIONS, +) if TYPE_CHECKING: from collections.abc import Iterator diff --git a/packages/zarr-metadata/tests/model/test_extension_points.py b/packages/zarr-metadata/tests/model/test_extension_points.py index e0a9fb3836..bfd52df7b0 100644 --- a/packages/zarr-metadata/tests/model/test_extension_points.py +++ b/packages/zarr-metadata/tests/model/test_extension_points.py @@ -25,7 +25,12 @@ validate_array_metadata_v3, validate_group_metadata_v3, ) -from zarr_metadata.v3.definition import CORE, CORE_AND_EXTENSIONS, CodecDefinition, Context +from zarr_metadata.v3.definition import ( + CORE, + CORE_AND_EXTENSIONS, + CodecDefinition, + Context, +) BYTES = {"name": "bytes", "configuration": {"endian": "little"}} diff --git a/packages/zarr-metadata/tests/model/test_group.py b/packages/zarr-metadata/tests/model/test_group.py index f367fa3d68..07e24b8d36 100644 --- a/packages/zarr-metadata/tests/model/test_group.py +++ b/packages/zarr-metadata/tests/model/test_group.py @@ -19,7 +19,9 @@ arrays_to_tuples, json_text, ) -from zarr_metadata.model import UNSET +from zarr_metadata.model import ( + UNSET, +) from zarr_metadata.model._array import ZarrV3ArrayMetadata, ZarrV3ArrayMetadataUpdate from zarr_metadata.model._group import ( ZarrV2ConsolidatedMetadata, diff --git a/packages/zarr-metadata/tests/model/test_pair_v2.py b/packages/zarr-metadata/tests/model/test_pair_v2.py index b03ab20c8a..1b1a29de4a 100644 --- a/packages/zarr-metadata/tests/model/test_pair_v2.py +++ b/packages/zarr-metadata/tests/model/test_pair_v2.py @@ -28,7 +28,9 @@ ZarrV2CodecDefinition, ZarrV2DataTypeDefinition, ) -from zarr_metadata.v3.definition import EmptyConfiguration +from zarr_metadata.v3.definition import ( + EmptyConfiguration, +) Loc = tuple[str | int, ...] ARRAY: dict[str, Any] = { @@ -424,10 +426,3 @@ def test_error_a_second_key_for_a_node_file_hides_no_other_problem() -> None: assert one == ZarrV2ConsolidatedMetadata( {"zarr_consolidated_format": 1, "metadata": {"a/.zarray": ZARRAY}} ) - - -def test_the_repair_shapes_are_exported() -> None: - """The JSON shape of each known v2 writer bug is exported beside the v3 ones.""" - from zarr_metadata import model - - assert "ZarrV2ZGroupWithConsolidatedMetadataJSON" in model.__all__ diff --git a/packages/zarr-metadata/tests/model/test_pydantic.py b/packages/zarr-metadata/tests/model/test_pydantic.py index 2cc8822fe1..bccbe528e1 100644 --- a/packages/zarr-metadata/tests/model/test_pydantic.py +++ b/packages/zarr-metadata/tests/model/test_pydantic.py @@ -42,7 +42,9 @@ ) from zarr_metadata import JSONValue -from zarr_metadata.model import ZarrV3ArrayMetadata +from zarr_metadata.model import ( + ZarrV3ArrayMetadata, +) # --- the integration (this is the example) ----------------------------------- diff --git a/packages/zarr-metadata/tests/model/test_pydantic_module.py b/packages/zarr-metadata/tests/model/test_pydantic_module.py index c75f4a9833..5b15ddbf64 100644 --- a/packages/zarr-metadata/tests/model/test_pydantic_module.py +++ b/packages/zarr-metadata/tests/model/test_pydantic_module.py @@ -27,7 +27,10 @@ ZarrV3GroupMetadata, ) from zarr_metadata.v3.codec.gzip import GZIP_CODEC -from zarr_metadata.v3.definition import CORE, Read +from zarr_metadata.v3.definition import ( + CORE, + Read, +) V3_ARRAY_DOC = dict(ZarrV3ArrayMetadata.create_default(shape=(4,)).to_json()) V2_ARRAY_DOC = dict(ZarrV2ArrayMetadata.create_default(shape=(4,), chunks=(2,)).to_json()) diff --git a/packages/zarr-metadata/tests/model/test_read_array_metadata.py b/packages/zarr-metadata/tests/model/test_read_array_metadata.py index 9beb44b35c..2d4c50e8d4 100644 --- a/packages/zarr-metadata/tests/model/test_read_array_metadata.py +++ b/packages/zarr-metadata/tests/model/test_read_array_metadata.py @@ -21,6 +21,10 @@ read_array_metadata_v3, read_group_metadata_v3, ) +from zarr_metadata.v3._definition import ( + fields_of, + with_problems, +) from zarr_metadata.v3.definition import ( CORE, CORE_AND_EXTENSIONS, @@ -34,9 +38,7 @@ Refused, StorageTransformerDefinition, Unclaimed, - fields_of, resolve, - with_problems, ) if TYPE_CHECKING: diff --git a/packages/zarr-metadata/tests/model/test_read_array_metadata_v2.py b/packages/zarr-metadata/tests/model/test_read_array_metadata_v2.py index 1eac349756..18e0d53b7e 100644 --- a/packages/zarr-metadata/tests/model/test_read_array_metadata_v2.py +++ b/packages/zarr-metadata/tests/model/test_read_array_metadata_v2.py @@ -24,7 +24,9 @@ Unclaimed, ZarrV2CodecDefinition, ) -from zarr_metadata.v3.definition import EmptyConfiguration +from zarr_metadata.v3.definition import ( + EmptyConfiguration, +) Loc = tuple[str | int, ...] BASE: dict[str, Any] = dict(ZarrV2ArrayMetadata.create_default(shape=(4,)).to_json()) diff --git a/packages/zarr-metadata/tests/model/test_scope_threading.py b/packages/zarr-metadata/tests/model/test_scope_threading.py index 8051dadb9d..334ba03025 100644 --- a/packages/zarr-metadata/tests/model/test_scope_threading.py +++ b/packages/zarr-metadata/tests/model/test_scope_threading.py @@ -8,8 +8,13 @@ import pytest import zarr_metadata.model as zm -from zarr_metadata.model import ZarrV3ArrayMetadata -from zarr_metadata.v3.definition import CORE_AND_EXTENSIONS, Context +from zarr_metadata.model import ( + ZarrV3ArrayMetadata, +) +from zarr_metadata.v3.definition import ( + CORE_AND_EXTENSIONS, + Context, +) if TYPE_CHECKING: from collections.abc import Callable diff --git a/packages/zarr-metadata/tests/test_json_schema.py b/packages/zarr-metadata/tests/test_json_schema.py index 960f32ddac..bf72501d32 100644 --- a/packages/zarr-metadata/tests/test_json_schema.py +++ b/packages/zarr-metadata/tests/test_json_schema.py @@ -23,8 +23,17 @@ from tests.v3.test_every_definition import CASES, KINDS from zarr_metadata._common import JSONValue from zarr_metadata._typed_json import Schemas -from zarr_metadata.model import node_metadata_json_schema_v3, validate_node_metadata_v3 -from zarr_metadata.typed_json import check, json_schema +from zarr_metadata.model import ( + node_metadata_json_schema_v3, + validate_node_metadata_v3, +) +from zarr_metadata.typed_json import ( + check, + json_schema, +) +from zarr_metadata.v3._definition import ( + field_json_schema, +) from zarr_metadata.v3.codec.gzip import GzipCodecConfiguration from zarr_metadata.v3.definition import ( CORE, @@ -36,7 +45,6 @@ DataTypeDefinition, Definition, StorageTransformerDefinition, - field_json_schema, resolve, ) diff --git a/packages/zarr-metadata/tests/test_problem_data.py b/packages/zarr-metadata/tests/test_problem_data.py index 29e59ea7b5..f3fe5b3c1a 100644 --- a/packages/zarr-metadata/tests/test_problem_data.py +++ b/packages/zarr-metadata/tests/test_problem_data.py @@ -20,7 +20,14 @@ from annotated_types import Le from typing_extensions import TypedDict -from zarr_metadata._json import arrays_to_tuples, is_canonical_json, value_at, with_input, within +from zarr_metadata._json import ( + arrays_to_tuples, + is_canonical_json, + validate_json, + value_at, + with_input, + within, +) from zarr_metadata._sentinel import UNSET from zarr_metadata.model import ( MetadataValidationError, @@ -33,13 +40,14 @@ validate_array_metadata_v3, validate_group_metadata_v2, validate_group_metadata_v3, - validate_json, validate_metadata_field_v3, validate_node_metadata_v3, validate_node_name_v3, validate_node_path_v3, ) -from zarr_metadata.typed_json import check +from zarr_metadata.typed_json import ( + check, +) from zarr_metadata.v3.codec.gzip import GZIP_CODEC from zarr_metadata.v3.definition import ( CORE_AND_EXTENSIONS, diff --git a/packages/zarr-metadata/tests/test_public_api.py b/packages/zarr-metadata/tests/test_public_api.py index f8ef36cb04..28bb326e14 100644 --- a/packages/zarr-metadata/tests/test_public_api.py +++ b/packages/zarr-metadata/tests/test_public_api.py @@ -333,7 +333,6 @@ def test_all_is_grouped_and_unique() -> None: "ShardingIndexLocation", "Struct", "StructField", - "TypedDictKeys", "ValidationProblem", } ) diff --git a/packages/zarr-metadata/tests/test_typed_json.py b/packages/zarr-metadata/tests/test_typed_json.py index 73b50f66d2..311d497395 100644 --- a/packages/zarr-metadata/tests/test_typed_json.py +++ b/packages/zarr-metadata/tests/test_typed_json.py @@ -59,8 +59,13 @@ shape_of, typeddict_keys, ) -from zarr_metadata.model import ZarrV2ArrayMetadata, ZarrV3ArrayMetadata -from zarr_metadata.typed_json import check +from zarr_metadata.model import ( + ZarrV2ArrayMetadata, + ZarrV3ArrayMetadata, +) +from zarr_metadata.typed_json import ( + check, +) from zarr_metadata.v2.array import ZarrV2ArrayMetadataJSON, ZarrV2DataTypeMetadata from zarr_metadata.v3.array import ZarrV3ArrayMetadataJSON from zarr_metadata.v3.data_type.float32 import Float32FillValue diff --git a/packages/zarr-metadata/tests/v2/test_codecs.py b/packages/zarr-metadata/tests/v2/test_codecs.py index d4987f2a38..79528f05a8 100644 --- a/packages/zarr-metadata/tests/v2/test_codecs.py +++ b/packages/zarr-metadata/tests/v2/test_codecs.py @@ -9,7 +9,16 @@ from zarr_metadata.v2._definition import ZarrV2CodecDefinition from zarr_metadata.v2.codec import V2_CODECS, ZarrV2CodecMetadata -from zarr_metadata.v3.definition import Context, Read, Refused, Unclaimed, canonical_of, resolve +from zarr_metadata.v3._definition import ( + canonical_of, +) +from zarr_metadata.v3.definition import ( + Context, + Read, + Refused, + Unclaimed, + resolve, +) SCOPE = Context.of(*V2_CODECS) Loc = tuple[str | int, ...] diff --git a/packages/zarr-metadata/tests/v2/test_data_types.py b/packages/zarr-metadata/tests/v2/test_data_types.py index c5f71a2ff8..c8a1bfa7e6 100644 --- a/packages/zarr-metadata/tests/v2/test_data_types.py +++ b/packages/zarr-metadata/tests/v2/test_data_types.py @@ -6,14 +6,16 @@ from zarr_metadata.v2._definition import ZarrV2DataTypeDefinition from zarr_metadata.v2.data_type import V2_DATA_TYPES +from zarr_metadata.v3._definition import ( + canonical_of, + fields_of, +) from zarr_metadata.v3.definition import ( Context, Read, Refused, Unclaimed, canonical_fill_value, - canonical_of, - fields_of, fill_value_problems, resolve, ) diff --git a/packages/zarr-metadata/tests/v2/test_definition.py b/packages/zarr-metadata/tests/v2/test_definition.py index 8b06c655fb..ec723b9515 100644 --- a/packages/zarr-metadata/tests/v2/test_definition.py +++ b/packages/zarr-metadata/tests/v2/test_definition.py @@ -17,7 +17,11 @@ resolve_dtype_v2, ) from zarr_metadata.v3.codec.gzip import GZIP_CODEC -from zarr_metadata.v3.definition import CORE_AND_EXTENSIONS, CodecDefinition, DataTypeDefinition +from zarr_metadata.v3.definition import ( + CORE_AND_EXTENSIONS, + CodecDefinition, + DataTypeDefinition, +) def test_core_v2_files_every_v2_definition_apart_from_v3() -> None: diff --git a/packages/zarr-metadata/tests/v2/test_kinds.py b/packages/zarr-metadata/tests/v2/test_kinds.py index 5b8140ed01..7955f7834a 100644 --- a/packages/zarr-metadata/tests/v2/test_kinds.py +++ b/packages/zarr-metadata/tests/v2/test_kinds.py @@ -11,7 +11,9 @@ ZarrV2DataTypeDefinition, parse_typestr, ) -from zarr_metadata.v3.definition import EmptyConfiguration +from zarr_metadata.v3.definition import ( + EmptyConfiguration, +) Loc = tuple[str | int, ...] diff --git a/packages/zarr-metadata/tests/v3/test_definitions.py b/packages/zarr-metadata/tests/v3/test_definitions.py index d961a81ce6..e8917be224 100644 --- a/packages/zarr-metadata/tests/v3/test_definitions.py +++ b/packages/zarr-metadata/tests/v3/test_definitions.py @@ -7,7 +7,7 @@ import math import pickle from collections.abc import ( - Mapping, # noqa: TC003 - a TypedDict's annotations are evaluated at run time + Mapping, ) from dataclasses import dataclass, replace from typing import TYPE_CHECKING, Annotated, Any, Final, NotRequired, cast @@ -23,6 +23,14 @@ validate_metadata_field_v3, ) from zarr_metadata.model._array import ZarrV3ArrayMetadata +from zarr_metadata.typed_json import ( + check, +) +from zarr_metadata.v3._common import ZarrV3MetadataFieldJSON +from zarr_metadata.v3._definition import ( + canonical_of, + configuration_of, +) from zarr_metadata.v3.array import ZarrV3ArrayMetadataJSON from zarr_metadata.v3.chunk_grid.regular import REGULAR_CHUNK_GRID from zarr_metadata.v3.codec.crc32c import Empty @@ -47,16 +55,11 @@ StorageTransformerDefinition, Unclaimed, ValidationProblem, - ZarrV3MetadataFieldJSON, - canonical_of, - check, - configuration_of, resolve, ) if TYPE_CHECKING: from collections.abc import Callable, Iterable, Iterator - from decimal import Decimal class AcmeStackConfiguration(TypedDict, closed=True): @@ -1132,7 +1135,7 @@ class AcmePlainConfiguration(TypedDict, closed=True): class AcmeDecimal(TypedDict, closed=True): - value: Decimal # a name the type checker sees, and the running module does not + value: Decimal # noqa: F821 - a name the running module does not define # pyright: ignore[reportUndefinedVariable] class AcmeUnresolvedConfiguration(TypedDict, closed=True): diff --git a/packages/zarr-metadata/tests/v3/test_every_definition.py b/packages/zarr-metadata/tests/v3/test_every_definition.py index 19661e355b..112fdec528 100644 --- a/packages/zarr-metadata/tests/v3/test_every_definition.py +++ b/packages/zarr-metadata/tests/v3/test_every_definition.py @@ -13,6 +13,11 @@ import pytest from zarr_metadata._json import value_at +from zarr_metadata.v3._definition import ( + canonical_of, + canonicalize, + configuration_of, +) from zarr_metadata.v3.codec.gzip import GZIP_CODEC from zarr_metadata.v3.data_type.raw import RAW_BYTES_DATA_TYPE, RawBytesConfiguration from zarr_metadata.v3.definition import ( @@ -27,9 +32,6 @@ Read, Refused, ValidationProblem, - canonical_of, - canonicalize, - configuration_of, fill_value_problems, resolve, ) diff --git a/packages/zarr-metadata/tests/v3/test_fill_values.py b/packages/zarr-metadata/tests/v3/test_fill_values.py index d871b106e9..0b397f6a18 100644 --- a/packages/zarr-metadata/tests/v3/test_fill_values.py +++ b/packages/zarr-metadata/tests/v3/test_fill_values.py @@ -20,7 +20,10 @@ from zarr_metadata._json import JSON_DEPTH, value_at from zarr_metadata._sentinel import UNSET -from zarr_metadata.model import validate_array_metadata_v3, validate_group_metadata_v3 +from zarr_metadata.model import ( + validate_array_metadata_v3, + validate_group_metadata_v3, +) from zarr_metadata.model._array import ZarrV3ArrayMetadata from zarr_metadata.v3.data_type._float import FloatWidth, complex_fill_value_rules, float_bits from zarr_metadata.v3.data_type.struct import STRUCT_DATA_TYPE diff --git a/packages/zarr-metadata/tests/v3/test_grid_shapes.py b/packages/zarr-metadata/tests/v3/test_grid_shapes.py index 93747545b5..00e0cf753e 100644 --- a/packages/zarr-metadata/tests/v3/test_grid_shapes.py +++ b/packages/zarr-metadata/tests/v3/test_grid_shapes.py @@ -11,13 +11,18 @@ import pytest -from zarr_metadata.model import ZarrV3ArrayMetadata, validate_array_metadata_v3 +from zarr_metadata.model import ( + ZarrV3ArrayMetadata, + validate_array_metadata_v3, +) +from zarr_metadata.v3._definition import ( + chunk_grid_lengths, +) from zarr_metadata.v3.codec.crc32c import Empty from zarr_metadata.v3.definition import ( CORE_AND_EXTENSIONS, ChunkGridDefinition, JSONValue, - chunk_grid_lengths, resolve, ) diff --git a/packages/zarr-metadata/tests/v3/test_kinds.py b/packages/zarr-metadata/tests/v3/test_kinds.py index 5be3d5dfb8..61623ef0f2 100644 --- a/packages/zarr-metadata/tests/v3/test_kinds.py +++ b/packages/zarr-metadata/tests/v3/test_kinds.py @@ -11,7 +11,12 @@ from typing_extensions import TypedDict from zarr_metadata._json import ValidationProblem -from zarr_metadata.v3._definition import as_kind, kind_of +from zarr_metadata.v3._definition import ( + WithFillValue, + as_kind, + field_json_schema, + kind_of, +) from zarr_metadata.v3._scope import kind_name from zarr_metadata.v3.codec.gzip import GZIP_CODEC from zarr_metadata.v3.definition import ( @@ -22,9 +27,7 @@ Read, Refused, Unclaimed, - WithFillValue, canonical_fill_value, - field_json_schema, fill_value_problems, resolve, ) diff --git a/packages/zarr-metadata/tests/v3/test_pipelines.py b/packages/zarr-metadata/tests/v3/test_pipelines.py index ae19e4eb50..5c733e430b 100644 --- a/packages/zarr-metadata/tests/v3/test_pipelines.py +++ b/packages/zarr-metadata/tests/v3/test_pipelines.py @@ -12,7 +12,13 @@ import pytest from typing_extensions import TypedDict -from zarr_metadata.model import ZarrV3ArrayMetadata, validate_array_metadata_v3 +from zarr_metadata.model import ( + ZarrV3ArrayMetadata, + validate_array_metadata_v3, +) +from zarr_metadata.v3._pipeline import ( + read_pipeline, +) from zarr_metadata.v3.codec.crc32c import Empty from zarr_metadata.v3.definition import ( CORE_AND_EXTENSIONS, @@ -24,7 +30,6 @@ JSONValue, Nested, Resolved, - read_pipeline, resolve, ) diff --git a/packages/zarr-metadata/tests/v3/test_scope.py b/packages/zarr-metadata/tests/v3/test_scope.py index 00fc877a85..c51daa1eda 100644 --- a/packages/zarr-metadata/tests/v3/test_scope.py +++ b/packages/zarr-metadata/tests/v3/test_scope.py @@ -9,7 +9,16 @@ from hypothesis import given from hypothesis import strategies as st -from zarr_metadata.v3._scope import kind_name +from zarr_metadata.v3._definition import ( + fields_of, +) +from zarr_metadata.v3._scope import ( + Claims, + Disagreements, + claims_of, + kind_name, + refines, +) from zarr_metadata.v3.codec.bytes import BYTES_CODEC from zarr_metadata.v3.codec.crc32c import CRC32C_CODEC, Empty from zarr_metadata.v3.codec.gzip import GZIP_CODEC @@ -22,20 +31,15 @@ CORE_AND_EXTENSIONS, ChunkGridDefinition, ChunkKeyEncodingDefinition, - Claims, CodecDefinition, Conflict, Context, DataTypeDefinition, Definition, - Disagreements, Refused, Resolved, ScopeConflictError, StorageTransformerDefinition, - claims_of, - fields_of, - refines, resolve, ) diff --git a/packages/zarr-metadata/tests/v3/test_sharding.py b/packages/zarr-metadata/tests/v3/test_sharding.py index 3afa98994b..4d94f8d105 100644 --- a/packages/zarr-metadata/tests/v3/test_sharding.py +++ b/packages/zarr-metadata/tests/v3/test_sharding.py @@ -11,7 +11,13 @@ import pytest -from zarr_metadata.model import ZarrV3ArrayMetadata, validate_array_metadata_v3 +from zarr_metadata.model import ( + ZarrV3ArrayMetadata, + validate_array_metadata_v3, +) +from zarr_metadata.v3._pipeline import ( + read_pipeline, +) from zarr_metadata.v3.definition import ( CORE_AND_EXTENSIONS, Chunk, @@ -19,7 +25,6 @@ DataTypeDefinition, JSONValue, Resolved, - read_pipeline, resolve, ) From 05cf27596a9c3598648742301f433f3d3c632953 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Thu, 8 Oct 2026 15:24:44 +0200 Subject: [PATCH 77/94] refactor(zarr-metadata): narrow JSON containers with type guards, not casts is_object, is_json_object, is_list_or_tuple, is_tuple and is_array in _json say what isinstance leaves unknown to a type checker, so the readers no longer cast a mapping or sequence to its element types after checking it: 239 casts become 179. The TypedDict parsers report a non-string key as a problem instead of assuming one is a string. The two loops kept for frame depth say so instead of carrying a PERF noqa. Assisted-by: ClaudeCode:claude-fable-5-1 Co-Authored-By: Claude Fable 5.1 --- .../zarr-metadata/src/zarr_metadata/_json.py | 77 ++++++++++++------- .../src/zarr_metadata/_typed_json.py | 38 +++++---- .../src/zarr_metadata/model/_array.py | 5 +- .../src/zarr_metadata/model/_group.py | 67 ++++++++-------- .../src/zarr_metadata/model/_repair.py | 34 ++++---- .../src/zarr_metadata/model/_validation.py | 30 ++++---- .../src/zarr_metadata/pydantic.py | 7 +- .../src/zarr_metadata/v2/_definition.py | 20 +++-- .../src/zarr_metadata/v3/_common.py | 26 +++---- .../src/zarr_metadata/v3/_definition.py | 18 +++-- .../src/zarr_metadata/v3/_pipeline.py | 10 +-- 11 files changed, 188 insertions(+), 144 deletions(-) diff --git a/packages/zarr-metadata/src/zarr_metadata/_json.py b/packages/zarr-metadata/src/zarr_metadata/_json.py index 3152454f5e..9c40c15141 100644 --- a/packages/zarr-metadata/src/zarr_metadata/_json.py +++ b/packages/zarr-metadata/src/zarr_metadata/_json.py @@ -244,18 +244,13 @@ def prefixed( def value_at(value: object, loc: tuple[str | int, ...]) -> object: """What `value` holds at `loc`, each key naming a member of an object and each index an element of an array; `UNSET` where it holds nothing.""" for part in loc: - if isinstance(part, str) and isinstance(value, Mapping): - members = cast("Mapping[object, object]", value) + if isinstance(part, str) and is_object(value): + members = value if part not in members: return UNSET value = members[part] - elif ( - isinstance(part, int) - and isinstance(value, Sequence) - and not isinstance(value, (str, bytes, bytearray)) - and 0 <= part < len(cast("Sequence[object]", value)) - ): - value = cast("Sequence[object]", value)[part] + elif isinstance(part, int) and is_array(value) and 0 <= part < len(value): + value = value[part] else: return UNSET return value @@ -307,6 +302,37 @@ def validate_json(value: object, loc: tuple[str | int, ...] = ()) -> tuple[Valid return with_input(refine_json(value, loc)[1], value, loc) +def is_object(value: object) -> TypeGuard[Mapping[object, object]]: + """Whether `value` is a JSON object as Python holds one: a mapping, of any keys. + + What `isinstance(value, Mapping)` says, narrowed to a mapping of + `object`, which the bare check leaves unknown to a type checker. + """ + return isinstance(value, Mapping) + + +def is_json_object(value: object) -> TypeGuard[Mapping[str, object]]: + """Whether `value` is a JSON object with the keys JSON gives one: a mapping whose keys are all strings.""" + return isinstance(value, Mapping) and all( + isinstance(key, str) for key in cast("Mapping[object, object]", value) + ) + + +def is_list_or_tuple(value: object) -> TypeGuard[list[object] | tuple[object, ...]]: + """Whether `value` is a list or a tuple: the two containers canonical JSON holds an array in.""" + return isinstance(value, (list, tuple)) + + +def is_tuple(value: object) -> TypeGuard[tuple[object, ...]]: + """Whether `value` is a tuple, narrowed to a tuple of `object`.""" + return isinstance(value, tuple) + + +def is_array(value: object) -> TypeGuard[Sequence[object]]: + """Whether `value` is a JSON array as Python holds one: a sequence that is not text or bytes.""" + return isinstance(value, Sequence) and not isinstance(value, (str, bytes, bytearray)) + + def refine_json( value: object, loc: tuple[str | int, ...] = () ) -> tuple[JSONValue | None, tuple[ValidationProblem, ...]]: @@ -383,12 +409,12 @@ def _refine(value: object, loc: tuple[str | int, ...], *, finite: bool) -> _Refi return value, () if (past := nested_past_the_levels(value, loc)) is not None: return None, (past,) - if isinstance(value, Mapping): + if is_object(value): # Walked here rather than through `_refine_members`, so that each # level of nesting costs one frame, `JSON_DEPTH` of them at most. members: dict[str, JSONValue] = {} found_in_members: list[ValidationProblem] = [] - for key, item in cast("Mapping[object, object]", value).items(): + for key, item in value.items(): if not isinstance(key, str): found_in_members.append( ValidationProblem( @@ -401,10 +427,10 @@ def _refine(value: object, loc: tuple[str | int, ...], *, finite: bool) -> _Refi if len(found) == 0: members[key] = member return (members if len(found_in_members) == 0 else None), tuple(found_in_members) - if isinstance(value, Sequence) and not isinstance(value, (bytes, bytearray)): + if is_array(value): entries: list[JSONValue] = [] found_in_entries: list[ValidationProblem] = [] - for index, item in enumerate(cast("Sequence[object]", value)): + for index, item in enumerate(value): entry, found = _refine(item, (*loc, index), finite=finite) found_in_entries.extend(found) if len(found) == 0: @@ -535,13 +561,13 @@ def _is_canonical(value: object, depth: int, *, finite: bool) -> bool: return True if depth >= JSON_DEPTH: return False - if isinstance(value, (list, tuple)): - for item in cast("list[object] | tuple[object, ...]", value): + if is_list_or_tuple(value): + for item in value: # noqa: SIM110 - a loop, not a generator, is one frame per level if not _is_canonical(item, depth + 1, finite=finite): return False return True - if isinstance(value, dict): - for key, item in cast("dict[object, object]", value).items(): + if is_object(value): + for key, item in value.items(): if not isinstance(key, str) or not _is_canonical(item, depth + 1, finite=finite): return False return True @@ -590,31 +616,26 @@ def copied(value: JSONValue) -> JSONValue: if isinstance(value, (tuple, list)): # A loop, not a comprehension, which is a frame of its own before # Python 3.12: one frame for each level. - entries: list[JSONValue] = [] - for item in value: - entries.append(copied(item)) # noqa: PERF401 + entries = list(map(copied, value)) return entries if isinstance(value, list) else tuple(entries) return value def arrays_to_tuples(obj: object) -> object: """Recursively materialize mappings and convert array-like values to tuples.""" - if isinstance(obj, Sequence) and not isinstance(obj, (str, bytes, bytearray)): - sequence = cast("Sequence[object]", obj) + if is_array(obj): + sequence = obj # Loops, not comprehensions, which are a frame of their own before # Python 3.12: one frame for each level, as `copied` takes. - converted_items: list[object] = [] - for item in sequence: - converted_items.append(arrays_to_tuples(item)) # noqa: PERF401 - converted_sequence = tuple(converted_items) + converted_sequence = tuple(map(arrays_to_tuples, sequence)) if isinstance(obj, tuple) and all( converted is original for converted, original in zip(converted_sequence, sequence, strict=True) ): return sequence return converted_sequence - if isinstance(obj, Mapping): - mapping = cast("Mapping[object, object]", obj) + if is_object(obj): + mapping = obj converted: dict[object, object] = {} for key, value in mapping.items(): converted[key] = arrays_to_tuples(value) diff --git a/packages/zarr-metadata/src/zarr_metadata/_typed_json.py b/packages/zarr-metadata/src/zarr_metadata/_typed_json.py index 468755e27d..b1fecb6c81 100644 --- a/packages/zarr-metadata/src/zarr_metadata/_typed_json.py +++ b/packages/zarr-metadata/src/zarr_metadata/_typed_json.py @@ -73,6 +73,8 @@ choices, copied, is_json, + is_list_or_tuple, + is_object, outside_of, refine_json, shown, @@ -582,9 +584,9 @@ def sequence_of(element: Parser) -> Parser: """A member whose type is an array of one element type, parsed element by element.""" def parse(value: object, loc: Loc) -> Parsed: - if not isinstance(value, (list, tuple)): + if not is_list_or_tuple(value): return value, problem(loc, f"expected an array, got {shown(value)}") - entries = cast("list[object] | tuple[object, ...]", value) + entries = value parsed: list[object] = [] found: list[ValidationProblem] = [] for index, entry in enumerate(entries): @@ -600,9 +602,9 @@ def fixed_tuple(elements: Sequence[Parser], description: str) -> Parser: """A member whose type is an array of a fixed length, parsed position by position.""" def parse(value: object, loc: Loc) -> Parsed: - if not isinstance(value, (list, tuple)): + if not is_list_or_tuple(value): return value, problem(loc, f"expected {description}, got {shown(value)}") - entries = tuple(cast("list[object] | tuple[object, ...]", value)) + entries = tuple(value) if len(entries) != len(elements): return entries, problem(loc, f"expected {description}, got {shown(entries)}") parsed: list[object] = [] @@ -641,7 +643,7 @@ def any_of(branches: Sequence[Branch], description: str, tag: Tag | None = None) def parse(value: object, loc: Loc) -> Parsed: present = _keys_of(value) if tag is not None and present is not None: - return _by_tag(branches, tag, cast("Mapping[str, object]", value), loc) + return _by_tag(branches, tag, value, loc) clean: list[tuple[int, int, Parsed]] = [] failed: list[tuple[int, int, Parsed]] = [] for index, (shape, branch, keys) in enumerate(branches): @@ -663,10 +665,10 @@ def parse(value: object, loc: Loc) -> Parsed: return parse -def _by_tag(branches: Sequence[Branch], tag: Tag, value: Mapping[str, object], loc: Loc) -> Parsed: +def _by_tag(branches: Sequence[Branch], tag: Tag, value: object, loc: Loc) -> Parsed: """`value` parsed by the branch its tag picks; a tag missing, or one no branch has, reported at it.""" key, picks = tag - if key not in value: + if not is_object(value) or key not in value: return value, problem((*loc, key), f"missing required key {key!r}", "missing_key") said = value[key] index = picks.get((type(said), said)) if _hashable(said) else None @@ -677,8 +679,8 @@ def _by_tag(branches: Sequence[Branch], tag: Tag, value: Mapping[str, object], l def _keys_of(value: object) -> AbstractSet[str] | None: """The keys of `value` if it is a JSON object, else None.""" - if isinstance(value, Mapping): - return cast("Mapping[str, object]", value).keys() + if is_object(value): + return {key for key in value if isinstance(key, str)} return None @@ -697,12 +699,15 @@ def object_of(members: Mapping[str, tuple[Parser, bool]], extra: Parser | None) """ def parse(value: object, loc: Loc) -> Parsed: - if not isinstance(value, Mapping): + if not is_object(value): return value, problem(loc, f"expected an object, got {shown(value)}") - entries = cast("Mapping[str, object]", value) + entries = value parsed: dict[str, object] = {} found: list[ValidationProblem] = [] for key, entry in entries.items(): + if not isinstance(key, str): + found.extend(problem(loc, f"non-string key {shown(key)}")) + continue if key in members: continue if extra is None: @@ -716,7 +721,9 @@ def parse(value: object, loc: Loc) -> Parsed: found.extend(problems) elif required: found.extend(problem((*loc, key), f"missing required key {key!r}", "missing_key")) - return {key: parsed[key] for key in entries if key in parsed}, tuple(found) + return { + key: parsed[key] for key in entries if isinstance(key, str) and key in parsed + }, tuple(found) return parse @@ -729,12 +736,15 @@ def mapping_of(value: Parser) -> Parser: """ def parse(candidate: object, loc: Loc) -> Parsed: - if not isinstance(candidate, Mapping): + if not is_object(candidate): return candidate, problem(loc, f"expected an object, got {shown(candidate)}") - entries = cast("Mapping[str, object]", candidate) + entries = candidate parsed: dict[str, object] = {} found: list[ValidationProblem] = [] for key, entry in entries.items(): + if not isinstance(key, str): + found.extend(problem(loc, f"non-string key {shown(key)}")) + continue item, problems = value(entry, (*loc, key)) parsed[key] = item found.extend(problems) diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_array.py b/packages/zarr-metadata/src/zarr_metadata/model/_array.py index 1155080b92..4fcb66d312 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_array.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/_array.py @@ -17,6 +17,7 @@ ValidationProblem, copied, frozen, + is_object, json_text, refine_user_data, with_input, @@ -851,9 +852,9 @@ def from_key_value( JSON, `.zarray` holds `attributes`, or the document is not valid. """ zarray_raw = load_store_json(mapping, ZARR_V2_ARRAY_METADATA_STORE_KEY) - if not isinstance(zarray_raw, Mapping): + if not is_object(zarray_raw): return cls(zarray_raw, context=context) - zarray = cast("Mapping[str, object]", zarray_raw) + zarray = zarray_raw if "attributes" in zarray: refused = ValidationProblem( ("attributes",), "unexpected document member", "invalid_value" diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_group.py b/packages/zarr-metadata/src/zarr_metadata/model/_group.py index f9706d782c..8a0bf42257 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_group.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/_group.py @@ -18,6 +18,7 @@ copied, frozen, is_canonical_json, + is_object, json_text, nested_past_the_levels, not_an_object, @@ -652,9 +653,9 @@ def _node_type(value: object) -> tuple[str | None, tuple[ValidationProblem, ...] A document that says none is judged by its `zarr_format` too, so one of another format says it is not v3. """ - if not isinstance(value, Mapping): + if not is_object(value): return None, not_an_object(value) - document = cast("Mapping[object, object]", value) + document = value node_type = document.get("node_type") if isinstance(node_type, str) and node_type in _NODE_TYPES: return node_type, () @@ -713,9 +714,9 @@ def read_group_v3( so a chain of them is bounded by the levels a reader walks, counted from the outermost root. """ - if not isinstance(value, Mapping): + if not is_object(value): return ZarrV3GroupMetadataReading(problems=not_an_object(value)), None - doc = cast("Mapping[object, object]", value) + doc = value found: list[ValidationProblem] = list(missing_keys(GROUP_METADATA_REQUIRED_KEYS_V3, doc)) past = members_past_the_levels(doc, at) found.extend(past.values()) @@ -760,9 +761,9 @@ def documents_for(value: object) -> object: What a group's own document is built from, so a document built of models is JSON as any other, each child written as the child wrote it. """ - if not isinstance(value, Mapping): + if not is_object(value): return value - document = cast("Mapping[object, object]", value) + document = value given = document.get(ZARR_V3_CONSOLIDATED_METADATA_KEY) member = _member_documents_for(given) if member is given: @@ -774,14 +775,14 @@ def _member_documents_for(member: object) -> object: """A `consolidated_metadata` member with each node model it lists replaced by its document; `member` itself when it lists none.""" if isinstance(member, ZarrV3ConsolidatedMetadata): return member._document # pyright: ignore[reportPrivateUsage] - if not isinstance(member, Mapping): + if not is_object(member): + return member + entries = member.get("metadata") + if not is_object(entries): return member - entries = cast("Mapping[object, object]", member).get("metadata") - if not isinstance(entries, Mapping): - return cast("object", member) replaced: dict[object, object] = {} changed = False - for path, entry in cast("Mapping[object, object]", entries).items(): + for path, entry in entries.items(): if isinstance(entry, (ZarrV3ArrayMetadata, ZarrV3GroupMetadata)): replaced[path] = entry._document # pyright: ignore[reportPrivateUsage] changed = True @@ -790,8 +791,8 @@ def _member_documents_for(member: object) -> object: replaced[path] = documents_for(entry) changed = changed or replaced[path] is not entry if not changed: - return cast("object", member) - return {**cast("Mapping[object, object]", member), "metadata": replaced} + return member + return {**member, "metadata": replaced} def _read_node_model( @@ -884,9 +885,9 @@ def _read_consolidated_v3( past = nested_past_the_levels(value, at) if past is not None: return {}, {}, within((past,), at) - if not isinstance(value, Mapping): + if not is_object(value): return {}, {}, (ValidationProblem((), "expected an object", "invalid_type"),) - env = cast("Mapping[object, object]", value) + env = value # Missing members are reported in the order the envelope declares them. problems: list[ValidationProblem] = [ ValidationProblem((key,), "missing required key", "missing_key") @@ -901,12 +902,12 @@ def _read_consolidated_v3( node_types: dict[str, NodeType | None] = {} entries = env.get("metadata") past = nested_past_the_levels(entries, (*at, "metadata")) - if "metadata" in env and not isinstance(entries, Mapping): + if "metadata" in env and not is_object(entries): problems.append(ValidationProblem(("metadata",), "expected an object", "invalid_type")) - elif isinstance(entries, Mapping) and past is not None: + elif is_object(entries) and past is not None: problems.extend(within((past,), at)) - elif isinstance(entries, Mapping): - for key, entry in cast("Mapping[object, object]", entries).items(): + elif is_object(entries): + for key, entry in entries.items(): if not isinstance(key, str): problems.append( ValidationProblem( @@ -1077,15 +1078,15 @@ def _with_models( document = cast("dict[str, JSONValue]", refined) model = ZarrV3GroupMetadata._of(document, context, reading, members) # pyright: ignore[reportPrivateUsage] return model.reading - if members.consolidated is UNSET or not isinstance(value, Mapping): + if members.consolidated is UNSET or not is_object(value): return reading - member = cast("Mapping[object, object]", value).get(ZARR_V3_CONSOLIDATED_METADATA_KEY) - if not isinstance(member, Mapping): + member = value.get(ZARR_V3_CONSOLIDATED_METADATA_KEY) + if not is_object(member): return reading - entries = cast("Mapping[object, object]", member).get("metadata") - if not isinstance(entries, Mapping): + entries = member.get("metadata") + if not is_object(entries): return reading - held_entries = cast("Mapping[object, object]", entries) + held_entries = entries documents: dict[str, JSONValue] = {} for path in members.consolidated: if path not in held_entries: @@ -1347,9 +1348,9 @@ def from_key_value( JSON, `.zgroup` holds `attributes`, or the document is not valid. """ zgroup_raw = load_store_json(mapping, ZARR_V2_GROUP_METADATA_STORE_KEY) - if not isinstance(zgroup_raw, Mapping): + if not is_object(zgroup_raw): return cls(zgroup_raw, context=context) - zgroup = cast("Mapping[str, object]", zgroup_raw) + zgroup = zgroup_raw if "attributes" in zgroup: # A key `.zgroup` does not declare: its attributes are `.zattrs`. refused = ValidationProblem( @@ -1596,9 +1597,9 @@ def _read_consolidated_v2( its entry and the attributes' under the `.zattrs` entry. The nodes are empty when anything is wrong. """ - if not isinstance(data, Mapping): + if not is_object(data): return {}, {}, not_an_object(data) - doc = cast("Mapping[object, object]", data) + doc = data problems: list[ValidationProblem] = [ ValidationProblem((key,), "missing required key", "missing_key") for key in ("zarr_consolidated_format", "metadata") @@ -1609,16 +1610,16 @@ def _read_consolidated_v2( refined: dict[str, JSONValue] = {} if "metadata" in doc: entries = doc["metadata"] - if not isinstance(entries, Mapping) or not all( - isinstance(k, str) for k in cast("Mapping[object, object]", entries) - ): + if not is_object(entries) or not all(isinstance(k, str) for k in entries): problems.append( ValidationProblem( ("metadata",), "expected an object with string keys", "invalid_type" ) ) else: - for key, value in cast("Mapping[str, object]", entries).items(): + for key, value in entries.items(): + if not isinstance(key, str): + continue refine = ( refine_user_data if key.rsplit("/", 1)[-1] == ZARR_V2_ATTRIBUTES_STORE_KEY diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_repair.py b/packages/zarr-metadata/src/zarr_metadata/model/_repair.py index 7e198497f1..baa5f3902a 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_repair.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/_repair.py @@ -24,7 +24,13 @@ from zarr_metadata._common import ( JSONValue, ) -from zarr_metadata._json import JSON_DEPTH, MetadataValidationError, ValidationProblem +from zarr_metadata._json import ( + JSON_DEPTH, + MetadataValidationError, + ValidationProblem, + is_json_object, + is_object, +) from zarr_metadata._typed_json import Loc, check from zarr_metadata.model._group import ( ZarrV2ConsolidatedMetadata, @@ -111,11 +117,11 @@ def repair_node_metadata_v3(value: object) -> tuple[object, tuple[Repair, ...]]: def _repaired(value: object, at: Loc) -> tuple[object, tuple[Repair, ...]]: - if not isinstance(value, Mapping) or len(at) >= JSON_DEPTH: + if not is_json_object(value) or len(at) >= JSON_DEPTH: # Not a document, or past the levels a reader walks, which the # strict read reports. - return cast("object", value), () - original = cast("Mapping[str, object]", value) + return value, () + original = value repairs: list[Repair] = [] document = _zero_chunk_length(original, at, repairs) document = _null_consolidated_metadata(document, at, repairs) @@ -185,15 +191,15 @@ def _consolidated( ) -> Mapping[str, object]: """`document` with each document its consolidated metadata holds repaired, where it holds any.""" member = document.get(ZARR_V3_CONSOLIDATED_METADATA_KEY) - if not isinstance(member, Mapping): + if not is_object(member): return document - envelope = cast("Mapping[str, object]", member) + envelope = member entries = envelope.get("metadata") - if not isinstance(entries, Mapping): + if not is_object(entries): return document held: dict[object, object] = {} found = len(repairs) - for key, entry in cast("Mapping[object, object]", entries).items(): + for key, entry in entries.items(): if isinstance(key, str): entry, inside = _repaired( entry, (*at, ZARR_V3_CONSOLIDATED_METADATA_KEY, "metadata", key) @@ -248,15 +254,15 @@ def repair_consolidated_metadata_v2(value: object) -> tuple[object, tuple[Repair applies to is left as it is, and `value` is not changed; a document with none of the bugs is given back, and no repairs. """ - if not isinstance(value, Mapping): + if not is_json_object(value): return value, () - document = cast("Mapping[str, object]", value) + document = value entries = document.get("metadata") - if not isinstance(entries, Mapping): + if not is_object(entries): return cast("object", value), () repairs: list[Repair] = [] held: dict[object, object] = {} - for key, entry in cast("Mapping[object, object]", entries).items(): + for key, entry in entries.items(): if isinstance(key, str): path, _, name = key.rpartition("/") # Below the root only: no writer puts the member in the root's .zgroup. @@ -270,9 +276,9 @@ def repair_consolidated_metadata_v2(value: object) -> tuple[object, tuple[Repair def _without_consolidated_metadata(entry: object, at: Loc, repairs: list[Repair]) -> object: """`entry`, a `.zgroup` entry, without the member zarr-python 3.x writes into it, when it is one such.""" - if not isinstance(entry, Mapping): + if not is_json_object(entry): return entry - group = cast("Mapping[str, object]", entry) + group = entry shaped, problems = check( _members(group, ("zarr_format", ZARR_V3_CONSOLIDATED_METADATA_KEY)), ZarrV2ZGroupWithConsolidatedMetadataJSON, diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_validation.py b/packages/zarr-metadata/src/zarr_metadata/model/_validation.py index 96b73f363f..f6a2315682 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_validation.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/_validation.py @@ -34,6 +34,8 @@ MetadataValidationError, ValidationProblem, arrays_to_tuples, + is_object, + is_tuple, nested_past_the_levels, not_an_object, outside_of, @@ -247,12 +249,12 @@ def _is_canonical_dtype_v2(value: object) -> bool: """Whether a validated v2 dtype uses the tuple-backed public representation.""" if isinstance(value, str): return True - if not isinstance(value, tuple): + if not is_tuple(value): return False - for record in cast("tuple[object, ...]", value): - if not isinstance(record, tuple): + for record in value: + if not is_tuple(record): return False - fields = cast("tuple[object, ...]", record) + fields = record if not _is_canonical_dtype_v2(fields[1]): return False if len(fields) == 3 and not isinstance(fields[2], tuple): @@ -346,9 +348,7 @@ def attributes_of( the levels a reader walks are counted from that one's root; the problems are located in the document holding it. """ - if not isinstance(value, Mapping) or not all( - isinstance(k, str) for k in cast("Mapping[object, object]", value) - ): + if not is_object(value) or not all(isinstance(k, str) for k in value): return None, ( ValidationProblem( ("attributes",), "expected an object with string keys", "invalid_type" @@ -356,7 +356,9 @@ def attributes_of( ) attributes: dict[str, JSONValue] = {} problems: list[ValidationProblem] = [] - for key, item in cast("Mapping[str, object]", value).items(): + for key, item in value.items(): + if not isinstance(key, str): + continue refined, found = refine_user_data(item, (*at, "attributes", key)) problems.extend(found) attributes[key] = refined @@ -510,9 +512,9 @@ def read_array_v3( its group's -- so the levels a reader walks are counted from that one's root; the problems are located in this document. """ - if not isinstance(value, Mapping): + if not is_object(value): return ZarrV3ArrayMetadataReading(problems=not_an_object(value)), None - doc = cast("Mapping[object, object]", value) + doc = value problems: list[ValidationProblem] = list(missing_keys(ARRAY_METADATA_REQUIRED_KEYS_V3, doc)) past = members_past_the_levels(doc, at) problems.extend(past.values()) @@ -684,9 +686,9 @@ def read_array_v2( `filters` are required keys that may be `None`. `at` is where the document sits in the one handed in, which prefixes every problem. """ - if not isinstance(value, Mapping): + if not is_object(value): return ZarrV2ArrayMetadataReading(problems=within(not_an_object(value), at)), None - doc = cast("Mapping[object, object]", value) + doc = value # Unlike the group document ("Other keys MUST NOT be present", # https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v2/v2.0.rst#L313), the v2 array document is open: other keys "SHOULD NOT be # present within the metadata object and SHOULD be ignored by @@ -824,9 +826,9 @@ def validate_group_metadata_v2( optional `attributes` mapping folded in from `.zattrs`. A group holds no field a scope reads; `context` is taken as every v2 reader takes it. """ - if not isinstance(value, Mapping): + if not is_object(value): return not_an_object(value) - doc = cast("Mapping[object, object]", value) + doc = value problems: list[ValidationProblem] = list(missing_keys(GROUP_METADATA_REQUIRED_KEYS_V2, doc)) problems.extend(unexpected_keys(GROUP_METADATA_STANDARD_KEYS_V2, doc)) problems.extend(check_literal(doc, "zarr_format", 2)) diff --git a/packages/zarr-metadata/src/zarr_metadata/pydantic.py b/packages/zarr-metadata/src/zarr_metadata/pydantic.py index 8c7bf064bf..01e972042f 100644 --- a/packages/zarr-metadata/src/zarr_metadata/pydantic.py +++ b/packages/zarr-metadata/src/zarr_metadata/pydantic.py @@ -51,14 +51,13 @@ class ArrayManifest(BaseModel): from __future__ import annotations -from collections.abc import Mapping from typing import TYPE_CHECKING, Annotated, Final, LiteralString, Protocol, TypeVar, cast from pydantic import BeforeValidator, InstanceOf, PlainSerializer, ValidationInfo from pydantic_core import InitErrorDetails, PydanticCustomError, ValidationError from zarr_metadata import model as _model -from zarr_metadata._json import MetadataValidationError, ValidationProblem, value_at +from zarr_metadata._json import MetadataValidationError, ValidationProblem, is_object, value_at from zarr_metadata._pydantic_schema import ( ZarrV2ArrayMetadataJSON as _ZarrV2ArrayMetadataSchema, ) @@ -166,9 +165,9 @@ def _scope(context: object, default: Context, key: str) -> Context: """The scope a validation context holds for one format: itself, a `Context`, the scope of every field type; its `key` item, the format's own; or, holding none, `default`, the format's core scope.""" if isinstance(context, Context): return context - if not isinstance(context, Mapping) or key not in context: + if not is_object(context) or key not in context: return default - scope = cast("Mapping[object, object]", context)[key] + scope = context[key] if not isinstance(scope, Context): msg = f"{key}: the scope to read in is a Context, got {scope!r}" raise TypeError(msg) diff --git a/packages/zarr-metadata/src/zarr_metadata/v2/_definition.py b/packages/zarr-metadata/src/zarr_metadata/v2/_definition.py index 0e52bf6d65..9bfb73b60c 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v2/_definition.py +++ b/packages/zarr-metadata/src/zarr_metadata/v2/_definition.py @@ -19,17 +19,18 @@ from __future__ import annotations import re -from collections.abc import Mapping from dataclasses import dataclass from typing import TYPE_CHECKING, ClassVar, Final, cast from typing_extensions import TypeAliasType -from zarr_metadata._json import ValidationProblem, shown +from zarr_metadata._json import ValidationProblem, is_object, shown from zarr_metadata.v2.array import ZarrV2DataTypeMetadata from zarr_metadata.v3._definition import C, Definition, WithFillValue if TYPE_CHECKING: + from collections.abc import Mapping + from zarr_metadata._common import JSONValue from zarr_metadata._typed_json import Loc from zarr_metadata.v3._definition import Problems @@ -220,23 +221,26 @@ def name_problem(cls, name: str, at: Loc) -> ValidationProblem | None: def named_configuration( cls, value: object ) -> tuple[str | None, Mapping[str, object] | None, Problems]: - if not isinstance(value, Mapping): + if not is_object(value): return None, None, () - entry = cast("Mapping[str, object]", value) - name = entry.get("id") + name = value.get("id") if not isinstance(name, str): return None, None, () - return name, {key: item for key, item in entry.items() if key != "id"}, () + return ( + name, + {key: item for key, item in value.items() if isinstance(key, str) and key != "id"}, + (), + ) @classmethod def envelope_problems(cls, value: object) -> Problems: - if not isinstance(value, Mapping): + if not is_object(value): return ( ValidationProblem( (), "expected a codec configuration with a string 'id'", "invalid_type" ), ) - entry = cast("Mapping[str, object]", value) + entry = value if "id" not in entry: return (ValidationProblem(("id",), "missing required key", "missing_key"),) if not isinstance(entry["id"], str): diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/_common.py b/packages/zarr-metadata/src/zarr_metadata/v3/_common.py index d433cbb270..f71b347d5c 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/_common.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/_common.py @@ -8,7 +8,6 @@ """ import re -from collections.abc import Mapping from typing import Final, TypeAlias, TypeGuard, cast from typing_extensions import TypeAliasType @@ -19,6 +18,7 @@ ValidationProblem, arrays_to_tuples, is_canonical_json, + is_object, shown, shown_key, validate_json, @@ -130,7 +130,7 @@ def envelope_problems( if isinstance(value, str): bad = name_problem(value, ()) return () if bad is None else (bad,) - if not isinstance(value, Mapping): + if not is_object(value): return ( ValidationProblem( (), @@ -138,7 +138,7 @@ def envelope_problems( "invalid_type", ), ) - field = cast("Mapping[object, object]", value) + field = value problems: list[ValidationProblem] = [] for key in field: if not isinstance(key, str): @@ -157,11 +157,11 @@ def envelope_problems( problems.append(bad) if "configuration" in field: configuration = field["configuration"] - if not isinstance(configuration, Mapping): + if not is_object(configuration): problems.append( ValidationProblem(("configuration",), "expected an object", "invalid_type") ) - elif not all(isinstance(k, str) for k in cast("Mapping[object, object]", configuration)): + elif not all(isinstance(k, str) for k in configuration): problems.append( ValidationProblem(("configuration",), "expected string keys", "invalid_type") ) @@ -184,17 +184,18 @@ def envelope_problems( def _configuration_json_problems(value: object) -> tuple[ValidationProblem, ...]: """Each member of `value`'s configuration that is not JSON, located; nothing where no object of string keys is there to walk.""" - if not isinstance(value, Mapping): + if not is_object(value): return () - configuration = cast("Mapping[object, object]", value).get("configuration") - if not isinstance(configuration, Mapping): + configuration = value.get("configuration") + if not is_object(configuration): return () - members = cast("Mapping[object, object]", configuration) + members = configuration if not all(isinstance(key, str) for key in members): return () return tuple( found - for key, item in cast("Mapping[str, object]", members).items() + for key, item in members.items() + if isinstance(key, str) for found in validate_json(item, ("configuration", key)) ) @@ -203,10 +204,9 @@ def is_metadata_field_v3(value: object) -> TypeGuard[ZarrV3MetadataFieldJSON]: """Whether `value` is a v3 metadata field: a bare name as the spec names an extension, or a named config.""" if isinstance(value, str): return well_named(value) - if not isinstance(value, dict): + if not is_object(value) or not isinstance(value, dict): return False - field = cast("dict[object, object]", value) - return is_canonical_json(field) and not validate_metadata_field_v3(field) + return is_canonical_json(value) and not validate_metadata_field_v3(value) def parse_metadata_field_v3(value: object) -> ZarrV3MetadataFieldJSON: diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py b/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py index af5b76d809..5706af3549 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py @@ -56,6 +56,8 @@ from zarr_metadata._json import ( ValidationProblem, copied, + is_object, + is_tuple, json_text, refine_json, shown, @@ -966,8 +968,8 @@ def _collect(value: object, nested: list[_NestedField]) -> None: """Each nested field the checker handed back in `value`, in order.""" if isinstance(value, _NestedField): nested.append(value) - elif isinstance(value, tuple): - for entry in cast("tuple[object, ...]", value): + elif is_tuple(value): + for entry in value: _collect(entry, nested) elif isinstance(value, dict): for entry in cast("dict[str, object]", value).values(): @@ -991,8 +993,8 @@ def _put_back(value: object, put: Callable[[_NestedField], JSONValue]) -> object """`value` with each nested field the checker handed back put back as `put` gives it.""" if isinstance(value, _NestedField): return put(value) - if isinstance(value, tuple): - return tuple(_put_back(entry, put) for entry in cast("tuple[object, ...]", value)) + if is_tuple(value): + return tuple(_put_back(entry, put) for entry in value) if isinstance(value, dict): entries = cast("dict[str, object]", value) return {key: _put_back(entry, put) for key, entry in entries.items()} @@ -1091,9 +1093,9 @@ def named_configuration( """ if isinstance(value, str): return value, None, () - if not isinstance(value, Mapping): + if not is_object(value): return None, None, () - entry = cast("Mapping[str, object]", value) + entry = value name = entry.get("name") if not isinstance(name, str): return None, None, () @@ -1408,9 +1410,9 @@ def _is_data_type_field(value: object) -> bool: def _is_lengths(value: object) -> TypeGuard[Lengths]: """Whether `value` is chunk lengths: per axis, a frozenset of integers, or None.""" - if not isinstance(value, tuple): + if not is_tuple(value): return False - for axis in cast("tuple[object, ...]", value): + for axis in value: if axis is None: continue if not isinstance(axis, frozenset) or not all( diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/_pipeline.py b/packages/zarr-metadata/src/zarr_metadata/v3/_pipeline.py index e80a50d1a2..0097f7e67e 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/_pipeline.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/_pipeline.py @@ -26,11 +26,10 @@ from __future__ import annotations import dataclasses -from collections.abc import Mapping from dataclasses import dataclass from typing import TYPE_CHECKING, Any, Final, TypeGuard, cast -from zarr_metadata._json import ValidationProblem, with_input +from zarr_metadata._json import ValidationProblem, is_object, with_input from zarr_metadata.v3._definition import ( Chunk, CodecDefinition, @@ -43,7 +42,7 @@ ) if TYPE_CHECKING: - from collections.abc import Iterator, Sequence + from collections.abc import Iterator, Mapping, Sequence from zarr_metadata._typed_json import Loc from zarr_metadata.v3._definition import Nested, Problems @@ -227,9 +226,8 @@ def _inner_pipelines( def _is_pipelines(value: object) -> TypeGuard[Mapping[str, Chunk]]: """Whether `value` maps members of a configuration to chunks.""" - return isinstance(value, Mapping) and all( - isinstance(member, str) and isinstance(chunk, Chunk) - for member, chunk in cast("Mapping[object, object]", value).items() + return is_object(value) and all( + isinstance(member, str) and isinstance(chunk, Chunk) for member, chunk in value.items() ) From 9bf67b5ec87386861d091441a6e7431e2d1abaa5 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Thu, 8 Oct 2026 15:35:34 +0200 Subject: [PATCH 78/94] refactor(zarr-metadata): type the alias registry and the model's held documents instead of casting frozen keeps an object's type; refined_object and object_at check what a read already established instead of asserting it; is_alias and is_field are type guards; field_aliases and the kind registry hold TypeAliasType. Casts in src go from 179 to 128. Assisted-by: ClaudeCode:claude-fable-5-1 Co-Authored-By: Claude Fable 5.1 --- .../zarr-metadata/src/zarr_metadata/_json.py | 28 +++++++- .../src/zarr_metadata/_typed_json.py | 24 ++++--- .../src/zarr_metadata/model/_array.py | 26 +++---- .../src/zarr_metadata/model/_group.py | 71 ++++++++++--------- .../src/zarr_metadata/model/_repair.py | 6 +- .../src/zarr_metadata/model/_validation.py | 9 +-- .../src/zarr_metadata/v2/_definition.py | 2 +- .../src/zarr_metadata/v3/_definition.py | 60 +++++++++------- 8 files changed, 131 insertions(+), 95 deletions(-) diff --git a/packages/zarr-metadata/src/zarr_metadata/_json.py b/packages/zarr-metadata/src/zarr_metadata/_json.py index 9c40c15141..3bc5a417d6 100644 --- a/packages/zarr-metadata/src/zarr_metadata/_json.py +++ b/packages/zarr-metadata/src/zarr_metadata/_json.py @@ -15,7 +15,7 @@ from collections.abc import Callable, Iterator, Mapping, Sequence from dataclasses import dataclass from types import MappingProxyType -from typing import Final, Literal, TypeGuard, cast, get_args +from typing import Final, Literal, TypeGuard, cast, get_args, overload from zarr_metadata._common import JSONValue from zarr_metadata._sentinel import UNSET @@ -311,6 +311,28 @@ def is_object(value: object) -> TypeGuard[Mapping[object, object]]: return isinstance(value, Mapping) +def refined_object(value: object) -> dict[str, JSONValue]: + """`value`, an object a read found nothing wrong with, refined as user data: the document a model holds. + + `TypeError` when it is not an object, which a read that found no + problem rules out. + """ + refined, _ = refine_user_data(value) + if not isinstance(refined, dict): + msg = f"expected an object a read found nothing wrong with, got {shown(value)}" + raise TypeError(msg) + return refined + + +def object_at(document: Mapping[str, JSONValue], key: str) -> dict[str, JSONValue]: + """The object `document`, refined, holds at `key`, which a read found there; `TypeError` when something else is, which that read rules out.""" + value = document[key] + if not isinstance(value, dict): + msg = f"expected an object at {key!r}, got {shown(value)}" + raise TypeError(msg) + return value + + def is_json_object(value: object) -> TypeGuard[Mapping[str, object]]: """Whether `value` is a JSON object with the keys JSON gives one: a mapping whose keys are all strings.""" return isinstance(value, Mapping) and all( @@ -587,6 +609,10 @@ def parse_json(value: object) -> JSONValue: return refined +@overload +def frozen(value: Mapping[str, JSONValue]) -> Mapping[str, JSONValue]: ... +@overload +def frozen(value: JSONValue) -> JSONValue: ... def frozen(value: JSONValue) -> JSONValue: """`value` as a read-only view at every level: each object a mapping proxy, each array a tuple, so nothing handed out can be changed in place. diff --git a/packages/zarr-metadata/src/zarr_metadata/_typed_json.py b/packages/zarr-metadata/src/zarr_metadata/_typed_json.py index b1fecb6c81..261e000028 100644 --- a/packages/zarr-metadata/src/zarr_metadata/_typed_json.py +++ b/packages/zarr-metadata/src/zarr_metadata/_typed_json.py @@ -56,6 +56,7 @@ NewType, NoReturn, TypeAlias, + TypeGuard, TypeVar, cast, get_args, @@ -111,9 +112,12 @@ # A `type` statement makes a `typing.TypeAliasType`, which is not the # `typing_extensions` one on every version that has both. -_ALIASES: Final[tuple[type, ...]] = ( +_ALIASES: Final[tuple[type[typing_extensions.TypeAliasType], ...]] = ( typing_extensions.TypeAliasType, - getattr(typing, "TypeAliasType", typing_extensions.TypeAliasType), + cast( + "type[typing_extensions.TypeAliasType]", + getattr(typing, "TypeAliasType", typing_extensions.TypeAliasType), + ), ) @@ -198,11 +202,9 @@ def is_union(annotation: object) -> bool: return get_origin(annotation) in (typing.Union, types.UnionType) -def is_alias(annotation: object) -> bool: +def is_alias(annotation: object) -> TypeGuard[typing_extensions.TypeAliasType]: """Whether `annotation` is a type alias that takes no type parameters: `type Level = int`.""" - if not isinstance(annotation, _ALIASES): - return False - return len(cast("typing_extensions.TypeAliasType", annotation).__type_params__) == 0 + return isinstance(annotation, _ALIASES) and len(annotation.__type_params__) == 0 @functools.cache @@ -458,7 +460,7 @@ def _named(annotation: object, seen: frozenset[object]) -> tuple[str, str]: if isinstance(inner, NewType): return _named(inner.__supertype__, seen) if is_alias(inner): - alias = cast("typing_extensions.TypeAliasType", inner) + alias = inner if alias in seen or _holds(alias_value(alias), alias, frozenset()): return f"a {alias.__name__}", f"{alias.__name__} values" return _named(alias_value(alias), seen | {alias}) @@ -479,7 +481,7 @@ def _holds(annotation: object, alias: object, seen: frozenset[object]) -> bool: if is_alias(inner): if inner in seen: return False - value = alias_value(cast("typing_extensions.TypeAliasType", inner)) + value = alias_value(inner) return _holds(value, alias, seen | {inner}) return any(_holds(argument, alias, seen) for argument in get_args(inner)) @@ -497,7 +499,7 @@ def shape_of(annotation: object) -> str | None: if isinstance(inner, NewType): inner = strip_annotation(inner.__supertype__)[0] else: - inner = strip_annotation(alias_value(cast("typing_extensions.TypeAliasType", inner)))[0] + inner = strip_annotation(alias_value(inner))[0] if inner is int: return "int" if inner is float: @@ -1064,7 +1066,7 @@ def _compile_type(inner: object, leaf: Leaf, building: _Building) -> Parser | No # the code's, for a value it has vouched for. return _compile(inner.__supertype__, leaf, building) if is_alias(inner): - return _alias(cast("typing_extensions.TypeAliasType", inner), leaf, building) + return _alias(inner, leaf, building) return None @@ -1339,7 +1341,7 @@ def _type(self, inner: object) -> JSONSchema: if isinstance(inner, NewType): return self.of(inner.__supertype__) if is_alias(inner): - alias = cast("typing_extensions.TypeAliasType", inner) + alias = inner return self.defined(alias, alias.__name__, lambda: self.of(alias_value(alias))) msg = f"{inner!r} is not a shape JSON takes" raise TypeError(msg) diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_array.py b/packages/zarr-metadata/src/zarr_metadata/model/_array.py index 4fcb66d312..de90712fff 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_array.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/_array.py @@ -19,7 +19,7 @@ frozen, is_object, json_text, - refine_user_data, + refined_object, with_input, ) from zarr_metadata._sentinel import UNSET @@ -150,8 +150,7 @@ def __init__(self, document: object, context: Context | None = None) -> None: reading, members = read_array_v3(document, scope) if members is None: raise MetadataValidationError(reading.problems) - refined, _ = refine_user_data(document) - self._adopt(cast("dict[str, JSONValue]", refined), scope, reading, members) + self._adopt(refined_object(document), scope, reading, members) @classmethod def _of( @@ -183,7 +182,7 @@ def _adopt( self._shown = ( frozen(members.fill_value), frozen(members.attributes), - frozen(cast("JSONValue", members.extra_fields)), + frozen(members.extra_fields), ) self._key = array_key(self) self._claims = MappingProxyType(claims_of(reading.fields())) @@ -251,12 +250,12 @@ def dimension_names(self) -> tuple[str | None, ...] | UNSET: @property def attributes(self) -> Mapping[str, JSONValue]: """The attributes, read-only at every level; empty when the document writes none.""" - return cast("Mapping[str, JSONValue]", self._shown[1]) + return self._shown[1] @property def extra_fields(self) -> Mapping[str, ZarrV3ExtensionField]: """Each member the spec does not define, by name, read-only at every level.""" - return cast("Mapping[str, ZarrV3ExtensionField]", self._shown[2]) + return self._shown[2] @property def must_understand_fields(self) -> dict[str, ZarrV3ExtensionField]: @@ -506,8 +505,7 @@ def read_array_metadata_v3( reading, members = read_array_v3(value, scope) if members is None: return reading - refined, _ = refine_user_data(value) - document = cast("dict[str, JSONValue]", refined) + document = refined_object(value) model = ZarrV3ArrayMetadata._of(document, scope, reading, members) # pyright: ignore[reportPrivateUsage] return model.reading @@ -527,8 +525,7 @@ def read_array_metadata_v2( reading, members = read_array_v2(value, scope) if members is None: return reading - refined, _ = refine_user_data(value) - document = cast("dict[str, JSONValue]", refined) + document = refined_object(value) if "dimension_separator" not in document: document = {**document, "dimension_separator": "."} model = ZarrV2ArrayMetadata._of(document, scope, reading, members) # pyright: ignore[reportPrivateUsage] @@ -588,8 +585,7 @@ def __init__(self, document: object, context: Context | None = None) -> None: reading, members = read_array_v2(document, scope) if members is None: raise MetadataValidationError(reading.problems) - refined, _ = refine_user_data(document) - held = cast("dict[str, JSONValue]", refined) + held = refined_object(document) if "dimension_separator" not in held: held = {**held, "dimension_separator": "."} self._adopt(held, scope, reading, members) @@ -622,7 +618,7 @@ def _adopt( self._shown = ( frozen(members.fill_value), UNSET if members.attributes is UNSET else frozen(members.attributes), - frozen(cast("JSONValue", members.extra_fields)), + frozen(members.extra_fields), ) self._key = array_key_v2(self) self._claims = MappingProxyType(claims_of(reading.fields())) @@ -707,12 +703,12 @@ def dimension_separator(self) -> ZarrV2ArrayDimensionSeparator: @property def attributes(self) -> Mapping[str, JSONValue] | UNSET: """The user attributes a `.zattrs` holds, read-only at every level; `UNSET` when there is no `.zattrs`.""" - return cast("Mapping[str, JSONValue] | UNSET", self._shown[1]) + return self._shown[1] @property def extra_fields(self) -> Mapping[str, JSONValue]: """Every member the spec does not define, as written, read-only at every level.""" - return cast("Mapping[str, JSONValue]", self._shown[2]) + return self._shown[2] @property def dtype(self) -> Read[ZarrV2DataTypeDefinition[Any]] | Unclaimed: diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_group.py b/packages/zarr-metadata/src/zarr_metadata/model/_group.py index 8a0bf42257..67ba7d8124 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_group.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/_group.py @@ -18,13 +18,16 @@ copied, frozen, is_canonical_json, + is_json_object, is_object, json_text, nested_past_the_levels, not_an_object, + object_at, outside_of, refine_json, refine_user_data, + refined_object, shown, shown_key, with_input, @@ -157,8 +160,7 @@ def __init__(self, document: object, context: Context | None = None) -> None: reading, members = read_group_v3(document, scope) if members is None or len(reading.problems) != 0: raise MetadataValidationError(reading.problems) - refined, _ = refine_user_data(documents_for(document)) - self._adopt(cast("dict[str, JSONValue]", refined), scope, reading, members) + self._adopt(refined_object(documents_for(document)), scope, reading, members) @classmethod def _of( @@ -183,17 +185,20 @@ def _adopt( self._document = document self._context = context self._members = members + if members.attributes is None: + msg = "a group model holds members a read found nothing wrong with" + raise TypeError(msg) # What the model shows of its members, read-only at every level. self._shown = ( - frozen(cast("JSONValue", members.attributes)), - frozen(cast("JSONValue", members.extra_fields)), + frozen(members.attributes), + frozen(members.extra_fields), ) held: Mapping[str, ZarrV3NodeMetadataReading] = reading.consolidated if members.consolidated is UNSET: self._consolidated: ZarrV3ConsolidatedMetadata | UNSET = UNSET else: - member = cast("dict[str, JSONValue]", document[ZARR_V3_CONSOLIDATED_METADATA_KEY]) - documents = cast("dict[str, JSONValue]", member["metadata"]) + member = object_at(document, ZARR_V3_CONSOLIDATED_METADATA_KEY) + documents = object_at(member, "metadata") models = _nested_models(documents, context, reading.consolidated, members.consolidated) # One model per document, of this scope: each nested reading # holds the model the group holds. @@ -261,12 +266,12 @@ def __reduce__(self) -> tuple[type[ZarrV3GroupMetadata], tuple[object, Context]] @property def attributes(self) -> Mapping[str, JSONValue]: """The attributes, read-only at every level; empty when the document writes none.""" - return cast("Mapping[str, JSONValue]", self._shown[0]) + return self._shown[0] @property def extra_fields(self) -> Mapping[str, ZarrV3ExtensionField]: """Each member the spec does not define, `consolidated_metadata` apart, by name: read-only at every level.""" - return cast("Mapping[str, ZarrV3ExtensionField]", self._shown[1]) + return self._shown[1] @property def consolidated_metadata(self) -> ZarrV3ConsolidatedMetadata | UNSET: @@ -326,9 +331,7 @@ def refines(self, other: ZarrV3GroupMetadata) -> bool: if type(other) is not type(self): return False mine, theirs = self._members, other._members - if json_text(cast("JSONValue", mine.attributes)) != json_text( - cast("JSONValue", theirs.attributes) - ): + if json_text(mine.attributes) != json_text(theirs.attributes): return False if json_text(mine.extra_fields) != json_text(theirs.extra_fields): return False @@ -402,9 +405,8 @@ def __init__(self, member: object, context: Context | None = None) -> None: ) if len(problems) != 0: raise MetadataValidationError(problems) - refined, _ = refine_user_data(_member_documents_for(member)) - document = cast("dict[str, JSONValue]", refined) - documents = cast("dict[str, JSONValue]", document["metadata"]) + document = refined_object(_member_documents_for(member)) + documents = object_at(document, "metadata") self._adopt(document, scope, _nested_models(documents, scope, readings, members)) @classmethod @@ -767,7 +769,7 @@ def documents_for(value: object) -> object: given = document.get(ZARR_V3_CONSOLIDATED_METADATA_KEY) member = _member_documents_for(given) if member is given: - return cast("object", value) + return value return {**document, ZARR_V3_CONSOLIDATED_METADATA_KEY: member} @@ -941,7 +943,7 @@ def group_key(model: ZarrV3GroupMetadata) -> tuple[object, ...]: consolidated = model.consolidated_metadata members = model._members # pyright: ignore[reportPrivateUsage] return ( - json_text(cast("JSONValue", members.attributes)), + json_text(members.attributes), UNSET if consolidated is UNSET else consolidated._key, # pyright: ignore[reportPrivateUsage] json_text(members.extra_fields), ) @@ -1074,8 +1076,7 @@ def _with_models( those of its own listing. """ if len(reading.problems) == 0: - refined, _ = refine_user_data(value) - document = cast("dict[str, JSONValue]", refined) + document = refined_object(value) model = ZarrV3GroupMetadata._of(document, context, reading, members) # pyright: ignore[reportPrivateUsage] return model.reading if members.consolidated is UNSET or not is_object(value): @@ -1140,7 +1141,7 @@ def _nested_models( if held is not None and held.context is context: models[path] = held continue - document = cast("dict[str, JSONValue]", documents[path]) + document = object_at(documents, path) if isinstance(reading, ZarrV3ArrayMetadataReading): models[path] = ZarrV3ArrayMetadata._of( # pyright: ignore[reportPrivateUsage] document, context, reading, cast("ArrayMembersV3", child) @@ -1217,7 +1218,7 @@ class ZarrV2GroupMetadata: _context: Context _document: dict[str, JSONValue] _key: tuple[object, ...] - _shown: object + _shown: Mapping[str, JSONValue] | UNSET zarr_format: Final = 2 @@ -1229,8 +1230,7 @@ def claims(self) -> Claims: def __init__(self, document: object, context: Context | None = None) -> None: scope = CORE_V2 if context is None else context parsed = parse_group_metadata_v2(document, context=scope) - refined, _ = refine_user_data(document) - self._adopt(cast("dict[str, JSONValue]", refined), scope, _v2_attributes(parsed)) + self._adopt(refined_object(document), scope, _v2_attributes(parsed)) @classmethod def _of( @@ -1264,7 +1264,7 @@ def context(self) -> Context: @property def attributes(self) -> Mapping[str, JSONValue] | UNSET: """The user attributes a `.zattrs` holds, read-only at every level; `UNSET` when there is no `.zattrs`.""" - return cast("Mapping[str, JSONValue] | UNSET", self._shown) + return self._shown def to_json(self) -> ZarrV2GroupMetadataJSON: """The merged document as written, refined, sharing nothing with the model. @@ -1395,7 +1395,7 @@ class ZarrV2ConsolidatedMetadata: _document: dict[str, JSONValue] _key: tuple[object, ...] _nodes: dict[str, ZarrV2NodeMetadata] - _shown: object + _shown: Mapping[str, JSONValue] zarr_consolidated_format: Final = 1 @@ -1421,7 +1421,7 @@ def _adopt( self._document = document self._context = context self._nodes = nodes - self._shown = frozen(document["metadata"]) + self._shown = frozen(object_at(document, "metadata")) self._key = consolidated_key_v2(self) @property @@ -1432,7 +1432,7 @@ def context(self) -> Context: @property def metadata(self) -> Mapping[str, JSONValue]: """The entries as written, refined, by store key; read-only at every level.""" - return cast("Mapping[str, JSONValue]", self._shown) + return self._shown @property def nodes(self) -> Mapping[str, ZarrV2NodeMetadata]: @@ -1478,7 +1478,7 @@ def refined_in(self, context: Context | None = None) -> ZarrV2ConsolidatedMetada """ scope = CORE_V2 if context is None else context conflicts: list[Conflict] = [] - entries = cast("Mapping[str, JSONValue]", self._document["metadata"]) + entries = object_at(self._document, "metadata") by_path, _ = _entries_by_path(entries) for path, node in self._nodes.items(): if not isinstance(node, ZarrV2ArrayMetadata): @@ -1569,7 +1569,7 @@ def _entries_by_path( def _other_entries_text(model: ZarrV2ConsolidatedMetadata) -> str: """The entries no node is read from -- an orphan `.zattrs`, any other key -- as JSON text: what `==` compares of them.""" - entries = cast("Mapping[str, JSONValue]", model._document["metadata"]) # pyright: ignore[reportPrivateUsage] + entries = object_at(model._document, "metadata") # pyright: ignore[reportPrivateUsage] consumed: set[str] = set() for path, names in _entries_by_path(entries)[0].items(): if path in model._nodes: # pyright: ignore[reportPrivateUsage] @@ -1665,13 +1665,13 @@ def _read_node_v2( if key is None: return None, [] document = entries[key] - if not isinstance(document, Mapping): + if not is_json_object(document): return None, [ ValidationProblem( ("metadata", key), f"expected an object, got {shown(document)}", "invalid_type" ) ] - merged = dict(cast("Mapping[str, JSONValue]", document)) + merged = dict(document) if "attributes" in merged: return None, [ ValidationProblem( @@ -1696,17 +1696,18 @@ def located(found: tuple[ValidationProblem, ...]) -> list[ValidationProblem]: reading, members = read_array_v2(merged, context) if members is None: return None, located(reading.problems) - held = merged if "dimension_separator" in merged else {**merged, "dimension_separator": "."} + held = refined_object(merged) + if "dimension_separator" not in held: + held["dimension_separator"] = "." return ZarrV2ArrayMetadata._of(held, context, reading, members), [] # pyright: ignore[reportPrivateUsage] found = validate_group_metadata_v2(merged, context=context) if len(found) != 0: return None, located(found) + document = refined_object(merged) attributes: dict[str, JSONValue] | UNSET = ( - dict(cast("Mapping[str, JSONValue]", merged["attributes"])) - if "attributes" in merged - else UNSET + object_at(document, "attributes") if "attributes" in document else UNSET ) - return ZarrV2GroupMetadata._of(merged, context, attributes), [] # pyright: ignore[reportPrivateUsage] + return ZarrV2GroupMetadata._of(document, context, attributes), [] # pyright: ignore[reportPrivateUsage] def _v2_attributes(document: ZarrV2GroupMetadataJSON) -> dict[str, JSONValue] | UNSET: diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_repair.py b/packages/zarr-metadata/src/zarr_metadata/model/_repair.py index baa5f3902a..75f532ea39 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_repair.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/_repair.py @@ -259,7 +259,7 @@ def repair_consolidated_metadata_v2(value: object) -> tuple[object, tuple[Repair document = value entries = document.get("metadata") if not is_object(entries): - return cast("object", value), () + return value, () repairs: list[Repair] = [] held: dict[object, object] = {} for key, entry in entries.items(): @@ -270,7 +270,7 @@ def repair_consolidated_metadata_v2(value: object) -> tuple[object, tuple[Repair entry = _without_consolidated_metadata(entry, ("metadata", key), repairs) held[key] = entry if len(repairs) == 0: - return cast("object", value), () + return value, () return {**document, "metadata": held}, tuple(repairs) @@ -284,7 +284,7 @@ def _without_consolidated_metadata(entry: object, at: Loc, repairs: list[Repair] ZarrV2ZGroupWithConsolidatedMetadataJSON, ) if shaped is None or len(problems) != 0: - return cast("object", entry) + return entry repairs.append( Repair( (*at, ZARR_V3_CONSOLIDATED_METADATA_KEY), diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_validation.py b/packages/zarr-metadata/src/zarr_metadata/model/_validation.py index f6a2315682..9e66137e8b 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_validation.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/_validation.py @@ -423,13 +423,14 @@ def fields(self) -> Iterator[tuple[Loc, Resolved[Any]]]: them: a shard's codecs, a struct's field types. `with_problems` gives each with its problems. """ - for key, field in ( + own: tuple[tuple[str, Resolved[Any] | UNSET], ...] = ( ("data_type", self.data_type), ("chunk_grid", self.chunk_grid), ("chunk_key_encoding", self.chunk_key_encoding), - ): + ) + for key, field in own: if field is not UNSET: - yield from fields_of(cast("Resolved[Any]", field), (key,)) + yield from fields_of(field, (key,)) for index, stage in enumerate(self.pipeline): yield from fields_of(stage.codec, ("codecs", index)) for index, transformer in enumerate(self.storage_transformers): @@ -493,7 +494,7 @@ def __reduce__(self) -> str | tuple[object, ...]: def fields(self) -> Iterator[tuple[Loc, Resolved[Any]]]: """Each field the document holds, as the scope read it, where it sits: the dtype, a struct's record types after it, the compressor, each filter at its index.""" if self.dtype is not UNSET: - yield from fields_of(cast("Resolved[Any]", self.dtype), ("dtype",)) + yield from fields_of(self.dtype, ("dtype",)) if self.compressor is not UNSET and self.compressor is not None: yield from fields_of(self.compressor, ("compressor",)) if self.filters is not UNSET and self.filters is not None: diff --git a/packages/zarr-metadata/src/zarr_metadata/v2/_definition.py b/packages/zarr-metadata/src/zarr_metadata/v2/_definition.py index 9bfb73b60c..4712fb44fe 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v2/_definition.py +++ b/packages/zarr-metadata/src/zarr_metadata/v2/_definition.py @@ -119,7 +119,7 @@ class ZarrV2DataTypeDefinition(WithFillValue[C]): is_kind: ClassVar[bool] = True label: ClassVar[str] = "v2 data type" - field_aliases: ClassVar[tuple[object, ...]] = (ZarrV2DataTypeField,) + field_aliases: ClassVar[tuple[TypeAliasType, ...]] = (ZarrV2DataTypeField,) def _refusal(self) -> str | None: if parse_typestr(self.name) is not None: diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py b/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py index 5706af3549..c2a0ebd361 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py @@ -71,6 +71,7 @@ Parser, SchemaLeaf, Schemas, + is_alias, parser, problem, typeddict_keys, @@ -174,7 +175,7 @@ class EmptyConfiguration(TypedDict, closed=True): """The configuration of a definition with nothing to configure: its field is written with its name alone.""" -_FIELD_KINDS: Final[dict[object, type[Definition[Any]]]] = {} +_FIELD_KINDS: Final[dict[TypeAliasType, type[Definition[Any]]]] = {} """Each field alias, and the kind a member annotated with it is read as; filed by each kind as its class is built.""" @@ -219,7 +220,7 @@ class Definition(Generic[C]): """ label: ClassVar[str] = "definition" """The kind as a message names it: "codec".""" - field_aliases: ClassVar[tuple[object, ...]] = () + field_aliases: ClassVar[tuple[TypeAliasType, ...]] = () """The field aliases a configuration member holding a field of this kind is annotated with: `CodecField` and `StaticCodecField` for a codec.""" name: str @@ -484,7 +485,7 @@ class DataTypeDefinition(WithFillValue[C]): is_kind: ClassVar[bool] = True label: ClassVar[str] = "data type" - field_aliases: ClassVar[tuple[object, ...]] = (DataTypeField,) + field_aliases: ClassVar[tuple[TypeAliasType, ...]] = (DataTypeField,) @classmethod def spelled(cls, name: str) -> tuple[str | None, dict[str, JSONValue] | None]: @@ -548,7 +549,7 @@ class ChunkGridDefinition(Definition[C]): is_kind: ClassVar[bool] = True label: ClassVar[str] = "chunk grid" - field_aliases: ClassVar[tuple[object, ...]] = (ChunkGridField,) + field_aliases: ClassVar[tuple[TypeAliasType, ...]] = (ChunkGridField,) shape_rules: Callable[[C, Nested, tuple[int, ...]], Iterable[ValidationProblem]] = no_rules """What the spec disallows in this grid over an array of a shape, located in the configuration.""" @@ -562,7 +563,7 @@ class ChunkKeyEncodingDefinition(Definition[C]): is_kind: ClassVar[bool] = True label: ClassVar[str] = "chunk key encoding" - field_aliases: ClassVar[tuple[object, ...]] = (ChunkKeyEncodingField,) + field_aliases: ClassVar[tuple[TypeAliasType, ...]] = (ChunkKeyEncodingField,) CodecKind = Literal["array_array", "array_bytes", "bytes_bytes"] @@ -617,7 +618,7 @@ class CodecDefinition(Definition[C]): is_kind: ClassVar[bool] = True label: ClassVar[str] = "codec" - field_aliases: ClassVar[tuple[object, ...]] = (CodecField, StaticCodecField) + field_aliases: ClassVar[tuple[TypeAliasType, ...]] = (CodecField, StaticCodecField) kind: CodecKind size: CodecSize @@ -648,7 +649,7 @@ class StorageTransformerDefinition(Definition[C]): is_kind: ClassVar[bool] = True label: ClassVar[str] = "storage transformer" - field_aliases: ClassVar[tuple[object, ...]] = (StorageTransformerField,) + field_aliases: ClassVar[tuple[TypeAliasType, ...]] = (StorageTransformerField,) KINDS: Final[tuple[type[Definition[Any]], ...]] = ( @@ -696,16 +697,15 @@ def as_kind(kind: object) -> type[Definition[Any]]: raise TypeError(msg) -_STATIC_SIZE: Final[frozenset[object]] = frozenset({StaticCodecField}) +_STATIC_SIZE: Final[frozenset[TypeAliasType]] = frozenset({StaticCodecField}) """The field aliases whose codec must be of static size.""" def field_kind(annotation: object) -> type[Definition[Any]] | None: """The kind of metadata field a member annotated `annotation` holds -- `CodecDefinition` for `CodecField` -- or None when it holds none.""" - try: - return _FIELD_KINDS.get(annotation) - except TypeError: # an unhashable annotation is no field alias + if not is_alias(annotation): return None + return _FIELD_KINDS.get(annotation) @dataclass(frozen=True, slots=True) @@ -758,7 +758,7 @@ def _vetting(annotation: object) -> Parser | None: msg = ( "a member typed ZarrV3MetadataFieldJSON checks as plain JSON, and is never read as " "a field; annotate it with the field alias of its kind: " - + ", ".join(cast("TypeAliasType", alias).__name__ for alias in _FIELD_KINDS) + + ", ".join(alias.__name__ for alias in _FIELD_KINDS) ) raise TypeError(msg) if ( @@ -778,8 +778,8 @@ def _vetting(annotation: object) -> Parser | None: def _no_field(annotation: object) -> Parser | None: """A leaf refusing a field alias: a fill value is a value of its data type, and holds no metadata field.""" - if field_kind(annotation) is not None: - name = cast("TypeAliasType", annotation).__name__ + if is_alias(annotation) and field_kind(annotation) is not None: + name = annotation.__name__ msg = f"{name} holds a metadata field, and a fill value is a value of its data type" raise TypeError(msg) return None @@ -846,9 +846,9 @@ def field_schemas(context: Context) -> SchemaLeaf: def leaf(annotation: object, schemas: Schemas) -> JSONSchema | None: kind = field_kind(annotation) - if kind is None: + if kind is None or not is_alias(annotation): return None - alias = cast("TypeAliasType", annotation) + alias = annotation static = annotation in _STATIC_SIZE return schemas.defined( alias, alias.__name__, lambda: _field_schema(kind, static, context, schemas) @@ -1155,9 +1155,9 @@ class Read(Generic[D]): """The kind of metadata it was read as: its definition's.""" def __eq__(self, other: object) -> bool: - if not isinstance(other, _FIELDS): + if not is_field(other): return NotImplemented - return field_key(self) == field_key(cast("Resolved[Any]", other)) + return field_key(self) == field_key(other) def __hash__(self) -> int: return hash(field_key(self)) @@ -1188,7 +1188,7 @@ def to_json(self) -> JSONValue: A name that carries its configuration, as raw bits' does, is written alone. """ - return copied(cast("JSONValue", document_json(self))) + return copied(written_json(self)) @dataclass(frozen=True, slots=True, kw_only=True) @@ -1210,9 +1210,9 @@ class Unclaimed: """The configuration as written, which nothing judged; empty when none is written.""" def __eq__(self, other: object) -> bool: - if not isinstance(other, _FIELDS): + if not is_field(other): return NotImplemented - return field_key(self) == field_key(cast("Resolved[Any]", other)) + return field_key(self) == field_key(other) def __hash__(self) -> int: return hash(field_key(self)) @@ -1244,7 +1244,7 @@ def nested(self) -> Nested: def to_json(self) -> JSONValue: """The field as a document writes it, sharing nothing with the field: its configuration as written, in the envelope every reader takes, as `Read.to_json` writes one.""" - return copied(cast("JSONValue", document_json(self))) + return copied(written_json(self)) @dataclass(frozen=True, slots=True, kw_only=True) @@ -1263,9 +1263,9 @@ class Refused(Generic[D]): """The fields its configuration holds, each as the scope read it; empty when its configuration was not checked against its TypedDict.""" def __eq__(self, other: object) -> bool: - if not isinstance(other, _FIELDS): + if not is_field(other): return NotImplemented - return field_key(self) == field_key(cast("Resolved[Any]", other)) + return field_key(self) == field_key(other) def __hash__(self) -> int: return hash(field_key(self)) @@ -1356,6 +1356,11 @@ def document_json(field: Resolved[Any]) -> JSONValue | UNSET: """ if isinstance(field, Refused): return field.json + return written_json(field) + + +def written_json(field: Read[Any] | Unclaimed) -> JSONValue: + """A field a scope read or left unclaimed, as a document writes it: `document_json` of one that is JSON.""" kind = field.read_as if isinstance(field, Read) and kind.spelled(field.name)[1] is not None: return kind.envelope_json(field.name, {}) @@ -1403,9 +1408,14 @@ def rank(self) -> int | None: return None if self.lengths is None else len(self.lengths) +def is_field(value: object) -> TypeGuard[Resolved[Any]]: + """Whether `value` is a field a scope read: `Read`, `Unclaimed` or `Refused`.""" + return isinstance(value, _FIELDS) + + def _is_data_type_field(value: object) -> bool: """Whether `value` is a field a scope read as a data type.""" - return isinstance(value, _FIELDS) and cast("Resolved[Any]", value).read_as is DataTypeDefinition + return is_field(value) and value.read_as is DataTypeDefinition def _is_lengths(value: object) -> TypeGuard[Lengths]: From 59282c5331cecbc31b4f5e9c105e694899d6e4f7 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Thu, 8 Oct 2026 15:42:28 +0200 Subject: [PATCH 79/94] refactor(zarr-metadata): models compute their own keys on a shared Keyed base Equality and hashing live once, in Keyed; each model computes its key in a method of its own class instead of a module function reading its private members. Entries given as node models are unwrapped through to_json. The remaining reportPrivateUsage sites are the readers building models through _of, which the docstrings now say. Assisted-by: ClaudeCode:claude-fable-5-1 Co-Authored-By: Claude Fable 5.1 --- .../src/zarr_metadata/model/_array.py | 209 ++++++++---------- .../src/zarr_metadata/model/_group.py | 149 +++++-------- .../src/zarr_metadata/model/_keyed.py | 22 ++ 3 files changed, 171 insertions(+), 209 deletions(-) create mode 100644 packages/zarr-metadata/src/zarr_metadata/model/_keyed.py diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_array.py b/packages/zarr-metadata/src/zarr_metadata/model/_array.py index de90712fff..3d7ccbcdd7 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_array.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/_array.py @@ -23,6 +23,7 @@ with_input, ) from zarr_metadata._sentinel import UNSET +from zarr_metadata.model._keyed import Keyed from zarr_metadata.model._validation import ( ArrayMembersV2, ArrayMembersV3, @@ -115,7 +116,7 @@ class ZarrV3ArrayMetadataUpdate(TypedDict, total=False, extra_items=ZarrV3Extens dimension_names: tuple[str | None, ...] | UNSET -class ZarrV3ArrayMetadata: +class ZarrV3ArrayMetadata(Keyed): """A v3 array document, and the scope it was read in. The model is the pair: `to_json` is the document as written, refined @@ -128,14 +129,14 @@ class ZarrV3ArrayMetadata: constructor reads `document` in `context` and raises `MetadataValidationError` with every problem, so no model is invalid. Two models are equal when their documents mean the same in their - scopes, as `array_key` says; the scope itself takes no part. `update` + scopes, as its key says; the scope itself takes no part. `update` reads new members in the model's own scope; `with_context` and `refined_in` read the document in another. A model pickles as its pair, when the definitions its scope holds do: ones whose functions are defined at a module's top level. """ - __slots__ = ("_claims", "_context", "_document", "_key", "_members", "_reading", "_shown") + __slots__ = ("_claims", "_context", "_document", "_members", "_reading", "_shown") zarr_format: Final = 3 node_type: Final = "array" @@ -160,7 +161,7 @@ def _of( reading: ZarrV3ArrayMetadataReading, members: ArrayMembersV3, ) -> ZarrV3ArrayMetadata: - """A model of a document a read found nothing wrong with, holding that reading: no second read.""" + """A model of a document a read found nothing wrong with, holding that reading: no second read. The readers of this package build models through this, the private use pyright reports.""" model = object.__new__(cls) model._adopt(document, context, reading, members) return model @@ -184,7 +185,7 @@ def _adopt( frozen(members.attributes), frozen(members.extra_fields), ) - self._key = array_key(self) + self._key = self._key_of() self._claims = MappingProxyType(claims_of(reading.fields())) # --- the pair --------------------------------------------------------- @@ -217,13 +218,46 @@ def to_key_value( def __repr__(self) -> str: return f"{type(self).__name__}({self._document!r}, context={self._context!r})" - def __eq__(self, other: object) -> bool: - if type(other) is not type(self): - return NotImplemented - return self._key == cast("ZarrV3ArrayMetadata", other)._key + def _key_of(self) -> tuple[object, ...]: + """What `==` and `hash` compare of a v3 array model: what its document means. + + Each field by its `field_key`, the fill value in its canonical spelling + as JSON text when a definition in scope read the data type, and every + other member as it is, the JSON ones as text. + """ + members = self._members + return ( + members.shape, + self._fill_value_key(), + field_key(self.data_type), + field_key(self.chunk_grid), + tuple(field_key(codec) for codec in self.codecs), + field_key(self.chunk_key_encoding), + members.dimension_names, + json_text(members.attributes), + tuple(field_key(transformer) for transformer in self.storage_transformers), + json_text(members.extra_fields), + ) - def __hash__(self) -> int: - return hash(self._key) + def _plain_key( + self, data_type: Read[DataTypeDefinition[Any]] | Unclaimed + ) -> tuple[object, ...]: + """What `refines` compares of a model other than its fields, the fill value spelled as `data_type` -- the more informed side's -- spells it.""" + members = self._members + return ( + members.shape, + json_text(spelled_canonically(data_type, members.fill_value)), + members.dimension_names, + json_text(members.attributes), + json_text(members.extra_fields), + ) + + def _fill_value_key(self) -> str: + """What `==` compares of the fill value: its canonical spelling as JSON text when a definition in scope read the data type, and the fill value as written when none did.""" + fill_value = self._members.fill_value + if isinstance(self.data_type, Read): + return json_text(spelled_canonically(self.data_type, fill_value)) + return json_text(fill_value) def __reduce__(self) -> tuple[type[ZarrV3ArrayMetadata], tuple[object, Context]]: # The pair, read again on load: a model's reading never disagrees @@ -365,7 +399,7 @@ def refines(self, other: ZarrV3ArrayMetadata) -> bool: return False if len(fill_value_problems(self.data_type, other._members.fill_value)) != 0: return False - return _plain_key(self, self.data_type) == _plain_key(other, self.data_type) + return self._plain_key(self.data_type) == other._plain_key(self.data_type) # --- constructors ----------------------------------------------------- @@ -429,28 +463,6 @@ def from_key_value( return cls(load_store_json(mapping, ZARR_V3_ARRAY_METADATA_STORE_KEY), context=context) -def array_key(model: ZarrV3ArrayMetadata) -> tuple[object, ...]: - """What `==` and `hash` compare of a v3 array model: what its document means. - - Each field by its `field_key`, the fill value in its canonical spelling - as JSON text when a definition in scope read the data type, and every - other member as it is, the JSON ones as text. - """ - members = model._members # pyright: ignore[reportPrivateUsage] - return ( - members.shape, - _fill_value_key(model), - field_key(model.data_type), - field_key(model.chunk_grid), - tuple(field_key(codec) for codec in model.codecs), - field_key(model.chunk_key_encoding), - members.dimension_names, - json_text(members.attributes), - tuple(field_key(transformer) for transformer in model.storage_transformers), - json_text(members.extra_fields), - ) - - def located_conflicts( fields: Iterable[tuple[Loc, Resolved[Any]]], conflicts: Sequence[Conflict] ) -> tuple[Conflict, ...]: @@ -465,28 +477,6 @@ def located_conflicts( return tuple(located) -def _plain_key( - model: ZarrV3ArrayMetadata, data_type: Read[DataTypeDefinition[Any]] | Unclaimed -) -> tuple[object, ...]: - """What `refines` compares of a model other than its fields, the fill value spelled as `data_type` -- the more informed side's -- spells it.""" - members = model._members # pyright: ignore[reportPrivateUsage] - return ( - members.shape, - json_text(spelled_canonically(data_type, members.fill_value)), - members.dimension_names, - json_text(members.attributes), - json_text(members.extra_fields), - ) - - -def _fill_value_key(model: ZarrV3ArrayMetadata) -> str: - """What `==` compares of `model`'s fill value: its canonical spelling as JSON text when a definition in scope read the data type, and the fill value as written when none did.""" - fill_value = model._members.fill_value # pyright: ignore[reportPrivateUsage] - if isinstance(model.data_type, Read): - return json_text(spelled_canonically(model.data_type, fill_value)) - return json_text(fill_value) - - def read_array_metadata_v3( value: object, *, context: Context | None = None ) -> ZarrV3ArrayMetadataReading: @@ -550,7 +540,7 @@ class ZarrV2ArrayMetadataUpdate(TypedDict, total=False, extra_items=JSONValue | attributes: Mapping[str, JSONValue] | UNSET -class ZarrV2ArrayMetadata: +class ZarrV2ArrayMetadata(Keyed): """A v2 array document, and the scope it was read in. The pair, as the v3 models are: `to_json` is the merged document -- @@ -566,12 +556,12 @@ class ZarrV2ArrayMetadata: only by reading: the constructor reads `document` in `context` and raises `MetadataValidationError` with every problem, so no model is invalid. Two models are equal when their documents mean the same in - their scopes, as `array_key_v2` says. `update` reads new members in + their scopes, as its key says. `update` reads new members in the model's own scope; `with_context` and `refined_in` read the document in another. A model pickles as its pair. """ - __slots__ = ("_claims", "_context", "_document", "_key", "_members", "_reading", "_shown") + __slots__ = ("_claims", "_context", "_document", "_members", "_reading", "_shown") zarr_format: Final = 2 @@ -598,7 +588,7 @@ def _of( reading: ZarrV2ArrayMetadataReading, members: ArrayMembersV2, ) -> ZarrV2ArrayMetadata: - """A model of a document a read found nothing wrong with, holding that reading: no second read.""" + """A model of a document a read found nothing wrong with, holding that reading: no second read. The readers of this package build models through this, the private use pyright reports.""" model = object.__new__(cls) model._adopt(document, context, reading, members) return model @@ -620,7 +610,7 @@ def _adopt( UNSET if members.attributes is UNSET else frozen(members.attributes), frozen(members.extra_fields), ) - self._key = array_key_v2(self) + self._key = self._key_of() self._claims = MappingProxyType(claims_of(reading.fields())) # --- the pair --------------------------------------------------------- @@ -662,13 +652,49 @@ def to_key_value( def __repr__(self) -> str: return f"{type(self).__name__}({self._document!r}, context={self._context!r})" - def __eq__(self, other: object) -> bool: - if type(other) is not type(self): - return NotImplemented - return self._key == cast("ZarrV2ArrayMetadata", other)._key + def _key_of(self) -> tuple[object, ...]: + """What `==` and `hash` compare of a v2 array model: what its document means. + + Each field by its `field_key`, the fill value in its canonical spelling + as JSON text when a definition in scope read the dtype, and every other + member as it is, the JSON ones as text; `attributes` as `UNSET` when + there is no `.zattrs`. + """ + members = self._members + return ( + members.shape, + members.chunks, + members.order, + members.dimension_separator, + self._fill_value_key(), + field_key(self.dtype), + None if self.compressor is None else field_key(self.compressor), + None if self.filters is None else tuple(field_key(entry) for entry in self.filters), + UNSET if members.attributes is UNSET else json_text(members.attributes), + json_text(members.extra_fields), + ) - def __hash__(self) -> int: - return hash(self._key) + def _plain_key( + self, dtype: Read[ZarrV2DataTypeDefinition[Any]] | Unclaimed + ) -> tuple[object, ...]: + """What `refines` compares of a model other than its fields, the fill value spelled as `dtype` -- the more informed side's -- spells it.""" + members = self._members + return ( + members.shape, + members.chunks, + members.order, + members.dimension_separator, + json_text(spelled_canonically(dtype, members.fill_value)), + UNSET if members.attributes is UNSET else json_text(members.attributes), + json_text(members.extra_fields), + ) + + def _fill_value_key(self) -> str: + """What `==` compares of the fill value: its canonical spelling as JSON text when a definition in scope read the dtype, and the fill value as written when none did.""" + fill_value = self._members.fill_value + if isinstance(self.dtype, Read): + return json_text(spelled_canonically(self.dtype, fill_value)) + return json_text(fill_value) def __reduce__(self) -> tuple[type[ZarrV2ArrayMetadata], tuple[object, Context]]: return type(self), (self._document, self._context) @@ -787,7 +813,7 @@ def refines(self, other: ZarrV2ArrayMetadata) -> bool: return False if len(fill_value_problems(self.dtype, other._members.fill_value)) != 0: return False - return _plain_key_v2(self, self.dtype) == _plain_key_v2(other, self.dtype) + return self._plain_key(self.dtype) == other._plain_key(self.dtype) # --- constructors ----------------------------------------------------- @@ -860,50 +886,3 @@ def from_key_value( zattrs = load_store_json(mapping, ZARR_V2_ATTRIBUTES_STORE_KEY) return cls({**zarray, "attributes": zattrs}, context=context) return cls(zarray, context=context) - - -def array_key_v2(model: ZarrV2ArrayMetadata) -> tuple[object, ...]: - """What `==` and `hash` compare of a v2 array model: what its document means. - - Each field by its `field_key`, the fill value in its canonical spelling - as JSON text when a definition in scope read the dtype, and every other - member as it is, the JSON ones as text; `attributes` as `UNSET` when - there is no `.zattrs`. - """ - members = model._members # pyright: ignore[reportPrivateUsage] - return ( - members.shape, - members.chunks, - members.order, - members.dimension_separator, - _fill_value_key_v2(model), - field_key(model.dtype), - None if model.compressor is None else field_key(model.compressor), - None if model.filters is None else tuple(field_key(entry) for entry in model.filters), - UNSET if members.attributes is UNSET else json_text(members.attributes), - json_text(members.extra_fields), - ) - - -def _plain_key_v2( - model: ZarrV2ArrayMetadata, dtype: Read[ZarrV2DataTypeDefinition[Any]] | Unclaimed -) -> tuple[object, ...]: - """What `refines` compares of a model other than its fields, the fill value spelled as `dtype` -- the more informed side's -- spells it.""" - members = model._members # pyright: ignore[reportPrivateUsage] - return ( - members.shape, - members.chunks, - members.order, - members.dimension_separator, - json_text(spelled_canonically(dtype, members.fill_value)), - UNSET if members.attributes is UNSET else json_text(members.attributes), - json_text(members.extra_fields), - ) - - -def _fill_value_key_v2(model: ZarrV2ArrayMetadata) -> str: - """What `==` compares of `model`'s fill value: its canonical spelling as JSON text when a definition in scope read the dtype, and the fill value as written when none did.""" - fill_value = model._members.fill_value # pyright: ignore[reportPrivateUsage] - if isinstance(model.dtype, Read): - return json_text(spelled_canonically(model.dtype, fill_value)) - return json_text(fill_value) diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_group.py b/packages/zarr-metadata/src/zarr_metadata/model/_group.py index 67ba7d8124..60cfea4f47 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_group.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/_group.py @@ -42,6 +42,7 @@ must_understand_subset, read_array_metadata_v3, ) +from zarr_metadata.model._keyed import Keyed from zarr_metadata.model._validation import ( GROUP_METADATA_REQUIRED_KEYS_V3, GROUP_METADATA_STANDARD_KEYS_V3, @@ -122,7 +123,7 @@ class ZarrV3GroupMetadataUpdate(TypedDict, total=False, extra_items=ZarrV3Extens consolidated_metadata: ZarrV3ConsolidatedMetadataInput | ZarrV3ConsolidatedMetadata | UNSET -class ZarrV3GroupMetadata: +class ZarrV3GroupMetadata(Keyed): """A v3 group document, and the scope it was read in. The model is the pair, as `ZarrV3ArrayMetadata` is: `to_json` is the @@ -141,7 +142,6 @@ class ZarrV3GroupMetadata: "_consolidated", "_context", "_document", - "_key", "_members", "_reading", "_shown", @@ -170,7 +170,7 @@ def _of( reading: ZarrV3GroupMetadataReading, members: GroupMembersV3, ) -> ZarrV3GroupMetadata: - """A model of a document a read found nothing wrong with, holding that reading: no second read.""" + """A model of a document a read found nothing wrong with, holding that reading: no second read. The readers of this package build models through this, the private use pyright reports.""" model = object.__new__(cls) model._adopt(document, context, reading, members) return model @@ -216,7 +216,7 @@ def _adopt( self._reading = dataclasses.replace( reading, consolidated=MappingProxyType(dict(held)), metadata=self ) - self._key = group_key(self) + self._key = self._key_of() self._claims = MappingProxyType(claims_of(reading.fields())) # --- the pair --------------------------------------------------------- @@ -249,13 +249,15 @@ def to_key_value( def __repr__(self) -> str: return f"{type(self).__name__}({self._document!r}, context={self._context!r})" - def __eq__(self, other: object) -> bool: - if type(other) is not type(self): - return NotImplemented - return self._key == cast("ZarrV3GroupMetadata", other)._key - - def __hash__(self) -> int: - return hash(self._key) + def _key_of(self) -> tuple[object, ...]: + """What `==` and `hash` compare of a v3 group model: its attributes and extra fields as JSON text, and what its consolidated metadata holds, by its key.""" + consolidated = self.consolidated_metadata + members = self._members + return ( + json_text(members.attributes), + UNSET if consolidated is UNSET else consolidated._key, + json_text(members.extra_fields), + ) def __reduce__(self) -> tuple[type[ZarrV3GroupMetadata], tuple[object, Context]]: # The pair, read again on load. @@ -377,7 +379,7 @@ def from_key_value( return cls(load_store_json(mapping, ZARR_V3_GROUP_METADATA_STORE_KEY), context=context) -class ZarrV3ConsolidatedMetadata: +class ZarrV3ConsolidatedMetadata(Keyed): """A group's inline `consolidated_metadata` member, and the scope it was read in. Models the reference-implementation convention where consolidated @@ -391,7 +393,7 @@ class ZarrV3ConsolidatedMetadata: path without the leading `/`: the node at `/a/b` at `a/b`. """ - __slots__ = ("_context", "_document", "_key", "_metadata") + __slots__ = ("_context", "_document", "_metadata") kind: Final = "inline" must_understand: Final = False @@ -416,7 +418,7 @@ def _of( context: Context, metadata: dict[str, ZarrV3NodeMetadata], ) -> ZarrV3ConsolidatedMetadata: - """The member of a group a read found nothing wrong with, holding the models that read built.""" + """The member of a group a read found nothing wrong with, holding the models that read built. The readers of this package build models through this, the private use pyright reports.""" model = object.__new__(cls) model._adopt(document, context, metadata) return model @@ -430,7 +432,7 @@ def _adopt( self._document = document self._context = context self._metadata = metadata - self._key = consolidated_key(self) + self._key = self._key_of() # Hidden from the readings, which hold their own models; see `metadata`. @property @@ -450,13 +452,12 @@ def to_json(self) -> ZarrV3ConsolidatedMetadataJSON: def __repr__(self) -> str: return f"{type(self).__name__}({self._document!r}, context={self._context!r})" - def __eq__(self, other: object) -> bool: - if type(other) is not type(self): - return NotImplemented - return self._key == cast("ZarrV3ConsolidatedMetadata", other)._key - - def __hash__(self) -> int: - return hash(self._key) + def _key_of(self) -> tuple[object, ...]: + """What `==` and `hash` compare of consolidated metadata: each document's key, by its path, in path order.""" + return tuple( + (path, node._key) + for path, node in sorted(self.metadata.items(), key=lambda item: item[0]) + ) def __reduce__(self) -> tuple[type[ZarrV3ConsolidatedMetadata], tuple[object, Context]]: return type(self), (self._document, self._context) @@ -776,7 +777,7 @@ def documents_for(value: object) -> object: def _member_documents_for(member: object) -> object: """A `consolidated_metadata` member with each node model it lists replaced by its document; `member` itself when it lists none.""" if isinstance(member, ZarrV3ConsolidatedMetadata): - return member._document # pyright: ignore[reportPrivateUsage] + return member.to_json() if not is_object(member): return member entries = member.get("metadata") @@ -786,7 +787,7 @@ def _member_documents_for(member: object) -> object: changed = False for path, entry in entries.items(): if isinstance(entry, (ZarrV3ArrayMetadata, ZarrV3GroupMetadata)): - replaced[path] = entry._document # pyright: ignore[reportPrivateUsage] + replaced[path] = entry.to_json() changed = True else: # A document listed here may list models of its own. @@ -806,7 +807,7 @@ def _read_node_model( scope. A gain reads the document again there, so a problem a newly claimed definition finds is reported where it sits. """ - document = entry._document # pyright: ignore[reportPrivateUsage] + document = entry.to_json() found = context.disagreements(entry.claims) if len(found.conflicts) != 0: # Read again in the group's scope, so the reading is of that scope, @@ -938,25 +939,6 @@ def _read_consolidated_v3( return readings, members, tuple(problems) -def group_key(model: ZarrV3GroupMetadata) -> tuple[object, ...]: - """What `==` and `hash` compare of a v3 group model: its attributes and extra fields as JSON text, and what its consolidated metadata holds, by `consolidated_key`.""" - consolidated = model.consolidated_metadata - members = model._members # pyright: ignore[reportPrivateUsage] - return ( - json_text(members.attributes), - UNSET if consolidated is UNSET else consolidated._key, # pyright: ignore[reportPrivateUsage] - json_text(members.extra_fields), - ) - - -def consolidated_key(model: ZarrV3ConsolidatedMetadata) -> tuple[object, ...]: - """What `==` and `hash` compare of consolidated metadata: each document's key, by its path, in path order.""" - return tuple( - (path, node._key) # pyright: ignore[reportPrivateUsage] - for path, node in sorted(model.metadata.items(), key=lambda item: item[0]) - ) - - def _key_problems(key: str) -> list[ValidationProblem]: """What keeps `key` from being where consolidated metadata keeps a document, said in one problem at the key. @@ -1198,7 +1180,7 @@ class ZarrV2GroupMetadataUpdate(TypedDict, total=False): attributes: Mapping[str, JSONValue] | UNSET -class ZarrV2GroupMetadata: +class ZarrV2GroupMetadata(Keyed): """A v2 group document, and the scope it was read in. The pair, as the v3 models are: `to_json` is the merged document -- @@ -1212,12 +1194,11 @@ class ZarrV2GroupMetadata: `group_key_v2` says. """ - __slots__ = ("_attributes", "_context", "_document", "_key", "_shown") + __slots__ = ("_attributes", "_context", "_document", "_shown") _attributes: dict[str, JSONValue] | UNSET _context: Context _document: dict[str, JSONValue] - _key: tuple[object, ...] _shown: Mapping[str, JSONValue] | UNSET zarr_format: Final = 2 @@ -1239,7 +1220,7 @@ def _of( context: Context, attributes: dict[str, JSONValue] | UNSET, ) -> ZarrV2GroupMetadata: - """A model of a document a read found nothing wrong with: no second read.""" + """A model of a document a read found nothing wrong with: no second read. The readers of this package build models through this, the private use pyright reports.""" model = object.__new__(cls) model._adopt(document, context, attributes) return model @@ -1254,7 +1235,7 @@ def _adopt( self._context = context self._attributes = attributes self._shown = UNSET if attributes is UNSET else frozen(attributes) - self._key = group_key_v2(self) + self._key = self._key_of() @property def context(self) -> Context: @@ -1293,13 +1274,10 @@ def to_key_value( def __repr__(self) -> str: return f"{type(self).__name__}({self._document!r}, context={self._context!r})" - def __eq__(self, other: object) -> bool: - if type(other) is not type(self): - return NotImplemented - return self._key == cast("ZarrV2GroupMetadata", other)._key - - def __hash__(self) -> int: - return hash(self._key) + def _key_of(self) -> tuple[object, ...]: + """What `==` and `hash` compare of a v2 group model: its attributes as JSON text, or `UNSET` when there is no `.zattrs`.""" + attributes = self._attributes + return (UNSET if attributes is UNSET else json_text(attributes),) def __reduce__(self) -> tuple[type[ZarrV2GroupMetadata], tuple[object, Context]]: return type(self), (self._document, self._context) @@ -1363,17 +1341,11 @@ def from_key_value( return cls(zgroup, context=context) -def group_key_v2(model: ZarrV2GroupMetadata) -> tuple[object, ...]: - """What `==` and `hash` compare of a v2 group model: its attributes as JSON text, or `UNSET` when there is no `.zattrs`.""" - attributes = model._attributes # pyright: ignore[reportPrivateUsage] - return (UNSET if attributes is UNSET else json_text(attributes),) - - ZarrV2NodeMetadata: TypeAlias = "ZarrV2ArrayMetadata | ZarrV2GroupMetadata" """The model of one node a v2 `.zmetadata` document holds: an array, or a group.""" -class ZarrV2ConsolidatedMetadata: +class ZarrV2ConsolidatedMetadata(Keyed): """A v2 `.zmetadata` document, and the scope its nodes were read in. `metadata` holds the flat file-keyed entries (`"path/.zarray"`, @@ -1385,15 +1357,14 @@ class ZarrV2ConsolidatedMetadata: Built only by reading: the constructor raises `MetadataValidationError` with every problem, each located under its entry. Two documents are equal when each node means the same and the other entries are written - alike, as `consolidated_key_v2` says; `refines`, `with_context` and + alike, as its key says; `refines`, `with_context` and `refined_in` go through the nodes. """ - __slots__ = ("_context", "_document", "_key", "_nodes", "_shown") + __slots__ = ("_context", "_document", "_nodes", "_shown") _context: Context _document: dict[str, JSONValue] - _key: tuple[object, ...] _nodes: dict[str, ZarrV2NodeMetadata] _shown: Mapping[str, JSONValue] @@ -1410,7 +1381,7 @@ def __init__(self, document: object, context: Context | None = None) -> None: def _of( cls, document: dict[str, JSONValue], context: Context, nodes: dict[str, ZarrV2NodeMetadata] ) -> ZarrV2ConsolidatedMetadata: - """A model of a document a read found nothing wrong with, holding the nodes that read built.""" + """A model of a document a read found nothing wrong with, holding the nodes that read built. The readers of this package build models through this, the private use pyright reports.""" model = object.__new__(cls) model._adopt(document, context, nodes) return model @@ -1422,7 +1393,7 @@ def _adopt( self._context = context self._nodes = nodes self._shown = frozen(object_at(document, "metadata")) - self._key = consolidated_key_v2(self) + self._key = self._key_of() @property def context(self) -> Context: @@ -1454,13 +1425,22 @@ def to_key_value( def __repr__(self) -> str: return f"{type(self).__name__}({self._document!r}, context={self._context!r})" - def __eq__(self, other: object) -> bool: - if type(other) is not type(self): - return NotImplemented - return self._key == cast("ZarrV2ConsolidatedMetadata", other)._key + def _key_of(self) -> tuple[object, ...]: + """What `==` and `hash` compare of v2 consolidated metadata: each node by its path and its own key, and every other entry as JSON text.""" + nodes = self._nodes + return ( + tuple(sorted((path, node._key) for path, node in nodes.items())), + self._other_entries_text(), + ) - def __hash__(self) -> int: - return hash(self._key) + def _other_entries_text(self) -> str: + """The entries no node is read from -- an orphan `.zattrs`, any other key -- as JSON text: what `==` compares of them.""" + entries = object_at(self._document, "metadata") + consumed: set[str] = set() + for path, names in _entries_by_path(entries)[0].items(): + if path in self._nodes: + consumed.update(names.values()) + return json_text({key: value for key, value in entries.items() if key not in consumed}) def __reduce__(self) -> tuple[type[ZarrV2ConsolidatedMetadata], tuple[object, Context]]: return type(self), (self._document, self._context) @@ -1502,7 +1482,7 @@ def refines(self, other: ZarrV2ConsolidatedMetadata) -> bool: return False if self._nodes.keys() != other._nodes.keys(): return False - if _other_entries_text(self) != _other_entries_text(other): + if self._other_entries_text() != other._other_entries_text(): return False return all(_v2_node_refines(self._nodes[path], other._nodes[path]) for path in self._nodes) @@ -1567,25 +1547,6 @@ def _entries_by_path( return by_path, problems -def _other_entries_text(model: ZarrV2ConsolidatedMetadata) -> str: - """The entries no node is read from -- an orphan `.zattrs`, any other key -- as JSON text: what `==` compares of them.""" - entries = object_at(model._document, "metadata") # pyright: ignore[reportPrivateUsage] - consumed: set[str] = set() - for path, names in _entries_by_path(entries)[0].items(): - if path in model._nodes: # pyright: ignore[reportPrivateUsage] - consumed.update(names.values()) - return json_text({key: value for key, value in entries.items() if key not in consumed}) - - -def consolidated_key_v2(model: ZarrV2ConsolidatedMetadata) -> tuple[object, ...]: - """What `==` and `hash` compare of v2 consolidated metadata: each node by its path and its own key, and every other entry as JSON text.""" - nodes = model._nodes # pyright: ignore[reportPrivateUsage] - return ( - tuple(sorted((path, node._key) for path, node in nodes.items())), # pyright: ignore[reportPrivateUsage] - _other_entries_text(model), - ) - - def _read_consolidated_v2( data: object, context: Context ) -> tuple[dict[str, JSONValue], dict[str, ZarrV2NodeMetadata], tuple[ValidationProblem, ...]]: diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_keyed.py b/packages/zarr-metadata/src/zarr_metadata/model/_keyed.py new file mode 100644 index 0000000000..f131c7748f --- /dev/null +++ b/packages/zarr-metadata/src/zarr_metadata/model/_keyed.py @@ -0,0 +1,22 @@ +"""What every model compares and hashes by: a key its constructor computes once of everything the model shows.""" + + +class Keyed: + """A model whose `==` and `hash` compare a key computed once, when it is built, of what its document means. + + Each model computes its own key in `_key_of` and stores it; two models + are equal when they are of one class and their keys are, so a model of + another class, or anything else, is never equal to one. + """ + + __slots__ = ("_key",) + + _key: tuple[object, ...] + + def __eq__(self, other: object) -> bool: + if type(other) is not type(self): + return NotImplemented + return self._key == other._key + + def __hash__(self) -> int: + return hash(self._key) From 3b69760decc72d6676b6ba59817e64ab74a2ea75 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Thu, 8 Oct 2026 15:46:27 +0200 Subject: [PATCH 80/94] refactor(zarr-metadata): check what a read established instead of casting to it The model's field properties go through held, which raises for a field a read refused or never read; the canonical checks and the nested-field walkers use the type guards; reading_of takes a Protocol. Casts in src go from 128 to 100. Assisted-by: ClaudeCode:claude-fable-5-1 Co-Authored-By: Claude Fable 5.1 --- .../src/zarr_metadata/model/_array.py | 39 +++++++------- .../src/zarr_metadata/model/_json_schema.py | 8 +-- .../src/zarr_metadata/model/_validation.py | 51 +++++++++++-------- .../src/zarr_metadata/v3/_definition.py | 28 ++++++---- .../src/zarr_metadata/v3/_pipeline.py | 8 +-- 5 files changed, 74 insertions(+), 60 deletions(-) diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_array.py b/packages/zarr-metadata/src/zarr_metadata/model/_array.py index 3d7ccbcdd7..2fbb1b9b92 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_array.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/_array.py @@ -56,6 +56,7 @@ Unclaimed, field_key, fill_value_problems, + held, spelled_canonically, ) from zarr_metadata.v3._registry import CORE_AND_EXTENSIONS, Context @@ -306,37 +307,29 @@ def must_understand_fields(self) -> dict[str, ZarrV3ExtensionField]: @property def data_type(self) -> Read[DataTypeDefinition[Any]] | Unclaimed: """The data type, as the scope read it.""" - return cast("Read[DataTypeDefinition[Any]] | Unclaimed", self._reading.data_type) + return held(self._reading.data_type) @property def chunk_grid(self) -> Read[ChunkGridDefinition[Any]] | Unclaimed: """The chunk grid, as the scope read it.""" - return cast("Read[ChunkGridDefinition[Any]] | Unclaimed", self._reading.chunk_grid) + return held(self._reading.chunk_grid) @property def chunk_key_encoding(self) -> Read[ChunkKeyEncodingDefinition[Any]] | Unclaimed: """The chunk key encoding, as the scope read it.""" - return cast( - "Read[ChunkKeyEncodingDefinition[Any]] | Unclaimed", self._reading.chunk_key_encoding - ) + return held(self._reading.chunk_key_encoding) @property def codecs(self) -> tuple[Read[CodecDefinition[Any]] | Unclaimed, ...]: """The codecs, each as the scope read it, in pipeline order.""" - return tuple( - cast("Read[CodecDefinition[Any]] | Unclaimed", stage.codec) - for stage in self._reading.pipeline - ) + return tuple(held(stage.codec) for stage in self._reading.pipeline) @property def storage_transformers( self, ) -> tuple[Read[StorageTransformerDefinition[Any]] | Unclaimed, ...]: """The storage transformers, each as the scope read it.""" - return cast( - "tuple[Read[StorageTransformerDefinition[Any]] | Unclaimed, ...]", - self._reading.storage_transformers, - ) + return tuple(held(entry) for entry in self._reading.storage_transformers) # --- changing --------------------------------------------------------- @@ -425,7 +418,7 @@ def create_default( """ # The grid derives from a shape the read takes; one it refuses is # reported by the read, and derives nothing. - lengths, _ = dimension_lengths(cast("Mapping[object, object]", overrides), "shape") + lengths, _ = dimension_lengths(overrides, "shape") document: dict[str, object] = { "zarr_format": 3, "node_type": "array", @@ -739,20 +732,24 @@ def extra_fields(self) -> Mapping[str, JSONValue]: @property def dtype(self) -> Read[ZarrV2DataTypeDefinition[Any]] | Unclaimed: """The dtype as the scope read it: by its family's definition, or unclaimed.""" - return cast("Read[ZarrV2DataTypeDefinition[Any]] | Unclaimed", self._reading.dtype) + return held(self._reading.dtype) @property def compressor(self) -> Read[ZarrV2CodecDefinition[Any]] | Unclaimed | None: """The compressor as the scope read it; None when written as `null`.""" - return cast("Read[ZarrV2CodecDefinition[Any]] | Unclaimed | None", self._reading.compressor) + compressor = self._reading.compressor + return None if compressor is None else held(compressor) @property def filters(self) -> tuple[Read[ZarrV2CodecDefinition[Any]] | Unclaimed, ...] | None: """The filters, each as the scope read it; None when written as `null`.""" - return cast( - "tuple[Read[ZarrV2CodecDefinition[Any]] | Unclaimed, ...] | None", - self._reading.filters, - ) + filters = self._reading.filters + if filters is None: + return None + if filters is UNSET: + msg = "expected filters a read found nothing wrong with, got UNSET" + raise TypeError(msg) + return tuple(held(entry) for entry in filters) # --- changing --------------------------------------------------------- @@ -844,7 +841,7 @@ def create_default( } given: dict[str, object] = dict(overrides) if "shape" in given and "chunks" not in given: - lengths, _ = dimension_lengths(cast("Mapping[object, object]", given), "shape") + lengths, _ = dimension_lengths(given, "shape") if lengths is not None: given["chunks"] = lengths if "dtype" in given and "fill_value" not in given: diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_json_schema.py b/packages/zarr-metadata/src/zarr_metadata/model/_json_schema.py index 5dd53e0f3b..936ab44c87 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_json_schema.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/_json_schema.py @@ -2,7 +2,7 @@ from __future__ import annotations -from typing import TYPE_CHECKING, Any, cast +from typing import TYPE_CHECKING, cast from zarr_metadata._typed_json import Schemas from zarr_metadata.v3._definition import DataTypeDefinition, field_schemas, written_name @@ -72,8 +72,10 @@ def _array(context: Context, schemas: Schemas) -> JSONSchema: """An array document: its TypedDict, and its fill value held to the data type it names, for each data type in scope.""" schema = schemas.object_of(ZarrV3ArrayMetadataJSON) held: list[JSONValue] = [] - for definition in context.tables.get(DataTypeDefinition, {}).values(): - fill_value = schemas.of(cast("DataTypeDefinition[Any]", definition).fill_value) + for definition in context.definitions(): + if not isinstance(definition, DataTypeDefinition): + continue + fill_value = schemas.of(definition.fill_value) if len(fill_value) == 0: continue # a data type that says nothing of its fill value takes any JSON name = written_name(definition) diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_validation.py b/packages/zarr-metadata/src/zarr_metadata/model/_validation.py index 9e66137e8b..833aca3bbd 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_validation.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/_validation.py @@ -28,7 +28,17 @@ import json from collections.abc import Mapping, Sequence from dataclasses import dataclass -from typing import TYPE_CHECKING, Any, Final, TypeGuard, TypeVar, cast, get_args, get_origin +from typing import ( + TYPE_CHECKING, + Any, + Final, + Protocol, + TypeGuard, + TypeVar, + cast, + get_args, + get_origin, +) from zarr_metadata._json import ( MetadataValidationError, @@ -228,7 +238,7 @@ def _is_int_sequence(value: object) -> TypeGuard[Sequence[int]]: def dimension_lengths( - doc: Mapping[object, object], key: str + doc: Mapping[object, object] | Mapping[str, object], key: str ) -> tuple[tuple[int, ...] | None, tuple[ValidationProblem, ...]]: """The dimension lengths `doc` holds at `key` (`shape`, `chunks`), and every problem with them. @@ -269,32 +279,27 @@ def _is_canonical_metadata_field_v3(value: object) -> bool: def _is_canonical_array_metadata_v3(value: object) -> bool: """Whether a validated v3 array document matches `ZarrV3ArrayMetadataJSON` at runtime.""" - if not isinstance(value, dict): + if not is_object(value) or not isinstance(value, dict): return False - doc = cast("dict[str, object]", value) - if not isinstance(doc["shape"], tuple) or not isinstance(doc["codecs"], tuple): + doc = value + codecs = doc["codecs"] + if not isinstance(doc["shape"], tuple) or not is_tuple(codecs): return False - if "storage_transformers" in doc and not isinstance(doc["storage_transformers"], tuple): + transformers = doc.get("storage_transformers", ()) + if not is_tuple(transformers): return False if "dimension_names" in doc and not isinstance(doc["dimension_names"], tuple): return False if not all(_is_canonical_metadata_field_v3(doc[key]) for key, _ in _EXTENSION_POINTS_V3): return False - if not all( - _is_canonical_metadata_field_v3(item) for item in cast("tuple[object, ...]", doc["codecs"]) - ): - return False - return "storage_transformers" not in doc or all( - _is_canonical_metadata_field_v3(item) - for item in cast("tuple[object, ...]", doc["storage_transformers"]) - ) + return all(_is_canonical_metadata_field_v3(item) for item in (*codecs, *transformers)) def _is_canonical_array_metadata_v2(value: object) -> bool: """Whether a validated v2 array document matches `ZarrV2ArrayMetadataJSON` at runtime.""" - if not isinstance(value, dict): + if not is_object(value) or not isinstance(value, dict): return False - doc = cast("dict[str, object]", value) + doc = value if not isinstance(doc["shape"], tuple) or not isinstance(doc["chunks"], tuple): return False if not _is_canonical_dtype_v2(doc["dtype"]): @@ -304,8 +309,7 @@ def _is_canonical_array_metadata_v2(value: object) -> bool: return False filters = doc["filters"] return filters is None or ( - isinstance(filters, tuple) - and all(isinstance(item, dict) for item in cast("tuple[object, ...]", filters)) + is_tuple(filters) and all(isinstance(item, dict) for item in filters) ) @@ -437,9 +441,16 @@ def fields(self) -> Iterator[tuple[Loc, Resolved[Any]]]: yield from fields_of(transformer, ("storage_transformers", index)) -def reading_of(model: object) -> object: +class _HoldsReading(Protocol): + """A model: what holds the reading it was built from.""" + + @property + def reading(self) -> object: ... + + +def reading_of(model: _HoldsReading) -> object: """The reading `model`, a model, holds: what a pickled reading that held a model is built again as.""" - return cast("Any", model).reading + return model.reading @dataclass(frozen=True, slots=True) diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py b/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py index c2a0ebd361..4d0203df08 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py @@ -971,8 +971,8 @@ def _collect(value: object, nested: list[_NestedField]) -> None: elif is_tuple(value): for entry in value: _collect(entry, nested) - elif isinstance(value, dict): - for entry in cast("dict[str, object]", value).values(): + elif is_object(value): + for entry in value.values(): _collect(entry, nested) @@ -995,9 +995,8 @@ def _put_back(value: object, put: Callable[[_NestedField], JSONValue]) -> object return put(value) if is_tuple(value): return tuple(_put_back(entry, put) for entry in value) - if isinstance(value, dict): - entries = cast("dict[str, object]", value) - return {key: _put_back(entry, put) for key, entry in entries.items()} + if is_object(value): + return {key: _put_back(entry, put) for key, entry in value.items()} return value @@ -1413,6 +1412,14 @@ def is_field(value: object) -> TypeGuard[Resolved[Any]]: return isinstance(value, _FIELDS) +def held(field: Resolved[D] | UNSET) -> Read[D] | Unclaimed: + """`field`, one a read found nothing wrong with: `Read` or `Unclaimed`; `TypeError` for one refused or never read, which such a read rules out.""" + if isinstance(field, (Read, Unclaimed)): + return field + msg = f"expected a field a read found nothing wrong with, got {field!r}" + raise TypeError(msg) + + def _is_data_type_field(value: object) -> bool: """Whether `value` is a field a scope read as a data type.""" return is_field(value) and value.read_as is DataTypeDefinition @@ -1910,11 +1917,12 @@ def _replaced(value: JSONValue, path: Loc, new: JSONValue) -> JSONValue: if len(path) == 0: return new step, rest = path[0], path[1:] - if isinstance(step, str): - members = cast("Mapping[str, JSONValue]", value) - return {**members, step: _replaced(members[step], rest, new)} - entries = cast("tuple[JSONValue, ...]", value) - return (*entries[:step], _replaced(entries[step], rest, new), *entries[step + 1 :]) + if isinstance(step, str) and isinstance(value, Mapping): + return {**value, step: _replaced(value[step], rest, new)} + if isinstance(step, int) and isinstance(value, tuple): + return (*value[:step], _replaced(value[step], rest, new), *value[step + 1 :]) + msg = f"{path!r} does not lead into {value!r}" + raise TypeError(msg) __all__ = [ diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/_pipeline.py b/packages/zarr-metadata/src/zarr_metadata/v3/_pipeline.py index 0097f7e67e..2417f0f557 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/_pipeline.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/_pipeline.py @@ -29,7 +29,7 @@ from dataclasses import dataclass from typing import TYPE_CHECKING, Any, Final, TypeGuard, cast -from zarr_metadata._json import ValidationProblem, is_object, with_input +from zarr_metadata._json import ValidationProblem, is_object, is_tuple, with_input from zarr_metadata.v3._definition import ( Chunk, CodecDefinition, @@ -240,11 +240,7 @@ def _held( `TypeError`: a fault in the definition that names it. """ entries: object = configuration.get(member) - places = ( - [(member, index) for index in range(len(cast("tuple[object, ...]", entries)))] - if isinstance(entries, tuple) - else None - ) + places = [(member, index) for index in range(len(entries))] if is_tuple(entries) else None if places is None or not all( place in nested and nested[place].read_as is CodecDefinition for place in places ): From 3c20c0f141da9da4f4ee6a83a93596217523ba15 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Thu, 8 Oct 2026 15:58:12 +0200 Subject: [PATCH 81/94] docs(zarr-metadata): consolidate the stack's changelog fragments under upstream PR 4490 Assisted-by: ClaudeCode:claude-fable-5-1 Co-Authored-By: Claude Fable 5.1 --- packages/zarr-metadata/changes/371.bugfix.md | 9 -------- packages/zarr-metadata/changes/371.doc.md | 7 ------- packages/zarr-metadata/changes/371.misc.md | 3 --- packages/zarr-metadata/changes/373.bugfix.md | 11 ---------- .../zarr-metadata/changes/373.feature.1.md | 15 ------------- .../zarr-metadata/changes/373.feature.2.md | 10 --------- packages/zarr-metadata/changes/373.feature.md | 14 ------------- packages/zarr-metadata/changes/373.removal.md | 10 --------- packages/zarr-metadata/changes/374.feature.md | 16 -------------- packages/zarr-metadata/changes/376.feature.md | 18 ---------------- packages/zarr-metadata/changes/376.removal.md | 9 -------- packages/zarr-metadata/changes/377.feature.md | 11 ---------- packages/zarr-metadata/changes/378.bugfix.md | 9 -------- packages/zarr-metadata/changes/378.feature.md | 21 ------------------- packages/zarr-metadata/changes/379.feature.md | 16 -------------- .../zarr-metadata/changes/381.bugfix.1.md | 6 ------ .../zarr-metadata/changes/381.bugfix.2.md | 9 -------- .../zarr-metadata/changes/381.bugfix.3.md | 14 ------------- packages/zarr-metadata/changes/381.bugfix.md | 9 -------- .../zarr-metadata/changes/381.feature.1.md | 10 --------- packages/zarr-metadata/changes/381.feature.md | 5 ----- .../zarr-metadata/changes/382.bugfix.1.md | 1 - .../zarr-metadata/changes/382.bugfix.2.md | 1 - .../zarr-metadata/changes/382.bugfix.3.md | 1 - .../zarr-metadata/changes/382.bugfix.4.md | 1 - packages/zarr-metadata/changes/382.bugfix.md | 1 - .../zarr-metadata/changes/382.feature.1.md | 1 - packages/zarr-metadata/changes/382.feature.md | 1 - packages/zarr-metadata/changes/391.feature.md | 1 - packages/zarr-metadata/changes/392.feature.md | 1 - packages/zarr-metadata/changes/393.feature.md | 1 - packages/zarr-metadata/changes/393.removal.md | 1 - .../zarr-metadata/changes/395.feature.1.md | 1 - packages/zarr-metadata/changes/395.feature.md | 1 - packages/zarr-metadata/changes/396.feature.md | 1 - packages/zarr-metadata/changes/397.feature.md | 1 - packages/zarr-metadata/changes/397.removal.md | 1 - packages/zarr-metadata/changes/398.removal.md | 1 - .../zarr-metadata/changes/4490.bugfix.1.md | 11 ++++++++++ .../zarr-metadata/changes/4490.bugfix.2.md | 7 +++++++ .../zarr-metadata/changes/4490.bugfix.3.md | 9 ++++++++ packages/zarr-metadata/changes/4490.bugfix.md | 7 +++++++ packages/zarr-metadata/changes/4490.doc.md | 7 +++++++ .../zarr-metadata/changes/4490.feature.1.md | 7 +++++++ .../zarr-metadata/changes/4490.feature.2.md | 13 ++++++++++++ .../zarr-metadata/changes/4490.feature.3.md | 9 ++++++++ .../zarr-metadata/changes/4490.feature.4.md | 12 +++++++++++ .../zarr-metadata/changes/4490.feature.5.md | 9 ++++++++ .../zarr-metadata/changes/4490.feature.6.md | 10 +++++++++ .../zarr-metadata/changes/4490.feature.7.md | 20 ++++++++++++++++++ .../zarr-metadata/changes/4490.feature.8.md | 10 +++++++++ .../zarr-metadata/changes/4490.feature.md | 15 +++++++++++++ .../zarr-metadata/changes/4490.removal.1.md | 8 +++++++ .../zarr-metadata/changes/4490.removal.2.md | 9 ++++++++ .../zarr-metadata/changes/4490.removal.md | 13 ++++++++++++ 55 files changed, 176 insertions(+), 249 deletions(-) delete mode 100644 packages/zarr-metadata/changes/371.bugfix.md delete mode 100644 packages/zarr-metadata/changes/371.doc.md delete mode 100644 packages/zarr-metadata/changes/371.misc.md delete mode 100644 packages/zarr-metadata/changes/373.bugfix.md delete mode 100644 packages/zarr-metadata/changes/373.feature.1.md delete mode 100644 packages/zarr-metadata/changes/373.feature.2.md delete mode 100644 packages/zarr-metadata/changes/373.feature.md delete mode 100644 packages/zarr-metadata/changes/373.removal.md delete mode 100644 packages/zarr-metadata/changes/374.feature.md delete mode 100644 packages/zarr-metadata/changes/376.feature.md delete mode 100644 packages/zarr-metadata/changes/376.removal.md delete mode 100644 packages/zarr-metadata/changes/377.feature.md delete mode 100644 packages/zarr-metadata/changes/378.bugfix.md delete mode 100644 packages/zarr-metadata/changes/378.feature.md delete mode 100644 packages/zarr-metadata/changes/379.feature.md delete mode 100644 packages/zarr-metadata/changes/381.bugfix.1.md delete mode 100644 packages/zarr-metadata/changes/381.bugfix.2.md delete mode 100644 packages/zarr-metadata/changes/381.bugfix.3.md delete mode 100644 packages/zarr-metadata/changes/381.bugfix.md delete mode 100644 packages/zarr-metadata/changes/381.feature.1.md delete mode 100644 packages/zarr-metadata/changes/381.feature.md delete mode 100644 packages/zarr-metadata/changes/382.bugfix.1.md delete mode 100644 packages/zarr-metadata/changes/382.bugfix.2.md delete mode 100644 packages/zarr-metadata/changes/382.bugfix.3.md delete mode 100644 packages/zarr-metadata/changes/382.bugfix.4.md delete mode 100644 packages/zarr-metadata/changes/382.bugfix.md delete mode 100644 packages/zarr-metadata/changes/382.feature.1.md delete mode 100644 packages/zarr-metadata/changes/382.feature.md delete mode 100644 packages/zarr-metadata/changes/391.feature.md delete mode 100644 packages/zarr-metadata/changes/392.feature.md delete mode 100644 packages/zarr-metadata/changes/393.feature.md delete mode 100644 packages/zarr-metadata/changes/393.removal.md delete mode 100644 packages/zarr-metadata/changes/395.feature.1.md delete mode 100644 packages/zarr-metadata/changes/395.feature.md delete mode 100644 packages/zarr-metadata/changes/396.feature.md delete mode 100644 packages/zarr-metadata/changes/397.feature.md delete mode 100644 packages/zarr-metadata/changes/397.removal.md delete mode 100644 packages/zarr-metadata/changes/398.removal.md create mode 100644 packages/zarr-metadata/changes/4490.bugfix.1.md create mode 100644 packages/zarr-metadata/changes/4490.bugfix.2.md create mode 100644 packages/zarr-metadata/changes/4490.bugfix.3.md create mode 100644 packages/zarr-metadata/changes/4490.bugfix.md create mode 100644 packages/zarr-metadata/changes/4490.doc.md create mode 100644 packages/zarr-metadata/changes/4490.feature.1.md create mode 100644 packages/zarr-metadata/changes/4490.feature.2.md create mode 100644 packages/zarr-metadata/changes/4490.feature.3.md create mode 100644 packages/zarr-metadata/changes/4490.feature.4.md create mode 100644 packages/zarr-metadata/changes/4490.feature.5.md create mode 100644 packages/zarr-metadata/changes/4490.feature.6.md create mode 100644 packages/zarr-metadata/changes/4490.feature.7.md create mode 100644 packages/zarr-metadata/changes/4490.feature.8.md create mode 100644 packages/zarr-metadata/changes/4490.feature.md create mode 100644 packages/zarr-metadata/changes/4490.removal.1.md create mode 100644 packages/zarr-metadata/changes/4490.removal.2.md create mode 100644 packages/zarr-metadata/changes/4490.removal.md diff --git a/packages/zarr-metadata/changes/371.bugfix.md b/packages/zarr-metadata/changes/371.bugfix.md deleted file mode 100644 index b1f1381086..0000000000 --- a/packages/zarr-metadata/changes/371.bugfix.md +++ /dev/null @@ -1,9 +0,0 @@ -A problem's message shows what the document holds as JSON -- `null`, -`[1, 2]`, `"C"` -- where it showed the Python value, `None`, `(1, 2)`, -`'C'`, and names what was expected in JSON's words: an array, not a -sequence; an object, not a mapping; a lone value alone, not a -one-element tuple; a union of objects once, not "an object or an -object". A value of another JSON type than each of a closed set's -- -`"zarr_format": "3"`, or `"version": 0.5` where a TypedDict says -`"0.5"` -- is `invalid_type`, where it was `invalid_value`, so the kind -tells a wrong type from a wrong value. diff --git a/packages/zarr-metadata/changes/371.doc.md b/packages/zarr-metadata/changes/371.doc.md deleted file mode 100644 index 94492433c2..0000000000 --- a/packages/zarr-metadata/changes/371.doc.md +++ /dev/null @@ -1,7 +0,0 @@ -The validation boundary says what the package decides where the specs -leave it open or settle it two ways: `attributes` may hold `NaN`, -`Infinity` and `-Infinity`, which `to_key_value` writes as bare tokens; -`must_understand: false` is refused at every extension point, codecs -too. The models' `from_json`, `to_json`, `from_key_value` and -`to_key_value` have docstrings, and no public docstring names a private -function. diff --git a/packages/zarr-metadata/changes/371.misc.md b/packages/zarr-metadata/changes/371.misc.md deleted file mode 100644 index a10dc222ac..0000000000 --- a/packages/zarr-metadata/changes/371.misc.md +++ /dev/null @@ -1,3 +0,0 @@ -A scope's repr says how many definitions it holds, `Context(<33 -definitions>)`, so `help` of a validator no longer prints every -definition in its default scope. diff --git a/packages/zarr-metadata/changes/373.bugfix.md b/packages/zarr-metadata/changes/373.bugfix.md deleted file mode 100644 index ed59a34206..0000000000 --- a/packages/zarr-metadata/changes/373.bugfix.md +++ /dev/null @@ -1,11 +0,0 @@ -`to_json`, `to_key_value` and `canonicalize` write every extension point -but a data type as an object -- `{"name": "crc32c"}` -- where they wrote -the bare name `"crc32c"` for a field with nothing to configure. A Zarr -v3.0 reader takes no short-hand name in `codecs`, and no zarr-python -release reads one there or in `chunk_key_encoding`, so a document the -package wrote could not be opened by them. A data type with nothing to -configure is still written by its bare name, as core data types are. A -field's own `to_json` writes it the same way, and each field its -configuration holds -- a shard's inner codecs -- is written so too, -however the document spelled it; `canonicalize` spells a field nothing -in scope claims by its kind as well. diff --git a/packages/zarr-metadata/changes/373.feature.1.md b/packages/zarr-metadata/changes/373.feature.1.md deleted file mode 100644 index 1bedd89a36..0000000000 --- a/packages/zarr-metadata/changes/373.feature.1.md +++ /dev/null @@ -1,15 +0,0 @@ -A field a scope reads is one of three frozen dataclasses, built with -keywords: `Read`, by the definition in scope that claims its name, -holding that definition and the configuration it allowed; `Unclaimed`, -when nothing in scope claims it, which is left unjudged; or `Refused`, -holding the definition that claims its name and refused it, if any. -`Resolved` is their union, for `match`. Each answers the kind it was -read as, `read_as`; the `name` it was written with -- `"r16"`, whose -definition is filed under `r*`; its `definition`; and the fields it -holds as the scope read them, `nested`, which a refused field keeps too. -Two fields are equal when they read the same, however each was spelled: -a configuration holds each field in it as a document writes it, and that -is what rules are handed. A codec that is not JSON takes its kind from -the definition that claims its name, as one refused for a configuration -that is JSON does, so the pipeline's order is judged with it. A scope -pickles, and the core definitions compare equal once pickled. diff --git a/packages/zarr-metadata/changes/373.feature.2.md b/packages/zarr-metadata/changes/373.feature.2.md deleted file mode 100644 index 29983eef86..0000000000 --- a/packages/zarr-metadata/changes/373.feature.2.md +++ /dev/null @@ -1,10 +0,0 @@ -A v3 model holds each extension point as a scope read it, and holds no -scope, as a pydantic model holds no validation context: -`update(context=..., **members)` reads only the members it is given, as -JSON, in the scope it is given, keeps each field the model holds as it -was read, and leaves out a member given as `UNSET`; a group keeps the -documents its consolidated metadata holds unless it is given others. The -pydantic field types read in the scope their validation context holds: -the context itself, or its `zarr_metadata_context` item. -`create_default`'s codec is `bytes` with a little `endian`, so the -default of any data type of fixed size is valid. diff --git a/packages/zarr-metadata/changes/373.feature.md b/packages/zarr-metadata/changes/373.feature.md deleted file mode 100644 index 081c563e57..0000000000 --- a/packages/zarr-metadata/changes/373.feature.md +++ /dev/null @@ -1,14 +0,0 @@ -`read_array_metadata_v3` reads a v3 array document once and returns -everything the read found, as a `ZarrV3ArrayMetadataReading`: each -extension point as the scope read it, the chunks the codecs are handed, -the codecs as a pipeline, each with the chunk it is handed, every -problem, and, when there is none, the document's model, built by the -same read. Its `fields()` gives each field with where it sits in the -document, and after it the fields it holds -- a shard's codecs, a -struct's field types -- as `fields_of` gives them for any field -`resolve` read, so a policy over a document's fields, the core spec's -alone, say, reads nothing twice. `read_group_metadata_v3` reads a v3 -group the same way, and each document its consolidated metadata holds -once, holding the model of each that has no problem. `from_json` is the -model, or the problems raised; `validate_*` are the problems, and build -no model. diff --git a/packages/zarr-metadata/changes/373.removal.md b/packages/zarr-metadata/changes/373.removal.md deleted file mode 100644 index 5ecdd43f76..0000000000 --- a/packages/zarr-metadata/changes/373.removal.md +++ /dev/null @@ -1,10 +0,0 @@ -`ZarrV3NamedConfig`, `ZarrV3MetadataField` -- the model's and the -pydantic one -- `ZarrV3ArrayMetadataPartial` and -`ZarrV3GroupMetadataPartial` are gone: a model holds each extension point -as a scope read it, a `Read` or an `Unclaimed`, and `update` takes -members of the document as JSON, as `ZarrV3ArrayMetadataUpdate` and -`ZarrV3GroupMetadataUpdate` type them, and as `create_default` takes -them. `Resolved` is no longer a class with a `resolution`: match on its -variant, each built with keywords. `update` takes the scope it reads in, -`context`, which has no default. `to_key_value` no longer takes a -`context`. diff --git a/packages/zarr-metadata/changes/374.feature.md b/packages/zarr-metadata/changes/374.feature.md deleted file mode 100644 index 2ba397eb40..0000000000 --- a/packages/zarr-metadata/changes/374.feature.md +++ /dev/null @@ -1,16 +0,0 @@ -`read_node_metadata_v3` reads a v3 `zarr.json` of either kind as the node -its `node_type` says it is -- the tag of a union, as pydantic's -discriminator and zod's discriminated union read one: an array as -`read_array_metadata_v3` reads it, a group as `read_group_metadata_v3` -does, and a document whose `node_type` says neither, or that is not an -object, as `ZarrV3UnknownNodeReading`, with the problem, nothing else of -it read. `ZarrV3NodeMetadataReading` is the three, and -`validate_node_metadata_v3` gives the problems. -`node_metadata_from_json_v3` and `node_metadata_from_key_value_v3` build -the model of either kind, `ZarrV3NodeMetadata`, as the models' own -`from_json` and `from_key_value` build one, so a node is read from a -store's bytes without its `node_type` read first. A group's consolidated -metadata reads each document it holds so: one of no node type is in -`reading.consolidated`, where it was missing, an entry with no -`node_type` has a `missing_key` problem, and an entry that is not an -object has an `invalid_type` problem at the entry. diff --git a/packages/zarr-metadata/changes/376.feature.md b/packages/zarr-metadata/changes/376.feature.md deleted file mode 100644 index ed04787613..0000000000 --- a/packages/zarr-metadata/changes/376.feature.md +++ /dev/null @@ -1,18 +0,0 @@ -A number's type carries its bounds, as pydantic reads them: -annotated-types' `Gt`, `Ge`, `Lt`, `Le` and `Interval`, one bound from -each side, at any depth, so a gzip `level` is -`Annotated[int, Interval(ge=0, le=9)]`. `check`, a definition's `judge` -and every reader hold a value to them: one problem, `invalid_value`, -saying what the type admits. `Annotated` may also carry a note, a string -or a `Doc`; any other metadata is a `TypeError` when the type is -compiled, so the checker and pydantic never read one type two ways. The -package's own bounds are declared so -- the gzip and zstd levels, -blosc's `clevel` and `blocksize`, the regular and rectilinear grids, the -shard's inner chunk shape, the integer fill values, byte values, and the -numpy time types' `scale_factor` and ticks -- where they were rules. A -`ValidationProblem` carries its message's data, as pydantic's errors and -zod's issues do: `input`, the JSON the value handed in holds at its -`loc`, `UNSET` where nothing is, and `ctx`, what was expected -- the -bounds by pydantic's names, `{"ge": 0, "le": 9}`, or the values of a -closed set, `{"expected": ["array", "group"]}`. Every reader fills -`input`, so a definition's rules say only where a problem is. diff --git a/packages/zarr-metadata/changes/376.removal.md b/packages/zarr-metadata/changes/376.removal.md deleted file mode 100644 index 699fe0718a..0000000000 --- a/packages/zarr-metadata/changes/376.removal.md +++ /dev/null @@ -1,9 +0,0 @@ -`typeddict_keys(...).members` keeps the `Annotated` metadata of a key's -type, which carries its bounds. A definition's rules are asked only of a -configuration within its bounds, as pydantic's after-validators are, so -a value out of bounds is the one problem reported until it is fixed. A -definition that reuses one of the package's configuration TypedDicts -takes its bounds with it. A v2 `order` or `dimension_separator`, and -the `must_understand` of a consolidated envelope, are reported as values -outside a closed set -- `expected one of ["C", "F"], got "Q"` -- and a -`must_understand` that is not a boolean as `invalid_type`. diff --git a/packages/zarr-metadata/changes/377.feature.md b/packages/zarr-metadata/changes/377.feature.md deleted file mode 100644 index c7a01fee0f..0000000000 --- a/packages/zarr-metadata/changes/377.feature.md +++ /dev/null @@ -1,11 +0,0 @@ -`with_problems(fields, problems)` gives each field of one read -- a -reading's `fields()` and `problems`, or `fields_of` a field and the -problems `resolve` gave with it -- with the problems located in it, in -the fields it holds too: those it was read with, and those the document -found with it where it stands, its place in the pipeline, the chunk it -is handed, the array's shape. A field with none is valid there; a -problem in no field is in none's. It groups as zod's `flattenError` -groups issues, at every depth. `canonical_of(resolved, problems)` spells -a field a scope has read already, given its problems, in its simplest -equivalent spelling without reading it again, and gives what -`canonicalize` gives: None for a field with a problem. diff --git a/packages/zarr-metadata/changes/378.bugfix.md b/packages/zarr-metadata/changes/378.bugfix.md deleted file mode 100644 index bde372fb24..0000000000 --- a/packages/zarr-metadata/changes/378.bugfix.md +++ /dev/null @@ -1,9 +0,0 @@ -The `regular` chunk grid refuses a chunk length of 0, along a dimension -of length 0 too, as its specification says: "Chunk sizes must be greater -than zero". `RegularChunkGridConfiguration.chunk_shape` is -`tuple[Annotated[int, Ge(1)], ...]`, so a 0 is an `invalid_value` at its -place, with the bound in its `ctx`. The package had read the core -specification's "The chunk shape elements are non-zero when the -corresponding dimensions of the arrays have non-zero length" as allowing -0 on an empty dimension, as zarr-python 3.0 and 3.1 wrote it; that -sentence says less than the grid's own, not something else. diff --git a/packages/zarr-metadata/changes/378.feature.md b/packages/zarr-metadata/changes/378.feature.md deleted file mode 100644 index 9943b052bd..0000000000 --- a/packages/zarr-metadata/changes/378.feature.md +++ /dev/null @@ -1,21 +0,0 @@ -JSON Schemas of what the package reads, draft 2020-12, as pydantic's -`TypeAdapter(...).json_schema()` and zod's `toJSONSchema` write theirs. -`node_metadata_json_schema_v3(context=...)`, in `zarr_metadata.model`, -is a `zarr.json`'s, an array's or a group's, for an editor or a -validator in another language: each extension point a field as its -scope reads it -- a configuration as its definition's TypedDict says, -bounds and all, or a name nothing in the scope claims, with any -configuration -- the fill value what the data type it names takes, and -a group's consolidated metadata the documents it holds. -`field_json_schema(kind, context)`, in `zarr_metadata.v3.definition`, is -one field's, and `json_schema(shape)`, in `zarr_metadata.typed_json`, -any TypedDict's, as `check` reads it. What the rules say of members -together is not in a schema, so a document it accepts may still have a -problem; a JSON document the validators accept, it accepts. JSON Schema -takes `1.0` for an integer, where the package wants `1`. A v3 array -document's extension points are annotated with the field aliases -- -`data_type: DataTypeField`, `codecs: tuple[CodecField, ...]` -- which -are the JSON a metadata field is, so a type checker and `check` read -them as before; its `shape` holds integers of at least 0, which `check` -now holds it to. A data type whose `fill_value` holds a metadata field -is refused when it is built: a fill value is a value of its data type. diff --git a/packages/zarr-metadata/changes/379.feature.md b/packages/zarr-metadata/changes/379.feature.md deleted file mode 100644 index 941dbd5310..0000000000 --- a/packages/zarr-metadata/changes/379.feature.md +++ /dev/null @@ -1,16 +0,0 @@ -**Breaking:** a model checks itself when it is built, as pydantic's -`__init__` does, and `to_key_value` writes it as it is. A model built by -hand, or changed by `dataclasses.replace` or a v2 model's `update`, whose -document has a problem raises `MetadataValidationError` with every -problem at the change, where it was refused only when written, so no -model is built invalid; `ZarrV2ArrayMetadata.create_default` refuses -`chunks` its default shape does not take, as the v3 model refuses a -grid. A model holds its members as that read refines them, in containers -of its own, as pydantic holds what its `__init__` coerced: a list given -for an array is held as a tuple, and a dict given is not the dict held. -Each field, a `Read` or an `Unclaimed`, is held as the scope read it. A -model a read builds is not read a second time, and writing one reads -nothing again. Change a model by building another: a container it -holds, changed in place, is not checked again, as a pydantic model's is -not. A path in `ZarrV3ConsolidatedMetadata.metadata` that is not a string -is a `TypeError`. diff --git a/packages/zarr-metadata/changes/381.bugfix.1.md b/packages/zarr-metadata/changes/381.bugfix.1.md deleted file mode 100644 index 6bfaeb669a..0000000000 --- a/packages/zarr-metadata/changes/381.bugfix.1.md +++ /dev/null @@ -1,6 +0,0 @@ -A `zarr.json` that says no node type the spec defines is judged by its -`zarr_format` too, so a document of another format says it is not v3. -The root `zarr.json` of zarr-python 2's draft of v3, whose `zarr_format` -is a URL, was reported only as missing its `node_type`; it now also has -an `invalid_type` problem at `zarr_format`, and a v2 document read as a -`zarr.json` an `invalid_value` there. diff --git a/packages/zarr-metadata/changes/381.bugfix.2.md b/packages/zarr-metadata/changes/381.bugfix.2.md deleted file mode 100644 index 9c5b0f86bb..0000000000 --- a/packages/zarr-metadata/changes/381.bugfix.2.md +++ /dev/null @@ -1,9 +0,0 @@ -A group's consolidated metadata is judged as the hierarchy below the -group, the group its root: zarr-python keeps the document of the node at -`/a/b` at the key `a/b`, and the documents and the group make a tree in -which only groups hold nodes and each node's parent is held. A key that -is no node's path below the group, such as `""`, `"__a"` or `"a/../b"`, -and a document below an array's were read without a problem, and so was -a document whose group is missing, which zarr-python fails to read. Each -is a problem now, a missing group a `missing_key`, and -`ZarrV3ConsolidatedMetadata` refuses them when it is built. diff --git a/packages/zarr-metadata/changes/381.bugfix.3.md b/packages/zarr-metadata/changes/381.bugfix.3.md deleted file mode 100644 index bc3f7ef3ed..0000000000 --- a/packages/zarr-metadata/changes/381.bugfix.3.md +++ /dev/null @@ -1,14 +0,0 @@ -Two models are equal when they mean the same document, however each is -spelled. What the package interprets compares by its canonical spelling: -`"NaN"` and `"0x7fc00000"` were two `float32` fill values and `0.0` and -`-0.0` one, and a blosc with and without the `typesize` that `noshuffle` -ignores, a key encoding with and without its default separator, a shard -with and without `index_location: "end"` and a zstd with and without -`checksum: false` were two fields each, though `canonical_of` spelled -them alike; each is one now, and `default`, `v2`, `sharding_indexed` -and `zstd` fold their spec defaults in a `canonical` of their own. What -it does not interpret -- attributes, extra fields, unclaimed -configurations, every member of a v2 document -- compares as JSON text, -which tells `true` from `1` and `-0.0` from `0.0` and takes `NaN` for -itself, so a model holding a NaN attribute equals its own copy. Equal -fields and models hash alike. diff --git a/packages/zarr-metadata/changes/381.bugfix.md b/packages/zarr-metadata/changes/381.bugfix.md deleted file mode 100644 index 5b3ebacb69..0000000000 --- a/packages/zarr-metadata/changes/381.bugfix.md +++ /dev/null @@ -1,9 +0,0 @@ -A member a closed object does not declare is an `unknown_key` wherever -it sits, as `ProblemKind` defines one: in a v2 `.zgroup`, a `.zmetadata` -envelope, a group's `consolidated_metadata` and a metadata field's -envelope, where it was an `invalid_value`, as it already was inside a v3 -configuration. So a reader that tolerates a key another writer added, -such as netCDF-C's `_nczarr_*` keys in a `.zgroup`, can filter by kind. -A definition's `check` and `judge` give a configuration back when a -field it holds has a stray member, which they report and leave out, as -the checker does a key a closed TypedDict does not declare. diff --git a/packages/zarr-metadata/changes/381.feature.1.md b/packages/zarr-metadata/changes/381.feature.1.md deleted file mode 100644 index 48a0348c8b..0000000000 --- a/packages/zarr-metadata/changes/381.feature.1.md +++ /dev/null @@ -1,10 +0,0 @@ -A data type's definition says which spelling of a fill value is its -value's own: `DataTypeDefinition.fill_value_canonical`, and -`canonical_fill_value(data_type, value)` in `zarr_metadata.v3.definition` -spells one, or gives `UNSET` for a fill value with a problem. The float -types read a number as a float64, as JSON parsers and numpy do, and -spell a value by its shortest number, a named value, or the bits of a -NaN the spec does not name; the complex types spell each part so; the -numpy time types spell `-2**63` as `"NaT"`; `bytes` spells its bytes as -base64; and a struct spells each field's fill value by that field's -type. diff --git a/packages/zarr-metadata/changes/381.feature.md b/packages/zarr-metadata/changes/381.feature.md deleted file mode 100644 index 03dd5fd425..0000000000 --- a/packages/zarr-metadata/changes/381.feature.md +++ /dev/null @@ -1,5 +0,0 @@ -`NodeName` and `NodePath`, in `zarr_metadata.v3`, are the names and paths -of the nodes of a v3 hierarchy, modeled on zarrs' types of those names, -and `zarr_metadata.model` judges a string by the spec's rules for them: -`validate_node_name_v3`, `is_node_name_v3` and `parse_node_name_v3`, and -their `node_path` twins. diff --git a/packages/zarr-metadata/changes/382.bugfix.1.md b/packages/zarr-metadata/changes/382.bugfix.1.md deleted file mode 100644 index e44aa01c12..0000000000 --- a/packages/zarr-metadata/changes/382.bugfix.1.md +++ /dev/null @@ -1 +0,0 @@ -A `Read` or `Unclaimed` object placed inside a document passed to a reader is now refused as not JSON. Before, it passed validation as if it had been read, and a model could then write a document that the same validator refused. diff --git a/packages/zarr-metadata/changes/382.bugfix.2.md b/packages/zarr-metadata/changes/382.bugfix.2.md deleted file mode 100644 index b4ef29356f..0000000000 --- a/packages/zarr-metadata/changes/382.bugfix.2.md +++ /dev/null @@ -1 +0,0 @@ -Absent members are now `UNSET` instead of `None`, so they can be told apart from a JSON `null`: a reading's `data_type`, `chunk_grid` and `chunk_key_encoding` when the key is missing, and `Refused.json` when the field was not JSON. A metadata field with no `name` is a `missing_key` problem instead of `invalid_type`. `ZarrV3ConsolidatedMetadata` declares `must_understand` as `False` instead of accepting and then refusing a value. diff --git a/packages/zarr-metadata/changes/382.bugfix.3.md b/packages/zarr-metadata/changes/382.bugfix.3.md deleted file mode 100644 index 8ccd49e534..0000000000 --- a/packages/zarr-metadata/changes/382.bugfix.3.md +++ /dev/null @@ -1 +0,0 @@ -A group's `consolidated_metadata` of `null` is now an `invalid_type` problem. It is no longer read as absent, and neither the package's JSON Schema nor pydantic's admits it. The spec says the member is an object. zarr-python 3.0.x wrote `null` by mistake; a reader of those stores should remove the key before reading. diff --git a/packages/zarr-metadata/changes/382.bugfix.4.md b/packages/zarr-metadata/changes/382.bugfix.4.md deleted file mode 100644 index c0d3d91e27..0000000000 --- a/packages/zarr-metadata/changes/382.bugfix.4.md +++ /dev/null @@ -1 +0,0 @@ -Definition functions (`rules`, `canonical`, and the rest) always receive a read-only view of a field's configuration, so a rule that assigns to it fails with an error naming the rule. A raw-bits name with more than 100 digits is treated as an extension name, not a size; before, it raised `ValueError`. Problem messages now show an over-long integer by its bit length, a too-deeply-nested value as "too deep to show", and a non-string key with Python's repr; before, each of these raised. diff --git a/packages/zarr-metadata/changes/382.bugfix.md b/packages/zarr-metadata/changes/382.bugfix.md deleted file mode 100644 index 9dbc638adb..0000000000 --- a/packages/zarr-metadata/changes/382.bugfix.md +++ /dev/null @@ -1 +0,0 @@ -Validators, readers and `from_key_value` no longer raise `RecursionError` on deeply nested documents, including a chain of groups each nested in the previous one's consolidated metadata. Nesting is capped at 256 levels. A value nested deeper is reported as an `invalid_value` problem at the level where the limit is passed. diff --git a/packages/zarr-metadata/changes/382.feature.1.md b/packages/zarr-metadata/changes/382.feature.1.md deleted file mode 100644 index eb59aa833c..0000000000 --- a/packages/zarr-metadata/changes/382.feature.1.md +++ /dev/null @@ -1 +0,0 @@ -The pydantic types raise `ValidationError` with one line error per problem, each carrying `type` (the problem's kind), `loc`, `input` and `ctx`. Before, every problem was concatenated into one `value_error`. For a missing key, `input` is the object that lacks it, as pydantic's own `missing` error does. A message that contains a placeholder such as `{expected}` is passed through unchanged. diff --git a/packages/zarr-metadata/changes/382.feature.md b/packages/zarr-metadata/changes/382.feature.md deleted file mode 100644 index 3f7e74d7d7..0000000000 --- a/packages/zarr-metadata/changes/382.feature.md +++ /dev/null @@ -1 +0,0 @@ -Extension names must match `^[a-z][a-z0-9-_.]+$` or be a URI in RFC 3986 characters. Any other name (`""`, `"foo/bar"`) is refused with an `invalid_value` problem at `name` before any definition is looked up; before, every string was read as an unknown extension. `well_named` checks a string, `validate_metadata_field_v3` enforces the rule, the JSON Schema carries the same pattern, and a definition built with an invalid name raises `TypeError`. diff --git a/packages/zarr-metadata/changes/391.feature.md b/packages/zarr-metadata/changes/391.feature.md deleted file mode 100644 index 12e826e910..0000000000 --- a/packages/zarr-metadata/changes/391.feature.md +++ /dev/null @@ -1 +0,0 @@ -`read_repaired_node_metadata_v3` reads a v3 `zarr.json` after fixing known writer bugs and returns the list of repairs it made. It covers a chunk length of 0 along an empty dimension (zarr-python 3.0 and 3.1) and `consolidated_metadata: null` (zarr-python 3.0.x). `repair_node_metadata_v3` does only the repair; the strict readers still refuse these documents. diff --git a/packages/zarr-metadata/changes/392.feature.md b/packages/zarr-metadata/changes/392.feature.md deleted file mode 100644 index e89f440e76..0000000000 --- a/packages/zarr-metadata/changes/392.feature.md +++ /dev/null @@ -1 +0,0 @@ -`Context` values are equal when they hold the same definitions, and equal values hash alike. `Context.joined` combines scopes and raises `ScopeConflictError` if two of them define one name differently. `Context.disagreements` reports where a scope would read a document's fields differently. `claims_of` lists the definition a reading used for each name, and `refines` says whether one reading of a field contains everything another does. diff --git a/packages/zarr-metadata/changes/393.feature.md b/packages/zarr-metadata/changes/393.feature.md deleted file mode 100644 index df16213512..0000000000 --- a/packages/zarr-metadata/changes/393.feature.md +++ /dev/null @@ -1 +0,0 @@ -A v3 model is now a pair: its document and the scope it was read in. `to_json` returns the document as it was written, and `update` reads the changed document in the model's own scope, so no scope is passed in. `with_context` and `refined_in` re-read a model in another scope, and `refines` compares two models by what they mean. diff --git a/packages/zarr-metadata/changes/393.removal.md b/packages/zarr-metadata/changes/393.removal.md deleted file mode 100644 index 2c58cf7ce8..0000000000 --- a/packages/zarr-metadata/changes/393.removal.md +++ /dev/null @@ -1 +0,0 @@ -The v3 models are no longer dataclasses built from typed fields. `ZarrV3ArrayMetadata(shape=..., data_type=, ...)`, `dataclasses.replace` and `dataclasses.fields` no longer work on them; build a model from a document and change it with `update`. `update` no longer takes `context`, and `to_json` no longer rewrites field spelling (`"bytes"` stays `"bytes"`). `attributes`, `extra_fields` and a struct `fill_value` are read-only at every level (objects as read-only mappings, arrays as tuples), so they cannot be modified in place, serialized or pickled directly; use `to_json()` for plain JSON containers. diff --git a/packages/zarr-metadata/changes/395.feature.1.md b/packages/zarr-metadata/changes/395.feature.1.md deleted file mode 100644 index f01fc9f35b..0000000000 --- a/packages/zarr-metadata/changes/395.feature.1.md +++ /dev/null @@ -1 +0,0 @@ -Every reader, validator and guard accepts `context=None` for the default scope, `CORE_AND_EXTENSIONS`, like the model constructors. diff --git a/packages/zarr-metadata/changes/395.feature.md b/packages/zarr-metadata/changes/395.feature.md deleted file mode 100644 index 9b7be745ee..0000000000 --- a/packages/zarr-metadata/changes/395.feature.md +++ /dev/null @@ -1 +0,0 @@ -A group's `consolidated_metadata` can be built from node models, in a constructor or in `update`. A model is accepted when the group's scope reads it the same way its own scope did, or claims names its scope left unclaimed. It is refused, with a problem at its path, when the two scopes read a name with different definitions, or when the group's scope does not claim a name the model's scope did. diff --git a/packages/zarr-metadata/changes/396.feature.md b/packages/zarr-metadata/changes/396.feature.md deleted file mode 100644 index d1b2b9c09d..0000000000 --- a/packages/zarr-metadata/changes/396.feature.md +++ /dev/null @@ -1 +0,0 @@ -Zarr v2 array metadata is now read against a scope, as v3 metadata is. `zarr_metadata.v2.definition` defines the data types zarr-python 2.x writes (one definition per NumPy family) and 21 of the codecs numcodecs 0.16 configures, files them in `CORE_V2`, and reads one `dtype` or codec with `resolve_dtype_v2` and `resolve_codec_v2`. `validate_array_metadata_v2` now reports a dtype string that is not a NumPy typestr or that its family does not take, a codec parameter outside what numcodecs accepts, and a fill value its dtype does not take; a dtype or codec id the package does not model is left unjudged as before. `ZarrV2ArrayMetadata.create_default` given a dtype whose family does not take `0` as a fill value, such as `|b1` or `|S3`, now sets `fill_value` to `null` unless one is given. diff --git a/packages/zarr-metadata/changes/397.feature.md b/packages/zarr-metadata/changes/397.feature.md deleted file mode 100644 index 182c72003d..0000000000 --- a/packages/zarr-metadata/changes/397.feature.md +++ /dev/null @@ -1 +0,0 @@ -The v2 models are now a document and the scope it was read in, as the v3 models are: `ZarrV2ArrayMetadata(document, context=None)` reads the document in `CORE_V2` or the scope given, holds `dtype`, `compressor` and each filter as the scope read them, and compares by what the document means, so ` Date: Thu, 8 Oct 2026 16:08:08 +0200 Subject: [PATCH 82/94] docs(zarr-metadata): one changelog fragment for the whole PR Assisted-by: ClaudeCode:claude-fable-5-1 Co-Authored-By: Claude Fable 5.1 --- .../zarr-metadata/changes/4490.bugfix.1.md | 11 ----- .../zarr-metadata/changes/4490.bugfix.2.md | 7 --- .../zarr-metadata/changes/4490.bugfix.3.md | 9 ---- packages/zarr-metadata/changes/4490.bugfix.md | 7 --- packages/zarr-metadata/changes/4490.doc.md | 7 --- .../zarr-metadata/changes/4490.feature.1.md | 7 --- .../zarr-metadata/changes/4490.feature.2.md | 13 ------ .../zarr-metadata/changes/4490.feature.3.md | 9 ---- .../zarr-metadata/changes/4490.feature.4.md | 12 ------ .../zarr-metadata/changes/4490.feature.5.md | 9 ---- .../zarr-metadata/changes/4490.feature.6.md | 10 ----- .../zarr-metadata/changes/4490.feature.7.md | 20 --------- .../zarr-metadata/changes/4490.feature.8.md | 10 ----- .../zarr-metadata/changes/4490.feature.md | 43 ++++++++++++------- .../zarr-metadata/changes/4490.removal.1.md | 8 ---- .../zarr-metadata/changes/4490.removal.2.md | 9 ---- .../zarr-metadata/changes/4490.removal.md | 13 ------ 17 files changed, 28 insertions(+), 176 deletions(-) delete mode 100644 packages/zarr-metadata/changes/4490.bugfix.1.md delete mode 100644 packages/zarr-metadata/changes/4490.bugfix.2.md delete mode 100644 packages/zarr-metadata/changes/4490.bugfix.3.md delete mode 100644 packages/zarr-metadata/changes/4490.bugfix.md delete mode 100644 packages/zarr-metadata/changes/4490.doc.md delete mode 100644 packages/zarr-metadata/changes/4490.feature.1.md delete mode 100644 packages/zarr-metadata/changes/4490.feature.2.md delete mode 100644 packages/zarr-metadata/changes/4490.feature.3.md delete mode 100644 packages/zarr-metadata/changes/4490.feature.4.md delete mode 100644 packages/zarr-metadata/changes/4490.feature.5.md delete mode 100644 packages/zarr-metadata/changes/4490.feature.6.md delete mode 100644 packages/zarr-metadata/changes/4490.feature.7.md delete mode 100644 packages/zarr-metadata/changes/4490.feature.8.md delete mode 100644 packages/zarr-metadata/changes/4490.removal.1.md delete mode 100644 packages/zarr-metadata/changes/4490.removal.2.md delete mode 100644 packages/zarr-metadata/changes/4490.removal.md diff --git a/packages/zarr-metadata/changes/4490.bugfix.1.md b/packages/zarr-metadata/changes/4490.bugfix.1.md deleted file mode 100644 index 0853f5c4a4..0000000000 --- a/packages/zarr-metadata/changes/4490.bugfix.1.md +++ /dev/null @@ -1,11 +0,0 @@ -Problem kinds and messages are more precise. A problem's message shows -what the document holds as JSON, `null` or `[1, 2]`, not the Python -value, and names what was expected in JSON's words. A value of another -JSON type than a closed set's, such as `"zarr_format": "3"`, is -`invalid_type`, where it was `invalid_value`. A member a closed object -does not declare is an `unknown_key` wherever it sits, in a `.zgroup`, -a `.zmetadata` envelope or a field's envelope, so a reader that -tolerates another writer's keys, such as netCDF-C's `_nczarr_*`, can -filter by kind. A metadata field with no `name` is a `missing_key`. A -member of a `.zarray` the spec does not define is kept and written -back. diff --git a/packages/zarr-metadata/changes/4490.bugfix.2.md b/packages/zarr-metadata/changes/4490.bugfix.2.md deleted file mode 100644 index 6a3bbbfcc4..0000000000 --- a/packages/zarr-metadata/changes/4490.bugfix.2.md +++ /dev/null @@ -1,7 +0,0 @@ -A group's consolidated metadata is judged as the hierarchy below the -group: a key that is no node's path, such as `""` or `"a/../b"`, a -document below an array's, and a document whose parent group is missing -are problems, where they were read without one. A `zarr.json` of no -node type the spec defines is judged by its `zarr_format` too, so a v2 -document, or the root of zarr-python 2's draft of v3, is reported as -not v3 rather than only as missing its `node_type`. diff --git a/packages/zarr-metadata/changes/4490.bugfix.3.md b/packages/zarr-metadata/changes/4490.bugfix.3.md deleted file mode 100644 index 704593bafa..0000000000 --- a/packages/zarr-metadata/changes/4490.bugfix.3.md +++ /dev/null @@ -1,9 +0,0 @@ -Validators and readers no longer raise `RecursionError` on deeply -nested documents, including a chain of groups each nested in the -previous one's consolidated metadata: nesting is capped at 256 levels -and a deeper value is an `invalid_value` problem where the limit is -passed. A raw-bits name with more than 100 digits is an extension name, -not a `ValueError`; an over-long integer and a non-string key are shown -in a message where each raised. A definition's rules receive a read-only -view of a configuration, so a rule that assigns to it fails with an -error naming the rule. diff --git a/packages/zarr-metadata/changes/4490.bugfix.md b/packages/zarr-metadata/changes/4490.bugfix.md deleted file mode 100644 index 37c3ae5f7c..0000000000 --- a/packages/zarr-metadata/changes/4490.bugfix.md +++ /dev/null @@ -1,7 +0,0 @@ -`to_json` and `to_key_value` write every extension point but a data -type as an object, `{"name": "crc32c"}`, where they wrote the bare name -for a field with nothing to configure. A Zarr v3.0 reader takes no -short-hand name in `codecs`, and no zarr-python release reads one there -or in `chunk_key_encoding`, so a document the package wrote could not be -opened by them. A data type with nothing to configure is still written -by its bare name. diff --git a/packages/zarr-metadata/changes/4490.doc.md b/packages/zarr-metadata/changes/4490.doc.md deleted file mode 100644 index 7a172756ec..0000000000 --- a/packages/zarr-metadata/changes/4490.doc.md +++ /dev/null @@ -1,7 +0,0 @@ -The README says what the core depends on, `typing-extensions` and -`annotated-types` only, and why it has no validation framework: a pin -would collide with consumers' pins. The validation boundary says what -the package decides where the specs leave it open: `attributes` may -hold `NaN` and `Infinity`, which `to_key_value` writes as bare tokens. -The models' `from_json`, `to_json`, `from_key_value` and `to_key_value` -have docstrings, and no public docstring names a private function. diff --git a/packages/zarr-metadata/changes/4490.feature.1.md b/packages/zarr-metadata/changes/4490.feature.1.md deleted file mode 100644 index 88ad9d9046..0000000000 --- a/packages/zarr-metadata/changes/4490.feature.1.md +++ /dev/null @@ -1,7 +0,0 @@ -A field a scope reads is `Read`, by the definition in scope that claims -its name, holding that definition and the configuration it allowed; -`Unclaimed`, when nothing in scope claims it; or `Refused`, with every -problem. `Resolved` is their union, for `match`. Each answers `read_as`, -`name`, `definition` and `nested`, the fields it holds as the scope read -them, such as a shard's inner codecs. Two fields are equal when they -read the same, however each was spelled. diff --git a/packages/zarr-metadata/changes/4490.feature.2.md b/packages/zarr-metadata/changes/4490.feature.2.md deleted file mode 100644 index d4f1e4845c..0000000000 --- a/packages/zarr-metadata/changes/4490.feature.2.md +++ /dev/null @@ -1,13 +0,0 @@ -`read_array_metadata_v3`, `read_group_metadata_v3` and -`read_node_metadata_v3` read a document once and return everything the -read found as a reading: each extension point as the scope read it, the -codecs as a pipeline with the chunk each is handed, every problem, and, -when there is none, the model built by the same read. A reading's -`fields()` gives each field with where it sits, the fields it holds -after it. `read_node_metadata_v3` reads a `zarr.json` of either kind by -its `node_type`, and gives a `ZarrV3UnknownNodeReading` for a document -that says neither; `node_metadata_from_json_v3` and -`node_metadata_from_key_value_v3` build the model of either kind. A -group's consolidated metadata is read the same way, each document it -holds once. Every reader, validator and guard takes `context=None` for -the default scope. diff --git a/packages/zarr-metadata/changes/4490.feature.3.md b/packages/zarr-metadata/changes/4490.feature.3.md deleted file mode 100644 index afc6c28aad..0000000000 --- a/packages/zarr-metadata/changes/4490.feature.3.md +++ /dev/null @@ -1,9 +0,0 @@ -Scopes have an algebra. `Context` values are equal when they hold the -same definitions, and hash alike; `Context.joined` combines scopes and -raises `ScopeConflictError` when two define one name differently. A -group's `consolidated_metadata` can be built from node models, in the -constructor or in `update`: a model is accepted when the group's scope -reads it as its own scope did, or claims names its scope left -unclaimed, and refused with a problem at its path when the two scopes -read a name differently or the group's scope claims less. A scope's -repr says how many definitions it holds. diff --git a/packages/zarr-metadata/changes/4490.feature.4.md b/packages/zarr-metadata/changes/4490.feature.4.md deleted file mode 100644 index 16da0bbfb5..0000000000 --- a/packages/zarr-metadata/changes/4490.feature.4.md +++ /dev/null @@ -1,12 +0,0 @@ -A number's type carries its bounds, as pydantic reads them: -annotated-types' `Gt`, `Ge`, `Lt`, `Le` and `Interval`, so a gzip -`level` is `Annotated[int, Interval(ge=0, le=9)]`, and every reader -holds a value to them with one `invalid_value` problem. The package's -own bounds are declared this way: codec levels, blosc's `clevel` and -`blocksize`, chunk and shard shapes, integer fill values, and the numpy -time types' `scale_factor`. A `ValidationProblem` carries its message's -data, as pydantic's errors do: `input`, the JSON at its `loc`, and -`ctx`, what was expected, such as `{"ge": 0, "le": 9}` or -`{"expected": ["array", "group"]}`. The pydantic types raise -`ValidationError` with one line error per problem, each carrying -`type`, `loc`, `input` and `ctx`. diff --git a/packages/zarr-metadata/changes/4490.feature.5.md b/packages/zarr-metadata/changes/4490.feature.5.md deleted file mode 100644 index 9eaff4b1f7..0000000000 --- a/packages/zarr-metadata/changes/4490.feature.5.md +++ /dev/null @@ -1,9 +0,0 @@ -JSON Schemas, draft 2020-12, of what the package reads. -`node_metadata_json_schema_v3(context=...)` in `zarr_metadata.model` is -a `zarr.json`'s, an array's or a group's, for an editor or a validator -in another language: each extension point a field as its scope reads -it, the fill value what its data type takes, and a group's consolidated -metadata the documents it holds. `json_schema(shape)` in -`zarr_metadata.typed_json` is any TypedDict's. What the rules say of -members together is not in a schema, so a document a schema accepts may -still have a problem. diff --git a/packages/zarr-metadata/changes/4490.feature.6.md b/packages/zarr-metadata/changes/4490.feature.6.md deleted file mode 100644 index b5cedaf409..0000000000 --- a/packages/zarr-metadata/changes/4490.feature.6.md +++ /dev/null @@ -1,10 +0,0 @@ -Known writer bugs are undone below the strict model. -`read_repaired_node_metadata_v3` reads a v3 `zarr.json` after fixing a -chunk length of 0 along an empty dimension (zarr-python 3.0 and 3.1) and -`consolidated_metadata: null` (zarr-python 3.0.x), and returns the -repairs it made; `repair_node_metadata_v3` does only the repair. -`read_repaired_consolidated_metadata_v2` removes the -`consolidated_metadata` member zarr-python 3.x writes into each -`.zgroup` entry below the root of a v2 `.zmetadata`, and -`repair_consolidated_metadata_v2` says what was changed. The strict -readers still refuse these documents. diff --git a/packages/zarr-metadata/changes/4490.feature.7.md b/packages/zarr-metadata/changes/4490.feature.7.md deleted file mode 100644 index 144349560e..0000000000 --- a/packages/zarr-metadata/changes/4490.feature.7.md +++ /dev/null @@ -1,20 +0,0 @@ -Zarr v2 metadata is read against a scope, as v3 metadata is. -`zarr_metadata.v2.definition` defines the data types zarr-python 2.x -writes, one per NumPy family, and 21 of the codecs numcodecs 0.16 -configures, files them in `CORE_V2`, and reads one `dtype` or codec with -`resolve_dtype_v2` and `resolve_codec_v2`. `validate_array_metadata_v2` -reports a dtype string that is not a NumPy typestr, a codec parameter -outside what numcodecs accepts, and a fill value its dtype does not -take; a dtype or codec id the package does not model is left unjudged. -`ZarrV2ArrayMetadata(document, context=None)` is the pair as the v3 -models are, holding `dtype`, `compressor` and each filter as the scope -read them, with `update`, `with_context`, `refined_in` and `refines`; -`ZarrV2ConsolidatedMetadata` keeps its entries as written and adds -`nodes`, each `.zarray` or `.zgroup` entry with its `.zattrs` as a -model; `read_array_metadata_v2` reads a document once and returns -everything it found. The v2 pydantic types read in the scope under the -validation context's `zarr_metadata_context_v2` key, the v3 ones under -`zarr_metadata_context`, and every field type reads in a bare `Context` -given as the context. `ZarrV2ArrayMetadata.create_default` given a -dtype whose family does not take `0`, such as `|b1` or `|S3`, sets -`fill_value` to `null` unless one is given. diff --git a/packages/zarr-metadata/changes/4490.feature.8.md b/packages/zarr-metadata/changes/4490.feature.8.md deleted file mode 100644 index 28a3bdd7ae..0000000000 --- a/packages/zarr-metadata/changes/4490.feature.8.md +++ /dev/null @@ -1,10 +0,0 @@ -`NodeName` and `NodePath` in `zarr_metadata.v3` are the names and paths -of the nodes of a v3 hierarchy, and `zarr_metadata.model` judges a -string by the spec's rules for them: `validate_node_name_v3`, -`is_node_name_v3`, `parse_node_name_v3` and their `node_path` twins. -`canonical_fill_value(data_type, value)` spells a fill value as its data -type's own spelling: the float types by the shortest number, a named -value or the bits of a NaN the spec does not name, the numpy time types -`-2**63` as `"NaT"`, `bytes` as base64, and a struct each field by its -type. `ZarrV3ArrayMetadata.create_default`'s codec is `bytes` with a -little `endian`, so the default of any data type of fixed size is valid. diff --git a/packages/zarr-metadata/changes/4490.feature.md b/packages/zarr-metadata/changes/4490.feature.md index 37431985b8..6051b47dfe 100644 --- a/packages/zarr-metadata/changes/4490.feature.md +++ b/packages/zarr-metadata/changes/4490.feature.md @@ -1,15 +1,28 @@ -Every v3 model is now a pair: the document as written and the scope it -was read in. `ZarrV3ArrayMetadata(document, context=None)` reads the -document in `CORE_AND_EXTENSIONS` or the scope given, holds each -extension point as that scope read it, and is never invalid: a document -with a problem raises `MetadataValidationError` with every problem. -`to_json` returns the document as written, refined; `update` reads new -members, given as JSON, in the model's own scope; `with_context` and -`refined_in` read the document in another scope; `refines` says whether -one model holds everything another does. Two models are equal, and hash -alike, when their documents mean the same in their scopes: `"NaN"` and -`"0x7fc00000"` are one `float32` fill value, a zstd with and without -`checksum: false` one codec, while attributes and unclaimed -configurations compare as JSON text. `attributes`, `extra_fields` and a -struct fill value are read-only at every level; use `to_json()` for -plain containers. +**Breaking:** metadata is read against definitions in a scope, and every +model is the pair of its document and the scope it was read in. A field +(a v3 data type, chunk grid, chunk key encoding, codec or storage +transformer; a v2 dtype or numcodecs codec) is `Read` by the definition +in scope that claims its name, `Unclaimed` when nothing does, or +`Refused` with every problem; `read_array_metadata_v3`, +`read_group_metadata_v3`, `read_node_metadata_v3` and +`read_array_metadata_v2` read a document once and return everything the +read found, the model among it. A model is built from a document in a +scope (`CORE_AND_EXTENSIONS` or `CORE_V2` by default) and is never +invalid; two models are equal when their documents mean the same; +`update` reads new members, given as JSON, in the model's own scope, +`with_context` and `refined_in` read the document in another, and +`refines` orders models by information. Scopes compare, join and report +disagreements, a group's consolidated metadata accepts node models, +number types carry their bounds and problems carry `input` and `ctx`, +JSON Schemas are exported for a TypedDict, a field and a `zarr.json`, +and `read_repaired_node_metadata_v3` and +`read_repaired_consolidated_metadata_v2` undo known writer bugs below +the strict model. Gone: the dataclass models and `construct`, the +`...Partial` TypedDicts (now `...Update`), `ZarrV3NamedConfig`, +`ZarrV3MetadataField`, `Resolution`/`Unread`, the `context` argument of +`update` and `to_key_value`, and the helpers `zarr_metadata.model`, +`zarr_metadata.v3.definition` and `zarr_metadata.typed_json` re-exported +from the core. Refused where accepted before: `must_understand: false`, +extension names outside `^[a-z][a-z0-9-_.]+$` or a URI, a chunk length +of 0, `consolidated_metadata: null`, and a structured v2 dtype with +`fill_value: 0`. diff --git a/packages/zarr-metadata/changes/4490.removal.1.md b/packages/zarr-metadata/changes/4490.removal.1.md deleted file mode 100644 index 4ce48e4467..0000000000 --- a/packages/zarr-metadata/changes/4490.removal.1.md +++ /dev/null @@ -1,8 +0,0 @@ -The public doors are smaller. `zarr_metadata.v3.definition` no longer -exports `check` (still in `zarr_metadata.typed_json`), `canonicalize`, -`configuration_of`, `chunk_grid_lengths`, `read_pipeline` or -`ZarrV3MetadataFieldJSON` (still in `zarr_metadata.v3`). -`zarr_metadata.model` no longer re-exports the `*_KEYS_*` sets or -`is_json`, `parse_json` and `validate_json` (still in `zarr_metadata`). -`zarr_metadata.typed_json` no longer exports `typeddict_keys` and -`TypedDictKeys`. diff --git a/packages/zarr-metadata/changes/4490.removal.2.md b/packages/zarr-metadata/changes/4490.removal.2.md deleted file mode 100644 index 2226671749..0000000000 --- a/packages/zarr-metadata/changes/4490.removal.2.md +++ /dev/null @@ -1,9 +0,0 @@ -Documents the specs do not allow are refused where they were accepted. -A `must_understand: false` is refused at every extension point, codecs -too. An extension name must match `^[a-z][a-z0-9-_.]+$` or be a URI; -any other name, such as `""` or `"foo/bar"`, is an `invalid_value` at -`name`. The `regular` chunk grid refuses a chunk length of 0, along a -dimension of length 0 too. A group's `consolidated_metadata` of `null` -is an `invalid_type`. A `.zarray` with a structured dtype and -`fill_value: 0` is refused; the spec requires base64 or null. Readers of -stores zarr-python 3.0 and 3.1 wrote can use the repair readers. diff --git a/packages/zarr-metadata/changes/4490.removal.md b/packages/zarr-metadata/changes/4490.removal.md deleted file mode 100644 index 12e87d09b9..0000000000 --- a/packages/zarr-metadata/changes/4490.removal.md +++ /dev/null @@ -1,13 +0,0 @@ -The models are no longer dataclasses built from typed members. -`ZarrV3ArrayMetadata(shape=..., data_type=...)`, -`ZarrV2ArrayMetadata(shape=..., dtype=...)`, `construct`, -`dataclasses.replace` and `dataclasses.fields` no longer work on them: -build a model from a document and change it with `update`, which takes -members as JSON, as `ZarrV3ArrayMetadataUpdate`, -`ZarrV3GroupMetadataUpdate`, `ZarrV2ArrayMetadataUpdate` and -`ZarrV2GroupMetadataUpdate` type them; the `...Partial` TypedDicts are -gone. `update` and `to_key_value` take no `context`. `ZarrV3NamedConfig` -and `ZarrV3MetadataField`, in the model and in `zarr_metadata.pydantic`, -are gone: a model holds each extension point as `Read` or `Unclaimed`. -`Resolution` and `Unread` are replaced by `Read`, `Unclaimed` and -`Refused`, each built with keywords. From fb6a18d17e8929524d78de7b92ae3d0f7e388767 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Thu, 8 Oct 2026 21:43:54 +0200 Subject: [PATCH 83/94] refactor(zarr-metadata): declare a kind with the class keyword kind=True A kind is declared in its class header, class CodecDefinition(Definition[C], kind=True), as typing.Protocol and SQLAlchemy's __abstract__ mark a class and not its subclasses, instead of an is_kind ClassVar that a subclass inherits but that did not count. The hook keeps the mark in the class namespace, which the slots rebuild of a dataclass carries over while the keyword is not. Assisted-by: ClaudeCode:claude-fable-5-1 Co-Authored-By: Claude Fable 5.1 --- .../src/zarr_metadata/v2/_definition.py | 6 +- .../src/zarr_metadata/v3/_definition.py | 57 +++++++++---------- packages/zarr-metadata/tests/v3/test_kinds.py | 17 +++--- 3 files changed, 35 insertions(+), 45 deletions(-) diff --git a/packages/zarr-metadata/src/zarr_metadata/v2/_definition.py b/packages/zarr-metadata/src/zarr_metadata/v2/_definition.py index 4712fb44fe..f83e550036 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v2/_definition.py +++ b/packages/zarr-metadata/src/zarr_metadata/v2/_definition.py @@ -106,7 +106,7 @@ def typestr_problem(name: str, at: Loc) -> ValidationProblem | None: @dataclass(frozen=True, kw_only=True, slots=True, repr=False) -class ZarrV2DataTypeDefinition(WithFillValue[C]): +class ZarrV2DataTypeDefinition(WithFillValue[C], kind=True): """A v2 data type: one family of NumPy types, and the fill value an array of it takes. Filed under the family -- `float` -- and read for every typestr of @@ -117,7 +117,6 @@ class ZarrV2DataTypeDefinition(WithFillValue[C]): `("dtype", "fields", 0, 1)` for the type of the first record. """ - is_kind: ClassVar[bool] = True label: ClassVar[str] = "v2 data type" field_aliases: ClassVar[tuple[TypeAliasType, ...]] = (ZarrV2DataTypeField,) @@ -201,14 +200,13 @@ def name_loc(cls, loc: Loc) -> Loc: @dataclass(frozen=True, kw_only=True, slots=True, repr=False) -class ZarrV2CodecDefinition(Definition[C]): +class ZarrV2CodecDefinition(Definition[C], kind=True): """A v2 codec: a numcodecs id, and the TypedDict its parameters are. A document writes `{"id": name, **parameters}`; the definition's configuration is the parameters, read at the field itself. """ - is_kind: ClassVar[bool] = True label: ClassVar[str] = "v2 codec" @classmethod diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py b/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py index 4d0203df08..b89d928182 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py @@ -212,11 +212,11 @@ class Definition(Generic[C]): and a `name` or rules that are not what they say. """ - is_kind: ClassVar[bool] = False - """Whether this class is a kind: what a scope files definitions by. + _is_kind: ClassVar[bool] = False + """Whether this class is a kind, what a scope files definitions by: set by `class Kind(Definition, kind=True)`. - Set in a kind's own body and read from it, never inherited: a - subclass of a kind is a definition of that kind. + Read from a class's own namespace, never inherited: a subclass of a + kind is a definition of that kind, and declares no `kind`. """ label: ClassVar[str] = "definition" """The kind as a message names it: "codec".""" @@ -290,12 +290,17 @@ def name_loc(cls, loc: Loc) -> Loc: """Where the name of a field at `loc`, written as an object, sits: under `name` for v3.""" return (*loc, "name") - def __init_subclass__(cls, **kwargs: object) -> None: + def __init_subclass__(cls, *, kind: bool = False, **kwargs: object) -> None: + """Files a subclass: `kind=True` declares a kind, as `typing.Protocol` and SQLAlchemy's `__abstract__` mark a class and not its subclasses.""" # Named, not `super()`: a dataclass with slots is rebuilt, and the # cell a bare `super()` reads names the class that was thrown away. super(Definition, cls).__init_subclass__(**kwargs) - # A dataclass with slots is built twice, and the class built last - # is the one a document is read with: it files its aliases last. + # A dataclass with slots is built twice, the second time without + # the class keywords but with the first class's namespace, and the + # class built last is the one a document is read with: the mark is + # kept in the namespace, and the aliases are filed last. + if kind: + cls._is_kind = True for alias in cls.__dict__.get("field_aliases", ()): _FIELD_KINDS[alias] = cls @@ -451,7 +456,7 @@ def _refusal(self) -> str | None: @dataclass(frozen=True, kw_only=True, slots=True, repr=False) -class DataTypeDefinition(WithFillValue[C]): +class DataTypeDefinition(WithFillValue[C], kind=True): """A data type, and the fill value an array of it takes. `fill_value` is the JSON shape of a fill value -- `Int8FillValue`, an @@ -483,7 +488,6 @@ class DataTypeDefinition(WithFillValue[C]): this definition. """ - is_kind: ClassVar[bool] = True label: ClassVar[str] = "data type" field_aliases: ClassVar[tuple[TypeAliasType, ...]] = (DataTypeField,) @@ -529,7 +533,7 @@ def _fill_value_parser(annotation: object) -> Parser: @dataclass(frozen=True, kw_only=True, slots=True, repr=False) -class ChunkGridDefinition(Definition[C]): +class ChunkGridDefinition(Definition[C], kind=True): """A chunk grid, and the arrays it fits. `shape_rules` is what the spec disallows in a grid of this @@ -547,7 +551,6 @@ class ChunkGridDefinition(Definition[C]): every axis unknown. """ - is_kind: ClassVar[bool] = True label: ClassVar[str] = "chunk grid" field_aliases: ClassVar[tuple[TypeAliasType, ...]] = (ChunkGridField,) @@ -558,10 +561,9 @@ class ChunkGridDefinition(Definition[C]): @dataclass(frozen=True, kw_only=True, slots=True, repr=False) -class ChunkKeyEncodingDefinition(Definition[C]): +class ChunkKeyEncodingDefinition(Definition[C], kind=True): """A chunk key encoding.""" - is_kind: ClassVar[bool] = True label: ClassVar[str] = "chunk key encoding" field_aliases: ClassVar[tuple[TypeAliasType, ...]] = (ChunkKeyEncodingField,) @@ -585,7 +587,7 @@ class ChunkKeyEncodingDefinition(Definition[C]): @dataclass(frozen=True, kw_only=True, slots=True, repr=False) -class CodecDefinition(Definition[C]): +class CodecDefinition(Definition[C], kind=True): """A codec: what it does to what it is handed, and whether the size of what it gives out is static. A codec handed an array -- array -> array, array -> bytes -- says what @@ -616,7 +618,6 @@ class CodecDefinition(Definition[C]): codec, which is handed bytes -- is refused. """ - is_kind: ClassVar[bool] = True label: ClassVar[str] = "codec" field_aliases: ClassVar[tuple[TypeAliasType, ...]] = (CodecField, StaticCodecField) @@ -644,10 +645,9 @@ def _refusal(self) -> str | None: @dataclass(frozen=True, kw_only=True, slots=True, repr=False) -class StorageTransformerDefinition(Definition[C]): +class StorageTransformerDefinition(Definition[C], kind=True): """A storage transformer.""" - is_kind: ClassVar[bool] = True label: ClassVar[str] = "storage transformer" field_aliases: ClassVar[tuple[TypeAliasType, ...]] = (StorageTransformerField,) @@ -663,19 +663,14 @@ class StorageTransformerDefinition(Definition[C]): def kind_of(definition: Definition[Any]) -> type[Definition[Any]] | None: - """The kind `definition` is: the nearest class in its MRO that declares `is_kind`; None for a definition of no kind, which no scope files.""" - return _kind_in(type(definition)) + """The kind `definition` is: the nearest class in its MRO declared with `kind=True`; None for a definition of no kind, which no scope files.""" + bases: tuple[type, ...] = type(definition).__mro__ + return next((base for base in bases if _declares_kind(base)), None) -def _kind_in(cls: type) -> type[Definition[Any]] | None: - return next( - ( - cast("type[Definition[Any]]", base) - for base in cls.__mro__ - if vars(base).get("is_kind") is True - ), - None, - ) +def _declares_kind(cls: type) -> TypeGuard[type[Definition[Any]]]: + """Whether `cls` itself was declared `kind=True`: the mark is read from its own namespace, which a subclass does not share.""" + return issubclass(cls, Definition) and vars(cls).get("_is_kind") is True def as_kind(kind: object) -> type[Definition[Any]]: @@ -687,12 +682,12 @@ def as_kind(kind: object) -> type[Definition[Any]]: filed: a field read as one would go unjudged. """ origin = get_origin(kind) or kind - if isinstance(origin, type) and vars(origin).get("is_kind") is True: - return cast("type[Definition[Any]]", origin) + if isinstance(origin, type) and _declares_kind(origin): + return origin names = ", ".join(known.__name__ for known in KINDS) msg = ( f"{kind!r} is not a kind of metadata; read a field as one of {names}, or as a " - "subclass of Definition that sets is_kind in its own body" + "subclass of Definition declared with kind=True" ) raise TypeError(msg) diff --git a/packages/zarr-metadata/tests/v3/test_kinds.py b/packages/zarr-metadata/tests/v3/test_kinds.py index 61623ef0f2..fca6a8ed7a 100644 --- a/packages/zarr-metadata/tests/v3/test_kinds.py +++ b/packages/zarr-metadata/tests/v3/test_kinds.py @@ -1,4 +1,4 @@ -"""Kinds of definition: the class that declares `is_kind`, open to kinds of another format.""" +"""Kinds of definition: the class declared with `kind=True`, open to kinds of another format.""" from __future__ import annotations @@ -46,10 +46,9 @@ class MyCodec(CodecDefinition[Any]): @dataclass(frozen=True, kw_only=True, slots=True, repr=False) -class Tag(Definition[C]): +class Tag(Definition[C], kind=True): """A kind of its own, filed apart from every v3 kind.""" - is_kind: ClassVar[bool] = True label: ClassVar[str] = "tag" @@ -59,7 +58,7 @@ class NoKind(Definition[Any]): def test_a_subclass_of_a_kind_is_a_definition_of_that_kind() -> None: - """A definition built as a subclass of `CodecDefinition` is filed, read and named as a codec: the kind is the nearest class in the MRO that declares `is_kind`, not the class of the definition.""" + """A definition built as a subclass of `CodecDefinition` is filed, read and named as a codec: the kind is the nearest class in the MRO declared with `kind=True`, not the class of the definition.""" mine = MyCodec(name="mine", configuration=EmptyConfiguration, kind="bytes_bytes", size="static") assert kind_of(mine) is CodecDefinition scope = Context.of(mine) @@ -68,7 +67,7 @@ def test_a_subclass_of_a_kind_is_a_definition_of_that_kind() -> None: def test_a_kind_of_its_own_is_filed_apart() -> None: - """A class that sets `is_kind` in its body is a kind: `as_kind` accepts it with or without type arguments, a scope files its definitions apart from every other kind's, scopes compare by what each files, and messages name the kind by its label.""" + """A class declared with `kind=True` is a kind: `as_kind` accepts it with or without type arguments, a scope files its definitions apart from every other kind's, scopes compare by what each files, and messages name the kind by its label.""" tag = Tag(name="tag1", configuration=EmptyConfiguration) scope = Context.of(tag, GZIP_CODEC) assert as_kind(Tag) is Tag @@ -81,7 +80,7 @@ def test_a_kind_of_its_own_is_filed_apart() -> None: def test_error_a_class_that_declares_no_kind_is_of_none() -> None: - """A `Definition` subclass that does not set `is_kind` is of no kind: `kind_of` is None, `Context.of` refuses a definition of it, and `as_kind` refuses the class.""" + """A `Definition` subclass declared without `kind=True` is of no kind: `kind_of` is None, `Context.of` refuses a definition of it, and `as_kind` refuses the class.""" none = NoKind(name="nokind", configuration=EmptyConfiguration) assert kind_of(none) is None with pytest.raises(TypeError, match="a definition of no kind"): @@ -98,10 +97,9 @@ class Params(TypedDict, closed=True): @dataclass(frozen=True, kw_only=True, slots=True, repr=False) -class Flat(Definition[C]): +class Flat(Definition[C], kind=True): """A kind whose format writes the parameters beside the name: `{"id": name, **parameters}`.""" - is_kind: ClassVar[bool] = True label: ClassVar[str] = "flat" @classmethod @@ -163,10 +161,9 @@ def test_a_kind_reads_the_envelope_its_format_writes( @dataclass(frozen=True, kw_only=True, slots=True, repr=False) -class Typed(WithFillValue[C]): +class Typed(WithFillValue[C], kind=True): """A kind of another format whose definitions take a fill value.""" - is_kind: ClassVar[bool] = True label: ClassVar[str] = "typed" From ae2597490a22bb6fca43d2e33bd6544c0bb8e0d8 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Thu, 8 Oct 2026 21:52:14 +0200 Subject: [PATCH 84/94] refactor(zarr-metadata)!: name a read field by what it is: AcceptedField, UnclaimedField, RefusedField Read, Unclaimed, Refused and their union Resolved become AcceptedField, UnclaimedField, RefusedField and ResolvedField: the noun says these are metadata fields as a scope read them, and Accepted is the opposite of Refused where Read read like a verb. Assisted-by: ClaudeCode:claude-fable-5-1 Co-Authored-By: Claude Fable 5.1 --- packages/zarr-metadata/README.md | 8 +- .../zarr-metadata/changes/4490.feature.md | 6 +- packages/zarr-metadata/docs/index.md | 8 +- .../src/zarr_metadata/model/_array.py | 50 +++--- .../src/zarr_metadata/model/_group.py | 6 +- .../src/zarr_metadata/model/_validation.py | 36 ++--- .../src/zarr_metadata/v2/definition.py | 26 ++-- .../src/zarr_metadata/v3/_definition.py | 146 +++++++++--------- .../src/zarr_metadata/v3/_pipeline.py | 16 +- .../src/zarr_metadata/v3/_registry.py | 2 +- .../src/zarr_metadata/v3/_scope.py | 26 ++-- .../src/zarr_metadata/v3/codec/cast_value.py | 6 +- .../zarr_metadata/v3/codec/scale_offset.py | 4 +- .../v3/codec/sharding_indexed.py | 4 +- .../src/zarr_metadata/v3/definition.py | 34 ++-- .../zarr-metadata/tests/model/test_array.py | 12 +- .../zarr-metadata/tests/model/test_group.py | 8 +- .../zarr-metadata/tests/model/test_pair.py | 14 +- .../zarr-metadata/tests/model/test_pair_v2.py | 20 +-- .../tests/model/test_pydantic_module.py | 18 ++- .../tests/model/test_read_array_metadata.py | 48 +++--- .../model/test_read_array_metadata_v2.py | 28 ++-- .../zarr-metadata/tests/test_public_api.py | 12 +- .../zarr-metadata/tests/v2/test_codecs.py | 14 +- .../zarr-metadata/tests/v2/test_data_types.py | 14 +- .../zarr-metadata/tests/v2/test_definition.py | 10 +- .../tests/v3/test_definitions.py | 138 +++++++++-------- .../tests/v3/test_every_definition.py | 20 +-- .../tests/v3/test_fill_values.py | 8 +- packages/zarr-metadata/tests/v3/test_kinds.py | 25 +-- .../zarr-metadata/tests/v3/test_pipelines.py | 6 +- packages/zarr-metadata/tests/v3/test_scope.py | 14 +- .../zarr-metadata/tests/v3/test_sharding.py | 8 +- 33 files changed, 412 insertions(+), 383 deletions(-) diff --git a/packages/zarr-metadata/README.md b/packages/zarr-metadata/README.md index 782eed1972..d997d2fb25 100644 --- a/packages/zarr-metadata/README.md +++ b/packages/zarr-metadata/README.md @@ -121,9 +121,9 @@ so a document with a 0 there, as zarr-python 3.0 and 3.1 wrote for an empty dimension, is refused. `read_array_metadata_v3` reads a document once and returns everything -the read found: each field as the scope read it -- `Read` by the -definition that claims its name, `Unclaimed` when none does, or -`Refused` -- with where it sits and the kind it was read as, each codec +the read found: each field as the scope read it -- `AcceptedField` by the +definition that claims its name, `UnclaimedField` when none does, or +`RefusedField` -- with where it sits and the kind it was read as, each codec with the chunk it is handed, every problem, and the model when there is none; `from_json` is that model, or the problems raised. A consumer's own policy is a walk over the fields, with nothing read twice: @@ -179,7 +179,7 @@ scope, `CORE_AND_EXTENSIONS` when none is given, and raises `MetadataValidationError` with every problem, so no model is built invalid; `to_json` writes the document as it was written, and `to_key_value` writes it as it is. Every typed member is a view of that -read: each field as the scope read it, a `Read` or an `Unclaimed`, and +read: each field as the scope read it, an `AcceptedField` or an `UnclaimedField`, and `shape`, `attributes` and the rest as the read refined them, read-only at every level: a list given for an array as a tuple, an object as a read-only mapping; `to_json` gives plain containers. A model is changed diff --git a/packages/zarr-metadata/changes/4490.feature.md b/packages/zarr-metadata/changes/4490.feature.md index 6051b47dfe..851d7c76f3 100644 --- a/packages/zarr-metadata/changes/4490.feature.md +++ b/packages/zarr-metadata/changes/4490.feature.md @@ -1,9 +1,9 @@ **Breaking:** metadata is read against definitions in a scope, and every model is the pair of its document and the scope it was read in. A field (a v3 data type, chunk grid, chunk key encoding, codec or storage -transformer; a v2 dtype or numcodecs codec) is `Read` by the definition -in scope that claims its name, `Unclaimed` when nothing does, or -`Refused` with every problem; `read_array_metadata_v3`, +transformer; a v2 dtype or numcodecs codec) is `AcceptedField` by the definition +in scope that claims its name, `UnclaimedField` when nothing does, or +`RefusedField` with every problem; `read_array_metadata_v3`, `read_group_metadata_v3`, `read_node_metadata_v3` and `read_array_metadata_v2` read a document once and return everything the read found, the model among it. A model is built from a document in a diff --git a/packages/zarr-metadata/docs/index.md b/packages/zarr-metadata/docs/index.md index 7ee50fb213..ef986e3867 100644 --- a/packages/zarr-metadata/docs/index.md +++ b/packages/zarr-metadata/docs/index.md @@ -123,9 +123,9 @@ so a document with a 0 there, as zarr-python 3.0 and 3.1 wrote for an empty dimension, is refused. `read_array_metadata_v3` reads a document once and returns everything -the read found: each field as the scope read it -- `Read` by the -definition that claims its name, `Unclaimed` when none does, or -`Refused` -- with where it sits and the kind it was read as, each codec +the read found: each field as the scope read it -- `AcceptedField` by the +definition that claims its name, `UnclaimedField` when none does, or +`RefusedField` -- with where it sits and the kind it was read as, each codec with the chunk it is handed, every problem, and the model when there is none; `from_json` is that model, or the problems raised. A consumer's own policy is a walk over the fields, with nothing read twice: @@ -181,7 +181,7 @@ scope, `CORE_AND_EXTENSIONS` when none is given, and raises `MetadataValidationError` with every problem, so no model is built invalid; `to_json` writes the document as it was written, and `to_key_value` writes it as it is. Every typed member is a view of that -read: each field as the scope read it, a `Read` or an `Unclaimed`, and +read: each field as the scope read it, an `AcceptedField` or an `UnclaimedField`, and `shape`, `attributes` and the rest as the read refined them, read-only at every level: a list given for an array as a tuple, an object as a read-only mapping; `to_json` gives plain containers. A model is changed diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_array.py b/packages/zarr-metadata/src/zarr_metadata/model/_array.py index 2fbb1b9b92..9ccd732ef5 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_array.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/_array.py @@ -47,13 +47,13 @@ from zarr_metadata.v2.definition import CORE_V2, resolve_dtype_v2 from zarr_metadata.v3._common import ZarrV3MetadataFieldJSON from zarr_metadata.v3._definition import ( + AcceptedField, ChunkGridDefinition, ChunkKeyEncodingDefinition, CodecDefinition, DataTypeDefinition, - Read, StorageTransformerDefinition, - Unclaimed, + UnclaimedField, field_key, fill_value_problems, held, @@ -71,7 +71,7 @@ from zarr_metadata.v2._definition import ZarrV2CodecDefinition, ZarrV2DataTypeDefinition from zarr_metadata.v2.array import ZarrV2ArrayMetadataJSON, ZarrV2ArrayMetadataStoreKey from zarr_metadata.v2.attributes import ZarrV2AttributesStoreKey - from zarr_metadata.v3._definition import Resolved + from zarr_metadata.v3._definition import ResolvedField from zarr_metadata.v3.array import ( ZarrV3ArrayMetadataJSON, ZarrV3ArrayMetadataJSONPartial, @@ -124,8 +124,8 @@ class ZarrV3ArrayMetadata(Keyed): -- arrays as tuples, string keys -- and `context` the scope. Every typed member is a view of the reading the pair gives: `data_type`, `chunk_grid`, `chunk_key_encoding`, each codec and storage transformer - as the scope read it, `Read` by the definition that claims its name or - `Unclaimed`; `shape`, `fill_value`, `dimension_names`, `attributes` and + as the scope read it, `AcceptedField` by the definition that claims its name or + `UnclaimedField`; `shape`, `fill_value`, `dimension_names`, `attributes` and `extra_fields` as the read refined them. Built only by reading: the constructor reads `document` in `context` and raises `MetadataValidationError` with every problem, so no model is invalid. @@ -241,7 +241,7 @@ def _key_of(self) -> tuple[object, ...]: ) def _plain_key( - self, data_type: Read[DataTypeDefinition[Any]] | Unclaimed + self, data_type: AcceptedField[DataTypeDefinition[Any]] | UnclaimedField ) -> tuple[object, ...]: """What `refines` compares of a model other than its fields, the fill value spelled as `data_type` -- the more informed side's -- spells it.""" members = self._members @@ -256,7 +256,7 @@ def _plain_key( def _fill_value_key(self) -> str: """What `==` compares of the fill value: its canonical spelling as JSON text when a definition in scope read the data type, and the fill value as written when none did.""" fill_value = self._members.fill_value - if isinstance(self.data_type, Read): + if isinstance(self.data_type, AcceptedField): return json_text(spelled_canonically(self.data_type, fill_value)) return json_text(fill_value) @@ -305,29 +305,29 @@ def must_understand_fields(self) -> dict[str, ZarrV3ExtensionField]: return must_understand_subset(self.extra_fields) @property - def data_type(self) -> Read[DataTypeDefinition[Any]] | Unclaimed: + def data_type(self) -> AcceptedField[DataTypeDefinition[Any]] | UnclaimedField: """The data type, as the scope read it.""" return held(self._reading.data_type) @property - def chunk_grid(self) -> Read[ChunkGridDefinition[Any]] | Unclaimed: + def chunk_grid(self) -> AcceptedField[ChunkGridDefinition[Any]] | UnclaimedField: """The chunk grid, as the scope read it.""" return held(self._reading.chunk_grid) @property - def chunk_key_encoding(self) -> Read[ChunkKeyEncodingDefinition[Any]] | Unclaimed: + def chunk_key_encoding(self) -> AcceptedField[ChunkKeyEncodingDefinition[Any]] | UnclaimedField: """The chunk key encoding, as the scope read it.""" return held(self._reading.chunk_key_encoding) @property - def codecs(self) -> tuple[Read[CodecDefinition[Any]] | Unclaimed, ...]: + def codecs(self) -> tuple[AcceptedField[CodecDefinition[Any]] | UnclaimedField, ...]: """The codecs, each as the scope read it, in pipeline order.""" return tuple(held(stage.codec) for stage in self._reading.pipeline) @property def storage_transformers( self, - ) -> tuple[Read[StorageTransformerDefinition[Any]] | Unclaimed, ...]: + ) -> tuple[AcceptedField[StorageTransformerDefinition[Any]] | UnclaimedField, ...]: """The storage transformers, each as the scope read it.""" return tuple(held(entry) for entry in self._reading.storage_transformers) @@ -457,7 +457,7 @@ def from_key_value( def located_conflicts( - fields: Iterable[tuple[Loc, Resolved[Any]]], conflicts: Sequence[Conflict] + fields: Iterable[tuple[Loc, ResolvedField[Any]]], conflicts: Sequence[Conflict] ) -> tuple[Conflict, ...]: """Each of `conflicts`, found against a reading's claims, once for each place among `fields` the name it is about sits: located, as a problem is.""" located: list[Conflict] = [] @@ -476,8 +476,8 @@ def read_array_metadata_v3( """`value`, a v3 array document, as `context` read it, whatever it holds. Everything a read finds, in one: each extension point as `context` - read it -- `Read` by the definition that claims its name, `Unclaimed`, - or `Refused` -- the chunks the codecs are handed, each codec with the + read it -- `AcceptedField` by the definition that claims its name, `UnclaimedField`, + or `RefusedField` -- the chunks the codecs are handed, each codec with the chunk it is handed, every problem `validate_array_metadata_v3` finds, and, when there is none, the document's model, holding the same reading. A policy over the fields, the core spec's alone, say, is a @@ -499,8 +499,8 @@ def read_array_metadata_v2( """`value`, a v2 array document, as `context` read it, `CORE_V2` when none is given, whatever it holds. Everything a read finds, in one: the dtype, the compressor and each - filter as the scope read them -- `Read` by the definition that claims - the typestr or id, `Unclaimed`, or `Refused` -- every problem + filter as the scope read them -- `AcceptedField` by the definition that claims + the typestr or id, `UnclaimedField`, or `RefusedField` -- every problem `validate_array_metadata_v2` finds, and, when there is none, the document's model. """ @@ -542,8 +542,8 @@ class ZarrV2ArrayMetadata(Keyed): `dimension_separator` means `"."` by the v2 convention (https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v2/v2.0.rst#L81-L86), which the model holds and writes. `dtype`, `compressor` and each - filter are views of the reading: `Read` by the definition in scope - that claims the typestr or id, or `Unclaimed`. `attributes` is + filter are views of the reading: `AcceptedField` by the definition in scope + that claims the typestr or id, or `UnclaimedField`. `attributes` is `UNSET` when no `.zattrs` exists, distinct from an empty one. A member the spec does not define is kept, in `extra_fields`. Built only by reading: the constructor reads `document` in `context` and @@ -668,7 +668,7 @@ def _key_of(self) -> tuple[object, ...]: ) def _plain_key( - self, dtype: Read[ZarrV2DataTypeDefinition[Any]] | Unclaimed + self, dtype: AcceptedField[ZarrV2DataTypeDefinition[Any]] | UnclaimedField ) -> tuple[object, ...]: """What `refines` compares of a model other than its fields, the fill value spelled as `dtype` -- the more informed side's -- spells it.""" members = self._members @@ -685,7 +685,7 @@ def _plain_key( def _fill_value_key(self) -> str: """What `==` compares of the fill value: its canonical spelling as JSON text when a definition in scope read the dtype, and the fill value as written when none did.""" fill_value = self._members.fill_value - if isinstance(self.dtype, Read): + if isinstance(self.dtype, AcceptedField): return json_text(spelled_canonically(self.dtype, fill_value)) return json_text(fill_value) @@ -730,18 +730,20 @@ def extra_fields(self) -> Mapping[str, JSONValue]: return self._shown[2] @property - def dtype(self) -> Read[ZarrV2DataTypeDefinition[Any]] | Unclaimed: + def dtype(self) -> AcceptedField[ZarrV2DataTypeDefinition[Any]] | UnclaimedField: """The dtype as the scope read it: by its family's definition, or unclaimed.""" return held(self._reading.dtype) @property - def compressor(self) -> Read[ZarrV2CodecDefinition[Any]] | Unclaimed | None: + def compressor(self) -> AcceptedField[ZarrV2CodecDefinition[Any]] | UnclaimedField | None: """The compressor as the scope read it; None when written as `null`.""" compressor = self._reading.compressor return None if compressor is None else held(compressor) @property - def filters(self) -> tuple[Read[ZarrV2CodecDefinition[Any]] | Unclaimed, ...] | None: + def filters( + self, + ) -> tuple[AcceptedField[ZarrV2CodecDefinition[Any]] | UnclaimedField, ...] | None: """The filters, each as the scope read it; None when written as `null`.""" filters = self._reading.filters if filters is None: diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_group.py b/packages/zarr-metadata/src/zarr_metadata/model/_group.py index 60cfea4f47..2f2acc5279 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_group.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/_group.py @@ -82,7 +82,7 @@ from zarr_metadata.v2.attributes import ZarrV2AttributesStoreKey from zarr_metadata.v2.consolidated import ZarrV2ConsolidatedMetadataStoreKey from zarr_metadata.v2.group import ZarrV2GroupMetadataJSON, ZarrV2GroupMetadataStoreKey - from zarr_metadata.v3._definition import Definition, Resolved + from zarr_metadata.v3._definition import Definition, ResolvedField from zarr_metadata.v3.array import ZarrV3ArrayMetadataJSON from zarr_metadata.v3.consolidated import ZarrV3ConsolidatedMetadataJSON from zarr_metadata.v3.group import ZarrV3GroupMetadataJSONPartial, ZarrV3GroupMetadataStoreKey @@ -505,7 +505,7 @@ class ZarrV3GroupMetadataReading: metadata: ZarrV3GroupMetadata | None = None """The document's model when there is no problem; None otherwise.""" - def fields(self) -> Iterator[tuple[Loc, Resolved[Any]]]: + def fields(self) -> Iterator[tuple[Loc, ResolvedField[Any]]]: """Each field of each document its consolidated metadata holds, as read, with where it sits in this document.""" for path, reading in self.consolidated.items(): for loc, node in reading.fields(): @@ -547,7 +547,7 @@ def metadata(self) -> None: """Its model: none, since no node type says which model it is.""" return None - def fields(self) -> Iterator[tuple[Loc, Resolved[Any]]]: + def fields(self) -> Iterator[tuple[Loc, ResolvedField[Any]]]: """Its fields as read: none, since none of them is read.""" return iter(()) diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_validation.py b/packages/zarr-metadata/src/zarr_metadata/model/_validation.py index 833aca3bbd..c59e5c8f72 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_validation.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/_validation.py @@ -74,7 +74,7 @@ DataTypeDefinition, Definition, Lengths, - Resolved, + ResolvedField, StorageTransformerDefinition, chunk_grid_lengths, field_kind, @@ -394,17 +394,17 @@ class ZarrV3ArrayMetadataReading: not hold as a list is empty. """ - data_type: Resolved[DataTypeDefinition[Any]] | UNSET = UNSET + data_type: ResolvedField[DataTypeDefinition[Any]] | UNSET = UNSET """The data type, as the scope read it.""" - chunk_grid: Resolved[ChunkGridDefinition[Any]] | UNSET = UNSET + chunk_grid: ResolvedField[ChunkGridDefinition[Any]] | UNSET = UNSET """The chunk grid, as the scope read it.""" - chunk_key_encoding: Resolved[ChunkKeyEncodingDefinition[Any]] | UNSET = UNSET + chunk_key_encoding: ResolvedField[ChunkKeyEncodingDefinition[Any]] | UNSET = UNSET """The chunk key encoding, as the scope read it.""" chunk: Chunk = dataclasses.field(default_factory=Chunk) """The chunks the codecs are handed: the lengths the grid's chunks take along each axis of the shape, of the data type.""" pipeline: tuple[Stage, ...] = () """The codecs, read as a pipeline: each as the scope read it, with the chunk it is handed.""" - storage_transformers: tuple[Resolved[StorageTransformerDefinition[Any]], ...] = () + storage_transformers: tuple[ResolvedField[StorageTransformerDefinition[Any]], ...] = () """The storage transformers, each as the scope read it.""" problems: tuple[ValidationProblem, ...] = () """Every reason the document is not a valid one.""" @@ -419,7 +419,7 @@ def __reduce__(self) -> str | tuple[object, ...]: return (reading_of, (self.metadata,)) return object.__reduce__(self) - def fields(self) -> Iterator[tuple[Loc, Resolved[Any]]]: + def fields(self) -> Iterator[tuple[Loc, ResolvedField[Any]]]: """Each field the document holds, as the scope read it, with where it sits in the document. The extension points, then each codec and storage transformer at its @@ -427,7 +427,7 @@ def fields(self) -> Iterator[tuple[Loc, Resolved[Any]]]: them: a shard's codecs, a struct's field types. `with_problems` gives each with its problems. """ - own: tuple[tuple[str, Resolved[Any] | UNSET], ...] = ( + own: tuple[tuple[str, ResolvedField[Any] | UNSET], ...] = ( ("data_type", self.data_type), ("chunk_grid", self.chunk_grid), ("chunk_key_encoding", self.chunk_key_encoding), @@ -485,11 +485,11 @@ class ZarrV2ArrayMetadataReading: `filters` written as `null` is None. """ - dtype: Resolved[ZarrV2DataTypeDefinition[Any]] | UNSET = UNSET + dtype: ResolvedField[ZarrV2DataTypeDefinition[Any]] | UNSET = UNSET """The dtype, as the scope read it.""" - compressor: Resolved[ZarrV2CodecDefinition[Any]] | UNSET | None = UNSET + compressor: ResolvedField[ZarrV2CodecDefinition[Any]] | UNSET | None = UNSET """The compressor, as the scope read it; None when written as `null`.""" - filters: tuple[Resolved[ZarrV2CodecDefinition[Any]], ...] | UNSET | None = UNSET + filters: tuple[ResolvedField[ZarrV2CodecDefinition[Any]], ...] | UNSET | None = UNSET """The filters, each as the scope read it; None when written as `null`.""" problems: tuple[ValidationProblem, ...] = () """Every reason the document is not a valid one.""" @@ -502,7 +502,7 @@ def __reduce__(self) -> str | tuple[object, ...]: return (reading_of, (self.metadata,)) return object.__reduce__(self) - def fields(self) -> Iterator[tuple[Loc, Resolved[Any]]]: + def fields(self) -> Iterator[tuple[Loc, ResolvedField[Any]]]: """Each field the document holds, as the scope read it, where it sits: the dtype, a struct's record types after it, the compressor, each filter at its index.""" if self.dtype is not UNSET: yield from fields_of(self.dtype, ("dtype",)) @@ -519,7 +519,7 @@ def read_array_v3( """`value`, a v3 array document, as `context` read it, without its model, and its other members refined; None when it has a problem. `read_array_metadata_v3` builds the model from the two. A field object - in the document -- a `Read` built by hand -- is not JSON, and is refused + in the document -- an `AcceptedField` built by hand -- is not JSON, and is refused as such. `at` is where the document sits in the one handed in -- a document consolidated metadata holds sits three levels below its group's -- so the levels a reader walks are counted from that one's root; the problems are located in this document. @@ -547,7 +547,7 @@ def read_array_v3( # configuration, against the definition in `context` that claims its # name. `must_understand: false` keeps its meaning where it has one: an # unknown top-level extension *field*, which a reader really can skip. - read: dict[str, Resolved[Any]] = {} + read: dict[str, ResolvedField[Any]] = {} for key, kind in _EXTENSION_POINTS_V3: if key in doc: read[key], found = resolve(doc[key], kind, context, (*at, key)) @@ -568,7 +568,7 @@ def read_array_v3( if "chunk_grid" in read and shape is not None: lengths, found = chunk_grid_lengths(read["chunk_grid"], shape, ("chunk_grid",)) problems.extend(found) - listed: dict[str, list[Resolved[Any]]] = {} + listed: dict[str, list[ResolvedField[Any]]] = {} for key, kind in _EXTENSION_LISTS_V3: if key in doc: entries = doc[key] @@ -723,20 +723,20 @@ def read_array_v2( "invalid_value", ) ) - dtype: Resolved[ZarrV2DataTypeDefinition[Any]] | UNSET = UNSET + dtype: ResolvedField[ZarrV2DataTypeDefinition[Any]] | UNSET = UNSET if "dtype" in doc: # A typestr by its family, field records as a struct. dtype, found = resolve_dtype_v2(doc["dtype"], context, ("dtype",)) problems.extend(found) if "order" in doc and doc["order"] not in ("C", "F"): problems.append(outside_of(("order",), doc["order"], ("C", "F"))) - compressor: Resolved[ZarrV2CodecDefinition[Any]] | UNSET | None = UNSET + compressor: ResolvedField[ZarrV2CodecDefinition[Any]] | UNSET | None = UNSET if "compressor" in doc: compressor = None if doc["compressor"] is not None: compressor, found = resolve_codec_v2(doc["compressor"], context, ("compressor",)) problems.extend(found) - filters: tuple[Resolved[ZarrV2CodecDefinition[Any]], ...] | UNSET | None = UNSET + filters: tuple[ResolvedField[ZarrV2CodecDefinition[Any]], ...] | UNSET | None = UNSET if "filters" in doc: filters = None entries = doc["filters"] @@ -752,7 +752,7 @@ def read_array_v2( else: # "A list of JSON objects providing codec configurations, or # null" (https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v2/v2.0.rst#L76-L79): an empty list is a list. - read: list[Resolved[ZarrV2CodecDefinition[Any]]] = [] + read: list[ResolvedField[ZarrV2CodecDefinition[Any]]] = [] for index, item in enumerate(entries): entry, found = resolve_codec_v2(item, context, ("filters", index)) read.append(entry) diff --git a/packages/zarr-metadata/src/zarr_metadata/v2/definition.py b/packages/zarr-metadata/src/zarr_metadata/v2/definition.py index cd7862b5e2..08230a180e 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v2/definition.py +++ b/packages/zarr-metadata/src/zarr_metadata/v2/definition.py @@ -5,8 +5,8 @@ has a scope all the same: the data types zarr-python 2.x writes, one definition per NumPy family, and the codecs numcodecs 0.16 configures, one per id this package models. `CORE_V2` is that scope. Read one field -in it with `resolve_dtype_v2` or `resolve_codec_v2`, which give `Read`, -`Unclaimed` or `Refused` as `zarr_metadata.v3.definition.resolve` does +in it with `resolve_dtype_v2` or `resolve_codec_v2`, which give `AcceptedField`, +`UnclaimedField` or `RefusedField` as `zarr_metadata.v3.definition.resolve` does for a v3 field; the scope algebra -- `Context.of`, `extended_with`, `joined`, `claimant` -- is the same `Context`. @@ -15,7 +15,7 @@ and its simplest spelling is the typestr again. A records array reads as `struct`, each record's type a nested field. A codec reads by its id, its other members the parameters. A typestr of a type code the spec does -not list, or a codec id the package does not model, is `Unclaimed`. +not list, or a codec id the package does not model, is `UnclaimedField`. """ from __future__ import annotations @@ -30,10 +30,10 @@ from zarr_metadata.v2.codec import V2_CODECS from zarr_metadata.v2.data_type import V2_DATA_TYPES from zarr_metadata.v3._definition import ( - Read, - Refused, - Resolved, - Unclaimed, + AcceptedField, + RefusedField, + ResolvedField, + UnclaimedField, canonical_fill_value, fill_value_problems, resolve, @@ -57,14 +57,14 @@ def resolve_dtype_v2( value: object, context: Context | None = None, loc: Loc = () -) -> tuple[Resolved[ZarrV2DataTypeDefinition[Any]], Problems]: +) -> tuple[ResolvedField[ZarrV2DataTypeDefinition[Any]], Problems]: """`value`, a v2 `dtype`, read in `context`, `CORE_V2` when none is given: what the scope made of it, and every problem, each prefixed with `loc`.""" return resolve(value, ZarrV2DataTypeDefinition, CORE_V2 if context is None else context, loc) def resolve_codec_v2( value: object, context: Context | None = None, loc: Loc = () -) -> tuple[Resolved[ZarrV2CodecDefinition[Any]], Problems]: +) -> tuple[ResolvedField[ZarrV2CodecDefinition[Any]], Problems]: """`value`, a v2 `compressor` or one of its `filters`, read in `context`, `CORE_V2` when none is given: what the scope made of it, and every problem, each prefixed with `loc`.""" return resolve(value, ZarrV2CodecDefinition, CORE_V2 if context is None else context, loc) @@ -73,16 +73,16 @@ def resolve_codec_v2( "CORE_V2", "V2_CODECS", "V2_DATA_TYPES", + "AcceptedField", "ClaimKey", "Claims", "Conflict", "Context", "Disagreements", - "Read", - "Refused", - "Resolved", + "RefusedField", + "ResolvedField", "ScopeConflictError", - "Unclaimed", + "UnclaimedField", "ZarrV2CodecDefinition", "ZarrV2DataTypeDefinition", "ZarrV2DataTypeField", diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py b/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py index b89d928182..3b6b112367 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py @@ -664,13 +664,13 @@ class StorageTransformerDefinition(Definition[C], kind=True): def kind_of(definition: Definition[Any]) -> type[Definition[Any]] | None: """The kind `definition` is: the nearest class in its MRO declared with `kind=True`; None for a definition of no kind, which no scope files.""" - bases: tuple[type, ...] = type(definition).__mro__ + bases: tuple[type[object], ...] = type(definition).__mro__ return next((base for base in bases if _declares_kind(base)), None) -def _declares_kind(cls: type) -> TypeGuard[type[Definition[Any]]]: +def _declares_kind(cls: type[object]) -> TypeGuard[type[Definition[Any]]]: """Whether `cls` itself was declared `kind=True`: the mark is read from its own namespace, which a subclass does not share.""" - return issubclass(cls, Definition) and vars(cls).get("_is_kind") is True + return vars(cls).get("_is_kind") is True and issubclass(cls, Definition) def as_kind(kind: object) -> type[Definition[Any]]: @@ -1111,7 +1111,7 @@ def _nothing_nested() -> Nested: @dataclass(frozen=True, slots=True, kw_only=True) -class Read(Generic[D]): +class AcceptedField(Generic[D]): """A field a definition in scope read: the name it is written with, the definition, and the configuration it allowed. A problem with the envelope around it -- a stray member, a @@ -1186,7 +1186,7 @@ def to_json(self) -> JSONValue: @dataclass(frozen=True, slots=True, kw_only=True) -class Unclaimed: +class UnclaimedField: """A field nothing in scope claims: an extension the scope leaves unjudged, which is what keeps the format open. Equal to another when it is written with the same name and @@ -1237,12 +1237,12 @@ def nested(self) -> Nested: return _nothing_nested() def to_json(self) -> JSONValue: - """The field as a document writes it, sharing nothing with the field: its configuration as written, in the envelope every reader takes, as `Read.to_json` writes one.""" + """The field as a document writes it, sharing nothing with the field: its configuration as written, in the envelope every reader takes, as `AcceptedField.to_json` writes one.""" return copied(written_json(self)) @dataclass(frozen=True, slots=True, kw_only=True) -class Refused(Generic[D]): +class RefusedField(Generic[D]): """A field that could not be read -- not a field at all, not JSON, or refused by the definition that claims its name -- as its problems say.""" json: JSONValue | UNSET @@ -1275,10 +1275,12 @@ def __post_init__(self) -> None: raise TypeError(refusal) -Resolved = TypeAliasType("Resolved", "Read[D] | Unclaimed | Refused[D]", type_params=(D,)) +ResolvedField = TypeAliasType( + "ResolvedField", "AcceptedField[D] | UnclaimedField | RefusedField[D]", type_params=(D,) +) """One metadata field as a scope read it: read by the definition that claims its name, claimed by nothing, or refused.""" -_FIELDS: Final = (Read, Unclaimed, Refused) +_FIELDS: Final = (AcceptedField, UnclaimedField, RefusedField) """The three things a scope makes of a field.""" @@ -1292,7 +1294,7 @@ def _misread(definition: object, kind: type[Definition[Any]], name: object) -> s return None -def field_key(field: Resolved[Any]) -> tuple[object, ...]: +def field_key(field: ResolvedField[Any]) -> tuple[object, ...]: """What `==` and `hash` compare of a field: what it means, not how it was spelled. A field read compares by the definition that read it, its @@ -1304,7 +1306,7 @@ def field_key(field: Resolved[Any]) -> tuple[object, ...]: fields are one when what a reader understands of them reads the same, and what none interprets is written alike. """ - if isinstance(field, Read): + if isinstance(field, AcceptedField): definition = cast("Definition[Any]", field.definition) configuration: JSONValue = dict(field.configuration) # A field it holds is compared by its own key, so its place holds @@ -1320,7 +1322,7 @@ def field_key(field: Resolved[Any]) -> tuple[object, ...]: lambda: json_text(cast("JSONValue", dict(definition.canonical(view)))), ) return ("read", definition, spelled, _nested_key(field.nested)) - if isinstance(field, Unclaimed): + if isinstance(field, UnclaimedField): return ("unclaimed", field.read_as, field.name, json_text(field.configuration)) return ( "refused", @@ -1332,7 +1334,7 @@ def field_key(field: Resolved[Any]) -> tuple[object, ...]: ) -def own_key(field: Read[Any]) -> tuple[object, ...]: +def own_key(field: AcceptedField[Any]) -> tuple[object, ...]: """What `field_key` compares of a read field without the fields it holds: the definition and the canonical spelling of its own members.""" return field_key(field)[:3] @@ -1342,26 +1344,26 @@ def _nested_key(nested: Nested) -> tuple[tuple[Loc, tuple[object, ...]], ...]: return tuple((loc, field_key(inner)) for loc, inner in nested.items()) -def document_json(field: Resolved[Any]) -> JSONValue | UNSET: +def document_json(field: ResolvedField[Any]) -> JSONValue | UNSET: """A field as a document writes it, holding the field's own values: as `to_json` writes it, or as it was written when it was refused, `UNSET` for one that was not JSON. What a writer serializes, which changes nothing, so it copies nothing; `to_json` is this, copied. """ - if isinstance(field, Refused): + if isinstance(field, RefusedField): return field.json return written_json(field) -def written_json(field: Read[Any] | Unclaimed) -> JSONValue: +def written_json(field: AcceptedField[Any] | UnclaimedField) -> JSONValue: """A field a scope read or left unclaimed, as a document writes it: `document_json` of one that is JSON.""" kind = field.read_as - if isinstance(field, Read) and kind.spelled(field.name)[1] is not None: + if isinstance(field, AcceptedField) and kind.spelled(field.name)[1] is not None: return kind.envelope_json(field.name, {}) return kind.envelope_json(field.name, field.configuration) -Nested: TypeAlias = Mapping[Loc, Resolved[Any]] +Nested: TypeAlias = Mapping[Loc, ResolvedField[Any]] """The fields a configuration holds, each as the scope read it, by where it sits in the configuration.""" @@ -1383,7 +1385,7 @@ class Chunk: lengths: Lengths | None = None """Per axis, the lengths the chunks take along it; None when not even the number of axes is known.""" - data_type: Resolved[DataTypeDefinition[Any]] | None = None + data_type: ResolvedField[DataTypeDefinition[Any]] | None = None """The data type field of the values, as a scope read it; None when no field says what they are: a document naming none, which its reading holds as `UNSET`, hands the pipeline a chunk of no known type.""" def __post_init__(self) -> None: @@ -1402,14 +1404,14 @@ def rank(self) -> int | None: return None if self.lengths is None else len(self.lengths) -def is_field(value: object) -> TypeGuard[Resolved[Any]]: - """Whether `value` is a field a scope read: `Read`, `Unclaimed` or `Refused`.""" +def is_field(value: object) -> TypeGuard[ResolvedField[Any]]: + """Whether `value` is a field a scope read: `AcceptedField`, `UnclaimedField` or `RefusedField`.""" return isinstance(value, _FIELDS) -def held(field: Resolved[D] | UNSET) -> Read[D] | Unclaimed: - """`field`, one a read found nothing wrong with: `Read` or `Unclaimed`; `TypeError` for one refused or never read, which such a read rules out.""" - if isinstance(field, (Read, Unclaimed)): +def held(field: ResolvedField[D] | UNSET) -> AcceptedField[D] | UnclaimedField: + """`field`, one a read found nothing wrong with: `AcceptedField` or `UnclaimedField`; `TypeError` for one refused or never read, which such a read rules out.""" + if isinstance(field, (AcceptedField, UnclaimedField)): return field msg = f"expected a field a read found nothing wrong with, got {field!r}" raise TypeError(msg) @@ -1435,7 +1437,9 @@ def _is_lengths(value: object) -> TypeGuard[Lengths]: return True -def fields_of(resolved: Resolved[Any], loc: Loc = ()) -> Iterator[tuple[Loc, Resolved[Any]]]: +def fields_of( + resolved: ResolvedField[Any], loc: Loc = () +) -> Iterator[tuple[Loc, ResolvedField[Any]]]: """`resolved`, a field a scope read, where it sits, then each field it holds and theirs in turn, each where it sits. `loc` is where `resolved` sits; a field it holds sits in its @@ -1449,8 +1453,8 @@ def fields_of(resolved: Resolved[Any], loc: Loc = ()) -> Iterator[tuple[Loc, Res def with_problems( - fields: Iterable[tuple[Loc, Resolved[Any]]], problems: Sequence[ValidationProblem] -) -> Iterator[tuple[Loc, Resolved[Any], Problems]]: + fields: Iterable[tuple[Loc, ResolvedField[Any]]], problems: Sequence[ValidationProblem] +) -> Iterator[tuple[Loc, ResolvedField[Any], Problems]]: """Each of `fields`, with where it sits, and the problems among `problems` located in it, in the fields it holds too. `fields` and `problems` are one read's: a reading's `fields()` and @@ -1477,20 +1481,20 @@ def with_problems( yield loc, field, tuple(held[loc]) -def configuration_of(resolved: Resolved[Any], definition: Definition[C]) -> C | None: +def configuration_of(resolved: ResolvedField[Any], definition: Definition[C]) -> C | None: """The configuration `resolved` holds, typed as `definition` declares it, if `definition` read it. - `Read` holds a configuration as the mapping every one is; asked with + `AcceptedField` holds a configuration as the mapping every one is; asked with the definition that read the field -- or one equal to it, as a pickled field's is -- this is the same mapping, as its TypedDict. None when another definition read it, or none did. """ - if not isinstance(resolved, Read) or resolved.definition != definition: + if not isinstance(resolved, AcceptedField) or resolved.definition != definition: return None return cast("C", resolved.configuration) -def fill_value_problems(data_type: Resolved[F], value: object, loc: Loc = ()) -> Problems: +def fill_value_problems(data_type: ResolvedField[F], value: object, loc: Loc = ()) -> Problems: """What is wrong with `value` as a fill value of `data_type`, a data type field a scope read. `value` is refined to JSON first: not JSON is the first verdict, @@ -1504,7 +1508,7 @@ def fill_value_problems(data_type: Resolved[F], value: object, loc: Loc = ()) -> value unjudged. `loc` prefixes every problem. """ refined, problems = refine_json(value, loc) - if len(problems) != 0 or not isinstance(data_type, Read): + if len(problems) != 0 or not isinstance(data_type, AcceptedField): return with_input(problems, value, loc) definition, configuration = data_type.definition, data_type.configuration typed, problems = _fill_value_parser(definition.fill_value)(refined, loc) @@ -1518,7 +1522,7 @@ def fill_value_problems(data_type: Resolved[F], value: object, loc: Loc = ()) -> return with_input((*problems, *refused), value, loc) -def canonical_fill_value(data_type: Resolved[F], value: object) -> JSONValue | UNSET: +def canonical_fill_value(data_type: ResolvedField[F], value: object) -> JSONValue | UNSET: """`value`, a fill value of `data_type`, a data type field a scope read, in the one spelling its value has; `UNSET` when it has a problem. As the data type's `fill_value_canonical` spells it, so two fill @@ -1536,14 +1540,14 @@ def canonical_fill_value(data_type: Resolved[F], value: object) -> JSONValue | U return spelled_canonically(data_type, refined) -def spelled_canonically(data_type: Resolved[F], value: JSONValue) -> JSONValue: +def spelled_canonically(data_type: ResolvedField[F], value: JSONValue) -> JSONValue: """`value`, a fill value of `data_type` with no problem, in its canonical spelling, as `canonical_fill_value` gives it, without judging it again. Its `fill_value_canonical` is the extension author's code: what it gives is checked to be JSON, and an error it raises says which data type's canonical spelling raised it. """ - if not isinstance(data_type, Read): + if not isinstance(data_type, AcceptedField): return value definition, configuration = data_type.definition, data_type.configuration spelled = asked( @@ -1564,7 +1568,7 @@ def spelled_canonically(data_type: Resolved[F], value: JSONValue) -> JSONValue: return refined -def storage_of(data_type: Resolved[DataTypeDefinition[Any]]) -> StorageClass | None: +def storage_of(data_type: ResolvedField[DataTypeDefinition[Any]]) -> StorageClass | None: """How the values of `data_type`, a data type field a scope read, are stored; None when unknown. Unknown when the scope did not read it, or its definition does not @@ -1572,7 +1576,7 @@ def storage_of(data_type: Resolved[DataTypeDefinition[Any]]) -> StorageClass | N checked to be a storage class, and an error it raises says which data type's storage raised it. """ - if not isinstance(data_type, Read): + if not isinstance(data_type, AcceptedField): return None definition, configuration = data_type.definition, data_type.configuration found = asked( @@ -1590,7 +1594,7 @@ def storage_of(data_type: Resolved[DataTypeDefinition[Any]]) -> StorageClass | N def chunk_grid_lengths( - chunk_grid: Resolved[ChunkGridDefinition[Any]], shape: tuple[int, ...], loc: Loc = () + chunk_grid: ResolvedField[ChunkGridDefinition[Any]], shape: tuple[int, ...], loc: Loc = () ) -> tuple[Lengths, Problems]: """The lengths the chunks of `chunk_grid`, a chunk grid field a scope read, take along each axis of an array of `shape`, and what is wrong with the grid over it. @@ -1608,7 +1612,7 @@ def chunk_grid_lengths( definition, not the field. """ unknown: Lengths = (None,) * len(shape) - if not isinstance(chunk_grid, Read): + if not isinstance(chunk_grid, AcceptedField): return unknown, () definition, configuration = chunk_grid.definition, chunk_grid.configuration at = (*loc, "configuration") @@ -1644,7 +1648,7 @@ def chunk_grid_lengths( def resolve( data: object, kind: type[D], context: Context, loc: Loc = () -) -> tuple[Resolved[D], Problems]: +) -> tuple[ResolvedField[D], Problems]: """`data`, one metadata field, read as a `kind` in `context`: what the scope made of it, and every problem. All three steps for one field. `data` is refined to JSON and its @@ -1654,10 +1658,10 @@ def resolve( configuration is checked against its TypedDict and judged by its rules; each nested field the check met is read the same way, in the same scope, and what is wrong with one is its own, reported where it - sits, as with a document's fields. What comes back is `Read` by the - definition that claims the name; `Unclaimed` when nothing in scope + sits, as with a document's fields. What comes back is `AcceptedField` by the + definition that claims the name; `UnclaimedField` when nothing in scope claims it, an unmodelled extension left unjudged, which is what keeps - the format open; or `Refused`, with the problems that say why. `loc` + the format open; or `RefusedField`, with the problems that say why. `loc` prefixes every problem. `kind` is one of the five kinds -- `CodecDefinition`, `DataTypeDefinition`, `ChunkGridDefinition`, `ChunkKeyEncodingDefinition`, `StorageTransformerDefinition` -- with @@ -1672,16 +1676,16 @@ def resolve( name = asked.named_configuration(data)[0] bad = None if name is None else asked.name_problem(name, asked.name_loc(loc)) claimant = None if name is None or bad is not None else context.claimant(asked, name) - refused = Refused(json=UNSET, name=name, read_as=asked, definition=claimant) + refused = RefusedField(json=UNSET, name=name, read_as=asked, definition=claimant) found = problems if bad is None else (bad, *problems) - return cast("Resolved[D]", refused), with_input(found, data, loc) + return cast("ResolvedField[D]", refused), with_input(found, data, loc) resolved, found = _resolve_field(refined, asked, context, loc) - return cast("Resolved[D]", resolved), with_input(found, data, loc) + return cast("ResolvedField[D]", resolved), with_input(found, data, loc) def _resolve_field( data: JSONValue, kind: type[Definition[Any]], context: Context, loc: Loc -) -> tuple[Resolved[Definition[Any]], Problems]: +) -> tuple[ResolvedField[Definition[Any]], Problems]: """A refined field with its envelope judged, then read. What the scope made of it is what became of the configuration. A @@ -1696,35 +1700,35 @@ def _resolve_field( def _read( data: JSONValue, kind: type[Definition[Any]], context: Context, loc: Loc -) -> tuple[Resolved[Definition[Any]], Problems]: +) -> tuple[ResolvedField[Definition[Any]], Problems]: name, given, malformed = kind.named_configuration(data) if name is None: - return Refused(json=data, name=None, read_as=kind), () + return RefusedField(json=data, name=None, read_as=kind), () if not kind.well_named(name): # The envelope rule every reader runs first reports it; no # definition is asked to claim it. - return Refused(json=data, name=name, read_as=kind), () + return RefusedField(json=data, name=name, read_as=kind), () definition = context.claimant(kind, name) if len(malformed) != 0: # A configuration that is not an object, which the envelope's # problems say; the name still says what claims the field. - return Refused(json=data, name=name, read_as=kind, definition=definition), () + return RefusedField(json=data, name=name, read_as=kind, definition=definition), () if definition is None: - return Unclaimed(json=data, name=name, read_as=kind), () + return UnclaimedField(json=data, name=name, read_as=kind), () _, carried = kind.spelled(name) if carried is not None: return _read_carried(data, name, kind, definition, given, carried, loc) at = kind.configuration_loc(loc) if given is None and definition.requires_configuration: missing = problem(at, f"{name!r} requires a configuration", "missing_key") - return Refused(json=data, name=name, read_as=kind, definition=definition), missing + return RefusedField(json=data, name=name, read_as=kind, definition=definition), missing typed, found, nested = _typed(definition.configuration, {} if given is None else given, at) # The fields it holds are read first, each a frame deeper than this # one, so the rules see them as the scope read them; each is put back # as a document writes it, so the configuration says what was read # however each was spelled. Their problems are reported after the # rules'. - within: dict[Loc, Resolved[Any]] = {} + within: dict[Loc, ResolvedField[Any]] = {} written: dict[Loc, JSONValue] = {} inside: list[ValidationProblem] = [] for field in nested: @@ -1751,9 +1755,11 @@ def _read( ruled(definition, lambda: definition.rules(read_only(configuration), within), at) ) if configuration is None or not _usable(own): - refused = Refused(json=data, name=name, read_as=kind, definition=definition, nested=within) + refused = RefusedField( + json=data, name=name, read_as=kind, definition=definition, nested=within + ) return refused, (*own, *inside) - read = Read( + read = AcceptedField( json=data, name=name, definition=definition, configuration=configuration, nested=within ) return read, (*own, *inside) @@ -1773,7 +1779,7 @@ def _read_carried( given: Mapping[str, object] | None, carried: Mapping[str, JSONValue], loc: Loc, -) -> tuple[Resolved[Definition[Any]], Problems]: +) -> tuple[ResolvedField[Definition[Any]], Problems]: """A field whose name carries its configuration -- raw bits, `r16` -- read by the definition its name is filed under. The document wrote a name, so what is wrong with what the name carries @@ -1798,12 +1804,14 @@ def _read_carried( # declare cannot be left out, as a stray key beside the name can, so # anything wrong with what the name carries refuses the field. if configuration is None or not _usable(beside) or len(judged) != 0: - refused = Refused(json=data, name=name, read_as=kind, definition=definition) + refused = RefusedField(json=data, name=name, read_as=kind, definition=definition) return refused, problems - return Read(json=data, name=name, definition=definition, configuration=configuration), problems + return AcceptedField( + json=data, name=name, definition=definition, configuration=configuration + ), problems -def _sized(field: _NestedField, inner: Resolved[Any]) -> Problems: +def _sized(field: _NestedField, inner: ResolvedField[Any]) -> Problems: """A codec of dynamic size in a member that takes codecs of static size, as a problem at the field. A name nothing in scope claims is left unjudged, its size unknown, as @@ -1852,7 +1860,7 @@ def canonicalize( def canonical_of( - resolved: Resolved[Any], problems: Sequence[ValidationProblem] + resolved: ResolvedField[Any], problems: Sequence[ValidationProblem] ) -> JSONValue | None: """`resolved`, a field a scope read, in its simplest equivalent spelling, as `canonicalize` spells one; None when it has a problem. @@ -1868,16 +1876,16 @@ def canonical_of( return _simplest(resolved) -def _simplest(field: Resolved[Any]) -> JSONValue | None: +def _simplest(field: ResolvedField[Any]) -> JSONValue | None: """A field in its simplest spelling; None when it, or a field it holds, was refused, which has none.""" - if isinstance(field, Read): + if isinstance(field, AcceptedField): return _canonical_field(field) - if isinstance(field, Unclaimed): + if isinstance(field, UnclaimedField): return field.to_json() return None -def _canonical_field(resolved: Read[Any]) -> JSONValue | None: +def _canonical_field(resolved: AcceptedField[Any]) -> JSONValue | None: """A field that read, in its simplest equivalent spelling: the fields it holds first, then its own members; None when one it holds was refused.""" definition, name = resolved.definition, resolved.name configuration: JSONValue = dict(resolved.configuration) @@ -1924,6 +1932,7 @@ def _replaced(value: JSONValue, path: Loc, new: JSONValue) -> JSONValue: "KINDS", "RAW_BYTES_NAME", "RAW_BYTES_NAME_PATTERN", + "AcceptedField", "Chunk", "ChunkGridDefinition", "ChunkGridField", @@ -1939,14 +1948,13 @@ def _replaced(value: JSONValue, path: Loc, new: JSONValue) -> JSONValue: "EmptyConfiguration", "Lengths", "Nested", - "Read", - "Refused", - "Resolved", + "RefusedField", + "ResolvedField", "StaticCodecField", "StorageClass", "StorageTransformerDefinition", "StorageTransformerField", - "Unclaimed", + "UnclaimedField", "as_kind", "asked", "canonical_fill_value", diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/_pipeline.py b/packages/zarr-metadata/src/zarr_metadata/v3/_pipeline.py index 2417f0f557..ec7eb3274b 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/_pipeline.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/_pipeline.py @@ -31,11 +31,11 @@ from zarr_metadata._json import ValidationProblem, is_object, is_tuple, with_input from zarr_metadata.v3._definition import ( + AcceptedField, Chunk, CodecDefinition, CodecKind, - Read, - Resolved, + ResolvedField, asked, read_only, ruled, @@ -57,7 +57,7 @@ def _no_stages() -> Mapping[str, tuple[Stage, ...]]: class Stage: """One codec of a pipeline, and the chunk it is handed.""" - codec: Resolved[CodecDefinition[Any]] + codec: ResolvedField[CodecDefinition[Any]] """The codec, as the scope read it.""" incoming: Chunk | None """The chunk it is handed. @@ -85,7 +85,7 @@ class Stage: def read_pipeline( - codecs: Sequence[Resolved[CodecDefinition[Any]]], chunk: Chunk, loc: Loc = () + codecs: Sequence[ResolvedField[CodecDefinition[Any]]], chunk: Chunk, loc: Loc = () ) -> tuple[tuple[Stage, ...], Problems]: """Each of `codecs`, codec fields a scope read, as a pipeline handed `chunk`, with the chunk it is handed, and what is wrong with them. @@ -115,7 +115,7 @@ def read_pipeline( incoming = Chunk() if handed is None else handed at = (*loc, index, "configuration") inner = _no_stages() - if isinstance(codec, Read): + if isinstance(codec, AcceptedField): configuration, nested = codec.configuration, codec.nested problems.extend(_chunk_problems(definition, configuration, nested, incoming, at)) inner, found = _inner_pipelines(definition, configuration, nested, incoming, at) @@ -123,7 +123,7 @@ def read_pipeline( stages.append(Stage(codec, incoming, inner)) if definition.kind == "array_bytes": handed = None - elif not isinstance(codec, Read): + elif not isinstance(codec, AcceptedField): handed = Chunk() else: handed = _handed_on(definition, codec.configuration, codec.nested, incoming, at) @@ -133,7 +133,7 @@ def read_pipeline( def _order_problems( - codecs: Sequence[Resolved[CodecDefinition[Any]]], loc: Loc + codecs: Sequence[ResolvedField[CodecDefinition[Any]]], loc: Loc ) -> Iterator[ValidationProblem]: """Array -> array codecs, then one array -> bytes codec, then bytes -> bytes codecs. @@ -233,7 +233,7 @@ def _is_pipelines(value: object) -> TypeGuard[Mapping[str, Chunk]]: def _held( definition: CodecDefinition[Any], configuration: Mapping[str, Any], nested: Nested, member: str -) -> tuple[Resolved[CodecDefinition[Any]], ...]: +) -> tuple[ResolvedField[CodecDefinition[Any]], ...]: """The codecs `member` of the configuration holds, as the scope read them. A member holding anything but a list of fields read as codecs is a diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/_registry.py b/packages/zarr-metadata/src/zarr_metadata/v3/_registry.py index 16341ea6a6..e01c4edf49 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/_registry.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/_registry.py @@ -11,7 +11,7 @@ uses nothing an implementation could refuse for being optional. `CORE_AND_EXTENSIONS` adds what `zarr-extensions` registers and this package defines. A name in neither is not refused -- that is what keeps -the format open -- it is read as `Unclaimed` and left unjudged. +the format open -- it is read as `UnclaimedField` and left unjudged. """ from __future__ import annotations diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/_scope.py b/packages/zarr-metadata/src/zarr_metadata/v3/_scope.py index 1488cb0f51..03214cecff 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/_scope.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/_scope.py @@ -16,10 +16,10 @@ from typing import TYPE_CHECKING, Any, TypeAlias from zarr_metadata.v3._definition import ( + AcceptedField, Definition, - Read, - Refused, - Unclaimed, + RefusedField, + UnclaimedField, field_key, own_key, spelled, @@ -29,7 +29,7 @@ from collections.abc import Callable, Iterable, Sequence from zarr_metadata._typed_json import Loc - from zarr_metadata.v3._definition import Resolved + from zarr_metadata.v3._definition import ResolvedField ClaimKey: TypeAlias = tuple[type[Definition[Any]], str] """A kind and the name a definition is filed under: what a scope answers `claimant` for.""" @@ -70,7 +70,7 @@ def __init__(self, conflicts: Sequence[Conflict]) -> None: super().__init__("; ".join(str(conflict) for conflict in self.conflicts)) -def claim_key(field: Resolved[Any]) -> ClaimKey | None: +def claim_key(field: ResolvedField[Any]) -> ClaimKey | None: """The key `field` is claimed under: its kind and the name its definition is filed under, `r*` for `r16`; None for a field that names nothing.""" if field.name is None: return None @@ -79,7 +79,7 @@ def claim_key(field: Resolved[Any]) -> ClaimKey | None: def claims_of( - fields: Iterable[tuple[Loc, Resolved[Any]]], + fields: Iterable[tuple[Loc, ResolvedField[Any]]], ) -> dict[ClaimKey, Definition[Any] | None]: """What `fields`, each with where it sits, claim of each name: the definition that read it, None where nothing claimed it. @@ -103,7 +103,7 @@ def claims_of( return claims -def refines(field: Resolved[Any], other: Resolved[Any]) -> bool: +def refines(field: ResolvedField[Any], other: ResolvedField[Any]) -> bool: """Whether `field` holds everything `other` holds: reads the same where both read, and reads what `other` left unclaimed. The order one reading of a document refines another in. A name nothing @@ -114,13 +114,13 @@ def refines(field: Resolved[Any], other: Resolved[Any]) -> bool: definitions is a conflict; a refused field refines itself alone. Two fields that refine each other are equal. """ - if isinstance(field, Refused) or isinstance(other, Refused): + if isinstance(field, RefusedField) or isinstance(other, RefusedField): return field == other - if isinstance(other, Unclaimed): - if isinstance(field, Unclaimed): + if isinstance(other, UnclaimedField): + if isinstance(field, UnclaimedField): return field_key(field) == field_key(other) return claim_key(field) == claim_key(other) and _as_unclaimed(field) == other - if isinstance(field, Unclaimed): + if isinstance(field, UnclaimedField): return False if field.definition != other.definition or own_key(field) != own_key(other): return False @@ -129,9 +129,9 @@ def refines(field: Resolved[Any], other: Resolved[Any]) -> bool: return all(refines(field.nested[loc], other.nested[loc]) for loc in field.nested) -def _as_unclaimed(field: Read[Any]) -> Unclaimed: +def _as_unclaimed(field: AcceptedField[Any]) -> UnclaimedField: """`field` as it would have been read had nothing claimed its name: what a gain is compared against.""" - return Unclaimed(json=field.json, name=field.name, read_as=field.read_as) + return UnclaimedField(json=field.json, name=field.name, read_as=field.read_as) @dataclass(frozen=True, slots=True) diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/codec/cast_value.py b/packages/zarr-metadata/src/zarr_metadata/v3/codec/cast_value.py index 1580074b76..921f05132f 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/codec/cast_value.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/codec/cast_value.py @@ -12,11 +12,11 @@ from zarr_metadata._common import JSONValue from zarr_metadata._json import ValidationProblem, shown from zarr_metadata.v3._definition import ( + AcceptedField, Chunk, CodecDefinition, DataTypeField, Nested, - Read, fill_value_problems, ) from zarr_metadata.v3.codec._arithmetic import COMPLEX, FLOATING_POINT, NOT_NUMBERS @@ -148,7 +148,7 @@ def _rules( models no real numbers is the one problem reported of it. """ target = nested.get(("data_type",)) - if not isinstance(target, Read): + if not isinstance(target, AcceptedField): return name, written = target.definition.name, target.name if name in _NO_REAL_NUMBERS: @@ -179,7 +179,7 @@ def _chunk_rules( so the data type it is handed is held to what the one it casts to is. """ source = chunk.data_type - if not isinstance(source, Read): + if not isinstance(source, AcceptedField): return if source.definition.name in _NO_REAL_NUMBERS: yield ValidationProblem( diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/codec/scale_offset.py b/packages/zarr-metadata/src/zarr_metadata/v3/codec/scale_offset.py index 72e6d50e14..46af385734 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/codec/scale_offset.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/codec/scale_offset.py @@ -12,10 +12,10 @@ from zarr_metadata._common import JSONValue from zarr_metadata._json import ValidationProblem, shown from zarr_metadata.v3._definition import ( + AcceptedField, Chunk, CodecDefinition, Nested, - Read, fill_value_problems, ) from zarr_metadata.v3.codec._arithmetic import NOT_NUMBERS @@ -95,7 +95,7 @@ def _chunk_rules( A null is the rules' to refuse. """ source = chunk.data_type - if not isinstance(source, Read): + if not isinstance(source, AcceptedField): return if source.definition.name in NOT_NUMBERS: yield ValidationProblem( diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/codec/sharding_indexed.py b/packages/zarr-metadata/src/zarr_metadata/v3/codec/sharding_indexed.py index 7402857221..e701c7d9de 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/codec/sharding_indexed.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/codec/sharding_indexed.py @@ -12,12 +12,12 @@ from zarr_metadata._json import ValidationProblem from zarr_metadata.v3._definition import ( + AcceptedField, Chunk, CodecDefinition, CodecField, Lengths, Nested, - Read, StaticCodecField, ) from zarr_metadata.v3.data_type.uint64 import UINT64_DATA_TYPE, UINT64_DATA_TYPE_NAME @@ -116,7 +116,7 @@ def _chunk_rules( ) -_UINT64: Final = Read( +_UINT64: Final = AcceptedField( json=UINT64_DATA_TYPE_NAME, name=UINT64_DATA_TYPE_NAME, definition=UINT64_DATA_TYPE, diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/definition.py b/packages/zarr-metadata/src/zarr_metadata/v3/definition.py index a1bd7905d7..75fc61b13b 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/definition.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/definition.py @@ -41,17 +41,17 @@ field in a scope: its envelope judged, its name related to a definition, its configuration judged, and each nested field read the same way. It returns what the scope made of the field, and every - problem: `Read` by the definition that claims its name, with the + problem: `AcceptedField` by the definition that claims its name, with the configuration it checked and allowed and the fields it holds, each - read the same way; `Unclaimed`, a name nothing in scope claims, left - unjudged, which is what keeps the format open; or `Refused`, whose + read the same way; `UnclaimedField`, a name nothing in scope claims, left + unjudged, which is what keeps the format open; or `RefusedField`, whose problems say why. Each has the field's JSON, the `name` it is written with, the kind it was read as, `read_as`, the `definition` - that claims its name -- None for `Unclaimed` -- and the fields it - holds as the scope read them, `nested`. The two a model holds, `Read` - and `Unclaimed`, have a `configuration` and `to_json()`, the field as + that claims its name -- None for `UnclaimedField` -- and the fields it + holds as the scope read them, `nested`. The two a model holds, `AcceptedField` + and `UnclaimedField`, have a `configuration` and `to_json()`, the field as a document writes it; two of them are equal when they read the same, - however each was spelled. `Resolved` is the three, for `match`. A + however each was spelled. `ResolvedField` is the three, for `match`. A field is read as one of the five kinds, with or without type arguments; `resolve(field, Definition, scope)` is a `TypeError`, since nothing @@ -64,7 +64,7 @@ resolved, problems = resolve({"name": "gzip", "configuration": {"level": 12}}, CodecDefinition, CORE_AND_EXTENSIONS) - resolved # Refused(..., definition=CodecDefinition(name='gzip'), ...) + resolved # RefusedField(..., definition=CodecDefinition(name='gzip'), ...) problems[0].loc # ('configuration', 'level') problems[0].input # 12 dict(problems[0].ctx) # {'ge': 0, 'le': 9} @@ -92,7 +92,7 @@ 728's, which `typing.TypedDict` does not take on the versions this package supports. The rules are handed the configuration and the fields it holds as the scope read them: a field that is read keeps what it read -inside it as `Read.nested`, a `Nested` mapping by where each sits, so a +inside it as `AcceptedField.nested`, a `Nested` mapping by where each sits, so a struct's rules reach its field types. `judge`, which reads in no scope, hands them none. A rule's message shows a value as the package's own messages do, as JSON, with `shown`: `null`, `[1, 2]`, `"C"`. A rule @@ -320,6 +320,7 @@ def acme_lz4_rules( StorageTransformerField, ) from zarr_metadata.v3._definition import ( + AcceptedField, Chunk, ChunkGridDefinition, ChunkKeyEncodingDefinition, @@ -331,12 +332,11 @@ def acme_lz4_rules( EmptyConfiguration, Lengths, Nested, - Read, - Refused, - Resolved, + RefusedField, + ResolvedField, StorageClass, StorageTransformerDefinition, - Unclaimed, + UnclaimedField, canonical_fill_value, fill_value_problems, resolve, @@ -355,6 +355,7 @@ def acme_lz4_rules( __all__ = [ "CORE", "CORE_AND_EXTENSIONS", + "AcceptedField", "Chunk", "ChunkGridDefinition", "ChunkGridField", @@ -379,16 +380,15 @@ def acme_lz4_rules( "MetadataValidationError", "Nested", "ProblemKind", - "Read", - "Refused", - "Resolved", + "RefusedField", + "ResolvedField", "ScopeConflictError", "Stage", "StaticCodecField", "StorageClass", "StorageTransformerDefinition", "StorageTransformerField", - "Unclaimed", + "UnclaimedField", "ValidationProblem", "canonical_fill_value", "fill_value_problems", diff --git a/packages/zarr-metadata/tests/model/test_array.py b/packages/zarr-metadata/tests/model/test_array.py index f9d667d471..f8b4590bd6 100644 --- a/packages/zarr-metadata/tests/model/test_array.py +++ b/packages/zarr-metadata/tests/model/test_array.py @@ -58,8 +58,8 @@ from zarr_metadata.v3.definition import ( CORE, CORE_AND_EXTENSIONS, - Read, - Unclaimed, + AcceptedField, + UnclaimedField, ) if TYPE_CHECKING: @@ -450,12 +450,12 @@ def test_update_reads_every_member_in_the_models_own_scope() -> None: little: ZarrV3NamedConfigJSON = {"name": "bytes", "configuration": {"endian": "little"}} zstd: ZarrV3NamedConfigJSON = {"name": "zstd", "configuration": {"level": 3, "checksum": False}} base = ZarrV3ArrayMetadata.create_default(context=CORE, codecs=(little, zstd)) - assert isinstance(base.codecs[1], Unclaimed) + assert isinstance(base.codecs[1], UnclaimedField) kept = base.update(attributes={"k": 1}) assert kept.codecs[1] == base.codecs[1] assert kept.context == CORE given = base.with_context(CORE_AND_EXTENSIONS).update(codecs=(little, zstd)) - assert isinstance(given.codecs[1], Read) + assert isinstance(given.codecs[1], AcceptedField) def test_update_leaves_out_a_member_given_as_unset() -> None: @@ -2393,10 +2393,10 @@ def test_error_a_document_nested_deeper_than_a_reader_walks_is_a_problem() -> No def test_error_a_field_object_in_a_document_is_not_json() -> None: - # A `Read` built by hand, with a configuration its definition refuses, + # A `AcceptedField` built by hand, with a configuration its definition refuses, # smuggled into a document: refused as what it is, so nothing built by # hand passes as read. A model holds its own fields as read. - smuggled = Read( + smuggled = AcceptedField( json="gzip", name="gzip", definition=GZIP_CODEC, configuration={"level": 99, "window": 1} ) document = { diff --git a/packages/zarr-metadata/tests/model/test_group.py b/packages/zarr-metadata/tests/model/test_group.py index 07e24b8d36..7521ca10e9 100644 --- a/packages/zarr-metadata/tests/model/test_group.py +++ b/packages/zarr-metadata/tests/model/test_group.py @@ -60,8 +60,8 @@ CodecDefinition, EmptyConfiguration, Nested, - Refused, - Unclaimed, + RefusedField, + UnclaimedField, resolve, ) from zarr_metadata.v3.group import ZarrV3GroupMetadataJSONPartial @@ -667,7 +667,7 @@ def test_group_update_reads_the_documents_it_is_given_in_its_scope() -> None: assert updated.consolidated_metadata is not UNSET child = updated.consolidated_metadata.metadata["a"] assert isinstance(child, ZarrV3ArrayMetadata) - assert isinstance(child.codecs[1], Unclaimed) + assert isinstance(child.codecs[1], UnclaimedField) removed = updated.update(consolidated_metadata=UNSET) assert removed.consolidated_metadata is UNSET @@ -955,7 +955,7 @@ def test_error_a_group_document_with_a_problem_reads_as_no_model( assert { path: read.metadata is not None for path, read in reading.consolidated.items() } == models - assert [loc for loc, field in reading.fields() if isinstance(field, Refused)] == refused + assert [loc for loc, field in reading.fields() if isinstance(field, RefusedField)] == refused # --- read_node_metadata_v3 ------------------------------------------------- diff --git a/packages/zarr-metadata/tests/model/test_pair.py b/packages/zarr-metadata/tests/model/test_pair.py index 6edfb19303..c74409cea0 100644 --- a/packages/zarr-metadata/tests/model/test_pair.py +++ b/packages/zarr-metadata/tests/model/test_pair.py @@ -28,12 +28,12 @@ from zarr_metadata.v3.definition import ( CORE, CORE_AND_EXTENSIONS, + AcceptedField, CodecDefinition, Context, Nested, - Read, ScopeConflictError, - Unclaimed, + UnclaimedField, ValidationProblem, ) @@ -96,8 +96,8 @@ def test_to_json_keeps_the_spelling_the_document_was_written_in( def test_properties_are_what_the_reading_holds() -> None: """The typed members -- fields as the scope read them, shape, fill value, attributes -- are views of the reading, read-only.""" model = ZarrV3ArrayMetadata({**ARRAY, "attributes": {"a": [1]}, "acme": 1}) - assert isinstance(model.data_type, Read) - assert isinstance(model.codecs[0], Read) + assert isinstance(model.data_type, AcceptedField) + assert isinstance(model.codecs[0], AcceptedField) assert model.shape == (4,) assert model.fill_value == 0 assert model.dimension_names is UNSET @@ -109,7 +109,7 @@ def test_properties_are_what_the_reading_holds() -> None: model.attributes["b"] = 1 # pyright: ignore[reportIndexIssue] with pytest.raises(AttributeError): model.shape = (5,) # pyright: ignore[reportAttributeAccessIssue] - assert isinstance(ZarrV3ArrayMetadata(ARRAY, context=Context.of()).data_type, Unclaimed) + assert isinstance(ZarrV3ArrayMetadata(ARRAY, context=Context.of()).data_type, UnclaimedField) def test_a_reading_without_problems_builds_the_model_without_reading_again() -> None: @@ -299,7 +299,7 @@ def test_with_context_reads_the_document_in_any_scope() -> None: private = model.with_context(PRIVATE) assert private.context == PRIVATE assert private.codecs[0].definition == MY_BYTES - assert isinstance(model.with_context(Context.of()).codecs[0], Unclaimed) + assert isinstance(model.with_context(Context.of()).codecs[0], UnclaimedField) assert model.with_context(None).context == CORE_AND_EXTENSIONS @@ -444,7 +444,7 @@ def test_group_update_with_context_and_refined_in_behave_as_the_arrays_do() -> N assert isinstance(held, ZarrV3ConsolidatedMetadata) array = held.metadata["a"] assert isinstance(array, ZarrV3ArrayMetadata) - assert isinstance(array.codecs[0], Read) + assert isinstance(array.codecs[0], AcceptedField) with pytest.raises(ScopeConflictError): ZarrV3GroupMetadata(CONSOLIDATED).refined_in(PRIVATE) private = group.with_context(PRIVATE).consolidated_metadata diff --git a/packages/zarr-metadata/tests/model/test_pair_v2.py b/packages/zarr-metadata/tests/model/test_pair_v2.py index 1b1a29de4a..96fec0df10 100644 --- a/packages/zarr-metadata/tests/model/test_pair_v2.py +++ b/packages/zarr-metadata/tests/model/test_pair_v2.py @@ -21,10 +21,10 @@ from zarr_metadata.v2.data_type.scalar import FLOAT_V2, UINT_V2 from zarr_metadata.v2.definition import ( CORE_V2, + AcceptedField, Context, - Read, ScopeConflictError, - Unclaimed, + UnclaimedField, ZarrV2CodecDefinition, ZarrV2DataTypeDefinition, ) @@ -72,12 +72,12 @@ def test_a_model_is_its_document_read_in_its_scope() -> None: def test_properties_are_what_the_reading_holds() -> None: """`dtype`, `compressor` and `filters` are the fields as the scope read them, None where `null` is written; `shape`, `chunks`, `fill_value`, `order`, `dimension_separator`, `attributes` and `extra_fields` are the members as the read refined them, read-only; `claims` is keyed as the scope files them.""" model = ZarrV2ArrayMetadata({**ARRAY, "filters": [{"id": "x"}], "extra": [1]}) - assert isinstance(model.dtype, Read) + assert isinstance(model.dtype, AcceptedField) assert model.dtype.definition is FLOAT_V2 - assert isinstance(model.compressor, Read) + assert isinstance(model.compressor, AcceptedField) assert model.compressor.definition is ZLIB_V2 assert model.filters is not None - assert isinstance(model.filters[0], Unclaimed) + assert isinstance(model.filters[0], UnclaimedField) members = (model.shape, model.chunks, model.fill_value, model.order, model.dimension_separator) assert members == ((4,), (2,), 0, "C", ".") assert model.attributes == {"a": (1, 2)} @@ -199,9 +199,9 @@ def test_error_update_refuses_a_document_with_a_problem(changes: dict[str, Any], def test_with_context_and_refined_in_read_the_document_in_another_scope() -> None: """`with_context` reads the document in any scope (a loss is allowed), `refined_in` only up the order: a scope that claims what this one left unclaimed is a gain, one that reads a name by another definition or by none is a `ScopeConflictError` naming where the name sits, and a gain that surfaces a problem is a `MetadataValidationError`.""" unclaimed = ZarrV2ArrayMetadata({**ARRAY, "compressor": {"id": "zlib"}}, SMALL) - assert isinstance(unclaimed.compressor, Unclaimed) + assert isinstance(unclaimed.compressor, UnclaimedField) gained = unclaimed.refined_in(PRIVATE) - assert isinstance(gained.compressor, Read) + assert isinstance(gained.compressor, AcceptedField) assert gained.compressor.definition is BARE_ZLIB assert unclaimed.refines(unclaimed) assert gained.refines(unclaimed) @@ -210,7 +210,7 @@ def test_with_context_and_refined_in_read_the_document_in_another_scope() -> Non gained.refined_in(CORE_V2) assert [c.loc for c in conflict.value.conflicts] == [("compressor",)] lost = gained.with_context(SMALL) - assert isinstance(lost.compressor, Unclaimed) + assert isinstance(lost.compressor, UnclaimedField) assert lost == unclaimed assert gained.with_context(PRIVATE) == gained with pytest.raises(MetadataValidationError): @@ -341,10 +341,10 @@ def test_consolidated_metadata_is_equal_by_its_nodes_and_moves_scope_with_them() assert ZarrV2ConsolidatedMetadata(CONSOLIDATED) != ZarrV2ConsolidatedMetadata(other) small = ZarrV2ConsolidatedMetadata(CONSOLIDATED, SMALL) assert isinstance(small.nodes["a"], ZarrV2ArrayMetadata) - assert isinstance(small.nodes["a"].compressor, Unclaimed) + assert isinstance(small.nodes["a"].compressor, UnclaimedField) gained = small.refined_in(PRIVATE) assert isinstance(gained.nodes["a"], ZarrV2ArrayMetadata) - assert isinstance(gained.nodes["a"].compressor, Read) + assert isinstance(gained.nodes["a"].compressor, AcceptedField) assert gained.refines(small) assert not small.refines(gained) with pytest.raises(ScopeConflictError) as conflict: diff --git a/packages/zarr-metadata/tests/model/test_pydantic_module.py b/packages/zarr-metadata/tests/model/test_pydantic_module.py index 5b15ddbf64..ca843a46a9 100644 --- a/packages/zarr-metadata/tests/model/test_pydantic_module.py +++ b/packages/zarr-metadata/tests/model/test_pydantic_module.py @@ -29,7 +29,7 @@ from zarr_metadata.v3.codec.gzip import GZIP_CODEC from zarr_metadata.v3.definition import ( CORE, - Read, + AcceptedField, ) V3_ARRAY_DOC = dict(ZarrV3ArrayMetadata.create_default(shape=(4,)).to_json()) @@ -320,7 +320,7 @@ def test_a_v3_field_type_reads_in_the_scope_the_validation_context_holds( ) -> None: """As pydantic hands any validator its context; `zstd` is an extension, which `CORE` leaves unclaimed.""" model = TypeAdapter(zmp.ZarrV3ArrayMetadata).validate_python(_WITH_ZSTD, context=context) - assert isinstance(model.codecs[1], Read) is read + assert isinstance(model.codecs[1], AcceptedField) is read def test_error_a_validation_context_holds_a_scope_that_is_not_one() -> None: @@ -390,9 +390,11 @@ def test_a_ctx_member_named_message_yields_to_a_message_holding_a_placeholder() def test_a_line_error_for_what_a_problem_could_not_hold_reports_what_sits_there() -> None: # A problem holds no input for what is not JSON a reader walks -- a - # `Read` built by hand among the codecs -- and the line error reports + # `AcceptedField` built by hand among the codecs -- and the line error reports # that object, where a missing key's reports the object missing it. - smuggled = Read(json="gzip", name="gzip", definition=GZIP_CODEC, configuration={"level": 1}) + smuggled = AcceptedField( + json="gzip", name="gzip", definition=GZIP_CODEC, configuration={"level": 1} + ) doc = { **V3_ARRAY_DOC, "codecs": [{"name": "bytes", "configuration": {"endian": "little"}}, smuggled], @@ -444,7 +446,7 @@ def test_the_pydantic_schema_names_an_extension_as_the_reader_does(field: object def test_v2_field_types_read_in_the_validation_contexts_scope() -> None: """The v2 field types read a document in the scope the validation context holds -- itself a `Context`, or its `zarr_metadata_context` item -- and in `CORE_V2` when it holds none, as the v3 field types read in theirs.""" from zarr_metadata.v2.data_type.scalar import UINT_V2 - from zarr_metadata.v2.definition import CORE_V2, Context, Unclaimed + from zarr_metadata.v2.definition import CORE_V2, Context, UnclaimedField adapter = TypeAdapter(zmp.ZarrV2ArrayMetadata) doc = json.loads(json.dumps(V2_ARRAY_DOC)) @@ -452,7 +454,7 @@ def test_v2_field_types_read_in_the_validation_contexts_scope() -> None: small = Context.of(UINT_V2) read = adapter.validate_python({**doc, "compressor": {"id": "zlib"}}, context=small) assert read.context is small - assert isinstance(read.compressor, Unclaimed) + assert isinstance(read.compressor, UnclaimedField) held = adapter.validate_python(doc, context={"zarr_metadata_context_v2": small}) assert held.context is small group = TypeAdapter(zmp.ZarrV2GroupMetadata).validate_python(V2_GROUP_DOC, context=small) @@ -462,7 +464,7 @@ def test_v2_field_types_read_in_the_validation_contexts_scope() -> None: def test_each_format_reads_in_its_own_context_key() -> None: """A mapping context names each format's scope by its own key -- `zarr_metadata_context` for v3, `zarr_metadata_context_v2` for v2 -- so a v3 scope given for the v3 fields leaves the v2 fields in `CORE_V2`; a bare `Context` is the scope of every field type.""" from zarr_metadata.v2.data_type.scalar import UINT_V2 - from zarr_metadata.v2.definition import CORE_V2, Context, Unclaimed + from zarr_metadata.v2.definition import CORE_V2, Context, UnclaimedField from zarr_metadata.v3.definition import CORE_AND_EXTENSIONS adapter = TypeAdapter(zmp.ZarrV2ArrayMetadata) @@ -478,5 +480,5 @@ def test_each_format_reads_in_its_own_context_key() -> None: assert own.context is small bare = adapter.validate_python({**doc, "compressor": {"id": "zlib"}}, context=small) assert bare.context is small - assert isinstance(bare.compressor, Unclaimed) + assert isinstance(bare.compressor, UnclaimedField) assert zmp.CONTEXT_KEY_V2 == "zarr_metadata_context_v2" diff --git a/packages/zarr-metadata/tests/model/test_read_array_metadata.py b/packages/zarr-metadata/tests/model/test_read_array_metadata.py index 2d4c50e8d4..6c0555c585 100644 --- a/packages/zarr-metadata/tests/model/test_read_array_metadata.py +++ b/packages/zarr-metadata/tests/model/test_read_array_metadata.py @@ -28,31 +28,31 @@ from zarr_metadata.v3.definition import ( CORE, CORE_AND_EXTENSIONS, + AcceptedField, ChunkGridDefinition, ChunkKeyEncodingDefinition, CodecDefinition, Context, DataTypeDefinition, Definition, - Read, - Refused, + RefusedField, StorageTransformerDefinition, - Unclaimed, + UnclaimedField, resolve, ) if TYPE_CHECKING: from collections.abc import Callable, Iterator - from zarr_metadata.v3.definition import Lengths, Loc, Resolved + from zarr_metadata.v3.definition import Lengths, Loc, ResolvedField LITTLE = {"name": "bytes", "configuration": {"endian": "little"}} ZSTD = {"name": "zstd", "configuration": {"level": 1}} POINTS: list[tuple[Loc, type[Definition[Any]], type]] = [ - (("data_type",), DataTypeDefinition, Read), - (("chunk_grid",), ChunkGridDefinition, Read), - (("chunk_key_encoding",), ChunkKeyEncodingDefinition, Read), + (("data_type",), DataTypeDefinition, AcceptedField), + (("chunk_grid",), ChunkGridDefinition, AcceptedField), + (("chunk_key_encoding",), ChunkKeyEncodingDefinition, AcceptedField), ] """A default document's single extension points, each read where it sits.""" @@ -63,7 +63,7 @@ def _document(shape: tuple[int, ...] = (4,), **fields: object) -> dict[str, Any] return cast("dict[str, Any]", arrays_to_tuples(document)) -def _codec(loc: Loc, variant: type = Read) -> tuple[Loc, type[Definition[Any]], type]: +def _codec(loc: Loc, variant: type = AcceptedField) -> tuple[Loc, type[Definition[Any]], type]: return (loc, CodecDefinition, variant) @@ -104,7 +104,7 @@ def _shard(chunk_shape: list[int], codecs: list[object] | None = None, **more: o *POINTS, _codec(("codecs", 0)), _codec(("codecs", 0, "configuration", "codecs", 0)), - _codec(("codecs", 0, "configuration", "codecs", 1), Unclaimed), + _codec(("codecs", 0, "configuration", "codecs", 1), UnclaimedField), _codec(("codecs", 0, "configuration", "index_codecs", 0)), _codec(("codecs", 0, "configuration", "index_codecs", 1)), ], @@ -140,9 +140,9 @@ def _shard(chunk_shape: list[int], codecs: list[object] | None = None, **more: o CORE, [ *POINTS, - _codec(("codecs", 0), Refused), + _codec(("codecs", 0), RefusedField), _codec(("codecs", 0, "configuration", "codecs", 0)), - _codec(("codecs", 0, "configuration", "codecs", 1), Unclaimed), + _codec(("codecs", 0, "configuration", "codecs", 1), UnclaimedField), _codec(("codecs", 0, "configuration", "index_codecs", 0)), ], [(frozenset({8}), frozenset({8}))], @@ -169,12 +169,12 @@ def _shard(chunk_shape: list[int], codecs: list[object] | None = None, **more: o ( ("data_type", "configuration", "fields", 0, "data_type"), DataTypeDefinition, - Read, + AcceptedField, ), ( ("data_type", "configuration", "fields", 1, "data_type"), DataTypeDefinition, - Unclaimed, + UnclaimedField, ), *POINTS[1:], _codec(("codecs", 0)), @@ -193,11 +193,11 @@ def _shard(chunk_shape: list[int], codecs: list[object] | None = None, **more: o CORE_AND_EXTENSIONS, [ POINTS[0], - (("chunk_grid",), ChunkGridDefinition, Unclaimed), + (("chunk_grid",), ChunkGridDefinition, UnclaimedField), POINTS[2], _codec(("codecs", 0)), - _codec(("codecs", 1), Refused), - (("storage_transformers", 0), StorageTransformerDefinition, Unclaimed), + _codec(("codecs", 1), RefusedField), + (("storage_transformers", 0), StorageTransformerDefinition, UnclaimedField), ], [(None,), None], ), @@ -205,7 +205,11 @@ def _shard(chunk_shape: list[int], codecs: list[object] | None = None, **more: o ( _document(data_type=float("nan")), CORE_AND_EXTENSIONS, - [(("data_type",), DataTypeDefinition, Refused), *POINTS[1:], _codec(("codecs", 0))], + [ + (("data_type",), DataTypeDefinition, RefusedField), + *POINTS[1:], + _codec(("codecs", 0)), + ], [(frozenset({4}),)], ), ], @@ -256,14 +260,14 @@ def test_a_document_reads_as_each_field_where_it_sits_and_its_codecs_as_a_pipeli def _array_fields( document: object, -) -> Iterator[tuple[Loc, Resolved[Any], tuple[ValidationProblem, ...]]]: +) -> Iterator[tuple[Loc, ResolvedField[Any], tuple[ValidationProblem, ...]]]: reading = read_array_metadata_v3(document) return with_problems(reading.fields(), reading.problems) def _group_fields( document: object, -) -> Iterator[tuple[Loc, Resolved[Any], tuple[ValidationProblem, ...]]]: +) -> Iterator[tuple[Loc, ResolvedField[Any], tuple[ValidationProblem, ...]]]: group = { "zarr_format": 3, "node_type": "group", @@ -279,7 +283,7 @@ def _group_fields( def _field_fields( document: object, -) -> Iterator[tuple[Loc, Resolved[Any], tuple[ValidationProblem, ...]]]: +) -> Iterator[tuple[Loc, ResolvedField[Any], tuple[ValidationProblem, ...]]]: resolved, problems = resolve( cast("dict[str, Any]", document)["codecs"][0], CodecDefinition, @@ -324,7 +328,9 @@ def _field_fields( ids=["an-array", "a-document-a-group-holds", "one-field"], ) def test_each_field_comes_with_the_problems_located_in_it( - fields: Callable[[object], Iterator[tuple[Loc, Resolved[Any], tuple[ValidationProblem, ...]]]], + fields: Callable[ + [object], Iterator[tuple[Loc, ResolvedField[Any], tuple[ValidationProblem, ...]]] + ], expected: dict[Loc, list[Loc]], ) -> None: # Those it was read with and those the document found with it where diff --git a/packages/zarr-metadata/tests/model/test_read_array_metadata_v2.py b/packages/zarr-metadata/tests/model/test_read_array_metadata_v2.py index 18e0d53b7e..dcd671a3b5 100644 --- a/packages/zarr-metadata/tests/model/test_read_array_metadata_v2.py +++ b/packages/zarr-metadata/tests/model/test_read_array_metadata_v2.py @@ -18,10 +18,10 @@ from zarr_metadata.v2.data_type.scalar import FLOAT_V2 from zarr_metadata.v2.definition import ( CORE_V2, + AcceptedField, Context, - Read, - Refused, - Unclaimed, + RefusedField, + UnclaimedField, ZarrV2CodecDefinition, ) from zarr_metadata.v3.definition import ( @@ -36,7 +36,7 @@ @pytest.mark.parametrize( ("changes", "context", "dtype", "compressor", "filters", "locs"), [ - ({}, None, Read, None, None, [("dtype",)]), + ({}, None, AcceptedField, None, None, [("dtype",)]), ( { "dtype": " None: - """`read_array_metadata_v2` reads the document once in the scope given (`CORE_V2` by default): `dtype`, `compressor` and `filters` are each `Read`, `Unclaimed` or None as written, `fields()` walks them in document order with a struct's record types after it, and a document with no problem has none.""" + """`read_array_metadata_v2` reads the document once in the scope given (`CORE_V2` by default): `dtype`, `compressor` and `filters` are each `AcceptedField`, `UnclaimedField` or None as written, `fields()` walks them in document order with a struct's record types after it, and a document with no problem has none.""" reading = read_array_metadata_v2({**BASE, **changes}, context=context) assert reading.problems == () assert type(reading.dtype) is dtype @@ -104,13 +104,13 @@ def test_a_reading_holds_each_field_as_the_scope_read_it( def test_error_a_reading_reports_what_the_validator_reports( value: object, problems: list[tuple[Loc, str]] ) -> None: - """A reading of a document with a problem holds every problem `validate_array_metadata_v2` finds, a `Refused` field where one was refused, and no model.""" + """A reading of a document with a problem holds every problem `validate_array_metadata_v2` finds, a `RefusedField` field where one was refused, and no model.""" reading = read_array_metadata_v2(value) assert [(p.loc, p.kind) for p in reading.problems] == problems assert reading.metadata is None assert [(p.loc, p.kind) for p in validate_array_metadata_v2(value)] == problems if isinstance(value, dict) and "dtype" in problems[0][0]: - assert isinstance(reading.dtype, Refused) + assert isinstance(reading.dtype, RefusedField) def test_the_v2_readers_read_in_the_scope_given() -> None: diff --git a/packages/zarr-metadata/tests/test_public_api.py b/packages/zarr-metadata/tests/test_public_api.py index 28bb326e14..7d95ef2cb4 100644 --- a/packages/zarr-metadata/tests/test_public_api.py +++ b/packages/zarr-metadata/tests/test_public_api.py @@ -302,14 +302,14 @@ def test_all_is_grouped_and_unique() -> None: "Nested", "NodeName", "NodePath", - # What a scope made of a field: `Read` by the definition that claims - # its name, `Unclaimed`, or `Refused`; `Resolved` is the three. - "Read", - "Refused", - "Resolved", + # What a scope made of a field: `AcceptedField` by the definition that claims + # its name, `UnclaimedField`, or `RefusedField`; `ResolvedField` is the three. + "AcceptedField", + "RefusedField", + "ResolvedField", "Stage", "StorageClass", - "Unclaimed", + "UnclaimedField", "CastOutOfRangeMode", "CastRoundingMode", "Endianness", diff --git a/packages/zarr-metadata/tests/v2/test_codecs.py b/packages/zarr-metadata/tests/v2/test_codecs.py index 79528f05a8..28447f887c 100644 --- a/packages/zarr-metadata/tests/v2/test_codecs.py +++ b/packages/zarr-metadata/tests/v2/test_codecs.py @@ -13,10 +13,10 @@ canonical_of, ) from zarr_metadata.v3.definition import ( + AcceptedField, Context, - Read, - Refused, - Unclaimed, + RefusedField, + UnclaimedField, resolve, ) @@ -70,7 +70,7 @@ def test_every_example_reads_and_is_written_back_as_it_was( ) -> None: """A numcodecs configuration reads by the definition its id names, with no problem, and is written back as it was: the id beside the parameters, which have one spelling.""" resolved, problems = resolve(field, ZarrV2CodecDefinition, SCOPE) - assert isinstance(resolved, Read) + assert isinstance(resolved, AcceptedField) assert resolved.definition.name == name assert problems == () assert resolved.to_json() == field @@ -82,9 +82,9 @@ def test_every_example_reads_and_is_written_back_as_it_was( [{"id": "categorize", "labels": ["a"]}, {"id": "pickle"}, {"id": "n5_wrapper", "inner": 1}], ) def test_an_id_the_package_does_not_model_is_unclaimed(field: dict[str, Any]) -> None: - """A codec id nothing in scope claims reads as `Unclaimed`, its parameters kept and unjudged, as a v3 extension nothing claims is.""" + """A codec id nothing in scope claims reads as `UnclaimedField`, its parameters kept and unjudged, as a v3 extension nothing claims is.""" resolved, problems = resolve(field, ZarrV2CodecDefinition, SCOPE) - assert isinstance(resolved, Unclaimed) + assert isinstance(resolved, UnclaimedField) assert problems == () assert json.dumps(resolved.to_json()) == json.dumps(field) @@ -124,4 +124,4 @@ def test_error_a_parameter_outside_what_numcodecs_takes_is_a_problem( """A parameter out of its range, of the wrong type, missing when numcodecs has no default, a dtype parameter that is no typestr, or a key no codec declares is reported at the parameter, beside the field.""" resolved, problems = resolve(field, ZarrV2CodecDefinition, SCOPE, ("c",)) assert [(p.loc, p.kind) for p in problems] == [(at, kind)] - assert isinstance(resolved, Read if kind == "unknown_key" else Refused) + assert isinstance(resolved, AcceptedField if kind == "unknown_key" else RefusedField) diff --git a/packages/zarr-metadata/tests/v2/test_data_types.py b/packages/zarr-metadata/tests/v2/test_data_types.py index c8a1bfa7e6..c339d13d35 100644 --- a/packages/zarr-metadata/tests/v2/test_data_types.py +++ b/packages/zarr-metadata/tests/v2/test_data_types.py @@ -11,10 +11,10 @@ fields_of, ) from zarr_metadata.v3.definition import ( + AcceptedField, Context, - Read, - Refused, - Unclaimed, + RefusedField, + UnclaimedField, canonical_fill_value, fill_value_problems, resolve, @@ -71,7 +71,7 @@ def _read(value: object) -> tuple[type, list[tuple[Loc, str]]]: def test_every_family_reads_its_typestrs(value: object, family: str, canonical: object) -> None: """Each typestr of a family reads by the family's definition, and its simplest spelling is the typestr NumPy writes: `|` for a type of one byte or no byte order, `us` for a microsecond unit; a records array reads as `struct`, each record's type read too.""" resolved, problems = resolve(value, ZarrV2DataTypeDefinition, SCOPE) - assert isinstance(resolved, Read) + assert isinstance(resolved, AcceptedField) assert resolved.definition.name == family assert problems == () assert canonical_of(resolved, problems) == canonical @@ -79,9 +79,9 @@ def test_every_family_reads_its_typestrs(value: object, family: str, canonical: @pytest.mark.parametrize("value", [" None: - """A typestr whose type code the v2 spec does not list is filed under itself, which nothing in scope claims: read as `Unclaimed`, with no problem, and written back as it was.""" + """A typestr whose type code the v2 spec does not list is filed under itself, which nothing in scope claims: read as `UnclaimedField`, with no problem, and written back as it was.""" resolved, problems = resolve(value, ZarrV2DataTypeDefinition, SCOPE) - assert isinstance(resolved, Unclaimed) + assert isinstance(resolved, UnclaimedField) assert problems == () assert resolved.to_json() == value @@ -140,7 +140,7 @@ def test_a_type_code_the_spec_does_not_list_is_unclaimed(value: str) -> None: def test_error_a_typestr_the_family_does_not_take_is_a_problem(value: object, at: Loc) -> None: """A typestr of a size, byte order or unit its family does not take is refused, the problem at the field; a struct's problems sit under `fields`, at the record and its position, and a record's type that is refused is reported there while the struct still reads.""" kind, problems = _read(value) - assert kind is Refused or "fields" in at + assert kind is RefusedField or "fields" in at assert next(loc for loc, _ in problems) == at diff --git a/packages/zarr-metadata/tests/v2/test_definition.py b/packages/zarr-metadata/tests/v2/test_definition.py index ec723b9515..7cb7e31cfb 100644 --- a/packages/zarr-metadata/tests/v2/test_definition.py +++ b/packages/zarr-metadata/tests/v2/test_definition.py @@ -8,9 +8,9 @@ CORE_V2, V2_CODECS, V2_DATA_TYPES, + AcceptedField, Context, - Read, - Unclaimed, + UnclaimedField, ZarrV2CodecDefinition, ZarrV2DataTypeDefinition, resolve_codec_v2, @@ -39,11 +39,11 @@ def test_core_v2_files_every_v2_definition_apart_from_v3() -> None: def test_the_v2_readers_read_one_field_in_core_v2_by_default() -> None: """`resolve_dtype_v2` and `resolve_codec_v2` read one field in `CORE_V2` when no scope is given, and in the scope given otherwise, prefixing every problem with `loc`.""" dtype, problems = resolve_dtype_v2(" bytes codecs", "invalid_value" @@ -209,7 +209,7 @@ def acme_paired_rules( ) -INT8: Final = Read(json="int8", name="int8", definition=INT8_DATA_TYPE, configuration={}) +INT8: Final = AcceptedField(json="int8", name="int8", definition=INT8_DATA_TYPE, configuration={}) """An `int8` field as a scope reads it: by its definition, holding nothing inside.""" @@ -232,7 +232,7 @@ def test_a_read_field_keeps_the_fields_it_read_inside() -> None: }, } resolved, _ = resolve(shard, CodecDefinition, CORE_AND_EXTENSIONS) - assert isinstance(resolved, Read) + assert isinstance(resolved, AcceptedField) assert {loc: inner.json for loc, inner in resolved.nested.items()} == { ("codecs", 0): "bytes", ("index_codecs", 0): "bytes", @@ -240,17 +240,17 @@ def test_a_read_field_keeps_the_fields_it_read_inside() -> None: } cast_value = {"name": "cast_value", "configuration": {"data_type": "int8"}} resolved, _ = resolve(cast_value, CodecDefinition, CORE_AND_EXTENSIONS) - assert isinstance(resolved, Read) + assert isinstance(resolved, AcceptedField) assert resolved.nested[("data_type",)] == INT8 # A field holding none has nothing inside, and one nothing in scope # claims holds no field at all; one its check or rules refuse keeps # what it read. assert resolve("int8", DataTypeDefinition, CORE)[0] == INT8 unclaimed = {"name": "acme.cast", "configuration": {"data_type": "int8"}} - assert isinstance(resolve(unclaimed, CodecDefinition, CORE_AND_EXTENSIONS)[0], Unclaimed) + assert isinstance(resolve(unclaimed, CodecDefinition, CORE_AND_EXTENSIONS)[0], UnclaimedField) refused = {"name": "cast_value", "configuration": {"data_type": "int8", "rounding": 1}} resolved, _ = resolve(refused, CodecDefinition, CORE_AND_EXTENSIONS) - assert isinstance(resolved, Refused) + assert isinstance(resolved, RefusedField) assert resolved.nested[("data_type",)] == INT8 @@ -283,18 +283,18 @@ def test_a_reading_says_the_name_the_field_was_written_with( (INT8, DataTypeDefinition), # A name read by the definition filed under another, as raw bits are. ( - Read( + AcceptedField( json="r16", name="r16", definition=RAW_BYTES_DATA_TYPE, configuration={"bits": 16} ), DataTypeDefinition, ), # Type arguments dropped, as `resolve` drops them. ( - Unclaimed(json="acme.t", name="acme.t", read_as=DataTypeDefinition[Any]), + UnclaimedField(json="acme.t", name="acme.t", read_as=DataTypeDefinition[Any]), DataTypeDefinition, ), ( - Refused( + RefusedField( json="int8", name="int8", read_as=DataTypeDefinition[Any], @@ -303,12 +303,12 @@ def test_a_reading_says_the_name_the_field_was_written_with( DataTypeDefinition, ), # Not a field, so named nothing and claimed by nothing. - (Refused(json=5, name=None, read_as=CodecDefinition), CodecDefinition), + (RefusedField(json=5, name=None, read_as=CodecDefinition), CodecDefinition), ], ids=["read", "raw-bits", "unclaimed", "refused", "refused-nameless"], ) def test_a_field_built_by_hand_is_of_the_kind_it_says( - built: Read[Any] | Unclaimed | Refused[Any], read_as: type[Definition[Any]] + built: AcceptedField[Any] | UnclaimedField | RefusedField[Any], read_as: type[Definition[Any]] ) -> None: assert built.read_as is read_as @@ -318,7 +318,7 @@ def test_error_a_field_read_by_hand_by_what_is_not_a_definition_of_a_kind( definition: Definition[Any] | None, ) -> None: with pytest.raises(TypeError, match="a field read is read by a definition of a kind"): - Read( + AcceptedField( json="acme.kindless", name="acme.kindless", definition=definition, # pyright: ignore[reportArgumentType] @@ -331,14 +331,14 @@ def test_error_a_field_read_by_hand_by_a_definition_filed_under_another_name(nam with pytest.raises( TypeError, match=f"a field named '{name}' is read by the definition filed under it" ): - Read(json=name, name=name, definition=INT8_DATA_TYPE, configuration={}) + AcceptedField(json=name, name=name, definition=INT8_DATA_TYPE, configuration={}) def test_error_a_field_refused_by_hand_by_a_definition_of_another_kind() -> None: with pytest.raises( TypeError, match="read as a DataTypeDefinition is read by one, got CodecDefinition" ): - Refused(json="gzip", name="gzip", read_as=DataTypeDefinition, definition=GZIP_CODEC) + RefusedField(json="gzip", name="gzip", read_as=DataTypeDefinition, definition=GZIP_CODEC) @pytest.mark.parametrize("name", ["int16", None]) @@ -348,14 +348,14 @@ def test_error_a_field_refused_by_hand_by_a_definition_filed_under_another_name( with pytest.raises( TypeError, match=f"a field named {name!r} is read by the definition filed under it" ): - Refused(json=name, name=name, read_as=DataTypeDefinition, definition=INT8_DATA_TYPE) + RefusedField(json=name, name=name, read_as=DataTypeDefinition, definition=INT8_DATA_TYPE) @pytest.mark.parametrize( "build", [ - lambda: Unclaimed(json="acme.t", name="acme.t", read_as=Definition), - lambda: Refused(json="acme.t", name="acme.t", read_as=Definition), + lambda: UnclaimedField(json="acme.t", name="acme.t", read_as=Definition), + lambda: RefusedField(json="acme.t", name="acme.t", read_as=Definition), ], ids=["unclaimed", "refused"], ) @@ -366,7 +366,7 @@ def test_error_a_field_built_by_hand_of_no_kind(build: Callable[[], object]) -> def test_error_a_field_nothing_claims_built_by_hand_without_a_name() -> None: with pytest.raises(TypeError, match="a field nothing in scope claims is named"): - Unclaimed(json=5, name=None, read_as=CodecDefinition) # pyright: ignore[reportArgumentType] + UnclaimedField(json=5, name=None, read_as=CodecDefinition) # pyright: ignore[reportArgumentType] @pytest.mark.parametrize( @@ -466,7 +466,7 @@ def test_a_field_is_written_as_every_reader_takes_it( field: JSONValue, kind: type[Definition[Any]], written: JSONValue ) -> None: resolved, _ = resolve(field, kind, SCOPE) - assert isinstance(resolved, (Read, Unclaimed)) + assert isinstance(resolved, (AcceptedField, UnclaimedField)) assert resolved.to_json() == written @@ -476,26 +476,26 @@ def test_a_field_is_written_as_every_reader_takes_it( ( {"name": "gzip", "configuration": {"level": 5}}, CodecDefinition, - Read, + AcceptedField, {"level": 5}, [], ), - ("crc32c", CodecDefinition, Read, {}, []), - ({"name": "crc32c"}, CodecDefinition, Read, {}, []), - ({"name": "crc32c", "configuration": {}}, CodecDefinition, Read, {}, []), - ({"name": "crc32c", "must_understand": True}, CodecDefinition, Read, {}, []), - ("bytes", CodecDefinition, Read, {}, []), + ("crc32c", CodecDefinition, AcceptedField, {}, []), + ({"name": "crc32c"}, CodecDefinition, AcceptedField, {}, []), + ({"name": "crc32c", "configuration": {}}, CodecDefinition, AcceptedField, {}, []), + ({"name": "crc32c", "must_understand": True}, CodecDefinition, AcceptedField, {}, []), + ("bytes", CodecDefinition, AcceptedField, {}, []), ( {"name": "bytes", "configuration": {"endian": "big"}}, CodecDefinition, - Read, + AcceptedField, {"endian": "big"}, [], ), ( {"name": "regular", "configuration": {"chunk_shape": [2, 3]}}, ChunkGridDefinition, - Read, + AcceptedField, {"chunk_shape": (2, 3)}, [], ), @@ -504,7 +504,7 @@ def test_a_field_is_written_as_every_reader_takes_it( ( {"name": "gzip", "configuration": {"level": 5, "extra": 1}}, CodecDefinition, - Read, + AcceptedField, {"level": 5}, [(("configuration", "extra"), "unknown_key")], ), @@ -512,19 +512,19 @@ def test_a_field_is_written_as_every_reader_takes_it( ( {"name": "acme.bounded", "configuration": {"level": 5, "windw": 10}}, CodecDefinition, - Read, + AcceptedField, {"level": 5}, [(("configuration", "windw"), "unknown_key")], ), # A key that is not required, written as a string under postponed # annotations, and every key of a `total=False` TypedDict. - ("acme.level", CodecDefinition, Read, {}, []), - ("acme.total", CodecDefinition, Read, {}, []), + ("acme.level", CodecDefinition, AcceptedField, {}, []), + ("acme.total", CodecDefinition, AcceptedField, {}, []), # A kind with type arguments is that kind. ( {"name": "gzip", "configuration": {"level": 5}}, CodecDefinition[Any], - Read, + AcceptedField, {"level": 5}, [], ), @@ -534,7 +534,7 @@ def test_a_field_is_written_as_every_reader_takes_it( "configuration": {"label": "a", "children": [{"label": "b", "children": []}]}, }, CodecDefinition, - Read, + AcceptedField, {"label": "a", "children": ({"label": "b", "children": ()},)}, [], ), @@ -546,7 +546,7 @@ def test_a_field_is_written_as_every_reader_takes_it( "configuration": {"fallback": {"codec": {"name": "gzip"}, "note": "x"}}, }, CodecDefinition, - Read, + AcceptedField, {"fallback": {"codec": {"name": "gzip"}, "note": "x"}}, [], ), @@ -555,18 +555,18 @@ def test_a_field_is_written_as_every_reader_takes_it( ( {"name": "acme.routes", "configuration": {"fast": "crc32c"}}, CodecDefinition, - Read, + AcceptedField, {"fast": {"name": "crc32c"}}, [], ), # Nothing in scope claims it: left unjudged, not refused. - ({"name": "zfpy", "configuration": {"x": 1}}, CodecDefinition, Unclaimed, None, []), + ({"name": "zfpy", "configuration": {"x": 1}}, CodecDefinition, UnclaimedField, None, []), # A nested field is read in the same scope, and held as a document # writes it; one out of scope is left be. ( {"name": "acme.stack", "configuration": {"codecs": ["crc32c", "zfpy"]}}, CodecDefinition, - Read, + AcceptedField, {"codecs": ({"name": "crc32c"}, {"name": "zfpy"})}, [], ), @@ -601,7 +601,9 @@ def test_a_field_is_read_in_scope( ) -> None: resolved, found = resolve(field, kind, SCOPE) assert type(resolved) is variant - assert (resolved.configuration if isinstance(resolved, Read) else None) == configuration + assert ( + resolved.configuration if isinstance(resolved, AcceptedField) else None + ) == configuration assert _locs(found) == problems @@ -609,7 +611,7 @@ def test_error_a_rule_refuses_a_value() -> None: resolved, found = resolve( {"name": "gzip", "configuration": {"level": 12}}, CodecDefinition, SCOPE ) - assert isinstance(resolved, Refused) + assert isinstance(resolved, RefusedField) assert resolved.definition is GZIP_CODEC assert _locs(found) == [(("configuration", "level"), "invalid_value")] @@ -618,20 +620,20 @@ def test_error_a_member_of_the_wrong_type_is_not_asked_of_the_rules() -> None: resolved, found = resolve( {"name": "gzip", "configuration": {"level": "5"}}, CodecDefinition, SCOPE ) - assert isinstance(resolved, Refused) + assert isinstance(resolved, RefusedField) assert _locs(found) == [(("configuration", "level"), "invalid_type")] def test_error_a_required_configuration_is_missing() -> None: resolved, found = resolve("gzip", CodecDefinition, SCOPE) - assert isinstance(resolved, Refused) + assert isinstance(resolved, RefusedField) assert _locs(found) == [(("configuration",), "missing_key")] def test_error_the_configuration_is_not_an_object() -> None: - # Refused, and still claimed by the definition its name names. + # RefusedField, and still claimed by the definition its name names. resolved, found = resolve({"name": "gzip", "configuration": 5}, CodecDefinition, SCOPE) - assert resolved == Refused( + assert resolved == RefusedField( json={"name": "gzip", "configuration": 5}, name="gzip", read_as=CodecDefinition, @@ -644,7 +646,7 @@ def test_error_must_understand_false_is_refused() -> None: # The envelope's problem, reported with the field; the configuration # was read, so a later layer can still judge the codec. resolved, found = resolve({"name": "crc32c", "must_understand": False}, CodecDefinition, SCOPE) - assert isinstance(resolved, Read) + assert isinstance(resolved, AcceptedField) assert _locs(found) == [(("must_understand",), "invalid_value")] @@ -733,7 +735,7 @@ def test_error_null_is_not_a_field() -> None: # a value of the wrong type. resolved, found = resolve(None, CodecDefinition, SCOPE) assert (resolved, _locs(found)) == ( - Refused(json=None, name=None, read_as=CodecDefinition), + RefusedField(json=None, name=None, read_as=CodecDefinition), [((), "invalid_type")], ) configuration, found = GZIP_CODEC.judge(None) @@ -746,7 +748,7 @@ def test_error_a_value_that_is_not_json() -> None: resolved, found = resolve( {"name": "gzip", "configuration": {"level": math.nan}}, CodecDefinition, SCOPE ) - assert resolved == Refused( + assert resolved == RefusedField( json=UNSET, name="gzip", read_as=CodecDefinition, definition=GZIP_CODEC ) assert _locs(found) == [(("configuration", "level"), "invalid_value")] @@ -754,7 +756,7 @@ def test_error_a_value_that_is_not_json() -> None: def test_error_a_value_that_is_not_a_field() -> None: resolved, found = resolve(5, CodecDefinition, SCOPE) - assert resolved == Refused(json=5, name=None, read_as=CodecDefinition) + assert resolved == RefusedField(json=5, name=None, read_as=CodecDefinition) assert len(found) == 1 @@ -762,7 +764,7 @@ def test_error_a_regular_grid_extent_is_negative() -> None: resolved, found = resolve( {"name": "regular", "configuration": {"chunk_shape": [2, -1]}}, ChunkGridDefinition, SCOPE ) - assert isinstance(resolved, Refused) + assert isinstance(resolved, RefusedField) assert _locs(found) == [(("configuration", "chunk_shape", 1), "invalid_value")] @@ -774,8 +776,8 @@ def test_error_a_nested_field_is_judged_where_it_sits() -> None: "configuration": {"codecs": ["crc32c", {"name": "gzip", "configuration": {"level": 12}}]}, } resolved, found = resolve(field, CodecDefinition, SCOPE) - assert isinstance(resolved, Read) - assert isinstance(resolved.nested[("codecs", 1)], Refused) + assert isinstance(resolved, AcceptedField) + assert isinstance(resolved.nested[("codecs", 1)], RefusedField) assert _locs(found) == [ (("configuration", "codecs", 1, "configuration", "level"), "invalid_value") ] @@ -787,8 +789,8 @@ def test_error_a_nested_field_in_extra_items_is_judged_where_it_sits() -> None: "configuration": {"slow": {"name": "gzip", "configuration": {"level": 12}}}, } resolved, found = resolve(field, CodecDefinition, SCOPE) - assert isinstance(resolved, Read) - assert isinstance(resolved.nested[("slow",)], Refused) + assert isinstance(resolved, AcceptedField) + assert isinstance(resolved.nested[("slow",)], RefusedField) assert _locs(found) == [(("configuration", "slow", "configuration", "level"), "invalid_value")] @@ -799,7 +801,7 @@ def test_error_a_nested_envelope_s_problem_is_its_own_and_the_rules_are_asked() resolved, found = resolve( {"name": "acme.stack", "configuration": {"codecs": [unread]}}, CodecDefinition, SCOPE ) - assert isinstance(resolved, Read) + assert isinstance(resolved, AcceptedField) assert _locs(found) == [(("configuration", "codecs", 0, "must_understand"), "invalid_value")] # Its rules are asked all the same: one refuses the array -> bytes # codec it holds, which is the stack's own problem. @@ -808,7 +810,7 @@ def test_error_a_nested_envelope_s_problem_is_its_own_and_the_rules_are_asked() CodecDefinition, SCOPE, ) - assert isinstance(resolved, Refused) + assert isinstance(resolved, RefusedField) assert _locs(found) == [ (("configuration", "codecs", 1), "invalid_value"), (("configuration", "codecs", 0, "must_understand"), "invalid_value"), @@ -820,7 +822,7 @@ def test_error_a_container_rule_is_not_asked_of_a_malformed_nested_field() -> No # reported where it sits, and the rule is not asked, as `judge` would not. field = {"name": "acme.stack", "configuration": {"codecs": [{"configuration": {}}]}} resolved, found = resolve(field, CodecDefinition, SCOPE) - assert isinstance(resolved, Refused) + assert isinstance(resolved, RefusedField) assert _locs(found) == [(("configuration", "codecs", 0, "name"), "missing_key")] assert ACME_STACK.judge(field["configuration"])[0] is None @@ -842,7 +844,7 @@ def test_error_a_rule_reads_the_fields_the_configuration_holds_as_the_scope_read # nothing read. field = {"name": "acme.stack", "configuration": {"codecs": ["crc32c", "bytes"]}} resolved, found = resolve(field, CodecDefinition, SCOPE) - assert isinstance(resolved, Refused) + assert isinstance(resolved, RefusedField) assert _locs(found) == [(("configuration", "codecs", 1), "invalid_value")] assert ACME_STACK.judge(field["configuration"])[1] == () @@ -851,7 +853,7 @@ def test_error_a_nested_member_is_not_a_field() -> None: resolved, found = resolve( {"name": "acme.stack", "configuration": {"codecs": [5]}}, CodecDefinition, SCOPE ) - assert isinstance(resolved, Refused) + assert isinstance(resolved, RefusedField) assert _locs(found) == [(("configuration", "codecs", 0), "invalid_type")] @@ -1101,8 +1103,8 @@ def test_two_fields_are_equal_when_they_read_the_same( """However each was spelled: equal fields have one simplest spelling and one hash, though each is written as read.""" first, _ = resolve(one, kind, CORE_AND_EXTENSIONS) second, _ = resolve(other, kind, CORE_AND_EXTENSIONS) - assert isinstance(first, (Read, Unclaimed)) - assert isinstance(second, (Read, Unclaimed)) + assert isinstance(first, (AcceptedField, UnclaimedField)) + assert isinstance(second, (AcceptedField, UnclaimedField)) assert (first == second) is equal assert (canonical_of(first, ()) == canonical_of(second, ())) is equal if equal: @@ -1311,10 +1313,10 @@ def test_an_extension_is_named_as_the_spec_names_one(name: str, valid: bool) -> for field, at in ((name, ()), ({"name": name}, ("name",))): resolved, problems = resolve(field, CodecDefinition, CORE) if valid: - assert not isinstance(resolved, Refused) + assert not isinstance(resolved, RefusedField) assert problems == () else: - assert isinstance(resolved, Refused) + assert isinstance(resolved, RefusedField) assert [(p.loc, p.kind) for p in problems] == [(at, "invalid_value")] assert "expected an extension name" in problems[0].message assert [(p.loc, p.kind) for p in validate_metadata_field_v3(field)] == ( @@ -1329,7 +1331,7 @@ def test_error_a_bad_name_is_a_problem_when_the_field_is_not_json_too() -> None: # and no definition is asked to claim it. field = {"name": "Acme", "configuration": {"x": float("nan")}} resolved, problems = resolve(field, CodecDefinition, CORE) - assert isinstance(resolved, Refused) + assert isinstance(resolved, RefusedField) assert resolved.definition is None assert [(p.loc, p.kind) for p in problems] == [ (("name",), "invalid_value"), @@ -1348,12 +1350,12 @@ def test_error_a_definition_is_named_as_the_spec_names_an_extension(name: str) - def test_error_a_field_built_by_hand_is_named_as_the_spec_names_one() -> None: with pytest.raises(TypeError, match="named as a document names a"): - Unclaimed(json="Int8", name="Int8", read_as=CodecDefinition) + UnclaimedField(json="Int8", name="Int8", read_as=CodecDefinition) def test_error_a_field_without_a_name_is_missing_one() -> None: resolved, problems = resolve({"configuration": {}}, CodecDefinition, CORE) - assert isinstance(resolved, Refused) + assert isinstance(resolved, RefusedField) assert [(p.loc, p.kind, p.message) for p in problems] == [ (("name",), "missing_key", "missing required key") ] @@ -1364,7 +1366,7 @@ def test_raw_bits_of_more_than_a_hundred_digits_are_no_size_but_a_name(digits: i # `int` refuses to convert more than 4,300 digits, and no size has a # hundred: such a name is an extension's, which nothing in scope claims. resolved, problems = resolve("r" + "1" * digits, DataTypeDefinition, CORE_AND_EXTENSIONS) - assert type(resolved) is Unclaimed + assert type(resolved) is UnclaimedField assert problems == () diff --git a/packages/zarr-metadata/tests/v3/test_every_definition.py b/packages/zarr-metadata/tests/v3/test_every_definition.py index 112fdec528..a7981b6b31 100644 --- a/packages/zarr-metadata/tests/v3/test_every_definition.py +++ b/packages/zarr-metadata/tests/v3/test_every_definition.py @@ -23,14 +23,14 @@ from zarr_metadata.v3.definition import ( CORE, CORE_AND_EXTENSIONS, + AcceptedField, ChunkGridDefinition, ChunkKeyEncodingDefinition, CodecDefinition, Context, DataTypeDefinition, Definition, - Read, - Refused, + RefusedField, ValidationProblem, fill_value_problems, resolve, @@ -146,7 +146,7 @@ def _read(key: str, field: object) -> tuple[type, list[tuple[tuple[str | int, ...], str]]]: - """What the scope made of `field` -- `Read`, `Unclaimed` or `Refused` -- and where each problem is.""" + """What the scope made of `field` -- `AcceptedField`, `UnclaimedField` or `RefusedField` -- and where each problem is.""" resolved, problems = resolve(field, KINDS[key.split(":")[0]], CORE_AND_EXTENSIONS) return type(resolved), [(found.loc, found.kind) for found in problems] @@ -175,9 +175,9 @@ def test_every_definition_in_scope_has_an_example() -> None: ("key", "field"), CASES, ids=[f"{k}:{i}" for i, (k, _) in enumerate(CASES)] ) def test_every_example_reads_and_its_simplest_spelling_is_stable(key: str, field: object) -> None: - # Read, with nothing wrong; and its simplest spelling reads the same, + # AcceptedField, with nothing wrong; and its simplest spelling reads the same, # and is its own simplest spelling. - assert _read(key, field) == (Read, []) + assert _read(key, field) == (AcceptedField, []) kind = KINDS[key.split(":")[0]] simplest, problems = canonicalize(field, kind, CORE_AND_EXTENSIONS) assert problems == () @@ -336,7 +336,7 @@ def test_raw_bits_read_as_r_star_with_the_size_their_name_carries( ) -> None: resolved, problems = resolve(field, DataTypeDefinition, CORE_AND_EXTENSIONS) assert problems == () - assert isinstance(resolved, Read) + assert isinstance(resolved, AcceptedField) assert resolved.definition is RAW_BYTES_DATA_TYPE assert resolved.json == field assert configuration_of(resolved, RAW_BYTES_DATA_TYPE) == {"bits": bits} @@ -350,11 +350,11 @@ def test_a_reader_reads_raw_bits_its_own_way_by_defining_r_star() -> None: scope = CORE_AND_EXTENSIONS.extended_with(mine) resolved, problems = resolve("r12", DataTypeDefinition, scope) assert problems == () - assert isinstance(resolved, Read) + assert isinstance(resolved, AcceptedField) assert resolved.definition is mine # The package's own takes no 12 bits, and refuses them. again, _ = resolve("r12", DataTypeDefinition, scope.extended_with(RAW_BYTES_DATA_TYPE)) - assert isinstance(again, Refused) + assert isinstance(again, RefusedField) assert again.definition is RAW_BYTES_DATA_TYPE @@ -367,7 +367,7 @@ def test_error_r_star_is_notation_and_no_name(field: object, at: tuple[str, ...] # How the specification's table writes raw bits, and no document's name # for them: `*` is no character of an extension name. resolved, problems = resolve(field, DataTypeDefinition, CORE_AND_EXTENSIONS) - assert type(resolved) is Refused + assert type(resolved) is RefusedField assert [(p.loc, p.kind) for p in problems] == [(at, "invalid_value")] @@ -863,7 +863,7 @@ def test_error_a_configuration_written_beside_a_raw_bits_name() -> None: # The name carries the configuration, so one written beside it holds # nothing: each member is a key nothing declares. assert _read("data_type:r*", {"name": "r16", "configuration": {"bits": 16}}) == ( - Read, + AcceptedField, [(("configuration", "bits"), "unknown_key")], ) diff --git a/packages/zarr-metadata/tests/v3/test_fill_values.py b/packages/zarr-metadata/tests/v3/test_fill_values.py index 0b397f6a18..f64aacf4f9 100644 --- a/packages/zarr-metadata/tests/v3/test_fill_values.py +++ b/packages/zarr-metadata/tests/v3/test_fill_values.py @@ -29,11 +29,11 @@ from zarr_metadata.v3.data_type.struct import STRUCT_DATA_TYPE from zarr_metadata.v3.definition import ( CORE_AND_EXTENSIONS, + AcceptedField, DataTypeDefinition, EmptyConfiguration, JSONValue, Nested, - Read, ValidationProblem, canonical_fill_value, fill_value_problems, @@ -322,7 +322,7 @@ def deep(levels: int) -> dict[str, object]: def test_a_struct_read_without_its_field_types_leaves_its_fields_unjudged() -> None: # A reading built by hand, holding no field type's reading. configuration = {"fields": ({"name": "a", "data_type": "int8"},)} - struct = Read( + struct = AcceptedField( json=STRUCT, name="struct", definition=STRUCT_DATA_TYPE, configuration=configuration ) assert fill_value_problems(struct, {"a": 300}) == () @@ -336,10 +336,10 @@ def _alike(left: object, right: object) -> bool: return json.dumps(left, sort_keys=True) == json.dumps(right, sort_keys=True) -def _read(data_type: JSONValue) -> Read[DataTypeDefinition[Any]]: +def _read(data_type: JSONValue) -> AcceptedField[DataTypeDefinition[Any]]: resolved, found = resolve(data_type, DataTypeDefinition, CORE_AND_EXTENSIONS) assert found == () - assert isinstance(resolved, Read) + assert isinstance(resolved, AcceptedField) return resolved diff --git a/packages/zarr-metadata/tests/v3/test_kinds.py b/packages/zarr-metadata/tests/v3/test_kinds.py index fca6a8ed7a..2d2546fbcb 100644 --- a/packages/zarr-metadata/tests/v3/test_kinds.py +++ b/packages/zarr-metadata/tests/v3/test_kinds.py @@ -20,13 +20,13 @@ from zarr_metadata.v3._scope import kind_name from zarr_metadata.v3.codec.gzip import GZIP_CODEC from zarr_metadata.v3.definition import ( + AcceptedField, CodecDefinition, Context, Definition, EmptyConfiguration, - Read, - Refused, - Unclaimed, + RefusedField, + UnclaimedField, canonical_fill_value, fill_value_problems, resolve, @@ -63,7 +63,7 @@ def test_a_subclass_of_a_kind_is_a_definition_of_that_kind() -> None: assert kind_of(mine) is CodecDefinition scope = Context.of(mine) assert scope.claimant(CodecDefinition, "mine") is mine - assert isinstance(resolve({"name": "mine"}, CodecDefinition, scope)[0], Read) + assert isinstance(resolve({"name": "mine"}, CodecDefinition, scope)[0], AcceptedField) def test_a_kind_of_its_own_is_filed_apart() -> None: @@ -140,11 +140,16 @@ def name_loc(cls, loc: Loc) -> Loc: @pytest.mark.parametrize( ("field", "kind", "problems", "written"), [ - ({"id": "flat", "level": 1}, Read, [], {"id": "flat", "level": 1}), - ({"id": "other", "x": 1}, Unclaimed, [], {"id": "other", "x": 1}), - ({"id": "flat", "level": -1}, Refused, [(("c", "level"), "invalid_value")], None), - ({"id": "flat", "payload": object()}, Refused, [(("c", "payload"), "invalid_type")], None), - ("flat", Refused, [(("c",), "invalid_type")], None), + ({"id": "flat", "level": 1}, AcceptedField, [], {"id": "flat", "level": 1}), + ({"id": "other", "x": 1}, UnclaimedField, [], {"id": "other", "x": 1}), + ({"id": "flat", "level": -1}, RefusedField, [(("c", "level"), "invalid_value")], None), + ( + {"id": "flat", "payload": object()}, + RefusedField, + [(("c", "payload"), "invalid_type")], + None, + ), + ("flat", RefusedField, [(("c",), "invalid_type")], None), ], ids=["read", "unclaimed", "out-of-range", "not-json", "not-an-object"], ) @@ -156,7 +161,7 @@ def test_a_kind_reads_the_envelope_its_format_writes( assert type(resolved) is kind assert [(problem.loc, problem.kind) for problem in found] == problems if written is not None: - assert not isinstance(resolved, Refused) + assert not isinstance(resolved, RefusedField) assert resolved.to_json() == written diff --git a/packages/zarr-metadata/tests/v3/test_pipelines.py b/packages/zarr-metadata/tests/v3/test_pipelines.py index 5c733e430b..f1f779ba5b 100644 --- a/packages/zarr-metadata/tests/v3/test_pipelines.py +++ b/packages/zarr-metadata/tests/v3/test_pipelines.py @@ -29,7 +29,7 @@ DataTypeField, JSONValue, Nested, - Resolved, + ResolvedField, resolve, ) @@ -456,7 +456,9 @@ class AcmeHolderConfiguration(TypedDict, closed=True): types: tuple[DataTypeField, ...] -def _holder(pipelines: object, types: JSONValue = ("uint8",)) -> Resolved[CodecDefinition[Any]]: +def _holder( + pipelines: object, types: JSONValue = ("uint8",) +) -> ResolvedField[CodecDefinition[Any]]: """A codec holding a pipeline of codecs and a list of data types, `types`, whose pipelines are `pipelines`.""" holder = CodecDefinition( name="acme.holder", diff --git a/packages/zarr-metadata/tests/v3/test_scope.py b/packages/zarr-metadata/tests/v3/test_scope.py index c51daa1eda..ea25d8e362 100644 --- a/packages/zarr-metadata/tests/v3/test_scope.py +++ b/packages/zarr-metadata/tests/v3/test_scope.py @@ -36,8 +36,8 @@ Context, DataTypeDefinition, Definition, - Refused, - Resolved, + RefusedField, + ResolvedField, ScopeConflictError, StorageTransformerDefinition, resolve, @@ -61,7 +61,7 @@ """ -def _read(data: object, kind: type[Definition[Any]], scope: Context) -> Resolved[Any]: +def _read(data: object, kind: type[Definition[Any]], scope: Context) -> ResolvedField[Any]: return resolve(data, kind, scope)[0] @@ -126,12 +126,12 @@ def test_error_a_scope_conflict_says_each_disagreement() -> None: _read({"name": "gzip", "configuration": {"level": 12}}, CodecDefinition, CORE), {(CodecDefinition, "gzip"): GZIP_CODEC}, ), - (Refused(json=3, name=None, read_as=CodecDefinition), {}), + (RefusedField(json=3, name=None, read_as=CodecDefinition), {}), ], ids=["read", "unclaimed", "raw-bits", "nested", "refused-claimed", "refused-nameless"], ) def test_claims_of_says_what_a_reading_claimed_of_each_name( - field: Resolved[Any], claims: dict[object, object] + field: ResolvedField[Any], claims: dict[object, object] ) -> None: """A reading's claims name the definition that read each name the field and the fields it holds write, keyed as the scope files it -- raw bits under `r*` -- and None where nothing claimed one; a field refused by a definition still claims it, and one that names nothing claims nothing.""" assert claims_of(fields_of(field)) == claims @@ -257,7 +257,7 @@ def test_error_claims_of_refuses_one_name_read_two_ways() -> None: ], ) def test_refines_orders_readings_by_information( - field: Resolved[Any], other: Resolved[Any], expected: bool + field: ResolvedField[Any], other: ResolvedField[Any], expected: bool ) -> None: """`field` refines `other` when it reads the same where both read and gains where `other` left a name unclaimed -- in the fields it holds too; a loss, a conflict, a refused field, or two unclaimed fields written differently do not.""" assert refines(field, other) is expected @@ -279,7 +279,7 @@ def test_refines_orders_readings_by_information( @given(st.sampled_from(READINGS), st.sampled_from(READINGS), st.sampled_from(READINGS)) def test_refines_is_a_partial_order_whose_bottom_is_equality( - a: Resolved[Any], b: Resolved[Any], c: Resolved[Any] + a: ResolvedField[Any], b: ResolvedField[Any], c: ResolvedField[Any] ) -> None: """Over readings of documents in several scopes and spellings, `refines` is reflexive and transitive, two fields that refine each other are equal, and equal fields refine the same fields.""" assert refines(a, a) diff --git a/packages/zarr-metadata/tests/v3/test_sharding.py b/packages/zarr-metadata/tests/v3/test_sharding.py index 4d94f8d105..bcf834b804 100644 --- a/packages/zarr-metadata/tests/v3/test_sharding.py +++ b/packages/zarr-metadata/tests/v3/test_sharding.py @@ -24,7 +24,7 @@ CodecDefinition, DataTypeDefinition, JSONValue, - Resolved, + ResolvedField, resolve, ) @@ -37,7 +37,7 @@ INDEX: list[JSONValue] = [LITTLE, "crc32c"] -def _dt(name: JSONValue) -> Resolved[DataTypeDefinition[Any]]: +def _dt(name: JSONValue) -> ResolvedField[DataTypeDefinition[Any]]: return resolve(name, DataTypeDefinition, CORE_AND_EXTENSIONS)[0] @@ -45,7 +45,9 @@ def _dt(name: JSONValue) -> Resolved[DataTypeDefinition[Any]]: UINT64 = _dt("uint64") -def _chunk(*axes: set[int] | None, data_type: Resolved[DataTypeDefinition[Any]] = FLOAT32) -> Chunk: +def _chunk( + *axes: set[int] | None, data_type: ResolvedField[DataTypeDefinition[Any]] = FLOAT32 +) -> Chunk: return Chunk(tuple(None if axis is None else frozenset(axis) for axis in axes), data_type) From bc4eced79bfebb22e7b0dc6327798c4aff0f1f59 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Fri, 9 Oct 2026 12:06:18 +0200 Subject: [PATCH 85/94] refactor(zarr-metadata): rename Definition.judge to read_configuration; its check step is private A definition reads a configuration as the model readers read a document: the pair of what was read and every problem, never raising. The type-check step that only read_configuration called is _check_configuration; the public shape check stays zarr_metadata.typed_json.check. Assisted-by: ClaudeCode:claude-fable-5-1 Co-Authored-By: Claude Fable 5.1 --- .../zarr-metadata/changes/4434.feature.md | 2 +- .../src/zarr_metadata/v3/_definition.py | 18 ++++---- .../src/zarr_metadata/v3/definition.py | 8 ++-- .../zarr-metadata/tests/test_problem_data.py | 4 +- .../tests/v3/test_definitions.py | 41 ++++++++++--------- .../tests/v3/test_every_definition.py | 2 +- 6 files changed, 39 insertions(+), 36 deletions(-) diff --git a/packages/zarr-metadata/changes/4434.feature.md b/packages/zarr-metadata/changes/4434.feature.md index 5768a29019..35981fae5e 100644 --- a/packages/zarr-metadata/changes/4434.feature.md +++ b/packages/zarr-metadata/changes/4434.feature.md @@ -17,7 +17,7 @@ when built, and nothing happens at class creation. Reading a field is three steps, each usable on its own by a caller holding nothing but JSON. `check(value, SomeTypedDict)`, from `zarr_metadata.typed_json`, type-checks JSON against a TypedDict and needs -nothing else; `definition.judge(configuration)` is the check, each nested +nothing else; `definition.read_configuration(configuration)` is the check, each nested field's envelope judged, and then the rules; `resolve(field, CodecDefinition, scope)` reads a whole field in a scope -- its envelope judged, its name related to a definition, its configuration judged, each diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py b/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py index 3b6b112367..815f9dedd4 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py @@ -13,7 +13,7 @@ needs none. 2. The rules need the definition: plain functions over the checked TypedDict, for everything finer than a type -- a bound, members read - together. `Definition.judge` is the check and then the rules, for a + together. `Definition.read_configuration` is the check and then the rules, for a caller holding one configuration. 3. The reading needs a scope. `resolve` relates the field's name to a definition through a `Context`, judges the configuration, and reads @@ -196,7 +196,7 @@ class Definition(Generic[C]): the TypedDict admits and nothing else, each member within the bounds its type carries, and the fields it holds as the scope read them -- a struct's field types -- which is nothing when no scope read it. - `judge` is the two, for a caller holding JSON. + `read_configuration` is the two, for a caller holding JSON. Each function is handed the configuration as a read-only view, no `dict`: `copy.deepcopy` and `json.dumps` refuse it, and a function that @@ -328,7 +328,7 @@ def requires_configuration(self) -> bool: """Whether a document must write a configuration: whether the TypedDict has a required key.""" return len(typeddict_keys(self.configuration).required) != 0 - def check(self, value: object, loc: Loc = ()) -> tuple[C | None, Problems]: + def _check_configuration(self, value: object, loc: Loc = ()) -> tuple[C | None, Problems]: """`value` type-checked as this definition's configuration, each nested field's envelope judged. `zarr_metadata.typed_json.check` is the type check alone; this also @@ -337,7 +337,7 @@ def check(self, value: object, loc: Loc = ()) -> tuple[C | None, Problems]: configuration, problems = _configuration_checked(value, self.configuration, loc) return configuration, with_input(problems, value, loc) - def judge(self, value: object, loc: Loc = ()) -> tuple[C | None, Problems]: + def read_configuration(self, value: object, loc: Loc = ()) -> tuple[C | None, Problems]: """`value` type-checked, then judged by the rules: the configuration if it holds, and every problem. The rules are asked only of a configuration that type-checked, @@ -349,7 +349,7 @@ def judge(self, value: object, loc: Loc = ()) -> tuple[C | None, Problems]: type whose values vary in size -- finds nothing to judge: `resolve` reads the field in a scope, and asks every rule. """ - configuration, problems = self.check(value, loc) + configuration, problems = self._check_configuration(value, loc) if configuration is None: return None, problems refused = ruled(self, lambda: self.rules(read_only(configuration), _nothing_nested()), loc) @@ -1056,7 +1056,7 @@ def _configuration_checked( ) -> tuple[T | None, Problems]: """`value` checked as `shape`, as `typed_json.check` checks it, and each nested field's envelope judged. - The step a definition's `judge` starts from. A member typed with a + The step a definition's `read_configuration` starts from. A member typed with a field alias holds a metadata field, whose envelope is judged as a document's is: a `must_understand` of `false` is a problem of the configuration, and the value does not come back; a stray member is an @@ -1500,7 +1500,7 @@ def fill_value_problems(data_type: ResolvedField[F], value: object, loc: Loc = ( `value` is refined to JSON first: not JSON is the first verdict, whatever the data type. It is then checked against the JSON shape the data type's definition declares, and judged by its fill value rules, as - `judge` judges a configuration: a key the shape does not declare is + `read_configuration` reads a configuration: a key the shape does not declare is reported and left out, and the rules still judge the rest. The rules see the fields the configuration holds as the scope read them: a struct judges each field's fill value by that field's own type. A data @@ -1792,7 +1792,7 @@ def _read_carried( {} if given is None else given, kind.configuration_loc(loc), ) - configuration, judged = definition.judge(carried) + configuration, judged = definition.read_configuration(carried) # What is wrong with what the name carries is the field's: found at # the field, where the name is what is there, and what was expected # of a member of the configuration is not expected of it. @@ -1902,7 +1902,7 @@ def _canonical_field(resolved: AcceptedField[Any]) -> JSONValue | None: "canonical", lambda: dict(cast("Mapping[str, JSONValue]", definition.canonical(view))), ) - _, refused = definition.judge(simplified) + _, refused = definition.read_configuration(simplified) if len(refused) != 0: msg = ( f"{definition.name!r}: its canonical gave {simplified!r}, which does not hold: " diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/definition.py b/packages/zarr-metadata/src/zarr_metadata/v3/definition.py index 75fc61b13b..7efdfbb506 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/definition.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/definition.py @@ -33,10 +33,10 @@ a value of the TypedDict or None, and every problem, each located. The value holds what the TypedDict admits and nothing else. A member typed with a field alias is checked as the JSON a metadata field is. -2. `definition.judge(configuration)` is the check, each nested field's - envelope judged -- a stray member, a `must_understand` of `false` -- +2. `definition.read_configuration(configuration)` is the configuration as + that definition reads it: type-checked, each nested field's envelope judged -- a stray member, a `must_understand` of `false` -- and then the rules, for one configuration: - `GZIP_CODEC.judge({"level": 12})`. + `GZIP_CODEC.read_configuration({"level": 12})`. 3. `resolve(field, CodecDefinition, CORE_AND_EXTENSIONS)` reads a whole field in a scope: its envelope judged, its name related to a definition, its configuration judged, and each nested field read the @@ -93,7 +93,7 @@ package supports. The rules are handed the configuration and the fields it holds as the scope read them: a field that is read keeps what it read inside it as `AcceptedField.nested`, a `Nested` mapping by where each sits, so a -struct's rules reach its field types. `judge`, which reads in no scope, +struct's rules reach its field types. `read_configuration`, which reads in no scope, hands them none. A rule's message shows a value as the package's own messages do, as JSON, with `shown`: `null`, `[1, 2]`, `"C"`. A rule reports where a problem is; what is found there is the problem's diff --git a/packages/zarr-metadata/tests/test_problem_data.py b/packages/zarr-metadata/tests/test_problem_data.py index f3fe5b3c1a..1fb1a9f611 100644 --- a/packages/zarr-metadata/tests/test_problem_data.py +++ b/packages/zarr-metadata/tests/test_problem_data.py @@ -219,7 +219,7 @@ def _raised(read: Callable[[], object]) -> Sequence[ValidationProblem]: READERS: list[tuple[Callable[[object], Sequence[ValidationProblem]], object]] = [ (lambda value: check(value, GzipLevelOnly)[1], {"level": 12, "extra": [1]}), - (lambda value: GZIP_CODEC.judge(value)[1], {"level": 12, "extra": [1]}), + (lambda value: GZIP_CODEC.read_configuration(value)[1], {"level": 12, "extra": [1]}), ( lambda value: resolve(value, CodecDefinition, CORE_AND_EXTENSIONS)[1], {"name": "gzip", "configuration": {"level": 12}, "must_understand": "yes"}, @@ -275,7 +275,7 @@ def _raised(read: Callable[[], object]) -> Sequence[ValidationProblem]: READERS, ids=[ "check", - "judge", + "read_configuration", "resolve", "fill-value-problems", "validate-json", diff --git a/packages/zarr-metadata/tests/v3/test_definitions.py b/packages/zarr-metadata/tests/v3/test_definitions.py index 85d71ca503..77af5e9d3d 100644 --- a/packages/zarr-metadata/tests/v3/test_definitions.py +++ b/packages/zarr-metadata/tests/v3/test_definitions.py @@ -738,7 +738,7 @@ def test_error_null_is_not_a_field() -> None: RefusedField(json=None, name=None, read_as=CodecDefinition), [((), "invalid_type")], ) - configuration, found = GZIP_CODEC.judge(None) + configuration, found = GZIP_CODEC.read_configuration(None) assert (configuration, _locs(found)) == (None, [((), "invalid_type")]) @@ -819,18 +819,18 @@ def test_error_a_nested_envelope_s_problem_is_its_own_and_the_rules_are_asked() def test_error_a_container_rule_is_not_asked_of_a_malformed_nested_field() -> None: # The stack's rule reads each nested field's name; one with no name is - # reported where it sits, and the rule is not asked, as `judge` would not. + # reported where it sits, and the rule is not asked, as `read_configuration` would not. field = {"name": "acme.stack", "configuration": {"codecs": [{"configuration": {}}]}} resolved, found = resolve(field, CodecDefinition, SCOPE) assert isinstance(resolved, RefusedField) assert _locs(found) == [(("configuration", "codecs", 0, "name"), "missing_key")] - assert ACME_STACK.judge(field["configuration"])[0] is None + assert ACME_STACK.read_configuration(field["configuration"])[0] is None def test_error_a_rule_about_the_whole_configuration_lands_on_it() -> None: # A rule reports relative to the configuration: an empty location is # the configuration, judged alone or read in a field. - _, judged = ACME_PAIRED.judge({"first": 1}) + _, judged = ACME_PAIRED.read_configuration({"first": 1}) _, read = resolve( {"name": "acme.paired", "configuration": {"first": 1}}, CodecDefinition, SCOPE ) @@ -840,13 +840,13 @@ def test_error_a_rule_about_the_whole_configuration_lands_on_it() -> None: def test_error_a_rule_reads_the_fields_the_configuration_holds_as_the_scope_read_them() -> None: # `bytes` is an array -> bytes codec, which the stack's rule refuses - # from what the scope read; `judge` reads in no scope, so its rule sees + # from what the scope read; `read_configuration` reads in no scope, so its rule sees # nothing read. field = {"name": "acme.stack", "configuration": {"codecs": ["crc32c", "bytes"]}} resolved, found = resolve(field, CodecDefinition, SCOPE) assert isinstance(resolved, RefusedField) assert _locs(found) == [(("configuration", "codecs", 1), "invalid_value")] - assert ACME_STACK.judge(field["configuration"])[1] == () + assert ACME_STACK.read_configuration(field["configuration"])[1] == () def test_error_a_nested_member_is_not_a_field() -> None: @@ -911,30 +911,33 @@ def test_error_check_judges_a_nested_envelope() -> None: assert [found.loc for found in problems] == [("codecs", 0, "configuration")] -def test_judge_leaves_out_a_member_a_nested_field_s_envelope_does_not_declare() -> None: +def test_read_configuration_leaves_out_a_member_a_nested_field_s_envelope_does_not_declare() -> ( + None +): # Reported as an unknown key and left out, as the checker leaves out a # key a closed TypedDict does not declare: the configuration comes back. - for given in (ACME_STACK.check, ACME_STACK.judge): - typed, problems = given({"codecs": [{"name": "crc32c", "x": 1}]}) - assert typed == {"codecs": ({"name": "crc32c"},)} - assert _locs(problems) == [(("codecs", 0, "x"), "unknown_key")] + typed, problems = ACME_STACK.read_configuration({"codecs": [{"name": "crc32c", "x": 1}]}) + assert typed == {"codecs": ({"name": "crc32c"},)} + assert _locs(problems) == [(("codecs", 0, "x"), "unknown_key")] -def test_error_judge_refuses_a_nested_field_that_need_not_be_understood() -> None: +def test_error_read_configuration_refuses_a_nested_field_that_need_not_be_understood() -> None: # A `must_understand` of false is no unknown key: the configuration # does not come back. - typed, problems = ACME_STACK.judge({"codecs": [{"name": "crc32c", "must_understand": False}]}) + typed, problems = ACME_STACK.read_configuration( + {"codecs": [{"name": "crc32c", "must_understand": False}]} + ) assert typed is None assert _locs(problems) == [(("codecs", 0, "must_understand"), "invalid_value")] -def test_judge_is_the_check_and_then_the_rules() -> None: +def test_read_configuration_is_the_check_and_then_the_rules() -> None: # A caller holding one configuration: the rules are asked only of a # configuration that type-checked, so they never meet a wrong type. - assert GZIP_CODEC.judge({"level": 5}) == ({"level": 5}, ()) - refused, problems = GZIP_CODEC.judge({"level": 12}) + assert GZIP_CODEC.read_configuration({"level": 5}) == ({"level": 5}, ()) + refused, problems = GZIP_CODEC.read_configuration({"level": 12}) assert (refused, _locs(problems)) == (None, [(("level",), "invalid_value")]) - mistyped, problems = GZIP_CODEC.judge({"level": "x"}) + mistyped, problems = GZIP_CODEC.read_configuration({"level": "x"}) assert (mistyped, _locs(problems)) == (None, [(("level",), "invalid_type")]) @@ -1386,7 +1389,7 @@ def writes( def test_error_a_function_that_writes_to_the_configuration_fails_on_every_path() -> None: - # `judge` hands the rules the same view `resolve` does, and `==` and + # `read_configuration` hands the rules the same view `resolve` does, and `==` and # `hash` hand `canonical` one, as `canonical_of` does. def writes( configuration: GzipCodecConfiguration, nested: Nested @@ -1395,7 +1398,7 @@ def writes( yield from () with pytest.raises(TypeError, match="does not support item assignment"): - dataclasses.replace(GZIP_CODEC, rules=writes).judge({"level": 1}) + dataclasses.replace(GZIP_CODEC, rules=writes).read_configuration({"level": 1}) def folds_in_place(configuration: GzipCodecConfiguration) -> GzipCodecConfiguration: cast("dict[str, object]", configuration)["level"] = 0 diff --git a/packages/zarr-metadata/tests/v3/test_every_definition.py b/packages/zarr-metadata/tests/v3/test_every_definition.py index a7981b6b31..2ea52fcc7f 100644 --- a/packages/zarr-metadata/tests/v3/test_every_definition.py +++ b/packages/zarr-metadata/tests/v3/test_every_definition.py @@ -879,7 +879,7 @@ def test_a_rule_is_a_function_over_the_typeddict() -> None: # configuration can ask them without a scope or a field around it. from zarr_metadata.v3.codec.blosc import BLOSC_CODEC - _, problems = BLOSC_CODEC.judge({**BLOSC, "clevel": 10}) + _, problems = BLOSC_CODEC.read_configuration({**BLOSC, "clevel": 10}) assert problems == ( ValidationProblem(("clevel",), "expected an integer in [0, 9], got 10", "invalid_value"), ) From 3c4be60161db24afd0c96eb63a8dfc9ba4aff9de Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Fri, 9 Oct 2026 21:28:20 +0200 Subject: [PATCH 86/94] feat(zarr-metadata)!: a scope is of one Zarr format, and a reader refuses a scope of the other A kind declares its format, class CodecDefinition(Definition[C], kind=True, format=3); Context.of refuses definitions of two formats and Context.format says which one a scope reads; ZarrV2Context and ZarrV3Context type the scopes, and every v2 and v3 entry point, resolve, and the pydantic bridge's bare-Context rule raise TypeError for a scope of the other format. A scope that files nothing is of no format and reads in either. Assisted-by: ClaudeCode:claude-fable-5-1 Co-Authored-By: Claude Fable 5.1 --- .../zarr-metadata/changes/4490.feature.md | 5 +- .../src/zarr_metadata/model/_array.py | 56 +++++--- .../src/zarr_metadata/model/_group.py | 100 +++++++------ .../src/zarr_metadata/model/_json_schema.py | 6 +- .../src/zarr_metadata/model/_repair.py | 10 +- .../src/zarr_metadata/model/_validation.py | 35 +++-- .../src/zarr_metadata/pydantic.py | 24 ++-- .../src/zarr_metadata/v2/_definition.py | 4 +- .../src/zarr_metadata/v2/definition.py | 16 ++- .../src/zarr_metadata/v3/_definition.py | 47 +++++-- .../src/zarr_metadata/v3/_registry.py | 78 +++++++++-- .../src/zarr_metadata/v3/definition.py | 16 ++- .../tests/model/test_scope_threading.py | 11 ++ .../tests/model/test_scope_threading_v2.py | 131 ++++++++++++++++++ .../zarr-metadata/tests/v2/test_definition.py | 19 +-- packages/zarr-metadata/tests/v3/test_kinds.py | 58 +++++++- 16 files changed, 469 insertions(+), 147 deletions(-) create mode 100644 packages/zarr-metadata/tests/model/test_scope_threading_v2.py diff --git a/packages/zarr-metadata/changes/4490.feature.md b/packages/zarr-metadata/changes/4490.feature.md index 851d7c76f3..3c52a10d36 100644 --- a/packages/zarr-metadata/changes/4490.feature.md +++ b/packages/zarr-metadata/changes/4490.feature.md @@ -11,8 +11,9 @@ scope (`CORE_AND_EXTENSIONS` or `CORE_V2` by default) and is never invalid; two models are equal when their documents mean the same; `update` reads new members, given as JSON, in the model's own scope, `with_context` and `refined_in` read the document in another, and -`refines` orders models by information. Scopes compare, join and report -disagreements, a group's consolidated metadata accepts node models, +`refines` orders models by information. A scope is of one Zarr format, `ZarrV2Context` or `ZarrV3Context`, and a +reader given a scope of the other format raises `TypeError`; scopes +compare, join and report disagreements, a group's consolidated metadata accepts node models, number types carry their bounds and problems carry `input` and `ctx`, JSON Schemas are exported for a TypedDict, a field and a `zarr.json`, and `read_repaired_node_metadata_v3` and diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_array.py b/packages/zarr-metadata/src/zarr_metadata/model/_array.py index 9ccd732ef5..fc14c77374 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_array.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/_array.py @@ -59,7 +59,13 @@ held, spelled_canonically, ) -from zarr_metadata.v3._registry import CORE_AND_EXTENSIONS, Context +from zarr_metadata.v3._registry import ( + CORE_AND_EXTENSIONS, + Context, + ZarrV2Context, + ZarrV3Context, + scoped, +) from zarr_metadata.v3._scope import Claims, Conflict, ScopeConflictError, claim_key, claims_of from zarr_metadata.v3._scope import refines as refines_field from zarr_metadata.v3.array import ZARR_V3_ARRAY_METADATA_STORE_KEY, ZarrV3ExtensionField @@ -147,8 +153,8 @@ def claims(self) -> Claims: """What the reading claimed of each name the document writes, keyed as the scope files it.""" return self._claims - def __init__(self, document: object, context: Context | None = None) -> None: - scope = CORE_AND_EXTENSIONS if context is None else context + def __init__(self, document: object, context: ZarrV3Context | None = None) -> None: + scope = scoped(context, CORE_AND_EXTENSIONS) reading, members = read_array_v3(document, scope) if members is None: raise MetadataValidationError(reading.problems) @@ -346,18 +352,18 @@ def update(self, **members: Unpack[ZarrV3ArrayMetadataUpdate]) -> ZarrV3ArrayMet del document[key] return type(self)(document, context=self._context) - def with_context(self, context: Context | None = None) -> ZarrV3ArrayMetadata: + def with_context(self, context: ZarrV3Context | None = None) -> ZarrV3ArrayMetadata: """This document read in `context`, whatever that changes: a gain, a loss, a conflict. `MetadataValidationError` when the document has a problem there. The reading is kept when `context` reads every claim identically. """ - scope = CORE_AND_EXTENSIONS if context is None else context + scope = scoped(context, CORE_AND_EXTENSIONS) if scope.disagreements(self._claims).agrees: return self._of(self._document, scope, self._reading, self._members) return type(self)(self._document, context=scope) - def refined_in(self, context: Context | None = None) -> ZarrV3ArrayMetadata: + def refined_in(self, context: ZarrV3Context | None = None) -> ZarrV3ArrayMetadata: """This document read in `context`, which may claim what this scope left unclaimed and contradict nothing. `ScopeConflictError` naming each name `context` reads by another @@ -367,7 +373,7 @@ def refined_in(self, context: Context | None = None) -> ZarrV3ArrayMetadata: claims refuses what was written under it: a gain can surface a problem. `with_context` reads the document in any scope. """ - scope = CORE_AND_EXTENSIONS if context is None else context + scope = scoped(context, CORE_AND_EXTENSIONS) found = scope.disagreements(self._claims) if len(found.conflicts) != 0: raise ScopeConflictError(located_conflicts(self._reading.fields(), found.conflicts)) @@ -400,7 +406,7 @@ def refines(self, other: ZarrV3ArrayMetadata) -> bool: def create_default( cls, *, - context: Context | None = None, + context: ZarrV3Context | None = None, **overrides: Unpack[ZarrV3ArrayMetadataJSONPartial], ) -> ZarrV3ArrayMetadata: """A scalar `uint8` array, or the one `overrides`, members of its document, make of it, read in `context`. @@ -435,7 +441,9 @@ def create_default( return cls({**document, **overrides}, context=context) @classmethod - def from_json(cls, data: object, *, context: Context | None = None) -> ZarrV3ArrayMetadata: + def from_json( + cls, data: object, *, context: ZarrV3Context | None = None + ) -> ZarrV3ArrayMetadata: """The model of `data`, a v3 array document read in `context`. `MetadataValidationError` with every problem the read finds. @@ -446,7 +454,7 @@ def from_json(cls, data: object, *, context: Context | None = None) -> ZarrV3Arr @classmethod def from_key_value( - cls, mapping: Mapping[StoreKey, bytes], *, context: Context | None = None + cls, mapping: Mapping[StoreKey, bytes], *, context: ZarrV3Context | None = None ) -> ZarrV3ArrayMetadata: """The model of the array document at `zarr.json` in `mapping`, read in `context`. @@ -471,7 +479,7 @@ def located_conflicts( def read_array_metadata_v3( - value: object, *, context: Context | None = None + value: object, *, context: ZarrV3Context | None = None ) -> ZarrV3ArrayMetadataReading: """`value`, a v3 array document, as `context` read it, whatever it holds. @@ -484,7 +492,7 @@ def read_array_metadata_v3( walk over its `fields()`. A value that is not an object holds no field. """ - scope = CORE_AND_EXTENSIONS if context is None else context + scope = scoped(context, CORE_AND_EXTENSIONS) reading, members = read_array_v3(value, scope) if members is None: return reading @@ -494,7 +502,7 @@ def read_array_metadata_v3( def read_array_metadata_v2( - value: object, *, context: Context | None = None + value: object, *, context: ZarrV2Context | None = None ) -> ZarrV2ArrayMetadataReading: """`value`, a v2 array document, as `context` read it, `CORE_V2` when none is given, whatever it holds. @@ -504,7 +512,7 @@ def read_array_metadata_v2( `validate_array_metadata_v2` finds, and, when there is none, the document's model. """ - scope = CORE_V2 if context is None else context + scope = scoped(context, CORE_V2) reading, members = read_array_v2(value, scope) if members is None: return reading @@ -563,8 +571,8 @@ def claims(self) -> Claims: """What the reading claimed of each typestr and codec id the document writes, keyed as the scope files them.""" return self._claims - def __init__(self, document: object, context: Context | None = None) -> None: - scope = CORE_V2 if context is None else context + def __init__(self, document: object, context: ZarrV2Context | None = None) -> None: + scope = scoped(context, CORE_V2) reading, members = read_array_v2(document, scope) if members is None: raise MetadataValidationError(reading.problems) @@ -768,18 +776,18 @@ def update(self, **members: Unpack[ZarrV2ArrayMetadataUpdate]) -> ZarrV2ArrayMet del document[key] return type(self)(document, context=self._context) - def with_context(self, context: Context | None = None) -> ZarrV2ArrayMetadata: + def with_context(self, context: ZarrV2Context | None = None) -> ZarrV2ArrayMetadata: """This document read in `context`, whatever that changes: a gain, a loss, a conflict. `MetadataValidationError` when the document has a problem there. The reading is kept when `context` reads every claim identically. """ - scope = CORE_V2 if context is None else context + scope = scoped(context, CORE_V2) if scope.disagreements(self._claims).agrees: return self._of(self._document, scope, self._reading, self._members) return type(self)(self._document, context=scope) - def refined_in(self, context: Context | None = None) -> ZarrV2ArrayMetadata: + def refined_in(self, context: ZarrV2Context | None = None) -> ZarrV2ArrayMetadata: """This document read in `context`, which may claim what this scope left unclaimed and contradict nothing. `ScopeConflictError` naming each typestr or id `context` reads by @@ -787,7 +795,7 @@ def refined_in(self, context: Context | None = None) -> ZarrV2ArrayMetadata: and where each sits in the document. `MetadataValidationError` when a definition `context` claims refuses what was written. """ - scope = CORE_V2 if context is None else context + scope = scoped(context, CORE_V2) found = scope.disagreements(self._claims) if len(found.conflicts) != 0: raise ScopeConflictError(located_conflicts(self._reading.fields(), found.conflicts)) @@ -818,7 +826,7 @@ def refines(self, other: ZarrV2ArrayMetadata) -> bool: @classmethod def create_default( - cls, *, context: Context | None = None, **overrides: Unpack[ZarrV2ArrayMetadataUpdate] + cls, *, context: ZarrV2Context | None = None, **overrides: Unpack[ZarrV2ArrayMetadataUpdate] ) -> ZarrV2ArrayMetadata: """A scalar `|u1` array, or the one `overrides`, members of its document, make of it, read in `context`. @@ -854,7 +862,9 @@ def create_default( return cls(merged, context=context) @classmethod - def from_json(cls, data: object, *, context: Context | None = None) -> ZarrV2ArrayMetadata: + def from_json( + cls, data: object, *, context: ZarrV2Context | None = None + ) -> ZarrV2ArrayMetadata: """The model of `data`, a v2 array document with its attributes under `attributes`, read in `context`. `MetadataValidationError` with every problem the read finds. @@ -865,7 +875,7 @@ def from_json(cls, data: object, *, context: Context | None = None) -> ZarrV2Arr @classmethod def from_key_value( - cls, mapping: Mapping[StoreKey, bytes], *, context: Context | None = None + cls, mapping: Mapping[StoreKey, bytes], *, context: ZarrV2Context | None = None ) -> ZarrV2ArrayMetadata: """The model of the array at `.zarray` in `mapping`, with the attributes at `.zattrs` when there is one, read in `context`. diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_group.py b/packages/zarr-metadata/src/zarr_metadata/model/_group.py index 2f2acc5279..dbf9b50671 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_group.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/_group.py @@ -69,7 +69,13 @@ from zarr_metadata.v2.definition import CORE_V2 from zarr_metadata.v2.group import ZARR_V2_GROUP_METADATA_STORE_KEY from zarr_metadata.v3._hierarchy import NodeType, hierarchy_problems, path_faults, said -from zarr_metadata.v3._registry import CORE_AND_EXTENSIONS, Context +from zarr_metadata.v3._registry import ( + CORE_AND_EXTENSIONS, + Context, + ZarrV2Context, + ZarrV3Context, + scoped, +) from zarr_metadata.v3._scope import Claims, Conflict, ScopeConflictError, claims_of, kind_name from zarr_metadata.v3.array import ZarrV3ExtensionField from zarr_metadata.v3.consolidated import ZARR_V3_CONSOLIDATED_METADATA_KEY @@ -155,8 +161,8 @@ def claims(self) -> Claims: """What the reading claimed of each name the document and its consolidated documents write, keyed as the scope files it.""" return self._claims - def __init__(self, document: object, context: Context | None = None) -> None: - scope = CORE_AND_EXTENSIONS if context is None else context + def __init__(self, document: object, context: ZarrV3Context | None = None) -> None: + scope = scoped(context, CORE_AND_EXTENSIONS) reading, members = read_group_v3(document, scope) if members is None or len(reading.problems) != 0: raise MetadataValidationError(reading.problems) @@ -307,14 +313,14 @@ def update(self, **members: Unpack[ZarrV3GroupMetadataUpdate]) -> ZarrV3GroupMet del document[key] return type(self)(document, context=self._context) - def with_context(self, context: Context | None = None) -> ZarrV3GroupMetadata: + def with_context(self, context: ZarrV3Context | None = None) -> ZarrV3GroupMetadata: """This document read in `context`, whatever that changes; `MetadataValidationError` when it has a problem there. The reading is kept when `context` reads every claim identically.""" - scope = CORE_AND_EXTENSIONS if context is None else context + scope = scoped(context, CORE_AND_EXTENSIONS) if scope.disagreements(self._claims).agrees: return self._of(self._document, scope, self._reading, self._members) return type(self)(self._document, context=scope) - def refined_in(self, context: Context | None = None) -> ZarrV3GroupMetadata: + def refined_in(self, context: ZarrV3Context | None = None) -> ZarrV3GroupMetadata: """This document read in `context`, which may claim what this scope left unclaimed and contradict nothing. `ScopeConflictError` naming each name `context` reads by another @@ -322,7 +328,7 @@ def refined_in(self, context: Context | None = None) -> ZarrV3GroupMetadata: consolidated metadata holds too; `MetadataValidationError` when a name `context` claims refuses what was written under it. """ - scope = CORE_AND_EXTENSIONS if context is None else context + scope = scoped(context, CORE_AND_EXTENSIONS) found = scope.disagreements(self._claims) if len(found.conflicts) != 0: raise ScopeConflictError(located_conflicts(self._reading.fields(), found.conflicts)) @@ -348,14 +354,16 @@ def refines(self, other: ZarrV3GroupMetadata) -> bool: def create_default( cls, *, - context: Context | None = None, + context: ZarrV3Context | None = None, **members: Unpack[ZarrV3GroupMetadataJSONPartial], ) -> ZarrV3GroupMetadata: """A group with no attributes, or the one `members` of its document make of it, read in `context`; `MetadataValidationError` when its document has a problem.""" return cls({"zarr_format": 3, "node_type": "group", **members}, context=context) @classmethod - def from_json(cls, data: object, *, context: Context | None = None) -> ZarrV3GroupMetadata: + def from_json( + cls, data: object, *, context: ZarrV3Context | None = None + ) -> ZarrV3GroupMetadata: """The model of `data`, a v3 group document read in `context`, with each document its consolidated metadata holds. `MetadataValidationError` with every problem the read finds. A @@ -369,7 +377,7 @@ def from_json(cls, data: object, *, context: Context | None = None) -> ZarrV3Gro @classmethod def from_key_value( - cls, mapping: Mapping[StoreKey, bytes], *, context: Context | None = None + cls, mapping: Mapping[StoreKey, bytes], *, context: ZarrV3Context | None = None ) -> ZarrV3GroupMetadata: """The model of the group document at `zarr.json` in `mapping`, read in `context`. @@ -398,8 +406,8 @@ class ZarrV3ConsolidatedMetadata(Keyed): kind: Final = "inline" must_understand: Final = False - def __init__(self, member: object, context: Context | None = None) -> None: - scope = CORE_AND_EXTENSIONS if context is None else context + def __init__(self, member: object, context: ZarrV3Context | None = None) -> None: + scope = scoped(context, CORE_AND_EXTENSIONS) # The member sits under a group's key wherever it is read, so the # levels a reader walks are counted from there, as in the group. readings, members, problems = _read_consolidated_v3( @@ -474,7 +482,7 @@ def refines(self, other: ZarrV3ConsolidatedMetadata) -> bool: @classmethod def from_json( - cls, data: object, *, context: Context | None = None + cls, data: object, *, context: ZarrV3Context | None = None ) -> ZarrV3ConsolidatedMetadata: """The model of `data`, a group's `consolidated_metadata` member, each document read once in `context`, as the array or group its `node_type` says; `MetadataValidationError` with every problem found.""" return cls(data, context=context) @@ -560,7 +568,7 @@ def fields(self) -> Iterator[tuple[Loc, ResolvedField[Any]]]: def read_node_metadata_v3( - value: object, *, context: Context | None = None + value: object, *, context: ZarrV3Context | None = None ) -> ZarrV3NodeMetadataReading: """`value`, a v3 `zarr.json`, read in `context` as the node its `node_type` says it is. @@ -572,7 +580,7 @@ def read_node_metadata_v3( among them. So no caller reads `node_type` from JSON it has not read, and a document of another format says it is not v3. """ - scope = CORE_AND_EXTENSIONS if context is None else context + scope = scoped(context, CORE_AND_EXTENSIONS) node_type, problems = _node_type(value) if node_type == "array": return read_array_metadata_v3(value, context=scope) @@ -588,7 +596,7 @@ def read_node_metadata_v3( def node_metadata_from_json_v3( - data: object, *, context: Context | None = None + data: object, *, context: ZarrV3Context | None = None ) -> ZarrV3NodeMetadata: """The model of `data`, a v3 `zarr.json` read in `context`, as the node its `node_type` says. @@ -597,7 +605,7 @@ def node_metadata_from_json_v3( `MetadataValidationError` with every problem `read_node_metadata_v3` finds, a `node_type` that says neither among them. """ - scope = CORE_AND_EXTENSIONS if context is None else context + scope = scoped(context, CORE_AND_EXTENSIONS) reading = read_node_metadata_v3(data, context=scope) if reading.metadata is None: raise MetadataValidationError(reading.problems) @@ -605,24 +613,24 @@ def node_metadata_from_json_v3( def node_metadata_from_key_value_v3( - mapping: Mapping[StoreKey, bytes], *, context: Context | None = None + mapping: Mapping[StoreKey, bytes], *, context: ZarrV3Context | None = None ) -> ZarrV3NodeMetadata: """The model of the document at `zarr.json` in `mapping`, read in `context` as the node its `node_type` says, as `node_metadata_from_json_v3` reads one. `MetadataValidationError` when the key is missing, its bytes are not JSON, or the document is not a valid array or group. """ - scope = CORE_AND_EXTENSIONS if context is None else context + scope = scoped(context, CORE_AND_EXTENSIONS) # An array's document and a group's are both at `zarr.json`. document = load_store_json(mapping, ZARR_V3_GROUP_METADATA_STORE_KEY) return node_metadata_from_json_v3(document, context=scope) def validate_node_metadata_v3( - value: object, *, context: Context | None = None + value: object, *, context: ZarrV3Context | None = None ) -> tuple[ValidationProblem, ...]: """Every reason `value` is not a valid v3 `zarr.json`: those `validate_array_metadata_v3` or `validate_group_metadata_v3` finds in the node its `node_type` says it is, or why it says neither.""" - scope = CORE_AND_EXTENSIONS if context is None else context + scope = scoped(context, CORE_AND_EXTENSIONS) return _read_node_v3(value, scope)[0].problems @@ -686,7 +694,7 @@ class GroupMembersV3: def read_group_metadata_v3( - value: object, *, context: Context | None = None + value: object, *, context: ZarrV3Context | None = None ) -> ZarrV3GroupMetadataReading: """`value`, a v3 group document, as `context` read it, whatever it holds. @@ -697,7 +705,7 @@ def read_group_metadata_v3( models of those documents, which their readings hold too. A value that is not an object holds nothing. """ - scope = CORE_AND_EXTENSIONS if context is None else context + scope = scoped(context, CORE_AND_EXTENSIONS) reading, members = read_group_v3(value, scope) if members is None: return reading @@ -1136,7 +1144,7 @@ def _nested_models( def validate_group_metadata_v3( - value: object, *, context: Context | None = None + value: object, *, context: ZarrV3Context | None = None ) -> tuple[ValidationProblem, ...]: """Return every reason `value` is not a valid v3 group document. @@ -1149,25 +1157,25 @@ def validate_group_metadata_v3( `problems` of `read_group_metadata_v3`, which holds what was read to find them. """ - scope = CORE_AND_EXTENSIONS if context is None else context + scope = scoped(context, CORE_AND_EXTENSIONS) return read_group_v3(value, scope)[0].problems def is_group_metadata_v3( - value: object, *, context: Context | None = None + value: object, *, context: ZarrV3Context | None = None ) -> TypeGuard[ZarrV3GroupMetadataJSON]: """Whether `value` is a v3 group document `validate_group_metadata_v3` finds nothing wrong with, written with tuples.""" - scope = CORE_AND_EXTENSIONS if context is None else context + scope = scoped(context, CORE_AND_EXTENSIONS) return is_canonical_json(value, finite=False) and not validate_group_metadata_v3( value, context=scope ) def parse_group_metadata_v3( - value: object, *, context: Context | None = None + value: object, *, context: ZarrV3Context | None = None ) -> ZarrV3GroupMetadataJSON: """Return `value` narrowed to `ZarrV3GroupMetadataJSON`, or raise `MetadataValidationError`.""" - scope = CORE_AND_EXTENSIONS if context is None else context + scope = scoped(context, CORE_AND_EXTENSIONS) problems = validate_group_metadata_v3(value, context=scope) if len(problems) != 0: raise MetadataValidationError(problems) @@ -1208,8 +1216,8 @@ def claims(self) -> Claims: """What the reading claimed: nothing, since a group holds no field.""" return MappingProxyType({}) - def __init__(self, document: object, context: Context | None = None) -> None: - scope = CORE_V2 if context is None else context + def __init__(self, document: object, context: ZarrV2Context | None = None) -> None: + scope = scoped(context, CORE_V2) parsed = parse_group_metadata_v2(document, context=scope) self._adopt(refined_object(document), scope, _v2_attributes(parsed)) @@ -1290,12 +1298,12 @@ def update(self, **members: Unpack[ZarrV2GroupMetadataUpdate]) -> ZarrV2GroupMet del document[key] return type(self)(document, context=self._context) - def with_context(self, context: Context | None = None) -> ZarrV2GroupMetadata: + def with_context(self, context: ZarrV2Context | None = None) -> ZarrV2GroupMetadata: """This document read in `context`: the same group, holding that scope.""" - scope = CORE_V2 if context is None else context + scope = scoped(context, CORE_V2) return self._of(self._document, scope, self._attributes) - def refined_in(self, context: Context | None = None) -> ZarrV2GroupMetadata: + def refined_in(self, context: ZarrV2Context | None = None) -> ZarrV2GroupMetadata: """This document read in `context`: a group holds no field, so no scope conflicts with its reading, and this is `with_context`.""" return self.with_context(context) @@ -1305,20 +1313,22 @@ def refines(self, other: ZarrV2GroupMetadata) -> bool: @classmethod def create_default( - cls, *, context: Context | None = None, **overrides: Unpack[ZarrV2GroupMetadataUpdate] + cls, *, context: ZarrV2Context | None = None, **overrides: Unpack[ZarrV2GroupMetadataUpdate] ) -> ZarrV2GroupMetadata: """A group with no `.zattrs`, or the one `overrides` make of it, read in `context`; `MetadataValidationError` when the document they make has a problem.""" given = {key: value for key, value in overrides.items() if value is not UNSET} return cls({"zarr_format": 2, **given}, context=context) @classmethod - def from_json(cls, data: object, *, context: Context | None = None) -> ZarrV2GroupMetadata: + def from_json( + cls, data: object, *, context: ZarrV2Context | None = None + ) -> ZarrV2GroupMetadata: """The model of `data`, a v2 group document with its attributes under `attributes`, read in `context`; `MetadataValidationError` with every problem.""" return cls(data, context=context) @classmethod def from_key_value( - cls, mapping: Mapping[StoreKey, bytes], *, context: Context | None = None + cls, mapping: Mapping[StoreKey, bytes], *, context: ZarrV2Context | None = None ) -> ZarrV2GroupMetadata: """The model of the group at `.zgroup` in `mapping`, with the attributes at `.zattrs` when there is one, read in `context`. @@ -1370,8 +1380,8 @@ class ZarrV2ConsolidatedMetadata(Keyed): zarr_consolidated_format: Final = 1 - def __init__(self, document: object, context: Context | None = None) -> None: - scope = CORE_V2 if context is None else context + def __init__(self, document: object, context: ZarrV2Context | None = None) -> None: + scope = scoped(context, CORE_V2) entries, nodes, problems = _read_consolidated_v2(document, scope) if len(problems) != 0: raise MetadataValidationError(problems) @@ -1445,18 +1455,18 @@ def _other_entries_text(self) -> str: def __reduce__(self) -> tuple[type[ZarrV2ConsolidatedMetadata], tuple[object, Context]]: return type(self), (self._document, self._context) - def with_context(self, context: Context | None = None) -> ZarrV2ConsolidatedMetadata: + def with_context(self, context: ZarrV2Context | None = None) -> ZarrV2ConsolidatedMetadata: """This document with every node read in `context`, whatever that changes; `MetadataValidationError` when a node has a problem there.""" - scope = CORE_V2 if context is None else context + scope = scoped(context, CORE_V2) return type(self)(self._document, context=scope) - def refined_in(self, context: Context | None = None) -> ZarrV2ConsolidatedMetadata: + def refined_in(self, context: ZarrV2Context | None = None) -> ZarrV2ConsolidatedMetadata: """This document with every node read in `context`, which may claim what this scope left unclaimed and contradict nothing. `ScopeConflictError` naming each conflict, located at the node's entry; `MetadataValidationError` when a gain surfaces a problem. """ - scope = CORE_V2 if context is None else context + scope = scoped(context, CORE_V2) conflicts: list[Conflict] = [] entries = object_at(self._document, "metadata") by_path, _ = _entries_by_path(entries) @@ -1488,14 +1498,14 @@ def refines(self, other: ZarrV2ConsolidatedMetadata) -> bool: @classmethod def from_json( - cls, data: object, *, context: Context | None = None + cls, data: object, *, context: ZarrV2Context | None = None ) -> ZarrV2ConsolidatedMetadata: """The model of `data`, a `.zmetadata` document, its nodes read in `context`; `MetadataValidationError` with every problem.""" return cls(data, context=context) @classmethod def from_key_value( - cls, mapping: Mapping[StoreKey, bytes], *, context: Context | None = None + cls, mapping: Mapping[StoreKey, bytes], *, context: ZarrV2Context | None = None ) -> ZarrV2ConsolidatedMetadata: """The model of the document at `.zmetadata` in `mapping`, read in `context`. diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_json_schema.py b/packages/zarr-metadata/src/zarr_metadata/model/_json_schema.py index 936ab44c87..0550e430e6 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_json_schema.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/_json_schema.py @@ -6,7 +6,7 @@ from zarr_metadata._typed_json import Schemas from zarr_metadata.v3._definition import DataTypeDefinition, field_schemas, written_name -from zarr_metadata.v3._registry import CORE_AND_EXTENSIONS, Context +from zarr_metadata.v3._registry import CORE_AND_EXTENSIONS, Context, ZarrV3Context, scoped from zarr_metadata.v3.array import ZarrV3ArrayMetadataJSON from zarr_metadata.v3.consolidated import ( ZARR_V3_CONSOLIDATED_METADATA_KEY, @@ -19,7 +19,7 @@ from zarr_metadata._typed_json import JSONSchema, SchemaLeaf -def node_metadata_json_schema_v3(*, context: Context | None = None) -> JSONSchema: +def node_metadata_json_schema_v3(*, context: ZarrV3Context | None = None) -> JSONSchema: """The JSON Schema of a v3 `zarr.json` read in `context`: an array document or a group document, as `validate_node_metadata_v3` reads one, but for the rules. For an editor that validates a `zarr.json` as it is written, or a @@ -45,7 +45,7 @@ def node_metadata_json_schema_v3(*, context: Context | None = None) -> JSONSchem as lists: a model's `to_json` writes tuples, which a Python validator does not take for arrays. """ - scope = CORE_AND_EXTENSIONS if context is None else context + scope = scoped(context, CORE_AND_EXTENSIONS) schemas = Schemas(_documents(scope)) array = schemas.of(ZarrV3ArrayMetadataJSON) group = schemas.of(ZarrV3GroupMetadataJSON) diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_repair.py b/packages/zarr-metadata/src/zarr_metadata/model/_repair.py index 75f532ea39..7089b1e41d 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_repair.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/_repair.py @@ -39,7 +39,7 @@ ) from zarr_metadata.v2.definition import CORE_V2 from zarr_metadata.v2.group import ZARR_V2_GROUP_METADATA_STORE_KEY -from zarr_metadata.v3._registry import CORE_AND_EXTENSIONS, Context +from zarr_metadata.v3._registry import CORE_AND_EXTENSIONS, ZarrV2Context, ZarrV3Context, scoped from zarr_metadata.v3.consolidated import ZARR_V3_CONSOLIDATED_METADATA_KEY RepairKind: TypeAlias = Literal[ @@ -222,7 +222,7 @@ class ZarrV3RepairedNodeMetadataReading: def read_repaired_node_metadata_v3( - value: object, *, context: Context | None = None + value: object, *, context: ZarrV3Context | None = None ) -> ZarrV3RepairedNodeMetadataReading: """`value`, a v3 `zarr.json`, read in `context` as `read_node_metadata_v3` reads it, once `repair_node_metadata_v3` has undone each known writer bug in it. @@ -231,7 +231,7 @@ def read_repaired_node_metadata_v3( applies to is read as it is, and reported as `read_node_metadata_v3` reports it. """ - scope = CORE_AND_EXTENSIONS if context is None else context + scope = scoped(context, CORE_AND_EXTENSIONS) repaired, repairs = repair_node_metadata_v3(value) return ZarrV3RepairedNodeMetadataReading( read_node_metadata_v3(repaired, context=scope), repairs @@ -312,7 +312,7 @@ class ZarrV2RepairedConsolidatedMetadataReading: def read_repaired_consolidated_metadata_v2( - value: object, *, context: Context | None = None + value: object, *, context: ZarrV2Context | None = None ) -> ZarrV2RepairedConsolidatedMetadataReading: """`value`, a v2 `.zmetadata`, read in `context` as `ZarrV2ConsolidatedMetadata` reads it, once `repair_consolidated_metadata_v2` has undone each known writer bug in it. @@ -320,7 +320,7 @@ def read_repaired_consolidated_metadata_v2( calling this rather than the strict model. Whatever no repair applies to is read as it is, and reported as the strict read reports it. """ - scope = CORE_V2 if context is None else context + scope = scoped(context, CORE_V2) repaired, repairs = repair_consolidated_metadata_v2(value) try: model: ZarrV2ConsolidatedMetadata | None = ZarrV2ConsolidatedMetadata(repaired, scope) diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_validation.py b/packages/zarr-metadata/src/zarr_metadata/model/_validation.py index c59e5c8f72..1831a2023c 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_validation.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/_validation.py @@ -83,7 +83,13 @@ resolve, ) from zarr_metadata.v3._pipeline import Stage, read_pipeline -from zarr_metadata.v3._registry import CORE_AND_EXTENSIONS, Context +from zarr_metadata.v3._registry import ( + CORE_AND_EXTENSIONS, + Context, + ZarrV2Context, + ZarrV3Context, + scoped, +) from zarr_metadata.v3.array import ZarrV3ArrayMetadataJSON from zarr_metadata.v3.group import ZarrV3GroupMetadataJSON @@ -638,7 +644,7 @@ def read_array_v3( def validate_array_metadata_v3( - value: object, *, context: Context | None = None + value: object, *, context: ZarrV3Context | None = None ) -> tuple[ValidationProblem, ...]: """Return every reason `value` is not a valid v3 array document. @@ -660,15 +666,15 @@ def validate_array_metadata_v3( reports as `must_understand_fields`. These are the `problems` of `read_array_metadata_v3`, which holds what was read to find them. """ - scope = CORE_AND_EXTENSIONS if context is None else context + scope = scoped(context, CORE_AND_EXTENSIONS) return read_array_v3(value, scope)[0].problems def is_array_metadata_v3( - value: object, *, context: Context | None = None + value: object, *, context: ZarrV3Context | None = None ) -> TypeGuard[ZarrV3ArrayMetadataJSON]: """Whether `value` is a v3 array document `validate_array_metadata_v3` finds nothing wrong with, written with tuples.""" - scope = CORE_AND_EXTENSIONS if context is None else context + scope = scoped(context, CORE_AND_EXTENSIONS) return ( _is_canonical_json(value, finite=False) and not validate_array_metadata_v3(value, context=scope) @@ -677,10 +683,10 @@ def is_array_metadata_v3( def parse_array_metadata_v3( - value: object, *, context: Context | None = None + value: object, *, context: ZarrV3Context | None = None ) -> ZarrV3ArrayMetadataJSON: """Return `value` as `ZarrV3ArrayMetadataJSON`, or raise `MetadataValidationError`.""" - scope = CORE_AND_EXTENSIONS if context is None else context + scope = scoped(context, CORE_AND_EXTENSIONS) problems = validate_array_metadata_v3(value, context=scope) if len(problems) != 0: raise MetadataValidationError(problems) @@ -797,7 +803,7 @@ def read_array_v2( def validate_array_metadata_v2( - value: object, *, context: Context | None = None + value: object, *, context: ZarrV2Context | None = None ) -> tuple[ValidationProblem, ...]: """Every reason `value` is not a valid v2 array document, read in `context`, `CORE_V2` when none is given. @@ -805,11 +811,11 @@ def validate_array_metadata_v2( codec the scope refuses is a problem, one it does not claim is not; `fill_value` is judged by the dtype the scope read. """ - return read_array_v2(value, CORE_V2 if context is None else context)[0].problems + return read_array_v2(value, scoped(context, CORE_V2))[0].problems def is_array_metadata_v2( - value: object, *, context: Context | None = None + value: object, *, context: ZarrV2Context | None = None ) -> TypeGuard[ZarrV2ArrayMetadataJSON]: """Whether `value` is a valid v2 array metadata document, read in `context`, `CORE_V2` when none is given.""" return ( @@ -820,7 +826,7 @@ def is_array_metadata_v2( def parse_array_metadata_v2( - value: object, *, context: Context | None = None + value: object, *, context: ZarrV2Context | None = None ) -> ZarrV2ArrayMetadataJSON: """`value` as `ZarrV2ArrayMetadataJSON`, read in `context`, `CORE_V2` when none is given; `MetadataValidationError` with every problem.""" problems = validate_array_metadata_v2(value, context=context) @@ -830,7 +836,7 @@ def parse_array_metadata_v2( def validate_group_metadata_v2( - value: object, *, context: Context | None = None + value: object, *, context: ZarrV2Context | None = None ) -> tuple[ValidationProblem, ...]: """Return every reason `value` is not a structurally-valid v2 group doc. @@ -838,6 +844,7 @@ def validate_group_metadata_v2( optional `attributes` mapping folded in from `.zattrs`. A group holds no field a scope reads; `context` is taken as every v2 reader takes it. """ + scoped(context, CORE_V2) if not is_object(value): return not_an_object(value) doc = value @@ -850,7 +857,7 @@ def validate_group_metadata_v2( def is_group_metadata_v2( - value: object, *, context: Context | None = None + value: object, *, context: ZarrV2Context | None = None ) -> TypeGuard[ZarrV2GroupMetadataJSON]: """Whether `value` is a structurally-valid v2 group metadata document; `context` is taken as every v2 reader takes it.""" return _is_canonical_json(value, finite=False) and not validate_group_metadata_v2( @@ -859,7 +866,7 @@ def is_group_metadata_v2( def parse_group_metadata_v2( - value: object, *, context: Context | None = None + value: object, *, context: ZarrV2Context | None = None ) -> ZarrV2GroupMetadataJSON: """`value` narrowed to `ZarrV2GroupMetadataJSON`, or `MetadataValidationError`; `context` is taken as every v2 reader takes it.""" problems = validate_group_metadata_v2(value, context=context) diff --git a/packages/zarr-metadata/src/zarr_metadata/pydantic.py b/packages/zarr-metadata/src/zarr_metadata/pydantic.py index 01e972042f..ccb8bddc76 100644 --- a/packages/zarr-metadata/src/zarr_metadata/pydantic.py +++ b/packages/zarr-metadata/src/zarr_metadata/pydantic.py @@ -11,7 +11,8 @@ field-level coercion can never bypass it). A field type reads the fields of its document in the scope pydantic's validation context holds, as pydantic hands any validator its context: the context itself, when it is -a `Context`, which every field type then reads in; or, when it is a +a `Context`, which every field type of its format then reads in, the +other format's refusing it with `TypeError`; or, when it is a mapping, its `"zarr_metadata_context"` item for the v3 field types and its `"zarr_metadata_context_v2"` item for the v2 ones, so a model holding both kinds of field names each format's scope; a format whose scope is @@ -51,7 +52,7 @@ class ArrayManifest(BaseModel): from __future__ import annotations -from typing import TYPE_CHECKING, Annotated, Final, LiteralString, Protocol, TypeVar, cast +from typing import TYPE_CHECKING, Annotated, Any, Final, LiteralString, Protocol, TypeVar, cast from pydantic import BeforeValidator, InstanceOf, PlainSerializer, ValidationInfo from pydantic_core import InitErrorDetails, PydanticCustomError, ValidationError @@ -78,7 +79,7 @@ class ArrayManifest(BaseModel): ) from zarr_metadata._sentinel import UNSET from zarr_metadata.v2.definition import CORE_V2 -from zarr_metadata.v3._registry import CORE_AND_EXTENSIONS, Context +from zarr_metadata.v3._registry import CORE_AND_EXTENSIONS, Context, is_scope, scoped if TYPE_CHECKING: from collections.abc import Callable @@ -147,7 +148,7 @@ def __call__(self, data: object, /, *, context: Context) -> _Read_co: ... def _read_in_scope( - cls: type[_M], read: _Reads[_M], default: Context, key: str + cls: type[_M], read: _Reads[_M], default: Context[Any], key: str ) -> Callable[[object, ValidationInfo], _M]: """A validator that passes instances of `cls` through and reads anything else in the scope the validation context holds under `key`, `default` when it holds none.""" @@ -161,14 +162,19 @@ def coerce(value: object, info: ValidationInfo) -> _M: return coerce -def _scope(context: object, default: Context, key: str) -> Context: - """The scope a validation context holds for one format: itself, a `Context`, the scope of every field type; its `key` item, the format's own; or, holding none, `default`, the format's core scope.""" - if isinstance(context, Context): - return context +def _scope(context: object, default: Context[Any], key: str) -> Context[Any]: + """The scope a validation context holds for one format: itself, a `Context`, the scope of every field type of its format; its `key` item, the format's own; or, holding none, `default`, the format's core scope. + + A bare `Context` of the other format is a `TypeError`, as it is at + every reader: a model holding fields of both formats names each + format's scope by its key. + """ + if is_scope(context): + return scoped(context, default) if not is_object(context) or key not in context: return default scope = context[key] - if not isinstance(scope, Context): + if not is_scope(scope): msg = f"{key}: the scope to read in is a Context, got {scope!r}" raise TypeError(msg) return scope diff --git a/packages/zarr-metadata/src/zarr_metadata/v2/_definition.py b/packages/zarr-metadata/src/zarr_metadata/v2/_definition.py index f83e550036..87dba74dc0 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v2/_definition.py +++ b/packages/zarr-metadata/src/zarr_metadata/v2/_definition.py @@ -106,7 +106,7 @@ def typestr_problem(name: str, at: Loc) -> ValidationProblem | None: @dataclass(frozen=True, kw_only=True, slots=True, repr=False) -class ZarrV2DataTypeDefinition(WithFillValue[C], kind=True): +class ZarrV2DataTypeDefinition(WithFillValue[C], kind=True, format=2): """A v2 data type: one family of NumPy types, and the fill value an array of it takes. Filed under the family -- `float` -- and read for every typestr of @@ -200,7 +200,7 @@ def name_loc(cls, loc: Loc) -> Loc: @dataclass(frozen=True, kw_only=True, slots=True, repr=False) -class ZarrV2CodecDefinition(Definition[C], kind=True): +class ZarrV2CodecDefinition(Definition[C], kind=True, format=2): """A v2 codec: a numcodecs id, and the TypedDict its parameters are. A document writes `{"id": name, **parameters}`; the definition's diff --git a/packages/zarr-metadata/src/zarr_metadata/v2/definition.py b/packages/zarr-metadata/src/zarr_metadata/v2/definition.py index 08230a180e..7398dc5b10 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v2/definition.py +++ b/packages/zarr-metadata/src/zarr_metadata/v2/definition.py @@ -8,7 +8,8 @@ in it with `resolve_dtype_v2` or `resolve_codec_v2`, which give `AcceptedField`, `UnclaimedField` or `RefusedField` as `zarr_metadata.v3.definition.resolve` does for a v3 field; the scope algebra -- `Context.of`, `extended_with`, -`joined`, `claimant` -- is the same `Context`. +`joined`, `claimant` -- is the same `Context`, of format 2: a v2 reader +refuses a v3 scope with `TypeError`, and `ZarrV2Context` is its type. A dtype reads as its family, the typestr's byte order, size and unit its configuration: ` tuple[ResolvedField[ZarrV2DataTypeDefinition[Any]], Problems]: """`value`, a v2 `dtype`, read in `context`, `CORE_V2` when none is given: what the scope made of it, and every problem, each prefixed with `loc`.""" - return resolve(value, ZarrV2DataTypeDefinition, CORE_V2 if context is None else context, loc) + return resolve(value, ZarrV2DataTypeDefinition, scoped(context, CORE_V2), loc) def resolve_codec_v2( - value: object, context: Context | None = None, loc: Loc = () + value: object, context: ZarrV2Context | None = None, loc: Loc = () ) -> tuple[ResolvedField[ZarrV2CodecDefinition[Any]], Problems]: """`value`, a v2 `compressor` or one of its `filters`, read in `context`, `CORE_V2` when none is given: what the scope made of it, and every problem, each prefixed with `loc`.""" - return resolve(value, ZarrV2CodecDefinition, CORE_V2 if context is None else context, loc) + return resolve(value, ZarrV2CodecDefinition, scoped(context, CORE_V2), loc) __all__ = [ @@ -84,6 +85,7 @@ def resolve_codec_v2( "ScopeConflictError", "UnclaimedField", "ZarrV2CodecDefinition", + "ZarrV2Context", "ZarrV2DataTypeDefinition", "ZarrV2DataTypeField", "canonical_fill_value", diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py b/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py index 815f9dedd4..1930e8ea2d 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py @@ -213,11 +213,13 @@ class Definition(Generic[C]): """ _is_kind: ClassVar[bool] = False - """Whether this class is a kind, what a scope files definitions by: set by `class Kind(Definition, kind=True)`. + """Whether this class is a kind, what a scope files definitions by: set by `class Kind(Definition, kind=True, format=3)`. Read from a class's own namespace, never inherited: a subclass of a kind is a definition of that kind, and declares no `kind`. """ + _format: ClassVar[Literal[2, 3] | None] = None + """The Zarr format whose documents hold fields of this kind, declared with the kind and inherited by its definitions: what a scope reads documents of.""" label: ClassVar[str] = "definition" """The kind as a message names it: "codec".""" field_aliases: ClassVar[tuple[TypeAliasType, ...]] = () @@ -290,8 +292,10 @@ def name_loc(cls, loc: Loc) -> Loc: """Where the name of a field at `loc`, written as an object, sits: under `name` for v3.""" return (*loc, "name") - def __init_subclass__(cls, *, kind: bool = False, **kwargs: object) -> None: - """Files a subclass: `kind=True` declares a kind, as `typing.Protocol` and SQLAlchemy's `__abstract__` mark a class and not its subclasses.""" + def __init_subclass__( + cls, *, kind: bool = False, format: Literal[2, 3] | None = None, **kwargs: object + ) -> None: + """Files a subclass: `kind=True` declares a kind, of the Zarr `format` whose documents hold its fields, as `typing.Protocol` and SQLAlchemy's `__abstract__` mark a class and not its subclasses.""" # Named, not `super()`: a dataclass with slots is rebuilt, and the # cell a bare `super()` reads names the class that was thrown away. super(Definition, cls).__init_subclass__(**kwargs) @@ -300,7 +304,17 @@ def __init_subclass__(cls, *, kind: bool = False, **kwargs: object) -> None: # class built last is the one a document is read with: the mark is # kept in the namespace, and the aliases are filed last. if kind: + if format not in (2, 3): + msg = ( + f"{cls.__name__}: a kind declares the Zarr format its fields belong to, " + f"format=2 or format=3, got {format!r}" + ) + raise TypeError(msg) cls._is_kind = True + cls._format = format + elif format is not None: + msg = f"{cls.__name__}: only a kind declares a format; a definition of a kind has its kind's" + raise TypeError(msg) for alias in cls.__dict__.get("field_aliases", ()): _FIELD_KINDS[alias] = cls @@ -456,7 +470,7 @@ def _refusal(self) -> str | None: @dataclass(frozen=True, kw_only=True, slots=True, repr=False) -class DataTypeDefinition(WithFillValue[C], kind=True): +class DataTypeDefinition(WithFillValue[C], kind=True, format=3): """A data type, and the fill value an array of it takes. `fill_value` is the JSON shape of a fill value -- `Int8FillValue`, an @@ -533,7 +547,7 @@ def _fill_value_parser(annotation: object) -> Parser: @dataclass(frozen=True, kw_only=True, slots=True, repr=False) -class ChunkGridDefinition(Definition[C], kind=True): +class ChunkGridDefinition(Definition[C], kind=True, format=3): """A chunk grid, and the arrays it fits. `shape_rules` is what the spec disallows in a grid of this @@ -561,7 +575,7 @@ class ChunkGridDefinition(Definition[C], kind=True): @dataclass(frozen=True, kw_only=True, slots=True, repr=False) -class ChunkKeyEncodingDefinition(Definition[C], kind=True): +class ChunkKeyEncodingDefinition(Definition[C], kind=True, format=3): """A chunk key encoding.""" label: ClassVar[str] = "chunk key encoding" @@ -587,7 +601,7 @@ class ChunkKeyEncodingDefinition(Definition[C], kind=True): @dataclass(frozen=True, kw_only=True, slots=True, repr=False) -class CodecDefinition(Definition[C], kind=True): +class CodecDefinition(Definition[C], kind=True, format=3): """A codec: what it does to what it is handed, and whether the size of what it gives out is static. A codec handed an array -- array -> array, array -> bytes -- says what @@ -645,7 +659,7 @@ def _refusal(self) -> str | None: @dataclass(frozen=True, kw_only=True, slots=True, repr=False) -class StorageTransformerDefinition(Definition[C], kind=True): +class StorageTransformerDefinition(Definition[C], kind=True, format=3): """A storage transformer.""" label: ClassVar[str] = "storage transformer" @@ -668,6 +682,15 @@ def kind_of(definition: Definition[Any]) -> type[Definition[Any]] | None: return next((base for base in bases if _declares_kind(base)), None) +def format_of(kind: type[Definition[Any]]) -> Literal[2, 3]: + """The Zarr format whose documents hold fields of `kind`, as the kind declared it; `TypeError` for a class that is no kind.""" + found = as_kind(kind)._format # pyright: ignore[reportPrivateUsage] + if found is None: # a kind cannot be declared without one + msg = f"{kind!r} declares no format" + raise TypeError(msg) + return found + + def _declares_kind(cls: type[object]) -> TypeGuard[type[Definition[Any]]]: """Whether `cls` itself was declared `kind=True`: the mark is read from its own namespace, which a subclass does not share.""" return vars(cls).get("_is_kind") is True and issubclass(cls, Definition) @@ -687,7 +710,7 @@ def as_kind(kind: object) -> type[Definition[Any]]: names = ", ".join(known.__name__ for known in KINDS) msg = ( f"{kind!r} is not a kind of metadata; read a field as one of {names}, or as a " - "subclass of Definition declared with kind=True" + "subclass of Definition declared with kind=True and its format" ) raise TypeError(msg) @@ -1668,6 +1691,12 @@ def resolve( or without type arguments; anything else is a `TypeError`. """ asked = as_kind(kind) + if context.format is not None and context.format != format_of(asked): + msg = ( + f"a {asked.label} is a field of a Zarr v{format_of(asked)} document, read in a scope " + f"of that format or of none, got a v{context.format} scope" + ) + raise TypeError(msg) refined, problems = refine_json(data, loc) if len(problems) != 0: # Not JSON, so not read; its name, if it has one, still says what diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/_registry.py b/packages/zarr-metadata/src/zarr_metadata/v3/_registry.py index e01c4edf49..6150cc99d1 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/_registry.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/_registry.py @@ -19,9 +19,11 @@ from collections.abc import Mapping from dataclasses import dataclass from types import MappingProxyType -from typing import TYPE_CHECKING, Any, Final, cast +from typing import TYPE_CHECKING, Any, Final, Generic, Literal, TypeAlias, TypeGuard, cast -from zarr_metadata.v3._definition import Definition, as_kind, kind_of, spelled +from typing_extensions import TypeVar + +from zarr_metadata.v3._definition import Definition, as_kind, format_of, kind_of, spelled from zarr_metadata.v3._scope import Conflict, ScopeConflictError, disagreements_of from zarr_metadata.v3.chunk_grid.rectilinear import RECTILINEAR_CHUNK_GRID from zarr_metadata.v3.chunk_grid.regular import REGULAR_CHUNK_GRID @@ -66,26 +68,38 @@ Tables = Mapping[type[Definition[Any]], Mapping[str, Definition[Any]]] """By kind, then by the name each definition is filed under.""" +F = TypeVar("F", bound=Literal[2, 3], default=Any) +"""The Zarr format a scope reads documents of: 2 or 3, and any when a scope is built from definitions rather than from a format's own.""" + @dataclass(frozen=True, slots=True, eq=False) -class Context: - """The definitions in scope while metadata is read. +class Context(Generic[F]): + """The definitions in scope while metadata is read, of one Zarr format. A value with no reading of its own: `resolve` reads a field in it, and `claimant` is the one question it answers, which definition a name belongs to. Built from definitions with `Context.of`, extended with more by `extended_with`; two scopes are equal when they file the - same definitions, and equal scopes hash alike. + same definitions, and equal scopes hash alike. Every kind a scope + files declares one format, which is the scope's: a v3 reader refuses + a v2 scope, and the other way round, while a scope that files nothing + is of no format and reads in either, claiming nothing. """ tables: Tables + @property + def format(self) -> Literal[2, 3] | None: + """The Zarr format the definitions in scope read, 2 or 3; None for a scope that files nothing.""" + return next((format_of(kind) for kind in self.tables), None) + @classmethod - def of(cls, *definitions: Definition[Any]) -> Context: + def of(cls, *definitions: Definition[Any]) -> Context[Any]: """A scope of exactly these definitions; a later one takes a name over from an earlier. `TypeError` for a definition of no kind, which no position in a - document could hold. + document could hold, and for definitions of kinds of two formats, + which no document holds together. """ tables: dict[type[Definition[Any]], dict[str, Definition[Any]]] = {} for definition in definitions: @@ -99,11 +113,18 @@ def of(cls, *definitions: Definition[Any]) -> Context: ) raise TypeError(msg) tables.setdefault(kind, {})[definition.name] = definition + formats = sorted({format_of(kind) for kind in tables}) + if len(formats) > 1: + msg = ( + "a scope reads documents of one Zarr format; these definitions are of kinds of " + f"formats {' and '.join(f'v{found}' for found in formats)}" + ) + raise TypeError(msg) return cls( MappingProxyType({kind: MappingProxyType(table) for kind, table in tables.items()}) ) - def extended_with(self, *definitions: Definition[Any]) -> Context: + def extended_with(self, *definitions: Definition[Any]) -> Context[F]: """This scope, plus definitions of your own. A name already filed under the same kind is taken over by what is @@ -160,7 +181,7 @@ def disagreements(self, claims: Claims) -> Disagreements: return disagreements_of(lambda kind, name: self.tables.get(kind, {}).get(name), claims) @classmethod - def joined(cls, *contexts: Context) -> Context: + def joined(cls, *contexts: Context[F]) -> Context[F]: """The least scope that files everything each of `contexts` files: their join. `ScopeConflictError` when two of them file different definitions @@ -194,6 +215,30 @@ def claimant(self, kind: type[D], name: str) -> D | None: return cast("D | None", self.tables.get(asked, {}).get(filed)) +ZarrV3Context: TypeAlias = Context[Literal[3]] +"""A scope that reads Zarr v3 documents: what every v3 reader takes.""" +ZarrV2Context: TypeAlias = Context[Literal[2]] +"""A scope that reads Zarr v2 documents: what every v2 reader takes.""" + + +def is_scope(value: object) -> TypeGuard[Context[Any]]: + """Whether `value` is a scope, of whatever format.""" + return isinstance(value, Context) + + +def scoped(context: Context[F] | None, default: Context[F]) -> Context[F]: + """The scope a reader of `default`'s format reads in: `context`, or `default` when none is given; `TypeError` for a scope of the other format.""" + if context is None: + return default + if context.format is not None and context.format != default.format: + msg = ( + f"a v{default.format} document is read in a scope of that format or of none, " + f"got a v{context.format} scope" + ) + raise TypeError(msg) + return context + + _CORE: Final[tuple[Definition[Any], ...]] = ( BLOSC_CODEC, BYTES_CODEC, @@ -235,10 +280,19 @@ def claimant(self, kind: type[D], name: str) -> D | None: ) """What `zarr-extensions` registers and this package defines.""" -CORE: Final = Context.of(*_CORE) +CORE: Final[ZarrV3Context] = Context.of(*_CORE) """Only what the Zarr v3 specification defines.""" -CORE_AND_EXTENSIONS: Final = Context.of(*_CORE, *_EXTENSIONS) +CORE_AND_EXTENSIONS: Final[ZarrV3Context] = Context.of(*_CORE, *_EXTENSIONS) """What the specification defines, plus what `zarr-extensions` registers.""" -__all__ = ["CORE", "CORE_AND_EXTENSIONS", "Context", "Tables"] +__all__ = [ + "CORE", + "CORE_AND_EXTENSIONS", + "Context", + "Tables", + "ZarrV2Context", + "ZarrV3Context", + "is_scope", + "scoped", +] diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/definition.py b/packages/zarr-metadata/src/zarr_metadata/v3/definition.py index 7efdfbb506..2a10ac806b 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/definition.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/definition.py @@ -282,8 +282,12 @@ def acme_lz4_rules( TypedDict, and each field alias, is written once, in `$defs`, under its name; the fill value is held to its data type's. -**Scopes as values.** Two scopes are equal when they file the same -definitions, and equal scopes hash alike. `Context.joined(*scopes)` is +**Scopes as values.** A scope reads documents of one Zarr format, the +`format` every kind it files declares -- `ZarrV3Context` is the type of +a v3 scope, `ZarrV2Context` of a v2 one -- and a reader refuses a scope +of the other format with `TypeError`, while a scope that files nothing +is of no format and reads in either. Two scopes are equal when they file +the same definitions, and equal scopes hash alike. `Context.joined(*scopes)` is the least scope above each, or a `ScopeConflictError` naming each name filed two ways; `extended_with` remains the way to take a name over on purpose. A model's `refined_in` moves it to a scope that claims more and @@ -304,8 +308,9 @@ def acme_lz4_rules( `"dynamic"`; a function no codec of its kind is asked -- chunk rules or pipelines of a bytes -> bytes codec, which is handed bytes, or a `transition` of a codec that hands on bytes; a data type named as raw -bits of one size are written. A scope refuses a definition of no kind. -Nothing happens at class creation. +bits of one size are written. A scope refuses a definition of no kind, +and a kind is declared with its format, `class Tag(Definition[C], +kind=True, format=3)`. Nothing else happens at class creation. """ from zarr_metadata._common import JSONValue @@ -343,7 +348,7 @@ def acme_lz4_rules( storage_of, ) from zarr_metadata.v3._pipeline import Stage -from zarr_metadata.v3._registry import CORE, CORE_AND_EXTENSIONS, Context +from zarr_metadata.v3._registry import CORE, CORE_AND_EXTENSIONS, Context, ZarrV3Context from zarr_metadata.v3._scope import ( ClaimKey, Claims, @@ -390,6 +395,7 @@ def acme_lz4_rules( "StorageTransformerField", "UnclaimedField", "ValidationProblem", + "ZarrV3Context", "canonical_fill_value", "fill_value_problems", "resolve", diff --git a/packages/zarr-metadata/tests/model/test_scope_threading.py b/packages/zarr-metadata/tests/model/test_scope_threading.py index 334ba03025..5955e745ed 100644 --- a/packages/zarr-metadata/tests/model/test_scope_threading.py +++ b/packages/zarr-metadata/tests/model/test_scope_threading.py @@ -126,3 +126,14 @@ def test_the_json_schema_is_written_in_the_scope_it_is_given() -> None: """`node_metadata_json_schema_v3` takes `None` for the default scope, and writes a different schema for an empty one.""" assert zm.node_metadata_json_schema_v3(context=None) == zm.node_metadata_json_schema_v3() assert zm.node_metadata_json_schema_v3(context=EMPTY) != zm.node_metadata_json_schema_v3() + + +@pytest.mark.parametrize("read", ENTRY_POINTS.values(), ids=ENTRY_POINTS.keys()) +def test_error_every_entry_point_refuses_a_scope_of_another_format( + read: Callable[..., object], +) -> None: + """A v3 entry point given a v2 scope raises `TypeError`: a scope reads documents of one format, and a v2 scope claims nothing a v3 document writes.""" + from zarr_metadata.v2.definition import CORE_V2 + + with pytest.raises(TypeError, match="format"): + read(context=CORE_V2) diff --git a/packages/zarr-metadata/tests/model/test_scope_threading_v2.py b/packages/zarr-metadata/tests/model/test_scope_threading_v2.py new file mode 100644 index 0000000000..f4f470458e --- /dev/null +++ b/packages/zarr-metadata/tests/model/test_scope_threading_v2.py @@ -0,0 +1,131 @@ +"""Every v2 entry point reads in the scope it is given, `CORE_V2` when given none, and refuses a scope of another format.""" + +from __future__ import annotations + +import json +from typing import TYPE_CHECKING, Any + +import pytest + +import zarr_metadata.model as zm +from zarr_metadata.model import ZarrV2ArrayMetadata +from zarr_metadata.v2.definition import CORE_V2, Context, resolve_codec_v2, resolve_dtype_v2 +from zarr_metadata.v3.definition import CORE_AND_EXTENSIONS + +if TYPE_CHECKING: + from collections.abc import Callable + +EMPTY = Context.of() +BASE: dict[str, Any] = dict(ZarrV2ArrayMetadata.create_default(shape=(4,), chunks=(2,)).to_json()) +BAD: dict[str, Any] = {**BASE, "compressor": {"id": "gzip", "level": "x"}} +"""An array the v2 scope refuses -- a gzip level that is no integer -- and an empty scope leaves unjudged.""" +GROUP: dict[str, Any] = {"zarr_format": 2} +CONSOLIDATED: dict[str, Any] = { + "zarr_consolidated_format": 1, + "metadata": {".zgroup": GROUP, "a/.zarray": BAD}, +} +STORE_A = {".zarray": json.dumps(BAD).encode()} +STORE_G = {".zgroup": json.dumps(GROUP).encode()} +STORE_C = {".zmetadata": json.dumps(CONSOLIDATED).encode()} + + +def _accepts(read: Callable[..., object]) -> Callable[[Context | None], bool]: + """Whether `read`, given `context`, finds nothing wrong: True for a model, a reading with a model, an empty problem tuple, a guard saying yes, or a field read or left unclaimed.""" + + def accepted(context: Context | None) -> bool: + try: + found = read(context=context) + except zm.MetadataValidationError: + return False + if isinstance(found, tuple) and len(found) == 2 and isinstance(found[1], tuple): + return len(found[1]) == 0 # a resolved field and its problems + if isinstance(found, tuple): + return len(found) == 0 + if isinstance(found, bool): + return found + reading = getattr(found, "reading", found) + metadata = getattr(reading, "metadata", found) + return metadata is not None + + return accepted + + +ENTRY_POINTS: dict[str, Callable[..., object]] = { + "validate_array_metadata_v2": lambda context: zm.validate_array_metadata_v2( + BAD, context=context + ), + "is_array_metadata_v2": lambda context: zm.is_array_metadata_v2( + zm.parse_array_metadata_v2(BAD, context=EMPTY), context=context + ), + "parse_array_metadata_v2": lambda context: zm.parse_array_metadata_v2(BAD, context=context), + "read_array_metadata_v2": lambda context: zm.read_array_metadata_v2(BAD, context=context), + "ZarrV2ArrayMetadata": lambda context: zm.ZarrV2ArrayMetadata(BAD, context=context), + "ZarrV2ArrayMetadata.from_json": lambda context: zm.ZarrV2ArrayMetadata.from_json( + BAD, context=context + ), + "ZarrV2ArrayMetadata.from_key_value": lambda context: zm.ZarrV2ArrayMetadata.from_key_value( + STORE_A, context=context + ), + "ZarrV2ArrayMetadata.create_default": lambda context: zm.ZarrV2ArrayMetadata.create_default( + shape=(4,), chunks=(2,), compressor=BAD["compressor"], context=context + ), + "ZarrV2ArrayMetadata.with_context": lambda context: zm.ZarrV2ArrayMetadata( + BAD, context=EMPTY + ).with_context(context), + "ZarrV2ArrayMetadata.refined_in": lambda context: zm.ZarrV2ArrayMetadata( + BAD, context=EMPTY + ).refined_in(context), + "validate_group_metadata_v2": lambda context: zm.validate_group_metadata_v2( + GROUP, context=context + ), + "is_group_metadata_v2": lambda context: zm.is_group_metadata_v2(GROUP, context=context), + "parse_group_metadata_v2": lambda context: zm.parse_group_metadata_v2(GROUP, context=context), + "ZarrV2GroupMetadata": lambda context: zm.ZarrV2GroupMetadata(GROUP, context=context), + "ZarrV2GroupMetadata.from_json": lambda context: zm.ZarrV2GroupMetadata.from_json( + GROUP, context=context + ), + "ZarrV2GroupMetadata.from_key_value": lambda context: zm.ZarrV2GroupMetadata.from_key_value( + STORE_G, context=context + ), + "ZarrV2GroupMetadata.create_default": lambda context: zm.ZarrV2GroupMetadata.create_default( + context=context + ), + "ZarrV2ConsolidatedMetadata": lambda context: zm.ZarrV2ConsolidatedMetadata( + CONSOLIDATED, context=context + ), + "ZarrV2ConsolidatedMetadata.from_json": lambda context: zm.ZarrV2ConsolidatedMetadata.from_json( + CONSOLIDATED, context=context + ), + "ZarrV2ConsolidatedMetadata.from_key_value": ( + lambda context: zm.ZarrV2ConsolidatedMetadata.from_key_value(STORE_C, context=context) + ), + "read_repaired_consolidated_metadata_v2": ( + lambda context: zm.read_repaired_consolidated_metadata_v2(CONSOLIDATED, context=context) + ), + "resolve_dtype_v2": lambda context: resolve_dtype_v2(" None: + """Each v2 entry point accepts its document in an empty scope, and refuses the gzip level that is no integer in `CORE_V2`, which `None` names too, when the document holds it: the scope it is given is the scope it reads in.""" + accepted = _accepts(ENTRY_POINTS[name]) + assert accepted(EMPTY) is True + assert accepted(CORE_V2) is (name not in REFUSES_BAD) + assert accepted(None) is (name not in REFUSES_BAD) + + +@pytest.mark.parametrize("name", ENTRY_POINTS.keys()) +def test_error_every_v2_entry_point_refuses_a_scope_of_another_format(name: str) -> None: + """A v2 entry point given a v3 scope raises `TypeError`: a scope reads documents of one format, and a v3 scope claims nothing a v2 document writes.""" + with pytest.raises(TypeError, match="format"): + ENTRY_POINTS[name](context=CORE_AND_EXTENSIONS) diff --git a/packages/zarr-metadata/tests/v2/test_definition.py b/packages/zarr-metadata/tests/v2/test_definition.py index 7cb7e31cfb..7e8332c3eb 100644 --- a/packages/zarr-metadata/tests/v2/test_definition.py +++ b/packages/zarr-metadata/tests/v2/test_definition.py @@ -4,6 +4,8 @@ import pickle +import pytest + from zarr_metadata.v2.definition import ( CORE_V2, V2_CODECS, @@ -20,19 +22,20 @@ from zarr_metadata.v3.definition import ( CORE_AND_EXTENSIONS, CodecDefinition, - DataTypeDefinition, ) def test_core_v2_files_every_v2_definition_apart_from_v3() -> None: - """`CORE_V2` files the 12 data types and 21 codecs by their v2 kinds; a scope joined with the v3 scope files all of them apart, since the kinds differ: a v2 `gzip` and a v3 `gzip` are two definitions.""" + """`CORE_V2` files the 12 data types and 21 codecs by their v2 kinds, and is a scope of format 2: a v2 `gzip` and a v3 `gzip` are two definitions of two kinds, and no scope files both, since `Context.joined` refuses two formats.""" assert set(CORE_V2.definitions()) == {*V2_DATA_TYPES, *V2_CODECS} - both = Context.joined(CORE_V2, CORE_AND_EXTENSIONS) - assert both.claimant(ZarrV2CodecDefinition, "gzip") is not GZIP_CODEC - assert both.claimant(ZarrV2CodecDefinition, "gzip") is not None - assert both.claimant(CodecDefinition, "gzip") is GZIP_CODEC - assert both.claimant(ZarrV2DataTypeDefinition, " None: ), ) assert any(p.message.endswith("got 3") for p in problems), [p.message for p in problems] + + +def test_error_a_kind_declares_its_format() -> None: + """A kind says which Zarr format its fields belong to, `format=2` or `format=3`, in the class header beside `kind=True`: a kind declared without one, or with a format that is neither, is a `TypeError` at class creation.""" + with pytest.raises(TypeError, match="format"): + type("NoFormat", (Definition,), {}, kind=True) + with pytest.raises(TypeError, match="format"): + type("WrongFormat", (Definition,), {}, kind=True, format=4) + with pytest.raises(TypeError, match="format"): + type("FormatOnADefinition", (CodecDefinition,), {}, format=3) + + +TAG = Tag(name="tag", configuration=EmptyConfiguration) + + +def test_a_scope_has_the_format_of_the_kinds_it_files() -> None: + """A scope's `format` is the one format of every kind it files: 3 for `CORE`, 2 for `CORE_V2`, and None for a scope that files nothing, which reads in either format.""" + from zarr_metadata.v2.definition import CORE_V2 + from zarr_metadata.v3.definition import CORE + + assert CORE.format == 3 + assert CORE_V2.format == 2 + assert Context.of().format is None + assert CORE.extended_with(TAG).format == 3 + + +def test_error_a_scope_files_one_format() -> None: + """Definitions of two formats cannot share a scope: `Context.of`, `extended_with` and `joined` each raise `TypeError` naming both formats.""" + from zarr_metadata.v2.data_type.scalar import UINT_V2 + from zarr_metadata.v2.definition import CORE_V2 + from zarr_metadata.v3.definition import CORE + + with pytest.raises(TypeError, match="format"): + Context.of(TAG, UINT_V2) + with pytest.raises(TypeError, match="format"): + CORE.extended_with(UINT_V2) + with pytest.raises(TypeError, match="format"): + Context.joined(CORE, CORE_V2) + + +def test_error_a_field_is_read_in_a_scope_of_its_format() -> None: + """`resolve` refuses a scope of another format than the kind's with `TypeError`, and reads in a scope of no format, which claims nothing.""" + from zarr_metadata.v2.definition import CORE_V2, ZarrV2DataTypeDefinition + from zarr_metadata.v3.definition import CORE + + with pytest.raises(TypeError, match="format"): + resolve(" Date: Fri, 9 Oct 2026 21:32:47 +0200 Subject: [PATCH 87/94] fix(zarr-metadata): a field's configuration, nested fields and a stage's inner pipelines are read-only A field handed out by a model or a reading could be edited in place, which left its cached key, == and hash unchanged while codecs == and refines drifted. Fields and stages now hold read-only views at every level, as the models' attributes already do, and pickle as what they were built from. Assisted-by: ClaudeCode:claude-fable-5-1 Co-Authored-By: Claude Fable 5.1 --- .../zarr-metadata/src/zarr_metadata/_json.py | 10 ++- .../src/zarr_metadata/v3/_definition.py | 68 ++++++++++++++++++- .../src/zarr_metadata/v3/_pipeline.py | 9 +++ .../zarr-metadata/tests/model/test_pair.py | 12 ++++ .../tests/v3/test_definitions.py | 51 ++++++++++++++ .../zarr-metadata/tests/v3/test_pipelines.py | 26 +++++++ 6 files changed, 174 insertions(+), 2 deletions(-) diff --git a/packages/zarr-metadata/src/zarr_metadata/_json.py b/packages/zarr-metadata/src/zarr_metadata/_json.py index 3bc5a417d6..9b54531efd 100644 --- a/packages/zarr-metadata/src/zarr_metadata/_json.py +++ b/packages/zarr-metadata/src/zarr_metadata/_json.py @@ -472,7 +472,15 @@ def json_text(value: JSONValue) -> str: `0.0` and `NaN` for no value at all, but what a document writes: two values written alike are one value to every reader. """ - return json.dumps(value, sort_keys=True, ensure_ascii=False) + return json.dumps(value, sort_keys=True, ensure_ascii=False, default=_as_object) + + +def _as_object(value: object) -> dict[object, object]: + """A read-only view of an object, as `json.dumps` is handed one, written as the object; anything else is the `TypeError` `json.dumps` raises.""" + if is_object(value): + return dict(value) + msg = f"{value!r} is not JSON" + raise TypeError(msg) def shown(value: object) -> str: diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py b/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py index 1930e8ea2d..5c8c268f19 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py @@ -56,6 +56,7 @@ from zarr_metadata._json import ( ValidationProblem, copied, + frozen, is_object, is_tuple, json_text, @@ -1192,6 +1193,21 @@ def __post_init__(self) -> None: if refusal is not None: raise TypeError(refusal) object.__setattr__(self, "read_as", kind) + # What the field hands out is read-only at every level, so a field + # cannot be put in a state its key, `==` and `refines` disagree about. + object.__setattr__(self, "json", frozen(self.json)) + object.__setattr__(self, "configuration", frozen(self.configuration)) + object.__setattr__(self, "nested", MappingProxyType(dict(self.nested))) + + def __reduce__(self) -> tuple[Callable[..., AcceptedField[Any]], tuple[object, ...]]: + # Read-only views do not pickle: the field pickles as what it was built from. + return _accepted_field, ( + copied(self.json), + self.name, + self.definition, + copied(self.configuration), + dict(self.nested), + ) def to_json(self) -> JSONValue: """The field as a document writes it, for every reader: its configuration as read, sharing nothing with the field. @@ -1247,7 +1263,14 @@ def __post_init__(self) -> None: raise TypeError(msg) _, written, _ = self.read_as.named_configuration(self.json) configuration: Mapping[str, object] = {} if written is None else written - object.__setattr__(self, "configuration", cast("Mapping[str, JSONValue]", configuration)) + object.__setattr__(self, "json", frozen(self.json)) + object.__setattr__( + self, "configuration", frozen(cast("Mapping[str, JSONValue]", configuration)) + ) + + def __reduce__(self) -> tuple[Callable[..., UnclaimedField], tuple[object, ...]]: + # Read-only views do not pickle: the field pickles as what it was built from. + return _unclaimed_field, (copied(self.json), self.name, self.read_as) @property def definition(self) -> None: @@ -1296,6 +1319,19 @@ def __post_init__(self) -> None: refusal = None if definition is None else _misread(definition, kind, self.name) if refusal is not None: raise TypeError(refusal) + if self.json is not UNSET: + object.__setattr__(self, "json", frozen(self.json)) + object.__setattr__(self, "nested", MappingProxyType(dict(self.nested))) + + def __reduce__(self) -> tuple[Callable[..., RefusedField[Any]], tuple[object, ...]]: + # Read-only views do not pickle: the field pickles as what it was built from. + return _refused_field, ( + UNSET if self.json is UNSET else copied(self.json), + self.name, + self.read_as, + self.definition, + dict(self.nested), + ) ResolvedField = TypeAliasType( @@ -1303,6 +1339,36 @@ def __post_init__(self) -> None: ) """One metadata field as a scope read it: read by the definition that claims its name, claimed by nothing, or refused.""" + +def _accepted_field( + json: JSONValue, + name: str, + definition: Definition[Any], + configuration: Mapping[str, JSONValue], + nested: Nested, +) -> AcceptedField[Any]: + """An accepted field built again from what it pickled as.""" + return AcceptedField( + json=json, name=name, definition=definition, configuration=configuration, nested=nested + ) + + +def _unclaimed_field(json: JSONValue, name: str, read_as: type[Definition[Any]]) -> UnclaimedField: + """An unclaimed field built again from what it pickled as.""" + return UnclaimedField(json=json, name=name, read_as=read_as) + + +def _refused_field( + json: JSONValue | UNSET, + name: str | None, + read_as: type[Definition[Any]], + definition: Definition[Any] | None, + nested: Nested, +) -> RefusedField[Any]: + """A refused field built again from what it pickled as.""" + return RefusedField(json=json, name=name, read_as=read_as, definition=definition, nested=nested) + + _FIELDS: Final = (AcceptedField, UnclaimedField, RefusedField) """The three things a scope makes of a field.""" diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/_pipeline.py b/packages/zarr-metadata/src/zarr_metadata/v3/_pipeline.py index ec7eb3274b..1b7857b5e6 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/_pipeline.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/_pipeline.py @@ -27,6 +27,7 @@ import dataclasses from dataclasses import dataclass +from types import MappingProxyType from typing import TYPE_CHECKING, Any, Final, TypeGuard, cast from zarr_metadata._json import ValidationProblem, is_object, is_tuple, with_input @@ -69,6 +70,14 @@ class Stage: inner: Mapping[str, tuple[Stage, ...]] = dataclasses.field(default_factory=_no_stages) """The pipelines it holds, by the member of its configuration that holds each: each codec with the chunk it is handed.""" + def __post_init__(self) -> None: + # Read-only, as everything a reading hands out is. + object.__setattr__(self, "inner", MappingProxyType(dict(self.inner))) + + def __reduce__(self) -> tuple[type[Stage], tuple[object, ...]]: + # A read-only view does not pickle: the stage pickles as what it was built from. + return Stage, (self.codec, self.incoming, dict(self.inner)) + _POSITIONS: Final[Mapping[CodecKind, int]] = { "array_array": 0, diff --git a/packages/zarr-metadata/tests/model/test_pair.py b/packages/zarr-metadata/tests/model/test_pair.py index c74409cea0..d30409a5a1 100644 --- a/packages/zarr-metadata/tests/model/test_pair.py +++ b/packages/zarr-metadata/tests/model/test_pair.py @@ -726,3 +726,15 @@ def test_an_array_reading_pickles_through_its_model() -> None: assert again == reading assert again.metadata is not None assert again.metadata.reading is again + + +def test_error_a_models_fields_cannot_be_changed_in_place() -> None: + """The fields a model hands out -- a codec's configuration, the fields a shard holds -- are read-only, so `codecs ==` and `refines` cannot drift from `==`.""" + model = ZarrV3ArrayMetadata(ARRAY) + same = ZarrV3ArrayMetadata(ARRAY) + codec = model.codecs[0] + assert isinstance(codec, AcceptedField) + with pytest.raises(TypeError): + codec.configuration["endian"] = "big" # pyright: ignore[reportIndexIssue] + assert model.codecs == same.codecs + assert model.refines(same) diff --git a/packages/zarr-metadata/tests/v3/test_definitions.py b/packages/zarr-metadata/tests/v3/test_definitions.py index 77af5e9d3d..67f1a5d528 100644 --- a/packages/zarr-metadata/tests/v3/test_definitions.py +++ b/packages/zarr-metadata/tests/v3/test_definitions.py @@ -1419,3 +1419,54 @@ def refuses(configuration: GzipCodecConfiguration) -> GzipCodecConfiguration: with pytest.raises(ValueError, match="no") as raised: canonical_of(read, ()) assert raised.value.__notes__ == ["raised by the canonical of 'gzip'"] + + +def test_error_a_field_s_configuration_and_nested_fields_are_read_only() -> None: + """What a field hands out -- its `json`, `configuration` and `nested` fields -- is read-only at every level, so a field cannot be put in a state its key, `==` and `refines` disagree about.""" + whole = {**SHARD, "index_codecs": [LE, {"name": "crc32c"}]} + shard, _ = resolve({"name": "sharding_indexed", "configuration": whole}, CodecDefinition, CORE) + assert isinstance(shard, AcceptedField) + with pytest.raises(TypeError): + shard.configuration["index_location"] = "start" # pyright: ignore[reportIndexIssue] + codecs = shard.configuration["codecs"] + assert isinstance(codecs, tuple) + inner = codecs[0] + assert isinstance(inner, Mapping) + with pytest.raises(TypeError): + inner["name"] = "crc32c" # pyright: ignore[reportIndexIssue] + with pytest.raises(TypeError): + del shard.nested[("codecs", 0)] # pyright: ignore[reportIndexIssue] + assert isinstance(shard.json, Mapping) + with pytest.raises(TypeError): + shard.json["name"] = "gzip" # pyright: ignore[reportIndexIssue] + unclaimed, _ = resolve({"name": "acme.x", "configuration": {"a": [1]}}, CodecDefinition, CORE) + assert isinstance(unclaimed, UnclaimedField) + with pytest.raises(TypeError): + unclaimed.configuration["a"] = 2 # pyright: ignore[reportIndexIssue] + refused, _ = resolve( + {"name": "sharding_indexed", "configuration": {**whole, "index_location": "x"}}, + CodecDefinition, + CORE, + ) + assert isinstance(refused, RefusedField) + with pytest.raises(TypeError): + del refused.nested[("codecs", 0)] # pyright: ignore[reportIndexIssue] + + +def test_a_field_holding_fields_is_copied_and_pickled_whole() -> None: + """A field pickles and deep-copies with the fields it holds, equal to itself, whether accepted, refused or left unclaimed.""" + whole = {**SHARD, "index_codecs": [LE, {"name": "crc32c"}]} + fields = [ + resolve({"name": "sharding_indexed", "configuration": whole}, CodecDefinition, CORE)[0], + resolve({"name": "acme.x", "configuration": {"a": [1]}}, CodecDefinition, CORE)[0], + resolve( + {"name": "sharding_indexed", "configuration": {**whole, "index_location": "x"}}, + CodecDefinition, + CORE, + )[0], + ] + for field in fields: + for again in (pickle.loads(pickle.dumps(field)), copy.deepcopy(field)): + assert again == field + assert type(again) is type(field) + assert again.nested == field.nested diff --git a/packages/zarr-metadata/tests/v3/test_pipelines.py b/packages/zarr-metadata/tests/v3/test_pipelines.py index f1f779ba5b..b49af3331b 100644 --- a/packages/zarr-metadata/tests/v3/test_pipelines.py +++ b/packages/zarr-metadata/tests/v3/test_pipelines.py @@ -501,3 +501,29 @@ def test_error_pipelines_that_raise_say_whose_they_are() -> None: assert raised.value.__notes__ == [ "raised by the pipelines of 'acme.holder', reading ('codecs', 0, 'configuration')" ] + + +def test_error_a_stage_s_inner_pipelines_are_read_only() -> None: + """The pipelines a stage holds of a shard's inner codecs cannot be changed in place, and a reading holding them pickles and deep-copies equal to itself.""" + import copy + import pickle + + shard: JSONValue = { + "name": "sharding_indexed", + "configuration": { + "chunk_shape": [2], + "codecs": [{"name": "bytes", "configuration": {"endian": "little"}}], + "index_codecs": [ + {"name": "bytes", "configuration": {"endian": "little"}}, + {"name": "crc32c"}, + ], + "index_location": "end", + }, + } + stages, _ = _read([shard], CHUNK) + stage = stages[0] + assert "codecs" in stage.inner + with pytest.raises(TypeError): + stage.inner["codecs"] = () # pyright: ignore[reportIndexIssue] + for again in (pickle.loads(pickle.dumps(stages)), copy.deepcopy(stages)): + assert again == stages From 16c0b5bd85d94df6e27f0674cd1cd2e700cc49fc Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Fri, 9 Oct 2026 21:35:55 +0200 Subject: [PATCH 88/94] fix(zarr-metadata): a gain is judged by what the definition reads, so refines is transitive through == MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit An accepted field refined an unclaimed one only when their JSON was written alike, while == compares canonical configurations, so a ⊑ b ⊑ u could hold with a ⋢ u. A gain now re-reads what the unclaimed field wrote in the scope the accepted field's own claims make, and holds when that read is the field. Reading one field asks a Scope for its format and claimant, which a Context is and a field's claims are. Assisted-by: ClaudeCode:claude-fable-5-1 Co-Authored-By: Claude Fable 5.1 --- .../src/zarr_metadata/v3/_definition.py | 18 +++++-- .../src/zarr_metadata/v3/_scope.py | 50 +++++++++++++++---- packages/zarr-metadata/tests/v3/test_scope.py | 31 ++++++++++++ 3 files changed, 85 insertions(+), 14 deletions(-) diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py b/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py index 5c8c268f19..2a32921910 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py @@ -42,6 +42,7 @@ Final, Generic, Literal, + Protocol, TypeAlias, TypeGuard, cast, @@ -727,6 +728,15 @@ def field_kind(annotation: object) -> type[Definition[Any]] | None: return _FIELD_KINDS.get(annotation) +class Scope(Protocol): + """What reading a field asks of a scope: its format, and which definition of a kind claims a name. A `Context` is one; so is anything else that answers the two.""" + + @property + def format(self) -> Literal[2, 3] | None: ... + + def claimant(self, kind: type[D], name: str) -> D | None: ... + + @dataclass(frozen=True, slots=True) class _NestedField: """A metadata field the check met inside a configuration: where it sits, its kind, its JSON. @@ -1736,7 +1746,7 @@ def chunk_grid_lengths( def resolve( - data: object, kind: type[D], context: Context, loc: Loc = () + data: object, kind: type[D], context: Scope, loc: Loc = () ) -> tuple[ResolvedField[D], Problems]: """`data`, one metadata field, read as a `kind` in `context`: what the scope made of it, and every problem. @@ -1779,7 +1789,7 @@ def resolve( def _resolve_field( - data: JSONValue, kind: type[Definition[Any]], context: Context, loc: Loc + data: JSONValue, kind: type[Definition[Any]], context: Scope, loc: Loc ) -> tuple[ResolvedField[Definition[Any]], Problems]: """A refined field with its envelope judged, then read. @@ -1794,7 +1804,7 @@ def _resolve_field( def _read( - data: JSONValue, kind: type[Definition[Any]], context: Context, loc: Loc + data: JSONValue, kind: type[Definition[Any]], context: Scope, loc: Loc ) -> tuple[ResolvedField[Definition[Any]], Problems]: name, given, malformed = kind.named_configuration(data) if name is None: @@ -1926,7 +1936,7 @@ def _sized(field: _NestedField, inner: ResolvedField[Any]) -> Problems: def canonicalize( - data: object, kind: type[D], context: Context, loc: Loc = () + data: object, kind: type[D], context: Scope, loc: Loc = () ) -> tuple[JSONValue | None, Problems]: """`data`, one metadata field, in its simplest equivalent spelling, and every problem. diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/_scope.py b/packages/zarr-metadata/src/zarr_metadata/v3/_scope.py index 03214cecff..b9bd878e0a 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/_scope.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/_scope.py @@ -13,15 +13,19 @@ from collections.abc import Mapping from dataclasses import dataclass -from typing import TYPE_CHECKING, Any, TypeAlias +from typing import TYPE_CHECKING, Any, Literal, TypeAlias, cast from zarr_metadata.v3._definition import ( AcceptedField, Definition, RefusedField, UnclaimedField, + as_kind, field_key, + fields_of, + format_of, own_key, + resolve, spelled, ) @@ -29,7 +33,7 @@ from collections.abc import Callable, Iterable, Sequence from zarr_metadata._typed_json import Loc - from zarr_metadata.v3._definition import ResolvedField + from zarr_metadata.v3._definition import D, ResolvedField ClaimKey: TypeAlias = tuple[type[Definition[Any]], str] """A kind and the name a definition is filed under: what a scope answers `claimant` for.""" @@ -107,10 +111,11 @@ def refines(field: ResolvedField[Any], other: ResolvedField[Any]) -> bool: """Whether `field` holds everything `other` holds: reads the same where both read, and reads what `other` left unclaimed. The order one reading of a document refines another in. A name nothing - claimed, read by a definition, is a gain: the read field is compared - as the unclaimed one would be, by its name and the configuration as - written, as JSON text, so two spellings of one field are one, and - `true` is not `1`. The reverse is a loss; one name read by two + claimed, read by a definition, is a gain when what the unclaimed field + wrote, read by that definition and those of the fields the read field + holds, is the read field: two spellings of one configuration are one + gain, as they are one field to `==`, so the order is transitive + through equality. The reverse is a loss; one name read by two definitions is a conflict; a refused field refines itself alone. Two fields that refine each other are equal. """ @@ -119,7 +124,7 @@ def refines(field: ResolvedField[Any], other: ResolvedField[Any]) -> bool: if isinstance(other, UnclaimedField): if isinstance(field, UnclaimedField): return field_key(field) == field_key(other) - return claim_key(field) == claim_key(other) and _as_unclaimed(field) == other + return claim_key(field) == claim_key(other) and _gained(field, other) if isinstance(field, UnclaimedField): return False if field.definition != other.definition or own_key(field) != own_key(other): @@ -129,9 +134,34 @@ def refines(field: ResolvedField[Any], other: ResolvedField[Any]) -> bool: return all(refines(field.nested[loc], other.nested[loc]) for loc in field.nested) -def _as_unclaimed(field: AcceptedField[Any]) -> UnclaimedField: - """`field` as it would have been read had nothing claimed its name: what a gain is compared against.""" - return UnclaimedField(json=field.json, name=field.name, read_as=field.read_as) +def _gained(field: AcceptedField[Any], other: UnclaimedField) -> bool: + """Whether `other`, read in the scope `field`'s own claims make, is `field`: what a gain is.""" + again, _ = resolve(other.json, field.read_as, _Claimed.of(field)) + return again == field + + +@dataclass(frozen=True, slots=True) +class _Claimed: + """The scope a field's own claims make: its definition and those of the fields it holds, by kind and filed name; what a gain re-reads in.""" + + filed: Mapping[ClaimKey, Definition[Any]] + format: Literal[2, 3] | None + + @classmethod + def of(cls, field: AcceptedField[Any]) -> _Claimed: + claimed = claims_of(fields_of(field)) + return cls( + {key: definition for key, definition in claimed.items() if definition is not None}, + format_of(field.read_as), + ) + + def claimant(self, kind: type[D], name: str) -> D | None: + """The definition of `kind` among the claims that reads `name`; None if none does, as `Context.claimant` answers.""" + asked = as_kind(kind) + filed, _ = spelled(asked, name) + if filed is None: + return None + return cast("D | None", self.filed.get((asked, filed))) @dataclass(frozen=True, slots=True) diff --git a/packages/zarr-metadata/tests/v3/test_scope.py b/packages/zarr-metadata/tests/v3/test_scope.py index ea25d8e362..d136cd6a8f 100644 --- a/packages/zarr-metadata/tests/v3/test_scope.py +++ b/packages/zarr-metadata/tests/v3/test_scope.py @@ -385,3 +385,34 @@ def test_a_kind_is_named_in_words(kind: type[Definition[Any]], said: str) -> Non """`kind_name` names each kind of definition as a message does: `ChunkKeyEncodingDefinition` is "chunk key encoding".""" assert kind_name(kind) == said assert said in str(Conflict((kind, "x"), None, None)) + + +def test_a_gain_is_judged_by_what_the_definition_reads_not_by_spelling() -> None: + """An accepted field refines an unclaimed one when its definition, reading what the unclaimed field wrote, reads the accepted field: two spellings of one configuration are one gain, so `refines` is transitive through `==`, and a spelling the definition reads otherwise is no gain.""" + from zarr_metadata.v3.definition import CORE, ChunkKeyEncodingDefinition, Context, resolve + + nothing = Context.of() + spelled_out, _ = resolve( + {"name": "default", "configuration": {"separator": "/"}}, ChunkKeyEncodingDefinition, CORE + ) + bare, _ = resolve({"name": "default"}, ChunkKeyEncodingDefinition, CORE) + unclaimed, _ = resolve({"name": "default"}, ChunkKeyEncodingDefinition, nothing) + assert spelled_out == bare + assert refines(bare, unclaimed) + assert refines(spelled_out, unclaimed) + other, _ = resolve( + {"name": "default", "configuration": {"separator": "."}}, ChunkKeyEncodingDefinition, CORE + ) + assert not refines(other, unclaimed) + shard = { + "name": "sharding_indexed", + "configuration": { + "chunk_shape": [2], + "codecs": [{"name": "bytes", "configuration": {"endian": "little"}}], + "index_codecs": [{"name": "bytes", "configuration": {"endian": "little"}}, "crc32c"], + "index_location": "end", + }, + } + read, _ = resolve(shard, CodecDefinition, CORE) + unread, _ = resolve({**shard, "must_understand": True}, CodecDefinition, nothing) + assert refines(read, unread) From d5436ea7a12ea2219943e4c559ab2be863c240bc Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Fri, 9 Oct 2026 21:39:52 +0200 Subject: [PATCH 89/94] fix(zarr-metadata): a listed group's own listing holds the documents the group lists Two documents for one node path, one in the group's flat listing and another in a listed group's own listing, were accepted; the nested entry is now a problem when the two read otherwise. Arrays compare by array_key_of, what their models compare, and groups by their own members. Assisted-by: ClaudeCode:claude-fable-5-1 Co-Authored-By: Claude Fable 5.1 --- .../src/zarr_metadata/model/_array.py | 58 ++++++++------ .../src/zarr_metadata/model/_group.py | 80 ++++++++++++++----- .../zarr-metadata/tests/model/test_group.py | 29 +++++++ 3 files changed, 123 insertions(+), 44 deletions(-) diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_array.py b/packages/zarr-metadata/src/zarr_metadata/model/_array.py index fc14c77374..027495087b 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_array.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/_array.py @@ -226,25 +226,8 @@ def __repr__(self) -> str: return f"{type(self).__name__}({self._document!r}, context={self._context!r})" def _key_of(self) -> tuple[object, ...]: - """What `==` and `hash` compare of a v3 array model: what its document means. - - Each field by its `field_key`, the fill value in its canonical spelling - as JSON text when a definition in scope read the data type, and every - other member as it is, the JSON ones as text. - """ - members = self._members - return ( - members.shape, - self._fill_value_key(), - field_key(self.data_type), - field_key(self.chunk_grid), - tuple(field_key(codec) for codec in self.codecs), - field_key(self.chunk_key_encoding), - members.dimension_names, - json_text(members.attributes), - tuple(field_key(transformer) for transformer in self.storage_transformers), - json_text(members.extra_fields), - ) + """What `==` and `hash` compare of a v3 array model: what its document means, as `array_key_of` says.""" + return array_key_of(self._reading, self._members) def _plain_key( self, data_type: AcceptedField[DataTypeDefinition[Any]] | UnclaimedField @@ -259,13 +242,6 @@ def _plain_key( json_text(members.extra_fields), ) - def _fill_value_key(self) -> str: - """What `==` compares of the fill value: its canonical spelling as JSON text when a definition in scope read the data type, and the fill value as written when none did.""" - fill_value = self._members.fill_value - if isinstance(self.data_type, AcceptedField): - return json_text(spelled_canonically(self.data_type, fill_value)) - return json_text(fill_value) - def __reduce__(self) -> tuple[type[ZarrV3ArrayMetadata], tuple[object, Context]]: # The pair, read again on load: a model's reading never disagrees # with its document. @@ -478,6 +454,36 @@ def located_conflicts( return tuple(located) +def array_key_of( + reading: ZarrV3ArrayMetadataReading, members: ArrayMembersV3 +) -> tuple[object, ...]: + """What a v3 array document means, as `reading` read it and `members` refine it: what `==` and `hash` compare of its model, and what two listings of one node are compared by. + + Each field by its `field_key`, the fill value in its canonical spelling + as JSON text when a definition in scope read the data type and as + written when none did, and every other member as it is, the JSON ones + as text. + """ + data_type = held(reading.data_type) + fill_value = ( + spelled_canonically(data_type, members.fill_value) + if isinstance(data_type, AcceptedField) + else members.fill_value + ) + return ( + members.shape, + json_text(fill_value), + field_key(data_type), + field_key(held(reading.chunk_grid)), + tuple(field_key(held(stage.codec)) for stage in reading.pipeline), + field_key(held(reading.chunk_key_encoding)), + members.dimension_names, + json_text(members.attributes), + tuple(field_key(held(entry)) for entry in reading.storage_transformers), + json_text(members.extra_fields), + ) + + def read_array_metadata_v3( value: object, *, context: ZarrV3Context | None = None ) -> ZarrV3ArrayMetadataReading: diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_group.py b/packages/zarr-metadata/src/zarr_metadata/model/_group.py index dbf9b50671..96d02815d4 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_group.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/_group.py @@ -6,7 +6,7 @@ from collections.abc import Callable, Mapping from dataclasses import dataclass from types import MappingProxyType -from typing import TYPE_CHECKING, Any, Final, Literal, TypeAlias, TypeGuard, TypeVar, cast +from typing import TYPE_CHECKING, Any, Final, Literal, TypeAlias, TypeGuard, cast from typing_extensions import TypeAliasType, TypedDict, Unpack @@ -38,6 +38,7 @@ from zarr_metadata.model._array import ( ZarrV2ArrayMetadata, ZarrV3ArrayMetadata, + array_key_of, located_conflicts, must_understand_subset, read_array_metadata_v3, @@ -757,8 +758,6 @@ def read_group_v3( return reading, GroupMembersV3(attributes, extra_fields, held) -T = TypeVar("T") - _CONSOLIDATED_MEMBERS: Final = ("kind", "must_understand", "metadata") """The members of an inline `consolidated_metadata`, in the order the convention declares them.""" @@ -941,7 +940,7 @@ def _read_consolidated_v3( for key in node_types: problems.extend( _nested_listing_problems( - key, _reading_listing(readings[key]), node_types, _node_type_of + key, readings[key], members.get(key), readings, members, node_types ) ) return readings, members, tuple(problems) @@ -964,24 +963,27 @@ def _key_problems(key: str) -> list[ValidationProblem]: def _nested_listing_problems( key: str, - listing: Mapping[str, T], + listed: ZarrV3NodeMetadataReading, + listed_members: ArrayMembersV3 | GroupMembersV3 | None, + readings: Mapping[str, ZarrV3NodeMetadataReading], + members: Mapping[str, ArrayMembersV3 | GroupMembersV3], node_types: Mapping[str, NodeType | None], - node_type: Callable[[T], NodeType | None], ) -> list[ValidationProblem]: - """What is wrong with `listing`, the own consolidated listing of the group at `key`, against `node_types`, the group's flat listing: each problem at the nested entry. + """What is wrong with the own consolidated listing of `listed`, the group at `key`, against the group's flat listing, `readings` and their `node_types`: each problem at the nested entry. The reference implementation lists every node below the group in the group's own listing, flat, and gives each group it lists an empty listing of its own. So a node a listed group lists is one the group - lists too, at the joined key, of the same node type: one it lists - alone would be dropped by the reference reader, and one it lists as - another type contradicts the tree. Only the listing's own entries are - judged: what a group listed there lists in turn is that group's own to - judge, when its document is read. `node_type` says what each entry is, - so readings and models are judged alike. + lists too, at the joined key, of the same node type, and the same + document, as each is read: one it lists alone would be dropped by the + reference reader, one it lists as another type contradicts the tree, + and one it lists otherwise would give a reader two answers for one + node. Only the listing's own entries are judged: what a group listed + there lists in turn is that group's own to judge, when its document is + read, and a document with a problem of its own is judged by that. """ problems: list[ValidationProblem] = [] - for path, entry in listing.items(): + for path, entry in _reading_listing(listed).items(): if len(_below_faults(path)) != 0: # Its own reader reports a key that is no node's path. continue @@ -994,16 +996,58 @@ def _nested_listing_problems( ) problems.append(ValidationProblem(here, message, "invalid_value")) continue - listed, nested = node_types[joined], node_type(entry) - if listed is not None and nested is not None and listed != nested: + flat, nested = node_types[joined], _node_type_of(entry) + if flat is not None and nested is not None and flat != nested: + message = ( + f"expected {_an(flat)}, as the group lists {shown(f'/{joined}')}, got {_an(nested)}" + ) + problems.append(ValidationProblem(here, message, "invalid_value")) + continue + own = _own_listing_members(listed_members).get(path) + flat_members = members.get(joined) + if ( + own is not None + and flat_members is not None + and not _read_alike(readings[joined], flat_members, entry, own) + ): message = ( - f"expected {_an(listed)}, as the group lists {shown(f'/{joined}')}, " - f"got {_an(nested)}" + f"expected the document the group lists at {shown(f'/{joined}')}, got one " + "that reads otherwise: one node, one document" ) problems.append(ValidationProblem(here, message, "invalid_value")) return problems +def _own_listing_members( + listed: ArrayMembersV3 | GroupMembersV3 | None, +) -> Mapping[str, ArrayMembersV3 | GroupMembersV3]: + """The members of each document a listed group's own listing holds that a model can be built of; none for an array, a group listing nothing, or a document with a problem.""" + if isinstance(listed, GroupMembersV3) and listed.consolidated is not UNSET: + return listed.consolidated + return {} + + +def _read_alike( + reading: ZarrV3NodeMetadataReading, + members: ArrayMembersV3 | GroupMembersV3, + other: ZarrV3NodeMetadataReading, + other_members: ArrayMembersV3 | GroupMembersV3, +) -> bool: + """Whether two documents of one node read the same: arrays by what their models compare, groups by their own members, each listing judged where it sits.""" + if isinstance(reading, ZarrV3ArrayMetadataReading) and isinstance( + other, ZarrV3ArrayMetadataReading + ): + if not isinstance(members, ArrayMembersV3) or not isinstance(other_members, ArrayMembersV3): + return True + return array_key_of(reading, members) == array_key_of(other, other_members) + if isinstance(members, GroupMembersV3) and isinstance(other_members, GroupMembersV3): + return (json_text(members.attributes), json_text(members.extra_fields)) == ( + json_text(other_members.attributes), + json_text(other_members.extra_fields), + ) + return True + + def _an(node_type: NodeType) -> str: return "an array" if node_type == "array" else "a group" diff --git a/packages/zarr-metadata/tests/model/test_group.py b/packages/zarr-metadata/tests/model/test_group.py index 7521ca10e9..66fb7cd23d 100644 --- a/packages/zarr-metadata/tests/model/test_group.py +++ b/packages/zarr-metadata/tests/model/test_group.py @@ -1414,3 +1414,32 @@ def test_a_group_writes_no_empty_attributes() -> None: assert model.attributes == {} assert "attributes" not in model.to_json() assert json.loads(model.to_key_value()["zarr.json"]) == {"zarr_format": 3, "node_type": "group"} + + +def test_error_a_listed_group_s_own_listing_holds_the_documents_the_group_lists() -> None: + """One node, one document: a group listed in consolidated metadata may list a node the group lists too, but with the document the group holds at the joined key, as it is read; a document that reads otherwise is a problem at the nested entry, in the validator, the group model and the models given as entries alike.""" + same = _array(shape=(4,), attributes={"a": 1}) + other = _array(shape=(9,), attributes={"a": 1}) + agreed = _group( + consolidated_metadata=_inline( + b=_group(consolidated_metadata=_inline(c=same)), **{"b/c": same} + ) + ) + assert validate_group_metadata_v3(agreed) == () + contradicted = _group( + consolidated_metadata=_inline( + b=_group(consolidated_metadata=_inline(c=other)), **{"b/c": same} + ) + ) + found = validate_group_metadata_v3(contradicted) + at = ("consolidated_metadata", "metadata", "b", "consolidated_metadata", "metadata", "c") + assert [(p.loc, p.kind) for p in found] == [(at, "invalid_value")] + assert "/b/c" in found[0].message + with pytest.raises(MetadataValidationError): + ZarrV3GroupMetadata(contradicted) + inner = ZarrV3GroupMetadata(_group(consolidated_metadata=_inline(c=other))) + with pytest.raises(MetadataValidationError) as raised: + ZarrV3GroupMetadata( + _group(consolidated_metadata=_inline(b=inner, **{"b/c": ZarrV3ArrayMetadata(same)})) + ) + assert [(p.loc, p.kind) for p in raised.value.problems] == [(at, "invalid_value")] From 1a3ddc2d6b4979ae49ee200cf762edf7bdeab1fc Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Fri, 9 Oct 2026 21:43:52 +0200 Subject: [PATCH 90/94] fix(zarr-metadata): an integer JSON text cannot hold, and a v2 float fill value past float64, are read, not raised on An integer of more digits than the interpreter converts to text crashed the readers and the model's key with ValueError while validate_* found no problem; it is an invalid_value problem where it sits, the dimension lengths included. A v2 float or complex fill value written as an integer past the largest float64 raised OverflowError; it reads as the infinity of its sign, as the v3 float types read one. Assisted-by: ClaudeCode:claude-fable-5-1 Co-Authored-By: Claude Fable 5.1 --- .../zarr-metadata/src/zarr_metadata/_json.py | 19 +++++++++++++++ .../src/zarr_metadata/model/_validation.py | 5 +++- .../src/zarr_metadata/v2/data_type/scalar.py | 9 ++++++-- .../tests/model/test_refine_json.py | 23 +++++++++++++++++++ .../zarr-metadata/tests/v2/test_data_types.py | 21 +++++++++++++++++ 5 files changed, 74 insertions(+), 3 deletions(-) diff --git a/packages/zarr-metadata/src/zarr_metadata/_json.py b/packages/zarr-metadata/src/zarr_metadata/_json.py index 9b54531efd..865adeb0f1 100644 --- a/packages/zarr-metadata/src/zarr_metadata/_json.py +++ b/packages/zarr-metadata/src/zarr_metadata/_json.py @@ -427,6 +427,12 @@ def _refine(value: object, loc: tuple[str | int, ...], *, finite: bool) -> _Refi return None, ( ValidationProblem(loc, f"non-finite float {value!r} is not JSON", "invalid_value"), ) + if isinstance(value, int) and not isinstance(value, bool) and not _writable(value): + message = ( + f"an integer of {value.bit_length()} bits has more digits than JSON text holds " + "here, as sys.get_int_max_str_digits bounds it" + ) + return None, (ValidationProblem(loc, message, "invalid_value"),) if isinstance(value, (str, int, bool)) or value is None: return value, () if (past := nested_past_the_levels(value, loc)) is not None: @@ -465,6 +471,17 @@ def _refine(value: object, loc: tuple[str | int, ...], *, finite: bool) -> _Refi ) +def _writable(value: int) -> bool: + """Whether the interpreter converts `value` to text: what `json.dumps` and `repr` do, which `sys.set_int_max_str_digits` bounds; an integer past the bound could never be written back.""" + if value.bit_length() <= 64: + return True + try: + str(value) + except ValueError: + return False + return True + + def json_text(value: JSONValue) -> str: """`value` as the JSON text `json.dumps` writes for it, an object's keys sorted: what `==` compares of a JSON value the package does not interpret. @@ -485,6 +502,8 @@ def _as_object(value: object) -> dict[object, object]: def shown(value: object) -> str: """`value` as a problem's message shows it: as the JSON a document writes, `null` and `[1, 2]`, or by its repr when it is not JSON; what the interpreter will not write, an integer of too many digits or a value nested too deep, by saying so.""" + if isinstance(value, int) and not isinstance(value, bool) and not _writable(value): + return f"an integer of {value.bit_length()} bits" refined, problems = _refine(value, (), finite=False) if any(problem.message == _PAST_THE_LEVELS for problem in problems): # Nested past the levels a reader walks: said so, not left to the diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_validation.py b/packages/zarr-metadata/src/zarr_metadata/model/_validation.py index 1831a2023c..c9017236d7 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_validation.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/_validation.py @@ -253,7 +253,10 @@ def dimension_lengths( """ if key not in doc: return None, () - value = doc[key] + value, found = refine_json(doc[key], (key,)) + if len(found) != 0: + # An integer JSON text does not hold, or a value nested too deep. + return None, found if not _is_int_sequence(value): return None, (ValidationProblem((key,), "expected an array of integers", "invalid_type"),) if any(item < 0 for item in value): diff --git a/packages/zarr-metadata/src/zarr_metadata/v2/data_type/scalar.py b/packages/zarr-metadata/src/zarr_metadata/v2/data_type/scalar.py index 82baf7cccc..ac3f53d7f8 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v2/data_type/scalar.py +++ b/packages/zarr-metadata/src/zarr_metadata/v2/data_type/scalar.py @@ -127,8 +127,13 @@ def __call__( def _float_canonical( configuration: ZarrV2ScalarConfiguration, nested: Nested, value: ZarrV2FloatFillValue ) -> ZarrV2FloatFillValue: - """An integer written for a float is the float: `0` and `0.0` are one value.""" - return float(value) if isinstance(value, int) and not isinstance(value, bool) else value + """An integer written for a float is the float: `0` and `0.0` are one value, and one past the largest float64 is the infinity of its sign, as the v3 float types read it.""" + if isinstance(value, int) and not isinstance(value, bool): + try: + return float(value) + except OverflowError: + return "-Infinity" if value < 0 else "Infinity" + return value def _complex_canonical( diff --git a/packages/zarr-metadata/tests/model/test_refine_json.py b/packages/zarr-metadata/tests/model/test_refine_json.py index aad8edd2cb..c609343514 100644 --- a/packages/zarr-metadata/tests/model/test_refine_json.py +++ b/packages/zarr-metadata/tests/model/test_refine_json.py @@ -164,3 +164,26 @@ def test_bytes_past_the_levels_a_reader_walks_are_not_json_rather_than_nested() assert [(len(p.loc), p.kind, p.message[:31]) for p in problems] == [ (JSON_DEPTH, "invalid_type", "not a JSON-serializable value: ") ] + + +def test_error_an_integer_of_more_digits_than_json_text_holds_is_a_problem( + interpreter_writes_4300_digits: None, +) -> None: + """An integer the interpreter will not convert to text, as `sys.get_int_max_str_digits` bounds one, is an `invalid_value` problem where it sits, so no reader raises on it: the document could not be written back, and `validate_*` and `read_*` agree.""" + from zarr_metadata.model import ( + ZarrV3ArrayMetadata, + read_array_metadata_v3, + validate_array_metadata_v3, + ) + + refined, problems = refine_json({"x": [10**5000]}) + assert refined is None + assert [(p.loc, p.kind) for p in problems] == [(("x", 0), "invalid_value")] + assert "digits" in problems[0].message + document = {**ZarrV3ArrayMetadata.create_default(shape=(4,)).to_json()} + document["attributes"] = {"x": 10**5000} + found = validate_array_metadata_v3(document) + assert [(p.loc, p.kind) for p in found] == [(("attributes", "x"), "invalid_value")] + assert read_array_metadata_v3(document).problems == found + shaped = {**document, "attributes": {}, "shape": [10**5000]} + assert [p.loc for p in validate_array_metadata_v3(shaped)] == [("shape", 0)] diff --git a/packages/zarr-metadata/tests/v2/test_data_types.py b/packages/zarr-metadata/tests/v2/test_data_types.py index c339d13d35..483863f874 100644 --- a/packages/zarr-metadata/tests/v2/test_data_types.py +++ b/packages/zarr-metadata/tests/v2/test_data_types.py @@ -2,6 +2,8 @@ from __future__ import annotations +from typing import TYPE_CHECKING + import pytest from zarr_metadata.v2._definition import ZarrV2DataTypeDefinition @@ -20,6 +22,9 @@ resolve, ) +if TYPE_CHECKING: + from zarr_metadata import JSONValue + SCOPE = Context.of(*V2_DATA_TYPES) Loc = tuple[str | int, ...] @@ -233,3 +238,19 @@ def test_a_struct_record_type_sits_where_the_document_writes_it() -> None: ("dtype", "fields", 0, 1), ("dtype", "fields", 1, 1), ] + + +def test_a_float_fill_value_past_the_largest_float64_reads_as_an_infinity() -> None: + """An integer written for a float fill value that no float64 holds is the infinity of its sign, as the v3 float types read one, for a float and for each component of a complex: `read_array_metadata_v2` reads it, and the model equals one written `"Infinity"`.""" + from zarr_metadata.model import ZarrV2ArrayMetadata, read_array_metadata_v2 + + cases: list[tuple[str, JSONValue, JSONValue]] = [ + (" Date: Fri, 9 Oct 2026 21:45:27 +0200 Subject: [PATCH 91/94] fix(zarr-metadata): ScopeConflictError pickles and copies with its conflicts An exception's default reduce calls the constructor with its message, so a pickled or copied ScopeConflictError held each character of the message as a conflict. It reduces to its conflicts. Assisted-by: ClaudeCode:claude-fable-5-1 Co-Authored-By: Claude Fable 5.1 --- .../src/zarr_metadata/v3/_scope.py | 5 +++++ packages/zarr-metadata/tests/v3/test_scope.py | 17 +++++++++++++++++ 2 files changed, 22 insertions(+) diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/_scope.py b/packages/zarr-metadata/src/zarr_metadata/v3/_scope.py index b9bd878e0a..e38ed3cd34 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/_scope.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/_scope.py @@ -73,6 +73,11 @@ def __init__(self, conflicts: Sequence[Conflict]) -> None: self.conflicts: tuple[Conflict, ...] = tuple(conflicts) super().__init__("; ".join(str(conflict) for conflict in self.conflicts)) + def __reduce__(self) -> tuple[type[ScopeConflictError], tuple[tuple[Conflict, ...]]]: + # Pickled and copied as it was raised: an exception's default + # reduce calls the constructor with its message, not its conflicts. + return type(self), (self.conflicts,) + def claim_key(field: ResolvedField[Any]) -> ClaimKey | None: """The key `field` is claimed under: its kind and the name its definition is filed under, `r*` for `r16`; None for a field that names nothing.""" diff --git a/packages/zarr-metadata/tests/v3/test_scope.py b/packages/zarr-metadata/tests/v3/test_scope.py index d136cd6a8f..7721dde249 100644 --- a/packages/zarr-metadata/tests/v3/test_scope.py +++ b/packages/zarr-metadata/tests/v3/test_scope.py @@ -416,3 +416,20 @@ def test_a_gain_is_judged_by_what_the_definition_reads_not_by_spelling() -> None read, _ = resolve(shard, CodecDefinition, CORE) unread, _ = resolve({**shard, "must_understand": True}, CodecDefinition, nothing) assert refines(read, unread) + + +OTHER_GZIP = CodecDefinition(name="gzip", configuration=Empty, kind="bytes_bytes", size="static") +"""A second definition under the core gzip's name: what a join conflicts on.""" + + +def test_a_scope_conflict_error_pickles_and_copies_with_its_conflicts() -> None: + """`ScopeConflictError` pickles and copies as it was raised: the same conflicts, the same message, as `MetadataValidationError` does, so a conflict reported in another process reads the same here.""" + import copy + + with pytest.raises(ScopeConflictError) as raised: + Context.joined(CORE, Context.of(OTHER_GZIP)) + error = raised.value + for again in (pickle.loads(pickle.dumps(error)), copy.copy(error), copy.deepcopy(error)): + assert again.conflicts == error.conflicts + assert str(again) == str(error) + assert str(again).startswith("codec 'gzip'") From 54a4723dd22530d04200ac8d9d17bf1de73e9214 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Fri, 9 Oct 2026 21:59:38 +0200 Subject: [PATCH 92/94] fix(zarr-metadata): a v2 float fill value overflows at its dtype's width; ScopeConflictError keeps its notes A number past the largest float32 or float16 reads as the infinity of its sign, as v3 reads one and NumPy stores it, not only past float64; a complex component is held at half the itemsize. A pickled ScopeConflictError keeps its notes and attributes, as a ValueError does. Assisted-by: ClaudeCode:claude-fable-5-1 Co-Authored-By: Claude Fable 5.1 --- .../src/zarr_metadata/v2/data_type/scalar.py | 41 ++++++++++++++----- .../src/zarr_metadata/v3/_scope.py | 11 +++-- .../zarr-metadata/tests/v2/test_data_types.py | 9 +++- packages/zarr-metadata/tests/v3/test_scope.py | 2 + 4 files changed, 47 insertions(+), 16 deletions(-) diff --git a/packages/zarr-metadata/src/zarr_metadata/v2/data_type/scalar.py b/packages/zarr-metadata/src/zarr_metadata/v2/data_type/scalar.py index ac3f53d7f8..0dc2d593ab 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v2/data_type/scalar.py +++ b/packages/zarr-metadata/src/zarr_metadata/v2/data_type/scalar.py @@ -2,6 +2,8 @@ from __future__ import annotations +import math +import struct from dataclasses import dataclass from typing import TYPE_CHECKING, Annotated, Final, Literal, cast @@ -12,7 +14,7 @@ from zarr_metadata.v2._definition import ZarrV2DataTypeDefinition if TYPE_CHECKING: - from collections.abc import Iterator + from collections.abc import Iterator, Mapping from zarr_metadata.v3._definition import Nested @@ -127,13 +129,31 @@ def __call__( def _float_canonical( configuration: ZarrV2ScalarConfiguration, nested: Nested, value: ZarrV2FloatFillValue ) -> ZarrV2FloatFillValue: - """An integer written for a float is the float: `0` and `0.0` are one value, and one past the largest float64 is the infinity of its sign, as the v3 float types read it.""" - if isinstance(value, int) and not isinstance(value, bool): - try: - return float(value) - except OverflowError: - return "-Infinity" if value < 0 else "Infinity" - return value + """An integer written for a float is the float, `0` and `0.0` one value, and a number past the largest float of the dtype's width is the infinity of its sign, as the v3 float types read one and NumPy stores it.""" + return _held_in(value, configuration["itemsize"]) + + +def _held_in(value: ZarrV2FloatFillValue, itemsize: int) -> ZarrV2FloatFillValue: + """`value` as a float of `itemsize` bytes holds it: itself as a float, or the infinity of its sign past the largest such float; a width no float has keeps the value.""" + if isinstance(value, bool) or not isinstance(value, (int, float)): + return value + try: + held = float(value) + narrowed = held + if itemsize in _FLOAT_CODES: + # `struct` refuses a float16 past the largest, and packs a + # float32 past the largest as an infinity. + code = _FLOAT_CODES[itemsize] + (narrowed,) = struct.unpack(code, struct.pack(code, held)) + except OverflowError: + return "-Infinity" if value < 0 else "Infinity" + if math.isinf(narrowed): + return "-Infinity" if value < 0 else "Infinity" + return held + + +_FLOAT_CODES: Final[Mapping[int, str]] = {2: "e", 4: "f", 8: "d"} +"""The `struct` code of the float each size in bytes is: what refuses a value past the largest.""" def _complex_canonical( @@ -143,9 +163,10 @@ def _complex_canonical( if value is None: return None real, imag = value + width = configuration["itemsize"] // 2 return ( - cast("ZarrV2ComplexComponent", _float_canonical(configuration, nested, real)), - cast("ZarrV2ComplexComponent", _float_canonical(configuration, nested, imag)), + cast("ZarrV2ComplexComponent", _held_in(real, width)), + cast("ZarrV2ComplexComponent", _held_in(imag, width)), ) diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/_scope.py b/packages/zarr-metadata/src/zarr_metadata/v3/_scope.py index e38ed3cd34..b2a0fda61a 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/_scope.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/_scope.py @@ -73,10 +73,13 @@ def __init__(self, conflicts: Sequence[Conflict]) -> None: self.conflicts: tuple[Conflict, ...] = tuple(conflicts) super().__init__("; ".join(str(conflict) for conflict in self.conflicts)) - def __reduce__(self) -> tuple[type[ScopeConflictError], tuple[tuple[Conflict, ...]]]: - # Pickled and copied as it was raised: an exception's default - # reduce calls the constructor with its message, not its conflicts. - return type(self), (self.conflicts,) + def __reduce__( + self, + ) -> tuple[type[ScopeConflictError], tuple[tuple[Conflict, ...]], dict[str, object]]: + # Pickled and copied as it was raised, its notes and attributes kept: + # an exception's default reduce calls the constructor with its + # message, not its conflicts. + return type(self), (self.conflicts,), dict(self.__dict__) def claim_key(field: ResolvedField[Any]) -> ClaimKey | None: diff --git a/packages/zarr-metadata/tests/v2/test_data_types.py b/packages/zarr-metadata/tests/v2/test_data_types.py index 483863f874..2dfc06c992 100644 --- a/packages/zarr-metadata/tests/v2/test_data_types.py +++ b/packages/zarr-metadata/tests/v2/test_data_types.py @@ -240,15 +240,20 @@ def test_a_struct_record_type_sits_where_the_document_writes_it() -> None: ] -def test_a_float_fill_value_past_the_largest_float64_reads_as_an_infinity() -> None: - """An integer written for a float fill value that no float64 holds is the infinity of its sign, as the v3 float types read one, for a float and for each component of a complex: `read_array_metadata_v2` reads it, and the model equals one written `"Infinity"`.""" +def test_a_float_fill_value_past_the_largest_of_its_width_reads_as_an_infinity() -> None: + """A number written for a float fill value that no float of the dtype's width holds -- ` None: with pytest.raises(ScopeConflictError) as raised: Context.joined(CORE, Context.of(OTHER_GZIP)) error = raised.value + error.add_note("seen in a join") for again in (pickle.loads(pickle.dumps(error)), copy.copy(error), copy.deepcopy(error)): assert again.conflicts == error.conflicts assert str(again) == str(error) assert str(again).startswith("codec 'gzip'") + assert again.__notes__ == ["seen in a join"] From f6007db493ea1827e47d7ae38ee4a66fd0cbde28 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Fri, 9 Oct 2026 22:46:04 +0200 Subject: [PATCH 93/94] docs(zarr-metadata): models are pairs, not dataclasses; 3.1 wrote null consolidated metadata too; depth notes Assisted-by: ClaudeCode:claude-fable-5-1 Co-Authored-By: Claude Fable 5.1 --- packages/zarr-metadata/README.md | 28 +++++++++---------- .../src/zarr_metadata/_typed_json.py | 6 ++-- .../src/zarr_metadata/model/__init__.py | 4 +-- .../src/zarr_metadata/model/_group.py | 2 +- .../src/zarr_metadata/model/_json_schema.py | 8 +++--- .../src/zarr_metadata/model/_repair.py | 6 ++-- .../src/zarr_metadata/pydantic.py | 6 ++++ .../zarr-metadata/tests/model/test_group.py | 2 +- 8 files changed, 35 insertions(+), 27 deletions(-) diff --git a/packages/zarr-metadata/README.md b/packages/zarr-metadata/README.md index d997d2fb25..db48871bf3 100644 --- a/packages/zarr-metadata/README.md +++ b/packages/zarr-metadata/README.md @@ -13,9 +13,9 @@ Two layers and an optional integration: and [Zarr v3](https://zarr-specs.readthedocs.io/en/latest/v3/core/index.html) specifications, plus types for [`zarr-extensions`](https://github.com/zarr-developers/zarr-extensions/) and a few widely-used-but-unspecified entities (e.g. consolidated metadata). -- **Document models** (`zarr_metadata.model`): canonical frozen-dataclass - models of whole metadata documents, with validators, loc-aware - parsers, and store-key (de)serialization. A document produced by `to_json` +- **Document models** (`zarr_metadata.model`): each model is a metadata + document and the scope it was read in, with validators, loc-aware + readers, and store-key (de)serialization. A document produced by `to_json` shares no mutable state with the model that produced it. - **Optional Pydantic integration** (`zarr_metadata.pydantic`, requires Pydantic 2.13 or newer): each model as a Pydantic field type that validates @@ -99,9 +99,10 @@ Two choices the specs' words leave open, or settle two ways: `copy.deepcopy`, and `pickle` before Python 3.12, two: a document at the cap takes about half of the interpreter's default limit, and the rest is the caller's. -- **`consolidated_metadata: null` is a problem.** A zarr-python 3.0.x bug - wrote it; the spec says an object, and the package models nothing else - as right. A reader of those stores strips the key before reading. +- **`consolidated_metadata: null` is a problem.** zarr-python 3.0 and 3.1 + wrote it on a group they had not consolidated; the spec says an object, + and the package models nothing else as right. + `read_repaired_node_metadata_v3` removes it before reading. - **An extension is named as the spec names one**, `^[a-z][a-z0-9-_.]+$`, or by a URI, which earlier versions of the spec required; any other name is refused before a definition is asked, so `""` and `"foo/bar"` @@ -202,14 +203,13 @@ against its data type -- compares by its canonical spelling, as `canonical_of` and `canonical_fill_value` give it: `"NaN"` and `"0x7fc00000"` are one `float32` fill value, `0.0` and `-0.0` two, and a blosc with and without the `typesize` that `noshuffle` ignores one -codec. What it does not interpret -- attributes, extra fields, the -configuration of a field nothing in scope claims, and every member of a -v2 document -- compares as JSON text, which tells `true` from `1` and -`-0.0` from `0.0`, and takes `NaN` for itself. Equal models hash alike, -and may write two documents: `to_json` writes each as it was given. A -v3 model holds nothing that can be changed in place; a v2 model's hash -is of what its containers held when it was hashed, so one in a set, or a -key of a dict, is not changed in place. +codec, and a v2 `dtype` by its family and size, ` JSO For an editor that validates a `zarr.json` as it is written, or a validator in another language. JSON Schema draft 2020-12, as - `json_schema` writes one. Each extension point is a field as - `field_json_schema` writes one in `context`: one a definition in scope - reads, or a name none of them claims. The fill value is the JSON shape + `json_schema` writes one. Each extension point is a field as a scope + reads one in `context`: one a definition in scope reads, with the + configuration its definition declares, or a name none of them claims. The fill value is the JSON shape the data type's definition declares for one -- an `int8`'s an integer in [-128, 127] -- when the document names a data type in scope. A group's `consolidated_metadata` holds array and group documents, by - path; a `null` one, which a zarr-python 3.0.x bug wrote, is refused, as + path; a `null` one, which zarr-python 3.0 and 3.1 wrote, is refused, as the validator refuses it. Each document is in `$defs` under the name of its TypedDict: `ZarrV3ArrayMetadataJSON` is an array's alone. diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_repair.py b/packages/zarr-metadata/src/zarr_metadata/model/_repair.py index 7089b1e41d..f0e8b91062 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_repair.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/_repair.py @@ -94,8 +94,8 @@ class ZarrV3ZeroChunkArrayMetadataJSON(TypedDict): class ZarrV3NullConsolidatedGroupMetadataJSON(TypedDict): """The members of a group document the `null_consolidated_metadata` repair reads. - zarr-python 3.0.x wrote `"consolidated_metadata": null` on a group it had - not consolidated, which the convention does not allow: the member is + zarr-python 3.0 and 3.1 wrote `"consolidated_metadata": null` on a group + they had not consolidated, which the convention does not allow: the member is an object, or absent. The repair removes it, which is what the writer meant. """ @@ -180,7 +180,7 @@ def _null_consolidated_metadata( Repair( (*at, ZARR_V3_CONSOLIDATED_METADATA_KEY), "null_consolidated_metadata", - "a consolidated_metadata of null, as zarr-python 3.0.x wrote it, removed", + "a consolidated_metadata of null, as zarr-python 3.0 and 3.1 wrote it, removed", ) ) return {key: item for key, item in document.items() if key != ZARR_V3_CONSOLIDATED_METADATA_KEY} diff --git a/packages/zarr-metadata/src/zarr_metadata/pydantic.py b/packages/zarr-metadata/src/zarr_metadata/pydantic.py index ccb8bddc76..ba40f4a610 100644 --- a/packages/zarr-metadata/src/zarr_metadata/pydantic.py +++ b/packages/zarr-metadata/src/zarr_metadata/pydantic.py @@ -38,6 +38,12 @@ `model_config`; a `TypeAdapter` over a field type takes no `config`, so write its value with the model's `to_key_value` instead. +Pydantic's own JSON paths bound nesting below the 256 levels the readers +walk: `validate_json` refuses a document nested about 200 levels deep, +and `dump_json` one nested about 255, each with its own error. A document +that deep goes through `validate_python` on parsed JSON, and is written +with the model's `to_key_value`. + Usage: import zarr_metadata.pydantic as zmp diff --git a/packages/zarr-metadata/tests/model/test_group.py b/packages/zarr-metadata/tests/model/test_group.py index 66fb7cd23d..93e0307c4c 100644 --- a/packages/zarr-metadata/tests/model/test_group.py +++ b/packages/zarr-metadata/tests/model/test_group.py @@ -1322,7 +1322,7 @@ def test_group_must_understand_fields_partition() -> None: def test_error_a_null_consolidated_metadata_is_a_value_the_document_wrote() -> None: - """A zarr-python 3.0.x bug wrote `consolidated_metadata: null`; the spec says an object, and the package models nothing else as right: a reader of those stores strips the key first.""" + """zarr-python 3.0 and 3.1 wrote `consolidated_metadata: null`; the spec says an object, and the package models nothing else as right: a reader of those stores strips the key first.""" null_doc = {"zarr_format": 3, "node_type": "group", "consolidated_metadata": None} assert [(p.loc, p.kind) for p in validate_group_metadata_v3(null_doc)] == [ (("consolidated_metadata",), "invalid_type") From a6664db7fcae1259c404edcb7a4c4c64e290040d Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Fri, 9 Oct 2026 22:55:17 +0200 Subject: [PATCH 94/94] fix(zarr-metadata): the cleanups a review found: messages, guards, v2 paths and hierarchy, sized fill values, enum values, scope invariants A conflict names the field as the document writes it and tells two definitions of one name apart. is_group_metadata_v3 asks tuples of the documents a group lists. A .zmetadata key with a '.' or '..' segment is a problem, and the nodes of a .zmetadata make a hierarchy, as a v3 group's consolidated metadata does. A void or struct fill value decodes to exactly the item's size and a byte string's to at most it. A str, int or float subclass, a StrEnum or IntEnum member, is refined to the value JSON writes for it. Context(tables) keeps the invariants Context.of keeps. The raw-bits pattern's docstring says what the package does: a hundred digits is more than any size, so more is a name. Assisted-by: ClaudeCode:claude-fable-5-1 Co-Authored-By: Claude Fable 5.1 --- .../zarr-metadata/src/zarr_metadata/_json.py | 9 +- .../src/zarr_metadata/model/_array.py | 6 +- .../src/zarr_metadata/model/_group.py | 93 ++++++++++++++++--- .../src/zarr_metadata/model/_validation.py | 11 ++- .../zarr_metadata/v2/data_type/fixed_width.py | 50 +++++++++- .../src/zarr_metadata/v2/data_type/struct.py | 38 +++++++- .../src/zarr_metadata/v3/_definition.py | 4 +- .../src/zarr_metadata/v3/_registry.py | 18 ++-- .../src/zarr_metadata/v3/_scope.py | 17 +++- .../zarr-metadata/tests/model/test_group.py | 19 ++++ .../zarr-metadata/tests/model/test_pair_v2.py | 48 +++++++++- .../tests/model/test_refine_json.py | 29 ++++++ .../tests/model/test_v2_scope_reads.py | 2 +- .../zarr-metadata/tests/v2/test_data_types.py | 44 +++++++++ packages/zarr-metadata/tests/v3/test_scope.py | 49 +++++++++- 15 files changed, 391 insertions(+), 46 deletions(-) diff --git a/packages/zarr-metadata/src/zarr_metadata/_json.py b/packages/zarr-metadata/src/zarr_metadata/_json.py index 865adeb0f1..dee2807796 100644 --- a/packages/zarr-metadata/src/zarr_metadata/_json.py +++ b/packages/zarr-metadata/src/zarr_metadata/_json.py @@ -423,7 +423,7 @@ def _refine(value: object, loc: tuple[str | int, ...], *, finite: bool) -> _Refi """`refine_json`, a non-finite number being JSON unless `finite`.""" if isinstance(value, float): if not finite or math.isfinite(value): - return value, () + return float(value), () return None, ( ValidationProblem(loc, f"non-finite float {value!r} is not JSON", "invalid_value"), ) @@ -433,8 +433,13 @@ def _refine(value: object, loc: tuple[str | int, ...], *, finite: bool) -> _Refi "here, as sys.get_int_max_str_digits bounds it" ) return None, (ValidationProblem(loc, message, "invalid_value"),) - if isinstance(value, (str, int, bool)) or value is None: + if isinstance(value, bool) or value is None: return value, () + if isinstance(value, str): + # A subclass -- a `StrEnum` member -- as the string JSON writes for it. + return str.__str__(value), () + if isinstance(value, int): + return int(value), () if (past := nested_past_the_levels(value, loc)) is not None: return None, (past,) if is_object(value): diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_array.py b/packages/zarr-metadata/src/zarr_metadata/model/_array.py index 027495087b..5350b10cdd 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_array.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/_array.py @@ -447,10 +447,12 @@ def located_conflicts( located: list[Conflict] = [] placed = list(fields) for conflict in conflicts: - places = [loc for loc, field in placed if claim_key(field) == conflict.key] + places = [(loc, field) for loc, field in placed if claim_key(field) == conflict.key] if len(places) == 0: located.append(conflict) - located.extend(dataclasses.replace(conflict, loc=loc) for loc in places) + located.extend( + dataclasses.replace(conflict, loc=loc, written=field.name) for loc, field in places + ) return tuple(located) diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_group.py b/packages/zarr-metadata/src/zarr_metadata/model/_group.py index 7ec214f65c..2be9088767 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_group.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/_group.py @@ -53,6 +53,7 @@ attributes_of, check_literal, dump_store_json, + is_canonical_array_metadata_v3, load_store_json, members_past_the_levels, missing_keys, @@ -77,7 +78,14 @@ ZarrV3Context, scoped, ) -from zarr_metadata.v3._scope import Claims, Conflict, ScopeConflictError, claims_of, kind_name +from zarr_metadata.v3._scope import ( + Claims, + Conflict, + ScopeConflictError, + claims_of, + definition_said, + kind_name, +) from zarr_metadata.v3.array import ZarrV3ExtensionField from zarr_metadata.v3.consolidated import ZARR_V3_CONSOLIDATED_METADATA_KEY from zarr_metadata.v3.group import ZARR_V3_GROUP_METADATA_STORE_KEY, ZarrV3GroupMetadataJSON @@ -89,7 +97,7 @@ from zarr_metadata.v2.attributes import ZarrV2AttributesStoreKey from zarr_metadata.v2.consolidated import ZarrV2ConsolidatedMetadataStoreKey from zarr_metadata.v2.group import ZarrV2GroupMetadataJSON, ZarrV2GroupMetadataStoreKey - from zarr_metadata.v3._definition import Definition, ResolvedField + from zarr_metadata.v3._definition import ResolvedField from zarr_metadata.v3.array import ZarrV3ArrayMetadataJSON from zarr_metadata.v3.consolidated import ZarrV3ConsolidatedMetadataJSON from zarr_metadata.v3.group import ZarrV3GroupMetadataJSONPartial, ZarrV3GroupMetadataStoreKey @@ -849,23 +857,16 @@ def _conflict_said(conflict: Conflict, written: str | None) -> str: """What a conflict between a model entry's scope and the group's says: the kind and the name as the document writes it, what read it, and how the group's scope reads it -- by another definition, told apart from the model's where the two print alike, or by none.""" kind, filed = conflict.key name = filed if written is None else written - claimed = _definition_said(conflict.claimed) + claimed = definition_said(conflict.claimed) head = f"expected a document read in the group's scope, got a model that reads the {kind_name(kind)} {name!r}" if conflict.found is None: return f"{head} by {claimed}, which the group's scope leaves unclaimed" - found = _definition_said(conflict.found) + found = definition_said(conflict.found) if claimed == found: return f"{head} by another definition than the one the group's scope reads it by, {found}" return f"{head} by {claimed}, which the group's scope reads by {found}" -def _definition_said(definition: Definition[Any] | None) -> str: - """A definition as a message tells it from another of the same name: by the TypedDict its configuration is.""" - if definition is None: - return "no definition" - return f"{definition!r} of {definition.configuration.__qualname__}" - - def _with_problems( reading: ZarrV3NodeMetadataReading, problems: tuple[ValidationProblem, ...] ) -> ZarrV3NodeMetadataReading: @@ -1210,11 +1211,34 @@ def is_group_metadata_v3( ) -> TypeGuard[ZarrV3GroupMetadataJSON]: """Whether `value` is a v3 group document `validate_group_metadata_v3` finds nothing wrong with, written with tuples.""" scope = scoped(context, CORE_AND_EXTENSIONS) - return is_canonical_json(value, finite=False) and not validate_group_metadata_v3( - value, context=scope + return ( + is_canonical_json(value, finite=False) + and _is_canonical_group_metadata_v3(value) + and not validate_group_metadata_v3(value, context=scope) ) +def _is_canonical_group_metadata_v3(value: object) -> bool: + """Whether `value`, a group document, is written with the containers `ZarrV3GroupMetadataJSON` declares: each document its consolidated metadata lists as its TypedDict has it, arrays as `is_canonical_array_metadata_v3` asks, groups so again.""" + if not is_object(value) or not isinstance(value, dict): + return False + member = value.get(ZARR_V3_CONSOLIDATED_METADATA_KEY) + if not is_object(member): + return True + entries = member.get("metadata") + if not is_object(entries): + return True + for entry in entries.values(): + if not is_object(entry): + continue + node_type = entry.get("node_type") + if node_type == "array" and not is_canonical_array_metadata_v3(entry): + return False + if node_type == "group" and not _is_canonical_group_metadata_v3(entry): + return False + return True + + def parse_group_metadata_v3( value: object, *, context: ZarrV3Context | None = None ) -> ZarrV3GroupMetadataJSON: @@ -1583,10 +1607,20 @@ def _entries_by_path( problems: list[ValidationProblem] = [] for key in entries: path, _, name = key.rpartition("/") - # A node's path, as a store names it: segments joined by one `/`. - path = "/".join(segment for segment in path.split("/") if segment != "") if name not in _NODE_FILES: continue + # A node's path, as a store names it: segments joined by one `/`. + segments = [segment for segment in path.split("/") if segment != ""] + if any(segment in (".", "..") for segment in segments): + problems.append( + ValidationProblem( + ("metadata", key), + f"expected the path of a node, got {key!r}, which has a '.' or '..' segment", + "invalid_value", + ) + ) + continue + path = "/".join(segments) files = by_path.setdefault(path, {}) if name in files: problems.append( @@ -1655,11 +1689,40 @@ def _read_consolidated_v2( problems.extend(found) if node is not None: nodes[path] = node + problems.extend(_hierarchy_problems_v2(by_path)) if len(problems) != 0: nodes = {} return refined, nodes, with_input(problems, doc) +def _hierarchy_problems_v2(by_path: Mapping[str, Mapping[str, str]]) -> list[ValidationProblem]: + """What keeps the nodes a `.zmetadata` holds from making a hierarchy, as `hierarchy_problems` judges one: a node below an array at its entry, a group missing above a node at the entry it would have; an orphan `.zattrs` is no node.""" + # The root is a group unless an entry says otherwise: a `.zmetadata` + # need not list its own `.zgroup`. + nodes: dict[str, NodeType] = {"/": "group"} + for path, names in by_path.items(): + if ZARR_V2_ARRAY_METADATA_STORE_KEY in names: + nodes[f"/{path}"] = "array" + elif ZARR_V2_GROUP_METADATA_STORE_KEY in names: + nodes[f"/{path}"] = "group" + problems: list[ValidationProblem] = [] + for found in hierarchy_problems(nodes): + at = found.loc[0] + path = at[1:] if isinstance(at, str) else "" + names = by_path.get(path, {}) + key = names.get( + ZARR_V2_ARRAY_METADATA_STORE_KEY, + names.get( + ZARR_V2_GROUP_METADATA_STORE_KEY, + f"{path}/{ZARR_V2_GROUP_METADATA_STORE_KEY}" + if path != "" + else ZARR_V2_GROUP_METADATA_STORE_KEY, + ), + ) + problems.append(ValidationProblem(("metadata", key), found.message, found.kind)) + return problems + + def _read_node_v2( path: str, names: Mapping[str, str], entries: Mapping[str, JSONValue], context: Context ) -> tuple[ZarrV2NodeMetadata | None, list[ValidationProblem]]: diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_validation.py b/packages/zarr-metadata/src/zarr_metadata/model/_validation.py index c9017236d7..bf7c68d38c 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_validation.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/_validation.py @@ -175,8 +175,11 @@ def unexpected_keys( def check_literal( doc: Mapping[object, object], key: str, expected: object ) -> tuple[ValidationProblem, ...]: - """One problem if `doc[key]` is present but not `expected`: of its type, when it is not of `expected`'s JSON type, else of its value.""" - if key in doc and (type(doc[key]) is not type(expected) or doc[key] != expected): + """One problem if `doc[key]` is present but not `expected`: of its type, when it is not of `expected`'s JSON type, else of its value; a value is judged as refined, so a `StrEnum` member is its string.""" + if key not in doc: + return () + value, found = refine_json(doc[key], (key,)) + if len(found) != 0 or type(value) is not type(expected) or value != expected: return (outside_of((key,), doc[key], (expected,)),) return () @@ -286,7 +289,7 @@ def _is_canonical_metadata_field_v3(value: object) -> bool: return isinstance(value, (str, dict)) -def _is_canonical_array_metadata_v3(value: object) -> bool: +def is_canonical_array_metadata_v3(value: object) -> bool: """Whether a validated v3 array document matches `ZarrV3ArrayMetadataJSON` at runtime.""" if not is_object(value) or not isinstance(value, dict): return False @@ -681,7 +684,7 @@ def is_array_metadata_v3( return ( _is_canonical_json(value, finite=False) and not validate_array_metadata_v3(value, context=scope) - and _is_canonical_array_metadata_v3(value) + and is_canonical_array_metadata_v3(value) ) diff --git a/packages/zarr-metadata/src/zarr_metadata/v2/data_type/fixed_width.py b/packages/zarr-metadata/src/zarr_metadata/v2/data_type/fixed_width.py index e870bde388..d1a1d0273f 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v2/data_type/fixed_width.py +++ b/packages/zarr-metadata/src/zarr_metadata/v2/data_type/fixed_width.py @@ -2,6 +2,7 @@ from __future__ import annotations +import base64 from typing import TYPE_CHECKING, Final from zarr_metadata._json import ValidationProblem, shown @@ -22,6 +23,13 @@ def base64_fill_value_rules( configuration: object, nested: Nested, value: ZarrV2Base64FillValue ) -> Iterator[ValidationProblem]: """A string of standard-alphabet base64, when it is not null.""" + yield from sized_base64_rules(value, None, exact=True) + + +def sized_base64_rules( + value: ZarrV2Base64FillValue, size: int | None, *, exact: bool +) -> Iterator[ValidationProblem]: + """A string of standard-alphabet base64 of `size` bytes, or of at most `size` when not `exact`, when it is not null; any size when `size` is None, which no item of a known size gives.""" if value is None: return try: @@ -30,6 +38,31 @@ def base64_fill_value_rules( yield ValidationProblem( (), f"expected standard-alphabet base64, got {shown(value)}", "invalid_value" ) + return + if size is None: + return + held = base64.b64decode(value) + if (len(held) != size) if exact else (len(held) > size): + bound = "" if exact else "at most " + yield ValidationProblem( + (), + f"expected base64 of {bound}{size} bytes, the item's size, got {len(held)} bytes", + "invalid_value", + ) + + +def _void_fill_value_rules( + configuration: ZarrV2ScalarConfiguration, nested: Nested, value: ZarrV2Base64FillValue +) -> Iterator[ValidationProblem]: + """Base64 of exactly the item's bytes.""" + yield from sized_base64_rules(value, configuration["itemsize"], exact=True) + + +def _bytes_fill_value_rules( + configuration: ZarrV2ScalarConfiguration, nested: Nested, value: ZarrV2Base64FillValue +) -> Iterator[ValidationProblem]: + """Base64 of at most the item's bytes: NumPy pads a shorter byte string with zeros.""" + yield from sized_base64_rules(value, configuration["itemsize"], exact=False) BYTES_V2: Final = ZarrV2DataTypeDefinition( @@ -38,9 +71,9 @@ def base64_fill_value_rules( rules=sized(None, None), canonical=orderless_at(None), fill_value=ZarrV2Base64FillValue, - fill_value_rules=base64_fill_value_rules, + fill_value_rules=_bytes_fill_value_rules, ) -"""`|S`: byte strings of `n` bytes; the fill value base64 of one.""" +"""`|S`: byte strings of `n` bytes; the fill value base64 of at most `n`.""" STR_V2: Final = ZarrV2DataTypeDefinition( name="str", @@ -56,8 +89,15 @@ def base64_fill_value_rules( rules=sized(None, None), canonical=orderless_at(None), fill_value=ZarrV2Base64FillValue, - fill_value_rules=base64_fill_value_rules, + fill_value_rules=_void_fill_value_rules, ) -"""`|V`: `n` bytes of no type; the fill value base64 of them.""" +"""`|V`: `n` bytes of no type; the fill value base64 of exactly `n`.""" -__all__ = ["BYTES_V2", "STR_V2", "VOID_V2", "ZarrV2Base64FillValue", "base64_fill_value_rules"] +__all__ = [ + "BYTES_V2", + "STR_V2", + "VOID_V2", + "ZarrV2Base64FillValue", + "base64_fill_value_rules", + "sized_base64_rules", +] diff --git a/packages/zarr-metadata/src/zarr_metadata/v2/data_type/struct.py b/packages/zarr-metadata/src/zarr_metadata/v2/data_type/struct.py index 08ff8e86d2..979e6eb474 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v2/data_type/struct.py +++ b/packages/zarr-metadata/src/zarr_metadata/v2/data_type/struct.py @@ -2,14 +2,16 @@ from __future__ import annotations -from typing import TYPE_CHECKING, Annotated, Final +import math +from typing import TYPE_CHECKING, Annotated, Any, Final, cast from annotated_types import Ge from typing_extensions import ReadOnly, TypedDict from zarr_metadata._json import ValidationProblem from zarr_metadata.v2._definition import STRUCT_NAME, ZarrV2DataTypeDefinition, ZarrV2DataTypeField -from zarr_metadata.v2.data_type.fixed_width import ZarrV2Base64FillValue, base64_fill_value_rules +from zarr_metadata.v2.data_type.fixed_width import ZarrV2Base64FillValue, sized_base64_rules +from zarr_metadata.v3._definition import AcceptedField if TYPE_CHECKING: from collections.abc import Iterator @@ -50,12 +52,42 @@ def _rules(configuration: ZarrV2StructConfiguration, nested: Nested) -> Iterator ) +def _fill_value_rules( + configuration: ZarrV2StructConfiguration, nested: Nested, value: ZarrV2Base64FillValue +) -> Iterator[ValidationProblem]: + """Base64 of exactly one record, when every field's size is known.""" + yield from sized_base64_rules(value, record_size(configuration, nested), exact=True) + + +def record_size(configuration: ZarrV2StructConfiguration, nested: Nested) -> int | None: + """The bytes one record of `configuration` takes: each field's item size times the product of its subarray shape, a nested struct's its own record; None when a field's size is not known, as `|O`'s is not, or its dtype was not read.""" + total = 0 + for index, record in enumerate(configuration["fields"]): + field = nested.get(("fields", index, 1)) + if not isinstance(field, AcceptedField): + return None + size = _item_size(field) + if size is None: + return None + shape = record[2] if len(record) == 3 else () + total += size * math.prod(shape) + return total + + +def _item_size(field: AcceptedField[Any]) -> int | None: + """The bytes one item of the dtype `field` read takes; None when not known.""" + if field.name == STRUCT_NAME or "fields" in field.configuration: + return record_size(cast("ZarrV2StructConfiguration", field.configuration), field.nested) + size = field.configuration.get("itemsize") + return size if isinstance(size, int) and not isinstance(size, bool) else None + + STRUCT_V2: Final = ZarrV2DataTypeDefinition( name=STRUCT_NAME, configuration=ZarrV2StructConfiguration, rules=_rules, fill_value=ZarrV2Base64FillValue, - fill_value_rules=base64_fill_value_rules, + fill_value_rules=_fill_value_rules, ) """A structured type: an array of field records; the fill value base64 of one record (https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v2/v2.0.rst#L191-L193).""" diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py b/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py index 2a32921910..4457eb8695 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py @@ -422,7 +422,9 @@ def _function_members(kind: type[Definition[Any]]) -> tuple[str, ...]: `r\uff11\uff16` would be read as sixteen bits, and a third-party name spelled that way would be taken for raw bits. A size the spec does not allow -- `r0`, `r12` -- is still raw bits, so it is reported as a bad -size rather than passed as an unknown extension. +size rather than passed as an unknown extension. A hundred digits is +more than any size: `r` and more digits than that is a name, as the spec +allows one, which nothing in scope claims. """ diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/_registry.py b/packages/zarr-metadata/src/zarr_metadata/v3/_registry.py index 6150cc99d1..38c9dfc012 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/_registry.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/_registry.py @@ -88,6 +88,17 @@ class Context(Generic[F]): tables: Tables + def __post_init__(self) -> None: + # What `of` builds, the constructor refuses to build otherwise: each + # key a kind, and every kind of one format. + formats = sorted({format_of(kind) for kind in self.tables}) + if len(formats) > 1: + msg = ( + "a scope reads documents of one Zarr format; these definitions are of kinds of " + f"formats {' and '.join(f'v{found}' for found in formats)}" + ) + raise TypeError(msg) + @property def format(self) -> Literal[2, 3] | None: """The Zarr format the definitions in scope read, 2 or 3; None for a scope that files nothing.""" @@ -113,13 +124,6 @@ def of(cls, *definitions: Definition[Any]) -> Context[Any]: ) raise TypeError(msg) tables.setdefault(kind, {})[definition.name] = definition - formats = sorted({format_of(kind) for kind in tables}) - if len(formats) > 1: - msg = ( - "a scope reads documents of one Zarr format; these definitions are of kinds of " - f"formats {' and '.join(f'v{found}' for found in formats)}" - ) - raise TypeError(msg) return cls( MappingProxyType({kind: MappingProxyType(table) for kind, table in tables.items()}) ) diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/_scope.py b/packages/zarr-metadata/src/zarr_metadata/v3/_scope.py index b2a0fda61a..5aa75c8e02 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/_scope.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/_scope.py @@ -50,11 +50,24 @@ class Conflict: claimed: Definition[Any] | None found: Definition[Any] | None loc: Loc | None = None + written: str | None = None + """The name the document writes where the conflict was found, `r16` for a key of `r*`; None when no document is at hand.""" def __str__(self) -> str: - kind, name = self.key + kind, filed = self.key + name = filed if self.written is None else self.written where = "" if self.loc is None else f" at {self.loc!r}" - return f"{kind_name(kind)} {name!r}{where}: claimed {self.claimed!r}, found {self.found!r}" + return ( + f"{kind_name(kind)} {name!r}{where}: claimed {definition_said(self.claimed)}, " + f"found {definition_said(self.found)}" + ) + + +def definition_said(definition: Definition[Any] | None) -> str: + """A definition as a message tells it from another of the same name: by the TypedDict its configuration is; "no definition" for None.""" + if definition is None: + return "no definition" + return f"{definition!r} of {definition.configuration.__qualname__}" def kind_name(kind: type[Definition[Any]]) -> str: diff --git a/packages/zarr-metadata/tests/model/test_group.py b/packages/zarr-metadata/tests/model/test_group.py index 93e0307c4c..d840574b5d 100644 --- a/packages/zarr-metadata/tests/model/test_group.py +++ b/packages/zarr-metadata/tests/model/test_group.py @@ -21,6 +21,7 @@ ) from zarr_metadata.model import ( UNSET, + is_array_metadata_v3, ) from zarr_metadata.model._array import ZarrV3ArrayMetadata, ZarrV3ArrayMetadataUpdate from zarr_metadata.model._group import ( @@ -1443,3 +1444,21 @@ def test_error_a_listed_group_s_own_listing_holds_the_documents_the_group_lists( _group(consolidated_metadata=_inline(b=inner, **{"b/c": ZarrV3ArrayMetadata(same)})) ) assert [(p.loc, p.kind) for p in raised.value.problems] == [(at, "invalid_value")] + + +def test_the_group_guard_asks_tuples_of_the_documents_the_group_lists() -> None: + """`is_group_metadata_v3` says no to a group whose consolidated metadata holds an array document written with lists, as `is_array_metadata_v3` says no to that document: a guard narrows to the TypedDict, whose arrays are tuples, at every level.""" + array = _array() + listed = dict(array) + listed["shape"] = list(array["shape"]) # pyright: ignore[reportArgumentType] + assert not is_array_metadata_v3(listed) + group = _group(consolidated_metadata=_inline(a=listed)) + assert validate_group_metadata_v3(group) == () + assert not is_group_metadata_v3(group) + assert is_group_metadata_v3(parse_group_metadata_v3(group)) + nested = _group( + consolidated_metadata=_inline( + g=_group(consolidated_metadata=_inline(a=listed)), **{"g/a": array} + ) + ) + assert not is_group_metadata_v3(nested) diff --git a/packages/zarr-metadata/tests/model/test_pair_v2.py b/packages/zarr-metadata/tests/model/test_pair_v2.py index 96fec0df10..b09502401b 100644 --- a/packages/zarr-metadata/tests/model/test_pair_v2.py +++ b/packages/zarr-metadata/tests/model/test_pair_v2.py @@ -419,10 +419,54 @@ def test_error_a_second_key_for_a_node_file_hides_no_other_problem() -> None: ) assert set( ZarrV2ConsolidatedMetadata( - {"zarr_consolidated_format": 1, "metadata": {"x//y/.zarray": ZARRAY}} + { + "zarr_consolidated_format": 1, + "metadata": {"x/.zgroup": {"zarr_format": 2}, "x//y/.zarray": ZARRAY}, + } ).nodes - ) == {"x/y"} + ) == {"x", "x/y"} assert set(one.nodes) == {"a"} assert one == ZarrV2ConsolidatedMetadata( {"zarr_consolidated_format": 1, "metadata": {"a/.zarray": ZARRAY}} ) + + +def test_error_a_key_with_a_dot_segment_is_no_node_s_path() -> None: + """A `.zmetadata` key whose path has a `.` or `..` segment names no node, as a leading or repeated `/` does not: it is a problem at the key, and not a second node beside the one it would name.""" + with pytest.raises(MetadataValidationError) as raised: + ZarrV2ConsolidatedMetadata( + { + "zarr_consolidated_format": 1, + "metadata": { + ".zgroup": {"zarr_format": 2}, + "./a/.zarray": ZARRAY, + "b/../a/.zarray": ZARRAY, + }, + } + ) + assert [(p.loc, p.kind) for p in raised.value.problems] == [ + (("metadata", "./a/.zarray"), "invalid_value"), + (("metadata", "b/../a/.zarray"), "invalid_value"), + ] + + +@pytest.mark.parametrize( + ("entries", "at", "kind"), + [ + ( + {".zgroup": {"zarr_format": 2}, "a/.zarray": ZARRAY, "a/b/.zarray": ZARRAY}, + "a/b/.zarray", + "invalid_value", + ), + ({".zarray": ZARRAY, "x/.zgroup": {"zarr_format": 2}}, "x/.zgroup", "invalid_value"), + ({".zgroup": {"zarr_format": 2}, "x/y/.zarray": ZARRAY}, "x/.zgroup", "missing_key"), + ], + ids=["below-an-array", "below-the-root-array", "missing-parent"], +) +def test_error_the_nodes_of_a_zmetadata_make_a_hierarchy( + entries: dict[str, Any], at: str, kind: str +) -> None: + """The nodes a `.zmetadata` holds make a hierarchy, as a v3 group's consolidated metadata does: a node below an array is a problem at its entry, and a group missing above a node a `missing_key` at the entry the group would have.""" + with pytest.raises(MetadataValidationError) as raised: + ZarrV2ConsolidatedMetadata({"zarr_consolidated_format": 1, "metadata": entries}) + assert [(p.loc, p.kind) for p in raised.value.problems] == [(("metadata", at), kind)] diff --git a/packages/zarr-metadata/tests/model/test_refine_json.py b/packages/zarr-metadata/tests/model/test_refine_json.py index c609343514..d5594d4ebd 100644 --- a/packages/zarr-metadata/tests/model/test_refine_json.py +++ b/packages/zarr-metadata/tests/model/test_refine_json.py @@ -187,3 +187,32 @@ def test_error_an_integer_of_more_digits_than_json_text_holds_is_a_problem( assert read_array_metadata_v3(document).problems == found shaped = {**document, "attributes": {}, "shape": [10**5000]} assert [p.loc for p in validate_array_metadata_v3(shaped)] == [("shape", 0)] + + +def test_a_str_or_int_subclass_is_refined_to_the_json_type_it_is() -> None: + """A `StrEnum` member, an `IntEnum` member, or any `str`, `int` or `float` subclass, is refined to the plain value JSON writes for it, so a `Literal` and a tag read it as the value, while `bool` stays apart from `int`.""" + import enum + + from zarr_metadata.model import ZarrV3GroupMetadata + + class NodeType(enum.StrEnum): + GROUP = "group" + + class Format(enum.IntEnum): + THREE = 3 + + class Score(float): + pass + + refined, problems = refine_json( + {"a": NodeType.GROUP, "b": Format.THREE, "c": Score(1.5), "d": True} + ) + assert problems == () + assert refined == {"a": "group", "b": 3, "c": 1.5, "d": True} + assert isinstance(refined, dict) + assert all(type(refined[key]) in (str, int, float, bool) for key in refined) + assert type(refined["d"]) is bool + model = ZarrV3GroupMetadata.from_json( + {"zarr_format": Format.THREE, "node_type": NodeType.GROUP} + ) + assert model.to_json() == {"zarr_format": 3, "node_type": "group"} diff --git a/packages/zarr-metadata/tests/model/test_v2_scope_reads.py b/packages/zarr-metadata/tests/model/test_v2_scope_reads.py index 58a708084c..03a2975d0f 100644 --- a/packages/zarr-metadata/tests/model/test_v2_scope_reads.py +++ b/packages/zarr-metadata/tests/model/test_v2_scope_reads.py @@ -20,7 +20,7 @@ "changes", [ {"dtype": " None: + """A void or struct fill value decodes to exactly the item's size -- a record's the sum of its fields', each times the product of its subarray shape, nested structs included -- and a byte string's to at most it, as NumPy pads a shorter one; a record holding a field of no fixed size, `|O`, is left unjudged. zarr-python 3 fails to open an array whose fill value is of another size.""" + from zarr_metadata.model import ZarrV2ArrayMetadata, validate_array_metadata_v2 + + document = { + **ZarrV2ArrayMetadata.create_default(shape=(4,), chunks=(2,)).to_json(), + "dtype": dtype, + "fill_value": fill_value, + } + found = validate_array_metadata_v2(document) + assert (len(found) == 0) is accepted, [p.message for p in found] + if not accepted: + assert [(p.loc, p.kind) for p in found] == [(("fill_value",), "invalid_value")] diff --git a/packages/zarr-metadata/tests/v3/test_scope.py b/packages/zarr-metadata/tests/v3/test_scope.py index d0b9b2eae2..c0a0872fca 100644 --- a/packages/zarr-metadata/tests/v3/test_scope.py +++ b/packages/zarr-metadata/tests/v3/test_scope.py @@ -100,8 +100,10 @@ def test_error_a_scope_conflict_says_each_disagreement() -> None: ) assert error.conflicts[0].key == (CodecDefinition, "bytes") assert str(error) == ( - "codec 'bytes' at ('codecs', 0): claimed CodecDefinition(name='bytes'), found None; " - "codec 'gzip': claimed None, found CodecDefinition(name='gzip')" + "codec 'bytes' at ('codecs', 0): claimed CodecDefinition(name='bytes') of " + "BytesCodecConfiguration, found no definition; " + "codec 'gzip': claimed no definition, found CodecDefinition(name='gzip') of " + "GzipCodecConfiguration" ) @@ -435,3 +437,46 @@ def test_a_scope_conflict_error_pickles_and_copies_with_its_conflicts() -> None: assert str(again) == str(error) assert str(again).startswith("codec 'gzip'") assert again.__notes__ == ["seen in a join"] + + +def test_a_conflict_names_the_field_as_the_document_writes_it_and_tells_the_definitions_apart() -> ( + None +): + """A conflict found in a document says the name the document writes, `r16`, not the name its definition is filed under, `r*`, and tells two definitions of one name apart by the configuration each declares, as the consolidated entries' messages do.""" + import dataclasses + + from zarr_metadata.model import ZarrV3ArrayMetadata + from zarr_metadata.v3.definition import CORE_AND_EXTENSIONS + + document = dict( + ZarrV3ArrayMetadata.create_default(shape=(2,), data_type="r16", fill_value=[0, 0]).to_json() + ) + model = ZarrV3ArrayMetadata(document, CORE_AND_EXTENSIONS) + other = dataclasses.replace(RAW_BYTES_DATA_TYPE, configuration=Empty) + with pytest.raises(ScopeConflictError) as raised: + model.refined_in(CORE_AND_EXTENSIONS.extended_with(other)) + message = str(raised.value) + assert message.startswith("data type 'r16'") + assert "RawBytesConfiguration" in message + assert "Empty" in message + with pytest.raises(ScopeConflictError) as joined: + Context.joined(Context.of(GZIP_CODEC), Context.of(OTHER_GZIP)) + assert str(joined.value).count("gzip") >= 2 + assert "Empty" in str(joined.value) + + +def test_error_a_scope_built_from_tables_keeps_the_invariants_of() -> None: + """`Context(tables)`, the constructor, refuses what `Context.of` refuses: a key that is no kind, and kinds of two formats, so no scope is built that `of` could not build.""" + from types import MappingProxyType + + from zarr_metadata.v2.data_type.scalar import UINT_V2 + from zarr_metadata.v2.definition import ZarrV2DataTypeDefinition + + with pytest.raises(TypeError, match="one Zarr format"): + Context( + MappingProxyType( + {CodecDefinition: {"gzip": GZIP_CODEC}, ZarrV2DataTypeDefinition: {"uint": UINT_V2}} + ) + ) + with pytest.raises(TypeError, match="kind"): + Context(MappingProxyType({int: {"gzip": GZIP_CODEC}})) # pyright: ignore[reportArgumentType]