From b2c50a87b888bdb0ad91ec2da9e02f8b4285efd7 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Sun, 27 Sep 2026 15:39:58 +0200 Subject: [PATCH 01/17] feat(zarr-metadata)!: a fill value is judged by the data type it fills Each data type's definition declares the JSON shape of its fill value and the rules for one of that shape, and the v3 array validators judge a document's fill value against the data type the scope read. A field that is read keeps the fields it read inside, by location, so a struct judges each field's fill value by that field's own type. Assisted-by: ClaudeCode:claude-opus-5-5 --- packages/zarr-metadata/README.md | 8 +- .../changes/+layer-4a.feature.md | 10 + packages/zarr-metadata/docs/index.md | 8 +- .../src/zarr_metadata/model/__init__.py | 17 +- .../src/zarr_metadata/model/_validation.py | 36 ++- .../src/zarr_metadata/v3/_definition.py | 120 ++++++++-- .../src/zarr_metadata/v3/data_type/_float.py | 73 ++++++ .../zarr_metadata/v3/data_type/_integer.py | 39 +++ .../zarr_metadata/v3/data_type/_numpy_time.py | 21 +- .../src/zarr_metadata/v3/data_type/bool.py | 2 +- .../src/zarr_metadata/v3/data_type/bytes.py | 25 +- .../zarr_metadata/v3/data_type/complex128.py | 10 +- .../zarr_metadata/v3/data_type/complex64.py | 10 +- .../src/zarr_metadata/v3/data_type/float16.py | 10 +- .../src/zarr_metadata/v3/data_type/float32.py | 10 +- .../src/zarr_metadata/v3/data_type/float64.py | 10 +- .../src/zarr_metadata/v3/data_type/int16.py | 8 +- .../src/zarr_metadata/v3/data_type/int32.py | 8 +- .../src/zarr_metadata/v3/data_type/int64.py | 8 +- .../src/zarr_metadata/v3/data_type/int8.py | 8 +- .../v3/data_type/numpy_datetime64.py | 7 +- .../v3/data_type/numpy_timedelta64.py | 7 +- .../src/zarr_metadata/v3/data_type/raw.py | 27 ++- .../src/zarr_metadata/v3/data_type/string.py | 2 +- .../src/zarr_metadata/v3/data_type/struct.py | 35 ++- .../src/zarr_metadata/v3/data_type/uint16.py | 8 +- .../src/zarr_metadata/v3/data_type/uint32.py | 8 +- .../src/zarr_metadata/v3/data_type/uint64.py | 8 +- .../src/zarr_metadata/v3/data_type/uint8.py | 8 +- .../src/zarr_metadata/v3/definition.py | 21 +- .../zarr-metadata/tests/model/test_array.py | 16 +- .../zarr-metadata/tests/test_public_api.py | 1 + .../tests/v3/test_definitions.py | 38 +++ .../tests/v3/test_fill_values.py | 225 ++++++++++++++++++ 34 files changed, 759 insertions(+), 93 deletions(-) create mode 100644 packages/zarr-metadata/changes/+layer-4a.feature.md create mode 100644 packages/zarr-metadata/src/zarr_metadata/v3/data_type/_float.py create mode 100644 packages/zarr-metadata/src/zarr_metadata/v3/data_type/_integer.py create mode 100644 packages/zarr-metadata/tests/v3/test_fill_values.py diff --git a/packages/zarr-metadata/README.md b/packages/zarr-metadata/README.md index 122d534a44..a669ea370b 100644 --- a/packages/zarr-metadata/README.md +++ b/packages/zarr-metadata/README.md @@ -63,10 +63,10 @@ codec and each storage transformer -- through the definition that claims its name in a scope, `CORE_AND_EXTENSIONS` unless a `context` is passed: a configuration its definition refuses is refused, and a key it does not declare is reported as `unknown_key`. A name nothing in the scope claims is -left unjudged, and whether to support it is the consumer's decision. The -validators do not judge fields against each other: a fill value against its -data type, a codec against the array it is handed, a chunk grid against the -shape. +left unjudged, and whether to support it is the consumer's decision. A v3 +fill value is judged against the data type it names, by that data type's +definition. The validators do not judge a codec against the array it is +handed, or a chunk grid against the shape. The Pydantic integration's generated JSON Schemas express independently checkable document structure and field constraints, but they are not a diff --git a/packages/zarr-metadata/changes/+layer-4a.feature.md b/packages/zarr-metadata/changes/+layer-4a.feature.md new file mode 100644 index 0000000000..d3eef59b49 --- /dev/null +++ b/packages/zarr-metadata/changes/+layer-4a.feature.md @@ -0,0 +1,10 @@ +**Breaking:** a v3 array's `fill_value` is judged against its `data_type`. +Each data type's definition declares the JSON shape of its fill value, +`fill_value`, and the rules for one of that shape, `fill_value_rules`: an +`int8` fill value of 300, a `float32` hex string of another width, and a +struct fill value missing a field are each a problem at `fill_value`, +where the package accepted them before. `fill_value_problems(data_type, +value)` judges a fill value against a data type field a scope read, and +a field that is read keeps the fields it read inside as +`Resolved.nested`, a `Nested` mapping by location. A data type nothing +in scope claims leaves its fill value unjudged. diff --git a/packages/zarr-metadata/docs/index.md b/packages/zarr-metadata/docs/index.md index 1e0346393f..497b43c11d 100644 --- a/packages/zarr-metadata/docs/index.md +++ b/packages/zarr-metadata/docs/index.md @@ -78,10 +78,10 @@ codec and each storage transformer -- through the definition that claims its name in a scope, `CORE_AND_EXTENSIONS` unless a `context` is passed: a configuration its definition refuses is refused, and a key it does not declare is reported as `unknown_key`. A name nothing in the scope claims is -left unjudged, and whether to support it is the consumer's decision. The -validators do not judge fields against each other: a fill value against its -data type, a codec against the array it is handed, a chunk grid against the -shape. +left unjudged, and whether to support it is the consumer's decision. A v3 +fill value is judged against the data type it names, by that data type's +definition. The validators do not judge a codec against the array it is +handed, or a chunk grid against the shape. ## Scope diff --git a/packages/zarr-metadata/src/zarr_metadata/model/__init__.py b/packages/zarr-metadata/src/zarr_metadata/model/__init__.py index 2c75d65b5c..2c2c6c849d 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/__init__.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/__init__.py @@ -4,14 +4,15 @@ representation of the JSON documents. Validators check a document's JSON structure and, in a v3 document, read each extension point (codecs, chunk grids, data types, ...) through the definition that claims its name in a -scope, `CORE_AND_EXTENSIONS` unless a `context` is passed; they do not judge -fields against each other. Each document concept gets a `validate_*` -function returning every problem found (a tuple of `ValidationProblem`, each -with a machine-readable `kind`), an `is_*` type guard, and a `parse_*` -function that narrows or raises `MetadataValidationError`. Model -`from_json` / `from_key_value` constructors raise `MetadataValidationError` -for every ingestion failure, including missing store keys and undecodable -bytes, and the v3 ones take the same `context`. +scope, `CORE_AND_EXTENSIONS` unless a `context` is passed, and judge the +fill value against the data type it names. Each document concept gets a +`validate_*` function returning every problem found (a tuple of +`ValidationProblem`, each with a machine-readable `kind`), an `is_*` type +guard, and a `parse_*` function that narrows or raises +`MetadataValidationError`. Model `from_json` / `from_key_value` constructors +raise `MetadataValidationError` for every ingestion failure, including +missing store keys and undecodable bytes, and the v3 ones take the same +`context`. """ from zarr_metadata._json import ( diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_validation.py b/packages/zarr-metadata/src/zarr_metadata/model/_validation.py index ff4f9a1f76..5c83f262c2 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_validation.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/_validation.py @@ -4,12 +4,12 @@ shapes, fixed literals like `zarr_format` -- and, in a v3 document, read each extension point through the definition that claims its name in a scope, so a configuration its definition refuses is refused here too. A -name nothing in the scope claims is left unjudged. Rules that read one -field against another -- a fill value against its data type, a codec -against the array it is handed, a grid against the shape -- are not -judged here. Each concept gets a `validate_*` function returning every -problem found, an `is_*` type guard, and a `parse_*` function that -narrows or raises `MetadataValidationError`. The guards are `TypeGuard`s, +name nothing in the scope claims is left unjudged. A v3 fill value is +judged against the data type it names; a codec against the array it is +handed, and a grid against the shape, are not judged here. Each concept +gets a `validate_*` function returning every problem found, an `is_*` +type guard, and a `parse_*` function that narrows or raises +`MetadataValidationError`. The guards are `TypeGuard`s, not `TypeIs`: True narrows a value to its document type, and False says nothing about its type, since a value can be well typed and still not a valid document. @@ -43,7 +43,9 @@ CodecDefinition, DataTypeDefinition, Definition, + Resolved, StorageTransformerDefinition, + fill_value_problems, resolve, ) from zarr_metadata.v3._registry import CORE_AND_EXTENSIONS, Context @@ -346,8 +348,10 @@ def validate_array_metadata_v3( Its structure, and each extension point read through the definition that claims its name in `context`: a gzip `level` out of range, a key a - codec's configuration does not declare. A name nothing in `context` - claims is left unjudged. Unknown top-level keys are allowed (they map + codec's configuration does not declare. The fill value is judged + against the data type as `context` read it: an `int8` fill value of 300. + A name nothing in `context` claims is left unjudged, with any fill value + of it. Unknown top-level keys are allowed (they map to `extra_fields`); a reader must understand each one that does not say `must_understand: false`, which the model reports as `must_understand_fields`. @@ -360,8 +364,6 @@ def validate_array_metadata_v3( problems.extend(_check_literal(doc, "zarr_format", 3)) problems.extend(_check_literal(doc, "node_type", "array")) problems.extend(_validate_dim_sequence(doc, "shape")) - if "fill_value" in doc: - problems.extend(_prefix("fill_value", validate_json(doc["fill_value"]))) # Each extension point is read by `resolve`, which judges its envelope # -- every extension *point* must be understood, so a `must_understand` # of `false` is refused at each: ignoring a codec gives wrong bytes as @@ -372,9 +374,21 @@ def validate_array_metadata_v3( # configuration, against the definition in `context` that claims its # name. `must_understand: false` keeps its meaning where it has one: an # unknown top-level extension *field*, which a reader really can skip. + read: dict[str, Resolved[Any]] = {} for key, kind in _EXTENSION_POINTS_V3: if key in doc: - problems.extend(resolve(doc[key], kind, context, (key,))[1]) + read[key], found = resolve(doc[key], kind, context, (key,)) + problems.extend(found) + # The fill value is JSON, and judged by the data type the scope read, + # when there is one: a data type nothing in scope claims leaves it + # unjudged. + if "fill_value" in doc: + if "data_type" in read: + problems.extend( + fill_value_problems(read["data_type"], doc["fill_value"], ("fill_value",)) + ) + else: + problems.extend(_prefix("fill_value", validate_json(doc["fill_value"]))) for key, kind in _EXTENSION_LISTS_V3: if key in doc: entries = doc[key] diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py b/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py index 4aa557cf84..f1cf15bb62 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py @@ -29,6 +29,7 @@ from __future__ import annotations +import dataclasses import functools import re from collections.abc import Callable, Iterable, Iterator, Mapping @@ -43,6 +44,7 @@ cast, get_args, get_origin, + get_type_hints, ) from typing_extensions import TypeAliasType, TypedDict, TypeVar, is_typeddict @@ -53,6 +55,7 @@ Loc, Parsed, Parser, + no_leaf, parser, problem, typeddict_keys, @@ -90,6 +93,13 @@ def unchanged(configuration: T) -> T: return configuration +def no_fill_value_rules( + configuration: object, nested: Nested, value: object +) -> Iterator[ValidationProblem]: + """The fill value rules of a data type with none: every fill value of its JSON shape is allowed.""" + yield from () + + class EmptyConfiguration(TypedDict, closed=True): """The configuration of a definition with nothing to configure, written as its bare name.""" @@ -181,14 +191,20 @@ def _malformed(definition: Definition[Any]) -> str | None: name = cast("object", definition.name) if not isinstance(name, str): return f"a definition's name is a string, got {name!r}" - functions: dict[str, object] = {"rules": definition.rules, "canonical": definition.canonical} - return next( - ( - f"{name!r}: {member} is a function, got {value!r}" - for member, value in functions.items() - if not callable(value) - ), - None, + for member in _function_members(type(definition)): + value = getattr(definition, member) + if not callable(value): + return f"{name!r}: {member} is a function, got {value!r}" + return None + + +@functools.cache +def _function_members(kind: type[Definition[Any]]) -> tuple[str, ...]: + """The members `kind` declares as functions: each one its annotation says is a `Callable`.""" + return tuple( + member + for member, annotation in get_type_hints(kind).items() + if get_origin(annotation) is Callable ) @@ -243,20 +259,43 @@ def _carrying_name( @dataclass(frozen=True, kw_only=True, slots=True) class DataTypeDefinition(Definition[C]): - """A data type. + """A data type, and the fill value an array of it takes. + + `fill_value` is the JSON shape of a fill value -- `Int8FillValue`, an + annotation the checker reads as it reads a configuration's members -- + and `fill_value_rules` is what the spec disallows in a fill value of + that shape: an integer out of range, a hex string of another width. + The rules are handed the configuration, the fields it holds as the + scope read them (a struct's field types), and the typed fill value. A + data type that says nothing of its fill value takes any JSON. One named as a document writes raw bits of one size -- `r16` -- is refused: that name reads as `r*`, so nothing would ever read it with this definition. """ + fill_value: object = JSONValue + """The JSON shape of a fill value, as an annotation: `Int8FillValue`.""" + fill_value_rules: Callable[[C, Nested, Any], Iterable[ValidationProblem]] = no_fill_value_rules + """What the spec disallows in a fill value of that shape, located in it.""" + def _refusal(self) -> str | None: - if RAW_BYTES_NAME_PATTERN.fullmatch(self.name) is None: - return None - return ( - f"{self.name!r} is how a document writes raw bits of one size, which read as " - f"{RAW_BYTES_NAME!r}; to read raw bits your own way, define {RAW_BYTES_NAME!r}" - ) + if RAW_BYTES_NAME_PATTERN.fullmatch(self.name) is not None: + return ( + f"{self.name!r} is how a document writes raw bits of one size, which read as " + f"{RAW_BYTES_NAME!r}; to read raw bits your own way, define {RAW_BYTES_NAME!r}" + ) + try: + _fill_value_parser(self.fill_value) + except TypeError as error: + return f"{self.name!r}: fill_value: {error}" + return None + + +@functools.cache +def _fill_value_parser(annotation: object) -> Parser: + """The checker for a fill value's JSON shape, compiled once; `TypeError` naming what no checker reads.""" + return parser(annotation, no_leaf) @dataclass(frozen=True, kw_only=True, slots=True) @@ -585,6 +624,11 @@ def named_configuration( """What a scope made of a field: read by the definition that claims it, or unread, and why.""" +def _nothing_nested() -> Nested: + """What a field that holds no field, or was not read, holds inside: nothing.""" + return {} + + @dataclass(frozen=True, slots=True) class Resolved(Generic[D]): """One metadata field, as read in a scope: its JSON, and what the scope made of it. @@ -603,6 +647,17 @@ class Resolved(Generic[D]): """The definition that claims the field's name; None when nothing in scope does, or it names none.""" configuration: Mapping[str, JSONValue] | None """The configuration, type-checked and allowed by the rules, when the field was read; None otherwise.""" + nested: Nested = dataclasses.field(default_factory=_nothing_nested) + """The fields the configuration holds, each as the scope read it, by where it sits in the configuration. + + A struct's field types at `("fields", 0, "data_type")`, a shard's + codecs at `("codecs", 0)`: what a definition's functions consult about + the fields inside its own. Empty unless the field was read. + """ + + +Nested: TypeAlias = Mapping[Loc, Resolved[Any]] +"""The fields a configuration holds, each as the scope read it, by where it sits in the configuration.""" def configuration_of(resolved: Resolved[Any], definition: Definition[C]) -> C | None: @@ -617,6 +672,34 @@ def configuration_of(resolved: Resolved[Any], definition: Definition[C]) -> C | return cast("C", resolved.configuration) +def fill_value_problems( + data_type: Resolved[DataTypeDefinition], value: object, loc: Loc = () +) -> Problems: + """What is wrong with `value` as a fill value of `data_type`, a data type field a scope read. + + `value` is refined to JSON first: not JSON is the first verdict, + whatever the data type. It is then checked against the JSON shape the + data type's definition declares, and judged by its fill value rules, + which see the fields its configuration holds as the scope read them: a + struct judges each field's fill value by that field's own type. A data + type the scope did not read, out of scope or invalid, leaves a JSON fill + value unjudged. `loc` prefixes every problem. + """ + refined, problems = refine_json(value, loc) + definition = data_type.definition + configuration = data_type.configuration + if len(problems) != 0 or definition is None or configuration is None: + return problems + typed, problems = _fill_value_parser(definition.fill_value)(refined, loc) + if len(problems) != 0: + return problems + return _ruled( + definition, + lambda: definition.fill_value_rules(configuration, data_type.nested, typed), + loc, + ) + + def resolve( data: object, kind: type[D], context: Context, loc: Loc = () ) -> tuple[Resolved[D], Problems]: @@ -678,14 +761,16 @@ def _read( configuration = cast("Mapping[str, JSONValue]", typed) if sound else None if configuration is not None: problems.extend(_ruled(definition, lambda: definition.rules(configuration), at)) + within: dict[Loc, Resolved[Any]] = {} for field, envelope in zip(nested, envelopes, strict=True): problems.extend(envelope) inner, found = _read(field.json, field.kind, context, field.loc) + within[field.loc[len(at) :]] = inner problems.extend(found) problems.extend(_sized(field, inner.definition)) if not _usable(problems): return Resolved(data, "invalid", definition, None), tuple(problems) - return Resolved(data, "read", definition, configuration), tuple(problems) + return Resolved(data, "read", definition, configuration, within), tuple(problems) def _read_carried( @@ -813,6 +898,7 @@ def _replaced(value: JSONValue, path: Loc, new: JSONValue) -> JSONValue: "DataTypeField", "Definition", "EmptyConfiguration", + "Nested", "Resolution", "Resolved", "StaticCodecField", @@ -822,8 +908,10 @@ def _replaced(value: JSONValue, path: Loc, new: JSONValue) -> JSONValue: "as_kind", "canonicalize", "configuration_of", + "fill_value_problems", "kind_of", "named_configuration", + "no_fill_value_rules", "no_rules", "resolve", "spelled", diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/_float.py b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/_float.py new file mode 100644 index 0000000000..eb61e2f46d --- /dev/null +++ b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/_float.py @@ -0,0 +1,73 @@ +"""The fill value rules floating-point numbers share, and complex numbers built of them. + +A float's fill value is a JSON number, which a reader rounds to the +nearest value the type represents; one of `"NaN"`, `"Infinity"` and +`"-Infinity"`; or `"0x"` and the hex digits of the value's bytes read as +an unsigned integer, as many as the type has bytes +(https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v3/data-types/index.rst#L63-L79). +A complex fill value is a pair of such components, real then imaginary +(https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v3/data-types/index.rst#L88-L91). +""" + +from collections.abc import Callable, Iterable, Iterator +from typing import Any, Literal, get_args + +from zarr_metadata._json import ValidationProblem +from zarr_metadata.v3._definition import EmptyConfiguration, Nested + +FloatSpecialFillValue = Literal["NaN", "Infinity", "-Infinity"] +"""The named non-finite fill values every IEEE 754 floating-point type takes.""" + + +def float_fill_value_rules( + name: str, hex_form: Callable[[str], str] +) -> Callable[[EmptyConfiguration, Nested, float | str], Iterator[ValidationProblem]]: + """The fill value rules of the floating-point type `name`, whose hex strings `hex_form` accepts. + + A number takes any value, since a reader rounds it. A string is one of + the named values, or a hex string of the type's own width: the + type-checked shape takes any string there, and `hex_form` raises + `ValueError` for one that is not. + """ + + def rules( + configuration: EmptyConfiguration, nested: Nested, value: float | str + ) -> Iterator[ValidationProblem]: + if not isinstance(value, str) or value in get_args(FloatSpecialFillValue): + return + try: + hex_form(value) + except ValueError: + yield ValidationProblem( + (), + f"expected a number, one of {get_args(FloatSpecialFillValue)!r}, or a {name} " + f"hex string, got {value!r}", + "invalid_value", + ) + + return rules + + +def complex_fill_value_rules( + component: Callable[[EmptyConfiguration, Nested, Any], Iterable[ValidationProblem]], +) -> Callable[ + [EmptyConfiguration, Nested, tuple[float | str, float | str]], Iterator[ValidationProblem] +]: + """The fill value rules of a complex type, whose components `component` judges, each at its index. + + `component` is the fill value rules of the component's own float type. + """ + + def rules( + configuration: EmptyConfiguration, + nested: Nested, + value: tuple[float | str, float | str], + ) -> Iterator[ValidationProblem]: + for index, part in enumerate(value): + for found in component(configuration, nested, part): + yield ValidationProblem((index, *found.loc), found.message, found.kind) + + return rules + + +__all__ = ["FloatSpecialFillValue", "complex_fill_value_rules", "float_fill_value_rules"] diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/_integer.py b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/_integer.py new file mode 100644 index 0000000000..ea6a1515e0 --- /dev/null +++ b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/_integer.py @@ -0,0 +1,39 @@ +"""The fill value rules integers share: a JSON integer within a range. + +The integer data types take one within their own range, and the byte +values of `r*` and `bytes` fill values are integers in `[0, 255]` +(https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v3/data-types/index.rst#L59-L61). +""" + +from collections.abc import Callable, Iterator + +from zarr_metadata._json import ValidationProblem +from zarr_metadata.v3._definition import EmptyConfiguration, Nested + + +def integer_fill_value_rules( + low: int, high: int +) -> Callable[[EmptyConfiguration, Nested, int], Iterator[ValidationProblem]]: + """The fill value rules of an integer data type whose range is `[low, high]`.""" + + def rules( + configuration: EmptyConfiguration, nested: Nested, value: int + ) -> Iterator[ValidationProblem]: + if not low <= value <= high: + yield ValidationProblem( + (), f"expected an integer in [{low}, {high}], got {value}", "invalid_value" + ) + + return rules + + +def byte_value_problems(values: tuple[int, ...]) -> Iterator[ValidationProblem]: + """Each of `values` that is not a byte, an integer in `[0, 255]`, located at its index.""" + for index, value in enumerate(values): + if not 0 <= value <= 255: + yield ValidationProblem( + (index,), f"expected an integer in [0, 255], got {value}", "invalid_value" + ) + + +__all__ = ["byte_value_problems", "integer_fill_value_rules"] diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/_numpy_time.py b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/_numpy_time.py index d2f7d02e53..d9d972284b 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/_numpy_time.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/_numpy_time.py @@ -11,6 +11,7 @@ from typing_extensions import ReadOnly, TypedDict from zarr_metadata._json import ValidationProblem +from zarr_metadata.v3._definition import Nested NUMPY_TIME_MAX_SCALE_FACTOR: Final = 2**31 - 1 """The largest `scale_factor` numpy stores: the field is a signed int32.""" @@ -34,4 +35,22 @@ def numpy_time_rules(configuration: NumpyTimeConfiguration) -> Iterator[Validati ) -__all__ = ["NUMPY_TIME_MAX_SCALE_FACTOR", "NumpyTimeConfiguration", "numpy_time_rules"] +def numpy_time_fill_value_rules( + configuration: NumpyTimeConfiguration, nested: Nested, value: int | str +) -> Iterator[ValidationProblem]: + """An integer fill value is a signed 64-bit one; `"NaT"` is the other form, which the shape admits. + + https://github.com/zarr-developers/zarr-extensions/blob/6a3adaeef244b3c76270dca52d6a849e88cf002c/data-types/numpy.datetime64/README.md?plain=1#L109-L112 + """ + if isinstance(value, int) and not -(2**63) <= value <= 2**63 - 1: + yield ValidationProblem( + (), f"expected a signed 64-bit integer or 'NaT', got {value}", "invalid_value" + ) + + +__all__ = [ + "NUMPY_TIME_MAX_SCALE_FACTOR", + "NumpyTimeConfiguration", + "numpy_time_fill_value_rules", + "numpy_time_rules", +] diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/bool.py b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/bool.py index 51787d9176..d42308c0b1 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/bool.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/bool.py @@ -19,7 +19,7 @@ BOOL_DATA_TYPE: Final = DataTypeDefinition( - name=BOOL_DATA_TYPE_NAME, configuration=EmptyConfiguration + name=BOOL_DATA_TYPE_NAME, configuration=EmptyConfiguration, fill_value=BoolFillValue ) """The `bool` data type: a bare name, with nothing to configure.""" diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/bytes.py b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/bytes.py index 9ff9258df1..06f1f2fbf2 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/bytes.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/bytes.py @@ -5,9 +5,12 @@ """ import re +from collections.abc import Iterator from typing import Final, Literal, NewType -from zarr_metadata.v3._definition import DataTypeDefinition, EmptyConfiguration +from zarr_metadata._json import ValidationProblem +from zarr_metadata.v3._definition import DataTypeDefinition, EmptyConfiguration, Nested +from zarr_metadata.v3.data_type._integer import byte_value_problems BYTES_DATA_TYPE_NAME: Final = "bytes" """The `data_type` value for the variable-length `bytes` type.""" @@ -41,8 +44,26 @@ def base64_bytes(value: str) -> Base64Bytes: """ +def _fill_value_rules( + configuration: EmptyConfiguration, nested: Nested, value: BytesFillValue +) -> Iterator[ValidationProblem]: + """Integers in `[0, 255]`, or a string of standard-alphabet base64.""" + if not isinstance(value, str): + yield from byte_value_problems(value) + return + try: + base64_bytes(value) + except ValueError: + yield ValidationProblem( + (), f"expected standard-alphabet base64, got {value!r}", "invalid_value" + ) + + BYTES_DATA_TYPE: Final = DataTypeDefinition( - name=BYTES_DATA_TYPE_NAME, configuration=EmptyConfiguration + name=BYTES_DATA_TYPE_NAME, + configuration=EmptyConfiguration, + fill_value=BytesFillValue, + fill_value_rules=_fill_value_rules, ) """The `bytes` data type: a bare name, with nothing to configure.""" diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/complex128.py b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/complex128.py index 2602b24e8d..d447fbf0bc 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/complex128.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/complex128.py @@ -7,7 +7,8 @@ from typing import Final, Literal from zarr_metadata.v3._definition import DataTypeDefinition, EmptyConfiguration -from zarr_metadata.v3.data_type.float64 import Float64FillValue +from zarr_metadata.v3.data_type._float import complex_fill_value_rules +from zarr_metadata.v3.data_type.float64 import FLOAT64_DATA_TYPE, Float64FillValue COMPLEX128_DATA_TYPE_NAME: Final = "complex128" """The `data_type` value for the `complex128` type.""" @@ -31,9 +32,12 @@ COMPLEX128_DATA_TYPE: Final = DataTypeDefinition( - name=COMPLEX128_DATA_TYPE_NAME, configuration=EmptyConfiguration + name=COMPLEX128_DATA_TYPE_NAME, + configuration=EmptyConfiguration, + fill_value=Complex128FillValue, + fill_value_rules=complex_fill_value_rules(FLOAT64_DATA_TYPE.fill_value_rules), ) -"""The `complex128` data type: a bare name, with nothing to configure.""" +"""The `complex128` data type: a bare name, with nothing to configure; its fill value a pair of `float64` components.""" __all__ = [ diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/complex64.py b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/complex64.py index 1254652eac..7967c4bdca 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/complex64.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/complex64.py @@ -7,7 +7,8 @@ from typing import Final, Literal from zarr_metadata.v3._definition import DataTypeDefinition, EmptyConfiguration -from zarr_metadata.v3.data_type.float32 import Float32FillValue +from zarr_metadata.v3.data_type._float import complex_fill_value_rules +from zarr_metadata.v3.data_type.float32 import FLOAT32_DATA_TYPE, Float32FillValue COMPLEX64_DATA_TYPE_NAME: Final = "complex64" """The `data_type` value for the `complex64` type.""" @@ -31,9 +32,12 @@ COMPLEX64_DATA_TYPE: Final = DataTypeDefinition( - name=COMPLEX64_DATA_TYPE_NAME, configuration=EmptyConfiguration + name=COMPLEX64_DATA_TYPE_NAME, + configuration=EmptyConfiguration, + fill_value=Complex64FillValue, + fill_value_rules=complex_fill_value_rules(FLOAT32_DATA_TYPE.fill_value_rules), ) -"""The `complex64` data type: a bare name, with nothing to configure.""" +"""The `complex64` data type: a bare name, with nothing to configure; its fill value a pair of `float32` components.""" __all__ = [ diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/float16.py b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/float16.py index 0a3843701a..4940969078 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/float16.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/float16.py @@ -8,6 +8,7 @@ from typing import Final, Literal, NewType from zarr_metadata.v3._definition import DataTypeDefinition, EmptyConfiguration +from zarr_metadata.v3.data_type._float import FloatSpecialFillValue, float_fill_value_rules FLOAT16_DATA_TYPE_NAME: Final = "float16" """The `data_type` value for the `float16` type.""" @@ -15,7 +16,7 @@ Float16DataTypeName = Literal["float16"] """Literal type of the `data_type` field for `float16`.""" -Float16SpecialFillValue = Literal["NaN", "Infinity", "-Infinity"] +Float16SpecialFillValue = FloatSpecialFillValue """Named non-finite fill values permitted by the spec for IEEE 754 floats. https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v3/data-types/index.rst#L63-L79 @@ -64,9 +65,12 @@ def hex_float16(value: str) -> HexFloat16: FLOAT16_DATA_TYPE: Final = DataTypeDefinition( - name=FLOAT16_DATA_TYPE_NAME, configuration=EmptyConfiguration + name=FLOAT16_DATA_TYPE_NAME, + configuration=EmptyConfiguration, + fill_value=Float16FillValue, + fill_value_rules=float_fill_value_rules("float16", hex_float16), ) -"""The `float16` data type: a bare name, with nothing to configure.""" +"""The `float16` data type: a bare name, with nothing to configure; its fill value a number, a named value or a hex string.""" __all__ = [ diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/float32.py b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/float32.py index 7fed625efb..b7e77b5475 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/float32.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/float32.py @@ -8,6 +8,7 @@ from typing import Final, Literal, NewType from zarr_metadata.v3._definition import DataTypeDefinition, EmptyConfiguration +from zarr_metadata.v3.data_type._float import FloatSpecialFillValue, float_fill_value_rules FLOAT32_DATA_TYPE_NAME: Final = "float32" """The `data_type` value for the `float32` type.""" @@ -15,7 +16,7 @@ Float32DataTypeName = Literal["float32"] """Literal type of the `data_type` field for `float32`.""" -Float32SpecialFillValue = Literal["NaN", "Infinity", "-Infinity"] +Float32SpecialFillValue = FloatSpecialFillValue """Named non-finite fill values permitted by the spec for IEEE 754 floats. https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v3/data-types/index.rst#L63-L79 @@ -64,9 +65,12 @@ def hex_float32(value: str) -> HexFloat32: FLOAT32_DATA_TYPE: Final = DataTypeDefinition( - name=FLOAT32_DATA_TYPE_NAME, configuration=EmptyConfiguration + name=FLOAT32_DATA_TYPE_NAME, + configuration=EmptyConfiguration, + fill_value=Float32FillValue, + fill_value_rules=float_fill_value_rules("float32", hex_float32), ) -"""The `float32` data type: a bare name, with nothing to configure.""" +"""The `float32` data type: a bare name, with nothing to configure; its fill value a number, a named value or a hex string.""" __all__ = [ diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/float64.py b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/float64.py index 72e596bd45..7f8deb3903 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/float64.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/float64.py @@ -8,6 +8,7 @@ from typing import Final, Literal, NewType from zarr_metadata.v3._definition import DataTypeDefinition, EmptyConfiguration +from zarr_metadata.v3.data_type._float import FloatSpecialFillValue, float_fill_value_rules FLOAT64_DATA_TYPE_NAME: Final = "float64" """The `data_type` value for the `float64` type.""" @@ -15,7 +16,7 @@ Float64DataTypeName = Literal["float64"] """Literal type of the `data_type` field for `float64`.""" -Float64SpecialFillValue = Literal["NaN", "Infinity", "-Infinity"] +Float64SpecialFillValue = FloatSpecialFillValue """Named non-finite fill values permitted by the spec for IEEE 754 floats. https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v3/data-types/index.rst#L63-L79 @@ -65,9 +66,12 @@ def hex_float64(value: str) -> HexFloat64: FLOAT64_DATA_TYPE: Final = DataTypeDefinition( - name=FLOAT64_DATA_TYPE_NAME, configuration=EmptyConfiguration + name=FLOAT64_DATA_TYPE_NAME, + configuration=EmptyConfiguration, + fill_value=Float64FillValue, + fill_value_rules=float_fill_value_rules("float64", hex_float64), ) -"""The `float64` data type: a bare name, with nothing to configure.""" +"""The `float64` data type: a bare name, with nothing to configure; its fill value a number, a named value or a hex string.""" __all__ = [ diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/int16.py b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/int16.py index 9129519594..a37c5dd1b0 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/int16.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/int16.py @@ -7,6 +7,7 @@ from typing import Final, Literal from zarr_metadata.v3._definition import DataTypeDefinition, EmptyConfiguration +from zarr_metadata.v3.data_type._integer import integer_fill_value_rules INT16_DATA_TYPE_NAME: Final = "int16" """The `data_type` value for the `int16` type.""" @@ -19,9 +20,12 @@ INT16_DATA_TYPE: Final = DataTypeDefinition( - name=INT16_DATA_TYPE_NAME, configuration=EmptyConfiguration + name=INT16_DATA_TYPE_NAME, + configuration=EmptyConfiguration, + fill_value=Int16FillValue, + fill_value_rules=integer_fill_value_rules(-(2**15), 2**15 - 1), ) -"""The `int16` data type: a bare name, with nothing to configure.""" +"""The `int16` data type: a bare name, with nothing to configure; its fill value an integer in its range.""" __all__ = [ diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/int32.py b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/int32.py index 85fdb9f1e3..ac25007462 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/int32.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/int32.py @@ -7,6 +7,7 @@ from typing import Final, Literal from zarr_metadata.v3._definition import DataTypeDefinition, EmptyConfiguration +from zarr_metadata.v3.data_type._integer import integer_fill_value_rules INT32_DATA_TYPE_NAME: Final = "int32" """The `data_type` value for the `int32` type.""" @@ -19,9 +20,12 @@ INT32_DATA_TYPE: Final = DataTypeDefinition( - name=INT32_DATA_TYPE_NAME, configuration=EmptyConfiguration + name=INT32_DATA_TYPE_NAME, + configuration=EmptyConfiguration, + fill_value=Int32FillValue, + fill_value_rules=integer_fill_value_rules(-(2**31), 2**31 - 1), ) -"""The `int32` data type: a bare name, with nothing to configure.""" +"""The `int32` data type: a bare name, with nothing to configure; its fill value an integer in its range.""" __all__ = [ diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/int64.py b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/int64.py index bb2439b862..f57ed8b8a9 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/int64.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/int64.py @@ -7,6 +7,7 @@ from typing import Final, Literal from zarr_metadata.v3._definition import DataTypeDefinition, EmptyConfiguration +from zarr_metadata.v3.data_type._integer import integer_fill_value_rules INT64_DATA_TYPE_NAME: Final = "int64" """The `data_type` value for the `int64` type.""" @@ -19,9 +20,12 @@ INT64_DATA_TYPE: Final = DataTypeDefinition( - name=INT64_DATA_TYPE_NAME, configuration=EmptyConfiguration + name=INT64_DATA_TYPE_NAME, + configuration=EmptyConfiguration, + fill_value=Int64FillValue, + fill_value_rules=integer_fill_value_rules(-(2**63), 2**63 - 1), ) -"""The `int64` data type: a bare name, with nothing to configure.""" +"""The `int64` data type: a bare name, with nothing to configure; its fill value an integer in its range.""" __all__ = [ diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/int8.py b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/int8.py index e1d2517ede..e170ee22fa 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/int8.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/int8.py @@ -7,6 +7,7 @@ from typing import Final, Literal from zarr_metadata.v3._definition import DataTypeDefinition, EmptyConfiguration +from zarr_metadata.v3.data_type._integer import integer_fill_value_rules INT8_DATA_TYPE_NAME: Final = "int8" """The `data_type` value for the `int8` type.""" @@ -19,9 +20,12 @@ INT8_DATA_TYPE: Final = DataTypeDefinition( - name=INT8_DATA_TYPE_NAME, configuration=EmptyConfiguration + name=INT8_DATA_TYPE_NAME, + configuration=EmptyConfiguration, + fill_value=Int8FillValue, + fill_value_rules=integer_fill_value_rules(-(2**7), 2**7 - 1), ) -"""The `int8` data type: a bare name, with nothing to configure.""" +"""The `int8` data type: a bare name, with nothing to configure; its fill value an integer in its range.""" __all__ = [ diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/numpy_datetime64.py b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/numpy_datetime64.py index 7f5cd9bc8a..543050c513 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/numpy_datetime64.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/numpy_datetime64.py @@ -9,7 +9,10 @@ from typing_extensions import ReadOnly, TypedDict from zarr_metadata.v3._definition import DataTypeDefinition -from zarr_metadata.v3.data_type._numpy_time import numpy_time_rules +from zarr_metadata.v3.data_type._numpy_time import ( + numpy_time_fill_value_rules, + numpy_time_rules, +) NUMPY_DATETIME64_DATA_TYPE_NAME: Final = "numpy.datetime64" """The `name` field value of the `numpy.datetime64` data type.""" @@ -58,6 +61,8 @@ class NumpyDatetime64(TypedDict, closed=True): name=NUMPY_DATETIME64_DATA_TYPE_NAME, configuration=NumpyDatetime64Configuration, rules=numpy_time_rules, + fill_value=NumpyDatetime64FillValue, + fill_value_rules=numpy_time_fill_value_rules, ) """The `numpy.datetime64` data type.""" diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/numpy_timedelta64.py b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/numpy_timedelta64.py index db6f9deed8..2a3ac45635 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/numpy_timedelta64.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/numpy_timedelta64.py @@ -9,7 +9,10 @@ from typing_extensions import ReadOnly, TypedDict from zarr_metadata.v3._definition import DataTypeDefinition -from zarr_metadata.v3.data_type._numpy_time import numpy_time_rules +from zarr_metadata.v3.data_type._numpy_time import ( + numpy_time_fill_value_rules, + numpy_time_rules, +) NUMPY_TIMEDELTA64_DATA_TYPE_NAME: Final = "numpy.timedelta64" """The `name` field value of the `numpy.timedelta64` data type.""" @@ -77,6 +80,8 @@ class NumpyTimedelta64(TypedDict, closed=True): name=NUMPY_TIMEDELTA64_DATA_TYPE_NAME, configuration=NumpyTimedelta64Configuration, rules=numpy_time_rules, + fill_value=NumpyTimedelta64FillValue, + fill_value_rules=numpy_time_fill_value_rules, ) """The `numpy.timedelta64` data type.""" diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/raw.py b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/raw.py index 98be8a2212..1cbca53607 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/raw.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/raw.py @@ -14,7 +14,13 @@ from typing_extensions import TypedDict from zarr_metadata._json import ValidationProblem -from zarr_metadata.v3._definition import RAW_BYTES_NAME, RAW_BYTES_NAME_PATTERN, DataTypeDefinition +from zarr_metadata.v3._definition import ( + RAW_BYTES_NAME, + RAW_BYTES_NAME_PATTERN, + DataTypeDefinition, + Nested, +) +from zarr_metadata.v3.data_type._integer import byte_value_problems RawBytesDataTypeName = NewType("RawBytesDataTypeName", str) """A spec-conformant `r` raw-bytes name (e.g. `"r8"`, `"r16"`). @@ -68,10 +74,29 @@ def _rules(configuration: RawBytesConfiguration) -> Iterator[ValidationProblem]: ) +def _fill_value_rules( + configuration: RawBytesConfiguration, nested: Nested, value: RawBytesFillValue +) -> Iterator[ValidationProblem]: + """One byte value, an integer in `[0, 255]`, for each 8 of the size. + + The spec's text says `N` values for `r`, but `N` counts bits + (https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v3/core/index.rst#L897-L900), + and one value is one byte. + """ + expected = configuration["bits"] // 8 + if len(value) != expected: + yield ValidationProblem( + (), f"expected {expected} byte values, got {len(value)}", "invalid_value" + ) + yield from byte_value_problems(value) + + RAW_BYTES_DATA_TYPE: Final = DataTypeDefinition( name=RAW_BYTES_NAME, configuration=RawBytesConfiguration, rules=_rules, + fill_value=RawBytesFillValue, + fill_value_rules=_fill_value_rules, ) """Raw bits, `r*`: one data type, whose name carries its size. diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/string.py b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/string.py index 43008e6645..0aa246bf15 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/string.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/string.py @@ -19,7 +19,7 @@ STRING_DATA_TYPE: Final = DataTypeDefinition( - name=STRING_DATA_TYPE_NAME, configuration=EmptyConfiguration + name=STRING_DATA_TYPE_NAME, configuration=EmptyConfiguration, fill_value=StringFillValue ) """The `string` data type: a bare name, with nothing to configure.""" diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/struct.py b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/struct.py index f6c9a6bbd6..8676fbd04c 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/struct.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/struct.py @@ -11,7 +11,12 @@ from zarr_metadata._common import JSONValue from zarr_metadata._json import ValidationProblem -from zarr_metadata.v3._definition import DataTypeDefinition, DataTypeField +from zarr_metadata.v3._definition import ( + DataTypeDefinition, + DataTypeField, + Nested, + fill_value_problems, +) STRUCT_DATA_TYPE_NAME: Final = "struct" """The `name` field value of the `struct` data type.""" @@ -85,8 +90,34 @@ def _rules(configuration: StructConfiguration) -> Iterator[ValidationProblem]: ) +def _fill_value_rules( + configuration: StructConfiguration, nested: Nested, value: StructFillValue +) -> Iterator[ValidationProblem]: + """A fill value for every field, each one judged by that field's own type, and for nothing else. + + Every field needs one, valid for its type + (https://github.com/zarr-developers/zarr-extensions/blob/6a3adaeef244b3c76270dca52d6a849e88cf002c/data-types/struct/README.md?plain=1#L221-L224). + A field type the scope did not read leaves its fill value unjudged. + """ + names = [member["name"] for member in configuration["fields"]] + for index, name in enumerate(names): + if name not in value: + yield ValidationProblem( + (name,), f"expected a fill value for struct field {name!r}", "missing_key" + ) + continue + yield from fill_value_problems(nested[("fields", index, "data_type")], value[name], (name,)) + for key in value: + if key not in names: + yield ValidationProblem((key,), f"no struct field is named {key!r}", "unknown_key") + + STRUCT_DATA_TYPE: Final = DataTypeDefinition( - name=STRUCT_DATA_TYPE_NAME, configuration=StructConfiguration, rules=_rules + name=STRUCT_DATA_TYPE_NAME, + configuration=StructConfiguration, + rules=_rules, + fill_value=StructFillValue, + fill_value_rules=_fill_value_rules, ) """The `struct` data type: a record of named fields, each field's type a nested field.""" diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/uint16.py b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/uint16.py index 0cccdab830..c0c3e865b6 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/uint16.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/uint16.py @@ -7,6 +7,7 @@ from typing import Final, Literal from zarr_metadata.v3._definition import DataTypeDefinition, EmptyConfiguration +from zarr_metadata.v3.data_type._integer import integer_fill_value_rules UINT16_DATA_TYPE_NAME: Final = "uint16" """The `data_type` value for the `uint16` type.""" @@ -19,9 +20,12 @@ UINT16_DATA_TYPE: Final = DataTypeDefinition( - name=UINT16_DATA_TYPE_NAME, configuration=EmptyConfiguration + name=UINT16_DATA_TYPE_NAME, + configuration=EmptyConfiguration, + fill_value=Uint16FillValue, + fill_value_rules=integer_fill_value_rules(0, 2**16 - 1), ) -"""The `uint16` data type: a bare name, with nothing to configure.""" +"""The `uint16` data type: a bare name, with nothing to configure; its fill value an integer in its range.""" __all__ = [ diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/uint32.py b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/uint32.py index db784d9b31..9bf43429a2 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/uint32.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/uint32.py @@ -7,6 +7,7 @@ from typing import Final, Literal from zarr_metadata.v3._definition import DataTypeDefinition, EmptyConfiguration +from zarr_metadata.v3.data_type._integer import integer_fill_value_rules UINT32_DATA_TYPE_NAME: Final = "uint32" """The `data_type` value for the `uint32` type.""" @@ -19,9 +20,12 @@ UINT32_DATA_TYPE: Final = DataTypeDefinition( - name=UINT32_DATA_TYPE_NAME, configuration=EmptyConfiguration + name=UINT32_DATA_TYPE_NAME, + configuration=EmptyConfiguration, + fill_value=Uint32FillValue, + fill_value_rules=integer_fill_value_rules(0, 2**32 - 1), ) -"""The `uint32` data type: a bare name, with nothing to configure.""" +"""The `uint32` data type: a bare name, with nothing to configure; its fill value an integer in its range.""" __all__ = [ diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/uint64.py b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/uint64.py index 9639352d63..768413de85 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/uint64.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/uint64.py @@ -7,6 +7,7 @@ from typing import Final, Literal from zarr_metadata.v3._definition import DataTypeDefinition, EmptyConfiguration +from zarr_metadata.v3.data_type._integer import integer_fill_value_rules UINT64_DATA_TYPE_NAME: Final = "uint64" """The `data_type` value for the `uint64` type.""" @@ -19,9 +20,12 @@ UINT64_DATA_TYPE: Final = DataTypeDefinition( - name=UINT64_DATA_TYPE_NAME, configuration=EmptyConfiguration + name=UINT64_DATA_TYPE_NAME, + configuration=EmptyConfiguration, + fill_value=Uint64FillValue, + fill_value_rules=integer_fill_value_rules(0, 2**64 - 1), ) -"""The `uint64` data type: a bare name, with nothing to configure.""" +"""The `uint64` data type: a bare name, with nothing to configure; its fill value an integer in its range.""" __all__ = [ diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/uint8.py b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/uint8.py index 6124d29b4a..5279b14523 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/uint8.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/uint8.py @@ -7,6 +7,7 @@ from typing import Final, Literal from zarr_metadata.v3._definition import DataTypeDefinition, EmptyConfiguration +from zarr_metadata.v3.data_type._integer import integer_fill_value_rules UINT8_DATA_TYPE_NAME: Final = "uint8" """The `data_type` value for the `uint8` type.""" @@ -19,9 +20,12 @@ UINT8_DATA_TYPE: Final = DataTypeDefinition( - name=UINT8_DATA_TYPE_NAME, configuration=EmptyConfiguration + name=UINT8_DATA_TYPE_NAME, + configuration=EmptyConfiguration, + fill_value=Uint8FillValue, + fill_value_rules=integer_fill_value_rules(0, 2**8 - 1), ) -"""The `uint8` data type: a bare name, with nothing to configure.""" +"""The `uint8` data type: a bare name, with nothing to configure; its fill value an integer in its range.""" __all__ = [ diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/definition.py b/packages/zarr-metadata/src/zarr_metadata/v3/definition.py index 420926d983..07a84e05b7 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/definition.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/definition.py @@ -110,8 +110,9 @@ def acme_lz4_rules(configuration: AcmeLz4Configuration) -> Iterator[ValidationPr `validate_array_metadata_v3(document, context=SCOPE)`, from `zarr_metadata.model`, reads each extension point of a v3 array document through the definitions in `SCOPE`, and so do the model's `from_json` and -`from_key_value`. A fill value is not judged against its data type there: -no rule here reads one field against another. +`from_key_value`, and a fill value is judged against the data type it +names, by that data type's definition. No rule here reads a codec against +the array it is handed, or a chunk grid against the shape. The TypedDict says what a key it does not declare is: with `closed=True`, a problem, as above; with `extra_items=`, a key holding @@ -134,6 +135,18 @@ def acme_lz4_rules(configuration: AcmeLz4Configuration) -> Iterator[ValidationPr member typed with it. An extension with nothing to configure takes `EmptyConfiguration`, and is written as its bare name. +A data type also says what its fill value is: `fill_value`, the JSON +shape of one as an annotation the checker reads -- `Int8FillValue` -- and +`fill_value_rules`, a function yielding what the spec disallows in a fill +value of that shape: an integer out of range, a hex string of another +width. The rules are handed the configuration, the fields it holds as the +scope read them, and the typed fill value; a field that is read keeps what +it read inside it as `Resolved.nested`, a `Nested` mapping by location, so +a struct judges each field's fill value by that field's own type. +`fill_value_problems(data_type, value)` judges a fill value against a data +type field the scope read; one nothing in scope claims leaves it unjudged. +A data type that says nothing of its fill value takes any JSON. + Raw bits are the one data type whose name carries its configuration: a document writes `r` and the size in bits, and `r16` reads as `r*`, as the specification's table writes raw bits, with `{"bits": 16}`. A reader that @@ -180,6 +193,7 @@ class creation. DataTypeField, Definition, EmptyConfiguration, + Nested, Resolution, Resolved, StaticCodecField, @@ -188,6 +202,7 @@ class creation. Unread, canonicalize, configuration_of, + fill_value_problems, resolve, ) from zarr_metadata.v3._registry import CORE, CORE_AND_EXTENSIONS, Context @@ -211,6 +226,7 @@ class creation. "JSONValue", "Loc", "MetadataValidationError", + "Nested", "ProblemKind", "Resolution", "Resolved", @@ -223,5 +239,6 @@ class creation. "canonicalize", "check", "configuration_of", + "fill_value_problems", "resolve", ] diff --git a/packages/zarr-metadata/tests/model/test_array.py b/packages/zarr-metadata/tests/model/test_array.py index 9f4097969a..a07bfa1c76 100644 --- a/packages/zarr-metadata/tests/model/test_array.py +++ b/packages/zarr-metadata/tests/model/test_array.py @@ -182,14 +182,15 @@ def test_json_value_type_accepts_json_shapes() -> None: def test_string_nan_fill_value_roundtrips() -> None: - # Non-finite floats are represented as the spec strings ("NaN", "Infinity", - # "-Infinity") by the caller — the metadata layer does not interpret dtypes. + # A float's non-finite fill values are the spec strings ("NaN", + # "Infinity", "-Infinity"): # https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v3/data-types/index.rst#L63-L79 # The string form round-trips cleanly under default dataclass equality, - # unlike a raw float('nan') (which is an invalid fill_value the caller must - # not pass). - """A string 'NaN' fill_value round-trips cleanly (non-finite floats are the caller's responsibility).""" - m = ZarrV3ArrayMetadata.create_default(fill_value="NaN") + # unlike a raw float('nan'), which is not JSON. + """A float array's string 'NaN' fill_value round-trips cleanly.""" + m = ZarrV3ArrayMetadata.create_default( + fill_value="NaN", data_type=ZarrV3NamedConfig(name="float32", configuration={}) + ) assert ZarrV3ArrayMetadata.from_json(m.to_json()) == m assert ZarrV3ArrayMetadata.from_json(m.to_json()).fill_value == "NaN" @@ -1797,7 +1798,8 @@ def test_array_parsers_normalize_json_lists_before_narrowing() -> None: def test_array_guards_reject_noncanonical_nested_json() -> None: """Document guards cannot narrow values that only parsers can materialize.""" - v3 = dict(ZarrV3ArrayMetadata.create_default().to_json()) + # Raw bits of 16, whose fill value is two byte values. + v3 = dict(ZarrV3ArrayMetadata.create_default().to_json()) | {"data_type": "r16"} v3["fill_value"] = range(2) v2 = dict(ZarrV2ArrayMetadata.create_default().to_json()) v2["fill_value"] = range(2) diff --git a/packages/zarr-metadata/tests/test_public_api.py b/packages/zarr-metadata/tests/test_public_api.py index 1e6e10bdff..7ba9777d78 100644 --- a/packages/zarr-metadata/tests/test_public_api.py +++ b/packages/zarr-metadata/tests/test_public_api.py @@ -290,6 +290,7 @@ def test_all_is_grouped_and_unique() -> None: "Context", "Definition", "Loc", + "Nested", "Resolution", "Resolved", "Unread", diff --git a/packages/zarr-metadata/tests/v3/test_definitions.py b/packages/zarr-metadata/tests/v3/test_definitions.py index 46abd2882f..bd56034d83 100644 --- a/packages/zarr-metadata/tests/v3/test_definitions.py +++ b/packages/zarr-metadata/tests/v3/test_definitions.py @@ -18,6 +18,7 @@ from zarr_metadata.v3.chunk_grid.regular import REGULAR_CHUNK_GRID from zarr_metadata.v3.codec.crc32c import Empty from zarr_metadata.v3.codec.gzip import GZIP_CODEC, GzipCodecConfiguration +from zarr_metadata.v3.data_type.int8 import INT8_DATA_TYPE from zarr_metadata.v3.definition import ( CORE, CORE_AND_EXTENSIONS, @@ -183,6 +184,31 @@ def test_core_is_a_subset_of_core_and_extensions() -> None: assert set(CORE.definitions()) <= set(CORE_AND_EXTENSIONS.definitions()) +def test_a_read_field_keeps_the_fields_it_read_inside() -> None: + # By where each sits in the configuration, as the scope read it. + shard = { + "name": "sharding_indexed", + "configuration": { + "chunk_shape": [2], + "codecs": ["bytes"], + "index_codecs": ["bytes", "crc32c"], + }, + } + resolved, _ = resolve(shard, CodecDefinition, CORE_AND_EXTENSIONS) + assert {loc: inner.json for loc, inner in resolved.nested.items()} == { + ("codecs", 0): "bytes", + ("index_codecs", 0): "bytes", + ("index_codecs", 1): "crc32c", + } + cast = {"name": "cast_value", "configuration": {"data_type": "int8"}} + resolved, _ = resolve(cast, CodecDefinition, CORE_AND_EXTENSIONS) + assert resolved.nested[("data_type",)].definition is INT8_DATA_TYPE + # A field holding none, or one that was not read, has nothing inside. + assert resolve("int8", DataTypeDefinition, CORE)[0].nested == {} + unread = {"name": "cast_value", "configuration": {"data_type": "int8", "rounding": 1}} + assert resolve(unread, CodecDefinition, CORE_AND_EXTENSIONS)[0].nested == {} + + @pytest.mark.parametrize( ("field", "kind", "resolution", "configuration", "problems"), [ @@ -679,6 +705,18 @@ def test_error_a_data_type_is_named_as_raw_bits_of_one_size_are_written() -> Non DataTypeDefinition(name="r16", configuration=Empty) +def test_error_a_data_type_fill_value_no_checker_reads() -> None: + with pytest.raises(TypeError, match="'acme.set': fill_value: "): + DataTypeDefinition(name="acme.set", configuration=Empty, fill_value=set[int]) + + +@pytest.mark.parametrize("member", ["rules", "canonical", "fill_value_rules"]) +def test_error_a_function_member_that_is_not_a_function(member: str) -> None: + # Each member a definition's annotations declare a `Callable`. + with pytest.raises(TypeError, match=f"'acme.t': {member} is a function, got 'none'"): + DataTypeDefinition(name="acme.t", configuration=Empty, **{member: "none"}) # pyright: ignore[reportArgumentType] + + @pytest.mark.parametrize("kind", [Definition, AcmeCodecDefinition]) def test_error_a_field_is_read_as_a_kind_of_metadata(kind: type[Definition[Any]]) -> None: # Nothing is filed under either, so the field would go unjudged. diff --git a/packages/zarr-metadata/tests/v3/test_fill_values.py b/packages/zarr-metadata/tests/v3/test_fill_values.py new file mode 100644 index 0000000000..e9746c932f --- /dev/null +++ b/packages/zarr-metadata/tests/v3/test_fill_values.py @@ -0,0 +1,225 @@ +"""Fill values, judged by the data type they fill. + +Each data type's definition declares the JSON shape of its fill value and +the rules for one of that shape; `fill_value_problems` judges a fill value +against a data type field a scope read, and the v3 array validators judge +a document's `fill_value` against its `data_type`. +""" + +from __future__ import annotations + +import math + +import pytest + +from zarr_metadata.model import validate_array_metadata_v3, validate_group_metadata_v3 +from zarr_metadata.model._array import ZarrV3ArrayMetadata +from zarr_metadata.v3.definition import ( + CORE_AND_EXTENSIONS, + DataTypeDefinition, + JSONValue, + fill_value_problems, + resolve, +) + +DATETIME: JSONValue = { + "name": "numpy.datetime64", + "configuration": {"unit": "s", "scale_factor": 1}, +} +TIMEDELTA: JSONValue = { + "name": "numpy.timedelta64", + "configuration": {"unit": "ms", "scale_factor": 1}, +} +STRUCT: JSONValue = { + "name": "struct", + "configuration": { + "fields": [ + {"name": "a", "data_type": "int8"}, + {"name": "b", "data_type": "float32"}, + ] + }, +} + + +def _problems(data_type: JSONValue, value: object) -> list[tuple[tuple[str | int, ...], str]]: + resolved, found = resolve(data_type, DataTypeDefinition, CORE_AND_EXTENSIONS) + assert found == () + return [(problem.loc, problem.kind) for problem in fill_value_problems(resolved, value)] + + +@pytest.mark.parametrize( + ("data_type", "fill_value"), + [ + ("bool", False), + ("int8", -128), + ("int8", 127), + ("uint8", 255), + ("int64", -(2**63)), + ("uint64", 2**64 - 1), + ("float16", 1.5), + # A number takes any value: a reader rounds it to the type. + ("float16", 1e10), + ("float32", 0), + ("float32", "NaN"), + ("float32", "-Infinity"), + ("float16", "0x7e00"), + ("float32", "0x7fc00000"), + ("float64", "0x7ff8000000000000"), + ("complex64", [1.5, "NaN"]), + ("complex128", ["0x7ff8000000000000", 0]), + ("r8", [255]), + ("r16", [0, 1]), + ("r008", [7]), + ("bytes", [0, 255]), + ("bytes", "AQI="), + ("bytes", ""), + ("string", "a"), + (DATETIME, "NaT"), + (DATETIME, -(2**63)), + (TIMEDELTA, 2**63 - 1), + (STRUCT, {"a": 1, "b": "NaN"}), + # A field type nothing in scope claims leaves its fill value unjudged. + ( + { + "name": "struct", + "configuration": {"fields": [{"name": "a", "data_type": "acme.decimal"}]}, + }, + {"a": "anything"}, + ), + # So does a data type nothing in scope claims, or one that is not read. + ("acme.decimal", "anything"), + ( + {"name": "numpy.datetime64", "configuration": {"unit": "s", "scale_factor": 0}}, + "anything", + ), + ], +) +def test_every_data_type_takes_its_fill_values(data_type: JSONValue, fill_value: object) -> None: + resolved, _ = resolve(data_type, DataTypeDefinition, CORE_AND_EXTENSIONS) + assert fill_value_problems(resolved, fill_value) == () + + +@pytest.mark.parametrize( + ("data_type", "fill_value"), + [ + ("bool", 0), + ("int8", True), + ("int8", 1.0), + ("int8", "5"), + ("float32", None), + ("string", 1), + ("complex64", [1]), + ("r16", "AQI="), + (DATETIME, 1.5), + (STRUCT, [1, 2]), + ], +) +def test_error_a_fill_value_of_the_wrong_json_type( + data_type: JSONValue, fill_value: object +) -> None: + assert _problems(data_type, fill_value) == [((), "invalid_type")] + + +def test_error_a_fill_value_that_is_not_json() -> None: + # A float's NaN is the string "NaN"; the number is not JSON. + assert _problems("float32", math.nan) == [((), "invalid_value")] + + +@pytest.mark.parametrize( + ("data_type", "fill_value"), + [ + ("int8", 128), + ("int8", -129), + ("uint8", -1), + ("uint64", 2**64), + ("int64", -(2**63) - 1), + ], +) +def test_error_an_integer_fill_value_out_of_its_type_s_range( + data_type: str, fill_value: int +) -> None: + assert _problems(data_type, fill_value) == [((), "invalid_value")] + + +@pytest.mark.parametrize( + ("data_type", "fill_value", "loc"), + [ + # Not one of the named values: the spec spells them this way only. + ("float32", "nan", ()), + # A hex string of another type's width. + ("float32", "0x7e00", ()), + ("float16", "0x7fc00000", ()), + ("complex64", [1, "0x7ff8000000000000"], (1,)), + ], +) +def test_error_a_float_string_that_is_not_a_named_value_or_a_hex_string_of_its_width( + data_type: str, fill_value: object, loc: tuple[int, ...] +) -> None: + assert _problems(data_type, fill_value) == [(loc, "invalid_value")] + + +@pytest.mark.parametrize(("data_type", "fill_value"), [("r16", [1]), ("r8", [1, 2])]) +def test_error_a_raw_bits_fill_value_of_another_size(data_type: str, fill_value: object) -> None: + # One byte value for each 8 bits of the size. + assert _problems(data_type, fill_value) == [((), "invalid_value")] + + +@pytest.mark.parametrize( + ("data_type", "fill_value", "loc"), [("r16", [1, 256], (1,)), ("bytes", [-1], (0,))] +) +def test_error_a_byte_value_out_of_range( + data_type: str, fill_value: object, loc: tuple[int, ...] +) -> None: + assert _problems(data_type, fill_value) == [(loc, "invalid_value")] + + +@pytest.mark.parametrize("fill_value", ["!!", "AQI"]) +def test_error_a_bytes_fill_value_that_is_not_base64(fill_value: str) -> None: + assert _problems("bytes", fill_value) == [((), "invalid_value")] + + +@pytest.mark.parametrize( + ("data_type", "fill_value"), [(DATETIME, 2**63), (TIMEDELTA, -(2**63) - 1)] +) +def test_error_a_time_fill_value_outside_a_signed_64_bit_integer( + data_type: JSONValue, fill_value: int +) -> None: + assert _problems(data_type, fill_value) == [((), "invalid_value")] + + +def test_error_a_struct_fill_value_missing_a_field() -> None: + assert _problems(STRUCT, {"a": 1}) == [(("b",), "missing_key")] + + +def test_error_a_struct_fill_value_for_no_field() -> None: + assert _problems(STRUCT, {"a": 1, "b": 0, "c": 2}) == [(("c",), "unknown_key")] + + +def test_error_a_struct_field_s_fill_value_its_own_type_refuses() -> None: + # Judged by the field's type, as the scope read it, at the field's name. + assert _problems(STRUCT, {"a": 300, "b": "x"}) == [ + (("a",), "invalid_value"), + (("b",), "invalid_value"), + ] + + +def test_error_an_array_document_s_fill_value_its_data_type_refuses() -> None: + # Judged against the data type the document names, and in a + # consolidated group, where the array sits. + document = dict(ZarrV3ArrayMetadata.create_default().to_json()) | {"fill_value": 300} + assert [(p.loc, p.kind) for p in validate_array_metadata_v3(document)] == [ + (("fill_value",), "invalid_value") + ] + group = { + "zarr_format": 3, + "node_type": "group", + "attributes": {}, + "consolidated_metadata": { + "kind": "inline", + "must_understand": False, + "metadata": {"a": document}, + }, + } + assert [(p.loc, p.kind) for p in validate_group_metadata_v3(group)] == [ + (("consolidated_metadata", "metadata", "a", "fill_value"), "invalid_value") + ] From 948236e4992811d1f9433ab79162ee936c604ea9 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Sun, 27 Sep 2026 15:56:40 +0200 Subject: [PATCH 02/17] refactor(zarr-metadata): canonicalize spells nested fields from what resolve kept The simplest spelling of a field that read spells each field it holds from its reading in `Resolved.nested`, rather than reading every subtree again at each level. A nested field of a field without problems has none, so the branch for one without a spelling could not be taken. Assisted-by: ClaudeCode:claude-opus-5-5 --- .../src/zarr_metadata/v3/_definition.py | 22 ++++++++++--------- 1 file changed, 12 insertions(+), 10 deletions(-) diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py b/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py index f1cf15bb62..f5bec2e820 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py @@ -839,21 +839,23 @@ def canonicalize( return None, problems if resolved.definition is None: return resolved.json, () - return _canonical_field(resolved.definition, resolved, context), () + return _canonical_field(resolved.definition, resolved), () -def _canonical_field( - definition: Definition[Any], resolved: Resolved[Any], context: Context -) -> JSONValue: - """A field that read, in its simplest equivalent spelling: nested fields first, then its own members.""" +def _canonical_field(definition: Definition[Any], resolved: Resolved[Any]) -> JSONValue: + """A field that read, in its simplest equivalent spelling: the fields it holds first, then its own members. + + Each field it holds is spelled from what `resolve` read of it, kept in + `nested`; one nothing in scope claims keeps the spelling it was written + in. + """ name, _, _ = named_configuration(resolved.json) configuration: JSONValue = dict(resolved.configuration or {}) - _, _, nested = _checked(definition.configuration, configuration, ()) - for field in nested: - simplest, _ = canonicalize(field.json, field.kind, context) - configuration = _replaced( - configuration, field.loc, field.json if simplest is None else simplest + for loc, inner in resolved.nested.items(): + simplest = ( + inner.json if inner.definition is None else _canonical_field(inner.definition, inner) ) + configuration = _replaced(configuration, loc, simplest) simplified = cast("Mapping[str, JSONValue]", definition.canonical(configuration)) _, refused = definition.judge(simplified) if len(refused) != 0: From b310c3635902be5208cf35681f5e859013c0eb5a Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Sun, 27 Sep 2026 16:06:31 +0200 Subject: [PATCH 03/17] fix(zarr-metadata): fill values read deep, definitions pickle, rules see past unknown keys - The JSON check walks one frame per level of nesting, as `refine_json` does, so a fill value checked against a JSON-typed shape reads as deep as `refine_json` reads: a struct fill value 600 levels deep raised `RecursionError`. - The integer, float and complex fill value rules are partial applications of module-level functions, so a scope's definitions pickle again. - A key a fill value's shape does not declare is reported and left out, and the fill value rules still judge the rest, as `judge` does for a configuration. - A struct whose reading holds no reading of a field's type leaves that field's fill value unjudged. - `fill_value_problems` takes the reading of any data type, whatever its configuration's type. - `create_default` says that an overridden `data_type` needs its own `fill_value`. Assisted-by: ClaudeCode:claude-opus-5-5 --- .../zarr-metadata/src/zarr_metadata/_json.py | 17 ++--- .../src/zarr_metadata/model/_array.py | 5 +- .../src/zarr_metadata/v3/_definition.py | 13 ++-- .../src/zarr_metadata/v3/data_type/_float.py | 61 ++++++++++-------- .../zarr_metadata/v3/data_type/_integer.py | 23 ++++--- .../src/zarr_metadata/v3/data_type/struct.py | 7 ++- .../tests/v3/test_every_definition.py | 13 ++++ .../tests/v3/test_fill_values.py | 63 +++++++++++++++++++ 8 files changed, 151 insertions(+), 51 deletions(-) diff --git a/packages/zarr-metadata/src/zarr_metadata/_json.py b/packages/zarr-metadata/src/zarr_metadata/_json.py index b4dcad614f..a5c0a9b6af 100644 --- a/packages/zarr-metadata/src/zarr_metadata/_json.py +++ b/packages/zarr-metadata/src/zarr_metadata/_json.py @@ -200,20 +200,23 @@ def is_canonical_json(value: object, *, finite: bool = True) -> TypeGuard[JSONVa A non-finite number counts only when `finite` is false, as a document's guard passes it: where one may be is the document's validator's to say. + One frame per level of nesting, as `refine_json` takes, so a value + `refine_json` reads is one this can walk. """ if isinstance(value, float): return not finite or math.isfinite(value) if isinstance(value, (str, int, bool)) or value is None: return True if isinstance(value, (list, tuple)): - sequence = cast("list[object] | tuple[object, ...]", value) - return all(is_canonical_json(item, finite=finite) for item in sequence) + for item in cast("list[object] | tuple[object, ...]", value): + if not is_canonical_json(item, finite=finite): + return False + return True if isinstance(value, dict): - mapping = cast("dict[object, object]", value) - return all( - isinstance(key, str) and is_canonical_json(item, finite=finite) - for key, item in mapping.items() - ) + for key, item in cast("dict[object, object]", value).items(): + if not isinstance(key, str) or not is_canonical_json(item, finite=finite): + return False + return True return False diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_array.py b/packages/zarr-metadata/src/zarr_metadata/model/_array.py index cdf5a684c4..e7c585c684 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_array.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/_array.py @@ -199,7 +199,10 @@ def create_default(cls, **overrides: Unpack[ZarrV3ArrayMetadataPartial]) -> Zarr layer never does (and cannot do for unrecognized grid names). So overriding `chunk_grid` without `shape` keeps the scalar default `shape=()`, and consistency between the two is the caller's - responsibility. + responsibility. So is a fill value for an overridden `data_type`: + the default `fill_value` is `0`, which a data type whose fill value + is not an integer -- `bool`, `string`, a complex or struct type -- + refuses, so pass the two together. """ if "shape" in overrides and "chunk_grid" not in overrides: overrides["chunk_grid"] = ZarrV3NamedConfig( diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py b/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py index f5bec2e820..088a141a37 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py @@ -673,14 +673,16 @@ def configuration_of(resolved: Resolved[Any], definition: Definition[C]) -> C | def fill_value_problems( - data_type: Resolved[DataTypeDefinition], value: object, loc: Loc = () + data_type: Resolved[DataTypeDefinition[Any]], value: object, loc: Loc = () ) -> Problems: """What is wrong with `value` as a fill value of `data_type`, a data type field a scope read. `value` is refined to JSON first: not JSON is the first verdict, whatever the data type. It is then checked against the JSON shape the - data type's definition declares, and judged by its fill value rules, - which see the fields its configuration holds as the scope read them: a + data type's definition declares, and judged by its fill value rules, as + `judge` judges a configuration: a key the shape does not declare is + reported and left out, and the rules still judge the rest. The rules + see the fields the configuration holds as the scope read them: a struct judges each field's fill value by that field's own type. A data type the scope did not read, out of scope or invalid, leaves a JSON fill value unjudged. `loc` prefixes every problem. @@ -691,13 +693,14 @@ def fill_value_problems( if len(problems) != 0 or definition is None or configuration is None: return problems typed, problems = _fill_value_parser(definition.fill_value)(refined, loc) - if len(problems) != 0: + if not _usable(problems): return problems - return _ruled( + refused = _ruled( definition, lambda: definition.fill_value_rules(configuration, data_type.nested, typed), loc, ) + return (*problems, *refused) def resolve( diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/_float.py b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/_float.py index eb61e2f46d..3b81bd7a7c 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/_float.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/_float.py @@ -9,6 +9,7 @@ (https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v3/data-types/index.rst#L88-L91). """ +import functools from collections.abc import Callable, Iterable, Iterator from typing import Any, Literal, get_args @@ -27,25 +28,30 @@ def float_fill_value_rules( A number takes any value, since a reader rounds it. A string is one of the named values, or a hex string of the type's own width: the type-checked shape takes any string there, and `hex_form` raises - `ValueError` for one that is not. + `ValueError` for one that is not. A partial application of a + module-level function, so a definition holding it pickles. """ - - def rules( - configuration: EmptyConfiguration, nested: Nested, value: float | str - ) -> Iterator[ValidationProblem]: - if not isinstance(value, str) or value in get_args(FloatSpecialFillValue): - return - try: - hex_form(value) - except ValueError: - yield ValidationProblem( - (), - f"expected a number, one of {get_args(FloatSpecialFillValue)!r}, or a {name} " - f"hex string, got {value!r}", - "invalid_value", - ) - - return rules + return functools.partial(_float_fill_value, name, hex_form) + + +def _float_fill_value( + name: str, + hex_form: Callable[[str], str], + configuration: EmptyConfiguration, + nested: Nested, + value: float | str, +) -> Iterator[ValidationProblem]: + if not isinstance(value, str) or value in get_args(FloatSpecialFillValue): + return + try: + hex_form(value) + except ValueError: + yield ValidationProblem( + (), + f"expected a number, one of {get_args(FloatSpecialFillValue)!r}, or a {name} " + f"hex string, got {value!r}", + "invalid_value", + ) def complex_fill_value_rules( @@ -57,17 +63,18 @@ def complex_fill_value_rules( `component` is the fill value rules of the component's own float type. """ + return functools.partial(_complex_fill_value, component) - def rules( - configuration: EmptyConfiguration, - nested: Nested, - value: tuple[float | str, float | str], - ) -> Iterator[ValidationProblem]: - for index, part in enumerate(value): - for found in component(configuration, nested, part): - yield ValidationProblem((index, *found.loc), found.message, found.kind) - return rules +def _complex_fill_value( + component: Callable[[EmptyConfiguration, Nested, Any], Iterable[ValidationProblem]], + configuration: EmptyConfiguration, + nested: Nested, + value: tuple[float | str, float | str], +) -> Iterator[ValidationProblem]: + for index, part in enumerate(value): + for found in component(configuration, nested, part): + yield ValidationProblem((index, *found.loc), found.message, found.kind) __all__ = ["FloatSpecialFillValue", "complex_fill_value_rules", "float_fill_value_rules"] diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/_integer.py b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/_integer.py index ea6a1515e0..4bf1005fd5 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/_integer.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/_integer.py @@ -5,6 +5,7 @@ (https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v3/data-types/index.rst#L59-L61). """ +import functools from collections.abc import Callable, Iterator from zarr_metadata._json import ValidationProblem @@ -14,17 +15,21 @@ def integer_fill_value_rules( low: int, high: int ) -> Callable[[EmptyConfiguration, Nested, int], Iterator[ValidationProblem]]: - """The fill value rules of an integer data type whose range is `[low, high]`.""" + """The fill value rules of an integer data type whose range is `[low, high]`. + + A partial application of a module-level function, so a definition + holding it pickles. + """ + return functools.partial(_in_range, low, high) - def rules( - configuration: EmptyConfiguration, nested: Nested, value: int - ) -> Iterator[ValidationProblem]: - if not low <= value <= high: - yield ValidationProblem( - (), f"expected an integer in [{low}, {high}], got {value}", "invalid_value" - ) - return rules +def _in_range( + low: int, high: int, configuration: EmptyConfiguration, nested: Nested, value: int +) -> Iterator[ValidationProblem]: + if not low <= value <= high: + yield ValidationProblem( + (), f"expected an integer in [{low}, {high}], got {value}", "invalid_value" + ) def byte_value_problems(values: tuple[int, ...]) -> Iterator[ValidationProblem]: diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/struct.py b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/struct.py index 8676fbd04c..9e52d59a96 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/struct.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/struct.py @@ -97,7 +97,8 @@ def _fill_value_rules( Every field needs one, valid for its type (https://github.com/zarr-developers/zarr-extensions/blob/6a3adaeef244b3c76270dca52d6a849e88cf002c/data-types/struct/README.md?plain=1#L221-L224). - A field type the scope did not read leaves its fill value unjudged. + A field type the scope did not read, or whose reading the struct's does + not hold, leaves its fill value unjudged. """ names = [member["name"] for member in configuration["fields"]] for index, name in enumerate(names): @@ -106,7 +107,9 @@ def _fill_value_rules( (name,), f"expected a fill value for struct field {name!r}", "missing_key" ) continue - yield from fill_value_problems(nested[("fields", index, "data_type")], value[name], (name,)) + field_type = nested.get(("fields", index, "data_type")) + if field_type is not None: + yield from fill_value_problems(field_type, value[name], (name,)) for key in value: if key not in names: yield ValidationProblem((key,), f"no struct field is named {key!r}", "unknown_key") diff --git a/packages/zarr-metadata/tests/v3/test_every_definition.py b/packages/zarr-metadata/tests/v3/test_every_definition.py index d85512ca16..3ecd25c473 100644 --- a/packages/zarr-metadata/tests/v3/test_every_definition.py +++ b/packages/zarr-metadata/tests/v3/test_every_definition.py @@ -7,6 +7,7 @@ from __future__ import annotations import dataclasses +import pickle from typing import Any import pytest @@ -19,11 +20,13 @@ ChunkGridDefinition, ChunkKeyEncodingDefinition, CodecDefinition, + Context, DataTypeDefinition, Definition, ValidationProblem, canonicalize, configuration_of, + fill_value_problems, resolve, ) @@ -145,6 +148,16 @@ def _problems(key: str, field: object) -> list[tuple[tuple[str | int, ...], str] return _read(key, field)[1] +def test_a_scope_s_definitions_pickle() -> None: + # So a scope can be sent to another process and filed again there. + definitions = pickle.loads(pickle.dumps(CORE_AND_EXTENSIONS.definitions())) + scope = Context.of(*definitions) + resolved, _ = resolve("complex64", DataTypeDefinition, scope) + assert [(p.loc, p.kind) for p in fill_value_problems(resolved, [1, "x"])] == [ + ((1,), "invalid_value") + ] + + def test_every_definition_in_scope_has_an_example() -> None: names = {definition.name for definition in CORE_AND_EXTENSIONS.definitions()} assert names == {key.split(":")[1] for key in EXAMPLES} diff --git a/packages/zarr-metadata/tests/v3/test_fill_values.py b/packages/zarr-metadata/tests/v3/test_fill_values.py index e9746c932f..3b379ff31b 100644 --- a/packages/zarr-metadata/tests/v3/test_fill_values.py +++ b/packages/zarr-metadata/tests/v3/test_fill_values.py @@ -9,19 +9,29 @@ from __future__ import annotations import math +from typing import TYPE_CHECKING import pytest +from typing_extensions import TypedDict from zarr_metadata.model import validate_array_metadata_v3, validate_group_metadata_v3 from zarr_metadata.model._array import ZarrV3ArrayMetadata +from zarr_metadata.v3.data_type.struct import STRUCT_DATA_TYPE from zarr_metadata.v3.definition import ( CORE_AND_EXTENSIONS, DataTypeDefinition, + EmptyConfiguration, JSONValue, + Nested, + Resolved, + ValidationProblem, fill_value_problems, resolve, ) +if TYPE_CHECKING: + from collections.abc import Iterator + DATETIME: JSONValue = { "name": "numpy.datetime64", "configuration": {"unit": "s", "scale_factor": 1}, @@ -41,6 +51,25 @@ } +class AcmePointFillValue(TypedDict, closed=True): + x: int + + +def acme_point_fill_value_rules( + configuration: EmptyConfiguration, nested: Nested, value: AcmePointFillValue +) -> Iterator[ValidationProblem]: + if value["x"] > 10: + yield ValidationProblem(("x",), "expected x <= 10", "invalid_value") + + +ACME_POINT = DataTypeDefinition( + name="acme.point", + configuration=EmptyConfiguration, + fill_value=AcmePointFillValue, + fill_value_rules=acme_point_fill_value_rules, +) + + def _problems(data_type: JSONValue, value: object) -> list[tuple[tuple[str | int, ...], str]]: resolved, found = resolve(data_type, DataTypeDefinition, CORE_AND_EXTENSIONS) assert found == () @@ -223,3 +252,37 @@ def test_error_an_array_document_s_fill_value_its_data_type_refuses() -> None: assert [(p.loc, p.kind) for p in validate_group_metadata_v3(group)] == [ (("consolidated_metadata", "metadata", "a", "fill_value"), "invalid_value") ] + + +def test_error_a_key_the_fill_value_shape_does_not_declare_hides_no_rule() -> None: + # Reported, and left out of what the rules see, which still judge the rest. + scope = CORE_AND_EXTENSIONS.extended_with(ACME_POINT) + resolved, _ = resolve("acme.point", DataTypeDefinition, scope) + found = fill_value_problems(resolved, {"x": 99, "extra": 1}) + assert [(problem.loc, problem.kind) for problem in found] == [ + (("extra",), "unknown_key"), + (("x",), "invalid_value"), + ] + + +def test_a_fill_value_nested_hundreds_deep_is_read() -> None: + def deep(levels: int) -> dict[str, object]: + value: dict[str, object] = {} + for _ in range(levels): + value = {"x": value} + return value + + document = dict(ZarrV3ArrayMetadata.create_default().to_json()) | { + "data_type": STRUCT, + "fill_value": {"a": 1, "b": 0.5, "c": deep(600)}, + } + assert [(p.loc, p.kind) for p in validate_array_metadata_v3(document)] == [ + (("fill_value", "c"), "unknown_key") + ] + + +def test_a_struct_read_without_its_field_types_leaves_its_fields_unjudged() -> None: + # A reading built by hand, holding no field type's reading. + configuration = {"fields": ({"name": "a", "data_type": "int8"},)} + struct = Resolved(STRUCT, "read", STRUCT_DATA_TYPE, configuration) + assert fill_value_problems(struct, {"a": 300}) == () From 4c24bb436953ff1aa291fb5832b443aacbe43d8f Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Sun, 27 Sep 2026 16:07:21 +0200 Subject: [PATCH 04/17] docs(zarr-metadata): the fill value fragment takes this PR's number Assisted-by: ClaudeCode:claude-opus-5-5 --- .../changes/{+layer-4a.feature.md => 4440.feature.md} | 0 1 file changed, 0 insertions(+), 0 deletions(-) rename packages/zarr-metadata/changes/{+layer-4a.feature.md => 4440.feature.md} (100%) diff --git a/packages/zarr-metadata/changes/+layer-4a.feature.md b/packages/zarr-metadata/changes/4440.feature.md similarity index 100% rename from packages/zarr-metadata/changes/+layer-4a.feature.md rename to packages/zarr-metadata/changes/4440.feature.md From bc1036cd42e38ae22a75a00f09dbbf5beb58c6d3 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Sun, 27 Sep 2026 16:42:45 +0200 Subject: [PATCH 05/17] feat(zarr-metadata)!: a chunk grid is judged against the shape it chunks A chunk grid's definition says which arrays it fits, and the v3 array validators judge a document's grid against its shape once both are read: a regular grid has a chunk length for each dimension, and 0 only for a dimension of length 0; a rectilinear grid has chunk lengths for each dimension that cover it. Assisted-by: ClaudeCode:claude-opus-5-5 --- packages/zarr-metadata/README.md | 3 +- .../changes/+layer-4b.feature.md | 7 ++ packages/zarr-metadata/docs/index.md | 3 +- .../src/zarr_metadata/model/__init__.py | 3 +- .../src/zarr_metadata/model/_array.py | 5 +- .../src/zarr_metadata/model/_validation.py | 53 ++++---- .../src/zarr_metadata/v3/_definition.py | 50 ++++++-- .../v3/chunk_grid/rectilinear.py | 34 ++++- .../zarr_metadata/v3/chunk_grid/regular.py | 40 +++++- .../zarr_metadata/v3/data_type/_numpy_time.py | 2 +- .../src/zarr_metadata/v3/data_type/struct.py | 2 +- .../src/zarr_metadata/v3/definition.py | 24 ++-- .../zarr-metadata/tests/model/test_array.py | 3 +- .../tests/v3/test_definitions.py | 16 ++- .../tests/v3/test_grid_shapes.py | 117 ++++++++++++++++++ 15 files changed, 300 insertions(+), 62 deletions(-) create mode 100644 packages/zarr-metadata/changes/+layer-4b.feature.md create mode 100644 packages/zarr-metadata/tests/v3/test_grid_shapes.py diff --git a/packages/zarr-metadata/README.md b/packages/zarr-metadata/README.md index a669ea370b..a2e7690a61 100644 --- a/packages/zarr-metadata/README.md +++ b/packages/zarr-metadata/README.md @@ -65,8 +65,9 @@ configuration its definition refuses is refused, and a key it does not declare is reported as `unknown_key`. A name nothing in the scope claims is left unjudged, and whether to support it is the consumer's decision. A v3 fill value is judged against the data type it names, by that data type's +definition, and the chunk grid against the shape, by the grid's definition. The validators do not judge a codec against the array it is -handed, or a chunk grid against the shape. +handed. The Pydantic integration's generated JSON Schemas express independently checkable document structure and field constraints, but they are not a diff --git a/packages/zarr-metadata/changes/+layer-4b.feature.md b/packages/zarr-metadata/changes/+layer-4b.feature.md new file mode 100644 index 0000000000..b899c84f09 --- /dev/null +++ b/packages/zarr-metadata/changes/+layer-4b.feature.md @@ -0,0 +1,7 @@ +**Breaking:** a v3 array's `chunk_grid` is judged against its `shape`, by +the grid's definition. A regular grid whose `chunk_shape` does not have +one length per dimension of the shape, or has a length of 0 for a +dimension that is not empty, and a rectilinear grid whose `chunk_shapes` +does not have one entry per dimension, or whose chunk lengths fall short +of their dimension, each have a problem in `chunk_grid.configuration`, +where the package accepted them before. diff --git a/packages/zarr-metadata/docs/index.md b/packages/zarr-metadata/docs/index.md index 497b43c11d..a79f2bf729 100644 --- a/packages/zarr-metadata/docs/index.md +++ b/packages/zarr-metadata/docs/index.md @@ -80,8 +80,9 @@ configuration its definition refuses is refused, and a key it does not declare is reported as `unknown_key`. A name nothing in the scope claims is left unjudged, and whether to support it is the consumer's decision. A v3 fill value is judged against the data type it names, by that data type's +definition, and the chunk grid against the shape, by the grid's definition. The validators do not judge a codec against the array it is -handed, or a chunk grid against the shape. +handed. ## Scope diff --git a/packages/zarr-metadata/src/zarr_metadata/model/__init__.py b/packages/zarr-metadata/src/zarr_metadata/model/__init__.py index 2c2c6c849d..18eebef5ba 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/__init__.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/__init__.py @@ -5,7 +5,8 @@ structure and, in a v3 document, read each extension point (codecs, chunk grids, data types, ...) through the definition that claims its name in a scope, `CORE_AND_EXTENSIONS` unless a `context` is passed, and judge the -fill value against the data type it names. Each document concept gets a +fill value against the data type it names and the chunk grid against +the shape. Each document concept gets a `validate_*` function returning every problem found (a tuple of `ValidationProblem`, each with a machine-readable `kind`), an `is_*` type guard, and a `parse_*` function that narrows or raises diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_array.py b/packages/zarr-metadata/src/zarr_metadata/model/_array.py index e7c585c684..6161d34eb2 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_array.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/_array.py @@ -198,8 +198,9 @@ def create_default(cls, **overrides: Unpack[ZarrV3ArrayMetadataPartial]) -> Zarr it would require interpreting the grid's configuration, which this layer never does (and cannot do for unrecognized grid names). So overriding `chunk_grid` without `shape` keeps the scalar default - `shape=()`, and consistency between the two is the caller's - responsibility. So is a fill value for an overridden `data_type`: + `shape=()`, which a grid of another rank does not fit: consistency + between the two is the caller's responsibility, so pass them + together. So is a fill value for an overridden `data_type`: the default `fill_value` is `0`, which a data type whose fill value is not an integer -- `bool`, `string`, a complex or struct type -- refuses, so pass the two together. diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_validation.py b/packages/zarr-metadata/src/zarr_metadata/model/_validation.py index 5c83f262c2..65ff07c1dc 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_validation.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/_validation.py @@ -5,8 +5,8 @@ each extension point through the definition that claims its name in a scope, so a configuration its definition refuses is refused here too. A name nothing in the scope claims is left unjudged. A v3 fill value is -judged against the data type it names; a codec against the array it is -handed, and a grid against the shape, are not judged here. Each concept +judged against the data type it names, and the chunk grid against the +shape; a codec against the array it is handed is not judged here. Each concept gets a `validate_*` function returning every problem found, an `is_*` type guard, and a `parse_*` function that narrows or raises `MetadataValidationError`. The guards are `TypeGuard`s, @@ -45,6 +45,7 @@ Definition, Resolved, StorageTransformerDefinition, + chunk_grid_problems, fill_value_problems, resolve, ) @@ -183,19 +184,22 @@ def _is_int_sequence(value: object) -> TypeGuard[Sequence[int]]: ) -def _validate_dim_sequence(doc: Mapping[object, object], key: str) -> tuple[ValidationProblem, ...]: - """Validate a dimension sequence (`shape` / `chunks`) if present in `doc`. +def _dimension_lengths( + doc: Mapping[object, object], key: str +) -> tuple[tuple[int, ...] | None, tuple[ValidationProblem, ...]]: + """The dimension lengths `doc` holds at `key` (`shape`, `chunks`), and every problem with them. - Dimension lengths are non-negative integers. + Dimension lengths are non-negative integers; the lengths are None when + `doc` holds none at `key`, or ones with a problem. """ if key not in doc: - return () + return None, () value = doc[key] if not _is_int_sequence(value): - return (ValidationProblem((key,), "expected a sequence of int", "invalid_type"),) + return None, (ValidationProblem((key,), "expected a sequence of int", "invalid_type"),) if any(item < 0 for item in value): - return (ValidationProblem((key,), "expected non-negative integers", "invalid_value"),) - return () + return None, (ValidationProblem((key,), "expected non-negative integers", "invalid_value"),) + return tuple(value), () def _is_dtype_v2(value: object) -> bool: @@ -349,9 +353,11 @@ def validate_array_metadata_v3( Its structure, and each extension point read through the definition that claims its name in `context`: a gzip `level` out of range, a key a codec's configuration does not declare. The fill value is judged - against the data type as `context` read it: an `int8` fill value of 300. - A name nothing in `context` claims is left unjudged, with any fill value - of it. Unknown top-level keys are allowed (they map + against the data type as `context` read it -- an `int8` fill value of + 300 -- and the chunk grid against the shape: a regular grid with a + chunk length for each of two dimensions, over an array of three. A name + nothing in `context` claims is left unjudged, with any fill value of + it. Unknown top-level keys are allowed (they map to `extra_fields`); a reader must understand each one that does not say `must_understand: false`, which the model reports as `must_understand_fields`. @@ -363,7 +369,8 @@ def validate_array_metadata_v3( problems.extend(_validate_other_members(doc, ARRAY_METADATA_STANDARD_KEYS_V3)) problems.extend(_check_literal(doc, "zarr_format", 3)) problems.extend(_check_literal(doc, "node_type", "array")) - problems.extend(_validate_dim_sequence(doc, "shape")) + shape, shape_problems = _dimension_lengths(doc, "shape") + problems.extend(shape_problems) # Each extension point is read by `resolve`, which judges its envelope # -- every extension *point* must be understood, so a `must_understand` # of `false` is refused at each: ignoring a codec gives wrong bytes as @@ -389,6 +396,9 @@ def validate_array_metadata_v3( ) else: problems.extend(_prefix("fill_value", validate_json(doc["fill_value"]))) + # The chunk grid is judged against the shape, once both are read. + if "chunk_grid" in read and shape is not None: + problems.extend(chunk_grid_problems(read["chunk_grid"], shape, ("chunk_grid",))) for key, kind in _EXTENSION_LISTS_V3: if key in doc: entries = doc[key] @@ -410,7 +420,6 @@ def validate_array_metadata_v3( # field-level loc, not per-bad-item locs; per-index locs are reserved for # the metadata-field lists (codecs, storage_transformers). names = doc["dimension_names"] - shape = doc.get("shape") if not _is_array(names): problems.append( ValidationProblem(("dimension_names",), "expected a sequence", "invalid_type") @@ -421,7 +430,7 @@ def validate_array_metadata_v3( ("dimension_names",), "expected items of str or None", "invalid_type" ) ) - elif _is_int_sequence(shape) and len(names) != len(shape): + elif shape is not None and len(names) != len(shape): problems.append( ValidationProblem( ("dimension_names",), @@ -474,19 +483,11 @@ def validate_array_metadata_v2(value: object) -> tuple[ValidationProblem, ...]: problems: list[ValidationProblem] = list(_missing_keys(ARRAY_METADATA_REQUIRED_KEYS_V2, doc)) problems.extend(_validate_other_members(doc, ARRAY_METADATA_STANDARD_KEYS_V2)) problems.extend(_check_literal(doc, "zarr_format", 2)) - shape_problems = _validate_dim_sequence(doc, "shape") - chunks_problems = _validate_dim_sequence(doc, "chunks") + shape, shape_problems = _dimension_lengths(doc, "shape") + chunks, chunks_problems = _dimension_lengths(doc, "chunks") problems.extend(shape_problems) problems.extend(chunks_problems) - shape = doc.get("shape") - chunks = doc.get("chunks") - if ( - len(shape_problems) == 0 - and len(chunks_problems) == 0 - and _is_int_sequence(shape) - and _is_int_sequence(chunks) - and len(shape) != len(chunks) - ): + if shape is not None and chunks is not None and len(shape) != len(chunks): problems.append( ValidationProblem( ("chunks",), diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py b/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py index 088a141a37..11541520e6 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py @@ -83,8 +83,8 @@ Problems: TypeAlias = tuple[ValidationProblem, ...] -def no_rules(configuration: object) -> Iterator[ValidationProblem]: - """The rules of a definition with none: every well-typed configuration is allowed.""" +def no_rules(*_: object) -> Iterator[ValidationProblem]: + """The rules of a definition with none, of any kind: everything well typed is allowed.""" yield from () @@ -93,13 +93,6 @@ def unchanged(configuration: T) -> T: return configuration -def no_fill_value_rules( - configuration: object, nested: Nested, value: object -) -> Iterator[ValidationProblem]: - """The fill value rules of a data type with none: every fill value of its JSON shape is allowed.""" - yield from () - - class EmptyConfiguration(TypedDict, closed=True): """The configuration of a definition with nothing to configure, written as its bare name.""" @@ -276,7 +269,7 @@ class DataTypeDefinition(Definition[C]): fill_value: object = JSONValue """The JSON shape of a fill value, as an annotation: `Int8FillValue`.""" - fill_value_rules: Callable[[C, Nested, Any], Iterable[ValidationProblem]] = no_fill_value_rules + fill_value_rules: Callable[[C, Nested, Any], Iterable[ValidationProblem]] = no_rules """What the spec disallows in a fill value of that shape, located in it.""" def _refusal(self) -> str | None: @@ -300,7 +293,18 @@ def _fill_value_parser(annotation: object) -> Parser: @dataclass(frozen=True, kw_only=True, slots=True) class ChunkGridDefinition(Definition[C]): - """A chunk grid.""" + """A chunk grid, and the arrays it fits. + + `shape_rules` is what the spec disallows in a grid of this + configuration over an array of a given shape: a dimension with no + chunk length, chunks that fall short of one. It is handed the + configuration, the fields it holds as the scope read them, and the + shape, and locates its problems in the configuration. A grid that + says nothing of the shape fits every one. + """ + + shape_rules: Callable[[C, Nested, tuple[int, ...]], Iterable[ValidationProblem]] = no_rules + """What the spec disallows in this grid over an array of a shape, located in the configuration.""" @dataclass(frozen=True, kw_only=True, slots=True) @@ -703,6 +707,28 @@ def fill_value_problems( return (*problems, *refused) +def chunk_grid_problems( + chunk_grid: Resolved[ChunkGridDefinition[Any]], shape: tuple[int, ...], loc: Loc = () +) -> Problems: + """What is wrong with `chunk_grid`, a chunk grid field a scope read, over an array of `shape`. + + Judged by the grid's shape rules, which locate their problems in the + configuration, under `loc`, where the field sits: a regular grid with + a chunk length for each of two dimensions, over an array of three. A + grid the scope did not read, out of scope or invalid, is left + unjudged. + """ + definition = chunk_grid.definition + configuration = chunk_grid.configuration + if definition is None or configuration is None: + return () + return _ruled( + definition, + lambda: definition.shape_rules(configuration, chunk_grid.nested, shape), + (*loc, "configuration"), + ) + + def resolve( data: object, kind: type[D], context: Context, loc: Loc = () ) -> tuple[Resolved[D], Problems]: @@ -912,11 +938,11 @@ def _replaced(value: JSONValue, path: Loc, new: JSONValue) -> JSONValue: "Unread", "as_kind", "canonicalize", + "chunk_grid_problems", "configuration_of", "fill_value_problems", "kind_of", "named_configuration", - "no_fill_value_rules", "no_rules", "resolve", "spelled", diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/chunk_grid/rectilinear.py b/packages/zarr-metadata/src/zarr_metadata/v3/chunk_grid/rectilinear.py index b685fca83f..744ce540f9 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/chunk_grid/rectilinear.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/chunk_grid/rectilinear.py @@ -11,7 +11,7 @@ from zarr_metadata._json import ValidationProblem from zarr_metadata._typed_json import Loc -from zarr_metadata.v3._definition import ChunkGridDefinition +from zarr_metadata.v3._definition import ChunkGridDefinition, Nested RECTILINEAR_CHUNK_GRID_NAME: Final = "rectilinear" """The `name` field value of the rectilinear chunk grid.""" @@ -123,11 +123,43 @@ def _canonical( ) +def _shape_rules( + configuration: RectilinearChunkGridConfiguration, nested: Nested, shape: tuple[int, ...] +) -> Iterator[ValidationProblem]: + """Chunk lengths for each of the array's dimensions, which cover it. + + "The length of `chunk_shapes` MUST match the number of dimensions of + the array", and "The sum of the edge lengths MUST equal or exceed `L`" + (https://github.com/zarr-developers/zarr-extensions/blob/4da7b37a84f76e660902f6d3de3eaef0e0febae6/chunk-grids/rectilinear/README.md?plain=1#L62-L91). + A bare integer repeats until it covers the dimension, so it always does. + """ + chunk_shapes = configuration["chunk_shapes"] + if len(chunk_shapes) != len(shape): + yield ValidationProblem( + ("chunk_shapes",), + f"expected one chunk_shapes entry per dimension of shape, got {len(chunk_shapes)}", + "invalid_value", + ) + return + for axis, (spec, extent) in enumerate(zip(chunk_shapes, shape, strict=True)): + if isinstance(spec, int): + continue + covered = sum(entry if isinstance(entry, int) else entry[0] * entry[1] for entry in spec) + if covered < extent: + yield ValidationProblem( + ("chunk_shapes", axis), + f"expected chunk lengths that cover the dimension's length {extent}, " + f"got lengths summing to {covered}", + "invalid_value", + ) + + RECTILINEAR_CHUNK_GRID: Final = ChunkGridDefinition( name=RECTILINEAR_CHUNK_GRID_NAME, configuration=RectilinearChunkGridConfiguration, rules=_rules, canonical=_canonical, + shape_rules=_shape_rules, ) """The `rectilinear` chunk grid.""" diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/chunk_grid/regular.py b/packages/zarr-metadata/src/zarr_metadata/v3/chunk_grid/regular.py index c6992888b6..705679b550 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/chunk_grid/regular.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/chunk_grid/regular.py @@ -10,7 +10,7 @@ from typing_extensions import TypedDict from zarr_metadata._json import ValidationProblem -from zarr_metadata.v3._definition import ChunkGridDefinition +from zarr_metadata.v3._definition import ChunkGridDefinition, Nested REGULAR_CHUNK_GRID_NAME: Final = "regular" """The `name` field value of the regular chunk grid.""" @@ -49,8 +49,8 @@ def _rules(configuration: RegularChunkGridConfiguration) -> Iterator[ValidationP "The chunk shape elements are non-zero when the corresponding dimensions of the arrays have non-zero length": an extent of 0 is right for a dimension of length 0, which zarr-python 3.0 and 3.1 - wrote, and which a grid alone cannot tell from one that is not. The - array's shape can, where the grid is read beside it. + wrote, and which a grid alone cannot tell from one that is not; the + shape rules can, beside the array's shape. """ for index, extent in enumerate(configuration["chunk_shape"]): if extent < 0: @@ -59,8 +59,40 @@ def _rules(configuration: RegularChunkGridConfiguration) -> Iterator[ValidationP ) +def _shape_rules( + configuration: RegularChunkGridConfiguration, nested: Nested, shape: tuple[int, ...] +) -> Iterator[ValidationProblem]: + """A chunk length for each of the array's dimensions, and 0 only for a dimension of length 0. + + "The dimensionality of the grid is the same as the dimensionality of + the array" + (https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v3/chunk-grids/regular-grid/index.rst#L29-L31), + and "The chunk shape elements are non-zero when the corresponding + dimensions of the arrays have non-zero length" + (https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v3/core/index.rst#L284-L285). + """ + chunk_shape = configuration["chunk_shape"] + if len(chunk_shape) != len(shape): + yield ValidationProblem( + ("chunk_shape",), + f"expected one chunk length per dimension of shape, got {len(chunk_shape)}", + "invalid_value", + ) + return + for axis, (length, extent) in enumerate(zip(chunk_shape, shape, strict=True)): + if length == 0 and extent != 0: + yield ValidationProblem( + ("chunk_shape", axis), + f"expected a chunk length >= 1 for a dimension of length {extent}, got 0", + "invalid_value", + ) + + REGULAR_CHUNK_GRID: Final = ChunkGridDefinition( - name=REGULAR_CHUNK_GRID_NAME, configuration=RegularChunkGridConfiguration, rules=_rules + name=REGULAR_CHUNK_GRID_NAME, + configuration=RegularChunkGridConfiguration, + rules=_rules, + shape_rules=_shape_rules, ) """The `regular` chunk grid.""" diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/_numpy_time.py b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/_numpy_time.py index d9d972284b..9b3b6671d0 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/_numpy_time.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/_numpy_time.py @@ -40,7 +40,7 @@ def numpy_time_fill_value_rules( ) -> Iterator[ValidationProblem]: """An integer fill value is a signed 64-bit one; `"NaT"` is the other form, which the shape admits. - https://github.com/zarr-developers/zarr-extensions/blob/6a3adaeef244b3c76270dca52d6a849e88cf002c/data-types/numpy.datetime64/README.md?plain=1#L109-L112 + https://github.com/zarr-developers/zarr-extensions/blob/4da7b37a84f76e660902f6d3de3eaef0e0febae6/data-types/numpy.datetime64/README.md?plain=1#L109-L112 """ if isinstance(value, int) and not -(2**63) <= value <= 2**63 - 1: yield ValidationProblem( diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/struct.py b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/struct.py index 9e52d59a96..789374158f 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/struct.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/struct.py @@ -96,7 +96,7 @@ def _fill_value_rules( """A fill value for every field, each one judged by that field's own type, and for nothing else. Every field needs one, valid for its type - (https://github.com/zarr-developers/zarr-extensions/blob/6a3adaeef244b3c76270dca52d6a849e88cf002c/data-types/struct/README.md?plain=1#L221-L224). + (https://github.com/zarr-developers/zarr-extensions/blob/4da7b37a84f76e660902f6d3de3eaef0e0febae6/data-types/struct/README.md?plain=1#L221-L224). A field type the scope did not read, or whose reading the struct's does not hold, leaves its fill value unjudged. """ diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/definition.py b/packages/zarr-metadata/src/zarr_metadata/v3/definition.py index 07a84e05b7..92623a04b8 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/definition.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/definition.py @@ -110,9 +110,10 @@ def acme_lz4_rules(configuration: AcmeLz4Configuration) -> Iterator[ValidationPr `validate_array_metadata_v3(document, context=SCOPE)`, from `zarr_metadata.model`, reads each extension point of a v3 array document through the definitions in `SCOPE`, and so do the model's `from_json` and -`from_key_value`, and a fill value is judged against the data type it -names, by that data type's definition. No rule here reads a codec against -the array it is handed, or a chunk grid against the shape. +`from_key_value`: a fill value is judged against the data type it names, +by that data type's definition, and the chunk grid against the shape, by +the grid's definition. No rule here reads a codec against the array it +is handed. The TypedDict says what a key it does not declare is: with `closed=True`, a problem, as above; with `extra_items=`, a key holding @@ -147,6 +148,12 @@ def acme_lz4_rules(configuration: AcmeLz4Configuration) -> Iterator[ValidationPr type field the scope read; one nothing in scope claims leaves it unjudged. A data type that says nothing of its fill value takes any JSON. +A chunk grid says which arrays it fits: `shape_rules`, a function +yielding what the spec disallows in a grid of its configuration over an +array of a given shape -- a dimension with no chunk length, chunks that +fall short of one -- located in the configuration. A grid that says +nothing of the shape fits every one. + Raw bits are the one data type whose name carries its configuration: a document writes `r` and the size in bits, and `r16` reads as `r*`, as the specification's table writes raw bits, with `{"bits": 16}`. A reader that @@ -169,11 +176,12 @@ def acme_lz4_rules(configuration: AcmeLz4Configuration) -> Iterator[ValidationPr `TypeError` saying what is wrong: a `configuration` that is not a TypedDict, says nothing of the keys it does not declare, or has a member no checker reads, named down to the TypedDict that holds it; a `name` -that is not a string; `rules` or `canonical` that are not functions; a -codec `kind` that is not one of the three, or a `size` that is not -`"static"` or `"dynamic"`; a data type named as raw bits of one size are -written. A scope refuses a definition of no kind. Nothing happens at -class creation. +that is not a string; a member declared as a function -- `rules`, +`canonical`, `fill_value_rules`, `shape_rules` -- that is not one; a data +type's `fill_value` no checker reads; a codec `kind` that is not one of +the three, or a `size` that is not `"static"` or `"dynamic"`; a data type +named as raw bits of one size are written. A scope refuses a definition +of no kind. Nothing happens at class creation. """ from zarr_metadata._common import JSONValue diff --git a/packages/zarr-metadata/tests/model/test_array.py b/packages/zarr-metadata/tests/model/test_array.py index a07bfa1c76..6db6bfa3d5 100644 --- a/packages/zarr-metadata/tests/model/test_array.py +++ b/packages/zarr-metadata/tests/model/test_array.py @@ -1572,7 +1572,8 @@ def test_error_array_v2_key_that_is_not_a_string(key: object) -> None: def test_array_v3_from_json_materializes_abstract_containers() -> None: """A flexible input mapping becomes the canonical dict/tuple model shape.""" - doc = UserDict(dict(ZarrV3ArrayMetadata.create_default(shape=(2,)).to_json())) + # `range(2)` is the shape (0, 1), which the default grid for it fits. + doc = UserDict(dict(ZarrV3ArrayMetadata.create_default(shape=(0, 1)).to_json())) doc["shape"] = range(2) model = ZarrV3ArrayMetadata.from_json(doc) diff --git a/packages/zarr-metadata/tests/v3/test_definitions.py b/packages/zarr-metadata/tests/v3/test_definitions.py index bd56034d83..4c5babfe2f 100644 --- a/packages/zarr-metadata/tests/v3/test_definitions.py +++ b/packages/zarr-metadata/tests/v3/test_definitions.py @@ -710,11 +710,21 @@ def test_error_a_data_type_fill_value_no_checker_reads() -> None: DataTypeDefinition(name="acme.set", configuration=Empty, fill_value=set[int]) -@pytest.mark.parametrize("member", ["rules", "canonical", "fill_value_rules"]) -def test_error_a_function_member_that_is_not_a_function(member: str) -> None: +@pytest.mark.parametrize( + ("kind", "member"), + [ + (DataTypeDefinition, "rules"), + (DataTypeDefinition, "canonical"), + (DataTypeDefinition, "fill_value_rules"), + (ChunkGridDefinition, "shape_rules"), + ], +) +def test_error_a_function_member_that_is_not_a_function( + kind: type[Definition[Any]], member: str +) -> None: # Each member a definition's annotations declare a `Callable`. with pytest.raises(TypeError, match=f"'acme.t': {member} is a function, got 'none'"): - DataTypeDefinition(name="acme.t", configuration=Empty, **{member: "none"}) # pyright: ignore[reportArgumentType] + kind(name="acme.t", configuration=Empty, **{member: "none"}) # pyright: ignore[reportArgumentType] @pytest.mark.parametrize("kind", [Definition, AcmeCodecDefinition]) diff --git a/packages/zarr-metadata/tests/v3/test_grid_shapes.py b/packages/zarr-metadata/tests/v3/test_grid_shapes.py new file mode 100644 index 0000000000..dc98c18018 --- /dev/null +++ b/packages/zarr-metadata/tests/v3/test_grid_shapes.py @@ -0,0 +1,117 @@ +"""Chunk grids, judged against the shape of the array they chunk. + +Each grid's definition says what the spec disallows in a grid of its +configuration over an array of a given shape; `chunk_grid_problems` +judges a grid field a scope read, and the v3 array validators judge a +document's `chunk_grid` against its `shape`. +""" + +from __future__ import annotations + +import pytest + +from zarr_metadata.model import ZarrV3ArrayMetadata, validate_array_metadata_v3 +from zarr_metadata.v3._definition import chunk_grid_problems +from zarr_metadata.v3.codec.crc32c import Empty +from zarr_metadata.v3.definition import ( + CORE_AND_EXTENSIONS, + ChunkGridDefinition, + JSONValue, + resolve, +) + + +def _regular(*lengths: int) -> JSONValue: + return {"name": "regular", "configuration": {"chunk_shape": list(lengths)}} + + +def _rectilinear(*specs: JSONValue) -> JSONValue: + return {"name": "rectilinear", "configuration": {"kind": "inline", "chunk_shapes": list(specs)}} + + +def _problems(grid: JSONValue, shape: tuple[int, ...]) -> list[tuple[tuple[str | int, ...], str]]: + resolved, found = resolve(grid, ChunkGridDefinition, CORE_AND_EXTENSIONS) + assert found == () + return [(problem.loc, problem.kind) for problem in chunk_grid_problems(resolved, shape)] + + +@pytest.mark.parametrize( + ("grid", "shape"), + [ + (_regular(), ()), + (_regular(4, 4), (10, 3)), + # A chunk longer than its dimension, and a chunk length of 0 for a + # dimension of length 0. + (_regular(8, 0), (3, 0)), + # A bare integer repeats until it covers its dimension. + (_rectilinear(4), (10,)), + (_rectilinear([4, 4, 2]), (10,)), + # Overflowing the dimension is allowed. + (_rectilinear([4, 4, 4]), (10,)), + (_rectilinear([[4, 2], 2]), (10,)), + # Nothing to cover. + (_rectilinear([]), (0,)), + (_rectilinear(2, [3, 3]), (7, 6)), + # A grid nothing in scope claims, or one that is not read, is left + # unjudged. + ({"name": "acme.grid", "configuration": {"x": 1}}, (3,)), + (_regular(-1), (3,)), + ], +) +def test_every_chunk_grid_fits_the_shapes_it_covers( + grid: JSONValue, shape: tuple[int, ...] +) -> None: + resolved, _ = resolve(grid, ChunkGridDefinition, CORE_AND_EXTENSIONS) + assert chunk_grid_problems(resolved, shape) == () + + +@pytest.mark.parametrize( + ("grid", "shape"), [(_regular(4), (10, 3)), (_regular(4, 4), ()), (_regular(), (1,))] +) +def test_error_a_regular_grid_of_another_rank(grid: JSONValue, shape: tuple[int, ...]) -> None: + assert _problems(grid, shape) == [(("configuration", "chunk_shape"), "invalid_value")] + + +def test_error_a_regular_chunk_length_of_0_for_a_dimension_that_is_not_empty() -> None: + assert _problems(_regular(4, 0), (4, 3)) == [ + (("configuration", "chunk_shape", 1), "invalid_value") + ] + + +@pytest.mark.parametrize( + ("grid", "shape"), [(_rectilinear(4), (10, 3)), (_rectilinear(4, 4), (1,))] +) +def test_error_a_rectilinear_grid_of_another_rank(grid: JSONValue, shape: tuple[int, ...]) -> None: + assert _problems(grid, shape) == [(("configuration", "chunk_shapes"), "invalid_value")] + + +@pytest.mark.parametrize("spec", [[4, 4], [[4, 2]], []]) +def test_error_rectilinear_chunks_that_do_not_cover_their_dimension(spec: JSONValue) -> None: + assert _problems(_rectilinear(2, spec), (4, 10)) == [ + (("configuration", "chunk_shapes", 1), "invalid_value") + ] + + +def test_a_grid_without_shape_rules_fits_every_shape() -> None: + lenient = ChunkGridDefinition(name="acme.grid", configuration=Empty) + scope = CORE_AND_EXTENSIONS.extended_with(lenient) + resolved, _ = resolve("acme.grid", ChunkGridDefinition, scope) + assert chunk_grid_problems(resolved, (3, 4)) == () + + +def test_error_an_array_document_s_chunk_grid_that_does_not_fit_its_shape() -> None: + # Located in the document's grid. + document = dict(ZarrV3ArrayMetadata.create_default(shape=(4, 4)).to_json()) + document["chunk_grid"] = _regular(4) + assert [(p.loc, p.kind) for p in validate_array_metadata_v3(document)] == [ + (("chunk_grid", "configuration", "chunk_shape"), "invalid_value") + ] + + +def test_error_an_array_document_s_shape_that_is_not_read_leaves_its_grid_unjudged() -> None: + document = dict(ZarrV3ArrayMetadata.create_default(shape=(4, 4)).to_json()) + document["chunk_grid"] = _regular(4) + document["shape"] = (4, -1) + assert [(p.loc, p.kind) for p in validate_array_metadata_v3(document)] == [ + (("shape",), "invalid_value") + ] From 23b64abbd5bb7bc04583a2f2d3e9ecfbc97b5508 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Sun, 27 Sep 2026 17:22:57 +0200 Subject: [PATCH 06/17] fix(zarr-metadata): create_default gives an empty dimension a chunk length of 1 The grid `create_default` derives from an overridden `shape` had a chunk length of 0 for a dimension of length 0. The regular grid asks for chunk lengths greater than zero, and `zarr` refuses to open such a grid, so the derived grid now has a chunk length of 1 there. It still fits the shape. Assisted-by: ClaudeCode:claude-opus-5-5 --- packages/zarr-metadata/changes/+layer-4b.bugfix.md | 3 +++ .../zarr-metadata/src/zarr_metadata/model/_array.py | 7 +++++-- packages/zarr-metadata/tests/model/test_array.py | 10 +++++----- 3 files changed, 13 insertions(+), 7 deletions(-) create mode 100644 packages/zarr-metadata/changes/+layer-4b.bugfix.md diff --git a/packages/zarr-metadata/changes/+layer-4b.bugfix.md b/packages/zarr-metadata/changes/+layer-4b.bugfix.md new file mode 100644 index 0000000000..645f8ec603 --- /dev/null +++ b/packages/zarr-metadata/changes/+layer-4b.bugfix.md @@ -0,0 +1,3 @@ +`ZarrV3ArrayMetadata.create_default(shape=...)` derives a chunk length of 1 +for a dimension of length 0, where it wrote 0: the regular grid asks for +chunk lengths greater than zero, and `zarr` does not open a grid with one. diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_array.py b/packages/zarr-metadata/src/zarr_metadata/model/_array.py index 6161d34eb2..4d5140eea0 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_array.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/_array.py @@ -191,7 +191,9 @@ def create_default(cls, **overrides: Unpack[ZarrV3ArrayMetadataPartial]) -> Zarr analog of `list()` returning `[]`. Any field can be overridden by keyword (the same fields accepted by `update`). Overriding `shape` without `chunk_grid` derives a consistent default grid: one regular chunk - covering the array (`chunk_shape` equal to `shape`). + covering the array (`chunk_shape` equal to `shape`, with a length of + 1 for a dimension of length 0, since a chunk length is at least 1: + https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v3/chunk-grids/regular-grid/index.rst#L40). The derivation is deliberately one-way. A user-supplied `chunk_grid` is an extension point and is taken verbatim — deriving `shape` from @@ -206,8 +208,9 @@ def create_default(cls, **overrides: Unpack[ZarrV3ArrayMetadataPartial]) -> Zarr refuses, so pass the two together. """ if "shape" in overrides and "chunk_grid" not in overrides: + chunk_shape = tuple(max(length, 1) for length in overrides["shape"]) overrides["chunk_grid"] = ZarrV3NamedConfig( - name="regular", configuration={"chunk_shape": tuple(overrides["shape"])} + name="regular", configuration={"chunk_shape": chunk_shape} ) default = cls( shape=(), diff --git a/packages/zarr-metadata/tests/model/test_array.py b/packages/zarr-metadata/tests/model/test_array.py index 6db6bfa3d5..17f81fc13f 100644 --- a/packages/zarr-metadata/tests/model/test_array.py +++ b/packages/zarr-metadata/tests/model/test_array.py @@ -1893,12 +1893,12 @@ def test_v2_create_default_explicit_chunks_respected() -> None: def test_v3_create_default_zero_length_dimensions() -> None: - """chunk_shape == shape is spec-sound even with zero-length dimensions: - 'The chunk shape elements are non-zero when the corresponding dimensions - of the arrays have non-zero length' — the constraint is conditional, so a - zero chunk length is permitted exactly where the dimension is empty.""" + """The derived grid gives a dimension of length 0 a chunk length of 1: + the regular grid asks for chunk sizes greater than zero, so the written + grid is one every reader takes, and it fits the shape it chunks.""" model = ZarrV3ArrayMetadata.create_default(shape=(0, 3)) - assert model.chunk_grid.configuration["chunk_shape"] == (0, 3) + assert model.chunk_grid.configuration["chunk_shape"] == (1, 3) + assert validate_array_metadata_v3(model.to_json()) == () def test_create_default_derivation_is_one_way() -> None: From e0f0db1c401e8c589e1696240dbe701941e83c3d Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Sun, 27 Sep 2026 17:25:57 +0200 Subject: [PATCH 07/17] docs(zarr-metadata): the grid fragments take this PR's number Assisted-by: ClaudeCode:claude-opus-5-5 --- .../zarr-metadata/changes/{+layer-4b.bugfix.md => 4441.bugfix.md} | 0 .../changes/{+layer-4b.feature.md => 4441.feature.md} | 0 2 files changed, 0 insertions(+), 0 deletions(-) rename packages/zarr-metadata/changes/{+layer-4b.bugfix.md => 4441.bugfix.md} (100%) rename packages/zarr-metadata/changes/{+layer-4b.feature.md => 4441.feature.md} (100%) diff --git a/packages/zarr-metadata/changes/+layer-4b.bugfix.md b/packages/zarr-metadata/changes/4441.bugfix.md similarity index 100% rename from packages/zarr-metadata/changes/+layer-4b.bugfix.md rename to packages/zarr-metadata/changes/4441.bugfix.md diff --git a/packages/zarr-metadata/changes/+layer-4b.feature.md b/packages/zarr-metadata/changes/4441.feature.md similarity index 100% rename from packages/zarr-metadata/changes/+layer-4b.feature.md rename to packages/zarr-metadata/changes/4441.feature.md From 3628f48b6ace955a569c46a47266208aeb82ef6f Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Sun, 27 Sep 2026 17:36:28 +0200 Subject: [PATCH 08/17] feat(zarr-metadata)!: a definition's rules see the fields its configuration holds `rules` is handed the configuration and the fields it holds as the scope read them (`Nested`), as `fill_value_rules` and `shape_rules` are, so a rule can judge a configuration by what is inside it: a struct's field types, a cast's target. `resolve` reads the nested fields before it asks the rules, and still reports the rules' problems first; `judge` reads in no scope, and hands them nothing read. Every definition's rules take the new argument. Assisted-by: ClaudeCode:claude-opus-5-5 --- .../src/zarr_metadata/v3/_definition.py | 34 ++++++++------ .../v3/chunk_grid/rectilinear.py | 4 +- .../zarr_metadata/v3/chunk_grid/regular.py | 4 +- .../src/zarr_metadata/v3/codec/blosc.py | 4 +- .../src/zarr_metadata/v3/codec/gzip.py | 4 +- .../zarr_metadata/v3/codec/scale_offset.py | 6 ++- .../v3/codec/sharding_indexed.py | 6 ++- .../src/zarr_metadata/v3/codec/transpose.py | 6 ++- .../src/zarr_metadata/v3/codec/zstd.py | 4 +- .../zarr_metadata/v3/data_type/_numpy_time.py | 4 +- .../src/zarr_metadata/v3/data_type/raw.py | 2 +- .../src/zarr_metadata/v3/data_type/struct.py | 2 +- .../src/zarr_metadata/v3/definition.py | 16 ++++--- .../tests/model/test_extension_points.py | 2 +- .../tests/v3/test_definitions.py | 45 +++++++++++++++---- 15 files changed, 99 insertions(+), 44 deletions(-) diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py b/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py index 11541520e6..262c06654d 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py @@ -111,8 +111,9 @@ class Definition(Generic[C]): type; with `closed=False`, anything. `rules` yields what the spec disallows in a configuration of that type, as it finds each; it is handed only a configuration that has passed the check, holding what - the TypedDict admits and nothing else -- `judge` is the two, for a - caller holding JSON. + the TypedDict admits and nothing else, and the fields it holds as the + scope read them -- a struct's field types -- which is nothing when no + scope read it. `judge` is the two, for a caller holding JSON. `canonical` is where two spellings of the configuration that mean the same thing are made one. @@ -127,8 +128,8 @@ class Definition(Generic[C]): """The name the metadata carries, which a scope files the definition under.""" configuration: type[C] """The TypedDict the configuration is.""" - rules: Callable[[C], Iterable[ValidationProblem]] = no_rules - """What the spec disallows in a well-typed configuration, located in it.""" + rules: Callable[[C, Nested], Iterable[ValidationProblem]] = no_rules + """What the spec disallows in a well-typed configuration and the fields it holds, located in it.""" canonical: Callable[[C], C] = unchanged """A well-typed, allowed configuration in its simplest equivalent spelling. @@ -170,12 +171,15 @@ def judge(self, value: object, loc: Loc = ()) -> tuple[C | None, Problems]: whose nested fields are well formed, holding what its TypedDict admits and nothing else, so a caller holding JSON never reaches a rule with a member of the wrong type, or one the type says cannot - be there. + be there. No scope reads the fields it holds, so the rules see + none of them read, and a rule about one -- a struct's field of a + type whose values vary in size -- finds nothing to judge: `resolve` + reads the field in a scope, and asks every rule. """ configuration, problems = self.check(value, loc) if configuration is None: return None, problems - refused = _ruled(self, lambda: self.rules(configuration), loc) + refused = _ruled(self, lambda: self.rules(configuration, _nothing_nested()), loc) return (configuration if len(refused) == 0 else None), (*problems, *refused) @@ -784,19 +788,23 @@ def _read( missing = problem(at, f"{name!r} requires a configuration", "missing_key") return Resolved(data, "invalid", definition, None), missing typed, found, nested = _checked(definition.configuration, {} if given is None else given, at) - problems = list(found) envelopes = [_envelope(field) for field in nested] sound = _usable(found) and all(_usable(envelope) for envelope in envelopes) configuration = cast("Mapping[str, JSONValue]", typed) if sound else None - if configuration is not None: - problems.extend(_ruled(definition, lambda: definition.rules(configuration), at)) + # The fields it holds are read first, so the rules see them as the + # scope read them; their problems are reported after the rules'. within: dict[Loc, Resolved[Any]] = {} + inside: list[ValidationProblem] = [] for field, envelope in zip(nested, envelopes, strict=True): - problems.extend(envelope) - inner, found = _read(field.json, field.kind, context, field.loc) + inside.extend(envelope) + inner, found_inside = _read(field.json, field.kind, context, field.loc) within[field.loc[len(at) :]] = inner - problems.extend(found) - problems.extend(_sized(field, inner.definition)) + inside.extend(found_inside) + inside.extend(_sized(field, inner.definition)) + problems = list(found) + if configuration is not None: + problems.extend(_ruled(definition, lambda: definition.rules(configuration, within), at)) + problems.extend(inside) if not _usable(problems): return Resolved(data, "invalid", definition, None), tuple(problems) return Resolved(data, "read", definition, configuration, within), tuple(problems) diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/chunk_grid/rectilinear.py b/packages/zarr-metadata/src/zarr_metadata/v3/chunk_grid/rectilinear.py index 744ce540f9..0818e90a5f 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/chunk_grid/rectilinear.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/chunk_grid/rectilinear.py @@ -57,7 +57,9 @@ def _not_positive(loc: Loc, value: int) -> ValidationProblem: return ValidationProblem(loc, f"expected an integer >= 1, got {value}", "invalid_value") -def _rules(configuration: RectilinearChunkGridConfiguration) -> Iterator[ValidationProblem]: +def _rules( + configuration: RectilinearChunkGridConfiguration, nested: Nested +) -> Iterator[ValidationProblem]: """Every extent, and every run's length and count, is at least 1.""" for axis, spec in enumerate(configuration["chunk_shapes"]): if isinstance(spec, int): diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/chunk_grid/regular.py b/packages/zarr-metadata/src/zarr_metadata/v3/chunk_grid/regular.py index 705679b550..8ee48ba300 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/chunk_grid/regular.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/chunk_grid/regular.py @@ -43,7 +43,9 @@ class RegularChunkGridObject(TypedDict, closed=True): """ -def _rules(configuration: RegularChunkGridConfiguration) -> Iterator[ValidationProblem]: +def _rules( + configuration: RegularChunkGridConfiguration, nested: Nested +) -> Iterator[ValidationProblem]: """No chunk extent is negative. "The chunk shape elements are non-zero when the corresponding diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/codec/blosc.py b/packages/zarr-metadata/src/zarr_metadata/v3/codec/blosc.py index 58462418a1..fb162c537d 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/codec/blosc.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/codec/blosc.py @@ -10,7 +10,7 @@ from typing_extensions import TypedDict from zarr_metadata._json import ValidationProblem -from zarr_metadata.v3._definition import CodecDefinition +from zarr_metadata.v3._definition import CodecDefinition, Nested BLOSC_CODEC_NAME: Final = "blosc" """The `name` field value of the `blosc` codec.""" @@ -68,7 +68,7 @@ class BloscCodecObject(TypedDict, closed=True): """ -def _rules(configuration: BloscCodecConfiguration) -> Iterator[ValidationProblem]: +def _rules(configuration: BloscCodecConfiguration, nested: Nested) -> Iterator[ValidationProblem]: """Bounds on `clevel` and `blocksize`; `typesize` against `shuffle`. Under `noshuffle` the spec says of `typesize` that "the value is diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/codec/gzip.py b/packages/zarr-metadata/src/zarr_metadata/v3/codec/gzip.py index 6811de9fbf..aa07d18421 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/codec/gzip.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/codec/gzip.py @@ -10,7 +10,7 @@ from typing_extensions import TypedDict from zarr_metadata._json import ValidationProblem -from zarr_metadata.v3._definition import CodecDefinition +from zarr_metadata.v3._definition import CodecDefinition, Nested GZIP_CODEC_NAME: Final = "gzip" """The `name` field value of the `gzip` codec.""" @@ -53,7 +53,7 @@ class GzipCodecObject(TypedDict, closed=True): """ -def _rules(configuration: GzipCodecConfiguration) -> Iterator[ValidationProblem]: +def _rules(configuration: GzipCodecConfiguration, nested: Nested) -> Iterator[ValidationProblem]: """`level` is an integer from 0 to 9.""" level = configuration["level"] if not 0 <= level <= 9: diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/codec/scale_offset.py b/packages/zarr-metadata/src/zarr_metadata/v3/codec/scale_offset.py index cb5460359f..d7dba8311d 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/codec/scale_offset.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/codec/scale_offset.py @@ -11,7 +11,7 @@ from zarr_metadata._common import JSONValue from zarr_metadata._json import ValidationProblem -from zarr_metadata.v3._definition import CodecDefinition +from zarr_metadata.v3._definition import CodecDefinition, Nested SCALE_OFFSET_CODEC_NAME: Final = "scale_offset" """The `name` field value of the `scale_offset` codec.""" @@ -58,7 +58,9 @@ class ScaleOffsetCodecObject(TypedDict, closed=True): """ -def _rules(configuration: ScaleOffsetCodecConfiguration) -> Iterator[ValidationProblem]: +def _rules( + configuration: ScaleOffsetCodecConfiguration, nested: Nested +) -> Iterator[ValidationProblem]: """Neither scalar is null. The registry says each is "JSON-encoded per the input array's diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/codec/sharding_indexed.py b/packages/zarr-metadata/src/zarr_metadata/v3/codec/sharding_indexed.py index cf6cf5c20c..5c6745a90e 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/codec/sharding_indexed.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/codec/sharding_indexed.py @@ -10,7 +10,7 @@ from typing_extensions import TypedDict from zarr_metadata._json import ValidationProblem -from zarr_metadata.v3._definition import CodecDefinition, CodecField, StaticCodecField +from zarr_metadata.v3._definition import CodecDefinition, CodecField, Nested, StaticCodecField SHARDING_INDEXED_CODEC_NAME: Final = "sharding_indexed" """The `name` field value of the `sharding_indexed` codec.""" @@ -69,7 +69,9 @@ class ShardingIndexedCodecObject(TypedDict, closed=True): """ -def _rules(configuration: ShardingIndexedCodecConfiguration) -> Iterator[ValidationProblem]: +def _rules( + configuration: ShardingIndexedCodecConfiguration, nested: Nested +) -> Iterator[ValidationProblem]: """Every inner chunk extent is at least 1.""" for index, extent in enumerate(configuration["chunk_shape"]): if extent < 1: diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/codec/transpose.py b/packages/zarr-metadata/src/zarr_metadata/v3/codec/transpose.py index def06bfc30..bb82fc268d 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/codec/transpose.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/codec/transpose.py @@ -10,7 +10,7 @@ from typing_extensions import TypedDict from zarr_metadata._json import ValidationProblem -from zarr_metadata.v3._definition import CodecDefinition +from zarr_metadata.v3._definition import CodecDefinition, Nested TRANSPOSE_CODEC_NAME: Final = "transpose" """The `name` field value of the `transpose` codec.""" @@ -48,7 +48,9 @@ class TransposeCodecObject(TypedDict, closed=True): """ -def _rules(configuration: TransposeCodecConfiguration) -> Iterator[ValidationProblem]: +def _rules( + configuration: TransposeCodecConfiguration, nested: Nested +) -> Iterator[ValidationProblem]: """`order` permutes its own axes. Whether it permutes the *array's* axes needs the array's rank, and is diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/codec/zstd.py b/packages/zarr-metadata/src/zarr_metadata/v3/codec/zstd.py index 059cf0b584..efeefb7edb 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/codec/zstd.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/codec/zstd.py @@ -12,7 +12,7 @@ from typing_extensions import TypedDict from zarr_metadata._json import ValidationProblem -from zarr_metadata.v3._definition import CodecDefinition +from zarr_metadata.v3._definition import CodecDefinition, Nested ZSTD_CODEC_NAME: Final = "zstd" """The `name` field value of the `zstd` codec.""" @@ -58,7 +58,7 @@ class ZstdCodecObject(TypedDict, closed=True): """The highest `level` zstd accepts: ZSTD_maxCLevel().""" -def _rules(configuration: ZstdCodecConfiguration) -> Iterator[ValidationProblem]: +def _rules(configuration: ZstdCodecConfiguration, nested: Nested) -> Iterator[ValidationProblem]: """`level` is one zstd accepts.""" level = configuration["level"] if not ZSTD_MIN_LEVEL <= level <= ZSTD_MAX_LEVEL: diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/_numpy_time.py b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/_numpy_time.py index 9b3b6671d0..5d60aa0978 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/_numpy_time.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/_numpy_time.py @@ -24,7 +24,9 @@ class NumpyTimeConfiguration(TypedDict): scale_factor: ReadOnly[int] -def numpy_time_rules(configuration: NumpyTimeConfiguration) -> Iterator[ValidationProblem]: +def numpy_time_rules( + configuration: NumpyTimeConfiguration, nested: Nested +) -> Iterator[ValidationProblem]: """`scale_factor` is a positive int32.""" scale_factor = configuration["scale_factor"] if not 1 <= scale_factor <= NUMPY_TIME_MAX_SCALE_FACTOR: diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/raw.py b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/raw.py index 1cbca53607..23bcf71c03 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/raw.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/raw.py @@ -63,7 +63,7 @@ class RawBytesConfiguration(TypedDict, closed=True): bits: int -def _rules(configuration: RawBytesConfiguration) -> Iterator[ValidationProblem]: +def _rules(configuration: RawBytesConfiguration, nested: Nested) -> Iterator[ValidationProblem]: """ "raw bits, variable size given by *, limited to be a multiple of 8" -- and zero bits is not a type.""" bits = configuration["bits"] if bits == 0 or bits % 8 != 0: diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/struct.py b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/struct.py index 789374158f..9e9f738a7a 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/struct.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/struct.py @@ -64,7 +64,7 @@ class Struct(TypedDict, closed=True): """ -def _rules(configuration: StructConfiguration) -> Iterator[ValidationProblem]: +def _rules(configuration: StructConfiguration, nested: Nested) -> Iterator[ValidationProblem]: """Fields exist, and their names are non-empty and distinct. A fill value addresses fields by name. Whether each field's type is diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/definition.py b/packages/zarr-metadata/src/zarr_metadata/v3/definition.py index 92623a04b8..24a1a4e903 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/definition.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/definition.py @@ -75,7 +75,11 @@ definition; then a scope that holds it. The TypedDict is a `typing_extensions.TypedDict`: `closed` and `extra_items` are PEP 728's, which `typing.TypedDict` does not take on the versions this package -supports. +supports. The rules are handed the configuration and the fields it holds +as the scope read them: a field that is read keeps what it read inside it +as `Resolved.nested`, a `Nested` mapping by where each sits, so a +struct's rules reach its field types. `judge`, which reads in no scope, +hands them none. from collections.abc import Iterator @@ -84,6 +88,7 @@ from zarr_metadata.v3.definition import ( CORE_AND_EXTENSIONS, CodecDefinition, + Nested, ValidationProblem, ) @@ -92,7 +97,9 @@ class AcmeLz4Configuration(TypedDict, closed=True): acceleration: int - def acme_lz4_rules(configuration: AcmeLz4Configuration) -> Iterator[ValidationProblem]: + def acme_lz4_rules( + configuration: AcmeLz4Configuration, nested: Nested + ) -> Iterator[ValidationProblem]: if configuration["acceleration"] < 1: yield ValidationProblem(("acceleration",), "expected an integer >= 1", "invalid_value") @@ -141,9 +148,8 @@ def acme_lz4_rules(configuration: AcmeLz4Configuration) -> Iterator[ValidationPr `fill_value_rules`, a function yielding what the spec disallows in a fill value of that shape: an integer out of range, a hex string of another width. The rules are handed the configuration, the fields it holds as the -scope read them, and the typed fill value; a field that is read keeps what -it read inside it as `Resolved.nested`, a `Nested` mapping by location, so -a struct judges each field's fill value by that field's own type. +scope read them, and the typed fill value, so a struct judges each +field's fill value by that field's own type. `fill_value_problems(data_type, value)` judges a fill value against a data type field the scope read; one nothing in scope claims leaves it unjudged. A data type that says nothing of its fill value takes any JSON. diff --git a/packages/zarr-metadata/tests/model/test_extension_points.py b/packages/zarr-metadata/tests/model/test_extension_points.py index 77424fb567..1459e4ec48 100644 --- a/packages/zarr-metadata/tests/model/test_extension_points.py +++ b/packages/zarr-metadata/tests/model/test_extension_points.py @@ -41,7 +41,7 @@ def _document(**fields: object) -> dict[str, Any]: configuration=GZIP_CODEC.configuration, kind="bytes_bytes", size="dynamic", - rules=lambda configuration: [], + rules=lambda configuration, nested: [], ) """A reader's own gzip, which takes any level: a scope can grow, and substitute.""" diff --git a/packages/zarr-metadata/tests/v3/test_definitions.py b/packages/zarr-metadata/tests/v3/test_definitions.py index 4c5babfe2f..a12323614c 100644 --- a/packages/zarr-metadata/tests/v3/test_definitions.py +++ b/packages/zarr-metadata/tests/v3/test_definitions.py @@ -30,6 +30,7 @@ DataTypeDefinition, Definition, JSONValue, + Nested, StorageTransformerDefinition, ValidationProblem, ZarrV3MetadataFieldJSON, @@ -49,7 +50,9 @@ class AcmeStackConfiguration(TypedDict, closed=True): codecs: tuple[CodecField, ...] -def acme_stack_rules(configuration: AcmeStackConfiguration) -> Iterator[ValidationProblem]: +def acme_stack_rules( + configuration: AcmeStackConfiguration, nested: Nested +) -> Iterator[ValidationProblem]: # A rule may read a nested field's name: it is asked only of a # configuration whose nested envelopes are sound. for index, codec in enumerate(configuration["codecs"]): @@ -57,6 +60,17 @@ def acme_stack_rules(configuration: AcmeStackConfiguration) -> Iterator[Validati yield ValidationProblem( ("codecs", index), "a stack does not hold itself", "invalid_value" ) + # And what the scope read of it: a codec of another kind is refused, + # one nothing in scope claims left be. + inner = nested.get(("codecs", index)) + if ( + inner is not None + and isinstance(inner.definition, CodecDefinition) + and inner.definition.kind != "bytes_bytes" + ): + yield ValidationProblem( + ("codecs", index), "a stack holds bytes -> bytes codecs", "invalid_value" + ) ACME_STACK = CodecDefinition( @@ -113,7 +127,9 @@ class AcmeBoundedConfiguration(TypedDict, closed=True): window: NotRequired[int] -def acme_bounded_rules(configuration: AcmeBoundedConfiguration) -> Iterator[ValidationProblem]: +def acme_bounded_rules( + configuration: AcmeBoundedConfiguration, nested: Nested +) -> Iterator[ValidationProblem]: # Every key it meets is one its TypedDict declares. for key, value in cast("Mapping[str, int]", configuration).items(): low, high = ACME_BOUNDS[key] @@ -126,7 +142,9 @@ class AcmePairedConfiguration(TypedDict, closed=True): second: NotRequired[int] -def acme_paired_rules(configuration: AcmePairedConfiguration) -> Iterator[ValidationProblem]: +def acme_paired_rules( + configuration: AcmePairedConfiguration, nested: Nested +) -> Iterator[ValidationProblem]: if ("first" in configuration) != ("second" in configuration): yield ValidationProblem((), "expected first and second, or neither", "invalid_value") @@ -419,14 +437,14 @@ def test_error_must_understand_false_is_refused_wherever_the_model_refuses_it( def _ruled_by( - rules: Callable[[GzipCodecConfiguration], Iterable[ValidationProblem]], + rules: Callable[[GzipCodecConfiguration, Nested], Iterable[ValidationProblem]], ) -> Context: lying = replace(GZIP_CODEC, name="acme.gzip", rules=rules) return CORE.extended_with(lying) def test_error_a_rule_that_yields_something_else() -> None: - def rules(configuration: GzipCodecConfiguration) -> Iterator[ValidationProblem]: + def rules(configuration: GzipCodecConfiguration, nested: Nested) -> Iterator[ValidationProblem]: yield "level is too high" # pyright: ignore[reportReturnType] with pytest.raises(TypeError, match="'acme.gzip': its rules yield ValidationProblem values"): @@ -438,7 +456,7 @@ def rules(configuration: GzipCodecConfiguration) -> Iterator[ValidationProblem]: def test_error_a_rule_that_raises_says_whose_it_is() -> None: - def rules(configuration: GzipCodecConfiguration) -> Iterator[ValidationProblem]: + def rules(configuration: GzipCodecConfiguration, nested: Nested) -> Iterator[ValidationProblem]: yield from ({}[configuration["level"]],) with pytest.raises(KeyError) as raised: @@ -454,7 +472,7 @@ def rules(configuration: GzipCodecConfiguration) -> Iterator[ValidationProblem]: def test_error_a_rule_that_returns_none_says_whose_it_is() -> None: # A plain function that forgot to yield: nothing to iterate. - def rules(configuration: GzipCodecConfiguration) -> None: + def rules(configuration: GzipCodecConfiguration, nested: Nested) -> None: return None with pytest.raises(TypeError, match="not iterable") as raised: @@ -543,6 +561,17 @@ def test_error_a_rule_about_the_whole_configuration_lands_on_it() -> None: assert _locs(read) == [(("configuration",), "invalid_value")] +def test_error_a_rule_reads_the_fields_the_configuration_holds_as_the_scope_read_them() -> None: + # `bytes` is an array -> bytes codec, which the stack's rule refuses + # from what the scope read; `judge` reads in no scope, so its rule sees + # nothing read. + field = {"name": "acme.stack", "configuration": {"codecs": ["crc32c", "bytes"]}} + resolved, found = resolve(field, CodecDefinition, SCOPE) + assert resolved.resolution == "invalid" + assert _locs(found) == [(("configuration", "codecs", 1), "invalid_value")] + assert ACME_STACK.judge(field["configuration"])[1] == () + + def test_error_a_nested_member_is_not_a_field() -> None: resolved, found = resolve( {"name": "acme.stack", "configuration": {"codecs": [5]}}, CodecDefinition, SCOPE @@ -612,7 +641,7 @@ def test_a_scope_takes_a_name_over() -> None: configuration=GzipCodecConfiguration, kind="bytes_bytes", size="dynamic", - rules=lambda configuration: ( + rules=lambda configuration, nested: ( [ValidationProblem(("level",), "level 0 stores uncompressed", "invalid_value")] if configuration["level"] == 0 else [] From b66021a4a300f6f15aee0898119db19486b8eb29 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Sun, 27 Sep 2026 17:47:27 +0200 Subject: [PATCH 09/17] feat(zarr-metadata)!: the codecs are read as a pipeline, each against the chunk it is handed A v3 array's codecs are read in order, and each codec handed an array is judged against the chunk it is handed. The array hands its first codec a `Chunk`: the lengths its grid's chunks take along each of the array's dimensions, which the grid's new `chunk_lengths` says, and its data type. A codec definition says what the spec disallows in it handed a chunk, `chunk_rules`, and an array -> array codec says what it hands on, `transition`, whatever its chunk rules found. `transpose` has both: its `order` has one entry per axis of its chunk, and it hands on the axes permuted. `read_pipeline(codecs, chunk)` judges the order first (array -> array codecs, one array -> bytes codec, bytes -> bytes codecs), which refuses an empty list too, and gives each codec's `Stage` with the chunk it is handed. Nothing is guessed: the codec after one the scope did not read, or after one that says nothing of what it hands on, is handed a chunk nothing is known of. `chunk_grid_lengths` replaces the private `chunk_grid_problems`, giving a grid field's lengths with its problems, and a `Chunk` checks what it holds, so a definition's fault is reported as the definition's. Assisted-by: ClaudeCode:claude-opus-5-5 --- packages/zarr-metadata/README.md | 11 +- packages/zarr-metadata/docs/api/index.md | 5 +- packages/zarr-metadata/docs/index.md | 11 +- .../src/zarr_metadata/model/__init__.py | 5 +- .../src/zarr_metadata/model/_validation.py | 51 ++-- .../src/zarr_metadata/v3/_definition.py | 199 ++++++++++++-- .../src/zarr_metadata/v3/_pipeline.py | 189 +++++++++++++ .../v3/chunk_grid/rectilinear.py | 22 +- .../zarr_metadata/v3/chunk_grid/regular.py | 17 +- .../src/zarr_metadata/v3/codec/transpose.py | 42 ++- .../src/zarr_metadata/v3/definition.py | 48 +++- .../zarr-metadata/tests/test_public_api.py | 3 + .../tests/v3/test_definitions.py | 6 +- .../tests/v3/test_grid_shapes.py | 91 +++++-- .../zarr-metadata/tests/v3/test_pipelines.py | 257 ++++++++++++++++++ 15 files changed, 863 insertions(+), 94 deletions(-) create mode 100644 packages/zarr-metadata/src/zarr_metadata/v3/_pipeline.py create mode 100644 packages/zarr-metadata/tests/v3/test_pipelines.py diff --git a/packages/zarr-metadata/README.md b/packages/zarr-metadata/README.md index a2e7690a61..bc67f170a5 100644 --- a/packages/zarr-metadata/README.md +++ b/packages/zarr-metadata/README.md @@ -56,8 +56,8 @@ members that the strict model parser rejects. The model validators enforce the declared document structure and a small set of context-free consistency rules, including fixed format literals, finite -JSON numbers, non-negative dimensions, non-empty v3 codec pipelines, and one -`dimension_names` entry per array dimension. In a v3 document they also read +JSON numbers, non-negative dimensions, and one `dimension_names` entry per +array dimension. In a v3 document they also read each extension point -- the data type, chunk grid, chunk key encoding, each codec and each storage transformer -- through the definition that claims its name in a scope, `CORE_AND_EXTENSIONS` unless a `context` is passed: a @@ -65,9 +65,10 @@ configuration its definition refuses is refused, and a key it does not declare is reported as `unknown_key`. A name nothing in the scope claims is left unjudged, and whether to support it is the consumer's decision. A v3 fill value is judged against the data type it names, by that data type's -definition, and the chunk grid against the shape, by the grid's -definition. The validators do not judge a codec against the array it is -handed. +definition, the chunk grid against the shape, by the grid's definition, +and the codecs as a pipeline: in order, each judged by its definition +against the chunk it is handed. The validators do not read a shard's +inner codecs as a pipeline. The Pydantic integration's generated JSON Schemas express independently checkable document structure and field constraints, but they are not a diff --git a/packages/zarr-metadata/docs/api/index.md b/packages/zarr-metadata/docs/api/index.md index 95e31491f5..66c7bda75b 100644 --- a/packages/zarr-metadata/docs/api/index.md +++ b/packages/zarr-metadata/docs/api/index.md @@ -21,8 +21,9 @@ The package is organized to mirror the structure of the Zarr specifications: and [data types](v3/data_type.md) - [`zarr_metadata.v3.definition`](v3/definition.md) — each extension's metadata as a definition: the TypedDict its configuration is, and the - rules on it; check JSON against a TypedDict, judge a configuration, or - read a whole field in a scope. Its module docstring is the guide + rules on it; check JSON against a TypedDict, judge a configuration, + read a whole field in a scope, or read a codec pipeline. Its module + docstring is the guide The document types, models, and spec vocabulary — including the store keys — are re-exported at the top level, so diff --git a/packages/zarr-metadata/docs/index.md b/packages/zarr-metadata/docs/index.md index a79f2bf729..f415878091 100644 --- a/packages/zarr-metadata/docs/index.md +++ b/packages/zarr-metadata/docs/index.md @@ -71,8 +71,8 @@ members that the strict model parser rejects. The model validators enforce the declared document structure and a small set of context-free consistency rules, including fixed format literals, finite -JSON numbers, non-negative dimensions, non-empty v3 codec pipelines, and one -`dimension_names` entry per array dimension. In a v3 document they also read +JSON numbers, non-negative dimensions, and one `dimension_names` entry per +array dimension. In a v3 document they also read each extension point -- the data type, chunk grid, chunk key encoding, each codec and each storage transformer -- through the definition that claims its name in a scope, `CORE_AND_EXTENSIONS` unless a `context` is passed: a @@ -80,9 +80,10 @@ configuration its definition refuses is refused, and a key it does not declare is reported as `unknown_key`. A name nothing in the scope claims is left unjudged, and whether to support it is the consumer's decision. A v3 fill value is judged against the data type it names, by that data type's -definition, and the chunk grid against the shape, by the grid's -definition. The validators do not judge a codec against the array it is -handed. +definition, the chunk grid against the shape, by the grid's definition, +and the codecs as a pipeline: in order, each judged by its definition +against the chunk it is handed. The validators do not read a shard's +inner codecs as a pipeline. ## Scope diff --git a/packages/zarr-metadata/src/zarr_metadata/model/__init__.py b/packages/zarr-metadata/src/zarr_metadata/model/__init__.py index 18eebef5ba..a148392584 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/__init__.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/__init__.py @@ -5,8 +5,9 @@ structure and, in a v3 document, read each extension point (codecs, chunk grids, data types, ...) through the definition that claims its name in a scope, `CORE_AND_EXTENSIONS` unless a `context` is passed, and judge the -fill value against the data type it names and the chunk grid against -the shape. Each document concept gets a +fill value against the data type it names, the chunk grid against +the shape, and the codecs as a pipeline, each against the chunk it is +handed. Each document concept gets a `validate_*` function returning every problem found (a tuple of `ValidationProblem`, each with a machine-readable `kind`), an `is_*` type guard, and a `parse_*` function that narrows or raises diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_validation.py b/packages/zarr-metadata/src/zarr_metadata/model/_validation.py index 65ff07c1dc..4f3b9d4c6b 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_validation.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/_validation.py @@ -5,8 +5,9 @@ each extension point through the definition that claims its name in a scope, so a configuration its definition refuses is refused here too. A name nothing in the scope claims is left unjudged. A v3 fill value is -judged against the data type it names, and the chunk grid against the -shape; a codec against the array it is handed is not judged here. Each concept +judged against the data type it names, the chunk grid against the +shape, and the codecs as a pipeline, each against the chunk it is +handed. Each concept gets a `validate_*` function returning every problem found, an `is_*` type guard, and a `parse_*` function that narrows or raises `MetadataValidationError`. The guards are `TypeGuard`s, @@ -38,17 +39,20 @@ from zarr_metadata.v2.array import ZarrV2ArrayMetadataJSON from zarr_metadata.v2.group import ZarrV2GroupMetadataJSON from zarr_metadata.v3._definition import ( + Chunk, ChunkGridDefinition, ChunkKeyEncodingDefinition, CodecDefinition, DataTypeDefinition, Definition, + Lengths, Resolved, StorageTransformerDefinition, - chunk_grid_problems, + chunk_grid_lengths, fill_value_problems, resolve, ) +from zarr_metadata.v3._pipeline import read_pipeline from zarr_metadata.v3._registry import CORE_AND_EXTENSIONS, Context from zarr_metadata.v3.array import ZarrV3ArrayMetadataJSON from zarr_metadata.v3.group import ZarrV3GroupMetadataJSON @@ -355,12 +359,15 @@ def validate_array_metadata_v3( codec's configuration does not declare. The fill value is judged against the data type as `context` read it -- an `int8` fill value of 300 -- and the chunk grid against the shape: a regular grid with a - chunk length for each of two dimensions, over an array of three. A name - nothing in `context` claims is left unjudged, with any fill value of - it. Unknown top-level keys are allowed (they map - to `extra_fields`); a reader must understand each one that does not - say `must_understand: false`, which the model reports as - `must_understand_fields`. + chunk length for each of two dimensions, over an array of three. The + codecs are read as a pipeline: in order, each judged against the chunk + it is handed -- a `transpose` whose `order` has another number of axes. + A name nothing in `context` claims is left unjudged, with any fill + value of it, and a codec of that name leaves the codec after it + handed a chunk nothing is known of. Unknown top-level keys are + allowed (they map to `extra_fields`); a reader must understand each + one that does not say `must_understand: false`, which the model + reports as `must_understand_fields`. """ if not isinstance(value, Mapping): return (ValidationProblem((), "expected a mapping", "invalid_type"),) @@ -396,23 +403,31 @@ def validate_array_metadata_v3( ) else: problems.extend(_prefix("fill_value", validate_json(doc["fill_value"]))) - # The chunk grid is judged against the shape, once both are read. + # The chunk grid is judged against the shape, once both are read, and + # says the lengths of the chunks the first codec is handed: an entry + # for each dimension of the shape, None where nothing says it. + lengths: Lengths | None = None if shape is None else (None,) * len(shape) if "chunk_grid" in read and shape is not None: - problems.extend(chunk_grid_problems(read["chunk_grid"], shape, ("chunk_grid",))) + lengths, found = chunk_grid_lengths(read["chunk_grid"], shape, ("chunk_grid",)) + problems.extend(found) + listed: dict[str, list[Resolved[Any]]] = {} for key, kind in _EXTENSION_LISTS_V3: if key in doc: entries = doc[key] if not _is_array(entries): problems.append(ValidationProblem((key,), "expected a sequence", "invalid_type")) else: - if key == "codecs" and len(entries) == 0: - problems.append( - ValidationProblem( - ("codecs",), "expected at least one codec", "invalid_value" - ) - ) + listed[key] = [] for index, entry in enumerate(entries): - problems.extend(resolve(entry, kind, context, (key, index))[1]) + resolved, found = resolve(entry, kind, context, (key, index)) + listed[key].append(resolved) + problems.extend(found) + # The codecs are read as a pipeline, the first handed the grid's chunks + # of the array's data type: in order, each judged against the chunk it + # is handed. That holds one array -> bytes codec, so it is not empty. + if "codecs" in listed: + chunk = Chunk(lengths, read.get("data_type")) + problems.extend(read_pipeline(listed["codecs"], chunk, ("codecs",))[1]) if "attributes" in doc: problems.extend(_validate_attributes(doc["attributes"])) if "dimension_names" in doc: diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py b/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py index 262c06654d..fa327bf3f7 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py @@ -41,6 +41,7 @@ Generic, Literal, TypeAlias, + TypeGuard, cast, get_args, get_origin, @@ -93,6 +94,16 @@ def unchanged(configuration: T) -> T: return configuration +def unknown_lengths(configuration: object, nested: object, shape: tuple[int, ...]) -> Lengths: + """The chunk lengths of a grid that says nothing of them: unknown, along each axis of the array.""" + return (None,) * len(shape) + + +def unknown_chunk(*_: object) -> Chunk: + """What an array -> array codec that says nothing of it hands on: a chunk nothing is known of.""" + return Chunk() + + class EmptyConfiguration(TypedDict, closed=True): """The configuration of a definition with nothing to configure, written as its bare name.""" @@ -179,7 +190,7 @@ def judge(self, value: object, loc: Loc = ()) -> tuple[C | None, Problems]: configuration, problems = self.check(value, loc) if configuration is None: return None, problems - refused = _ruled(self, lambda: self.rules(configuration, _nothing_nested()), loc) + refused = ruled(self, lambda: self.rules(configuration, _nothing_nested()), loc) return (configuration if len(refused) == 0 else None), (*problems, *refused) @@ -305,10 +316,19 @@ class ChunkGridDefinition(Definition[C]): configuration, the fields it holds as the scope read them, and the shape, and locates its problems in the configuration. A grid that says nothing of the shape fits every one. + + `chunk_lengths` is what the first codec of the array's pipeline is + handed: the lengths the grid's chunks take along each axis of an + array of a shape it fits -- one for each axis of a regular grid, every + length a rectilinear grid lists. It is asked only of a grid its shape + rules accept. A grid that says nothing of it leaves the lengths along + every axis unknown. """ shape_rules: Callable[[C, Nested, tuple[int, ...]], Iterable[ValidationProblem]] = no_rules """What the spec disallows in this grid over an array of a shape, located in the configuration.""" + chunk_lengths: Callable[[C, Nested, tuple[int, ...]], Lengths] = unknown_lengths + """The lengths its chunks take along each axis of an array of a shape it fits, None where unknown.""" @dataclass(frozen=True, kw_only=True, slots=True) @@ -329,10 +349,33 @@ class ChunkKeyEncodingDefinition(Definition[C]): @dataclass(frozen=True, kw_only=True, slots=True) class CodecDefinition(Definition[C]): - """A codec: what it does to what it is handed, and whether the size of what it gives out is static.""" + """A codec: what it does to what it is handed, and whether the size of what it gives out is static. + + A codec handed an array -- array -> array, array -> bytes -- says what + the spec disallows in it handed a `Chunk`: `chunk_rules`, handed the + configuration, the fields it holds as the scope read them, and the + chunk, and locating its problems in the configuration -- a `bytes` + codec without `endian`, handed a multi-byte data type. An array -> + array codec also says what it hands on: `transition`, the chunk the + next codec is handed, given the one it is handed -- `transpose` + permutes the axes, `cast_value` changes the data type. The two are the + spec's pair: a codec computes what it gives from the shape and data + type it is handed, and "If the decoded_representation_type is not + supported, this algorithm must fail with an error" + (https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v3/core/index.rst#L987-L994). + The transition is asked of every chunk the codec is handed, whatever + its chunk rules found, so it gives only what holds either way: a + `transpose` whose `order` has another number of axes hands on lengths + nothing is known of. A codec that says nothing of what it hands on + hands the next a chunk nothing is known of. + """ kind: CodecKind size: CodecSize + chunk_rules: Callable[[C, Nested, Chunk], Iterable[ValidationProblem]] = no_rules + """What the spec disallows in this codec handed a chunk, located in the configuration.""" + transition: Callable[[C, Nested, Chunk], Chunk] = unknown_chunk + """The chunk the next codec is handed, given the one this array -> array codec is handed.""" def _refusal(self) -> str | None: kind: object = self.kind @@ -341,6 +384,10 @@ def _refusal(self) -> str | None: size: object = self.size if size not in get_args(CodecSize): return f"{self.name!r}: size is one of {get_args(CodecSize)!r}, got {size!r}" + if kind == "bytes_bytes" and self.chunk_rules is not no_rules: + return f"{self.name!r}: chunk_rules, of a bytes -> bytes codec, which is handed bytes" + if kind != "array_array" and self.transition is not unknown_chunk: + return f"{self.name!r}: a transition, of a codec that hands on bytes" return None @@ -548,7 +595,21 @@ def _usable(problems: Sequence[ValidationProblem]) -> bool: return all(found.kind == "unknown_key" for found in problems) -def _ruled( +def asked(definition: Definition[Any], what: str, ask: Callable[[], T], at: Loc) -> T: + """What `ask`, a call of `definition`'s `what`, gives. + + A definition's functions are the extension author's code: an error one + raises says which definition's function raised it, and where it was + reading. + """ + try: + return ask() + except Exception as error: + error.add_note(f"raised by the {what} of {definition.name!r}, reading {at!r}") + raise + + +def ruled( definition: Definition[Any], ask: Callable[[], Iterable[ValidationProblem]], at: Loc ) -> Problems: """What `ask`, a call of `definition`'s rules, finds, located under `at`. @@ -557,11 +618,7 @@ def _ruled( what they declare, and an error one raises says which definition's rules raised it, and where they were reading. """ - try: - found = tuple(cast("Iterable[object]", ask())) - except Exception as error: - error.add_note(f"raised by the rules of {definition.name!r}, reading {at!r}") - raise + found = asked(definition, "rules", lambda: tuple(cast("Iterable[object]", ask())), at) for item in found: if not isinstance(item, ValidationProblem): msg = f"{definition.name!r}: its rules yield ValidationProblem values, got {item!r}" @@ -668,6 +725,66 @@ class Resolved(Generic[D]): """The fields a configuration holds, each as the scope read it, by where it sits in the configuration.""" +Lengths: TypeAlias = tuple[frozenset[int] | None, ...] +"""Per axis, every length chunks take along it -- a set, since a rectilinear grid's differ -- or None where unknown.""" + + +@dataclass(frozen=True, slots=True) +class Chunk: + """What a codec is handed: chunks of some lengths along each axis, of a data type. + + What nothing says is None: the lengths along an axis the grid does not + say, and every part of the chunk handed on by a codec that says nothing + of what it hands on. A data type field the scope did not read is held + as written, and says nothing of the values either. A codec's chunk + rules judge what is known and leave the rest, so a chunk nothing is + known of, `Chunk()`, is refused nothing. + """ + + lengths: Lengths | None = None + """Per axis, the lengths the chunks take along it; None when not even the number of axes is known.""" + data_type: Resolved[DataTypeDefinition[Any]] | None = None + """The data type field of the values, as a scope read it; None when no field says what they are.""" + + def __post_init__(self) -> None: + lengths = cast("object", self.lengths) + if lengths is not None and not _is_lengths(lengths): + msg = f"a chunk's lengths are a frozenset of integers or None per axis, got {lengths!r}" + raise TypeError(msg) + data_type = cast("object", self.data_type) + if data_type is not None and not _is_data_type_field(data_type): + msg = f"a chunk's data type is a data type field a scope read, got {data_type!r}" + raise TypeError(msg) + + @property + def rank(self) -> int | None: + """The number of axes; None when unknown.""" + return None if self.lengths is None else len(self.lengths) + + +def _is_data_type_field(value: object) -> bool: + """Whether `value` is a data type field a scope read: one read as a data type, or by nothing.""" + if not isinstance(value, Resolved): + return False + definition = cast("Resolved[Any]", value).definition + return definition is None or isinstance(definition, DataTypeDefinition) + + +def _is_lengths(value: object) -> TypeGuard[Lengths]: + """Whether `value` is chunk lengths: per axis, a frozenset of integers, or None.""" + if not isinstance(value, tuple): + return False + for axis in cast("tuple[object, ...]", value): + if axis is None: + continue + if not isinstance(axis, frozenset) or not all( + isinstance(length, int) and not isinstance(length, bool) + for length in cast("frozenset[object]", axis) + ): + return False + return True + + def configuration_of(resolved: Resolved[Any], definition: Definition[C]) -> C | None: """The configuration `resolved` holds, typed as `definition` declares it, if `definition` read it. @@ -703,7 +820,7 @@ def fill_value_problems( typed, problems = _fill_value_parser(definition.fill_value)(refined, loc) if not _usable(problems): return problems - refused = _ruled( + refused = ruled( definition, lambda: definition.fill_value_rules(configuration, data_type.nested, typed), loc, @@ -711,26 +828,54 @@ def fill_value_problems( return (*problems, *refused) -def chunk_grid_problems( +def chunk_grid_lengths( chunk_grid: Resolved[ChunkGridDefinition[Any]], shape: tuple[int, ...], loc: Loc = () -) -> Problems: - """What is wrong with `chunk_grid`, a chunk grid field a scope read, over an array of `shape`. - - Judged by the grid's shape rules, which locate their problems in the - configuration, under `loc`, where the field sits: a regular grid with - a chunk length for each of two dimensions, over an array of three. A - grid the scope did not read, out of scope or invalid, is left - unjudged. +) -> tuple[Lengths, Problems]: + """The lengths the chunks of `chunk_grid`, a chunk grid field a scope read, take along each axis of an array of `shape`, and what is wrong with the grid over it. + + A chunk has an extent "for each dimension of the array" + (https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v3/core/index.rst#L277-L281), + so the lengths have an entry for each dimension of `shape`, None where + nothing says them. The grid's shape rules judge it first, and locate + their problems in the configuration, under `loc`, where the field + sits: a regular grid with a chunk length for each of two dimensions, + over an array of three. Only a grid that fits the shape says its + lengths; a grid the scope did not read, out of scope or invalid, is + left unjudged and says none. What a grid's `chunk_lengths` gives is + checked: lengths of another type are a `TypeError`, and of another + number of axes than `shape` has a `ValueError`, each a fault in the + definition, not the field. """ + unknown: Lengths = (None,) * len(shape) definition = chunk_grid.definition configuration = chunk_grid.configuration if definition is None or configuration is None: - return () - return _ruled( + return unknown, () + at = (*loc, "configuration") + problems = ruled( + definition, lambda: definition.shape_rules(configuration, chunk_grid.nested, shape), at + ) + if len(problems) != 0: + return unknown, problems + lengths = asked( definition, - lambda: definition.shape_rules(configuration, chunk_grid.nested, shape), - (*loc, "configuration"), + "chunk lengths", + lambda: cast("object", definition.chunk_lengths(configuration, chunk_grid.nested, shape)), + at, ) + if not _is_lengths(lengths): + msg = ( + f"{definition.name!r}: its chunk_lengths give a frozenset of integers or None " + f"per axis, got {lengths!r}" + ) + raise TypeError(msg) + if len(lengths) != len(shape): + msg = ( + f"{definition.name!r}: its chunk_lengths gave {len(lengths)} axes, " + f"for a shape of {len(shape)}" + ) + raise ValueError(msg) + return lengths, () def resolve( @@ -803,7 +948,7 @@ def _read( inside.extend(_sized(field, inner.definition)) problems = list(found) if configuration is not None: - problems.extend(_ruled(definition, lambda: definition.rules(configuration, within), at)) + problems.extend(ruled(definition, lambda: definition.rules(configuration, within), at)) problems.extend(inside) if not _usable(problems): return Resolved(data, "invalid", definition, None), tuple(problems) @@ -925,6 +1070,7 @@ def _replaced(value: JSONValue, path: Loc, new: JSONValue) -> JSONValue: "KINDS", "RAW_BYTES_NAME", "RAW_BYTES_NAME_PATTERN", + "Chunk", "ChunkGridDefinition", "ChunkGridField", "ChunkKeyEncodingDefinition", @@ -937,6 +1083,7 @@ def _replaced(value: JSONValue, path: Loc, new: JSONValue) -> JSONValue: "DataTypeField", "Definition", "EmptyConfiguration", + "Lengths", "Nested", "Resolution", "Resolved", @@ -945,14 +1092,18 @@ def _replaced(value: JSONValue, path: Loc, new: JSONValue) -> JSONValue: "StorageTransformerField", "Unread", "as_kind", + "asked", "canonicalize", - "chunk_grid_problems", + "chunk_grid_lengths", "configuration_of", "fill_value_problems", "kind_of", "named_configuration", "no_rules", "resolve", + "ruled", "spelled", "unchanged", + "unknown_chunk", + "unknown_lengths", ] diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/_pipeline.py b/packages/zarr-metadata/src/zarr_metadata/v3/_pipeline.py new file mode 100644 index 0000000000..6b588c9694 --- /dev/null +++ b/packages/zarr-metadata/src/zarr_metadata/v3/_pipeline.py @@ -0,0 +1,189 @@ +"""A codec pipeline, read: each codec with the chunk it is handed. + +A pipeline is array -> array codecs, then one array -> bytes codec, then +bytes -> bytes codecs. The array hands its first codec a `Chunk` -- "the +same data type as the Zarr array, and shape equal to the chunk shape" +(https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v3/core/index.rst#L1048-L1050), +the lengths its grid's chunks take along each axis -- and each array -> +array codec hands the next the chunk its `transition` says. +Each codec handed an array is judged by its chunk rules against the chunk +it is handed: a `bytes` codec handed a multi-byte data type without an +`endian`, a `transpose` whose `order` has another number of axes. + +Nothing is guessed. A codec the scope did not read -- nothing in scope +claims it, or it has a problem of its own -- might do anything, so the +codec after it is handed a chunk nothing is known of, as is the codec +after one that says nothing of what it hands on; a codec after that +hands on only what it says of its own accord, as `cast_value` names its +data type. One nothing in scope claims might be of any of the three +kinds, so the order is judged without it. A chunk rule judges what is +known of its chunk and leaves the rest. +""" + +from __future__ import annotations + +from dataclasses import dataclass +from typing import TYPE_CHECKING, Any, Final, cast + +from zarr_metadata._json import ValidationProblem +from zarr_metadata.v3._definition import Chunk, CodecDefinition, CodecKind, Resolved, asked, ruled + +if TYPE_CHECKING: + from collections.abc import Iterator, Mapping, Sequence + + from zarr_metadata._typed_json import Loc + from zarr_metadata.v3._definition import Nested, Problems + + +@dataclass(frozen=True, slots=True) +class Stage: + """One codec of a pipeline, and the chunk it is handed.""" + + codec: Resolved[CodecDefinition[Any]] + """The codec, as the scope read it.""" + incoming: Chunk | None + """The chunk it is handed. + + None for a codec handed bytes -- a bytes -> bytes codec, or any codec + after the array -> bytes codec -- and for a codec nothing in scope + claims when what it is handed is not known to be an array. + """ + + +_POSITIONS: Final[Mapping[CodecKind, int]] = { + "array_array": 0, + "array_bytes": 1, + "bytes_bytes": 2, +} +"""Where each kind of codec comes in a pipeline.""" + +_SPOKEN: Final[Mapping[CodecKind, str]] = { + "array_array": "array -> array", + "array_bytes": "array -> bytes", + "bytes_bytes": "bytes -> bytes", +} + + +def read_pipeline( + codecs: Sequence[Resolved[CodecDefinition[Any]]], chunk: Chunk, loc: Loc = () +) -> tuple[tuple[Stage, ...], Problems]: + """Each of `codecs`, codec fields a scope read, as a pipeline handed `chunk`, with the chunk it is handed, and what is wrong with them. + + The order first: array -> array codecs, then one array -> bytes codec, + then bytes -> bytes codecs. Then each codec in turn, handed the chunk + the one before it handed on: judged by its chunk rules, which locate + their problems in its configuration, and, an array -> array codec, + asked what it hands on. Problems are located under `loc`, where the + codecs sit, each codec at its index. + """ + problems = list(_order_problems(codecs, loc)) + stages: list[Stage] = [] + # What the next codec is handed: None once that is not known to be an + # array -- past the array -> bytes codec, or a codec of unknown kind. + handed: Chunk | None = chunk + for index, codec in enumerate(codecs): + definition, configuration = codec.definition, codec.configuration + if definition is None: + stages.append(Stage(codec, handed)) + handed = None + continue + if definition.kind == "bytes_bytes": + stages.append(Stage(codec, None)) + handed = None + continue + incoming = Chunk() if handed is None else handed + stages.append(Stage(codec, incoming)) + at = (*loc, index, "configuration") + if configuration is not None: + problems.extend(_chunk_problems(definition, configuration, codec.nested, incoming, at)) + if definition.kind == "array_bytes": + handed = None + elif configuration is None: + handed = Chunk() + else: + handed = _handed_on(definition, configuration, codec.nested, incoming, at) + return tuple(stages), tuple(problems) + + +def _order_problems( + codecs: Sequence[Resolved[CodecDefinition[Any]]], loc: Loc +) -> Iterator[ValidationProblem]: + """Array -> array codecs, then one array -> bytes codec, then bytes -> bytes codecs. + + "the list of codecs must be of the following form: zero or more + array -> array codecs; followed by exactly one array -> bytes codec; + followed by zero or more bytes -> bytes codecs" + (https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v3/core/index.rst#L974-L983). + A codec nothing in scope claims is left out: it might be any of the + three, so a pipeline that holds one is not refused for holding no + array -> bytes codec. A codec out of place is reported where it sits, + and so is an array -> bytes codec after the first. + """ + furthest: CodecDefinition[Any] | None = None + encoder: CodecDefinition[Any] | None = None + unclaimed = False + for index, codec in enumerate(codecs): + definition = codec.definition + if definition is None: + unclaimed = True + continue + if furthest is not None and _POSITIONS[definition.kind] < _POSITIONS[furthest.kind]: + yield ValidationProblem( + (*loc, index), + f"expected {_SPOKEN[definition.kind]} codecs before " + f"{_SPOKEN[furthest.kind]} codecs, got {definition.name!r} after " + f"{furthest.name!r}", + "invalid_value", + ) + elif definition.kind == "array_bytes" and encoder is not None: + yield ValidationProblem( + (*loc, index), + f"expected one array -> bytes codec, got {definition.name!r} after " + f"{encoder.name!r}", + "invalid_value", + ) + if furthest is None or _POSITIONS[definition.kind] > _POSITIONS[furthest.kind]: + furthest = definition + if definition.kind == "array_bytes" and encoder is None: + encoder = definition + if encoder is None and not unclaimed: + yield ValidationProblem(loc, "expected an array -> bytes codec, got none", "invalid_value") + + +def _chunk_problems( + definition: CodecDefinition[Any], + configuration: Mapping[str, Any], + nested: Nested, + chunk: Chunk, + at: Loc, +) -> Problems: + """What `definition`'s chunk rules find in a codec handed `chunk`, located under `at`.""" + return ruled(definition, lambda: definition.chunk_rules(configuration, nested, chunk), at) + + +def _handed_on( + definition: CodecDefinition[Any], + configuration: Mapping[str, Any], + nested: Nested, + chunk: Chunk, + at: Loc, +) -> Chunk: + """The chunk an array -> array codec hands on, handed `chunk`. + + Its `transition` is the extension author's code: what it gives is + checked to be a chunk, and an error it raises says which codec's + transition raised it, and where it was reading. + """ + given = asked( + definition, + "transition", + lambda: cast("object", definition.transition(configuration, nested, chunk)), + at, + ) + if not isinstance(given, Chunk): + msg = f"{definition.name!r}: its transition gives a Chunk, got {given!r}" + raise TypeError(msg) + return given + + +__all__ = ["Stage", "read_pipeline"] diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/chunk_grid/rectilinear.py b/packages/zarr-metadata/src/zarr_metadata/v3/chunk_grid/rectilinear.py index 0818e90a5f..a8191727a8 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/chunk_grid/rectilinear.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/chunk_grid/rectilinear.py @@ -11,7 +11,7 @@ from zarr_metadata._json import ValidationProblem from zarr_metadata._typed_json import Loc -from zarr_metadata.v3._definition import ChunkGridDefinition, Nested +from zarr_metadata.v3._definition import ChunkGridDefinition, Lengths, Nested RECTILINEAR_CHUNK_GRID_NAME: Final = "rectilinear" """The `name` field value of the rectilinear chunk grid.""" @@ -156,12 +156,32 @@ def _shape_rules( ) +def _chunk_lengths( + configuration: RectilinearChunkGridConfiguration, nested: Nested, shape: tuple[int, ...] +) -> Lengths: + """Along each axis, every chunk length its entry lists. + + A bare integer is the length of every chunk along its axis; a list + gives each chunk's length, a `[length, count]` pair `count` of them + (https://github.com/zarr-developers/zarr-extensions/blob/4da7b37a84f76e660902f6d3de3eaef0e0febae6/chunk-grids/rectilinear/README.md?plain=1#L62-L91). + Every length listed counts, a chunk past the array's edge too, which + the array grows into when it is resized. + """ + return tuple( + frozenset({spec}) + if isinstance(spec, int) + else frozenset(entry if isinstance(entry, int) else entry[0] for entry in spec) + for spec in configuration["chunk_shapes"] + ) + + RECTILINEAR_CHUNK_GRID: Final = ChunkGridDefinition( name=RECTILINEAR_CHUNK_GRID_NAME, configuration=RectilinearChunkGridConfiguration, rules=_rules, canonical=_canonical, shape_rules=_shape_rules, + chunk_lengths=_chunk_lengths, ) """The `rectilinear` chunk grid.""" diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/chunk_grid/regular.py b/packages/zarr-metadata/src/zarr_metadata/v3/chunk_grid/regular.py index 8ee48ba300..444ec11f4a 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/chunk_grid/regular.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/chunk_grid/regular.py @@ -10,7 +10,7 @@ from typing_extensions import TypedDict from zarr_metadata._json import ValidationProblem -from zarr_metadata.v3._definition import ChunkGridDefinition, Nested +from zarr_metadata.v3._definition import ChunkGridDefinition, Lengths, Nested REGULAR_CHUNK_GRID_NAME: Final = "regular" """The `name` field value of the regular chunk grid.""" @@ -90,11 +90,26 @@ def _shape_rules( ) +def _chunk_lengths( + configuration: RegularChunkGridConfiguration, nested: Nested, shape: tuple[int, ...] +) -> Lengths: + """One length along each axis: every chunk has the grid's `chunk_shape`. + + "each chunk is a hyperrectangle of the same shape" + (https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v3/chunk-grids/regular-grid/index.rst#L28-L29), + a chunk at the array's edge too, where "the grid will overhang the + edge of the array space" + (https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v3/chunk-grids/regular-grid/index.rst#L43-L46). + """ + return tuple(frozenset({length}) for length in configuration["chunk_shape"]) + + REGULAR_CHUNK_GRID: Final = ChunkGridDefinition( name=REGULAR_CHUNK_GRID_NAME, configuration=RegularChunkGridConfiguration, rules=_rules, shape_rules=_shape_rules, + chunk_lengths=_chunk_lengths, ) """The `regular` chunk grid.""" diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/codec/transpose.py b/packages/zarr-metadata/src/zarr_metadata/v3/codec/transpose.py index bb82fc268d..2fb88338f9 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/codec/transpose.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/codec/transpose.py @@ -10,7 +10,7 @@ from typing_extensions import TypedDict from zarr_metadata._json import ValidationProblem -from zarr_metadata.v3._definition import CodecDefinition, Nested +from zarr_metadata.v3._definition import Chunk, CodecDefinition, Nested TRANSPOSE_CODEC_NAME: Final = "transpose" """The `name` field value of the `transpose` codec.""" @@ -53,8 +53,8 @@ def _rules( ) -> Iterator[ValidationProblem]: """`order` permutes its own axes. - Whether it permutes the *array's* axes needs the array's rank, and is - asked where the codec meets the array. + Whether it permutes the axes of the chunk the codec is handed is a + chunk rule. """ order = configuration["order"] if sorted(order) != list(range(len(order))): @@ -65,12 +65,48 @@ def _rules( ) +def _chunk_rules( + configuration: TransposeCodecConfiguration, nested: Nested, chunk: Chunk +) -> Iterator[ValidationProblem]: + """`order` has an entry for each axis of the chunk the codec is handed. + + "a permutation of 0, 1, ..., n-1, where n is the number of dimensions + in the decoded chunk representation provided as input to this codec" + (https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v3/codecs/transpose/index.rst#L63-L66). + """ + order = configuration["order"] + if chunk.rank is not None and len(order) != chunk.rank: + yield ValidationProblem( + ("order",), + f"expected {chunk.rank} entries, one per axis of the chunk the codec is " + f"handed, got {len(order)}", + "invalid_value", + ) + + +def _transition(configuration: TransposeCodecConfiguration, nested: Nested, chunk: Chunk) -> Chunk: + """The chunk's axes in `order`, its data type the same. + + "B_shape[i] = A_shape[order[i]]", where A is the chunk the codec is + handed and B the one it hands on, of "the same data type as A" + (https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v3/codecs/transpose/index.rst#L73-L79). + An order of another number of axes leaves the lengths unknown. + """ + lengths = chunk.lengths + order = configuration["order"] + if lengths is None or len(order) != len(lengths): + return Chunk(None, chunk.data_type) + return Chunk(tuple(lengths[axis] for axis in order), chunk.data_type) + + TRANSPOSE_CODEC: Final = CodecDefinition( name=TRANSPOSE_CODEC_NAME, configuration=TransposeCodecConfiguration, kind="array_array", size="static", rules=_rules, + chunk_rules=_chunk_rules, + transition=_transition, ) """The `transpose` codec.""" diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/definition.py b/packages/zarr-metadata/src/zarr_metadata/v3/definition.py index 24a1a4e903..ed263c84ea 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/definition.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/definition.py @@ -118,9 +118,9 @@ def acme_lz4_rules( `zarr_metadata.model`, reads each extension point of a v3 array document through the definitions in `SCOPE`, and so do the model's `from_json` and `from_key_value`: a fill value is judged against the data type it names, -by that data type's definition, and the chunk grid against the shape, by -the grid's definition. No rule here reads a codec against the array it -is handed. +by that data type's definition, the chunk grid against the shape, by +the grid's definition, and the codecs as a pipeline, each by its +definition against the chunk it is handed. The TypedDict says what a key it does not declare is: with `closed=True`, a problem, as above; with `extra_items=`, a key holding @@ -158,7 +158,27 @@ def acme_lz4_rules( yielding what the spec disallows in a grid of its configuration over an array of a given shape -- a dimension with no chunk length, chunks that fall short of one -- located in the configuration. A grid that says -nothing of the shape fits every one. +nothing of the shape fits every one. It also says the lengths its chunks +take along each axis of an array it fits, `chunk_lengths`: a set per +axis, since a rectilinear grid's chunks differ. `chunk_grid_lengths(grid, +shape)` gives both of a chunk grid field the scope read: an entry for +each dimension of the shape, None where nothing says the lengths. + +A codec is judged against what it is handed. The array hands its first +codec a `Chunk`: the lengths of its grid's chunks along each of the +array's dimensions, and its data type field, with None for what nothing +says. A codec handed an array says what the spec disallows in it handed +a chunk: `chunk_rules`, located in its configuration -- a `transpose` +whose `order` has another number of axes. An array -> array codec says +what it hands the next, whatever its chunk rules found: `transition` -- +`transpose` permutes the axes. `read_pipeline(codecs, chunk)` reads codec +fields the scope read as a pipeline: their order -- array -> array +codecs, one array -> bytes codec, bytes -> bytes codecs -- and then each +against the chunk it is handed, giving each codec's `Stage` with that +chunk. Nothing is guessed: the codec after one the scope did not read, +or after one that says nothing of what it hands on, is handed a chunk +nothing is known of, `Chunk()`, which is refused nothing; a codec after +that hands on only what it says of its own accord. Raw bits are the one data type whose name carries its configuration: a document writes `r` and the size in bits, and `r16` reads as `r*`, as the @@ -183,10 +203,13 @@ def acme_lz4_rules( TypedDict, says nothing of the keys it does not declare, or has a member no checker reads, named down to the TypedDict that holds it; a `name` that is not a string; a member declared as a function -- `rules`, -`canonical`, `fill_value_rules`, `shape_rules` -- that is not one; a data -type's `fill_value` no checker reads; a codec `kind` that is not one of -the three, or a `size` that is not `"static"` or `"dynamic"`; a data type -named as raw bits of one size are written. A scope refuses a definition +`canonical`, `fill_value_rules`, `shape_rules`, `chunk_lengths`, +`chunk_rules`, `transition` -- that is not one; a data type's +`fill_value` no checker reads; a codec `kind` that is not one of the +three, or a `size` that is not `"static"` or `"dynamic"`; chunk rules of +a bytes -> bytes codec, which is handed bytes, or a `transition` of a +codec that hands on bytes; a data type named as raw bits of one size are +written. A scope refuses a definition of no kind. Nothing happens at class creation. """ @@ -195,6 +218,7 @@ def acme_lz4_rules( from zarr_metadata._typed_json import Loc, check from zarr_metadata.v3._common import ZarrV3MetadataFieldJSON from zarr_metadata.v3._definition import ( + Chunk, ChunkGridDefinition, ChunkGridField, ChunkKeyEncodingDefinition, @@ -207,6 +231,7 @@ def acme_lz4_rules( DataTypeField, Definition, EmptyConfiguration, + Lengths, Nested, Resolution, Resolved, @@ -215,15 +240,18 @@ def acme_lz4_rules( StorageTransformerField, Unread, canonicalize, + chunk_grid_lengths, configuration_of, fill_value_problems, resolve, ) +from zarr_metadata.v3._pipeline import Stage, read_pipeline from zarr_metadata.v3._registry import CORE, CORE_AND_EXTENSIONS, Context __all__ = [ "CORE", "CORE_AND_EXTENSIONS", + "Chunk", "ChunkGridDefinition", "ChunkGridField", "ChunkKeyEncodingDefinition", @@ -238,12 +266,14 @@ def acme_lz4_rules( "Definition", "EmptyConfiguration", "JSONValue", + "Lengths", "Loc", "MetadataValidationError", "Nested", "ProblemKind", "Resolution", "Resolved", + "Stage", "StaticCodecField", "StorageTransformerDefinition", "StorageTransformerField", @@ -252,7 +282,9 @@ def acme_lz4_rules( "ZarrV3MetadataFieldJSON", "canonicalize", "check", + "chunk_grid_lengths", "configuration_of", "fill_value_problems", + "read_pipeline", "resolve", ] diff --git a/packages/zarr-metadata/tests/test_public_api.py b/packages/zarr-metadata/tests/test_public_api.py index 7ba9777d78..e729991b6f 100644 --- a/packages/zarr-metadata/tests/test_public_api.py +++ b/packages/zarr-metadata/tests/test_public_api.py @@ -285,14 +285,17 @@ def test_all_is_grouped_and_unique() -> None: "Base64Bytes", "BloscCName", "BloscShuffle", + "Chunk", "CodecKind", "CodecSize", "Context", "Definition", + "Lengths", "Loc", "Nested", "Resolution", "Resolved", + "Stage", "Unread", "CastOutOfRangeMode", "CastRoundingMode", diff --git a/packages/zarr-metadata/tests/v3/test_definitions.py b/packages/zarr-metadata/tests/v3/test_definitions.py index a12323614c..970f4a8346 100644 --- a/packages/zarr-metadata/tests/v3/test_definitions.py +++ b/packages/zarr-metadata/tests/v3/test_definitions.py @@ -746,14 +746,18 @@ def test_error_a_data_type_fill_value_no_checker_reads() -> None: (DataTypeDefinition, "canonical"), (DataTypeDefinition, "fill_value_rules"), (ChunkGridDefinition, "shape_rules"), + (ChunkGridDefinition, "chunk_lengths"), + (CodecDefinition, "chunk_rules"), + (CodecDefinition, "transition"), ], ) def test_error_a_function_member_that_is_not_a_function( kind: type[Definition[Any]], member: str ) -> None: # Each member a definition's annotations declare a `Callable`. + codec = {"kind": "array_array", "size": "static"} if kind is CodecDefinition else {} with pytest.raises(TypeError, match=f"'acme.t': {member} is a function, got 'none'"): - kind(name="acme.t", configuration=Empty, **{member: "none"}) # pyright: ignore[reportArgumentType] + kind(name="acme.t", configuration=Empty, **codec, **{member: "none"}) # pyright: ignore[reportArgumentType] @pytest.mark.parametrize("kind", [Definition, AcmeCodecDefinition]) diff --git a/packages/zarr-metadata/tests/v3/test_grid_shapes.py b/packages/zarr-metadata/tests/v3/test_grid_shapes.py index dc98c18018..6797fcb47e 100644 --- a/packages/zarr-metadata/tests/v3/test_grid_shapes.py +++ b/packages/zarr-metadata/tests/v3/test_grid_shapes.py @@ -1,8 +1,9 @@ """Chunk grids, judged against the shape of the array they chunk. Each grid's definition says what the spec disallows in a grid of its -configuration over an array of a given shape; `chunk_grid_problems` -judges a grid field a scope read, and the v3 array validators judge a +configuration over an array of a given shape, and the lengths its chunks +take along each axis of one it fits; `chunk_grid_lengths` reads both of +a grid field a scope read, and the v3 array validators judge a document's `chunk_grid` against its `shape`. """ @@ -11,12 +12,12 @@ import pytest from zarr_metadata.model import ZarrV3ArrayMetadata, validate_array_metadata_v3 -from zarr_metadata.v3._definition import chunk_grid_problems from zarr_metadata.v3.codec.crc32c import Empty from zarr_metadata.v3.definition import ( CORE_AND_EXTENSIONS, ChunkGridDefinition, JSONValue, + chunk_grid_lengths, resolve, ) @@ -32,37 +33,43 @@ def _rectilinear(*specs: JSONValue) -> JSONValue: def _problems(grid: JSONValue, shape: tuple[int, ...]) -> list[tuple[tuple[str | int, ...], str]]: resolved, found = resolve(grid, ChunkGridDefinition, CORE_AND_EXTENSIONS) assert found == () - return [(problem.loc, problem.kind) for problem in chunk_grid_problems(resolved, shape)] + lengths, problems = chunk_grid_lengths(resolved, shape) + # A grid that does not fit the shape says no lengths along any axis. + assert lengths == (None,) * len(shape) + return [(problem.loc, problem.kind) for problem in problems] @pytest.mark.parametrize( - ("grid", "shape"), + ("grid", "shape", "lengths"), [ - (_regular(), ()), - (_regular(4, 4), (10, 3)), + (_regular(), (), ()), + (_regular(4, 4), (10, 3), ({4}, {4})), # A chunk longer than its dimension, and a chunk length of 0 for a # dimension of length 0. - (_regular(8, 0), (3, 0)), + (_regular(8, 0), (3, 0), ({8}, {0})), # A bare integer repeats until it covers its dimension. - (_rectilinear(4), (10,)), - (_rectilinear([4, 4, 2]), (10,)), - # Overflowing the dimension is allowed. - (_rectilinear([4, 4, 4]), (10,)), - (_rectilinear([[4, 2], 2]), (10,)), - # Nothing to cover. - (_rectilinear([]), (0,)), - (_rectilinear(2, [3, 3]), (7, 6)), + (_rectilinear(4), (10,), ({4},)), + (_rectilinear([4, 4, 2]), (10,), ({4, 2},)), + # Overflowing the dimension is allowed, and the chunk past the edge + # counts. + (_rectilinear([4, 4, 4]), (10,), ({4},)), + (_rectilinear([4, 4, 2, 7]), (10,), ({4, 2, 7},)), + (_rectilinear([[4, 2], 2]), (10,), ({4, 2},)), + # Nothing to cover, and no chunk. + (_rectilinear([]), (0,), (set(),)), + (_rectilinear(2, [3, 3]), (7, 6), ({2}, {3})), # A grid nothing in scope claims, or one that is not read, is left - # unjudged. - ({"name": "acme.grid", "configuration": {"x": 1}}, (3,)), - (_regular(-1), (3,)), + # unjudged, its lengths unknown along each axis of the array. + ({"name": "acme.grid", "configuration": {"x": 1}}, (3, 2), (None, None)), + (_regular(-1), (3,), (None,)), ], ) -def test_every_chunk_grid_fits_the_shapes_it_covers( - grid: JSONValue, shape: tuple[int, ...] +def test_every_chunk_grid_gives_the_lengths_of_its_chunks_over_a_shape_it_fits( + grid: JSONValue, shape: tuple[int, ...], lengths: tuple[set[int] | None, ...] ) -> None: resolved, _ = resolve(grid, ChunkGridDefinition, CORE_AND_EXTENSIONS) - assert chunk_grid_problems(resolved, shape) == () + expected = tuple(None if axis is None else frozenset(axis) for axis in lengths) + assert chunk_grid_lengths(resolved, shape) == (expected, ()) @pytest.mark.parametrize( @@ -92,11 +99,22 @@ def test_error_rectilinear_chunks_that_do_not_cover_their_dimension(spec: JSONVa ] -def test_a_grid_without_shape_rules_fits_every_shape() -> None: +def test_a_grid_that_says_nothing_of_the_shape_fits_every_one_its_lengths_unknown() -> None: lenient = ChunkGridDefinition(name="acme.grid", configuration=Empty) scope = CORE_AND_EXTENSIONS.extended_with(lenient) resolved, _ = resolve("acme.grid", ChunkGridDefinition, scope) - assert chunk_grid_problems(resolved, (3, 4)) == () + assert chunk_grid_lengths(resolved, (3, 4)) == ((None, None), ()) + + +def test_error_a_grid_whose_chunk_lengths_have_another_number_of_axes() -> None: + # A fault in the definition, not the field. + flat = ChunkGridDefinition( + name="acme.grid", configuration=Empty, chunk_lengths=lambda c, n, s: (frozenset({1}),) + ) + scope = CORE_AND_EXTENSIONS.extended_with(flat) + resolved, _ = resolve("acme.grid", ChunkGridDefinition, scope) + with pytest.raises(ValueError, match="'acme.grid': its chunk_lengths gave 1 axes"): + chunk_grid_lengths(resolved, (3, 4)) def test_error_an_array_document_s_chunk_grid_that_does_not_fit_its_shape() -> None: @@ -115,3 +133,28 @@ def test_error_an_array_document_s_shape_that_is_not_read_leaves_its_grid_unjudg assert [(p.loc, p.kind) for p in validate_array_metadata_v3(document)] == [ (("shape",), "invalid_value") ] + + +@pytest.mark.parametrize( + ("chunk_lengths", "match"), + [ + (lambda configuration, nested, shape: (4, 4), "its chunk_lengths give a frozenset"), + (lambda configuration, nested, shape: {}["axis"], "'axis'"), + ], +) +def test_error_a_grid_whose_chunk_lengths_give_something_else_or_raise( + chunk_lengths: object, match: str +) -> None: + # A fault in the definition, and an error raised says whose. + odd = ChunkGridDefinition( + name="acme.grid", + configuration=Empty, + chunk_lengths=chunk_lengths, # pyright: ignore[reportArgumentType] + ) + resolved, _ = resolve("acme.grid", ChunkGridDefinition, CORE_AND_EXTENSIONS.extended_with(odd)) + with pytest.raises((TypeError, KeyError), match=match) as raised: + chunk_grid_lengths(resolved, (3, 4), ("chunk_grid",)) + if isinstance(raised.value, KeyError): + assert raised.value.__notes__ == [ + "raised by the chunk lengths of 'acme.grid', reading ('chunk_grid', 'configuration')" + ] diff --git a/packages/zarr-metadata/tests/v3/test_pipelines.py b/packages/zarr-metadata/tests/v3/test_pipelines.py new file mode 100644 index 0000000000..7aa653ab66 --- /dev/null +++ b/packages/zarr-metadata/tests/v3/test_pipelines.py @@ -0,0 +1,257 @@ +"""Codec pipelines, read: each codec with the chunk it is handed. + +`read_pipeline` judges a pipeline's order, and each codec against the +chunk it is handed; the v3 array validators read a document's `codecs` +as the pipeline its grid's chunks, of its data type, go through. +""" + +from __future__ import annotations + +from typing import TYPE_CHECKING + +import pytest + +from zarr_metadata.model import ZarrV3ArrayMetadata, validate_array_metadata_v3 +from zarr_metadata.v3.codec.crc32c import Empty +from zarr_metadata.v3.definition import ( + CORE_AND_EXTENSIONS, + Chunk, + CodecDefinition, + DataTypeDefinition, + JSONValue, + Nested, + read_pipeline, + resolve, +) + +if TYPE_CHECKING: + from zarr_metadata.v3.definition import Stage, ValidationProblem + + +def _tiled(configuration: Empty, nested: Nested, chunk: Chunk) -> Chunk: + """Each chunk twice as long along every axis, of the same data type.""" + if chunk.lengths is None: + return Chunk(None, chunk.data_type) + doubled = tuple( + None if axis is None else frozenset(2 * length for length in axis) for axis in chunk.lengths + ) + return Chunk(doubled, chunk.data_type) + + +ACME_TILE = CodecDefinition( + name="acme.tile", configuration=Empty, kind="array_array", size="static", transition=_tiled +) +"""An array -> array codec that says what it hands on.""" + +ACME_SHUFFLE = CodecDefinition( + name="acme.shuffle", configuration=Empty, kind="array_array", size="static" +) +"""An array -> array codec that says nothing of what it hands on.""" + +SCOPE = CORE_AND_EXTENSIONS.extended_with(ACME_TILE, ACME_SHUFFLE) + +UINT8 = resolve("uint8", DataTypeDefinition, SCOPE)[0] + +CHUNK = Chunk((frozenset({4}), frozenset({2})), UINT8) +"""Chunks of 4 by 2 `uint8` values.""" + +GZIP: JSONValue = {"name": "gzip", "configuration": {"level": 1}} + + +def _transpose(*order: int) -> JSONValue: + return {"name": "transpose", "configuration": {"order": list(order)}} + + +def _read( + codecs: list[JSONValue], chunk: Chunk +) -> tuple[tuple[Stage, ...], tuple[ValidationProblem, ...]]: + """The stages, and every problem: each codec's own, where it sits, then the pipeline's.""" + read = [resolve(codec, CodecDefinition, SCOPE, (index,)) for index, codec in enumerate(codecs)] + stages, found = read_pipeline([resolved for resolved, _ in read], chunk) + return stages, (*(problem for _, own in read for problem in own), *found) + + +def _problems( + codecs: list[JSONValue], chunk: Chunk = CHUNK +) -> list[tuple[tuple[str | int, ...], str]]: + return [(problem.loc, problem.kind) for problem in _read(codecs, chunk)[1]] + + +def _lengths(*axes: set[int]) -> tuple[frozenset[int], ...]: + return tuple(frozenset(axis) for axis in axes) + + +@pytest.mark.parametrize( + ("codecs", "chunk", "incoming"), + [ + (["bytes"], CHUNK, [CHUNK]), + # Each array -> array codec hands on the chunk its transition says: + # transpose permutes the axes, twice back to where they were. + ( + [_transpose(1, 0), "bytes"], + CHUNK, + [CHUNK, Chunk(_lengths({2}, {4}), UINT8)], + ), + ( + [_transpose(1, 0), _transpose(1, 0), "bytes"], + CHUNK, + [CHUNK, Chunk(_lengths({2}, {4}), UINT8), CHUNK], + ), + (["acme.tile", "bytes"], CHUNK, [CHUNK, Chunk(_lengths({8}, {4}), UINT8)]), + # After the array -> bytes codec, a codec is handed bytes. + (["bytes", GZIP, "crc32c"], CHUNK, [CHUNK, None, None]), + # One that says nothing of what it hands on hands the next a chunk + # nothing is known of. + (["acme.shuffle", "bytes"], CHUNK, [CHUNK, Chunk()]), + # So does one nothing in scope claims, which might be of any kind: + # a pipeline that holds one may hold its array -> bytes codec. + (["zfpy", "bytes"], CHUNK, [CHUNK, Chunk()]), + (["zfpy"], CHUNK, [CHUNK]), + (["bytes", "zfpy"], CHUNK, [CHUNK, None]), + # A chunk nothing is known of stays that way, and is refused nothing. + ([_transpose(1, 0, 2), "bytes"], Chunk(), [Chunk(), Chunk()]), + ], +) +def test_every_pipeline_hands_each_codec_the_chunk_the_one_before_hands_on( + codecs: list[JSONValue], chunk: Chunk, incoming: list[Chunk | None] +) -> None: + stages, problems = _read(codecs, chunk) + assert problems == () + assert [stage.incoming for stage in stages] == incoming + + +@pytest.mark.parametrize( + ("codecs", "index"), + [ + ([GZIP, "bytes"], 1), + (["bytes", _transpose(0, 1)], 1), + (["bytes", "crc32c", _transpose(0, 1)], 2), + (["bytes", GZIP, "bytes"], 2), + ], +) +def test_error_a_codec_out_of_order(codecs: list[JSONValue], index: int) -> None: + # Located where it sits. + assert _problems(codecs) == [((index,), "invalid_value")] + + +def test_error_a_second_array_to_bytes_codec() -> None: + assert _problems(["bytes", "bytes"]) == [((1,), "invalid_value")] + + +@pytest.mark.parametrize("codecs", [[], [GZIP], [_transpose(0, 1)]]) +def test_error_a_pipeline_without_an_array_to_bytes_codec(codecs: list[JSONValue]) -> None: + assert _problems(codecs) == [((), "invalid_value")] + + +def test_error_a_transpose_order_with_another_number_of_axes_than_its_chunk() -> None: + # Located in the configuration of the codec, where it sits. + assert _problems([_transpose(1, 0, 2), "bytes"]) == [ + ((0, "configuration", "order"), "invalid_value") + ] + # Judged against the chunk it is handed, which the codec before it + # made: here, still two axes. + assert _problems([_transpose(1, 0), _transpose(0), "bytes"]) == [ + ((1, "configuration", "order"), "invalid_value") + ] + + +def test_error_a_codec_with_a_problem_of_its_own_hands_on_a_chunk_nothing_is_known_of() -> None: + # Nothing is guessed of what it does, so the codec after it is judged + # against nothing. + stages, problems = _read([_transpose(0, 0), "bytes"], CHUNK) + assert [(problem.loc, problem.kind) for problem in problems] == [ + ((0, "configuration", "order"), "invalid_value") + ] + assert [stage.incoming for stage in stages] == [CHUNK, Chunk()] + + +def test_error_a_transition_that_gives_something_else() -> None: + lying = CodecDefinition( + name="acme.lying", + configuration=Empty, + kind="array_array", + size="static", + transition=lambda configuration, nested, chunk: "a chunk", # pyright: ignore[reportArgumentType] + ) + codec = resolve("acme.lying", CodecDefinition, SCOPE.extended_with(lying))[0] + with pytest.raises(TypeError, match="'acme.lying': its transition gives a Chunk"): + read_pipeline([codec], CHUNK) + + +def test_error_a_transition_that_raises_says_whose_it_is() -> None: + def transition(configuration: Empty, nested: Nested, chunk: Chunk) -> Chunk: + raise KeyError("axis") + + raising = CodecDefinition( + name="acme.raising", + configuration=Empty, + kind="array_array", + size="static", + transition=transition, + ) + codec = resolve("acme.raising", CodecDefinition, SCOPE.extended_with(raising))[0] + with pytest.raises(KeyError) as raised: + read_pipeline([codec], CHUNK, ("codecs",)) + assert raised.value.__notes__ == [ + "raised by the transition of 'acme.raising', reading ('codecs', 0, 'configuration')" + ] + + +@pytest.mark.parametrize( + ("lengths", "data_type", "match"), + [((4, 2), None, "a chunk's lengths are"), (None, "float32", "a chunk's data type is")], +) +def test_error_a_transition_that_builds_a_chunk_of_something_else( + lengths: object, data_type: object, match: str +) -> None: + # A chunk checks what it holds, so the fault is the transition's. + def transition(configuration: Empty, nested: Nested, chunk: Chunk) -> Chunk: + return Chunk(lengths, data_type) # pyright: ignore[reportArgumentType] + + odd = CodecDefinition( + name="acme.odd", + configuration=Empty, + kind="array_array", + size="static", + transition=transition, + ) + codec = resolve("acme.odd", CodecDefinition, SCOPE.extended_with(odd))[0] + with pytest.raises(TypeError, match=match) as raised: + read_pipeline([codec], CHUNK, ("codecs",)) + assert raised.value.__notes__ == [ + "raised by the transition of 'acme.odd', reading ('codecs', 0, 'configuration')" + ] + + +@pytest.mark.parametrize( + ("kind", "member"), + [("bytes_bytes", "chunk_rules"), ("bytes_bytes", "transition"), ("array_bytes", "transition")], +) +def test_error_a_codec_hook_nothing_would_ask(kind: str, member: str) -> None: + # A bytes -> bytes codec is handed bytes, and only an array -> array + # codec hands on a chunk. + hook = {member: lambda configuration, nested, chunk: ()} + with pytest.raises(TypeError, match=f"'acme.x': .*{member.split('_')[0]}"): + CodecDefinition(name="acme.x", configuration=Empty, kind=kind, size="static", **hook) # pyright: ignore[reportArgumentType] + + +def test_error_a_transpose_of_another_rank_than_the_array_under_a_grid_nothing_claims() -> None: + # A chunk has an extent for each dimension of the array, whatever the + # grid says of the lengths. + document = dict(ZarrV3ArrayMetadata.create_default(shape=(4, 4)).to_json()) + document["chunk_grid"] = {"name": "acme.grid", "configuration": {}} + document["codecs"] = [_transpose(0), "bytes"] + assert [(p.loc, p.kind) for p in validate_array_metadata_v3(document)] == [ + (("codecs", 0, "configuration", "order"), "invalid_value") + ] + + +def test_error_an_array_document_s_codecs_out_of_order_or_not_fitting_its_chunks() -> None: + # Located in the document's codecs; the chunks are the grid's, of the + # document's shape. + document = dict(ZarrV3ArrayMetadata.create_default(shape=(4, 4)).to_json()) + document["codecs"] = [_transpose(0), "bytes", "crc32c", _transpose(0, 1)] + assert [(p.loc, p.kind) for p in validate_array_metadata_v3(document)] == [ + (("codecs", 3), "invalid_value"), + (("codecs", 0, "configuration", "order"), "invalid_value"), + ] From 6e52c7a1090f852e543c10ebb2d123c65a69747d Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Sun, 27 Sep 2026 17:52:41 +0200 Subject: [PATCH 10/17] feat(zarr-metadata)!: data types say how their values are stored A data type definition says how its values are stored, `storage`: in single bytes, in several bytes at a time, or each in as many as it needs. `bool`, `int8`, `uint8` and `r8` are made of single bytes; the other core types, and the numpy time types, hold numbers of several bytes; `string` and `bytes` vary. Of raw bits wider than a byte the spec does not say whether a byte order applies, so their storage is unknown. A struct's is its fields', packed together. `storage_of` asks it of a data type field a scope read. Two rules follow. The `bytes` codec takes an `endian` when it is handed numbers of several bytes, and is not handed values that vary in size. A struct's field types are of fixed size. Assisted-by: ClaudeCode:claude-opus-5-5 --- .../src/zarr_metadata/model/_array.py | 4 +- .../src/zarr_metadata/v3/_definition.py | 77 ++++++++++++- .../src/zarr_metadata/v3/codec/bytes.py | 62 ++++++++-- .../zarr_metadata/v3/data_type/_numpy_time.py | 17 ++- .../src/zarr_metadata/v3/data_type/bool.py | 7 +- .../src/zarr_metadata/v3/data_type/bytes.py | 15 ++- .../zarr_metadata/v3/data_type/complex128.py | 3 +- .../zarr_metadata/v3/data_type/complex64.py | 3 +- .../src/zarr_metadata/v3/data_type/float16.py | 3 +- .../src/zarr_metadata/v3/data_type/float32.py | 3 +- .../src/zarr_metadata/v3/data_type/float64.py | 3 +- .../src/zarr_metadata/v3/data_type/int16.py | 3 +- .../src/zarr_metadata/v3/data_type/int32.py | 3 +- .../src/zarr_metadata/v3/data_type/int64.py | 3 +- .../src/zarr_metadata/v3/data_type/int8.py | 3 +- .../v3/data_type/numpy_datetime64.py | 2 + .../v3/data_type/numpy_timedelta64.py | 2 + .../src/zarr_metadata/v3/data_type/raw.py | 14 +++ .../src/zarr_metadata/v3/data_type/string.py | 14 ++- .../src/zarr_metadata/v3/data_type/struct.py | 51 ++++++++- .../src/zarr_metadata/v3/data_type/uint16.py | 3 +- .../src/zarr_metadata/v3/data_type/uint32.py | 3 +- .../src/zarr_metadata/v3/data_type/uint64.py | 3 +- .../src/zarr_metadata/v3/data_type/uint8.py | 3 +- .../src/zarr_metadata/v3/definition.py | 16 ++- .../zarr-metadata/tests/model/test_array.py | 6 +- .../zarr-metadata/tests/test_public_api.py | 1 + .../tests/v3/test_definitions.py | 1 + .../tests/v3/test_every_definition.py | 22 ++++ .../tests/v3/test_fill_values.py | 1 + .../zarr-metadata/tests/v3/test_pipelines.py | 44 ++++++- .../zarr-metadata/tests/v3/test_storage.py | 107 ++++++++++++++++++ 32 files changed, 460 insertions(+), 42 deletions(-) create mode 100644 packages/zarr-metadata/tests/v3/test_storage.py diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_array.py b/packages/zarr-metadata/src/zarr_metadata/model/_array.py index 4d5140eea0..6d3979dba3 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_array.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/_array.py @@ -205,7 +205,9 @@ def create_default(cls, **overrides: Unpack[ZarrV3ArrayMetadataPartial]) -> Zarr together. So is a fill value for an overridden `data_type`: the default `fill_value` is `0`, which a data type whose fill value is not an integer -- `bool`, `string`, a complex or struct type -- - refuses, so pass the two together. + refuses, so pass the two together; and so are its codecs: the + default `bytes` codec has no `endian`, which a data type whose + values take several bytes needs. """ if "shape" in overrides and "chunk_grid" not in overrides: chunk_shape = tuple(max(length, 1) for length in overrides["shape"]) diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py b/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py index fa327bf3f7..e11aae60ac 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py @@ -104,6 +104,35 @@ def unknown_chunk(*_: object) -> Chunk: return Chunk() +StorageClass = Literal["single_byte", "multi_byte", "variable_length"] +"""How a data type's values are stored: in single bytes, in several bytes at a time, or each in as many as it needs. + +A number of several bytes is stored in a byte order, which the `bytes` +codec's `endian` says. A value made of single bytes -- a `uint8`, or a +struct of `int8` fields -- has no byte order, and a value whose size +varies takes a codec of its own. +""" + + +def single_byte(*_: object) -> StorageClass: + """The storage of a data type made of single bytes, which no byte order applies to: `uint8`.""" + return "single_byte" + + +def multi_byte(*_: object) -> StorageClass: + """The storage of a data type holding numbers of several bytes, which a byte order applies to: `int16`.""" + return "multi_byte" + + +def variable_length(*_: object) -> StorageClass: + """The storage of a data type whose values vary in size: `string`.""" + return "variable_length" + + +def unknown_storage(*_: object) -> None: + """The storage of a data type that says nothing of it: unknown.""" + + class EmptyConfiguration(TypedDict, closed=True): """The configuration of a definition with nothing to configure, written as its bare name.""" @@ -277,6 +306,13 @@ class DataTypeDefinition(Definition[C]): scope read them (a struct's field types), and the typed fill value. A data type that says nothing of its fill value takes any JSON. + `storage` says how its values are stored -- in single bytes, in + several bytes at a time, or each in as many as it needs -- which is + what the `bytes` codec asks of the data type it is handed: an + `endian`, for numbers of several bytes. A struct's is its fields', so + it is handed the fields the configuration holds as the scope read + them. A data type that says nothing of it leaves it unknown. + One named as a document writes raw bits of one size -- `r16` -- is refused: that name reads as `r*`, so nothing would ever read it with this definition. @@ -286,6 +322,8 @@ class DataTypeDefinition(Definition[C]): """The JSON shape of a fill value, as an annotation: `Int8FillValue`.""" fill_value_rules: Callable[[C, Nested, Any], Iterable[ValidationProblem]] = no_rules """What the spec disallows in a fill value of that shape, located in it.""" + storage: Callable[[C, Nested], StorageClass | None] = unknown_storage + """How its values are stored, given the configuration and the fields it holds; None when unknown.""" def _refusal(self) -> str | None: if RAW_BYTES_NAME_PATTERN.fullmatch(self.name) is not None: @@ -595,17 +633,18 @@ def _usable(problems: Sequence[ValidationProblem]) -> bool: return all(found.kind == "unknown_key" for found in problems) -def asked(definition: Definition[Any], what: str, ask: Callable[[], T], at: Loc) -> T: +def asked(definition: Definition[Any], what: str, ask: Callable[[], T], at: Loc | None = None) -> T: """What `ask`, a call of `definition`'s `what`, gives. A definition's functions are the extension author's code: an error one raises says which definition's function raised it, and where it was - reading. + reading, when that is known. """ try: return ask() except Exception as error: - error.add_note(f"raised by the {what} of {definition.name!r}, reading {at!r}") + where = "" if at is None else f", reading {at!r}" + error.add_note(f"raised by the {what} of {definition.name!r}{where}") raise @@ -828,6 +867,32 @@ def fill_value_problems( return (*problems, *refused) +def storage_of(data_type: Resolved[DataTypeDefinition[Any]]) -> StorageClass | None: + """How the values of `data_type`, a data type field a scope read, are stored; None when unknown. + + Unknown when the scope did not read it, or its definition does not + say. Its `storage` is the extension author's code: what it gives is + checked to be a storage class, and an error it raises says which data + type's storage raised it. + """ + definition = data_type.definition + configuration = data_type.configuration + if definition is None or configuration is None: + return None + found = asked( + definition, + "storage", + lambda: cast("object", definition.storage(configuration, data_type.nested)), + ) + if found is not None and found not in get_args(StorageClass): + msg = ( + f"{definition.name!r}: its storage gives one of {get_args(StorageClass)!r} or None, " + f"got {found!r}" + ) + raise TypeError(msg) + return cast("StorageClass | None", found) + + def chunk_grid_lengths( chunk_grid: Resolved[ChunkGridDefinition[Any]], shape: tuple[int, ...], loc: Loc = () ) -> tuple[Lengths, Problems]: @@ -1088,6 +1153,7 @@ def _replaced(value: JSONValue, path: Loc, new: JSONValue) -> JSONValue: "Resolution", "Resolved", "StaticCodecField", + "StorageClass", "StorageTransformerDefinition", "StorageTransformerField", "Unread", @@ -1098,12 +1164,17 @@ def _replaced(value: JSONValue, path: Loc, new: JSONValue) -> JSONValue: "configuration_of", "fill_value_problems", "kind_of", + "multi_byte", "named_configuration", "no_rules", "resolve", "ruled", + "single_byte", "spelled", + "storage_of", "unchanged", "unknown_chunk", "unknown_lengths", + "unknown_storage", + "variable_length", ] diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/codec/bytes.py b/packages/zarr-metadata/src/zarr_metadata/v3/codec/bytes.py index c8676c9bd5..c98f1d712d 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/codec/bytes.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/codec/bytes.py @@ -4,11 +4,19 @@ See https://zarr-specs.readthedocs.io/en/latest/v3/codecs/bytes/index.html """ +from collections.abc import Iterator from typing import Final, Literal, NotRequired from typing_extensions import TypedDict -from zarr_metadata.v3._definition import CodecDefinition +from zarr_metadata._json import ValidationProblem +from zarr_metadata.v3._definition import ( + Chunk, + CodecDefinition, + Nested, + named_configuration, + storage_of, +) BYTES_CODEC_NAME: Final = "bytes" """The `name` field value of the `bytes` codec.""" @@ -58,14 +66,54 @@ class BytesCodecObject(TypedDict, closed=True): https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v3/core/index.rst#L1562-L1564 """ + +def _chunk_rules( + configuration: BytesCodecConfiguration, nested: Nested, chunk: Chunk +) -> Iterator[ValidationProblem]: + """An `endian` for a data type whose values take several bytes, and a data type of fixed size. + + `endian` is "Required for data types for which endianness is + applicable" + (https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v3/codecs/bytes/index.rst#L64-L69): + every core type of numbers but `bool`, `int8` and `uint8`, whose + values take one byte and do "not depend on endian" + (https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v3/codecs/bytes/index.rst#L84-L101). + Raw bits `r8` take one byte too; of wider raw bits the spec does not + yet say. + The codec "encodes arrays of fixed-size numeric data types" + (https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v3/codecs/bytes/index.rst#L28-L30), + and a type whose values vary in size takes a codec of its own: + `string` "is only compatible with the vlen-utf8 codec" + (https://github.com/zarr-developers/zarr-extensions/blob/4da7b37a84f76e660902f6d3de3eaef0e0febae6/data-types/string/README.md?plain=1#L25). + Asked of what the data type says of its storage; unknown, it is left + be. + """ + if chunk.data_type is None: + return + storage = storage_of(chunk.data_type) + written, _, _ = named_configuration(chunk.data_type.json) + if storage == "multi_byte" and "endian" not in configuration: + yield ValidationProblem( + ("endian",), + f"expected an endian, since each {written!r} value takes several bytes", + "missing_key", + ) + elif storage == "variable_length": + yield ValidationProblem( + (), + f"expected a data type of fixed size, got {written!r}, whose values vary in size", + "invalid_value", + ) + + BYTES_CODEC: Final = CodecDefinition( - name=BYTES_CODEC_NAME, configuration=BytesCodecConfiguration, kind="array_bytes", size="static" + name=BYTES_CODEC_NAME, + configuration=BytesCodecConfiguration, + kind="array_bytes", + size="static", + chunk_rules=_chunk_rules, ) -"""The `bytes` codec. - -No rule of its own: whether `endian` is required depends on the data type -the codec is handed, which is a question about the array, not the field. -""" +"""The `bytes` codec.""" __all__ = [ "BYTES_CODEC", diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/_numpy_time.py b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/_numpy_time.py index 5d60aa0978..7023495b93 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/_numpy_time.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/_numpy_time.py @@ -2,7 +2,7 @@ Both types' configurations have these two members and the one rule on them, so the rule is written here once and neither sibling imports it -from the other. +from the other. So is how their values are stored. """ from collections.abc import Iterator @@ -11,11 +11,23 @@ from typing_extensions import ReadOnly, TypedDict from zarr_metadata._json import ValidationProblem -from zarr_metadata.v3._definition import Nested +from zarr_metadata.v3._definition import Nested, multi_byte NUMPY_TIME_MAX_SCALE_FACTOR: Final = 2**31 - 1 """The largest `scale_factor` numpy stores: the field is a signed int32.""" +numpy_time_storage: Final = multi_byte +"""Signed 64-bit integers, in the byte order the codecs say. + +Each type "is compatible with any codec that supports arrays of signed +64-bit integers" +(https://github.com/zarr-developers/zarr-extensions/blob/4da7b37a84f76e660902f6d3de3eaef0e0febae6/data-types/numpy.datetime64/README.md?plain=1#L120), +and "the endianness of numpy.datetime64 arrays is determined by the +configuration of the codecs defined in metadata" +(https://github.com/zarr-developers/zarr-extensions/blob/4da7b37a84f76e660902f6d3de3eaef0e0febae6/data-types/numpy.datetime64/README.md?plain=1#L83-L85); +`numpy.timedelta64` says the same. +""" + class NumpyTimeConfiguration(TypedDict): """The members both numpy time types' configurations have, read-only so either type fits.""" @@ -55,4 +67,5 @@ def numpy_time_fill_value_rules( "NumpyTimeConfiguration", "numpy_time_fill_value_rules", "numpy_time_rules", + "numpy_time_storage", ] diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/bool.py b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/bool.py index d42308c0b1..e47eee98c9 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/bool.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/bool.py @@ -6,7 +6,7 @@ from typing import Final, Literal -from zarr_metadata.v3._definition import DataTypeDefinition, EmptyConfiguration +from zarr_metadata.v3._definition import DataTypeDefinition, EmptyConfiguration, single_byte BOOL_DATA_TYPE_NAME: Final = "bool" """The `data_type` value for the `bool` type.""" @@ -19,7 +19,10 @@ BOOL_DATA_TYPE: Final = DataTypeDefinition( - name=BOOL_DATA_TYPE_NAME, configuration=EmptyConfiguration, fill_value=BoolFillValue + name=BOOL_DATA_TYPE_NAME, + configuration=EmptyConfiguration, + fill_value=BoolFillValue, + storage=single_byte, ) """The `bool` data type: a bare name, with nothing to configure.""" diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/bytes.py b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/bytes.py index 06f1f2fbf2..530b88a5c2 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/bytes.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/bytes.py @@ -9,7 +9,12 @@ from typing import Final, Literal, NewType from zarr_metadata._json import ValidationProblem -from zarr_metadata.v3._definition import DataTypeDefinition, EmptyConfiguration, Nested +from zarr_metadata.v3._definition import ( + DataTypeDefinition, + EmptyConfiguration, + Nested, + variable_length, +) from zarr_metadata.v3.data_type._integer import byte_value_problems BYTES_DATA_TYPE_NAME: Final = "bytes" @@ -64,8 +69,14 @@ def _fill_value_rules( configuration=EmptyConfiguration, fill_value=BytesFillValue, fill_value_rules=_fill_value_rules, + storage=variable_length, ) -"""The `bytes` data type: a bare name, with nothing to configure.""" +"""The `bytes` data type: a bare name, with nothing to configure. + +Its values are "variable-length byte strings" +(https://github.com/zarr-developers/zarr-extensions/blob/4da7b37a84f76e660902f6d3de3eaef0e0febae6/data-types/bytes/README.md?plain=1#L3), +each stored in as many bytes as it needs. +""" __all__ = [ diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/complex128.py b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/complex128.py index d447fbf0bc..509b12f30a 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/complex128.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/complex128.py @@ -6,7 +6,7 @@ from typing import Final, Literal -from zarr_metadata.v3._definition import DataTypeDefinition, EmptyConfiguration +from zarr_metadata.v3._definition import DataTypeDefinition, EmptyConfiguration, multi_byte from zarr_metadata.v3.data_type._float import complex_fill_value_rules from zarr_metadata.v3.data_type.float64 import FLOAT64_DATA_TYPE, Float64FillValue @@ -36,6 +36,7 @@ configuration=EmptyConfiguration, fill_value=Complex128FillValue, fill_value_rules=complex_fill_value_rules(FLOAT64_DATA_TYPE.fill_value_rules), + storage=multi_byte, ) """The `complex128` data type: a bare name, with nothing to configure; its fill value a pair of `float64` components.""" diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/complex64.py b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/complex64.py index 7967c4bdca..ceb1576473 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/complex64.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/complex64.py @@ -6,7 +6,7 @@ from typing import Final, Literal -from zarr_metadata.v3._definition import DataTypeDefinition, EmptyConfiguration +from zarr_metadata.v3._definition import DataTypeDefinition, EmptyConfiguration, multi_byte from zarr_metadata.v3.data_type._float import complex_fill_value_rules from zarr_metadata.v3.data_type.float32 import FLOAT32_DATA_TYPE, Float32FillValue @@ -36,6 +36,7 @@ configuration=EmptyConfiguration, fill_value=Complex64FillValue, fill_value_rules=complex_fill_value_rules(FLOAT32_DATA_TYPE.fill_value_rules), + storage=multi_byte, ) """The `complex64` data type: a bare name, with nothing to configure; its fill value a pair of `float32` components.""" diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/float16.py b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/float16.py index 4940969078..33ba182dc4 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/float16.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/float16.py @@ -7,7 +7,7 @@ import re from typing import Final, Literal, NewType -from zarr_metadata.v3._definition import DataTypeDefinition, EmptyConfiguration +from zarr_metadata.v3._definition import DataTypeDefinition, EmptyConfiguration, multi_byte from zarr_metadata.v3.data_type._float import FloatSpecialFillValue, float_fill_value_rules FLOAT16_DATA_TYPE_NAME: Final = "float16" @@ -69,6 +69,7 @@ def hex_float16(value: str) -> HexFloat16: configuration=EmptyConfiguration, fill_value=Float16FillValue, fill_value_rules=float_fill_value_rules("float16", hex_float16), + storage=multi_byte, ) """The `float16` data type: a bare name, with nothing to configure; its fill value a number, a named value or a hex string.""" diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/float32.py b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/float32.py index b7e77b5475..7eb923d581 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/float32.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/float32.py @@ -7,7 +7,7 @@ import re from typing import Final, Literal, NewType -from zarr_metadata.v3._definition import DataTypeDefinition, EmptyConfiguration +from zarr_metadata.v3._definition import DataTypeDefinition, EmptyConfiguration, multi_byte from zarr_metadata.v3.data_type._float import FloatSpecialFillValue, float_fill_value_rules FLOAT32_DATA_TYPE_NAME: Final = "float32" @@ -69,6 +69,7 @@ def hex_float32(value: str) -> HexFloat32: configuration=EmptyConfiguration, fill_value=Float32FillValue, fill_value_rules=float_fill_value_rules("float32", hex_float32), + storage=multi_byte, ) """The `float32` data type: a bare name, with nothing to configure; its fill value a number, a named value or a hex string.""" diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/float64.py b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/float64.py index 7f8deb3903..0363439cfa 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/float64.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/float64.py @@ -7,7 +7,7 @@ import re from typing import Final, Literal, NewType -from zarr_metadata.v3._definition import DataTypeDefinition, EmptyConfiguration +from zarr_metadata.v3._definition import DataTypeDefinition, EmptyConfiguration, multi_byte from zarr_metadata.v3.data_type._float import FloatSpecialFillValue, float_fill_value_rules FLOAT64_DATA_TYPE_NAME: Final = "float64" @@ -70,6 +70,7 @@ def hex_float64(value: str) -> HexFloat64: configuration=EmptyConfiguration, fill_value=Float64FillValue, fill_value_rules=float_fill_value_rules("float64", hex_float64), + storage=multi_byte, ) """The `float64` data type: a bare name, with nothing to configure; its fill value a number, a named value or a hex string.""" diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/int16.py b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/int16.py index a37c5dd1b0..22ff1fd8a4 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/int16.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/int16.py @@ -6,7 +6,7 @@ from typing import Final, Literal -from zarr_metadata.v3._definition import DataTypeDefinition, EmptyConfiguration +from zarr_metadata.v3._definition import DataTypeDefinition, EmptyConfiguration, multi_byte from zarr_metadata.v3.data_type._integer import integer_fill_value_rules INT16_DATA_TYPE_NAME: Final = "int16" @@ -24,6 +24,7 @@ configuration=EmptyConfiguration, fill_value=Int16FillValue, fill_value_rules=integer_fill_value_rules(-(2**15), 2**15 - 1), + storage=multi_byte, ) """The `int16` data type: a bare name, with nothing to configure; its fill value an integer in its range.""" diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/int32.py b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/int32.py index ac25007462..98a039a5d9 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/int32.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/int32.py @@ -6,7 +6,7 @@ from typing import Final, Literal -from zarr_metadata.v3._definition import DataTypeDefinition, EmptyConfiguration +from zarr_metadata.v3._definition import DataTypeDefinition, EmptyConfiguration, multi_byte from zarr_metadata.v3.data_type._integer import integer_fill_value_rules INT32_DATA_TYPE_NAME: Final = "int32" @@ -24,6 +24,7 @@ configuration=EmptyConfiguration, fill_value=Int32FillValue, fill_value_rules=integer_fill_value_rules(-(2**31), 2**31 - 1), + storage=multi_byte, ) """The `int32` data type: a bare name, with nothing to configure; its fill value an integer in its range.""" diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/int64.py b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/int64.py index f57ed8b8a9..b49d5021b7 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/int64.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/int64.py @@ -6,7 +6,7 @@ from typing import Final, Literal -from zarr_metadata.v3._definition import DataTypeDefinition, EmptyConfiguration +from zarr_metadata.v3._definition import DataTypeDefinition, EmptyConfiguration, multi_byte from zarr_metadata.v3.data_type._integer import integer_fill_value_rules INT64_DATA_TYPE_NAME: Final = "int64" @@ -24,6 +24,7 @@ configuration=EmptyConfiguration, fill_value=Int64FillValue, fill_value_rules=integer_fill_value_rules(-(2**63), 2**63 - 1), + storage=multi_byte, ) """The `int64` data type: a bare name, with nothing to configure; its fill value an integer in its range.""" diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/int8.py b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/int8.py index e170ee22fa..7748a491f8 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/int8.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/int8.py @@ -6,7 +6,7 @@ from typing import Final, Literal -from zarr_metadata.v3._definition import DataTypeDefinition, EmptyConfiguration +from zarr_metadata.v3._definition import DataTypeDefinition, EmptyConfiguration, single_byte from zarr_metadata.v3.data_type._integer import integer_fill_value_rules INT8_DATA_TYPE_NAME: Final = "int8" @@ -24,6 +24,7 @@ configuration=EmptyConfiguration, fill_value=Int8FillValue, fill_value_rules=integer_fill_value_rules(-(2**7), 2**7 - 1), + storage=single_byte, ) """The `int8` data type: a bare name, with nothing to configure; its fill value an integer in its range.""" diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/numpy_datetime64.py b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/numpy_datetime64.py index 543050c513..f8b2e21280 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/numpy_datetime64.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/numpy_datetime64.py @@ -12,6 +12,7 @@ from zarr_metadata.v3.data_type._numpy_time import ( numpy_time_fill_value_rules, numpy_time_rules, + numpy_time_storage, ) NUMPY_DATETIME64_DATA_TYPE_NAME: Final = "numpy.datetime64" @@ -63,6 +64,7 @@ class NumpyDatetime64(TypedDict, closed=True): rules=numpy_time_rules, fill_value=NumpyDatetime64FillValue, fill_value_rules=numpy_time_fill_value_rules, + storage=numpy_time_storage, ) """The `numpy.datetime64` data type.""" diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/numpy_timedelta64.py b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/numpy_timedelta64.py index 2a3ac45635..5b0ee23626 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/numpy_timedelta64.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/numpy_timedelta64.py @@ -12,6 +12,7 @@ from zarr_metadata.v3.data_type._numpy_time import ( numpy_time_fill_value_rules, numpy_time_rules, + numpy_time_storage, ) NUMPY_TIMEDELTA64_DATA_TYPE_NAME: Final = "numpy.timedelta64" @@ -82,6 +83,7 @@ class NumpyTimedelta64(TypedDict, closed=True): rules=numpy_time_rules, fill_value=NumpyTimedelta64FillValue, fill_value_rules=numpy_time_fill_value_rules, + storage=numpy_time_storage, ) """The `numpy.timedelta64` data type.""" diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/raw.py b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/raw.py index 23bcf71c03..e23e523f8c 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/raw.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/raw.py @@ -19,6 +19,7 @@ RAW_BYTES_NAME_PATTERN, DataTypeDefinition, Nested, + StorageClass, ) from zarr_metadata.v3.data_type._integer import byte_value_problems @@ -91,12 +92,25 @@ def _fill_value_rules( yield from byte_value_problems(value) +def _storage(configuration: RawBytesConfiguration, nested: Nested) -> StorageClass | None: + """One byte, for `r8`; unknown for wider raw bits, of whose byte order the spec says nothing yet. + + "r8, r16, and r24 should be understood as fall-back types of + respectively 1, 2, and 3 byte length", and of byte order the spec + says only that it looks for "more feedback and prototypes of code + using the r*, raw bits, for various endianness" + (https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v3/core/index.rst#L897-L908). + """ + return "single_byte" if configuration["bits"] == 8 else None + + RAW_BYTES_DATA_TYPE: Final = DataTypeDefinition( name=RAW_BYTES_NAME, configuration=RawBytesConfiguration, rules=_rules, fill_value=RawBytesFillValue, fill_value_rules=_fill_value_rules, + storage=_storage, ) """Raw bits, `r*`: one data type, whose name carries its size. diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/string.py b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/string.py index 0aa246bf15..583160bda7 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/string.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/string.py @@ -6,7 +6,7 @@ from typing import Final, Literal -from zarr_metadata.v3._definition import DataTypeDefinition, EmptyConfiguration +from zarr_metadata.v3._definition import DataTypeDefinition, EmptyConfiguration, variable_length STRING_DATA_TYPE_NAME: Final = "string" """The `data_type` value for the `string` type.""" @@ -19,9 +19,17 @@ STRING_DATA_TYPE: Final = DataTypeDefinition( - name=STRING_DATA_TYPE_NAME, configuration=EmptyConfiguration, fill_value=StringFillValue + name=STRING_DATA_TYPE_NAME, + configuration=EmptyConfiguration, + fill_value=StringFillValue, + storage=variable_length, ) -"""The `string` data type: a bare name, with nothing to configure.""" +"""The `string` data type: a bare name, with nothing to configure. + +Its values are "variable-length UTF8 strings" +(https://github.com/zarr-developers/zarr-extensions/blob/4da7b37a84f76e660902f6d3de3eaef0e0febae6/data-types/string/README.md?plain=1#L3), +each stored in as many bytes as it needs. +""" __all__ = [ diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/struct.py b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/struct.py index 9e9f738a7a..3f3d284204 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/struct.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/struct.py @@ -15,7 +15,10 @@ DataTypeDefinition, DataTypeField, Nested, + StorageClass, fill_value_problems, + named_configuration, + storage_of, ) STRUCT_DATA_TYPE_NAME: Final = "struct" @@ -65,11 +68,14 @@ class Struct(TypedDict, closed=True): def _rules(configuration: StructConfiguration, nested: Nested) -> Iterator[ValidationProblem]: - """Fields exist, and their names are non-empty and distinct. - - A fill value addresses fields by name. Whether each field's type is - fixed-size is a question about that type, asked where types are read - together. + """Fields exist, their names are non-empty and distinct, and their types of fixed size. + + A fill value addresses fields by name. "Variable-length data types + (e.g. "string") MUST NOT be used as field types, as they do not have a + fixed encoded size" + (https://github.com/zarr-developers/zarr-extensions/blob/4da7b37a84f76e660902f6d3de3eaef0e0febae6/data-types/struct/README.md?plain=1#L42-L49): + judged of each field type the scope read, and one whose storage is + unknown is left be. """ fields = configuration["fields"] if len(fields) == 0: @@ -88,6 +94,14 @@ def _rules(configuration: StructConfiguration, nested: Nested) -> Iterator[Valid f"duplicate field name {name!r}, already used by field {first}", "invalid_value", ) + field_type = nested.get(("fields", index, "data_type")) + if field_type is not None and storage_of(field_type) == "variable_length": + written, _, _ = named_configuration(field_type.json) + yield ValidationProblem( + ("fields", index, "data_type"), + f"expected a data type of fixed size, got {written!r}, whose values vary in size", + "invalid_value", + ) def _fill_value_rules( @@ -115,12 +129,39 @@ def _fill_value_rules( yield ValidationProblem((key,), f"no struct field is named {key!r}", "unknown_key") +def _storage(configuration: StructConfiguration, nested: Nested) -> StorageClass | None: + """Its fields' values, packed together: numbers of several bytes if any field holds them, single bytes if every field is made of them. + + A nested struct "is encoded as the packed concatenation of its own + sub-fields, recursively" + (https://github.com/zarr-developers/zarr-extensions/blob/4da7b37a84f76e660902f6d3de3eaef0e0febae6/data-types/struct/README.md?plain=1#L142-L144), + and the `bytes` codec "MUST be configured with an explicit endian" + for a struct that "contains multi-byte numeric fields", while one + "composed entirely of single-byte fields" may go without + (https://github.com/zarr-developers/zarr-extensions/blob/4da7b37a84f76e660902f6d3de3eaef0e0febae6/data-types/struct/README.md?plain=1#L211-L217). + A field whose storage is unknown leaves the struct's unknown, unless + another field's settles it. A struct with a field whose values vary in + size is refused by its rules, so is never asked. + """ + found = { + None if field_type is None else storage_of(field_type) + for field_type in ( + nested.get(("fields", index, "data_type")) + for index in range(len(configuration["fields"])) + ) + } + if "multi_byte" in found: + return "multi_byte" + return "single_byte" if found == {"single_byte"} else None + + STRUCT_DATA_TYPE: Final = DataTypeDefinition( name=STRUCT_DATA_TYPE_NAME, configuration=StructConfiguration, rules=_rules, fill_value=StructFillValue, fill_value_rules=_fill_value_rules, + storage=_storage, ) """The `struct` data type: a record of named fields, each field's type a nested field.""" diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/uint16.py b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/uint16.py index c0c3e865b6..3578ff03fe 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/uint16.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/uint16.py @@ -6,7 +6,7 @@ from typing import Final, Literal -from zarr_metadata.v3._definition import DataTypeDefinition, EmptyConfiguration +from zarr_metadata.v3._definition import DataTypeDefinition, EmptyConfiguration, multi_byte from zarr_metadata.v3.data_type._integer import integer_fill_value_rules UINT16_DATA_TYPE_NAME: Final = "uint16" @@ -24,6 +24,7 @@ configuration=EmptyConfiguration, fill_value=Uint16FillValue, fill_value_rules=integer_fill_value_rules(0, 2**16 - 1), + storage=multi_byte, ) """The `uint16` data type: a bare name, with nothing to configure; its fill value an integer in its range.""" diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/uint32.py b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/uint32.py index 9bf43429a2..771d86804d 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/uint32.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/uint32.py @@ -6,7 +6,7 @@ from typing import Final, Literal -from zarr_metadata.v3._definition import DataTypeDefinition, EmptyConfiguration +from zarr_metadata.v3._definition import DataTypeDefinition, EmptyConfiguration, multi_byte from zarr_metadata.v3.data_type._integer import integer_fill_value_rules UINT32_DATA_TYPE_NAME: Final = "uint32" @@ -24,6 +24,7 @@ configuration=EmptyConfiguration, fill_value=Uint32FillValue, fill_value_rules=integer_fill_value_rules(0, 2**32 - 1), + storage=multi_byte, ) """The `uint32` data type: a bare name, with nothing to configure; its fill value an integer in its range.""" diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/uint64.py b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/uint64.py index 768413de85..4e72073b64 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/uint64.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/uint64.py @@ -6,7 +6,7 @@ from typing import Final, Literal -from zarr_metadata.v3._definition import DataTypeDefinition, EmptyConfiguration +from zarr_metadata.v3._definition import DataTypeDefinition, EmptyConfiguration, multi_byte from zarr_metadata.v3.data_type._integer import integer_fill_value_rules UINT64_DATA_TYPE_NAME: Final = "uint64" @@ -24,6 +24,7 @@ configuration=EmptyConfiguration, fill_value=Uint64FillValue, fill_value_rules=integer_fill_value_rules(0, 2**64 - 1), + storage=multi_byte, ) """The `uint64` data type: a bare name, with nothing to configure; its fill value an integer in its range.""" diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/uint8.py b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/uint8.py index 5279b14523..54ea7c59a7 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/data_type/uint8.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/data_type/uint8.py @@ -6,7 +6,7 @@ from typing import Final, Literal -from zarr_metadata.v3._definition import DataTypeDefinition, EmptyConfiguration +from zarr_metadata.v3._definition import DataTypeDefinition, EmptyConfiguration, single_byte from zarr_metadata.v3.data_type._integer import integer_fill_value_rules UINT8_DATA_TYPE_NAME: Final = "uint8" @@ -24,6 +24,7 @@ configuration=EmptyConfiguration, fill_value=Uint8FillValue, fill_value_rules=integer_fill_value_rules(0, 2**8 - 1), + storage=single_byte, ) """The `uint8` data type: a bare name, with nothing to configure; its fill value an integer in its range.""" diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/definition.py b/packages/zarr-metadata/src/zarr_metadata/v3/definition.py index ed263c84ea..445cd3df76 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/definition.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/definition.py @@ -154,6 +154,14 @@ def acme_lz4_rules( type field the scope read; one nothing in scope claims leaves it unjudged. A data type that says nothing of its fill value takes any JSON. +A data type says how its values are stored, too: `storage`, a function +of its configuration and the fields it holds, giving a `StorageClass` -- +in single bytes, in several bytes at a time, or each in as many as it +needs. A struct's is its fields'. `storage_of(data_type)` asks it of a +data type field the scope read: the `bytes` codec takes an `endian` for +numbers of several bytes, and a struct refuses a field whose values vary +in size. A data type that says nothing of it leaves it unknown. + A chunk grid says which arrays it fits: `shape_rules`, a function yielding what the spec disallows in a grid of its configuration over an array of a given shape -- a dimension with no chunk length, chunks that @@ -203,8 +211,8 @@ def acme_lz4_rules( TypedDict, says nothing of the keys it does not declare, or has a member no checker reads, named down to the TypedDict that holds it; a `name` that is not a string; a member declared as a function -- `rules`, -`canonical`, `fill_value_rules`, `shape_rules`, `chunk_lengths`, -`chunk_rules`, `transition` -- that is not one; a data type's +`canonical`, `fill_value_rules`, `storage`, `shape_rules`, +`chunk_lengths`, `chunk_rules`, `transition` -- that is not one; a data type's `fill_value` no checker reads; a codec `kind` that is not one of the three, or a `size` that is not `"static"` or `"dynamic"`; chunk rules of a bytes -> bytes codec, which is handed bytes, or a `transition` of a @@ -236,6 +244,7 @@ def acme_lz4_rules( Resolution, Resolved, StaticCodecField, + StorageClass, StorageTransformerDefinition, StorageTransformerField, Unread, @@ -244,6 +253,7 @@ def acme_lz4_rules( configuration_of, fill_value_problems, resolve, + storage_of, ) from zarr_metadata.v3._pipeline import Stage, read_pipeline from zarr_metadata.v3._registry import CORE, CORE_AND_EXTENSIONS, Context @@ -275,6 +285,7 @@ def acme_lz4_rules( "Resolved", "Stage", "StaticCodecField", + "StorageClass", "StorageTransformerDefinition", "StorageTransformerField", "Unread", @@ -287,4 +298,5 @@ def acme_lz4_rules( "fill_value_problems", "read_pipeline", "resolve", + "storage_of", ] diff --git a/packages/zarr-metadata/tests/model/test_array.py b/packages/zarr-metadata/tests/model/test_array.py index 17f81fc13f..768a9c0d0c 100644 --- a/packages/zarr-metadata/tests/model/test_array.py +++ b/packages/zarr-metadata/tests/model/test_array.py @@ -189,7 +189,9 @@ def test_string_nan_fill_value_roundtrips() -> None: # unlike a raw float('nan'), which is not JSON. """A float array's string 'NaN' fill_value round-trips cleanly.""" m = ZarrV3ArrayMetadata.create_default( - fill_value="NaN", data_type=ZarrV3NamedConfig(name="float32", configuration={}) + fill_value="NaN", + data_type=ZarrV3NamedConfig(name="float32", configuration={}), + codecs=(ZarrV3NamedConfig(name="bytes", configuration={"endian": "little"}),), ) assert ZarrV3ArrayMetadata.from_json(m.to_json()) == m assert ZarrV3ArrayMetadata.from_json(m.to_json()).fill_value == "NaN" @@ -556,6 +558,7 @@ def test_v3_from_json_reconstructs_required_fields() -> None: shape=(7,), attributes={"a": 1}, data_type=ZarrV3NamedConfig(name="int32", configuration={}), + codecs=(ZarrV3NamedConfig(name="bytes", configuration={"endian": "little"}),), ).to_json() model = ZarrV3ArrayMetadata.from_json(doc) assert model.shape == (7,) @@ -761,6 +764,7 @@ def test_v3_parser_accepts_bare_string_data_type() -> None: """V3 from_json accepts a bare-string data_type and re-serializes it canonically.""" doc = ZarrV3ArrayMetadata.create_default().to_json() doc["data_type"] = "int32" + doc["codecs"] = ({"name": "bytes", "configuration": {"endian": "little"}},) model = ZarrV3ArrayMetadata.from_json(doc) assert model.data_type == ZarrV3NamedConfig(name="int32", configuration={}) assert model.to_json()["data_type"] == "int32" diff --git a/packages/zarr-metadata/tests/test_public_api.py b/packages/zarr-metadata/tests/test_public_api.py index e729991b6f..cb88193103 100644 --- a/packages/zarr-metadata/tests/test_public_api.py +++ b/packages/zarr-metadata/tests/test_public_api.py @@ -296,6 +296,7 @@ def test_all_is_grouped_and_unique() -> None: "Resolution", "Resolved", "Stage", + "StorageClass", "Unread", "CastOutOfRangeMode", "CastRoundingMode", diff --git a/packages/zarr-metadata/tests/v3/test_definitions.py b/packages/zarr-metadata/tests/v3/test_definitions.py index 970f4a8346..65a139ed2f 100644 --- a/packages/zarr-metadata/tests/v3/test_definitions.py +++ b/packages/zarr-metadata/tests/v3/test_definitions.py @@ -745,6 +745,7 @@ def test_error_a_data_type_fill_value_no_checker_reads() -> None: (DataTypeDefinition, "rules"), (DataTypeDefinition, "canonical"), (DataTypeDefinition, "fill_value_rules"), + (DataTypeDefinition, "storage"), (ChunkGridDefinition, "shape_rules"), (ChunkGridDefinition, "chunk_lengths"), (CodecDefinition, "chunk_rules"), diff --git a/packages/zarr-metadata/tests/v3/test_every_definition.py b/packages/zarr-metadata/tests/v3/test_every_definition.py index 3ecd25c473..ac56427712 100644 --- a/packages/zarr-metadata/tests/v3/test_every_definition.py +++ b/packages/zarr-metadata/tests/v3/test_every_definition.py @@ -542,6 +542,28 @@ def test_error_struct_field_name_is_repeated() -> None: ] +@pytest.mark.parametrize( + ("field_type", "at"), + [ + ("string", ()), + ("bytes", ()), + # A struct holding a struct that holds a string is refused where + # the string sits. + ( + {"name": "struct", "configuration": {"fields": [{"name": "s", "data_type": "string"}]}}, + ("configuration", "fields", 0, "data_type"), + ), + ], +) +def test_error_a_struct_field_whose_values_vary_in_size( + field_type: object, at: tuple[str | int, ...] +) -> None: + fields = [{"name": "a", "data_type": "int8"}, {"name": "b", "data_type": field_type}] + assert _one("data_type:struct", {"fields": fields}) == [ + (("configuration", "fields", 1, "data_type", *at), "invalid_value") + ] + + def test_error_struct_field_type_is_judged_where_it_sits() -> None: fields = [ { diff --git a/packages/zarr-metadata/tests/v3/test_fill_values.py b/packages/zarr-metadata/tests/v3/test_fill_values.py index 3b379ff31b..10be671772 100644 --- a/packages/zarr-metadata/tests/v3/test_fill_values.py +++ b/packages/zarr-metadata/tests/v3/test_fill_values.py @@ -274,6 +274,7 @@ def deep(levels: int) -> dict[str, object]: document = dict(ZarrV3ArrayMetadata.create_default().to_json()) | { "data_type": STRUCT, + "codecs": [{"name": "bytes", "configuration": {"endian": "little"}}], "fill_value": {"a": 1, "b": 0.5, "c": deep(600)}, } assert [(p.loc, p.kind) for p in validate_array_metadata_v3(document)] == [ diff --git a/packages/zarr-metadata/tests/v3/test_pipelines.py b/packages/zarr-metadata/tests/v3/test_pipelines.py index 7aa653ab66..1f2e6df334 100644 --- a/packages/zarr-metadata/tests/v3/test_pipelines.py +++ b/packages/zarr-metadata/tests/v3/test_pipelines.py @@ -57,11 +57,25 @@ def _tiled(configuration: Empty, nested: Nested, chunk: Chunk) -> Chunk: GZIP: JSONValue = {"name": "gzip", "configuration": {"level": 1}} +LITTLE: JSONValue = {"name": "bytes", "configuration": {"endian": "little"}} + def _transpose(*order: int) -> JSONValue: return {"name": "transpose", "configuration": {"order": list(order)}} +def _of(data_type: JSONValue) -> Chunk: + """Chunks of 4 by 2 values of `data_type`.""" + return Chunk(CHUNK.lengths, resolve(data_type, DataTypeDefinition, SCOPE)[0]) + + +def _struct(*field_types: JSONValue) -> JSONValue: + fields: list[JSONValue] = [ + {"name": f"f{index}", "data_type": dt} for index, dt in enumerate(field_types) + ] + return {"name": "struct", "configuration": {"fields": fields}} + + def _read( codecs: list[JSONValue], chunk: Chunk ) -> tuple[tuple[Stage, ...], tuple[ValidationProblem, ...]]: @@ -110,6 +124,12 @@ def _lengths(*axes: set[int]) -> tuple[frozenset[int], ...]: (["bytes", "zfpy"], CHUNK, [CHUNK, None]), # A chunk nothing is known of stays that way, and is refused nothing. ([_transpose(1, 0, 2), "bytes"], Chunk(), [Chunk(), Chunk()]), + # `bytes` takes an `endian` for values of several bytes, and any + # for values of one; of wider raw bits the spec says nothing. + ([LITTLE], _of("float32"), [_of("float32")]), + ([LITTLE], _of("uint8"), [_of("uint8")]), + (["bytes"], _of(_struct("int8", "uint8")), [_of(_struct("int8", "uint8"))]), + (["bytes"], _of("r16"), [_of("r16")]), ], ) def test_every_pipeline_hands_each_codec_the_chunk_the_one_before_hands_on( @@ -165,6 +185,26 @@ def test_error_a_codec_with_a_problem_of_its_own_hands_on_a_chunk_nothing_is_kno assert [stage.incoming for stage in stages] == [CHUNK, Chunk()] +@pytest.mark.parametrize( + "data_type", + [ + "float32", + "int16", + _struct("int8", "float32"), + {"name": "numpy.datetime64", "configuration": {"unit": "s", "scale_factor": 1}}, + ], +) +def test_error_a_bytes_codec_without_endian_handed_values_of_several_bytes( + data_type: JSONValue, +) -> None: + assert _problems(["bytes"], _of(data_type)) == [((0, "configuration", "endian"), "missing_key")] + + +@pytest.mark.parametrize("data_type", ["string", "bytes"]) +def test_error_a_bytes_codec_handed_values_that_vary_in_size(data_type: JSONValue) -> None: + assert _problems([LITTLE], _of(data_type)) == [((0, "configuration"), "invalid_value")] + + def test_error_a_transition_that_gives_something_else() -> None: lying = CodecDefinition( name="acme.lying", @@ -248,10 +288,12 @@ def test_error_a_transpose_of_another_rank_than_the_array_under_a_grid_nothing_c def test_error_an_array_document_s_codecs_out_of_order_or_not_fitting_its_chunks() -> None: # Located in the document's codecs; the chunks are the grid's, of the - # document's shape. + # document's shape and data type. document = dict(ZarrV3ArrayMetadata.create_default(shape=(4, 4)).to_json()) + document["data_type"] = "int16" document["codecs"] = [_transpose(0), "bytes", "crc32c", _transpose(0, 1)] assert [(p.loc, p.kind) for p in validate_array_metadata_v3(document)] == [ (("codecs", 3), "invalid_value"), (("codecs", 0, "configuration", "order"), "invalid_value"), + (("codecs", 1, "configuration", "endian"), "missing_key"), ] diff --git a/packages/zarr-metadata/tests/v3/test_storage.py b/packages/zarr-metadata/tests/v3/test_storage.py new file mode 100644 index 0000000000..c8b83ca36f --- /dev/null +++ b/packages/zarr-metadata/tests/v3/test_storage.py @@ -0,0 +1,107 @@ +"""How a data type's values are stored: in single bytes, in several bytes at a time, or each in as many as it needs. + +Each data type's definition says, and `storage_of` asks it of a data type +field a scope read; the `bytes` codec asks it of the data type it is +handed, and a struct of each field's. +""" + +from __future__ import annotations + +import pytest + +from zarr_metadata.v3.codec.crc32c import Empty +from zarr_metadata.v3.definition import ( + CORE_AND_EXTENSIONS, + DataTypeDefinition, + JSONValue, + Nested, + StorageClass, + resolve, + storage_of, +) + + +def _struct(*field_types: JSONValue) -> JSONValue: + fields: list[JSONValue] = [ + {"name": f"f{index}", "data_type": dt} for index, dt in enumerate(field_types) + ] + return {"name": "struct", "configuration": {"fields": fields}} + + +DATETIME: JSONValue = { + "name": "numpy.datetime64", + "configuration": {"unit": "s", "scale_factor": 1}, +} + + +@pytest.mark.parametrize( + ("data_type", "storage"), + [ + ("bool", "single_byte"), + ("int8", "single_byte"), + ("uint8", "single_byte"), + ("r8", "single_byte"), + ("int16", "multi_byte"), + ("uint64", "multi_byte"), + ("float16", "multi_byte"), + ("float64", "multi_byte"), + ("complex64", "multi_byte"), + (DATETIME, "multi_byte"), + ("string", "variable_length"), + ("bytes", "variable_length"), + # Of raw bits wider than a byte, the spec does not say whether a + # byte order applies. + ("r16", None), + # A struct's values are its fields', packed together. + (_struct("int8", "uint8"), "single_byte"), + (_struct("int8", "float32"), "multi_byte"), + (_struct("int8", _struct("uint8", "int16")), "multi_byte"), + # A field of unknown storage leaves the struct's unknown, unless + # another field settles it. + (_struct("int8", "r16"), None), + (_struct("float32", "r16"), "multi_byte"), + (_struct("int8", "acme.decimal"), None), + # A data type nothing in scope claims, or one that is not read. + ("acme.decimal", None), + ({"name": "numpy.datetime64", "configuration": {"unit": "s", "scale_factor": 0}}, None), + ], +) +def test_every_data_type_says_how_its_values_are_stored( + data_type: JSONValue, storage: StorageClass | None +) -> None: + resolved, _ = resolve(data_type, DataTypeDefinition, CORE_AND_EXTENSIONS) + assert storage_of(resolved) == storage + + +def test_a_data_type_that_says_nothing_of_its_storage_leaves_it_unknown() -> None: + quiet = DataTypeDefinition(name="acme.quiet", configuration=Empty) + resolved, _ = resolve( + "acme.quiet", DataTypeDefinition, CORE_AND_EXTENSIONS.extended_with(quiet) + ) + assert storage_of(resolved) is None + + +def test_error_a_storage_that_gives_something_else() -> None: + lying = DataTypeDefinition( + name="acme.lying", + configuration=Empty, + storage=lambda configuration, nested: "two bytes", # pyright: ignore[reportArgumentType] + ) + resolved, _ = resolve( + "acme.lying", DataTypeDefinition, CORE_AND_EXTENSIONS.extended_with(lying) + ) + with pytest.raises(TypeError, match="'acme.lying': its storage gives one of"): + storage_of(resolved) + + +def test_error_a_storage_that_raises_says_whose_it_is() -> None: + def storage(configuration: Empty, nested: Nested) -> StorageClass: + raise KeyError("bits") + + raising = DataTypeDefinition(name="acme.raising", configuration=Empty, storage=storage) + resolved, _ = resolve( + "acme.raising", DataTypeDefinition, CORE_AND_EXTENSIONS.extended_with(raising) + ) + with pytest.raises(KeyError) as raised: + storage_of(resolved) + assert raised.value.__notes__ == ["raised by the storage of 'acme.raising'"] From 77893c909b019858289b1654ede9c3dc8bbf4ebb Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Sun, 27 Sep 2026 17:57:23 +0200 Subject: [PATCH 11/17] feat(zarr-metadata)!: cast_value and scale_offset are judged against the data types they meet `cast_value` casts to a data type that models real numbers, wraps only to an integral one, and is handed one that models real numbers; each scalar of its `scalar_map` is a fill value of the data type on its side of the cast. It hands on its data type, whatever it is handed. `scale_offset` is handed a data type with arithmetic, and its `offset` and `scale` are fill values of it; it hands on what it is handed. Each codec names the data types it takes "defined in this repository", so one it does not name is refused only where the spec's own words refuse it: truth values, raw bits, text, byte strings and records are no numbers, and complex numbers model no real number. The numpy time types say they take any codec of 64-bit integers, and scale_offset does not say whether complex numbers are its, so those are left be. A scalar is judged by the core fill value encoding, which writes a float's infinity `"Infinity"`, where the cast_value codec's own example writes `"+Infinity"`. Assisted-by: ClaudeCode:claude-opus-5-5 --- packages/zarr-metadata/README.md | 3 +- packages/zarr-metadata/docs/index.md | 3 +- .../src/zarr_metadata/v3/codec/_arithmetic.py | 77 +++++++++++ .../src/zarr_metadata/v3/codec/cast_value.py | 129 +++++++++++++++++- .../zarr_metadata/v3/codec/scale_offset.py | 66 ++++++++- .../tests/v3/test_every_definition.py | 42 ++++++ .../zarr-metadata/tests/v3/test_pipelines.py | 125 +++++++++++++++++ 7 files changed, 432 insertions(+), 13 deletions(-) create mode 100644 packages/zarr-metadata/src/zarr_metadata/v3/codec/_arithmetic.py diff --git a/packages/zarr-metadata/README.md b/packages/zarr-metadata/README.md index bc67f170a5..853331f9a1 100644 --- a/packages/zarr-metadata/README.md +++ b/packages/zarr-metadata/README.md @@ -68,7 +68,8 @@ fill value is judged against the data type it names, by that data type's definition, the chunk grid against the shape, by the grid's definition, and the codecs as a pipeline: in order, each judged by its definition against the chunk it is handed. The validators do not read a shard's -inner codecs as a pipeline. +inner codecs as a pipeline, and do none of a codec's arithmetic: whether +a fill value survives a `cast_value` round trip is not judged. The Pydantic integration's generated JSON Schemas express independently checkable document structure and field constraints, but they are not a diff --git a/packages/zarr-metadata/docs/index.md b/packages/zarr-metadata/docs/index.md index f415878091..ffd767162f 100644 --- a/packages/zarr-metadata/docs/index.md +++ b/packages/zarr-metadata/docs/index.md @@ -83,7 +83,8 @@ fill value is judged against the data type it names, by that data type's definition, the chunk grid against the shape, by the grid's definition, and the codecs as a pipeline: in order, each judged by its definition against the chunk it is handed. The validators do not read a shard's -inner codecs as a pipeline. +inner codecs as a pipeline, and do none of a codec's arithmetic: whether +a fill value survives a `cast_value` round trip is not judged. ## Scope diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/codec/_arithmetic.py b/packages/zarr-metadata/src/zarr_metadata/v3/codec/_arithmetic.py new file mode 100644 index 0000000000..47dd9d27f1 --- /dev/null +++ b/packages/zarr-metadata/src/zarr_metadata/v3/codec/_arithmetic.py @@ -0,0 +1,77 @@ +"""The data types the arithmetic codecs, `cast_value` and `scale_offset`, take. + +`cast_value` "is only defined for data types that model real numbers: +floating-point and integral data types" +(https://github.com/zarr-developers/zarr-extensions/blob/4da7b37a84f76e660902f6d3de3eaef0e0febae6/codecs/cast_value/README.md?plain=1#L5-L9), +and `scale_offset` "for data types where multiplication, division, +addition, and subtraction are well-defined" +(https://github.com/zarr-developers/zarr-extensions/blob/4da7b37a84f76e660902f6d3de3eaef0e0febae6/codecs/scale_offset/README.md?plain=1#L5-L9); +each then lists "the following data types defined in this repository", +the same list for both. A data type defined elsewhere may qualify, so one +the list does not name is refused only where the spec's own words refuse +it. + +The numpy time types are refused by neither: each "is compatible with any +codec that supports arrays of signed 64-bit integers" +(https://github.com/zarr-developers/zarr-extensions/blob/4da7b37a84f76e660902f6d3de3eaef0e0febae6/data-types/numpy.datetime64/README.md?plain=1#L120), +which both codecs do, while neither lists them. +""" + +from typing import Any, Final + +from zarr_metadata.v3._definition import RAW_BYTES_NAME, Resolved +from zarr_metadata.v3.data_type.bool import BOOL_DATA_TYPE_NAME +from zarr_metadata.v3.data_type.bytes import BYTES_DATA_TYPE_NAME +from zarr_metadata.v3.data_type.complex64 import COMPLEX64_DATA_TYPE_NAME +from zarr_metadata.v3.data_type.complex128 import COMPLEX128_DATA_TYPE_NAME +from zarr_metadata.v3.data_type.string import STRING_DATA_TYPE_NAME +from zarr_metadata.v3.data_type.struct import STRUCT_DATA_TYPE_NAME + +FLOATING_POINT: Final = frozenset( + { + "float4_e2m1fn", + "float6_e2m3fn", + "float6_e3m2fn", + "float8_e3m4", + "float8_e4m3", + "float8_e4m3b11fnuz", + "float8_e4m3fnuz", + "float8_e5m2", + "float8_e5m2fnuz", + "float8_e8m0fnu", + "bfloat16", + "float16", + "float32", + "float64", + } +) +"""The floating-point data types the lists name: no integral type, which `cast_value` may wrap to.""" + +NOT_NUMBERS: Final = frozenset( + { + BOOL_DATA_TYPE_NAME, + RAW_BYTES_NAME, + STRING_DATA_TYPE_NAME, + BYTES_DATA_TYPE_NAME, + STRUCT_DATA_TYPE_NAME, + } +) +"""Data types whose values are no numbers -- truth values, raw bits, text, byte strings, records -- which neither codec takes.""" + +COMPLEX: Final = frozenset({COMPLEX64_DATA_TYPE_NAME, COMPLEX128_DATA_TYPE_NAME}) +"""Complex numbers, which model no real number, so `cast_value` does not take them. + +Addition, subtraction, multiplication and division are well-defined on +them, so whether `scale_offset` does, its spec leaves open: the list +does not name them. +""" + + +def read_name(data_type: Resolved[Any] | None) -> str | None: + """The name the definition that read `data_type` is filed under; None when no definition read it.""" + if data_type is None or data_type.resolution != "read" or data_type.definition is None: + return None + return data_type.definition.name + + +__all__ = ["COMPLEX", "FLOATING_POINT", "NOT_NUMBERS", "read_name"] diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/codec/cast_value.py b/packages/zarr-metadata/src/zarr_metadata/v3/codec/cast_value.py index a4a2283ed3..d5a018a1e6 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/codec/cast_value.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/codec/cast_value.py @@ -4,12 +4,27 @@ See https://github.com/zarr-developers/zarr-extensions/blob/4da7b37a84f76e660902f6d3de3eaef0e0febae6/codecs/cast_value/README.md """ +from collections.abc import Iterator from typing import Final, Literal, NotRequired from typing_extensions import TypedDict from zarr_metadata._common import JSONValue -from zarr_metadata.v3._definition import CodecDefinition, DataTypeField +from zarr_metadata._json import ValidationProblem +from zarr_metadata.v3._definition import ( + Chunk, + CodecDefinition, + DataTypeField, + Nested, + fill_value_problems, + named_configuration, +) +from zarr_metadata.v3.codec._arithmetic import ( + COMPLEX, + FLOATING_POINT, + NOT_NUMBERS, + read_name, +) CAST_VALUE_CODEC_NAME: Final = "cast_value" """The `name` field value of the `cast_value` codec.""" @@ -50,8 +65,9 @@ ScalarMapEntry = tuple[JSONValue, JSONValue] """A single `[input, output]` mapping in a `scalar_map` direction. -Each scalar is JSON-encoded per its data type's fill-value rules (so -e.g. `"NaN"` and `"+Infinity"` are permitted). +Each scalar is a fill value of its data type: a float's infinities are +`"Infinity"` and `"-Infinity"`, as the core encoding writes them, though +the codec's own example writes `"+Infinity"`. """ @@ -95,17 +111,120 @@ class CastValueCodecObject(TypedDict, closed=True): """ +_NO_REAL_NUMBERS: Final = NOT_NUMBERS | COMPLEX +"""The data types the codec takes neither from nor to: none of them models real numbers.""" + + +def _scalars( + configuration: CastValueCodecConfiguration, side: Literal["source", "target"] +) -> Iterator[tuple[tuple[str | int, ...], JSONValue]]: + """Each scalar of `scalar_map` written in the data type on one side of the cast, with where it sits. + + "For encode, the input data type is the array data type before + casting and the output data type is the target data_type. For decode, + the input data type is the target data_type and the output data type + is the array data type before casting" + (https://github.com/zarr-developers/zarr-extensions/blob/4da7b37a84f76e660902f6d3de3eaef0e0febae6/codecs/cast_value/README.md?plain=1#L107-L109). + """ + scalar_map = configuration.get("scalar_map") + if scalar_map is None: + return + for direction, entries in ( + ("encode", scalar_map.get("encode", ())), + ("decode", scalar_map.get("decode", ())), + ): + at = int((direction == "encode") == (side == "target")) + for index, entry in enumerate(entries): + yield ("scalar_map", direction, index, at), entry[at] + + +def _rules( + configuration: CastValueCodecConfiguration, nested: Nested +) -> Iterator[ValidationProblem]: + """It casts to a data type that models real numbers, `wrap`s only to an integral one, and maps to values of it. + + The codec "is only defined for data types that model real numbers" + (https://github.com/zarr-developers/zarr-extensions/blob/4da7b37a84f76e660902f6d3de3eaef0e0febae6/codecs/cast_value/README.md?plain=1#L5-L9); + `wrap` is "Only permitted when data_type is an integral type" + (https://github.com/zarr-developers/zarr-extensions/blob/4da7b37a84f76e660902f6d3de3eaef0e0febae6/codecs/cast_value/README.md?plain=1#L84); + and a scalar of the target type is encoded with its fill value + encoding. Judged of the data type it casts to, which the scope read; + of the one it is handed, the chunk rules judge the rest. A target that + models no real numbers is the one problem reported of it. + """ + target = nested.get(("data_type",)) + name = read_name(target) + if target is None or name is None: + return + written, _, _ = named_configuration(target.json) + if name in _NO_REAL_NUMBERS: + yield ValidationProblem( + ("data_type",), + f"expected a data type that models real numbers, got {written!r}", + "invalid_value", + ) + return + if configuration.get("out_of_range") == "wrap" and name in FLOATING_POINT: + yield ValidationProblem( + ("out_of_range",), + f"expected an integral data_type to wrap to, got {written!r}", + "invalid_value", + ) + for at, scalar in _scalars(configuration, "target"): + yield from fill_value_problems(target, scalar, at) + + +def _chunk_rules( + configuration: CastValueCodecConfiguration, nested: Nested, chunk: Chunk +) -> Iterator[ValidationProblem]: + """It is handed a data type that models real numbers, and maps from values of it. + + "The same ordered procedure applies during decoding, with input and + output data types swapped" + (https://github.com/zarr-developers/zarr-extensions/blob/4da7b37a84f76e660902f6d3de3eaef0e0febae6/codecs/cast_value/README.md?plain=1#L13), + so the data type it is handed is held to what the one it casts to is. + """ + source = chunk.data_type + name = read_name(source) + if source is None or name is None: + return + if name in _NO_REAL_NUMBERS: + written, _, _ = named_configuration(source.json) + yield ValidationProblem( + (), + f"expected a chunk of a data type that models real numbers, got {written!r}", + "invalid_value", + ) + return + for at, scalar in _scalars(configuration, "source"): + yield from fill_value_problems(source, scalar, at) + + +def _transition(configuration: CastValueCodecConfiguration, nested: Nested, chunk: Chunk) -> Chunk: + """The chunk's values cast to the data type it names, and nothing else changed. + + It "converts (casts) the values of the input array to a new data + type ... and it leaves all other array properties intact" + (https://github.com/zarr-developers/zarr-extensions/blob/4da7b37a84f76e660902f6d3de3eaef0e0febae6/codecs/cast_value/README.md?plain=1#L3). + It hands on the data type field as the scope read it, which says + nothing of the values when the scope did not read it. + """ + return Chunk(chunk.lengths, nested.get(("data_type",))) + + CAST_VALUE_CODEC: Final = CodecDefinition( name=CAST_VALUE_CODEC_NAME, configuration=CastValueCodecConfiguration, kind="array_array", size="static", + rules=_rules, + chunk_rules=_chunk_rules, + transition=_transition, ) """The `cast_value` codec. The data type it casts to is a nested field, read in the scope the codec -is read in. Whether `out_of_range: "wrap"` suits that type is a question -about the type, asked where types are read together. +is read in, and the data type of the chunk it hands on. """ diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/codec/scale_offset.py b/packages/zarr-metadata/src/zarr_metadata/v3/codec/scale_offset.py index d7dba8311d..e4807fe6fe 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/codec/scale_offset.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/codec/scale_offset.py @@ -11,7 +11,14 @@ from zarr_metadata._common import JSONValue from zarr_metadata._json import ValidationProblem -from zarr_metadata.v3._definition import CodecDefinition, Nested +from zarr_metadata.v3._definition import ( + Chunk, + CodecDefinition, + Nested, + fill_value_problems, + named_configuration, +) +from zarr_metadata.v3.codec._arithmetic import NOT_NUMBERS, read_name SCALE_OFFSET_CODEC_NAME: Final = "scale_offset" """The `name` field value of the `scale_offset` codec.""" @@ -26,9 +33,9 @@ class ScaleOffsetCodecConfiguration(TypedDict, closed=True): Both fields are optional. A missing `offset` is the additive identity (e.g. 0 for numeric types); a missing `scale` is the multiplicative - identity (e.g. 1). Each scalar is JSON-encoded per the input array's - fill-value rules, so `"NaN"` and `"+Infinity"` style strings are - permitted in addition to numbers. + identity (e.g. 1). Each scalar is a fill value of the data type the + codec is handed: an integer for an integer type, and for a float type + a number, `"NaN"`, `"Infinity"`, `"-Infinity"` or a hex string. """ offset: NotRequired[JSONValue] @@ -65,8 +72,8 @@ def _rules( The registry says each is "JSON-encoded per the input array's fill-value rules", and no data type admits `null` as a fill value. - Which scalar it must be needs the data type the codec is handed, and - is asked where the codec meets the array. + Which scalar it must be needs the data type the codec is handed: a + chunk rule. """ if "offset" in configuration and configuration["offset"] is None: yield ValidationProblem(("offset",), "expected a scalar, got null", "invalid_value") @@ -74,12 +81,59 @@ def _rules( yield ValidationProblem(("scale",), "expected a scalar, got null", "invalid_value") +def _chunk_rules( + configuration: ScaleOffsetCodecConfiguration, nested: Nested, chunk: Chunk +) -> Iterator[ValidationProblem]: + """It is handed a data type with arithmetic, and `offset` and `scale` are values of it. + + The codec "is only defined for data types where multiplication, + division, addition, and subtraction are well-defined" + (https://github.com/zarr-developers/zarr-extensions/blob/4da7b37a84f76e660902f6d3de3eaef0e0febae6/codecs/scale_offset/README.md?plain=1#L5-L9), + and each of `offset` and `scale` "MUST be encoded to JSON using the + Zarr V3 fill value encoding for the input array's data type" + (https://github.com/zarr-developers/zarr-extensions/blob/4da7b37a84f76e660902f6d3de3eaef0e0febae6/codecs/scale_offset/README.md?plain=1#L39-L43). + A null is the rules' to refuse. + """ + source = chunk.data_type + name = read_name(source) + if source is None or name is None: + return + if name in NOT_NUMBERS: + written, _, _ = named_configuration(source.json) + yield ValidationProblem( + (), + f"expected a chunk of a data type with arithmetic, got {written!r}", + "invalid_value", + ) + return + for member in ("offset", "scale"): + value = configuration.get(member) + if value is not None: + yield from fill_value_problems(source, value, (member,)) + + +def _transition( + configuration: ScaleOffsetCodecConfiguration, nested: Nested, chunk: Chunk +) -> Chunk: + """The chunk as it is handed: the values change, not their shape or data type. + + Its arithmetic is "performed using the arithmetic semantics of the + input array's data type" + (https://github.com/zarr-developers/zarr-extensions/blob/4da7b37a84f76e660902f6d3de3eaef0e0febae6/codecs/scale_offset/README.md?plain=1#L47), + and a narrower data type is a codec after it to make + (https://github.com/zarr-developers/zarr-extensions/blob/4da7b37a84f76e660902f6d3de3eaef0e0febae6/codecs/scale_offset/README.md?plain=1#L153). + """ + return chunk + + SCALE_OFFSET_CODEC: Final = CodecDefinition( name=SCALE_OFFSET_CODEC_NAME, configuration=ScaleOffsetCodecConfiguration, kind="array_array", size="static", rules=_rules, + chunk_rules=_chunk_rules, + transition=_transition, ) """The `scale_offset` codec.""" diff --git a/packages/zarr-metadata/tests/v3/test_every_definition.py b/packages/zarr-metadata/tests/v3/test_every_definition.py index ac56427712..9df43313f0 100644 --- a/packages/zarr-metadata/tests/v3/test_every_definition.py +++ b/packages/zarr-metadata/tests/v3/test_every_definition.py @@ -500,6 +500,48 @@ def test_error_a_codec_inside_a_shard_is_judged_where_it_sits() -> None: ] +@pytest.mark.parametrize( + "configuration", + [ + {"data_type": "bool"}, + {"data_type": "complex64"}, + {"data_type": "string"}, + {"data_type": "r16"}, + # One problem: that it wraps follows from the target. + {"data_type": "bool", "out_of_range": "wrap"}, + ], +) +def test_error_a_cast_to_a_data_type_that_models_no_real_numbers(configuration: object) -> None: + assert _one("codecs:cast_value", configuration) == [ + (("configuration", "data_type"), "invalid_value") + ] + + +def test_error_a_cast_that_wraps_to_a_data_type_that_is_not_integral() -> None: + assert _one("codecs:cast_value", {"data_type": "float32", "out_of_range": "wrap"}) == [ + (("configuration", "out_of_range"), "invalid_value") + ] + + +@pytest.mark.parametrize( + ("data_type", "scalar_map", "loc"), + [ + # The output of encoding and the input of decoding are values of + # the data type cast to. + ("uint8", {"encode": [[0, 300]]}, ("scalar_map", "encode", 0, 1)), + ("uint8", {"decode": [[-1, 0]]}, ("scalar_map", "decode", 0, 0)), + # A float's infinity is "Infinity", as the core encoding writes it. + ("float32", {"decode": [["+Infinity", 0]]}, ("scalar_map", "decode", 0, 0)), + ], +) +def test_error_a_scalar_the_cast_maps_to_that_is_not_of_its_data_type( + data_type: str, scalar_map: object, loc: tuple[str | int, ...] +) -> None: + assert _one("codecs:cast_value", {"data_type": data_type, "scalar_map": scalar_map}) == [ + (("configuration", *loc), "invalid_value") + ] + + def test_error_a_cast_target_is_judged_where_it_sits() -> None: target = {"name": "numpy.datetime64", "configuration": {"unit": "s", "scale_factor": 0}} assert _one("codecs:cast_value", {"data_type": target}) == [ diff --git a/packages/zarr-metadata/tests/v3/test_pipelines.py b/packages/zarr-metadata/tests/v3/test_pipelines.py index 1f2e6df334..ad49e12ce9 100644 --- a/packages/zarr-metadata/tests/v3/test_pipelines.py +++ b/packages/zarr-metadata/tests/v3/test_pipelines.py @@ -64,6 +64,14 @@ def _transpose(*order: int) -> JSONValue: return {"name": "transpose", "configuration": {"order": list(order)}} +def _cast(data_type: JSONValue, **members: JSONValue) -> JSONValue: + return {"name": "cast_value", "configuration": {"data_type": data_type, **members}} + + +def _scale(**members: JSONValue) -> JSONValue: + return {"name": "scale_offset", "configuration": members} + + def _of(data_type: JSONValue) -> Chunk: """Chunks of 4 by 2 values of `data_type`.""" return Chunk(CHUNK.lengths, resolve(data_type, DataTypeDefinition, SCOPE)[0]) @@ -130,6 +138,50 @@ def _lengths(*axes: set[int]) -> tuple[frozenset[int], ...]: ([LITTLE], _of("uint8"), [_of("uint8")]), (["bytes"], _of(_struct("int8", "uint8")), [_of(_struct("int8", "uint8"))]), (["bytes"], _of("r16"), [_of("r16")]), + # cast_value hands on its data type, which the next codec is judged + # against; scale_offset hands on what it is handed. + ([_cast("float32"), LITTLE], _of("int16"), [_of("int16"), _of("float32")]), + ( + [_cast("uint8", out_of_range="wrap"), "bytes"], + _of("float32"), + [_of("float32"), _of("uint8")], + ), + ( + [_cast("float32"), _scale(offset=0.5, scale=2), _cast("int8"), "bytes"], + _of("int8"), + [_of("int8"), _of("float32"), _of("float32"), _of("int8")], + ), + ( + [_cast("uint8", scalar_map={"encode": [["NaN", 0]], "decode": [[0, "NaN"]]}), "bytes"], + _of("float32"), + [_of("float32"), _of("uint8")], + ), + # A cast to a data type nothing in scope claims hands on that field, + # which says nothing of the values. + ([_cast("acme.decimal"), "bytes"], _of("float32"), [_of("float32"), _of("acme.decimal")]), + # After a codec nothing in scope claims, a cast still names the data + # type it hands on; an arithmetic codec handed nothing known is + # refused nothing. + ( + ["zfpy", _cast("float32"), LITTLE], + CHUNK, + [CHUNK, Chunk(), Chunk(None, _of("float32").data_type)], + ), + ([_scale(offset=1), "vlen-utf8"], Chunk(), [Chunk(), Chunk()]), + # Where the specs disagree, or say nothing, a data type is left be: + # the numpy time types say they take any codec of 64-bit integers, + # and scale_offset does not say whether complex numbers are its. + ( + [_cast("int64"), LITTLE], + _of({"name": "numpy.datetime64", "configuration": {"unit": "s", "scale_factor": 1}}), + [ + _of( + {"name": "numpy.datetime64", "configuration": {"unit": "s", "scale_factor": 1}} + ), + _of("int64"), + ], + ), + ([_scale(offset=[1, 0]), LITTLE], _of("complex64"), [_of("complex64"), _of("complex64")]), ], ) def test_every_pipeline_hands_each_codec_the_chunk_the_one_before_hands_on( @@ -200,11 +252,75 @@ def test_error_a_bytes_codec_without_endian_handed_values_of_several_bytes( assert _problems(["bytes"], _of(data_type)) == [((0, "configuration", "endian"), "missing_key")] +def test_error_a_bytes_codec_without_endian_after_a_cast_past_a_codec_nothing_claims() -> None: + # The cast names the data type it hands on, whatever it is handed. + assert _problems(["zfpy", _cast("float32"), "bytes"]) == [ + ((2, "configuration", "endian"), "missing_key") + ] + + @pytest.mark.parametrize("data_type", ["string", "bytes"]) def test_error_a_bytes_codec_handed_values_that_vary_in_size(data_type: JSONValue) -> None: assert _problems([LITTLE], _of(data_type)) == [((0, "configuration"), "invalid_value")] +@pytest.mark.parametrize( + ("scalar_map", "loc"), + [ + # The input of encoding and the output of decoding are values of + # the data type it is handed. + ({"encode": [[300, 0]]}, ("scalar_map", "encode", 0, 0)), + ({"decode": [[0, -129]]}, ("scalar_map", "decode", 0, 1)), + ], +) +def test_error_a_scalar_the_cast_maps_from_that_is_not_of_the_data_type_it_is_handed( + scalar_map: JSONValue, loc: tuple[str | int, ...] +) -> None: + codecs = [_cast("uint8", scalar_map=scalar_map), "bytes"] + assert _problems(codecs, _of("int8")) == [((0, "configuration", *loc), "invalid_value")] + + +@pytest.mark.parametrize("codec", [_cast("float32"), _scale()]) +@pytest.mark.parametrize("data_type", ["bool", "string", _struct("int8")]) +def test_error_an_arithmetic_codec_handed_values_that_are_no_numbers( + codec: JSONValue, data_type: JSONValue +) -> None: + # Before an array -> bytes codec nothing in scope claims, which is + # judged against nothing. + assert _problems([codec, "vlen-utf8"], _of(data_type)) == [ + ((0, "configuration"), "invalid_value") + ] + + +def test_error_a_codec_handed_on_what_another_could_not_take_is_judged_too() -> None: + # scale_offset hands on what it is handed, which the bytes codec after + # it cannot take either. + assert _problems([_scale(), LITTLE], _of("string")) == [ + ((0, "configuration"), "invalid_value"), + ((1, "configuration"), "invalid_value"), + ] + + +def test_error_a_cast_handed_complex_numbers() -> None: + assert _problems([_cast("float32"), "vlen-utf8"], _of("complex64")) == [ + ((0, "configuration"), "invalid_value") + ] + + +@pytest.mark.parametrize( + ("members", "data_type", "found"), + [ + ({"offset": 0.5}, "int16", [((0, "configuration", "offset"), "invalid_type")]), + ({"scale": 70000}, "int16", [((0, "configuration", "scale"), "invalid_value")]), + ({"offset": "nan"}, "float32", [((0, "configuration", "offset"), "invalid_value")]), + ], +) +def test_error_a_scale_or_offset_that_is_not_a_value_of_the_data_type_it_is_handed( + members: dict[str, JSONValue], data_type: str, found: list[tuple[tuple[str | int, ...], str]] +) -> None: + assert _problems([_scale(**members), LITTLE], _of(data_type)) == found + + def test_error_a_transition_that_gives_something_else() -> None: lying = CodecDefinition( name="acme.lying", @@ -275,6 +391,15 @@ def test_error_a_codec_hook_nothing_would_ask(kind: str, member: str) -> None: CodecDefinition(name="acme.x", configuration=Empty, kind=kind, size="static", **hook) # pyright: ignore[reportArgumentType] +def test_an_array_document_s_codecs_are_each_judged_against_the_chunk_they_are_handed() -> None: + # scale_offset's offset is a value of the float32 the first cast hands + # it, not of the array's int8. + document = dict(ZarrV3ArrayMetadata.create_default(shape=(4,)).to_json()) + document["data_type"] = "int8" + document["codecs"] = [_cast("float32"), _scale(offset=0.5), _cast("int8"), "bytes"] + assert validate_array_metadata_v3(document) == () + + def test_error_a_transpose_of_another_rank_than_the_array_under_a_grid_nothing_claims() -> None: # A chunk has an extent for each dimension of the array, whatever the # grid says of the lengths. From f004256cbdb07fa8b3466586f3e4a602b800f484 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Sun, 27 Sep 2026 17:57:47 +0200 Subject: [PATCH 12/17] docs(zarr-metadata): a changelog fragment for the codec pipeline Assisted-by: ClaudeCode:claude-opus-5-5 --- .../changes/+layer-4c.feature.md | 20 +++++++++++++++++++ 1 file changed, 20 insertions(+) create mode 100644 packages/zarr-metadata/changes/+layer-4c.feature.md diff --git a/packages/zarr-metadata/changes/+layer-4c.feature.md b/packages/zarr-metadata/changes/+layer-4c.feature.md new file mode 100644 index 0000000000..69b58fbb77 --- /dev/null +++ b/packages/zarr-metadata/changes/+layer-4c.feature.md @@ -0,0 +1,20 @@ +**Breaking:** a v3 array's `codecs` are read as a pipeline: in order -- +array -> array codecs, then one array -> bytes codec, then bytes -> bytes +codecs -- and each codec judged against the chunk it is handed, which is +the grid's chunks of the array's data type, as the array -> array codecs +before it hand them on. A codec out of order, a second array -> bytes +codec or none at all; a `bytes` codec without an `endian` handed numbers +of several bytes, or handed values that vary in size; a `transpose` whose +`order` has another number of axes than its chunk; a `cast_value` to or +from a data type that models no real numbers, wrapping to one that is +not integral, or mapping a scalar that is not a fill value of the data +type on its side; a `scale_offset` handed values that are no numbers, or +with an `offset` or `scale` that is not a value of its data type; and a +struct field whose values vary in size are each a problem where they +sit, where the package accepted them before. `read_pipeline(codecs, +chunk)` gives each codec with the chunk it is handed, +`chunk_grid_lengths(grid, shape)` the lengths a grid's chunks take, and +`storage_of(data_type)` how a data type's values are stored. A +definition's `rules` are handed the fields its configuration holds as +the scope read them, and the kinds gain `chunk_lengths` (grids), +`storage` (data types), and `chunk_rules` and `transition` (codecs). From 025de4fe1daeb3c701580423b8bba096ec1e48af Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Sun, 27 Sep 2026 18:43:47 +0200 Subject: [PATCH 13/17] fix(zarr-metadata): a field whose configuration is not an object is still claimed by its name `resolve` read a field whose configuration was not an object as claimed by nothing, although its name names a definition, so the codec order was judged without it: `[{"name": "gzip", "configuration": 5}, "bytes"]` drew no problem for the gzip before the bytes codec. The field keeps the definition its name names, unread. Assisted-by: ClaudeCode:claude-opus-5-5 --- .../zarr-metadata/src/zarr_metadata/v3/_definition.py | 6 +++++- packages/zarr-metadata/tests/v3/test_definitions.py | 3 ++- packages/zarr-metadata/tests/v3/test_pipelines.py | 8 ++++++++ 3 files changed, 15 insertions(+), 2 deletions(-) diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py b/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py index e11aae60ac..3ce2731680 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py @@ -985,9 +985,13 @@ def _read( data: JSONValue, kind: type[Definition[Any]], context: Context, loc: Loc ) -> tuple[Resolved[Definition[Any]], Problems]: name, given, malformed = named_configuration(data) - if name is None or len(malformed) != 0: + if name is None: return Resolved(data, "invalid", None, None), () definition = context.claimant(kind, name) + if len(malformed) != 0: + # A configuration that is not an object, which the envelope's + # problems say; the name still says what claims the field. + return Resolved(data, "invalid", definition, None), () if definition is None: return Resolved(data, "out_of_scope", None, None), () _, carried = spelled(kind, name) diff --git a/packages/zarr-metadata/tests/v3/test_definitions.py b/packages/zarr-metadata/tests/v3/test_definitions.py index 65a139ed2f..65761f3156 100644 --- a/packages/zarr-metadata/tests/v3/test_definitions.py +++ b/packages/zarr-metadata/tests/v3/test_definitions.py @@ -394,8 +394,9 @@ def test_error_a_required_configuration_is_missing() -> None: def test_error_the_configuration_is_not_an_object() -> None: + # Unread, and still claimed by the definition its name names. resolved, found = resolve({"name": "gzip", "configuration": 5}, CodecDefinition, SCOPE) - assert resolved.resolution == "invalid" + assert (resolved.resolution, resolved.definition) == ("invalid", GZIP_CODEC) assert [found.loc for found in found] == [("configuration",)] diff --git a/packages/zarr-metadata/tests/v3/test_pipelines.py b/packages/zarr-metadata/tests/v3/test_pipelines.py index ad49e12ce9..ece311bc1c 100644 --- a/packages/zarr-metadata/tests/v3/test_pipelines.py +++ b/packages/zarr-metadata/tests/v3/test_pipelines.py @@ -206,6 +206,14 @@ def test_error_a_codec_out_of_order(codecs: list[JSONValue], index: int) -> None assert _problems(codecs) == [((index,), "invalid_value")] +def test_error_a_codec_out_of_order_whose_configuration_is_not_an_object() -> None: + # Its name still says what it is. + assert _problems([{"name": "gzip", "configuration": 5}, "bytes"]) == [ + ((0, "configuration"), "invalid_type"), + ((1,), "invalid_value"), + ] + + def test_error_a_second_array_to_bytes_codec() -> None: assert _problems(["bytes", "bytes"]) == [((1,), "invalid_value")] From fb7064a906a4d0b3ea706886bd8e6a2744814c3e Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Sun, 27 Sep 2026 18:47:00 +0200 Subject: [PATCH 14/17] docs(zarr-metadata): the pipeline fragment takes this PR's number Assisted-by: ClaudeCode:claude-opus-5-5 --- .../changes/{+layer-4c.feature.md => 4442.feature.md} | 0 1 file changed, 0 insertions(+), 0 deletions(-) rename packages/zarr-metadata/changes/{+layer-4c.feature.md => 4442.feature.md} (100%) diff --git a/packages/zarr-metadata/changes/+layer-4c.feature.md b/packages/zarr-metadata/changes/4442.feature.md similarity index 100% rename from packages/zarr-metadata/changes/+layer-4c.feature.md rename to packages/zarr-metadata/changes/4442.feature.md From 1932fe5203d895daa910fd0e3ede0d5f015b2ebb Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Sun, 27 Sep 2026 19:33:27 +0200 Subject: [PATCH 15/17] feat(zarr-metadata)!: a problem of a field held inside is that field's own `resolve` read a field as invalid when anything inside it had a problem: a codec a shard holds with a gzip `level` out of range, or a `must_understand` of false on one, left the shard unread, so nothing else about it was judged. A document's fields are not read that way: a problem with one codec leaves the others read. Now the fields a configuration holds are read the same way. A problem of one is its own, reported where it sits with its own resolution in `nested`, and the field holding it is read when its own configuration and rules are sound. Its rules are still asked only when each field it holds is named, since a rule may read one by name. Assisted-by: ClaudeCode:claude-opus-5-5 --- .../changes/+layer-4d-fields.feature.md | 5 +++ .../src/zarr_metadata/v3/_definition.py | 34 ++++++++++++------- .../src/zarr_metadata/v3/definition.py | 20 ++++++----- .../tests/v3/test_definitions.py | 31 +++++++++++++++-- 4 files changed, 67 insertions(+), 23 deletions(-) create mode 100644 packages/zarr-metadata/changes/+layer-4d-fields.feature.md diff --git a/packages/zarr-metadata/changes/+layer-4d-fields.feature.md b/packages/zarr-metadata/changes/+layer-4d-fields.feature.md new file mode 100644 index 0000000000..f51e88bc2d --- /dev/null +++ b/packages/zarr-metadata/changes/+layer-4d-fields.feature.md @@ -0,0 +1,5 @@ +**Breaking:** a problem of a field a configuration holds is that field's +own, reported where it sits, and the field holding it is read when its +own configuration and rules are sound -- as a document's fields are. A +codec a shard holds with a `level` out of range, or a `must_understand` +of false, no longer hides the shard's other problems. diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py b/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py index 3ce2731680..ed571f05b3 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py @@ -741,7 +741,8 @@ class Resolved(Generic[D]): definition that claims the name, claimed by nothing, or not readable. A problem with the envelope around it -- a stray member, a `must_understand` of `false` -- is reported with the field, and leaves - the resolution as it is. + the resolution as it is; so is a problem of a field the configuration + holds, which is that field's own, with its own resolution in `nested`. """ json: JSONValue @@ -954,7 +955,9 @@ def resolve( a problem. The name is related to a definition in `context`; the configuration is checked against its TypedDict and judged by its rules; each nested field the check met is read the same way, in the - same scope. A name nothing claims is `out_of_scope`: an unmodelled + same scope, and what is wrong with one is its own, reported where it + sits, as with a document's fields. A name nothing claims is + `out_of_scope`: an unmodelled extension, left unjudged, which is what keeps the format open. `loc` prefixes every problem. `kind` is one of `KINDS`, with or without type arguments; anything else is a `TypeError`. @@ -1002,26 +1005,33 @@ def _read( missing = problem(at, f"{name!r} requires a configuration", "missing_key") return Resolved(data, "invalid", definition, None), missing typed, found, nested = _checked(definition.configuration, {} if given is None else given, at) - envelopes = [_envelope(field) for field in nested] - sound = _usable(found) and all(_usable(envelope) for envelope in envelopes) + # The rules may read a field the configuration holds by its name, so + # they are asked only when each one is named; any other problem with + # one is its own, reported where it sits, as a document's fields are. + sound = _usable(found) and all(_named(field) for field in nested) configuration = cast("Mapping[str, JSONValue]", typed) if sound else None # The fields it holds are read first, so the rules see them as the # scope read them; their problems are reported after the rules'. within: dict[Loc, Resolved[Any]] = {} inside: list[ValidationProblem] = [] - for field, envelope in zip(nested, envelopes, strict=True): - inside.extend(envelope) + for field in nested: + inside.extend(_envelope(field)) inner, found_inside = _read(field.json, field.kind, context, field.loc) within[field.loc[len(at) :]] = inner inside.extend(found_inside) inside.extend(_sized(field, inner.definition)) - problems = list(found) + own = list(found) if configuration is not None: - problems.extend(ruled(definition, lambda: definition.rules(configuration, within), at)) - problems.extend(inside) - if not _usable(problems): - return Resolved(data, "invalid", definition, None), tuple(problems) - return Resolved(data, "read", definition, configuration, within), tuple(problems) + own.extend(ruled(definition, lambda: definition.rules(configuration, within), at)) + if configuration is None or not _usable(own): + return Resolved(data, "invalid", definition, None), (*own, *inside) + return Resolved(data, "read", definition, configuration, within), (*own, *inside) + + +def _named(field: _NestedField) -> bool: + """Whether a field a configuration holds is named, with an object for its configuration if it has one.""" + name, _, malformed = named_configuration(field.json) + return name is not None and len(malformed) == 0 def _read_carried( diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/definition.py b/packages/zarr-metadata/src/zarr_metadata/v3/definition.py index 445cd3df76..daff576631 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/definition.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/definition.py @@ -133,15 +133,17 @@ def acme_lz4_rules( A member holding another metadata field is annotated with the field alias of its kind -- a shard's `codecs: tuple[CodecField, ...]` -- and read in -the scope its field is read in. A member that takes codecs of static size -only is annotated `StaticCodecField` -- a shard's `index_codecs`, since a -reader finds the index by a size it knows before reading it -- and a codec -of dynamic size there is a problem at its place, where the field is read -in a scope; a name nothing claims is left unjudged, its size unknown with -the rest of it. `ZarrV3MetadataFieldJSON` is the same -JSON, but checks as JSON and nothing more, so a definition refuses a -member typed with it. An extension with nothing to configure takes -`EmptyConfiguration`, and is written as its bare name. +the scope its field is read in. What is wrong with the field it holds is +that field's own, reported where it sits: the field holding it is still +read, as a document holding it would be. A member that takes codecs of +static size only is annotated `StaticCodecField` -- a shard's +`index_codecs`, since a reader finds the index by a size it knows before +reading it -- and a codec of dynamic size there is a problem at its +place, where the field is read in a scope; a name nothing claims is left +unjudged, its size unknown with the rest of it. `ZarrV3MetadataFieldJSON` +is the same JSON, but checks as JSON and nothing more, so a definition +refuses a member typed with it. An extension with nothing to configure +takes `EmptyConfiguration`, and is written as its bare name. A data type also says what its fill value is: `fill_value`, the JSON shape of one as an annotation the checker reads -- `Int8FillValue` -- and diff --git a/packages/zarr-metadata/tests/v3/test_definitions.py b/packages/zarr-metadata/tests/v3/test_definitions.py index 65761f3156..bc06f78791 100644 --- a/packages/zarr-metadata/tests/v3/test_definitions.py +++ b/packages/zarr-metadata/tests/v3/test_definitions.py @@ -520,12 +520,15 @@ def test_error_a_regular_grid_extent_is_negative() -> None: def test_error_a_nested_field_is_judged_where_it_sits() -> None: + # Its problem is its own: the field holding it is read, as a document + # holding it would be. field = { "name": "acme.stack", "configuration": {"codecs": ["crc32c", {"name": "gzip", "configuration": {"level": 12}}]}, } resolved, found = resolve(field, CodecDefinition, SCOPE) - assert resolved.resolution == "invalid" + assert resolved.resolution == "read" + assert resolved.nested[("codecs", 1)].resolution == "invalid" assert _locs(found) == [ (("configuration", "codecs", 1, "configuration", "level"), "invalid_value") ] @@ -537,10 +540,34 @@ def test_error_a_nested_field_in_extra_items_is_judged_where_it_sits() -> None: "configuration": {"slow": {"name": "gzip", "configuration": {"level": 12}}}, } resolved, found = resolve(field, CodecDefinition, SCOPE) - assert resolved.resolution == "invalid" + assert resolved.resolution == "read" + assert resolved.nested[("slow",)].resolution == "invalid" assert _locs(found) == [(("configuration", "slow", "configuration", "level"), "invalid_value")] +def test_error_a_nested_envelope_s_problem_is_its_own_and_the_rules_are_asked() -> None: + # A `must_understand` of false inside, as in a document, is reported + # where it sits, and the stack is read. + unread = {"name": "crc32c", "must_understand": False} + resolved, found = resolve( + {"name": "acme.stack", "configuration": {"codecs": [unread]}}, CodecDefinition, SCOPE + ) + assert resolved.resolution == "read" + assert _locs(found) == [(("configuration", "codecs", 0, "must_understand"), "invalid_value")] + # Its rules are asked all the same: one refuses the array -> bytes + # codec it holds, which is the stack's own problem. + resolved, found = resolve( + {"name": "acme.stack", "configuration": {"codecs": [unread, "bytes"]}}, + CodecDefinition, + SCOPE, + ) + assert resolved.resolution == "invalid" + assert _locs(found) == [ + (("configuration", "codecs", 1), "invalid_value"), + (("configuration", "codecs", 0, "must_understand"), "invalid_value"), + ] + + def test_error_a_container_rule_is_not_asked_of_a_malformed_nested_field() -> None: # The stack's rule reads each nested field's name; one with no name is # reported where it sits, and the rule is not asked, as `judge` would not. From 4963e998ad546bf5874793e7af2198fcc32f2e7e Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Sun, 27 Sep 2026 18:53:47 +0200 Subject: [PATCH 16/17] feat(zarr-metadata)!: a shard is judged against the chunk it is handed, and its pipelines are read The sharding codec's `chunk_shape` has an inner chunk length for each axis of the shard it is handed, dividing each length the shards take along it. The shard is the chunk the codec is handed, so a transposed shard is divided along its transposed axes. A codec definition says what the pipelines it holds are handed, `pipelines`, by the member of its configuration that holds each: the shard's inner codecs are handed its inner chunks, of its data type, and its index codecs the shard index, of `uint64` with an axis more than the shard. Each is read as the array's pipeline is, its problems where it sits, and a `Stage` keeps its stages as `inner`. Like the transition, `pipelines` gives only what holds whatever the chunk rules found: a shard of another number of axes than its `chunk_shape` hands both pipelines lengths nothing is known of. The functions no codec of a kind is asked are one declared table. Assisted-by: ClaudeCode:claude-opus-5-5 --- packages/zarr-metadata/README.md | 6 +- .../changes/+layer-4d.feature.md | 13 + packages/zarr-metadata/docs/index.md | 6 +- .../src/zarr_metadata/model/_validation.py | 4 +- .../src/zarr_metadata/v3/_definition.py | 32 ++- .../src/zarr_metadata/v3/_pipeline.py | 101 +++++++- .../v3/codec/sharding_indexed.py | 99 +++++++- .../src/zarr_metadata/v3/definition.py | 24 +- .../tests/v3/test_definitions.py | 1 + .../zarr-metadata/tests/v3/test_pipelines.py | 55 +++- .../zarr-metadata/tests/v3/test_sharding.py | 234 ++++++++++++++++++ 11 files changed, 542 insertions(+), 33 deletions(-) create mode 100644 packages/zarr-metadata/changes/+layer-4d.feature.md create mode 100644 packages/zarr-metadata/tests/v3/test_sharding.py diff --git a/packages/zarr-metadata/README.md b/packages/zarr-metadata/README.md index 853331f9a1..1febfe80c3 100644 --- a/packages/zarr-metadata/README.md +++ b/packages/zarr-metadata/README.md @@ -67,9 +67,9 @@ left unjudged, and whether to support it is the consumer's decision. A v3 fill value is judged against the data type it names, by that data type's definition, the chunk grid against the shape, by the grid's definition, and the codecs as a pipeline: in order, each judged by its definition -against the chunk it is handed. The validators do not read a shard's -inner codecs as a pipeline, and do none of a codec's arithmetic: whether -a fill value survives a `cast_value` round trip is not judged. +against the chunk it is handed, a shard's inner and index codecs too. +The validators do no arithmetic on values: whether a fill value survives +a `cast_value` round trip is not judged. The Pydantic integration's generated JSON Schemas express independently checkable document structure and field constraints, but they are not a diff --git a/packages/zarr-metadata/changes/+layer-4d.feature.md b/packages/zarr-metadata/changes/+layer-4d.feature.md new file mode 100644 index 0000000000..db1618e3ec --- /dev/null +++ b/packages/zarr-metadata/changes/+layer-4d.feature.md @@ -0,0 +1,13 @@ +**Breaking:** a v3 array's sharding codec is judged against the shard it +is handed, and its two pipelines are read as the array's is. Its +`chunk_shape` has an inner chunk length for each axis of the shard, +dividing each length the shards take along it -- the shard being the +chunk the codec is handed, so a transposed shard is divided along its +transposed axes. Its inner codecs are handed the inner chunks, of the +shard's data type, and its index codecs the shard index, of `uint64`, +with an axis more than the shard. A shard that does not divide, or a +pipeline that does not fit what it is handed -- an index `bytes` codec +without an `endian` -- has a problem where it sits, where the package +accepted it before. A codec definition +says what the pipelines it holds are handed, `pipelines`, and each +codec's `Stage` keeps the stages of those pipelines as `inner`. diff --git a/packages/zarr-metadata/docs/index.md b/packages/zarr-metadata/docs/index.md index ffd767162f..d0fe4a7fec 100644 --- a/packages/zarr-metadata/docs/index.md +++ b/packages/zarr-metadata/docs/index.md @@ -82,9 +82,9 @@ left unjudged, and whether to support it is the consumer's decision. A v3 fill value is judged against the data type it names, by that data type's definition, the chunk grid against the shape, by the grid's definition, and the codecs as a pipeline: in order, each judged by its definition -against the chunk it is handed. The validators do not read a shard's -inner codecs as a pipeline, and do none of a codec's arithmetic: whether -a fill value survives a `cast_value` round trip is not judged. +against the chunk it is handed, a shard's inner and index codecs too. +The validators do no arithmetic on values: whether a fill value survives +a `cast_value` round trip is not judged. ## Scope diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_validation.py b/packages/zarr-metadata/src/zarr_metadata/model/_validation.py index 4f3b9d4c6b..d1ae10b83a 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_validation.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/_validation.py @@ -361,7 +361,9 @@ def validate_array_metadata_v3( 300 -- and the chunk grid against the shape: a regular grid with a chunk length for each of two dimensions, over an array of three. The codecs are read as a pipeline: in order, each judged against the chunk - it is handed -- a `transpose` whose `order` has another number of axes. + it is handed -- a `transpose` whose `order` has another number of + axes, a shard its inner chunks do not divide -- and a shard's inner + and index codecs too. A name nothing in `context` claims is left unjudged, with any fill value of it, and a codec of that name leaves the codec after it handed a chunk nothing is known of. Unknown top-level keys are diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py b/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py index ed571f05b3..c18a0bd260 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py @@ -104,6 +104,11 @@ def unknown_chunk(*_: object) -> Chunk: return Chunk() +def no_pipelines(*_: object) -> Mapping[str, Chunk]: + """The pipelines of a codec that holds none: none.""" + return {} + + StorageClass = Literal["single_byte", "multi_byte", "variable_length"] """How a data type's values are stored: in single bytes, in several bytes at a time, or each in as many as it needs. @@ -384,6 +389,13 @@ class ChunkKeyEncodingDefinition(Definition[C]): four bytes. `dynamic`: it depends on the values -- every compressor. """ +_UNASKED: Final[Mapping[CodecKind, tuple[str, ...]]] = { + "array_array": (), + "array_bytes": ("transition",), + "bytes_bytes": ("chunk_rules", "transition", "pipelines"), +} +"""The functions no codec of a kind is asked: a bytes -> bytes codec is handed bytes, and only an array -> array codec hands on a chunk.""" + @dataclass(frozen=True, kw_only=True, slots=True) class CodecDefinition(Definition[C]): @@ -406,6 +418,15 @@ class CodecDefinition(Definition[C]): `transpose` whose `order` has another number of axes hands on lengths nothing is known of. A codec that says nothing of what it hands on hands the next a chunk nothing is known of. + + A codec that holds pipelines of its own says what each is handed: + `pipelines`, by the member of its configuration that holds each, the + chunk its first codec is handed, given the chunk the codec is handed + -- a shard's inner codecs are handed its inner chunks, and its index + codecs the shard index. Like the transition, it is asked whatever the + chunk rules found, and gives only what holds either way. A function + no codec of its kind is asked -- the chunk rules of a bytes -> bytes + codec, which is handed bytes -- is refused. """ kind: CodecKind @@ -414,6 +435,8 @@ class CodecDefinition(Definition[C]): """What the spec disallows in this codec handed a chunk, located in the configuration.""" transition: Callable[[C, Nested, Chunk], Chunk] = unknown_chunk """The chunk the next codec is handed, given the one this array -> array codec is handed.""" + pipelines: Callable[[C, Nested, Chunk], Mapping[str, Chunk]] = no_pipelines + """The pipelines it holds, by the member of its configuration that holds each, and the chunk each is handed.""" def _refusal(self) -> str | None: kind: object = self.kind @@ -422,10 +445,10 @@ def _refusal(self) -> str | None: size: object = self.size if size not in get_args(CodecSize): return f"{self.name!r}: size is one of {get_args(CodecSize)!r}, got {size!r}" - if kind == "bytes_bytes" and self.chunk_rules is not no_rules: - return f"{self.name!r}: chunk_rules, of a bytes -> bytes codec, which is handed bytes" - if kind != "array_array" and self.transition is not unknown_chunk: - return f"{self.name!r}: a transition, of a codec that hands on bytes" + defaults = {member.name: member.default for member in dataclasses.fields(CodecDefinition)} + for member in _UNASKED[self.kind]: + if getattr(self, member) is not defaults[member]: + return f"{self.name!r}: {member}, which no codec of kind {kind!r} is asked" return None @@ -1180,6 +1203,7 @@ def _replaced(value: JSONValue, path: Loc, new: JSONValue) -> JSONValue: "kind_of", "multi_byte", "named_configuration", + "no_pipelines", "no_rules", "resolve", "ruled", diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/_pipeline.py b/packages/zarr-metadata/src/zarr_metadata/v3/_pipeline.py index 6b588c9694..a36017dc73 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/_pipeline.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/_pipeline.py @@ -8,7 +8,10 @@ array codec hands the next the chunk its `transition` says. Each codec handed an array is judged by its chunk rules against the chunk it is handed: a `bytes` codec handed a multi-byte data type without an -`endian`, a `transpose` whose `order` has another number of axes. +`endian`, a `transpose` whose `order` has another number of axes. A codec +that holds pipelines of its own has each read the same way, handed the +chunk its `pipelines` says: a shard's inner codecs its inner chunks, its +index codecs the shard index. Nothing is guessed. A codec the scope did not read -- nothing in scope claims it, or it has a problem of its own -- might do anything, so the @@ -22,19 +25,26 @@ from __future__ import annotations +import dataclasses +from collections.abc import Mapping from dataclasses import dataclass -from typing import TYPE_CHECKING, Any, Final, cast +from typing import TYPE_CHECKING, Any, Final, TypeGuard, cast from zarr_metadata._json import ValidationProblem from zarr_metadata.v3._definition import Chunk, CodecDefinition, CodecKind, Resolved, asked, ruled if TYPE_CHECKING: - from collections.abc import Iterator, Mapping, Sequence + from collections.abc import Iterator, Sequence from zarr_metadata._typed_json import Loc from zarr_metadata.v3._definition import Nested, Problems +def _no_stages() -> Mapping[str, tuple[Stage, ...]]: + """The pipelines of a codec that holds none, or was not read: none.""" + return {} + + @dataclass(frozen=True, slots=True) class Stage: """One codec of a pipeline, and the chunk it is handed.""" @@ -48,6 +58,8 @@ class Stage: after the array -> bytes codec -- and for a codec nothing in scope claims when what it is handed is not known to be an array. """ + inner: Mapping[str, tuple[Stage, ...]] = dataclasses.field(default_factory=_no_stages) + """The pipelines it holds, by the member of its configuration that holds each: each codec with the chunk it is handed.""" _POSITIONS: Final[Mapping[CodecKind, int]] = { @@ -72,9 +84,10 @@ def read_pipeline( The order first: array -> array codecs, then one array -> bytes codec, then bytes -> bytes codecs. Then each codec in turn, handed the chunk the one before it handed on: judged by its chunk rules, which locate - their problems in its configuration, and, an array -> array codec, - asked what it hands on. Problems are located under `loc`, where the - codecs sit, each codec at its index. + their problems in its configuration; each pipeline it holds read the + same way, where it sits in the configuration; and, an array -> array + codec, asked what it hands on. Problems are located under `loc`, where + the codecs sit, each codec at its index. """ problems = list(_order_problems(codecs, loc)) stages: list[Stage] = [] @@ -92,10 +105,13 @@ def read_pipeline( handed = None continue incoming = Chunk() if handed is None else handed - stages.append(Stage(codec, incoming)) at = (*loc, index, "configuration") + inner = _no_stages() if configuration is not None: problems.extend(_chunk_problems(definition, configuration, codec.nested, incoming, at)) + inner, found = _inner_pipelines(definition, configuration, codec.nested, incoming, at) + problems.extend(found) + stages.append(Stage(codec, incoming, inner)) if definition.kind == "array_bytes": handed = None elif configuration is None: @@ -161,6 +177,77 @@ def _chunk_problems( return ruled(definition, lambda: definition.chunk_rules(configuration, nested, chunk), at) +def _inner_pipelines( + definition: CodecDefinition[Any], + configuration: Mapping[str, Any], + nested: Nested, + chunk: Chunk, + at: Loc, +) -> tuple[Mapping[str, tuple[Stage, ...]], Problems]: + """Each pipeline a codec handed `chunk` holds, read handed the chunk its `pipelines` says. + + Its codecs are the fields the member holding it holds, as the scope + read them, and its problems are located there, under `at`. What + `pipelines` gives is the extension author's code: it is checked to map + members of the configuration to chunks, each member a list of codecs. + """ + given = asked( + definition, + "pipelines", + lambda: cast("object", definition.pipelines(configuration, nested, chunk)), + at, + ) + if not _is_pipelines(given): + msg = ( + f"{definition.name!r}: its pipelines give a mapping of members to chunks, got {given!r}" + ) + raise TypeError(msg) + read: dict[str, tuple[Stage, ...]] = {} + problems: list[ValidationProblem] = [] + for member, handed in given.items(): + codecs = _held(definition, configuration, nested, member) + read[member], found = read_pipeline(codecs, handed, (*at, member)) + problems.extend(found) + return read, tuple(problems) + + +def _is_pipelines(value: object) -> TypeGuard[Mapping[str, Chunk]]: + """Whether `value` maps members of a configuration to chunks.""" + return isinstance(value, Mapping) and all( + isinstance(member, str) and isinstance(chunk, Chunk) + for member, chunk in cast("Mapping[object, object]", value).items() + ) + + +def _held( + definition: CodecDefinition[Any], configuration: Mapping[str, Any], nested: Nested, member: str +) -> tuple[Resolved[CodecDefinition[Any]], ...]: + """The codecs `member` of the configuration holds, as the scope read them. + + A member holding anything but a list of fields, or fields a definition + of another kind read, is a `TypeError`: a fault in the definition that + names it. Fields nothing in scope claims tell nothing of their kind, + and are read as codecs nothing claims. + """ + entries: object = configuration.get(member) + places = ( + [(member, index) for index in range(len(cast("tuple[object, ...]", entries)))] + if isinstance(entries, tuple) + else None + ) + if places is None or not all( + place in nested + and ( + nested[place].definition is None + or isinstance(nested[place].definition, CodecDefinition) + ) + for place in places + ): + msg = f"{definition.name!r}: its pipelines name {member!r}, which holds no list of codecs" + raise TypeError(msg) + return tuple(nested[place] for place in places) + + def _handed_on( definition: CodecDefinition[Any], configuration: Mapping[str, Any], diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/codec/sharding_indexed.py b/packages/zarr-metadata/src/zarr_metadata/v3/codec/sharding_indexed.py index 5c6745a90e..98dd1560cf 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/codec/sharding_indexed.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/codec/sharding_indexed.py @@ -4,13 +4,22 @@ See https://zarr-specs.readthedocs.io/en/latest/v3/codecs/sharding-indexed/index.html """ -from collections.abc import Iterator +from collections.abc import Iterator, Mapping from typing import Final, Literal, NotRequired from typing_extensions import TypedDict from zarr_metadata._json import ValidationProblem -from zarr_metadata.v3._definition import CodecDefinition, CodecField, Nested, StaticCodecField +from zarr_metadata.v3._definition import ( + Chunk, + CodecDefinition, + CodecField, + Lengths, + Nested, + Resolved, + StaticCodecField, +) +from zarr_metadata.v3.data_type.uint64 import UINT64_DATA_TYPE, UINT64_DATA_TYPE_NAME SHARDING_INDEXED_CODEC_NAME: Final = "sharding_indexed" """The `name` field value of the `sharding_indexed` codec.""" @@ -80,17 +89,101 @@ def _rules( ) +def _chunk_rules( + configuration: ShardingIndexedCodecConfiguration, nested: Nested, chunk: Chunk +) -> Iterator[ValidationProblem]: + """An inner chunk length for each axis of the shard it is handed, dividing each length the shards take along it. + + "The length of the chunk_shape array must match the number of + dimensions of the shard shape to which this sharding codec is applied, + and the inner chunk shape along each dimension must evenly divide the + size of the shard shape" + (https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v3/codecs/sharding-indexed/index.rst#L131-L135). + The shard is the chunk the codec is handed, after any codec before it: + a transposed shard is divided along its transposed axes. + """ + chunk_shape = configuration["chunk_shape"] + lengths = chunk.lengths + if lengths is None: + return + if len(chunk_shape) != len(lengths): + yield ValidationProblem( + ("chunk_shape",), + f"expected {len(lengths)} inner chunk lengths, one per axis of the chunk the codec " + f"is handed, got {len(chunk_shape)}", + "invalid_value", + ) + return + for axis, (inner, shard) in enumerate(zip(chunk_shape, lengths, strict=True)): + undivided = [] if shard is None else [length for length in shard if length % inner != 0] + if len(undivided) != 0: + yield ValidationProblem( + ("chunk_shape", axis), + f"expected an inner chunk length that divides each length the shards take " + f"along this axis, got {inner}, which does not divide {min(undivided)}", + "invalid_value", + ) + + +_UINT64: Final = Resolved(UINT64_DATA_TYPE_NAME, "read", UINT64_DATA_TYPE, {}) +"""The data type of a shard index, which the spec fixes whatever the scope holds.""" + + +def _pipelines( + configuration: ShardingIndexedCodecConfiguration, nested: Nested, chunk: Chunk +) -> Mapping[str, Chunk]: + """The inner codecs are handed the inner chunks; the index codecs, the shard index. + + An inner chunk is "a chunk within the shard", of the shape + `chunk_shape` gives, and so of the shard's data type + (https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v3/codecs/sharding-indexed/index.rst#L168-L170). + The index is "an array with 64-bit unsigned integers with a shape that + matches the chunks per shard tuple with an appended dimension of size + 2" + (https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v3/codecs/sharding-indexed/index.rst#L186-L187), + chunks per shard being "the element-wise division of the shard shape by + the inner chunk shape" + (https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v3/codecs/sharding-indexed/index.rst#L173-L174): + unknown along an axis whose shard lengths are unknown, or that the + inner chunk length does not divide. Handed a shard of another number + of axes than `chunk_shape` has, it hands both pipelines chunks of + lengths nothing is known of: which of the two is wrong is not known. + """ + chunk_shape = configuration["chunk_shape"] + lengths = chunk.lengths + if lengths is not None and len(lengths) != len(chunk_shape): + return {"codecs": Chunk(None, chunk.data_type), "index_codecs": Chunk(None, _UINT64)} + inner = Chunk(tuple(frozenset({length}) for length in chunk_shape), chunk.data_type) + index = Chunk((*_per_shard(chunk_shape, lengths), frozenset({2})), _UINT64) + return {"codecs": inner, "index_codecs": index} + + +def _per_shard(chunk_shape: tuple[int, ...], lengths: Lengths | None) -> Lengths: + """Along each axis, how many inner chunks a shard holds; None where that is unknown.""" + if lengths is None: + return (None,) * len(chunk_shape) + return tuple( + None + if shard is None or any(length % inner != 0 for length in shard) + else frozenset(length // inner for length in shard) + for inner, shard in zip(chunk_shape, lengths, strict=True) + ) + + SHARDING_INDEXED_CODEC: Final = CodecDefinition( name=SHARDING_INDEXED_CODEC_NAME, configuration=ShardingIndexedCodecConfiguration, kind="array_bytes", size="dynamic", rules=_rules, + chunk_rules=_chunk_rules, + pipelines=_pipelines, ) """The `sharding_indexed` codec. Its two pipelines are nested fields, each codec in them read in the -scope the shard is read in. +scope the shard is read in, and each read as a pipeline: the inner codecs +handed the inner chunks, the index codecs the shard index. """ diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/definition.py b/packages/zarr-metadata/src/zarr_metadata/v3/definition.py index daff576631..bdd28ba803 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/definition.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/definition.py @@ -185,10 +185,15 @@ def acme_lz4_rules( fields the scope read as a pipeline: their order -- array -> array codecs, one array -> bytes codec, bytes -> bytes codecs -- and then each against the chunk it is handed, giving each codec's `Stage` with that -chunk. Nothing is guessed: the codec after one the scope did not read, -or after one that says nothing of what it hands on, is handed a chunk -nothing is known of, `Chunk()`, which is refused nothing; a codec after -that hands on only what it says of its own accord. +chunk. A codec that holds pipelines of its own says what each is +handed: `pipelines`, by the member of its configuration that holds each +-- a shard's inner codecs its inner chunks, its index codecs the shard +index -- and each is read the same way, its stages kept as the codec's +`Stage.inner`. +Nothing is guessed: the codec after one the scope did not read, or after +one that says nothing of what it hands on, is handed a chunk nothing is +known of, `Chunk()`, which is refused nothing; a codec after that hands +on only what it says of its own accord. Raw bits are the one data type whose name carries its configuration: a document writes `r` and the size in bits, and `r16` reads as `r*`, as the @@ -214,13 +219,14 @@ def acme_lz4_rules( no checker reads, named down to the TypedDict that holds it; a `name` that is not a string; a member declared as a function -- `rules`, `canonical`, `fill_value_rules`, `storage`, `shape_rules`, -`chunk_lengths`, `chunk_rules`, `transition` -- that is not one; a data type's -`fill_value` no checker reads; a codec `kind` that is not one of the -three, or a `size` that is not `"static"` or `"dynamic"`; chunk rules of +`chunk_lengths`, `chunk_rules`, `transition`, `pipelines` -- that is not +one; a data type's `fill_value` no checker reads; a codec `kind` that is +not one of the three, or a `size` that is not `"static"` or `"dynamic"`; +a function no codec of its kind is asked -- chunk rules or pipelines of a bytes -> bytes codec, which is handed bytes, or a `transition` of a codec that hands on bytes; a data type named as raw bits of one size are -written. A scope refuses a definition -of no kind. Nothing happens at class creation. +written. A scope refuses a definition of no kind. Nothing happens at +class creation. """ from zarr_metadata._common import JSONValue diff --git a/packages/zarr-metadata/tests/v3/test_definitions.py b/packages/zarr-metadata/tests/v3/test_definitions.py index bc06f78791..87626bcab9 100644 --- a/packages/zarr-metadata/tests/v3/test_definitions.py +++ b/packages/zarr-metadata/tests/v3/test_definitions.py @@ -778,6 +778,7 @@ def test_error_a_data_type_fill_value_no_checker_reads() -> None: (ChunkGridDefinition, "chunk_lengths"), (CodecDefinition, "chunk_rules"), (CodecDefinition, "transition"), + (CodecDefinition, "pipelines"), ], ) def test_error_a_function_member_that_is_not_a_function( diff --git a/packages/zarr-metadata/tests/v3/test_pipelines.py b/packages/zarr-metadata/tests/v3/test_pipelines.py index ece311bc1c..acdf015a2c 100644 --- a/packages/zarr-metadata/tests/v3/test_pipelines.py +++ b/packages/zarr-metadata/tests/v3/test_pipelines.py @@ -7,9 +7,10 @@ from __future__ import annotations -from typing import TYPE_CHECKING +from typing import TYPE_CHECKING, Any import pytest +from typing_extensions import TypedDict from zarr_metadata.model import ZarrV3ArrayMetadata, validate_array_metadata_v3 from zarr_metadata.v3.codec.crc32c import Empty @@ -17,9 +18,12 @@ CORE_AND_EXTENSIONS, Chunk, CodecDefinition, + CodecField, DataTypeDefinition, + DataTypeField, JSONValue, Nested, + Resolved, read_pipeline, resolve, ) @@ -389,13 +393,18 @@ def transition(configuration: Empty, nested: Nested, chunk: Chunk) -> Chunk: @pytest.mark.parametrize( ("kind", "member"), - [("bytes_bytes", "chunk_rules"), ("bytes_bytes", "transition"), ("array_bytes", "transition")], + [ + ("bytes_bytes", "chunk_rules"), + ("bytes_bytes", "transition"), + ("bytes_bytes", "pipelines"), + ("array_bytes", "transition"), + ], ) def test_error_a_codec_hook_nothing_would_ask(kind: str, member: str) -> None: # A bytes -> bytes codec is handed bytes, and only an array -> array # codec hands on a chunk. hook = {member: lambda configuration, nested, chunk: ()} - with pytest.raises(TypeError, match=f"'acme.x': .*{member.split('_')[0]}"): + with pytest.raises(TypeError, match=f"'acme.x': {member}, "): CodecDefinition(name="acme.x", configuration=Empty, kind=kind, size="static", **hook) # pyright: ignore[reportArgumentType] @@ -430,3 +439,43 @@ def test_error_an_array_document_s_codecs_out_of_order_or_not_fitting_its_chunks (("codecs", 0, "configuration", "order"), "invalid_value"), (("codecs", 1, "configuration", "endian"), "missing_key"), ] + + +class AcmeHolderConfiguration(TypedDict, closed=True): + codecs: tuple[CodecField, ...] + types: tuple[DataTypeField, ...] + + +def _holder(pipelines: object) -> Resolved[CodecDefinition[Any]]: + """A codec holding a pipeline of codecs and a list of data types, whose pipelines are `pipelines`.""" + holder = CodecDefinition( + name="acme.holder", + configuration=AcmeHolderConfiguration, + kind="array_bytes", + size="dynamic", + pipelines=pipelines, # pyright: ignore[reportArgumentType] + ) + field = {"name": "acme.holder", "configuration": {"codecs": [LITTLE], "types": ["uint8"]}} + return resolve(field, CodecDefinition, SCOPE.extended_with(holder))[0] + + +@pytest.mark.parametrize("given", ["a chunk", {"codecs": "a chunk"}, {("codecs",): Chunk()}]) +def test_error_pipelines_that_give_something_else(given: object) -> None: + with pytest.raises(TypeError, match="'acme.holder': its pipelines give a mapping of members"): + read_pipeline([_holder(lambda configuration, nested, chunk: given)], CHUNK) + + +@pytest.mark.parametrize("member", ["nowhere", "types"]) +def test_error_pipelines_that_name_a_member_holding_no_codecs(member: str) -> None: + holder = _holder(lambda configuration, nested, chunk: {member: Chunk()}) + with pytest.raises(TypeError, match=f"its pipelines name {member!r}, which holds no list"): + read_pipeline([holder], CHUNK) + + +def test_error_pipelines_that_raise_say_whose_they_are() -> None: + holder = _holder(lambda configuration, nested, chunk: {}["codecs"]) + with pytest.raises(KeyError) as raised: + read_pipeline([holder], CHUNK, ("codecs",)) + assert raised.value.__notes__ == [ + "raised by the pipelines of 'acme.holder', reading ('codecs', 0, 'configuration')" + ] diff --git a/packages/zarr-metadata/tests/v3/test_sharding.py b/packages/zarr-metadata/tests/v3/test_sharding.py new file mode 100644 index 0000000000..3afa98994b --- /dev/null +++ b/packages/zarr-metadata/tests/v3/test_sharding.py @@ -0,0 +1,234 @@ +"""Sharding, read: the shard it is handed, and the two pipelines it holds. + +`sharding_indexed` divides the chunk it is handed into inner chunks, which +its inner codecs are handed, and indexes them in a shard index, which its +index codecs are handed; each pipeline is read as the array's is. +""" + +from __future__ import annotations + +from typing import TYPE_CHECKING, Any + +import pytest + +from zarr_metadata.model import ZarrV3ArrayMetadata, validate_array_metadata_v3 +from zarr_metadata.v3.definition import ( + CORE_AND_EXTENSIONS, + Chunk, + CodecDefinition, + DataTypeDefinition, + JSONValue, + Resolved, + read_pipeline, + resolve, +) + +if TYPE_CHECKING: + from zarr_metadata.v3.definition import Stage, ValidationProblem + +Where = tuple[str | int, ...] + +LITTLE: JSONValue = {"name": "bytes", "configuration": {"endian": "little"}} +INDEX: list[JSONValue] = [LITTLE, "crc32c"] + + +def _dt(name: JSONValue) -> Resolved[DataTypeDefinition[Any]]: + return resolve(name, DataTypeDefinition, CORE_AND_EXTENSIONS)[0] + + +FLOAT32 = _dt("float32") +UINT64 = _dt("uint64") + + +def _chunk(*axes: set[int] | None, data_type: Resolved[DataTypeDefinition[Any]] = FLOAT32) -> Chunk: + return Chunk(tuple(None if axis is None else frozenset(axis) for axis in axes), data_type) + + +def _index(*axes: set[int] | None) -> Chunk: + return _chunk(*axes, data_type=UINT64) + + +def _shard( + chunk_shape: list[int], + codecs: list[JSONValue] | None = None, + index: list[JSONValue] | None = None, +) -> JSONValue: + configuration: dict[str, JSONValue] = { + "chunk_shape": list[JSONValue](chunk_shape), + "codecs": [LITTLE] if codecs is None else codecs, + "index_codecs": INDEX if index is None else index, + } + return {"name": "sharding_indexed", "configuration": configuration} + + +def _transpose(*order: int) -> JSONValue: + return {"name": "transpose", "configuration": {"order": list(order)}} + + +def _read( + codecs: list[JSONValue], chunk: Chunk +) -> tuple[tuple[Stage, ...], tuple[ValidationProblem, ...]]: + """The stages, and every problem: each codec's own, where it sits, then the pipeline's.""" + read = [ + resolve(codec, CodecDefinition, CORE_AND_EXTENSIONS, (index,)) + for index, codec in enumerate(codecs) + ] + stages, found = read_pipeline([resolved for resolved, _ in read], chunk) + return stages, (*(problem for _, own in read for problem in own), *found) + + +def _problems(codecs: list[JSONValue], chunk: Chunk) -> list[tuple[Where, str]]: + return [(problem.loc, problem.kind) for problem in _read(codecs, chunk)[1]] + + +def _handed(stages: tuple[Stage, ...], at: Where = ()) -> dict[Where, Chunk | None]: + """What each codec of each pipeline the codecs hold is handed, by where it sits.""" + found: dict[Where, Chunk | None] = {} + for index, stage in enumerate(stages): + for member, held in stage.inner.items(): + found |= {(*at, index, member, place): s.incoming for place, s in enumerate(held)} + found |= _handed(held, (*at, index, member)) + return found + + +@pytest.mark.parametrize( + ("codecs", "chunk", "handed"), + [ + # Chunks of 8 by 8, shards of 2 by 2 inner chunks. + ( + [_shard([4, 4])], + _chunk({8}, {8}), + { + (0, "codecs", 0): _chunk({4}, {4}), + (0, "index_codecs", 0): _index({2}, {2}, {2}), + (0, "index_codecs", 1): None, + }, + ), + # The shard is the chunk it is handed: transposed, 4 by 6 divides + # into inner chunks of 4 by 3 though 6 by 4 would not. + ( + [_transpose(1, 0), _shard([4, 3])], + _chunk({6}, {4}), + { + (1, "codecs", 0): _chunk({4}, {3}), + (1, "index_codecs", 0): _index({1}, {2}, {2}), + (1, "index_codecs", 1): None, + }, + ), + # A rectilinear grid's shards differ; so do their counts of inner + # chunks. + ( + [_shard([4])], + _chunk({8, 4}), + { + (0, "codecs", 0): _chunk({4}), + (0, "index_codecs", 0): _index({2, 1}, {2}), + (0, "index_codecs", 1): None, + }, + ), + # Handed a chunk nothing is known of, a shard still has the inner + # chunks it declares, and an index of uint64. + ( + [_shard([4])], + Chunk(), + { + (0, "codecs", 0): Chunk((frozenset({4}),)), + (0, "index_codecs", 0): _index(None, {2}), + (0, "index_codecs", 1): None, + }, + ), + # A shard within a shard is read the same way. + ( + [_shard([4, 4], codecs=[_shard([2, 2])])], + _chunk({8}, {8}), + { + (0, "codecs", 0): _chunk({4}, {4}), + (0, "codecs", 0, "codecs", 0): _chunk({2}, {2}), + (0, "codecs", 0, "index_codecs", 0): _index({2}, {2}, {2}), + (0, "codecs", 0, "index_codecs", 1): None, + (0, "index_codecs", 0): _index({2}, {2}, {2}), + (0, "index_codecs", 1): None, + }, + ), + ], +) +def test_every_shard_hands_its_inner_codecs_its_inner_chunks_and_its_index_codecs_its_index( + codecs: list[JSONValue], chunk: Chunk, handed: dict[Where, Chunk | None] +) -> None: + stages, problems = _read(codecs, chunk) + assert problems == () + assert _handed(stages) == handed + + +def test_error_a_shard_whose_inner_chunks_have_another_number_of_axes() -> None: + # One problem: which of the two is wrong is not known, so what the + # pipelines are handed is not either, and they are judged against + # nothing -- these transposes fit once `chunk_shape` has two lengths. + shard = _shard([4], codecs=[_transpose(1, 0), LITTLE], index=[_transpose(2, 1, 0), LITTLE]) + assert _problems([shard], _chunk({8}, {8})) == [ + ((0, "configuration", "chunk_shape"), "invalid_value") + ] + + +@pytest.mark.parametrize( + ("codecs", "chunk", "axis"), + [ + ([_shard([4, 4])], _chunk({8}, {6}), 1), + ([_shard([4])], _chunk({8, 6}), 0), + # Divided along its axes as it is handed them: transposed, a shard + # of 6 by 2 does not divide into inner chunks of 2 by 3, though 2 + # by 6 would. + ([_transpose(1, 0), _shard([2, 3])], _chunk({2}, {6}), 1), + ], +) +def test_error_an_inner_chunk_length_that_does_not_divide_the_shard( + codecs: list[JSONValue], chunk: Chunk, axis: int +) -> None: + at = len(codecs) - 1 + assert _problems(codecs, chunk) == [ + ((at, "configuration", "chunk_shape", axis), "invalid_value") + ] + + +@pytest.mark.parametrize( + ("shard", "found"), + [ + # Each pipeline is read as the array's is, where it sits. + (_shard([4, 4], codecs=[]), [((0, "configuration", "codecs"), "invalid_value")]), + ( + _shard([4, 4], codecs=["crc32c", LITTLE]), + [((0, "configuration", "codecs", 1), "invalid_value")], + ), + ( + _shard([4, 4], codecs=["bytes"]), + [((0, "configuration", "codecs", 0, "configuration", "endian"), "missing_key")], + ), + # The index is of uint64, whose values take several bytes, and has + # an axis more than the shard. + ( + _shard([4, 4], index=["bytes"]), + [((0, "configuration", "index_codecs", 0, "configuration", "endian"), "missing_key")], + ), + ( + _shard([4, 4], index=[_transpose(1, 0), LITTLE]), + [((0, "configuration", "index_codecs", 0, "configuration", "order"), "invalid_value")], + ), + ], +) +def test_error_a_pipeline_a_shard_holds_that_does_not_fit_what_it_is_handed( + shard: JSONValue, found: list[tuple[Where, str]] +) -> None: + assert _problems([shard], _chunk({8}, {8})) == found + + +def test_error_an_array_document_s_shard_is_read_where_it_sits() -> None: + document = dict(ZarrV3ArrayMetadata.create_default(shape=(16, 16)).to_json()) + document["chunk_grid"] = {"name": "regular", "configuration": {"chunk_shape": [8, 6]}} + document["codecs"] = [_shard([4, 4], index=["bytes"])] + assert [(p.loc, p.kind) for p in validate_array_metadata_v3(document)] == [ + (("codecs", 0, "configuration", "chunk_shape", 1), "invalid_value"), + ( + ("codecs", 0, "configuration", "index_codecs", 0, "configuration", "endian"), + "missing_key", + ), + ] From 074c7c9950347c38bab9eac481fb8ff6c1104d5d Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Sun, 27 Sep 2026 19:36:35 +0200 Subject: [PATCH 17/17] docs(zarr-metadata): the sharding fragments take this PR's number Assisted-by: ClaudeCode:claude-opus-5-5 --- .../changes/{+layer-4d-fields.feature.md => 4443.feature.1.md} | 0 .../changes/{+layer-4d.feature.md => 4443.feature.md} | 0 2 files changed, 0 insertions(+), 0 deletions(-) rename packages/zarr-metadata/changes/{+layer-4d-fields.feature.md => 4443.feature.1.md} (100%) rename packages/zarr-metadata/changes/{+layer-4d.feature.md => 4443.feature.md} (100%) diff --git a/packages/zarr-metadata/changes/+layer-4d-fields.feature.md b/packages/zarr-metadata/changes/4443.feature.1.md similarity index 100% rename from packages/zarr-metadata/changes/+layer-4d-fields.feature.md rename to packages/zarr-metadata/changes/4443.feature.1.md diff --git a/packages/zarr-metadata/changes/+layer-4d.feature.md b/packages/zarr-metadata/changes/4443.feature.md similarity index 100% rename from packages/zarr-metadata/changes/+layer-4d.feature.md rename to packages/zarr-metadata/changes/4443.feature.md