diff --git a/packages/zarr-metadata/README.md b/packages/zarr-metadata/README.md index 217e3f6e01..d91a85a443 100644 --- a/packages/zarr-metadata/README.md +++ b/packages/zarr-metadata/README.md @@ -71,7 +71,7 @@ against the chunk it is handed, a shard's inner and index codecs too. The validators do no arithmetic on values: whether a fill value survives a `cast_value` round trip is not judged. -Three choices the specs' words leave open, or settle two ways: +Two choices the specs' words leave open, or settle two ways: - **`attributes` may hold `NaN`, `Infinity` and `-Infinity`.** The spec interprets no attribute, and zarr-python and xarray write those numbers @@ -85,11 +85,13 @@ Three choices the specs' words leave open, or settle two ways: reads wrong bytes as surely as one that skips a data type reads wrong values. It keeps its meaning on an unknown top-level member, which a reader can skip. -- **A chunk length of 0 is allowed along a dimension of length 0.** The - core spec asks for non-zero chunk lengths only "when the corresponding - dimensions of the arrays have non-zero length"; the regular grid spec - says chunk sizes are greater than zero. The package follows the core - spec, which zarr-python 3.0 and 3.1 wrote for an empty dimension. + +A regular grid's chunk lengths are at least 1, along a dimension of +length 0 too: "Chunk sizes must be greater than zero", the regular grid +spec says. The core spec's "non-zero when the corresponding dimensions +of the arrays have non-zero length" says less, and allows nothing more, +so a document with a 0 there, as zarr-python 3.0 and 3.1 wrote for an +empty dimension, is refused. `read_array_metadata_v3` reads a document once and returns everything the read found: each field as the scope read it -- `Read` by the @@ -134,13 +136,37 @@ build the model of either kind, as the models' own `from_json` and A member the spec does not define is not a field; the model's `must_understand_fields` names those a reader must understand. -The Pydantic integration's generated JSON Schemas express independently -checkable document structure and field constraints, but they are not a -replacement for runtime model validation. Standard JSON Schema treats a -mathematically integral number such as `1.0` as an integer, while the runtime -boundary requires Python `int` values, and it cannot express arbitrary -same-length relations such as `dimension_names` versus `shape` or v2 `chunks` -versus `shape`. Consumers should run the model parser after schema validation. +`node_metadata_json_schema_v3` writes what the validators read as a +JSON Schema, draft 2020-12, for an editor that checks a `zarr.json` as it +is written, or a validator in another language. Each extension point is +a field as its scope reads it: a configuration as its definition's +TypedDict says, bounds and all, and a name nothing in the scope claims +with any configuration. The fill value is what the data type it names +takes. `field_json_schema(kind, context)`, in +`zarr_metadata.v3.definition`, writes one field's schema, and +`json_schema`, in `zarr_metadata.typed_json`, any TypedDict's, as `check` +reads it. A schema says what each member is, and not what the rules say +of members together, so a document it accepts may still have a problem; +a JSON document the validators accept, it accepts. A validator reads +JSON as a parser gives it, arrays as lists: a model's `to_json` writes +tuples, which a Python validator does not take for arrays. + +```python +import json +from zarr_metadata.model import node_metadata_json_schema_v3 + +with open("zarr.schema.json", "w") as f: + json.dump(node_metadata_json_schema_v3(), f, indent=2) +``` + +The Pydantic integration's field types have JSON Schemas of their own, +for a model that holds them: an extension point there is a name and any +configuration, read in no scope, and v2 documents have one too. For a +`zarr.json`, use `node_metadata_json_schema_v3`. Neither replaces the +validators: JSON Schema takes a number such as `1.0` for an integer, +where the models require an `int`, and says nothing of what members read +together say, such as `dimension_names` against `shape` or v2 `chunks` +against `shape`. Run the model parser after schema validation. ## Scope diff --git a/packages/zarr-metadata/changes/371.doc.md b/packages/zarr-metadata/changes/371.doc.md index d0017875f5..94492433c2 100644 --- a/packages/zarr-metadata/changes/371.doc.md +++ b/packages/zarr-metadata/changes/371.doc.md @@ -2,6 +2,6 @@ The validation boundary says what the package decides where the specs leave it open or settle it two ways: `attributes` may hold `NaN`, `Infinity` and `-Infinity`, which `to_key_value` writes as bare tokens; `must_understand: false` is refused at every extension point, codecs -too; a chunk length of 0 is allowed along a dimension of length 0. The -models' `from_json`, `to_json`, `from_key_value` and `to_key_value` -have docstrings, and no public docstring names a private function. +too. The models' `from_json`, `to_json`, `from_key_value` and +`to_key_value` have docstrings, and no public docstring names a private +function. diff --git a/packages/zarr-metadata/changes/378.bugfix.md b/packages/zarr-metadata/changes/378.bugfix.md new file mode 100644 index 0000000000..bde372fb24 --- /dev/null +++ b/packages/zarr-metadata/changes/378.bugfix.md @@ -0,0 +1,9 @@ +The `regular` chunk grid refuses a chunk length of 0, along a dimension +of length 0 too, as its specification says: "Chunk sizes must be greater +than zero". `RegularChunkGridConfiguration.chunk_shape` is +`tuple[Annotated[int, Ge(1)], ...]`, so a 0 is an `invalid_value` at its +place, with the bound in its `ctx`. The package had read the core +specification's "The chunk shape elements are non-zero when the +corresponding dimensions of the arrays have non-zero length" as allowing +0 on an empty dimension, as zarr-python 3.0 and 3.1 wrote it; that +sentence says less than the grid's own, not something else. diff --git a/packages/zarr-metadata/changes/378.feature.md b/packages/zarr-metadata/changes/378.feature.md new file mode 100644 index 0000000000..9943b052bd --- /dev/null +++ b/packages/zarr-metadata/changes/378.feature.md @@ -0,0 +1,21 @@ +JSON Schemas of what the package reads, draft 2020-12, as pydantic's +`TypeAdapter(...).json_schema()` and zod's `toJSONSchema` write theirs. +`node_metadata_json_schema_v3(context=...)`, in `zarr_metadata.model`, +is a `zarr.json`'s, an array's or a group's, for an editor or a +validator in another language: each extension point a field as its +scope reads it -- a configuration as its definition's TypedDict says, +bounds and all, or a name nothing in the scope claims, with any +configuration -- the fill value what the data type it names takes, and +a group's consolidated metadata the documents it holds. +`field_json_schema(kind, context)`, in `zarr_metadata.v3.definition`, is +one field's, and `json_schema(shape)`, in `zarr_metadata.typed_json`, +any TypedDict's, as `check` reads it. What the rules say of members +together is not in a schema, so a document it accepts may still have a +problem; a JSON document the validators accept, it accepts. JSON Schema +takes `1.0` for an integer, where the package wants `1`. A v3 array +document's extension points are annotated with the field aliases -- +`data_type: DataTypeField`, `codecs: tuple[CodecField, ...]` -- which +are the JSON a metadata field is, so a type checker and `check` read +them as before; its `shape` holds integers of at least 0, which `check` +now holds it to. A data type whose `fill_value` holds a metadata field +is refused when it is built: a fill value is a value of its data type. diff --git a/packages/zarr-metadata/changes/4443.feature.3.md b/packages/zarr-metadata/changes/4443.feature.3.md index b899c84f09..f7c8bcd78b 100644 --- a/packages/zarr-metadata/changes/4443.feature.3.md +++ b/packages/zarr-metadata/changes/4443.feature.3.md @@ -1,7 +1,6 @@ **Breaking:** a v3 array's `chunk_grid` is judged against its `shape`, by the grid's definition. A regular grid whose `chunk_shape` does not have -one length per dimension of the shape, or has a length of 0 for a -dimension that is not empty, and a rectilinear grid whose `chunk_shapes` -does not have one entry per dimension, or whose chunk lengths fall short -of their dimension, each have a problem in `chunk_grid.configuration`, -where the package accepted them before. +one length per dimension of the shape, and a rectilinear grid whose +`chunk_shapes` does not have one entry per dimension, or whose chunk +lengths fall short of their dimension, each have a problem in +`chunk_grid.configuration`, where the package accepted them before. diff --git a/packages/zarr-metadata/docs/api/index.md b/packages/zarr-metadata/docs/api/index.md index 66c7bda75b..7f4c73c7dc 100644 --- a/packages/zarr-metadata/docs/api/index.md +++ b/packages/zarr-metadata/docs/api/index.md @@ -7,12 +7,14 @@ title: API reference The package is organized to mirror the structure of the Zarr specifications: - [`zarr_metadata.model`](model.md) — frozen-dataclass document models, - validators, loc-aware parsers, and the `UNSET` sentinel + validators, loc-aware parsers, a `zarr.json`'s JSON Schema, and the + `UNSET` sentinel - [`zarr_metadata.pydantic`](pydantic.md) — optional Pydantic field types over the models - [`zarr_metadata.typed_json`](typed_json.md) — `check`, which type-checks a JSON value against any of the package's `TypedDict`s, read as the - typing spec defines them, with every problem located + typing spec defines them, with every problem located, and + `json_schema`, which writes what `check` reads as a JSON Schema - [`zarr_metadata.v2`](v2.md) — `TypedDict` shapes for Zarr v2 documents (`.zarray`, `.zgroup`, `.zattrs`, `.zmetadata`) - [`zarr_metadata.v3`](v3/index.md) — `TypedDict` shapes for Zarr v3 @@ -22,8 +24,8 @@ The package is organized to mirror the structure of the Zarr specifications: - [`zarr_metadata.v3.definition`](v3/definition.md) — each extension's metadata as a definition: the TypedDict its configuration is, and the rules on it; check JSON against a TypedDict, judge a configuration, - read a whole field in a scope, or read a codec pipeline. Its module - docstring is the guide + read a whole field in a scope, read a codec pipeline, or write a + scope's fields as a JSON Schema. Its module docstring is the guide The document types, models, and spec vocabulary — including the store keys — are re-exported at the top level, so diff --git a/packages/zarr-metadata/docs/index.md b/packages/zarr-metadata/docs/index.md index fa3c686217..627b37ed56 100644 --- a/packages/zarr-metadata/docs/index.md +++ b/packages/zarr-metadata/docs/index.md @@ -86,7 +86,7 @@ against the chunk it is handed, a shard's inner and index codecs too. The validators do no arithmetic on values: whether a fill value survives a `cast_value` round trip is not judged. -Three choices the specs' words leave open, or settle two ways: +Two choices the specs' words leave open, or settle two ways: - **`attributes` may hold `NaN`, `Infinity` and `-Infinity`.** The spec interprets no attribute, and zarr-python and xarray write those numbers @@ -100,11 +100,13 @@ Three choices the specs' words leave open, or settle two ways: reads wrong bytes as surely as one that skips a data type reads wrong values. It keeps its meaning on an unknown top-level member, which a reader can skip. -- **A chunk length of 0 is allowed along a dimension of length 0.** The - core spec asks for non-zero chunk lengths only "when the corresponding - dimensions of the arrays have non-zero length"; the regular grid spec - says chunk sizes are greater than zero. The package follows the core - spec, which zarr-python 3.0 and 3.1 wrote for an empty dimension. + +A regular grid's chunk lengths are at least 1, along a dimension of +length 0 too: "Chunk sizes must be greater than zero", the regular grid +spec says. The core spec's "non-zero when the corresponding dimensions +of the arrays have non-zero length" says less, and allows nothing more, +so a document with a 0 there, as zarr-python 3.0 and 3.1 wrote for an +empty dimension, is refused. `read_array_metadata_v3` reads a document once and returns everything the read found: each field as the scope read it -- `Read` by the @@ -149,6 +151,29 @@ build the model of either kind, as the models' own `from_json` and A member the spec does not define is not a field; the model's `must_understand_fields` names those a reader must understand. +`node_metadata_json_schema_v3` writes what the validators read as a +JSON Schema, draft 2020-12, for an editor that checks a `zarr.json` as it +is written, or a validator in another language. Each extension point is +a field as its scope reads it: a configuration as its definition's +TypedDict says, bounds and all, and a name nothing in the scope claims +with any configuration. The fill value is what the data type it names +takes. `field_json_schema(kind, context)`, in +`zarr_metadata.v3.definition`, writes one field's schema, and +`json_schema`, in `zarr_metadata.typed_json`, any TypedDict's, as `check` +reads it. A schema says what each member is, and not what the rules say +of members together, so a document it accepts may still have a problem; +a JSON document the validators accept, it accepts. A validator reads +JSON as a parser gives it, arrays as lists: a model's `to_json` writes +tuples, which a Python validator does not take for arrays. + +```python +import json +from zarr_metadata.model import node_metadata_json_schema_v3 + +with open("zarr.schema.json", "w") as f: + json.dump(node_metadata_json_schema_v3(), f, indent=2) +``` + ## Scope At minimum, this library supports what Zarr-Python needs: the complete diff --git a/packages/zarr-metadata/src/zarr_metadata/_typed_json.py b/packages/zarr-metadata/src/zarr_metadata/_typed_json.py index a5c38f2476..d2dca722e7 100644 --- a/packages/zarr-metadata/src/zarr_metadata/_typed_json.py +++ b/packages/zarr-metadata/src/zarr_metadata/_typed_json.py @@ -28,6 +28,10 @@ depth; the parser it returns is used as it is. Parsers are compiled once per annotation and are pure functions of the value, so the branch of a union that did not match leaves nothing behind. + +`json_schema` writes the same reading as a JSON Schema: `Schemas` writes +each shape as the checker reads it, asking a caller's `SchemaLeaf` first, +as a parser asks a `Leaf`. """ from __future__ import annotations @@ -38,6 +42,7 @@ import sys import types import typing +import urllib.parse from collections.abc import Callable, Iterator, Mapping, Sequence from collections.abc import Set as AbstractSet from dataclasses import dataclass @@ -66,6 +71,7 @@ from zarr_metadata._json import ( ValidationProblem, choices, + copied, is_json, outside_of, refine_json, @@ -1135,14 +1141,253 @@ def check( return (cast("T", typed) if readable else None), with_input(found, value, loc) +# --- JSON Schema --------------------------------------------------------- + +JSONSchema: TypeAlias = dict[str, JSONValue] +"""A JSON Schema, as the JSON object it is: arrays as lists, as validators take them.""" + +SchemaLeaf: TypeAlias = Callable[[object, "Schemas"], "JSONSchema | None"] +"""A caller's own shapes, written into a schema: asked first for every annotation, as a `Leaf` is, None to decline. + +Handed the schema being written, so a shape of the caller's own can hold +others, written with `of`, or be written once, in `$defs`, with `defined`. +""" + +DIALECT: Final = "https://json-schema.org/draft/2020-12/schema" +"""The dialect every schema is written in: JSON Schema draft 2020-12, as pydantic and zod write theirs.""" + +_KEYWORDS: Final[Mapping[str, str]] = { + "gt": "exclusiveMinimum", + "ge": "minimum", + "lt": "exclusiveMaximum", + "le": "maximum", +} +"""JSON Schema's keyword for each bound.""" + +_STRICTER: Final[Mapping[str, Callable[[float, float], float]]] = { + "exclusiveMinimum": max, + "minimum": max, + "exclusiveMaximum": min, + "maximum": min, +} +"""Of two bounds a keyword says, the one a value in both keeps within.""" + + +def no_schema_leaf(annotation: object, schemas: Schemas) -> JSONSchema | None: + """The schema leaf of a caller with no shapes of its own.""" + return None + + +class Schemas: + """One JSON Schema being written, and the `$defs` it holds so far. + + `of` writes an annotation as the checker reads it, asking the leaf + first at every depth, as `parser_for` asks a `Leaf`. A TypedDict and a + type alias are each written once, in `$defs`, under its name -- or its + name and a number, when another holds that one -- and referred to + wherever they occur, so one that holds itself is a schema that refers + to itself. `document` is the whole schema. + """ + + __slots__ = ("_defs", "_leaf", "_names", "_uses") + + def __init__(self, leaf: SchemaLeaf = no_schema_leaf) -> None: + self._leaf = leaf + self._defs: dict[str, JSONSchema] = {} + self._names: dict[object, str] = {} + self._uses: dict[str, int] = {} + + def of(self, annotation: object) -> JSONSchema: + """`annotation` as the checker reads it: its type, with the bounds and notes `Annotated` carries. + + A bound is the keyword JSON Schema has for it -- `Ge(0)` is + `minimum` -- and a `Doc` is the `description`. A type's bounds and + the bounds its `NewType` holds are both kept, as the checker holds + a value to both: where the two say one keyword, the stricter. An + annotation the checker reads is written; any other is a + `TypeError`, which a caller that vetted it through `parser` never + meets. + """ + inner, metadata = strip_annotation(annotation) + schema = self._type(inner) + if len(metadata) == 0: + return schema + notes = [ + item.documentation + for item in _unpacked(metadata) + if isinstance(item, typing_extensions.Doc) + ] + if len(notes) != 0: + schema = {**schema, "description": "\n\n".join(notes)} + for name, bound in constraints_of(metadata).items(): + keyword = _KEYWORDS[name] + held = cast("int | float | None", schema.get(keyword)) + schema = {**schema, keyword: bound if held is None else _STRICTER[keyword](held, bound)} + return schema + + def object_of(self, typeddict: type) -> JSONSchema: + """`typeddict` written in place: its keys, those it requires, and what any other key may hold.""" + keys = typeddict_keys(typeddict) + schema: JSONSchema = {"type": "object"} + if len(keys.members) != 0: + schema["properties"] = { + key: self.of(annotation) for key, (annotation, _) in keys.members.items() + } + required: list[JSONValue] = [key for key in keys.members if key in keys.required] + if len(required) != 0: + schema["required"] = required + if keys.closed: + schema["additionalProperties"] = False + elif not keys.open: + extra = self.of(keys.extra_items) + if len(extra) != 0: + schema["additionalProperties"] = extra + return schema + + def defined(self, key: object, name: str, write: Callable[[], JSONSchema]) -> JSONSchema: + """A reference to the entry in `$defs` for `key`, which `write` writes the first time `key` is asked for. + + The entry is named `name`, or `name` and a number when another key + holds that name, and it is reserved before it is written, so a + schema that holds itself refers to itself. One `write` fails to + write is not left reserved: a later reference to it would be to an + empty schema, which takes anything. + """ + name_held = self._names.get(key) + if name_held is None: + name_held, number = name, 1 + while name_held in self._defs: + number += 1 + name_held = f"{name}{number}" + self._names[key] = name_held + self._defs[name_held] = {} + try: + self._defs[name_held] = write() + except BaseException: + del self._names[key], self._defs[name_held] + raise + self._uses[name_held] = self._uses.get(name_held, 0) + 1 + return {"$ref": _pointer(name_held)} + + def document(self, root: JSONSchema) -> JSONSchema: + """The whole schema: its dialect, `root`, and the `$defs`, by name. + + `root` is written in place when it refers to an entry nothing else + refers to, as pydantic writes a model that does not hold itself. + What comes back shares nothing with what was written, nor one part + of it with another, so a caller may change it where it likes. + """ + defs = dict(self._defs) + target = next((name for name in defs if root == {"$ref": _pointer(name)}), None) + if target is not None and self._uses[target] == 1: + root = defs.pop(target) + whole: JSONSchema = {"$schema": DIALECT, **root} + if len(defs) != 0: + whole["$defs"] = {name: defs[name] for name in sorted(defs)} + return cast("JSONSchema", copied(whole)) + + def _type(self, inner: object) -> JSONSchema: + """The schema of a type, `Annotated` peeled from it.""" + found = self._leaf(inner, self) + if found is not None: + return found + if inner is int: + return {"type": "integer"} + if inner is float: + return {"type": "number"} + if inner is bool: + return {"type": "boolean"} + if inner is str: + return {"type": "string"} + if inner is None or inner is types.NoneType: + return {"type": "null"} + if inner is JSONValue: + return {} + origin = get_origin(inner) + if origin is Literal: + # Sorted, as `_literal` sorts them: `get_args` reports a + # `Literal`'s values in the order the first one built wrote them. + values: list[JSONValue] = sorted(get_args(inner), key=repr) + return {"const": values[0]} if len(values) == 1 else {"enum": values} + if is_union(inner): + return {"anyOf": [self.of(branch) for branch in get_args(inner)]} + if origin is tuple: + return self._tuple(get_args(inner)) + if origin in (Mapping, dict): + value = self.of(get_args(inner)[1]) + if len(value) == 0: + return {"type": "object"} + return {"type": "object", "additionalProperties": value} + if isinstance(inner, type) and is_typeddict(inner): + typeddict = inner + return self.defined(typeddict, typeddict.__name__, lambda: self.object_of(typeddict)) + if isinstance(inner, NewType): + return self.of(inner.__supertype__) + if is_alias(inner): + alias = cast("typing_extensions.TypeAliasType", inner) + return self.defined(alias, alias.__name__, lambda: self.of(alias_value(alias))) + msg = f"{inner!r} is not a shape JSON takes" + raise TypeError(msg) + + def _tuple(self, arguments: tuple[object, ...]) -> JSONSchema: + if len(arguments) == 2 and arguments[1] is Ellipsis: + items = self.of(arguments[0]) + return {"type": "array"} if len(items) == 0 else {"type": "array", "items": items} + if len(arguments) == 0: + return {"type": "array", "maxItems": 0} + return { + "type": "array", + "prefixItems": [self.of(argument) for argument in arguments], + "items": False, + "minItems": len(arguments), + } + + +def _pointer(name: str) -> str: + """The reference to the entry in `$defs` named `name`: a JSON pointer, escaped as a URI fragment.""" + escaped = name.replace("~", "~0").replace("/", "~1") + return f"#/$defs/{urllib.parse.quote(escaped, safe='')}" + + +def json_schema(shape: type) -> JSONSchema: + """The JSON Schema of the JSON `check` finds no problem with as `shape`, a TypedDict. + + Draft 2020-12, as a JSON object: arrays as lists, and `$schema` + first. A TypedDict is an object of its keys, those it requires, and + what any other key may hold -- nothing, in a closed one; a bound is + the keyword JSON Schema has for it, `Ge(0)` a `minimum`; a `Doc` is + the `description`, which is all that says one, as zod writes only + what `.describe()` said: a docstring is written for Python's readers; + a `Literal` is its values; a union is `anyOf` its branches. A + TypedDict or type alias is written once in `$defs`, under its name, + and referred to wherever it occurs, but for `shape` itself, which is + written in place unless it holds itself. + + One difference is JSON Schema's own: it takes a number with no + fraction, `1.0`, for an integer, where `check` wants `1`. `TypeError` + for a `shape` that is not a TypedDict, or holds something no parser + reads, as `check` raises it. + """ + if not is_typeddict(shape): + msg = f"{shape!r} is not a TypedDict" + raise TypeError(msg) + _checker(shape) + schemas = Schemas() + return schemas.document(schemas.of(shape)) + + __all__ = [ + "DIALECT", "Branch", "Constraints", + "JSONSchema", "Leaf", "Loc", "Parsed", "Parser", "Qualifier", + "SchemaLeaf", + "Schemas", "Tag", "TypedDictKeys", "alias_value", @@ -1156,8 +1401,10 @@ def check( "is_alias", "is_integer", "is_union", + "json_schema", "mapping_of", "no_leaf", + "no_schema_leaf", "object_of", "one_of", "parser", diff --git a/packages/zarr-metadata/src/zarr_metadata/model/__init__.py b/packages/zarr-metadata/src/zarr_metadata/model/__init__.py index c5f45073fc..0b4ec7344f 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/__init__.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/__init__.py @@ -13,7 +13,9 @@ that narrows or raises `MetadataValidationError`; a v3 array or group document also gets `read_array_metadata_v3` or `read_group_metadata_v3`, one read that returns what it read, the problems, and the model when -there are none. Model `from_json` / `from_key_value` constructors raise +there are none. `node_metadata_json_schema_v3` writes what the v3 +validators read as a JSON Schema, but for the rules. Model `from_json` / +`from_key_value` constructors raise `MetadataValidationError` for every ingestion failure, including missing store keys and undecodable bytes, and the v3 ones take the same `context`. @@ -55,6 +57,7 @@ validate_group_metadata_v3, validate_node_metadata_v3, ) +from zarr_metadata.model._json_schema import node_metadata_json_schema_v3 from zarr_metadata.model._validation import ( ARRAY_METADATA_OPTIONAL_KEYS_V3, ARRAY_METADATA_REQUIRED_KEYS_V2, @@ -159,6 +162,7 @@ "is_metadata_field_v3", "node_metadata_from_json_v3", "node_metadata_from_key_value_v3", + "node_metadata_json_schema_v3", "parse_array_metadata_v2", "parse_array_metadata_v3", "parse_group_metadata_v2", diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_array.py b/packages/zarr-metadata/src/zarr_metadata/model/_array.py index 13d31ded89..4cbc93c597 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_array.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/_array.py @@ -152,8 +152,8 @@ def create_default( type of any fixed size. Overriding `shape` without `chunk_grid` derives a consistent default grid: one regular chunk covering the array (`chunk_shape` equal to `shape`, with a length of 1 for a - dimension of length 0, which every reader takes: the core spec - allows 0 there and the regular grid spec does not, + dimension of length 0, since "Chunk sizes must be greater than + zero", https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v3/chunk-grids/regular-grid/index.rst#L40). """ # The grid derives from a shape the read takes; one it refuses is diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_json_schema.py b/packages/zarr-metadata/src/zarr_metadata/model/_json_schema.py new file mode 100644 index 0000000000..3b6b015fd2 --- /dev/null +++ b/packages/zarr-metadata/src/zarr_metadata/model/_json_schema.py @@ -0,0 +1,98 @@ +"""The JSON Schema of a v3 `zarr.json`, as a scope reads one.""" + +from __future__ import annotations + +from typing import TYPE_CHECKING, Any, cast + +from zarr_metadata._typed_json import Schemas +from zarr_metadata.v3._definition import DataTypeDefinition, field_schemas, written_name +from zarr_metadata.v3._registry import CORE_AND_EXTENSIONS, Context +from zarr_metadata.v3.array import ZarrV3ArrayMetadataJSON +from zarr_metadata.v3.consolidated import ( + ZARR_V3_CONSOLIDATED_METADATA_KEY, + ZarrV3ConsolidatedMetadataJSON, +) +from zarr_metadata.v3.group import ZarrV3GroupMetadataJSON + +if TYPE_CHECKING: + from zarr_metadata._common import JSONValue + from zarr_metadata._typed_json import JSONSchema, SchemaLeaf + + +def node_metadata_json_schema_v3(*, context: Context = CORE_AND_EXTENSIONS) -> JSONSchema: + """The JSON Schema of a v3 `zarr.json` read in `context`: an array document or a group document, as `validate_node_metadata_v3` reads one, but for the rules. + + For an editor that validates a `zarr.json` as it is written, or a + validator in another language. JSON Schema draft 2020-12, as + `json_schema` writes one. Each extension point is a field as + `field_json_schema` writes one in `context`: one a definition in scope + reads, or a name none of them claims. The fill value is the JSON shape + the data type's definition declares for one -- an `int8`'s an integer + in [-128, 127] -- when the document names a data type in scope. A + group's `consolidated_metadata` holds array and group documents, by + path, or is `null`. Each document is in `$defs` under the name of its + TypedDict: `ZarrV3ArrayMetadataJSON` is an array's alone. + + A JSON Schema says what each member is, and what the rules say of + members read together is not in it: one dimension name per dimension + of the shape, a chunk grid that fits the shape, codecs in the order a + pipeline takes them, each against the chunk it is handed, and what a + definition's `rules` say. So a document it accepts may still have a + problem, and a JSON document `validate_node_metadata_v3` finds none + with, it accepts. A validator reads JSON as a parser gives it, arrays + as lists: a model's `to_json` writes tuples, which a Python validator + does not take for arrays. + """ + schemas = Schemas(_documents(context)) + array = schemas.of(ZarrV3ArrayMetadataJSON) + group = schemas.of(ZarrV3GroupMetadataJSON) + return schemas.document({"anyOf": [array, group]}) + + +def _documents(context: Context) -> SchemaLeaf: + """The schema leaf of the documents read in `context`: an array's fill value held to its data type, and a group's consolidated metadata; each field as `context` reads one.""" + fields = field_schemas(context) + + def leaf(annotation: object, schemas: Schemas) -> JSONSchema | None: + if annotation is ZarrV3ArrayMetadataJSON: + return schemas.defined( + annotation, annotation.__name__, lambda: _array(context, schemas) + ) + if annotation is ZarrV3GroupMetadataJSON: + return schemas.defined(annotation, annotation.__name__, lambda: _group(schemas)) + return fields(annotation, schemas) + + return leaf + + +def _array(context: Context, schemas: Schemas) -> JSONSchema: + """An array document: its TypedDict, and its fill value held to the data type it names, for each data type in scope.""" + schema = schemas.object_of(ZarrV3ArrayMetadataJSON) + held: list[JSONValue] = [] + for definition in context.tables.get(DataTypeDefinition, {}).values(): + fill_value = schemas.of(cast("DataTypeDefinition[Any]", definition).fill_value) + if len(fill_value) == 0: + continue # a data type that says nothing of its fill value takes any JSON + name = written_name(definition) + named: JSONSchema = { + "anyOf": [name, {"type": "object", "properties": {"name": name}, "required": ["name"]}] + } + condition: JSONSchema = { + "if": {"properties": {"data_type": named}, "required": ["data_type"]}, + "then": {"properties": {"fill_value": fill_value}}, + } + held.append(condition) + return schema if len(held) == 0 else {**schema, "allOf": held} + + +def _group(schemas: Schemas) -> JSONSchema: + """A group document: its TypedDict, and the consolidated metadata the model reads, which a historical zarr-python bug wrote as `null`.""" + schema = schemas.object_of(ZarrV3GroupMetadataJSON) + properties = cast("dict[str, JSONValue]", schema.get("properties", {})) + consolidated: JSONSchema = { + "anyOf": [schemas.of(ZarrV3ConsolidatedMetadataJSON), {"type": "null"}] + } + return {**schema, "properties": {**properties, ZARR_V3_CONSOLIDATED_METADATA_KEY: consolidated}} + + +__all__ = ["node_metadata_json_schema_v3"] diff --git a/packages/zarr-metadata/src/zarr_metadata/model/_validation.py b/packages/zarr-metadata/src/zarr_metadata/model/_validation.py index 2f8823cd7c..4fc1c70f91 100644 --- a/packages/zarr-metadata/src/zarr_metadata/model/_validation.py +++ b/packages/zarr-metadata/src/zarr_metadata/model/_validation.py @@ -28,7 +28,7 @@ import json from collections.abc import Mapping, Sequence from dataclasses import dataclass -from typing import TYPE_CHECKING, Any, Final, TypeGuard, TypeVar, cast +from typing import TYPE_CHECKING, Any, Final, TypeGuard, TypeVar, cast, get_args, get_origin from zarr_metadata._json import ( MetadataValidationError, @@ -44,13 +44,13 @@ from zarr_metadata._json import is_canonical_json as _is_canonical_json from zarr_metadata._json import prefixed as _prefix from zarr_metadata._sentinel import UNSET +from zarr_metadata._typed_json import typeddict_keys from zarr_metadata.v2.array import ZarrV2ArrayMetadataJSON from zarr_metadata.v2.group import ZarrV2GroupMetadataJSON from zarr_metadata.v3._definition import ( Chunk, ChunkGridDefinition, ChunkKeyEncodingDefinition, - CodecDefinition, DataTypeDefinition, Definition, Lengths, @@ -59,6 +59,7 @@ StorageTransformerDefinition, Unclaimed, chunk_grid_lengths, + field_kind, fields_of, fill_value_problems, resolve, @@ -375,18 +376,21 @@ def attributes_of( return (attributes if len(problems) == 0 else None), tuple(problems) -_EXTENSION_POINTS_V3: Final[tuple[tuple[str, type[Definition[Any]]], ...]] = ( - ("data_type", DataTypeDefinition), - ("chunk_grid", ChunkGridDefinition), - ("chunk_key_encoding", ChunkKeyEncodingDefinition), +_MEMBERS_V3: Final = typeddict_keys(ZarrV3ArrayMetadataJSON).members + +_EXTENSION_POINTS_V3: Final[tuple[tuple[str, type[Definition[Any]]], ...]] = tuple( + (key, kind) + for key, (annotation, _) in _MEMBERS_V3.items() + if (kind := field_kind(annotation)) is not None ) -"""A v3 array document's single extension points, and the kind each is read as.""" +"""A v3 array document's single extension points, and the kind each is read as: each member its TypedDict annotates with a field alias.""" -_EXTENSION_LISTS_V3: Final[tuple[tuple[str, type[Definition[Any]]], ...]] = ( - ("codecs", CodecDefinition), - ("storage_transformers", StorageTransformerDefinition), +_EXTENSION_LISTS_V3: Final[tuple[tuple[str, type[Definition[Any]]], ...]] = tuple( + (key, kind) + for key, (annotation, _) in _MEMBERS_V3.items() + if get_origin(annotation) is tuple and (kind := field_kind(get_args(annotation)[0])) is not None ) -"""Its lists of extension points, and the kind each entry is read as.""" +"""Its lists of extension points, and the kind each entry is read as: each member it annotates as a tuple of a field alias.""" @dataclass(frozen=True, slots=True) diff --git a/packages/zarr-metadata/src/zarr_metadata/typed_json.py b/packages/zarr-metadata/src/zarr_metadata/typed_json.py index 097411a898..3ca22ad4f8 100644 --- a/packages/zarr-metadata/src/zarr_metadata/typed_json.py +++ b/packages/zarr-metadata/src/zarr_metadata/typed_json.py @@ -60,6 +60,22 @@ holds at `loc`, and `ctx`, what was expected there -- a type's bounds, or the values of a `Literal`. +`json_schema` writes what `check` reads as a JSON Schema, draft 2020-12, +for a validator in another language, or an editor: the JSON Schema of +the values `check` finds no problem with. A TypedDict is an object of its +keys, closed or not as it says; a bound is JSON Schema's keyword for it, +`Interval(ge=0, le=9)` a `minimum` and a `maximum`; a TypedDict or a type +alias is written once, in `$defs`, under its name. JSON Schema takes a +number with no fraction, `1.0`, for an integer, where `check` wants `1`: + + from zarr_metadata.typed_json import json_schema + + json_schema(GzipCodecConfiguration) + # {'$schema': 'https://json-schema.org/draft/2020-12/schema', + # 'type': 'object', + # 'properties': {'level': {'type': 'integer', 'minimum': 0, 'maximum': 9}}, + # 'required': ['level'], 'additionalProperties': False} + `check` reads the shapes JSON takes and no others -- `int`, `float` for any number, `bool`, `str`, `None`, `JSONValue`, a `Literal`, `tuple[T, ...]` and `tuple[T1, T2]`, a union, a TypedDict, @@ -77,14 +93,23 @@ from zarr_metadata._common import JSONValue from zarr_metadata._json import ProblemKind, ValidationProblem -from zarr_metadata._typed_json import Loc, TypedDictKeys, check, typeddict_keys +from zarr_metadata._typed_json import ( + JSONSchema, + Loc, + TypedDictKeys, + check, + json_schema, + typeddict_keys, +) __all__ = [ + "JSONSchema", "JSONValue", "Loc", "ProblemKind", "TypedDictKeys", "ValidationProblem", "check", + "json_schema", "typeddict_keys", ] diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/_common.py b/packages/zarr-metadata/src/zarr_metadata/v3/_common.py index 04a0874bf6..1347130a6b 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/_common.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/_common.py @@ -1,14 +1,17 @@ -"""The v3 metadata field: its JSON, and the validators that judge one on its own. +"""The v3 metadata field: its JSON, the aliases a member holding one is annotated with, and the validators that judge one on its own. Private, and below both readers of a field: the model, which judges the fields of a document, and the definitions, which read a field's configuration. Public consumers import `ZarrV3MetadataFieldJSON` from -`zarr_metadata.v3`, and the validators from `zarr_metadata.model`. +`zarr_metadata.v3`, the aliases from `zarr_metadata.v3.definition`, and +the validators from `zarr_metadata.model`. """ from collections.abc import Mapping from typing import TypeGuard, cast +from typing_extensions import TypeAliasType + from zarr_metadata._common import ZarrV3NamedConfigJSON from zarr_metadata._json import ( MetadataValidationError, @@ -31,6 +34,26 @@ """ +# A member holding a metadata field is annotated with the alias of its +# kind, which a scope reads it as. Each alias is the JSON a field is, so to +# a type checker, and to `check`, it is `ZarrV3MetadataFieldJSON`. + +DataTypeField = TypeAliasType("DataTypeField", ZarrV3MetadataFieldJSON) +"""A member holding a data type: a document's `data_type`, or a struct field's; read in the scope what holds it is read in.""" +ChunkGridField = TypeAliasType("ChunkGridField", ZarrV3MetadataFieldJSON) +"""A member holding a chunk grid: a document's `chunk_grid`.""" +ChunkKeyEncodingField = TypeAliasType("ChunkKeyEncodingField", ZarrV3MetadataFieldJSON) +"""A member holding a chunk key encoding: a document's `chunk_key_encoding`.""" +CodecField = TypeAliasType("CodecField", ZarrV3MetadataFieldJSON) +"""A member holding a codec: a document's `codecs` is `tuple[CodecField, ...]`, and so is a shard's.""" +StaticCodecField = TypeAliasType("StaticCodecField", ZarrV3MetadataFieldJSON) +"""A member holding a codec of static size: a shard's `index_codecs` is one, +since a reader finds the index by a size it knows before reading it. +""" +StorageTransformerField = TypeAliasType("StorageTransformerField", ZarrV3MetadataFieldJSON) +"""A member holding a storage transformer: a document's `storage_transformers` is `tuple[StorageTransformerField, ...]`.""" + + def validate_metadata_field_v3( value: object, *, allow_must_understand_false: bool = True ) -> tuple[ValidationProblem, ...]: @@ -142,6 +165,12 @@ def parse_metadata_field_v3(value: object) -> ZarrV3MetadataFieldJSON: __all__ = [ + "ChunkGridField", + "ChunkKeyEncodingField", + "CodecField", + "DataTypeField", + "StaticCodecField", + "StorageTransformerField", "ZarrV3MetadataFieldJSON", "envelope_problems", "is_metadata_field_v3", diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py b/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py index 89410bb3ee..0c589b9db5 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/_definition.py @@ -54,16 +54,26 @@ from zarr_metadata._json import ValidationProblem, copied, refine_json, shown, with_input from zarr_metadata._sentinel import UNSET from zarr_metadata._typed_json import ( + JSONSchema, Loc, Parsed, Parser, - no_leaf, + SchemaLeaf, + Schemas, parser, problem, typeddict_keys, unread_in, ) -from zarr_metadata.v3._common import ZarrV3MetadataFieldJSON, envelope_problems +from zarr_metadata.v3._common import ( + ChunkGridField, + ChunkKeyEncodingField, + CodecField, + DataTypeField, + StaticCodecField, + StorageTransformerField, + envelope_problems, +) if TYPE_CHECKING: from collections.abc import Sequence @@ -357,8 +367,8 @@ def _refusal(self) -> str | None: @functools.cache def _fill_value_parser(annotation: object) -> Parser: - """The checker for a fill value's JSON shape, compiled once; `TypeError` naming what no checker reads.""" - return parser(annotation, no_leaf) + """The checker for a fill value's JSON shape, compiled once; `TypeError` naming what no checker reads, or a metadata field in it.""" + return parser(annotation, _no_field) @dataclass(frozen=True, kw_only=True, slots=True, repr=False) @@ -501,21 +511,6 @@ def as_kind(kind: object) -> type[Definition[Any]]: return found -DataTypeField = TypeAliasType("DataTypeField", ZarrV3MetadataFieldJSON) -"""A configuration member holding a data type, read in the scope the member's field is read in.""" -ChunkGridField = TypeAliasType("ChunkGridField", ZarrV3MetadataFieldJSON) -"""A configuration member holding a chunk grid.""" -ChunkKeyEncodingField = TypeAliasType("ChunkKeyEncodingField", ZarrV3MetadataFieldJSON) -"""A configuration member holding a chunk key encoding.""" -CodecField = TypeAliasType("CodecField", ZarrV3MetadataFieldJSON) -"""A configuration member holding a codec: a shard's `codecs` is `tuple[CodecField, ...]`.""" -StaticCodecField = TypeAliasType("StaticCodecField", ZarrV3MetadataFieldJSON) -"""A configuration member holding a codec of static size: a shard's `index_codecs` is one, -since a reader finds the index by a size it knows before reading it. -""" -StorageTransformerField = TypeAliasType("StorageTransformerField", ZarrV3MetadataFieldJSON) -"""A configuration member holding a storage transformer.""" - _FIELD_KINDS: Final[Mapping[object, type[Definition[Any]]]] = { DataTypeField: DataTypeDefinition, ChunkGridField: ChunkGridDefinition, @@ -529,6 +524,14 @@ def as_kind(kind: object) -> type[Definition[Any]]: """The field aliases whose codec must be of static size.""" +def field_kind(annotation: object) -> type[Definition[Any]] | None: + """The kind of metadata field a member annotated `annotation` holds -- `CodecDefinition` for `CodecField` -- or None when it holds none.""" + try: + return _FIELD_KINDS.get(annotation) + except TypeError: # an unhashable annotation is no field alias + return None + + @dataclass(frozen=True, slots=True) class _NestedField: """A metadata field the check met inside a configuration: where it sits, its kind, its JSON. @@ -553,10 +556,7 @@ def _field(annotation: object) -> Parser | None: judged, and the name related to a definition, by whoever reads the field -- `check` without a scope, `resolve` in one. """ - try: - kind = _FIELD_KINDS.get(annotation) - except TypeError: # an unhashable annotation is no field alias - return None + kind = field_kind(annotation) if kind is None: return None static = annotation in _STATIC_SIZE @@ -601,6 +601,15 @@ def _vetting(annotation: object) -> Parser | None: return _field(annotation) +def _no_field(annotation: object) -> Parser | None: + """A leaf refusing a field alias: a fill value is a value of its data type, and holds no metadata field.""" + if field_kind(annotation) is not None: + name = cast("TypeAliasType", annotation).__name__ + msg = f"{name} holds a metadata field, and a fill value is a value of its data type" + raise TypeError(msg) + return None + + @functools.cache def _vet(configuration: type) -> None: """Refuse a configuration no definition could read with, saying what is wrong with it.""" @@ -616,6 +625,120 @@ def _vet(configuration: type) -> None: raise TypeError(msg) +_KIND_FIELDS: Final[Mapping[type[Definition[Any]], object]] = { + kind: alias for alias, kind in _FIELD_KINDS.items() if alias not in _STATIC_SIZE +} +"""The field alias of each kind: the one a member holding any field of the kind is annotated with.""" + +_RAW_BYTES_SCHEMA_PATTERN: Final = f"^{RAW_BYTES_NAME_PATTERN.pattern}(?![\\s\\S])" +"""`RAW_BYTES_NAME_PATTERN`, matched whole, as a JSON Schema writes a pattern. + +Held to the end of the name by a lookahead for no character at all: a +`$` there would also match before a final newline in a validator that +matches patterns as Python does, so `"r16\\n"`, which names nothing, +would read as raw bits. +""" + + +def field_json_schema(kind: type[Definition[Any]], context: Context) -> JSONSchema: + """The JSON Schema of one metadata field read as `kind` in `context`: what `resolve` reads, but for the rules. + + A field one of the definitions in scope reads -- its name, its + configuration as the TypedDict says, a `must_understand` of `true` + if any, and its bare name when it needs no configuration -- or a + name none of them claims, with any configuration: what keeps the + format open. A field a configuration holds is written the same way, + in the same scope, and a member taking codecs of static size only + takes those. JSON Schema draft 2020-12, as `json_schema` writes one; + the fields it holds, and the configuration of each definition, are in + `$defs`, under the name of the field alias or TypedDict. What only a + rule says -- a blosc `typesize` against its `shuffle` -- is not in it, + so a field it accepts may still have a problem. + """ + schemas = Schemas(field_schemas(context)) + return schemas.document(schemas.of(_KIND_FIELDS[as_kind(kind)])) + + +def field_schemas(context: Context) -> SchemaLeaf: + """The schema leaf that writes each field alias as a field of its kind, as `context` reads one: `field_json_schema`'s.""" + + def leaf(annotation: object, schemas: Schemas) -> JSONSchema | None: + kind = field_kind(annotation) + if kind is None: + return None + alias = cast("TypeAliasType", annotation) + static = annotation in _STATIC_SIZE + return schemas.defined( + alias, alias.__name__, lambda: _field_schema(kind, static, context, schemas) + ) + + return leaf + + +def _field_schema( + kind: type[Definition[Any]], static: bool, context: Context, schemas: Schemas +) -> JSONSchema: + """A field of `kind` as `context` reads it: one a definition in scope reads, or one none of them claims.""" + table = context.tables.get(kind, {}) + branches: list[JSONValue] = [] + for definition in table.values(): + if static and cast("CodecDefinition[Any]", definition).size != "static": + continue + branches.extend(_read_by(definition, schemas)) + branches.extend(_unclaimed(table)) + return {"anyOf": branches} + + +def written_name(definition: Definition[Any]) -> JSONSchema: + """The JSON Schema of each name a document writes for `definition`: its name, or `r` and a size for raw bits.""" + if isinstance(definition, DataTypeDefinition) and definition.name == RAW_BYTES_NAME: + return {"type": "string", "pattern": _RAW_BYTES_SCHEMA_PATTERN} + return {"const": definition.name} + + +def _read_by(definition: Definition[Any], schemas: Schemas) -> list[JSONValue]: + """The fields `definition` reads: an object of its name and configuration, and its bare name when it needs no configuration. + + Raw bits' name carries their configuration, so what is written beside + it holds nothing. + """ + carried = isinstance(definition, DataTypeDefinition) and definition.name == RAW_BYTES_NAME + name = written_name(definition) + bare = carried or not definition.requires_configuration + envelope: JSONSchema = { + "type": "object", + "properties": { + "name": name, + "configuration": schemas.of( + EmptyConfiguration if carried else definition.configuration + ), + "must_understand": {"const": True}, + }, + "required": ["name"] if bare else ["name", "configuration"], + "additionalProperties": False, + } + return [name, envelope] if bare else [envelope] + + +def _unclaimed(table: Mapping[str, Definition[Any]]) -> list[JSONValue]: + """The fields no definition in `table` claims: a name none of them is written with, bare or with any configuration.""" + claimed: list[JSONValue] = [written_name(definition) for definition in table.values()] + name: JSONSchema = {"type": "string"} + if len(claimed) != 0: + name["not"] = {"anyOf": claimed} + envelope: JSONSchema = { + "type": "object", + "properties": { + "name": name, + "configuration": {"type": "object"}, + "must_understand": {"const": True}, + }, + "required": ["name"], + "additionalProperties": False, + } + return [name, envelope] + + @dataclass(frozen=True, slots=True) class _Checker: """A TypedDict's checker, compiled once, and whether a value of it can hold a nested field.""" @@ -1469,6 +1592,9 @@ def _replaced(value: JSONValue, path: Loc, new: JSONValue) -> JSONValue: "canonicalize", "chunk_grid_lengths", "configuration_of", + "field_json_schema", + "field_kind", + "field_schemas", "fill_value_problems", "kind_of", "multi_byte", @@ -1486,4 +1612,5 @@ def _replaced(value: JSONValue, path: Loc, new: JSONValue) -> JSONValue: "unknown_storage", "variable_length", "with_problems", + "written_name", ] diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/array.py b/packages/zarr-metadata/src/zarr_metadata/v3/array.py index 31a5f6b755..79328d4eb7 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/array.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/array.py @@ -1,12 +1,19 @@ """Zarr v3 array metadata types.""" from collections.abc import Mapping -from typing import Final, Literal, NotRequired, TypeAlias +from typing import Annotated, Final, Literal, NotRequired, TypeAlias +from annotated_types import Ge from typing_extensions import TypedDict from zarr_metadata._common import JSONValue -from zarr_metadata.v3._common import ZarrV3MetadataFieldJSON +from zarr_metadata.v3._common import ( + ChunkGridField, + ChunkKeyEncodingField, + CodecField, + DataTypeField, + StorageTransformerField, +) ZarrV3ExtensionField: TypeAlias = JSONValue """The JSON value of an unknown top-level v3 metadata field. @@ -21,21 +28,24 @@ class ZarrV3ArrayMetadataJSON(TypedDict, extra_items=ZarrV3ExtensionField): """ Zarr v3 array metadata document (the `zarr.json` content for an array). - Extra keys may contain arbitrary JSON values. + Extra keys may contain arbitrary JSON values. Each extension point is + annotated with the field alias of its kind -- `data_type` a + `DataTypeField`, each of `codecs` a `CodecField` -- which is the JSON + a metadata field is, and says what a scope reads it as. See https://zarr-specs.readthedocs.io/en/latest/v3/core/index.html#array-metadata """ zarr_format: Literal[3] node_type: Literal["array"] - data_type: ZarrV3MetadataFieldJSON - shape: tuple[int, ...] - chunk_grid: ZarrV3MetadataFieldJSON - chunk_key_encoding: ZarrV3MetadataFieldJSON + data_type: DataTypeField + shape: tuple[Annotated[int, Ge(0)], ...] + chunk_grid: ChunkGridField + chunk_key_encoding: ChunkKeyEncodingField fill_value: JSONValue - codecs: tuple[ZarrV3MetadataFieldJSON, ...] + codecs: tuple[CodecField, ...] attributes: NotRequired[Mapping[str, JSONValue]] - storage_transformers: NotRequired[tuple[ZarrV3MetadataFieldJSON, ...]] + storage_transformers: NotRequired[tuple[StorageTransformerField, ...]] dimension_names: NotRequired[tuple[str | None, ...]] @@ -64,14 +74,14 @@ class ZarrV3ArrayMetadataJSONPartial(TypedDict, total=False, extra_items=ZarrV3E zarr_format: Literal[3] node_type: Literal["array"] - data_type: ZarrV3MetadataFieldJSON - shape: tuple[int, ...] - chunk_grid: ZarrV3MetadataFieldJSON - chunk_key_encoding: ZarrV3MetadataFieldJSON + data_type: DataTypeField + shape: tuple[Annotated[int, Ge(0)], ...] + chunk_grid: ChunkGridField + chunk_key_encoding: ChunkKeyEncodingField fill_value: JSONValue - codecs: tuple[ZarrV3MetadataFieldJSON, ...] + codecs: tuple[CodecField, ...] attributes: NotRequired[Mapping[str, JSONValue]] - storage_transformers: NotRequired[tuple[ZarrV3MetadataFieldJSON, ...]] + storage_transformers: NotRequired[tuple[StorageTransformerField, ...]] dimension_names: NotRequired[tuple[str | None, ...]] diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/chunk_grid/regular.py b/packages/zarr-metadata/src/zarr_metadata/v3/chunk_grid/regular.py index 179fdcc5cc..299603475d 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/chunk_grid/regular.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/chunk_grid/regular.py @@ -23,14 +23,16 @@ class RegularChunkGridConfiguration(TypedDict, closed=True): """Configuration for the regular chunk grid. - No chunk extent is negative. "The chunk shape elements are non-zero - when the corresponding dimensions of the arrays have non-zero length": - an extent of 0 is right for a dimension of length 0, which zarr-python - 3.0 and 3.1 wrote, and which a grid alone cannot tell from one that is - not; the shape rules can, beside the array's shape. + Every chunk length is at least 1, along a dimension of length 0 too: + "Chunk sizes must be greater than zero" + (https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v3/chunk-grids/regular-grid/index.rst#L40). + The core spec's "The chunk shape elements are non-zero when the + corresponding dimensions of the arrays have non-zero length" + (https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v3/core/index.rst#L284-L285) + says less of an empty dimension, and allows nothing the grid does not. """ - chunk_shape: tuple[Annotated[int, Ge(0)], ...] + chunk_shape: tuple[Annotated[int, Ge(1)], ...] class RegularChunkGridObject(TypedDict, closed=True): @@ -54,14 +56,11 @@ class RegularChunkGridObject(TypedDict, closed=True): def _shape_rules( configuration: RegularChunkGridConfiguration, nested: Nested, shape: tuple[int, ...] ) -> Iterator[ValidationProblem]: - """A chunk length for each of the array's dimensions, and 0 only for a dimension of length 0. + """A chunk length for each of the array's dimensions. "The dimensionality of the grid is the same as the dimensionality of the array" - (https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v3/chunk-grids/regular-grid/index.rst#L29-L31), - and "The chunk shape elements are non-zero when the corresponding - dimensions of the arrays have non-zero length" - (https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v3/core/index.rst#L284-L285). + (https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v3/chunk-grids/regular-grid/index.rst#L29-L31). """ chunk_shape = configuration["chunk_shape"] if len(chunk_shape) != len(shape): @@ -70,15 +69,6 @@ def _shape_rules( f"expected one chunk length per dimension of shape, got {len(chunk_shape)}", "invalid_value", ) - return - for axis, (length, extent) in enumerate(zip(chunk_shape, shape, strict=True)): - if length == 0 and extent != 0: - yield ValidationProblem( - ("chunk_shape", axis), - f"expected a chunk length >= 1 for a dimension of length {extent}, got 0", - "invalid_value", - ctx={"ge": 1}, - ) def _chunk_lengths( diff --git a/packages/zarr-metadata/src/zarr_metadata/v3/definition.py b/packages/zarr-metadata/src/zarr_metadata/v3/definition.py index ccd233b56b..65e766a55b 100644 --- a/packages/zarr-metadata/src/zarr_metadata/v3/definition.py +++ b/packages/zarr-metadata/src/zarr_metadata/v3/definition.py @@ -268,6 +268,21 @@ def acme_lz4_rules( again, and gives what `canonicalize` gives: None for a field with a problem. +**JSON Schema.** `field_json_schema(CodecDefinition, SCOPE)` writes the +fields of one kind a scope reads as a JSON Schema, draft 2020-12, for a +validator in another language or an editor: each definition's field -- +its name, its configuration as its TypedDict says, bounds and all, a +`must_understand` of `true`, and its bare name when it needs no +configuration -- and a name nothing in scope claims, with any +configuration. A field a configuration holds is written in the same +scope, and a member taking codecs of static size only takes those. The +rules are not in it, so a field it accepts may still have a problem; +one `resolve` reads without a problem, it accepts, as JSON: arrays as +lists, as a parser gives them. Each configuration +TypedDict, and each field alias, is written once, in `$defs`, under its +name. `node_metadata_json_schema_v3`, in `zarr_metadata.model`, writes a +whole `zarr.json`, its fill value held to its data type's. + A definition checks itself when it is built, and each of these is a `TypeError` saying what is wrong: a `configuration` that is not a TypedDict, says nothing of the keys it does not declare, or has a member @@ -275,31 +290,36 @@ def acme_lz4_rules( that is not a string; a member declared as a function -- `rules`, `canonical`, `fill_value_rules`, `storage`, `shape_rules`, `chunk_lengths`, `chunk_rules`, `transition`, `pipelines` -- that is not -one; a data type's `fill_value` no checker reads; a codec `kind` that is -not one of the three, or a `size` that is not `"static"` or `"dynamic"`; -a function no codec of its kind is asked -- chunk rules or pipelines of -a bytes -> bytes codec, which is handed bytes, or a `transition` of a -codec that hands on bytes; a data type named as raw bits of one size are -written. A scope refuses a definition of no kind. Nothing happens at -class creation. +one; a data type's `fill_value` no checker reads, or one holding a +metadata field, which a value of the data type never is; a codec `kind` +that is not one of the three, or a `size` that is not `"static"` or +`"dynamic"`; a function no codec of its kind is asked -- chunk rules or +pipelines of a bytes -> bytes codec, which is handed bytes, or a +`transition` of a codec that hands on bytes; a data type named as raw +bits of one size are written. A scope refuses a definition of no kind. +Nothing happens at class creation. """ from zarr_metadata._common import JSONValue from zarr_metadata._json import MetadataValidationError, ProblemKind, ValidationProblem, shown from zarr_metadata._typed_json import Loc, check -from zarr_metadata.v3._common import ZarrV3MetadataFieldJSON +from zarr_metadata.v3._common import ( + ChunkGridField, + ChunkKeyEncodingField, + CodecField, + DataTypeField, + StaticCodecField, + StorageTransformerField, + ZarrV3MetadataFieldJSON, +) from zarr_metadata.v3._definition import ( Chunk, ChunkGridDefinition, - ChunkGridField, ChunkKeyEncodingDefinition, - ChunkKeyEncodingField, CodecDefinition, - CodecField, CodecKind, CodecSize, DataTypeDefinition, - DataTypeField, Definition, EmptyConfiguration, Lengths, @@ -307,15 +327,14 @@ class creation. Read, Refused, Resolved, - StaticCodecField, StorageClass, StorageTransformerDefinition, - StorageTransformerField, Unclaimed, canonical_of, canonicalize, chunk_grid_lengths, configuration_of, + field_json_schema, fields_of, fill_value_problems, resolve, @@ -364,6 +383,7 @@ class creation. "check", "chunk_grid_lengths", "configuration_of", + "field_json_schema", "fields_of", "fill_value_problems", "read_pipeline", diff --git a/packages/zarr-metadata/tests/test_json_schema.py b/packages/zarr-metadata/tests/test_json_schema.py new file mode 100644 index 0000000000..d2baa4b225 --- /dev/null +++ b/packages/zarr-metadata/tests/test_json_schema.py @@ -0,0 +1,899 @@ +"""JSON Schemas of what the package reads: a TypedDict, a field in a scope, a `zarr.json`. + +A schema is held to the reader it is written from: what the reader finds +nothing wrong with, the schema accepts, and what the type says -- a +bound, a key a closed TypedDict does not declare, a name a scope claims +-- the schema says too. What only a rule says, the schema does not. +""" + +from __future__ import annotations + +import json +import types +from collections.abc import Callable, Mapping +from typing import Annotated, Any, Literal, NewType, NotRequired, cast + +import pytest +from annotated_types import Ge, Gt, Interval, Le, Lt, MinLen +from hypothesis import HealthCheck, event, given, settings +from hypothesis import strategies as st +from jsonschema import Draft202012Validator +from typing_extensions import Doc, TypeAliasType, TypedDict + +from tests.v3.test_every_definition import CASES, KINDS +from zarr_metadata._common import JSONValue +from zarr_metadata._typed_json import Schemas +from zarr_metadata.model import node_metadata_json_schema_v3, validate_node_metadata_v3 +from zarr_metadata.typed_json import check, json_schema +from zarr_metadata.v3.codec.gzip import GzipCodecConfiguration +from zarr_metadata.v3.definition import ( + CORE, + CORE_AND_EXTENSIONS, + ChunkGridDefinition, + ChunkKeyEncodingDefinition, + CodecDefinition, + Context, + DataTypeDefinition, + Definition, + StorageTransformerDefinition, + field_json_schema, + resolve, +) + +DIALECT = "https://json-schema.org/draft/2020-12/schema" + +# Built at run time, where a type checker reads each special form's +# arguments as it would in an annotation. +_annotated: Any = Annotated +_new_type: Callable[[str, object], object] = cast("Any", NewType) +_alias_type: Callable[[str, object], object] = cast("Any", TypeAliasType) + +Level = TypeAliasType("Level", Annotated[int, Interval(ge=0, le=9)]) +Tree = TypeAliasType("Tree", "int | tuple[Tree, ...]") +Name = NewType("Name", str) +Count = _new_type("Count", Annotated[int, Ge(0)]) + + +class Inner(TypedDict, closed=True): + x: int + + +class Open(TypedDict, closed=False): + x: NotRequired[int] + + +class Extra(TypedDict, extra_items=int): + x: str + + +class ExtraJSON(TypedDict, extra_items=JSONValue): + x: str + + +class Node(TypedDict, closed=True): + children: tuple[Node, ...] + + +def _closed(name: str, annotations: dict[str, object]) -> type: + """A closed TypedDict of these required keys, made here, so its annotations resolve in this module.""" + base: object = TypedDict + namespace = {"__annotations__": annotations, "__module__": __name__} + return types.new_class(name, (base,), {"closed": True}, lambda body: body.update(namespace)) + + +def _holding(annotation: object) -> type: + """A closed TypedDict of one required key, `value`, holding `annotation`.""" + return _closed("Holder", {"value": annotation}) + + +# Another class of the name `Inner`, which the schema tells from the first. +Both = _closed("Both", {"first": Inner, "second": _closed("Inner", {"y": str})}) + + +def _held(member: dict[str, Any], defs: dict[str, Any] | None = None) -> dict[str, Any]: + """The schema of `_holding` an annotation whose schema is `member`.""" + schema: dict[str, Any] = { + "$schema": DIALECT, + "type": "object", + "properties": {"value": member}, + "required": ["value"], + "additionalProperties": False, + } + return schema if defs is None else {**schema, "$defs": defs} + + +INNER = { + "type": "object", + "properties": {"x": {"type": "integer"}}, + "required": ["x"], + "additionalProperties": False, +} + +SHAPES: list[tuple[str, type, dict[str, Any]]] = [ + ("int", _holding(int), _held({"type": "integer"})), + ("float", _holding(float), _held({"type": "number"})), + ("bool", _holding(bool), _held({"type": "boolean"})), + ("str", _holding(str), _held({"type": "string"})), + ("null", _holding(None), _held({"type": "null"})), + ("json", _holding(JSONValue), _held({})), + ("one-value", _holding(Literal["a"]), _held({"const": "a"})), + # Sorted as the checker sorts them, so the order written does not show. + ("values", _holding(Literal["b", "a", 1]), _held({"enum": ["a", "b", 1]})), + ("array", _holding(tuple[int, ...]), _held({"type": "array", "items": {"type": "integer"}})), + ("array-of-json", _holding(tuple[JSONValue, ...]), _held({"type": "array"})), + ( + "pair", + _holding(tuple[int, str]), + _held( + { + "type": "array", + "prefixItems": [{"type": "integer"}, {"type": "string"}], + "items": False, + "minItems": 2, + } + ), + ), + ("empty-array", _holding(tuple[()]), _held({"type": "array", "maxItems": 0})), + ( + "union", + _holding(int | None), + _held({"anyOf": [{"type": "integer"}, {"type": "null"}]}), + ), + ( + "mapping", + _holding(Mapping[str, int]), + _held({"type": "object", "additionalProperties": {"type": "integer"}}), + ), + ("mapping-of-json", _holding(Mapping[str, JSONValue]), _held({"type": "object"})), + ("newtype", _holding(Name), _held({"type": "string"})), + ( + "interval", + _holding(Annotated[int, Interval(ge=0, le=9)]), + _held({"type": "integer", "minimum": 0, "maximum": 9}), + ), + ( + "exclusive-bounds", + _holding(Annotated[float, Gt(0), Lt(1)]), + _held({"type": "number", "exclusiveMinimum": 0, "exclusiveMaximum": 1}), + ), + ( + "doc", + _holding(Annotated[int, Ge(0), Doc("a count")]), + _held({"type": "integer", "description": "a count", "minimum": 0}), + ), + # A note that is no `Doc` says nothing a schema writes. + ("note", _holding(Annotated[str, "a note"]), _held({"type": "string"})), + # A value is held to its type's bounds and to its `NewType`'s: of two + # of one keyword, the stricter. + ( + "bounds-on-a-bounded-newtype", + _holding(_annotated[Count, Ge(-5), Lt(9)]), + _held({"type": "integer", "minimum": 0, "exclusiveMaximum": 9}), + ), + ( + "alias", + _holding(Level), + _held( + {"$ref": "#/$defs/Level"}, + {"Level": {"type": "integer", "minimum": 0, "maximum": 9}}, + ), + ), + ( + "alias-holding-itself", + _holding(Tree), + _held( + {"$ref": "#/$defs/Tree"}, + { + "Tree": { + "anyOf": [ + {"type": "integer"}, + {"type": "array", "items": {"$ref": "#/$defs/Tree"}}, + ] + } + }, + ), + ), + ("typeddict", _holding(Inner), _held({"$ref": "#/$defs/Inner"}, {"Inner": INNER})), + ( + "open", + Open, + {"$schema": DIALECT, "type": "object", "properties": {"x": {"type": "integer"}}}, + ), + ( + "extra-items", + Extra, + { + "$schema": DIALECT, + "type": "object", + "properties": {"x": {"type": "string"}}, + "required": ["x"], + "additionalProperties": {"type": "integer"}, + }, + ), + ( + "extra-items-of-json", + ExtraJSON, + { + "$schema": DIALECT, + "type": "object", + "properties": {"x": {"type": "string"}}, + "required": ["x"], + }, + ), + # A TypedDict that holds itself is referred to, not written in place. + ( + "holding-itself", + Node, + { + "$schema": DIALECT, + "$ref": "#/$defs/Node", + "$defs": { + "Node": { + "type": "object", + "properties": { + "children": {"type": "array", "items": {"$ref": "#/$defs/Node"}} + }, + "required": ["children"], + "additionalProperties": False, + } + }, + }, + ), + ( + "one-name-two-classes", + Both, + { + "$schema": DIALECT, + "type": "object", + "properties": { + "first": {"$ref": "#/$defs/Inner"}, + "second": {"$ref": "#/$defs/Inner2"}, + }, + "required": ["first", "second"], + "additionalProperties": False, + "$defs": { + "Inner": INNER, + "Inner2": { + "type": "object", + "properties": {"y": {"type": "string"}}, + "required": ["y"], + "additionalProperties": False, + }, + }, + }, + ), + ( + "a-definition-s-configuration", + GzipCodecConfiguration, + { + "$schema": DIALECT, + "type": "object", + "properties": {"level": {"type": "integer", "minimum": 0, "maximum": 9}}, + "required": ["level"], + "additionalProperties": False, + }, + ), +] + + +@pytest.mark.parametrize( + ("shape", "expected"), [case[1:] for case in SHAPES], ids=[case[0] for case in SHAPES] +) +def test_json_schema_writes_what_check_reads(shape: type, expected: dict[str, Any]) -> None: + schema = json_schema(shape) + assert schema == expected + Draft202012Validator.check_schema(schema) + assert json.loads(json.dumps(schema)) == schema + + +def test_json_schema_refuses_what_is_not_a_typeddict() -> None: + with pytest.raises(TypeError, match="is not a TypedDict"): + json_schema(int) + + +def test_json_schema_refuses_what_check_cannot_read() -> None: + unread = _holding(bytes) + with pytest.raises(TypeError) as raised: + check({}, unread) + with pytest.raises(TypeError, match=str(raised.value)): + json_schema(unread) + + +def test_error_a_schema_that_fails_to_write_leaves_nothing_behind() -> None: + # Reserved before it is written, so that it can refer to itself, and + # given up when the writing fails, so that nothing refers to an empty + # schema, which takes anything. + schemas = Schemas() + unread = _closed("Unread", {"value": bytes}) + with pytest.raises(TypeError, match="is not a shape JSON takes"): + schemas.of(unread) + assert schemas.document({}) == {"$schema": DIALECT} + with pytest.raises(TypeError, match="is not a shape JSON takes"): + schemas.of(unread) + + +def test_json_schema_refuses_metadata_check_does_not_hold_a_value_to() -> None: + with pytest.raises(TypeError, match="is not a constraint the checker reads"): + json_schema(_holding(Annotated[tuple[int, ...], MinLen(1)])) + + +_PROPERTY = settings(max_examples=300, deadline=None, suppress_health_check=[HealthCheck.too_slow]) + +_MARKERS: dict[str, Callable[[int], object]] = {"ge": Ge, "gt": Gt, "le": Le, "lt": Lt} +_LOW = st.none() | st.tuples(st.sampled_from(("ge", "gt")), st.integers(-3, 3)) +_HIGH = st.none() | st.tuples(st.sampled_from(("le", "lt")), st.integers(-3, 3)) +_LAYERS = st.lists( + st.tuples(st.sampled_from(("annotated", "newtype", "alias")), _LOW, _HIGH), + min_size=1, + max_size=3, +) +_NUMBERS = st.integers(-5, 5) | st.floats(-5, 5, allow_nan=False) | st.sampled_from((True, "1")) + + +def _integral_as_int(value: object) -> object: + """`value` as JSON Schema reads a number: `1.0` is the integer 1.""" + return int(value) if isinstance(value, float) and value.is_integer() else value + + +@_PROPERTY +@given(base=st.sampled_from((int, float)), layers=_LAYERS, values=st.lists(_NUMBERS, max_size=8)) +def test_bounds_in_layers_are_written_as_check_holds_a_value_to_them( + base: type, layers: list[tuple[str, object, object]], values: list[object] +) -> None: + # A number's bounds, on it, on a `NewType` of it, on an alias of it, + # layer on layer: the schema accepts what `check` does, a number with no + # fraction read as the integer it equals. + annotation: object = base + for index, (how, low, high) in enumerate(layers): + sides = [cast("tuple[str, int]", side) for side in (low, high) if side is not None] + bounds = [_MARKERS[name](bound) for name, bound in sides] + bounded = _annotated[(annotation, *bounds)] if len(bounds) != 0 else annotation + if how == "newtype": + annotation = _new_type(f"Bounded{index}", bounded) + elif how == "alias": + annotation = _alias_type(f"Bounded{index}", bounded) + else: + annotation = bounded + holder = _holding(annotation) + try: + schema = json_schema(holder) + except TypeError: + # A second bound from one side, which neither reads. + event("refused") + with pytest.raises(TypeError): + check({}, holder) + return + validator = Draft202012Validator(schema) + for value in values: + accepted = check({"value": _integral_as_int(value)}, holder)[1] == () + event("accepted" if accepted else "refused a value") + assert validator.is_valid(cast("Any", {"value": value})) == accepted, (value, schema) + + +# --- values changed in one place -------------------------------------------- + +_GONE = object() +_NAMES = ("r16", "r16\n", "r*", "gzip", "bytes", "crc32c", "int8", "regular", "default", "acme.x") +_KEYS = ("name", "configuration", "must_understand", "level", "x") +_SCALARS = ( + st.none() + | st.booleans() + | st.integers(-3, 300) + | st.floats(-3, 3, allow_nan=False) + | st.sampled_from(_NAMES) + | st.text(max_size=3) +) +_VALUES = st.recursive( + _SCALARS, + lambda inner: ( + st.lists(inner, max_size=2) + | st.dictionaries(st.sampled_from(_KEYS) | st.text(max_size=2), inner, max_size=2) + ), + max_leaves=4, +) + + +def _places(value: object, path: tuple[str | int, ...] = ()) -> list[tuple[str | int, ...]]: + """Every place in `value`: itself, and each member or entry inside it, depth first.""" + found = [path] + if isinstance(value, dict): + for key, entry in cast("dict[str, object]", value).items(): + found += _places(entry, (*path, key)) + elif isinstance(value, list): + for index, entry in enumerate(cast("list[object]", value)): + found += _places(entry, (*path, index)) + return found + + +def _at(value: object, path: tuple[str | int, ...]) -> object: + for step in path: + value = cast("Any", value)[step] + return value + + +def _put(value: object, path: tuple[str | int, ...], new: object) -> object: + """`value` with what sits at `path` replaced by `new`, or removed when `new` is `_GONE`.""" + if len(path) == 0: + return new + step, rest = path[0], path[1:] + copy: Any = ( + dict(cast("dict[str, object]", value)) + if isinstance(value, dict) + else list(cast("list[object]", value)) + ) + if len(rest) == 0 and new is _GONE: + del copy[step] + else: + copy[step] = _put(copy[step], rest, new) + return copy + + +@st.composite +def _changed(draw: st.DrawFn, value: object) -> object: + """`value` changed in one or two places: something replaced, dropped or added, a string with a character more, or a field renamed.""" + for _ in range(draw(st.integers(1, 2))): + path = draw(st.sampled_from(_places(value))) + here = _at(value, path) + how = draw(st.sampled_from(("replace", "drop", "add", "extend", "rename"))) + if how == "drop" and len(path) != 0: + value = _put(value, path, _GONE) + elif how == "add" and isinstance(here, dict): + members = cast("dict[str, object]", here) + value = _put(value, path, {**members, draw(st.sampled_from(_KEYS)): draw(_VALUES)}) + elif how == "add" and isinstance(here, list): + value = _put(value, path, [*cast("list[object]", here), draw(_VALUES)]) + elif how == "extend" and isinstance(here, str): + value = _put(value, path, here + draw(st.sampled_from(("\n", " ", "0", "x")))) + elif how == "rename" and isinstance(here, dict) and "name" in here: + value = _put(value, (*path, "name"), draw(st.sampled_from(_NAMES))) + elif how == "rename" and isinstance(here, str): + value = _put(value, path, draw(st.sampled_from(_NAMES))) + else: + value = _put(value, path, draw(_VALUES)) + return value + + +_SCOPES = (CORE, CORE_AND_EXTENSIONS, Context.of()) +_FIELD_BASES: tuple[tuple[str, object], ...] = ( + *CASES, + # Names nothing claims, whose configuration nothing judges. + ("codecs:acme.x", {"name": "acme.x", "configuration": {"x": [1]}}), + ("data_type:acme.t", {"name": "acme.t", "configuration": {"bits": 8}}), + ("chunk_grid:acme.g", {"name": "acme.g", "configuration": {"chunk_shape": [0]}}), +) +_FIELD_SCHEMAS: dict[tuple[type[Definition[Any]], int], Draft202012Validator] = {} + + +def _field_validator(kind: type[Definition[Any]], scope: int) -> Draft202012Validator: + if (kind, scope) not in _FIELD_SCHEMAS: + schema = field_json_schema(kind, _SCOPES[scope]) + _FIELD_SCHEMAS[(kind, scope)] = Draft202012Validator(schema) + return _FIELD_SCHEMAS[(kind, scope)] + + +# --- a field in a scope ----------------------------------------------------- + +_INDEXED = {"chunk_shape": [2], "codecs": ["bytes"]} + +# Each field, the kind and scope it is read in, whether the schema accepts +# it, and whether the scope reads it without a problem. Where the two +# differ, a rule found what the schema cannot say. +FIELDS: list[tuple[str, object, type[Definition[Any]], Context, bool, bool]] = [ + ("object", {"name": "gzip", "configuration": {"level": 5}}, CodecDefinition, CORE, True, True), + ( + "out-of-bounds", + {"name": "gzip", "configuration": {"level": 12}}, + CodecDefinition, + CORE, + False, + False, + ), + ( + "unknown-key", + {"name": "gzip", "configuration": {"level": 5, "window": 15}}, + CodecDefinition, + CORE, + False, + False, + ), + ("bare-name-needing-a-configuration", "gzip", CodecDefinition, CORE, False, False), + ("bare-name", "bytes", CodecDefinition, CORE, True, True), + ( + "understood", + {"name": "gzip", "configuration": {"level": 5}, "must_understand": True}, + CodecDefinition, + CORE, + True, + True, + ), + ( + "not-understood", + {"name": "gzip", "configuration": {"level": 5}, "must_understand": False}, + CodecDefinition, + CORE, + False, + False, + ), + ( + "stray-member", + {"name": "gzip", "configuration": {"level": 5}, "version": 2}, + CodecDefinition, + CORE, + False, + False, + ), + ( + "unclaimed", + {"name": "zfpy", "configuration": {"mode": 4}}, + CodecDefinition, + CORE, + True, + True, + ), + ("unclaimed-bare", "zfpy", CodecDefinition, CORE, True, True), + ( + "unclaimed-not-understood", + {"name": "zfpy", "must_understand": False}, + CodecDefinition, + CORE, + False, + False, + ), + # Nothing in an empty scope claims gzip, so nothing judges it. + ( + "out-of-scope", + {"name": "gzip", "configuration": {"level": 12}}, + CodecDefinition, + Context.of(), + True, + True, + ), + ( + "a-rule", + { + "name": "blosc", + "configuration": {"cname": "lz4", "clevel": 1, "shuffle": "shuffle", "blocksize": 0}, + }, + CodecDefinition, + CORE, + True, + False, + ), + ( + "static-index-codec", + { + "name": "sharding_indexed", + "configuration": {**_INDEXED, "index_codecs": ["bytes", "crc32c"]}, + }, + CodecDefinition, + CORE, + True, + True, + ), + ( + "dynamic-index-codec", + { + "name": "sharding_indexed", + "configuration": { + **_INDEXED, + "index_codecs": [{"name": "gzip", "configuration": {"level": 1}}], + }, + }, + CodecDefinition, + CORE, + False, + False, + ), + ( + "unclaimed-index-codec", + { + "name": "sharding_indexed", + "configuration": {**_INDEXED, "index_codecs": ["bytes", "acme.sum"]}, + }, + CodecDefinition, + CORE, + True, + True, + ), + ( + "inner-codec-out-of-bounds", + { + "name": "sharding_indexed", + "configuration": { + **_INDEXED, + "codecs": ["bytes", {"name": "gzip", "configuration": {"level": 12}}], + "index_codecs": ["bytes"], + }, + }, + CodecDefinition, + CORE, + False, + False, + ), + ("raw-bits", "r16", DataTypeDefinition, CORE, True, True), + ("raw-bits-object", {"name": "r16", "configuration": {}}, DataTypeDefinition, CORE, True, True), + # What a raw-bits name carries is not written beside it. + ( + "raw-bits-configured", + {"name": "r16", "configuration": {"bits": 16}}, + DataTypeDefinition, + CORE, + False, + False, + ), + ("raw-bits-of-a-size-the-spec-refuses", "r12", DataTypeDefinition, CORE, True, False), + ("raw-bits-notation", "r*", DataTypeDefinition, CORE, True, True), + # Matched to the end of the name, as the package matches it, so a + # validator that matches as Python does takes no final newline for it: + # a name nothing claims, which any configuration goes with. + ( + "raw-bits-and-a-newline", + {"name": "r16\n", "configuration": {"x": 1}}, + DataTypeDefinition, + CORE, + True, + True, + ), + ( + "struct-field-out-of-bounds", + { + "name": "struct", + "configuration": { + "fields": [ + { + "name": "t", + "data_type": { + "name": "numpy.datetime64", + "configuration": {"unit": "s", "scale_factor": 0}, + }, + } + ] + }, + }, + DataTypeDefinition, + CORE_AND_EXTENSIONS, + False, + False, + ), + ( + "grid-out-of-bounds", + {"name": "regular", "configuration": {"chunk_shape": [0]}}, + ChunkGridDefinition, + CORE, + False, + False, + ), + ( + "encoding", + {"name": "v2", "configuration": {"separator": "/"}}, + ChunkKeyEncodingDefinition, + CORE, + True, + True, + ), + ("transformer", {"name": "acme.cache"}, StorageTransformerDefinition, CORE, True, True), +] + + +@pytest.mark.parametrize( + ("field", "kind", "scope", "accepted", "clean"), + [case[1:] for case in FIELDS], + ids=[case[0] for case in FIELDS], +) +def test_field_json_schema_says_what_the_scope_reads( + field: object, kind: type[Definition[Any]], scope: Context, accepted: bool, clean: bool +) -> None: + schema = field_json_schema(kind, scope) + Draft202012Validator.check_schema(schema) + assert Draft202012Validator(schema).is_valid(cast("Any", field)) == accepted + assert (resolve(field, kind, scope)[1] == ()) == clean + + +@pytest.mark.parametrize( + ("key", "field"), CASES, ids=[f"{key}:{index}" for index, (key, _) in enumerate(CASES)] +) +def test_every_example_field_is_one_its_schema_accepts(key: str, field: object) -> None: + schema = field_json_schema(KINDS[key.split(":")[0]], CORE_AND_EXTENSIONS) + assert Draft202012Validator(schema).is_valid(cast("Any", field)) + + +@_PROPERTY +@given(st.data()) +def test_a_field_read_without_a_problem_is_one_its_schema_accepts(data: st.DataObject) -> None: + # Each example changed in one place, in one of three scopes: whatever + # the scope still reads without a problem, the schema accepts. + key, example = data.draw(st.sampled_from(_FIELD_BASES), label="example") + kind = KINDS[key.split(":")[0]] + scope = data.draw(st.sampled_from(range(len(_SCOPES))), label="scope") + field = data.draw(_changed(json.loads(json.dumps(example))), label="field") + read = resolve(field, kind, _SCOPES[scope])[1] == () + event("read without a problem" if read else "a problem") + if read: + assert _field_validator(kind, scope).is_valid(cast("Any", field)) + + +def test_field_json_schema_refuses_what_is_not_a_kind() -> None: + with pytest.raises(TypeError, match="is not a kind of metadata"): + field_json_schema(Definition, CORE) + + +# --- a zarr.json ------------------------------------------------------------ + +ARRAY: dict[str, Any] = { + "zarr_format": 3, + "node_type": "array", + "shape": [4, 4], + "data_type": "int8", + "chunk_grid": {"name": "regular", "configuration": {"chunk_shape": [2, 2]}}, + "chunk_key_encoding": {"name": "default"}, + "fill_value": 0, + "codecs": [{"name": "bytes"}, {"name": "gzip", "configuration": {"level": 5}}], +} +FLOATS: dict[str, Any] = { + **ARRAY, + "data_type": "float32", + "codecs": [{"name": "bytes", "configuration": {"endian": "little"}}], +} +GROUP: dict[str, Any] = {"zarr_format": 3, "node_type": "group"} + + +def _consolidated(**metadata: object) -> dict[str, Any]: + return { + **GROUP, + "consolidated_metadata": {"kind": "inline", "must_understand": False, "metadata": metadata}, + } + + +# Each document, whether the schema accepts it, and whether +# `validate_node_metadata_v3` finds nothing wrong with it. +DOCUMENTS: list[tuple[str, dict[str, Any], bool, bool]] = [ + ("array", ARRAY, True, True), + ("group", {**GROUP, "attributes": {"a": [1, None]}}, True, True), + ("fill-value-out-of-range", {**ARRAY, "fill_value": 300}, False, False), + ("fill-value-of-the-wrong-type", {**ARRAY, "fill_value": "0"}, False, False), + ("float-fill-value", {**FLOATS, "fill_value": "NaN"}, True, True), + ("hex-fill-value", {**FLOATS, "fill_value": "0x7fc00000"}, True, True), + # A string that is no float's is a rule's to find. + ("fill-value-a-rule-refuses", {**FLOATS, "fill_value": "nan"}, True, False), + ("unclaimed-data-type", {**ARRAY, "data_type": "acme.int7", "fill_value": "any"}, True, True), + ( + "a-name-raw-bits-and-a-newline", + {**ARRAY, "data_type": "r16\n", "fill_value": "any"}, + True, + True, + ), + ("raw-bits", {**ARRAY, "data_type": "r16", "fill_value": [0, 255]}, True, True), + ( + "raw-bits-byte-out-of-range", + {**ARRAY, "data_type": "r16", "fill_value": [0, 256]}, + False, + False, + ), + ( + "codec-not-understood", + {**ARRAY, "codecs": [{"name": "bytes", "must_understand": False}]}, + False, + False, + ), + ( + "codec-out-of-bounds", + {**ARRAY, "codecs": [{"name": "bytes"}, {"name": "gzip", "configuration": {"level": 12}}]}, + False, + False, + ), + ("negative-shape", {**ARRAY, "shape": [-1, 4]}, False, False), + ("wrong-format", {**ARRAY, "zarr_format": 2}, False, False), + ( + "no-node-type", + {key: value for key, value in ARRAY.items() if key != "node_type"}, + False, + False, + ), + ("an-extension-field", {**ARRAY, "acme": {"must_understand": False}}, True, True), + # What members read together say is the rules'. + ("names-for-another-rank", {**ARRAY, "dimension_names": ["x"]}, True, False), + ( + "codecs-out-of-order", + {**ARRAY, "codecs": [{"name": "gzip", "configuration": {"level": 5}}, "bytes"]}, + True, + False, + ), + ("consolidated", _consolidated(a=ARRAY, b=GROUP), True, True), + ("consolidated-null", {**GROUP, "consolidated_metadata": None}, True, True), + ("consolidated-bad-array", _consolidated(a={**ARRAY, "fill_value": 300}), False, False), + ( + "consolidated-within-consolidated", + _consolidated(b=_consolidated(c={**ARRAY, "fill_value": 300})), + False, + False, + ), + ( + "consolidated-kind", + { + **GROUP, + "consolidated_metadata": {"kind": "sidecar", "must_understand": False, "metadata": {}}, + }, + False, + False, + ), +] + +NODE_SCHEMA = node_metadata_json_schema_v3() + + +@pytest.mark.parametrize( + ("document", "accepted", "clean"), + [case[1:] for case in DOCUMENTS], + ids=[case[0] for case in DOCUMENTS], +) +def test_node_metadata_json_schema_says_what_a_zarr_json_holds( + document: dict[str, Any], accepted: bool, clean: bool +) -> None: + assert Draft202012Validator(NODE_SCHEMA).is_valid(document) == accepted + assert (validate_node_metadata_v3(document) == ()) == clean + + +_BASES: tuple[dict[str, Any], ...] = ( + ARRAY, + FLOATS, + {**ARRAY, "data_type": "r16", "fill_value": [0, 0]}, + { + **ARRAY, + "codecs": [ + { + "name": "sharding_indexed", + "configuration": { + "chunk_shape": [1, 1], + "codecs": ["bytes", {"name": "gzip", "configuration": {"level": 1}}], + "index_codecs": ["bytes", "crc32c"], + }, + } + ], + }, + _consolidated(a=ARRAY, b=GROUP), +) +_NODE_SCHEMAS: dict[int, Draft202012Validator] = {} + + +def _node_validator(scope: int) -> Draft202012Validator: + if scope not in _NODE_SCHEMAS: + _NODE_SCHEMAS[scope] = Draft202012Validator( + node_metadata_json_schema_v3(context=_SCOPES[scope]) + ) + return _NODE_SCHEMAS[scope] + + +@_PROPERTY +@given(st.data()) +def test_a_document_read_without_a_problem_is_one_its_schema_accepts(data: st.DataObject) -> None: + # Each document changed in one place, in one of three scopes: whatever + # `validate_node_metadata_v3` finds nothing wrong with, the schema accepts. + base = data.draw(st.sampled_from(_BASES), label="base") + scope = data.draw(st.sampled_from(range(len(_SCOPES))), label="scope") + document = data.draw(_changed(json.loads(json.dumps(base))), label="document") + read = validate_node_metadata_v3(document, context=_SCOPES[scope]) == () + event("read without a problem" if read else "a problem") + if read: + assert _node_validator(scope).is_valid(cast("Any", document)) + + +@pytest.mark.parametrize( + "scope", + [CORE, CORE_AND_EXTENSIONS, Context.of()], + ids=["core", "core-and-extensions", "empty"], +) +def test_every_schema_is_a_json_schema_written_the_same_each_time(scope: Context) -> None: + schema = node_metadata_json_schema_v3(context=scope) + Draft202012Validator.check_schema(schema) + assert json.loads(json.dumps(schema)) == schema + assert schema == node_metadata_json_schema_v3(context=scope) + defs = cast("dict[str, JSONValue]", schema["$defs"]) + assert {"ZarrV3ArrayMetadataJSON", "ZarrV3GroupMetadataJSON"} <= defs.keys() + for kind in ( + CodecDefinition, + DataTypeDefinition, + ChunkGridDefinition, + ChunkKeyEncodingDefinition, + StorageTransformerDefinition, + ): + Draft202012Validator.check_schema(field_json_schema(kind, scope)) diff --git a/packages/zarr-metadata/tests/test_public_api.py b/packages/zarr-metadata/tests/test_public_api.py index 22fdd0c863..82448e356c 100644 --- a/packages/zarr-metadata/tests/test_public_api.py +++ b/packages/zarr-metadata/tests/test_public_api.py @@ -305,6 +305,7 @@ def test_all_is_grouped_and_unique() -> None: "HexFloat16", "HexFloat32", "HexFloat64", + "JSONSchema", "JSONValue", "MetadataValidationError", "NumpyDatetime64", diff --git a/packages/zarr-metadata/tests/test_typed_json_properties.py b/packages/zarr-metadata/tests/test_typed_json_properties.py index 00a137f560..b228546baf 100644 --- a/packages/zarr-metadata/tests/test_typed_json_properties.py +++ b/packages/zarr-metadata/tests/test_typed_json_properties.py @@ -13,27 +13,31 @@ Values are JSON, their depth capped: drawn from the spec, so they conform; drawn at large, so most do not; or drawn from the spec and changed in one place, so they nearly do. On every one the checker has to -agree with the reference, and the two builds with each other. +agree with the reference, and the two builds with each other; so does +the JSON Schema written of the type, but for JSON Schema's own reading of +a number with no fraction, `1.0`, as the integer it equals. """ from __future__ import annotations import copy import itertools +import json import sys import types import typing from collections.abc import Mapping from dataclasses import dataclass, field -from typing import Annotated, Literal, NewType, NotRequired, Required, TypeAlias, Union, cast +from typing import Annotated, Any, Literal, NewType, NotRequired, Required, TypeAlias, Union, cast import pytest from hypothesis import HealthCheck, event, find, given, note, settings from hypothesis import strategies as st +from jsonschema import Draft202012Validator from typing_extensions import ReadOnly, TypeAliasType, TypedDict from zarr_metadata._common import JSONValue -from zarr_metadata._typed_json import Loc, Parsed, no_leaf, parser, typeddict_keys +from zarr_metadata._typed_json import Loc, Parsed, Schemas, no_leaf, parser, typeddict_keys # --- specs ----------------------------------------------------------------- @@ -664,6 +668,38 @@ def test_the_checker_agrees_with_the_reference(data: st.DataObject) -> None: assert conforms(typed, spec) +def _as_json_schema_reads(value: object) -> object: + """`value` as JSON Schema reads it: a number with no fraction is an integer, `1.0` the `1` it equals.""" + if isinstance(value, float) and value.is_integer(): + return int(value) + if isinstance(value, list): + return [_as_json_schema_reads(entry) for entry in cast("list[object]", value)] + if isinstance(value, dict): + entries = cast("dict[str, object]", value) + return {key: _as_json_schema_reads(entry) for key, entry in entries.items()} + return value + + +@_EXAMPLES +@given(st.data()) +def test_the_json_schema_agrees_with_the_reference(data: st.DataObject) -> None: + # The schema of a type accepts exactly the values of it, as the checker + # does, but that JSON Schema takes `1.0` for the integer it equals, which + # the checker does not. + spec = data.draw(specs(_DEPTH), label="spec") + build = Build(data.draw(st.sampled_from(_MODES), label="mode")) + annotation = build.annotation(spec) + note(build.text_of()) + schemas = Schemas() + schema = schemas.document(schemas.of(annotation)) + note(json.dumps(schema, indent=1)) + Draft202012Validator.check_schema(schema) + value = data.draw(_values(spec), label="value") + expected = conforms(_as_json_schema_reads(value), spec) + event("conforms" if expected else "does not conform") + assert Draft202012Validator(schema).is_valid(cast("Any", value)) == expected + + @_EXAMPLES @given(st.data()) def test_a_postponed_typeddict_reads_as_an_evaluated_one(data: st.DataObject) -> None: diff --git a/packages/zarr-metadata/tests/v3/test_definitions.py b/packages/zarr-metadata/tests/v3/test_definitions.py index a8a9271b5b..a262f152d8 100644 --- a/packages/zarr-metadata/tests/v3/test_definitions.py +++ b/packages/zarr-metadata/tests/v3/test_definitions.py @@ -489,15 +489,6 @@ def test_a_field_is_written_as_every_reader_takes_it( {"chunk_shape": (2, 3)}, [], ), - # An extent of 0 is right on a dimension of length 0, which only the - # array's shape can tell. - ( - {"name": "regular", "configuration": {"chunk_shape": [0, 3]}}, - ChunkGridDefinition, - Read, - {"chunk_shape": (0, 3)}, - [], - ), # An unknown key is survivable: reported, left out of the # configuration, and the field still read. ( @@ -579,7 +570,6 @@ def test_a_field_is_written_as_every_reader_takes_it( "bytes-bare", "bytes-endian", "regular-grid", - "regular-grid-zero-extent", "unknown-key", "unknown-key-before-the-rules", "not-required-postponed", @@ -866,17 +856,26 @@ def test_check_needs_nothing_but_the_value_and_a_typeddict() -> None: def test_check_reads_a_whole_array_document() -> None: - # A null in `dimension_names`, and a top-level key the document does - # not declare, typed by its `extra_items`. + # A null in `dimension_names`, a top-level key the document does not + # declare, typed by its `extra_items`, and a dimension of length 0. document = { - **ZarrV3ArrayMetadata.create_default(shape=(4,)).to_json(), - "dimension_names": [None], + **ZarrV3ArrayMetadata.create_default(shape=(0, 4)).to_json(), + "dimension_names": [None, "x"], "acme": {"must_understand": False}, } typed, problems = check(document, ZarrV3ArrayMetadataJSON) assert problems == () assert typed is not None - assert typed.get("dimension_names") == (None,) + assert typed.get("dimension_names") == (None, "x") + + +def test_error_check_holds_an_array_document_s_shape_to_its_bound() -> None: + document = {**ZarrV3ArrayMetadata.create_default(shape=(4,)).to_json(), "shape": [-1]} + typed, problems = check(document, ZarrV3ArrayMetadataJSON) + assert typed is None + assert [(found.loc, found.kind, dict(found.ctx)) for found in problems] == [ + (("shape", 0), "invalid_value", {"ge": 0}) + ] def test_configuration_of_types_what_its_definition_read() -> None: @@ -1129,6 +1128,12 @@ def test_error_a_data_type_fill_value_no_checker_reads() -> None: DataTypeDefinition(name="acme.set", configuration=Empty, fill_value=set[int]) +def test_error_a_data_type_fill_value_holding_a_metadata_field() -> None: + # A value of the data type, which no scope reads as a field. + with pytest.raises(TypeError, match="'acme.f': fill_value: CodecField holds a metadata field"): + DataTypeDefinition(name="acme.f", configuration=Empty, fill_value=tuple[CodecField, ...]) + + @pytest.mark.parametrize( ("kind", "member"), [ diff --git a/packages/zarr-metadata/tests/v3/test_every_definition.py b/packages/zarr-metadata/tests/v3/test_every_definition.py index 8617799364..5744860327 100644 --- a/packages/zarr-metadata/tests/v3/test_every_definition.py +++ b/packages/zarr-metadata/tests/v3/test_every_definition.py @@ -493,6 +493,13 @@ def test_error_scale_offset_scalar_is_null() -> None: ] +def test_error_regular_chunk_length_is_zero() -> None: + # Along a dimension of length 0 too: a grid's chunks have a size. + assert _one("chunk_grid:regular", {"chunk_shape": [4, 0]}) == [ + (("configuration", "chunk_shape", 1), "invalid_value") + ] + + def test_error_sharding_inner_chunk_extent_is_zero() -> None: configuration = {"chunk_shape": [0], "codecs": ["bytes"], "index_codecs": ["bytes"]} assert _one("codecs:sharding_indexed", configuration) == [ @@ -542,9 +549,9 @@ def test_error_sharding_inner_chunk_extent_is_zero() -> None: ), ( ChunkGridDefinition, - {"name": "regular", "configuration": {"chunk_shape": [2, -1]}}, + {"name": "regular", "configuration": {"chunk_shape": [2, 0]}}, ("configuration", "chunk_shape", 1), - {"ge": 0}, + {"ge": 1}, ), ( ChunkGridDefinition, diff --git a/packages/zarr-metadata/tests/v3/test_grid_shapes.py b/packages/zarr-metadata/tests/v3/test_grid_shapes.py index 6797fcb47e..93747545b5 100644 --- a/packages/zarr-metadata/tests/v3/test_grid_shapes.py +++ b/packages/zarr-metadata/tests/v3/test_grid_shapes.py @@ -44,9 +44,9 @@ def _problems(grid: JSONValue, shape: tuple[int, ...]) -> list[tuple[tuple[str | [ (_regular(), (), ()), (_regular(4, 4), (10, 3), ({4}, {4})), - # A chunk longer than its dimension, and a chunk length of 0 for a - # dimension of length 0. - (_regular(8, 0), (3, 0), ({8}, {0})), + # A chunk longer than its dimension, and a chunk over a dimension of + # length 0. + (_regular(8, 1), (3, 0), ({8}, {1})), # A bare integer repeats until it covers its dimension. (_rectilinear(4), (10,), ({4},)), (_rectilinear([4, 4, 2]), (10,), ({4, 2},)), @@ -79,12 +79,6 @@ def test_error_a_regular_grid_of_another_rank(grid: JSONValue, shape: tuple[int, assert _problems(grid, shape) == [(("configuration", "chunk_shape"), "invalid_value")] -def test_error_a_regular_chunk_length_of_0_for_a_dimension_that_is_not_empty() -> None: - assert _problems(_regular(4, 0), (4, 3)) == [ - (("configuration", "chunk_shape", 1), "invalid_value") - ] - - @pytest.mark.parametrize( ("grid", "shape"), [(_rectilinear(4), (10, 3)), (_rectilinear(4, 4), (1,))] )