Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
16 changes: 12 additions & 4 deletions packages/zarr-metadata/README.md
Original file line number Diff line number Diff line change
Expand Up @@ -97,20 +97,28 @@ definition that claims its name, `Unclaimed` when none does, or
`Refused` -- with where it sits and the kind it was read as, each codec
with the chunk it is handed, every problem, and the model when there is
none; `from_json` is that model, or the problems raised. A consumer's
own policy is a walk over the fields, with nothing read twice. Which
fields go beyond the core spec, say -- a field that names nothing is a
problem already:
own policy is a walk over the fields, with nothing read twice:
`with_problems` gives each with its problems, those located in it and in
the fields it holds, as zod's `flattenError` groups issues, and
`canonical_of` spells a field with none in the fewest words, without
reading it again. Which fields go beyond the core spec, say -- a field
that names nothing is a problem already -- and how each is spelled most
simply:

```python
from zarr_metadata.model import read_array_metadata_v3
from zarr_metadata.v3.definition import CORE
from zarr_metadata.v3.definition import CORE, canonical_of, with_problems

reading = read_array_metadata_v3(raw)
beyond_core = [
loc
for loc, field in reading.fields()
if field.name is not None and CORE.claimant(field.read_as, field.name) is None
]
simplest = {
loc: canonical_of(field, problems)
for loc, field, problems in with_problems(reading.fields(), reading.problems)
}
metadata = reading.metadata # None when reading.problems is not empty
```

Expand Down
11 changes: 11 additions & 0 deletions packages/zarr-metadata/changes/377.feature.md
Original file line number Diff line number Diff line change
@@ -0,0 +1,11 @@
`with_problems(fields, problems)` gives each field of one read -- a
reading's `fields()` and `problems`, or `fields_of` a field and the
problems `resolve` gave with it -- with the problems located in it, in
the fields it holds too: those it was read with, and those the document
found with it where it stands, its place in the pipeline, the chunk it
is handed, the array's shape. A field with none is valid there; a
problem in no field is in none's. It groups as zod's `flattenError`
groups issues, at every depth. `canonical_of(resolved, problems)` spells
a field a scope has read already, given its problems, in its simplest
equivalent spelling without reading it again, and gives what
`canonicalize` gives: None for a field with a problem.
16 changes: 12 additions & 4 deletions packages/zarr-metadata/docs/index.md
Original file line number Diff line number Diff line change
Expand Up @@ -112,20 +112,28 @@ definition that claims its name, `Unclaimed` when none does, or
`Refused` -- with where it sits and the kind it was read as, each codec
with the chunk it is handed, every problem, and the model when there is
none; `from_json` is that model, or the problems raised. A consumer's
own policy is a walk over the fields, with nothing read twice. Which
fields go beyond the core spec, say -- a field that names nothing is a
problem already:
own policy is a walk over the fields, with nothing read twice:
`with_problems` gives each with its problems, those located in it and in
the fields it holds, as zod's `flattenError` groups issues, and
`canonical_of` spells a field with none in the fewest words, without
reading it again. Which fields go beyond the core spec, say -- a field
that names nothing is a problem already -- and how each is spelled most
simply:

```python
from zarr_metadata.model import read_array_metadata_v3
from zarr_metadata.v3.definition import CORE
from zarr_metadata.v3.definition import CORE, canonical_of, with_problems

reading = read_array_metadata_v3(raw)
beyond_core = [
loc
for loc, field in reading.fields()
if field.name is not None and CORE.claimant(field.read_as, field.name) is None
]
simplest = {
loc: canonical_of(field, problems)
for loc, field, problems in with_problems(reading.fields(), reading.problems)
}
metadata = reading.metadata # None when reading.problems is not empty
```

Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -419,7 +419,8 @@ def fields(self) -> Iterator[tuple[Loc, Resolved[Any]]]:

The extension points, then each codec and storage transformer at its
index, each followed by the fields it holds, as `fields_of` gives
them: a shard's codecs, a struct's field types.
them: a shard's codecs, a struct's field types. `with_problems`
gives each with its problems.
"""
for key, field in (
("data_type", self.data_type),
Expand Down
69 changes: 60 additions & 9 deletions packages/zarr-metadata/src/zarr_metadata/v3/_definition.py
Original file line number Diff line number Diff line change
Expand Up @@ -1037,6 +1037,35 @@ def fields_of(resolved: Resolved[Any], loc: Loc = ()) -> Iterator[tuple[Loc, Res
yield from fields_of(inner, (*loc, "configuration", *place))


def with_problems(
fields: Iterable[tuple[Loc, Resolved[Any]]], problems: Sequence[ValidationProblem]
) -> Iterator[tuple[Loc, Resolved[Any], Problems]]:
"""Each of `fields`, with where it sits, and the problems among `problems` located in it, in the fields it holds too.

`fields` and `problems` are one read's: a reading's `fields()` and
`problems`, or `fields_of` a field and the problems `resolve` gave
with it. A field's problems are those it was read with, and those the
document found with it where it stands -- its place in the pipeline,
the chunk it is handed, the array's shape -- so a field with none is
valid there, and `canonical_of` spells it. A function of the problems,
as zod's `flattenError` is of the issues, grouping them at every
depth: a problem with a shard's inner codec is the inner codec's, and
the shard's. Each field comes before the fields it holds, as
`fields_of` gives them, so the last field whose problems hold a
problem is the innermost field holding it. A problem in no field --
with the fill value, with the shape -- is in none's.
"""
located = list(fields)
held: dict[Loc, list[ValidationProblem]] = {loc: [] for loc, _ in located}
for found in problems:
for depth in range(len(found.loc) + 1):
holder = held.get(found.loc[:depth])
if holder is not None:
holder.append(found)
for loc, field in located:
yield loc, field, tuple(held[loc])


def configuration_of(resolved: Resolved[Any], definition: Definition[C]) -> C | None:
"""The configuration `resolved` holds, typed as `definition` declares it, if `definition` read it.

Expand Down Expand Up @@ -1338,29 +1367,49 @@ def canonicalize(
(https://github.com/zarr-developers/zarr-specs/blob/fc7dd9c9beb5a50b87f9b08b00bf50fc0048482f/docs/v3/core/index.rst#L585-L592);
and there is no `must_understand`, since `true` is what absence means.
A name nothing in scope claims keeps the configuration it was written
with, since what it simplifies to is its own definition's call.
with, since what it simplifies to is its own definition's call. A
field a scope has read already is spelled so by `canonical_of`,
given its problems.
"""
resolved, problems = resolve(data, kind, context, loc)
return canonical_of(resolved, problems), problems


def canonical_of(
resolved: Resolved[Any], problems: Sequence[ValidationProblem]
) -> JSONValue | None:
"""`resolved`, a field a scope read, in its simplest equivalent spelling, as `canonicalize` spells one; None when it has a problem.

`problems` are the field's, as `resolve` gives them, or
`with_problems` gives each field of a reading. Only a field with none
has a simplest spelling: a simpler spelling of one with a problem
would erase what its author wrote -- a key its TypedDict does not
declare -- or spell what does not hold. What is spelled is what the
scope read, without reading the field again.
"""
if len(problems) != 0:
return None, problems
return _simplest(resolved), ()
return None
return _simplest(resolved)


def _simplest(field: Resolved[Any]) -> JSONValue:
"""A field without problems in its simplest spelling: one nothing in scope claims as a document writes it."""
def _simplest(field: Resolved[Any]) -> JSONValue | None:
"""A field in its simplest spelling; None when it, or a field it holds, was refused, which has none."""
if isinstance(field, Read):
return _canonical_field(field)
if isinstance(field, Unclaimed):
return field.to_json()
return field.json
return None


def _canonical_field(resolved: Read[Any]) -> JSONValue:
"""A field that read, in its simplest equivalent spelling: the fields it holds first, then its own members."""
def _canonical_field(resolved: Read[Any]) -> JSONValue | None:
"""A field that read, in its simplest equivalent spelling: the fields it holds first, then its own members; None when one it holds was refused."""
definition, name = resolved.definition, resolved.name
configuration: JSONValue = dict(resolved.configuration)
for loc, inner in resolved.nested.items():
configuration = _replaced(configuration, loc, _simplest(inner))
simplest = _simplest(inner)
if simplest is None:
return None
configuration = _replaced(configuration, loc, simplest)
simplified = cast("Mapping[str, JSONValue]", definition.canonical(configuration))
_, refused = definition.judge(simplified)
if len(refused) != 0:
Expand Down Expand Up @@ -1416,6 +1465,7 @@ def _replaced(value: JSONValue, path: Loc, new: JSONValue) -> JSONValue:
"Unclaimed",
"as_kind",
"asked",
"canonical_of",
"canonicalize",
"chunk_grid_lengths",
"configuration_of",
Expand All @@ -1435,4 +1485,5 @@ def _replaced(value: JSONValue, path: Loc, new: JSONValue) -> JSONValue:
"unknown_lengths",
"unknown_storage",
"variable_length",
"with_problems",
]
17 changes: 13 additions & 4 deletions packages/zarr-metadata/src/zarr_metadata/v3/definition.py
Original file line number Diff line number Diff line change
Expand Up @@ -258,10 +258,15 @@ def acme_lz4_rules(
words every reader takes: a data type with nothing to configure is its
bare name, any other field an object, `{"name": ...}`, as a Zarr v3.0
reader takes no short-hand name in `codecs`; raw bits write their size
back into the name, in decimal, so `r008` is `r8`. A field with any problem, an unknown key included, has
none: a simpler spelling of it would erase what its author wrote. What
`canonical` gives is judged again: one that does not hold is a
`ValueError`, a fault in the definition.
back into the name, in decimal, so `r008` is `r8`. A field with any
problem, an unknown key included, has none: a simpler spelling of it
would erase what its author wrote. What `canonical` gives is judged
again: one that does not hold is a `ValueError`, a fault in the
definition. `canonical_of(resolved, problems)` spells a field a scope
has read already, given its problems -- as `resolve` gives them, or
`with_problems` gives each field of a reading -- without reading it
again, and gives what `canonicalize` gives: None for a field with a
problem.

A definition checks itself when it is built, and each of these is a
`TypeError` saying what is wrong: a `configuration` that is not a
Expand Down Expand Up @@ -307,13 +312,15 @@ class creation.
StorageTransformerDefinition,
StorageTransformerField,
Unclaimed,
canonical_of,
canonicalize,
chunk_grid_lengths,
configuration_of,
fields_of,
fill_value_problems,
resolve,
storage_of,
with_problems,
)
from zarr_metadata.v3._pipeline import Stage, read_pipeline
from zarr_metadata.v3._registry import CORE, CORE_AND_EXTENSIONS, Context
Expand Down Expand Up @@ -352,6 +359,7 @@ class creation.
"Unclaimed",
"ValidationProblem",
"ZarrV3MetadataFieldJSON",
"canonical_of",
"canonicalize",
"check",
"chunk_grid_lengths",
Expand All @@ -362,4 +370,5 @@ class creation.
"resolve",
"shown",
"storage_of",
"with_problems",
]
105 changes: 104 additions & 1 deletion packages/zarr-metadata/tests/model/test_read_array_metadata.py
Original file line number Diff line number Diff line change
Expand Up @@ -18,6 +18,7 @@
ZarrV3ArrayMetadata,
ZarrV3ArrayMetadataReading,
read_array_metadata_v3,
read_group_metadata_v3,
validate_array_metadata_v3,
)
from zarr_metadata.v3.definition import (
Expand All @@ -33,10 +34,15 @@
Refused,
StorageTransformerDefinition,
Unclaimed,
fields_of,
resolve,
with_problems,
)

if TYPE_CHECKING:
from zarr_metadata.v3.definition import Lengths, Loc
from collections.abc import Callable, Iterator

from zarr_metadata.v3.definition import Lengths, Loc, Resolved

LITTLE = {"name": "bytes", "configuration": {"endian": "little"}}
ZSTD = {"name": "zstd", "configuration": {"level": 1}}
Expand Down Expand Up @@ -228,6 +234,103 @@ def test_a_document_reads_as_each_field_where_it_sits_and_its_codecs_as_a_pipeli
assert [None if s.incoming is None else s.incoming.lengths for s in reading.pipeline] == handed


_INNER_GZIP = {"name": "gzip", "configuration": {"level": 12}}
_WITH_PROBLEMS = _document(
(4, 4),
chunk_grid={"name": "regular", "configuration": {"chunk_shape": [4]}},
fill_value="high",
codecs=[
_shard([2, 2], [LITTLE, _INNER_GZIP]),
ZSTD,
{"name": "transpose", "configuration": {"order": [1, 0]}},
],
)
"""An array document with a problem in each place a field can have one, and one in no field."""


def _array_fields(
document: object,
) -> Iterator[tuple[Loc, Resolved[Any], tuple[ValidationProblem, ...]]]:
reading = read_array_metadata_v3(document)
return with_problems(reading.fields(), reading.problems)


def _group_fields(
document: object,
) -> Iterator[tuple[Loc, Resolved[Any], tuple[ValidationProblem, ...]]]:
group = {
"zarr_format": 3,
"node_type": "group",
"consolidated_metadata": {
"kind": "inline",
"must_understand": False,
"metadata": {"a": document},
},
}
reading = read_group_metadata_v3(group)
return with_problems(reading.fields(), reading.problems)


def _field_fields(
document: object,
) -> Iterator[tuple[Loc, Resolved[Any], tuple[ValidationProblem, ...]]]:
resolved, problems = resolve(
cast("dict[str, Any]", document)["codecs"][0],
CodecDefinition,
CORE_AND_EXTENSIONS,
("codecs", 0),
)
return with_problems(fields_of(resolved, ("codecs", 0)), problems)


_INNER_LEVEL = ("codecs", 0, "configuration", "codecs", 1, "configuration", "level")
_EACH_FIELD_S = {
("data_type",): [],
("chunk_grid",): [("chunk_grid", "configuration", "chunk_shape")],
("chunk_key_encoding",): [],
("codecs", 0): [_INNER_LEVEL],
("codecs", 0, "configuration", "codecs", 0): [],
("codecs", 0, "configuration", "codecs", 1): [_INNER_LEVEL],
("codecs", 0, "configuration", "index_codecs", 0): [],
("codecs", 1): [],
("codecs", 2): [("codecs", 2)],
}
"""Each field of `_WITH_PROBLEMS`, and where each of its problems is."""
_IN_A_GROUP = ("consolidated_metadata", "metadata", "a")


@pytest.mark.parametrize(
("fields", "expected"),
[
(_array_fields, _EACH_FIELD_S),
(
_group_fields,
{
(*_IN_A_GROUP, *loc): [(*_IN_A_GROUP, *problem) for problem in problems]
for loc, problems in _EACH_FIELD_S.items()
},
),
(
_field_fields,
{loc: problems for loc, problems in _EACH_FIELD_S.items() if loc[:2] == ("codecs", 0)},
),
],
ids=["an-array", "a-document-a-group-holds", "one-field"],
)
def test_each_field_comes_with_the_problems_located_in_it(
fields: Callable[[object], Iterator[tuple[Loc, Resolved[Any], tuple[ValidationProblem, ...]]]],
expected: dict[Loc, list[Loc]],
) -> None:
# Those it was read with and those the document found with it where
# it stands -- a transpose out of the pipeline's order, at its own
# place, and the chunk grid over the shape -- the fields it holds
# too: a shard with a bad inner codec has that problem as well. A
# problem in no field -- the fill value -- is in none's.
assert {
loc: [p.loc for p in problems] for loc, _, problems in fields(_WITH_PROBLEMS)
} == expected


def test_a_document_with_no_problem_reads_as_its_model_holding_the_fields_read() -> None:
# The fields the read made, not a second reading of them.
document = _document(codecs=[{"name": "transpose", "configuration": {"order": [0]}}, LITTLE])
Expand Down
Loading
Loading