From cd7d2641935a8178d4591ff81ed49584e4e3c3b9 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Wed, 9 Sep 2026 20:16:37 +0200 Subject: [PATCH 01/76] fix(chunk-grids): enforce one zero-length-axis invariant across model, clamps and metadata Invariant: a chunk edge length is always >= 1; a dimension's extent may be 0, in which case the dimension has zero chunks (ceildiv(0, size) == 0). Zero-length-axis bugs have recurred since 2017 (#150, #241, #303, #972, #1977, #2434, #3711, #4305, #4307, #4328) because the layers disagreed on this invariant and every span-derived chunk spelling clamped on its own: - The metadata layer (common.py, metadata/v3.py) required chunk edges >= 1, but the in-memory FixedDimension allowed size == 0 with four special-case branches left over from #2434, so normalization could build a grid the metadata constructor then rejected. FixedDimension now rejects size < 1 and the four `if self.size == 0` branches are gone. VaryingDimension already required edges > 0 and is unchanged. - `chunks=-1`, `chunks=False`, `chunks="auto"` (_guess_regular_chunks, both the typesize == 0 early return and the np.maximum line) and `shards="auto"` each derived "one chunk covering the axis" independently. They now all go through one helper, `_full_span_chunk_size(span) = max(span, 1)`, which is the single definition of that phrase for a possibly zero-length axis. - Zarr format 2 metadata had no chunk >= 1 check, so a legacy `chunks: [0]` document opened fine and read uninitialised memory after a resize. It now raises a clear ValueError at parse time, matching the format 3 grid. - Rectilinear grids had no creation-time spelling for a zero-length axis: normalize_chunks_1d required sum(edges) == span, which no list of positive edges can satisfy for span 0, even though the same state is reachable via resize((0,)) and round-trips through reopen. For span == 0 any non-empty list of positive edges is now accepted verbatim, producing the same VaryingDimension(edges, extent=0) that resize produces; the strict sum check is kept for span > 0. Tests: the per-spelling regression test from #4328 is replaced by one matrix over {-1, False, "auto", 1, (1,...), [[2, 2]]} x {(0,), (0, 4), (4, 0), (0, 0), ()} x {v2, v3} x {no shards, shards="auto" with and without a byte budget, explicit shards}, with separate small tests for each error case. Tests that constructed FixedDimension(size=0) now assert it raises, and a zero-extent test covers the behaviour the old special cases were guarding. Assisted-by: ClaudeCode:claude-fable-5-1 --- changes/0000.bugfix.md | 1 + docs/user-guide/arrays.md | 6 + src/zarr/core/chunk_grids.py | 75 ++++++---- src/zarr/core/metadata/v2.py | 7 + tests/test_chunk_grids.py | 229 ++++++++++++++++++++++++------- tests/test_metadata/test_v2.py | 12 ++ tests/test_unified_chunk_grid.py | 77 +++++------ 7 files changed, 284 insertions(+), 123 deletions(-) create mode 100644 changes/0000.bugfix.md diff --git a/changes/0000.bugfix.md b/changes/0000.bugfix.md new file mode 100644 index 0000000000..8b224c3f76 --- /dev/null +++ b/changes/0000.bugfix.md @@ -0,0 +1 @@ +Zero-length dimensions are now handled by a single rule instead of a per-spelling patch: a chunk edge length is always at least 1, while a dimension's extent may be 0 (such a dimension simply has zero chunks). Every way of asking for "one chunk covering the axis" — `chunks=-1`, `chunks=False`, `chunks="auto"`, and `shards="auto"` — now derives the chunk size from the same helper, so they agree on chunk size 1 for a zero-length axis in both Zarr formats. Zarr format 2 metadata now rejects a chunk edge length of 0 with a clear error, matching Zarr format 3, instead of reading uninitialised data after a resize. Rectilinear chunk grids (`chunks=[[...], ...]`) can now be created on a zero-length dimension: since no list of positive edge lengths can sum to 0, the given edge lengths are stored as-is and describe the chunks the dimension will grow into on `append` or `resize`, exactly the state a rectilinear dimension is in after being resized down to 0. The private `FixedDimension(size=0, ...)` model, which previously carried its own zero-size special cases, now raises `ValueError`. diff --git a/docs/user-guide/arrays.md b/docs/user-guide/arrays.md index 707fc1a1a5..ebc4b9aee1 100644 --- a/docs/user-guide/arrays.md +++ b/docs/user-guide/arrays.md @@ -708,6 +708,12 @@ z.append(np.arange(10, dtype='float64')) print(f"After append: shape={z.shape}, chunk_sizes={z.write_chunk_sizes}") ``` +A rectilinear array can also be created with a zero-length dimension: because no +list of positive chunk sizes can sum to 0, the chunk sizes given for such a +dimension are stored as-is and describe the chunks the dimension will grow into +on `append` or `resize` — the same state as resizing an existing rectilinear +dimension down to 0. + ### Compressors and filters Rectilinear arrays work with all codecs — compressors, filters, and checksums. diff --git a/src/zarr/core/chunk_grids.py b/src/zarr/core/chunk_grids.py index 6242994fdf..f759e914b6 100644 --- a/src/zarr/core/chunk_grids.py +++ b/src/zarr/core/chunk_grids.py @@ -45,22 +45,24 @@ @dataclass(frozen=True) class FixedDimension: """Uniform chunk size. Boundary chunks contain less data but are - encoded at full size by the codec pipeline.""" + encoded at full size by the codec pipeline. - size: int # chunk edge length (>= 0) - extent: int # array dimension length + The chunk edge length is always at least 1, matching the invariant the + metadata layer enforces for every stored chunk grid. The extent may be 0: + a zero-length axis simply has zero chunks (``ceildiv(0, size) == 0``). + """ + + size: int # chunk edge length (>= 1) + extent: int # array dimension length (>= 0) nchunks: int = field(init=False, repr=False) ngridcells: int = field(init=False, repr=False) def __post_init__(self) -> None: - if self.size < 0: - raise ValueError(f"FixedDimension size must be >= 0, got {self.size}") + if self.size < 1: + raise ValueError(f"FixedDimension size must be >= 1, got {self.size}") if self.extent < 0: raise ValueError(f"FixedDimension extent must be >= 0, got {self.extent}") - if self.size == 0: - n = 0 - else: - n = ceildiv(self.extent, self.size) + n = ceildiv(self.extent, self.size) object.__setattr__(self, "nchunks", n) object.__setattr__(self, "ngridcells", n) @@ -69,8 +71,6 @@ def index_to_chunk(self, idx: int) -> int: raise IndexError(f"Negative index {idx} is not allowed") if idx >= self.extent: raise IndexError(f"Index {idx} is out of bounds for extent {self.extent}") - if self.size == 0: - return 0 return idx // self.size def chunk_offset(self, chunk_ix: int) -> int: @@ -95,8 +95,6 @@ def data_size(self, chunk_ix: int) -> int: Does not validate *chunk_ix* — callers must ensure it is in ``[0, nchunks)``. Use ``ChunkGrid.__getitem__`` for safe access. """ - if self.size == 0: - return 0 return max(0, min(self.size, self.extent - chunk_ix * self.size)) @property @@ -110,8 +108,6 @@ def _unique_edge_lengths(self) -> Iterable[int]: return (self.size,) def indices_to_chunks(self, indices: npt.NDArray[np.intp]) -> npt.NDArray[np.intp]: - if self.size == 0: - return np.zeros_like(indices) return indices // self.size def with_extent(self, new_extent: int) -> FixedDimension: @@ -640,6 +636,20 @@ class ChunkLayout(NamedTuple): inner: ChunkLayout | None = None +def _full_span_chunk_size(span: int) -> int: + """The edge length of one chunk covering an entire axis of length *span*. + + This is *the* definition of "one chunk spans the axis" for a possibly + zero-length axis. Chunk edge lengths must be at least 1 (the invariant + shared by `FixedDimension`, `VaryingDimension` and the stored chunk grid + metadata), so a zero-length axis gets chunk size 1 and zero chunks. Every + spelling that derives a chunk size from a span — ``chunks=-1``, + ``chunks=False``, ``chunks="auto"``, ``shards="auto"`` — must route + through this helper rather than clamping on its own. + """ + return max(span, 1) + + def _guess_regular_chunks( shape: tuple[int, ...] | int, typesize: int, @@ -677,11 +687,10 @@ def _guess_regular_chunks( shape = (shape,) if typesize == 0: - return shape + return tuple(_full_span_chunk_size(s) for s in shape) ndims = len(shape) - # require chunks to have non-zero length for all dimensions - chunks = np.maximum(np.array(shape, dtype="=f8"), 1) + chunks = np.array([_full_span_chunk_size(s) for s in shape], dtype="=f8") # Determine the optimal chunk size in bytes using a PyTables expression. # This is kept as a float. @@ -724,13 +733,21 @@ def normalize_chunks_1d(chunks: int | Iterable[object], span: int) -> DimensionG the span, and the uniform form is O(1) in the number of chunks — a dimension with `2**62` chunks must not materialize one entry per chunk. - `-1` means "one chunk covering the entire span." + `-1` means "one chunk covering the entire span" (see `_full_span_chunk_size` + for what that means on a zero-length span). Explicit chunk size lists must sum to the span exactly and always produce `VaryingDimension`, even when the sizes happen to be uniform: the input syntax declares the grid kind, so a per-chunk list is preserved as a rectilinear dimension rather than silently collapsed to a regular one, which would change how the dimension grows on resize. For scalar sizes the last chunk may overhang the span. + + The one exception to the sum rule is a zero-length span: no list of + positive edges can sum to 0, so any non-empty list is accepted verbatim + and the edges describe the chunks the axis will grow into on `append` / + `resize`. This is the same state a rectilinear axis reaches when it is + resized down to 0 — `VaryingDimension` allows trailing edges beyond the + extent — so creating at length 0 and shrinking to 0 are indistinguishable. """ # `numbers.Integral` rather than `int` so that numpy integer scalars (which are not # `int` subclasses) take the uniform-chunk path instead of being treated as a sequence. @@ -741,9 +758,7 @@ def normalize_chunks_1d(chunks: int | Iterable[object], span: int) -> DimensionG if chunk_size < -1 or chunk_size == 0: raise ValueError(f"Chunk size must be positive or -1, got {chunk_size}") if chunk_size == -1: - # A zero-length span still gets chunk size 1 (chunk sizes must be positive), - # matching the auto-chunking clamp in _guess_regular_chunks. - return FixedDimension(size=max(span, 1), extent=span) + return FixedDimension(size=_full_span_chunk_size(span), extent=span) return FixedDimension(size=chunk_size, extent=span) else: try: @@ -768,7 +783,9 @@ def normalize_chunks_1d(chunks: int | Iterable[object], span: int) -> DimensionG ints: list[int] = [int(c) for c in chunk_list] # type: ignore[call-overload] if any(c <= 0 for c in ints): raise ValueError(f"All chunk sizes must be positive, got {ints}") - if sum(ints) != span: + # A zero-length span cannot be covered by positive edges; the edges are the + # chunks the axis will grow into, exactly as after ``resize(0)``. + if span > 0 and sum(ints) != span: raise ValueError(f"Chunk sizes {ints} do not sum to span {span}") return VaryingDimension(ints, extent=span) @@ -809,7 +826,7 @@ def normalize_chunks_nd( ) # handle no chunking: one chunk covering every axis. Routed through the -1 sentinel so - # the zero-length-axis clamp lives in one place (normalize_chunks_1d). + # the zero-length-axis rule lives in one place (_full_span_chunk_size). if chunks is False: chunks = -1 @@ -864,8 +881,10 @@ def _guess_num_chunks_per_axis_shard( In other words the shard would be a (2,2,2) grid of (2,2,2) chunks i.e., prod(chunk_shape) * (returned_val ** len(chunk_shape)) * item_size = 256 bytes. - Degenerate chunk shapes — a 0-dimensional shape, or one containing a zero-length - axis — return 1, as the search loop's stopping conditions can never be met. + Degenerate inputs — a 0-dimensional chunk shape, or a zero-byte chunk (``item_size`` + of 0; chunk edge lengths themselves are always at least 1) — return 1, as the + search loop's stopping conditions can never be met. A zero-length *array* axis + needs no special case: the array-bound check fails immediately for it. Parameters ---------- @@ -886,8 +905,8 @@ def _guess_num_chunks_per_axis_shard( if max_bytes < bytes_per_chunk: return 1 num_axes = len(chunk_shape) - # For a 0-dimensional chunk shape or one with a zero-length axis, both loop - # conditions below are constant, so the loop would never terminate. + # For a 0-dimensional chunk shape or a zero-byte chunk, both loop conditions + # below are constant, so the loop would never terminate. if num_axes == 0 or bytes_per_chunk == 0: return 1 chunks_per_shard = 1 diff --git a/src/zarr/core/metadata/v2.py b/src/zarr/core/metadata/v2.py index 5822a228b9..b1fa2d866d 100644 --- a/src/zarr/core/metadata/v2.py +++ b/src/zarr/core/metadata/v2.py @@ -89,6 +89,13 @@ def __init__( """ shape_parsed = parse_shapelike(shape) chunks_parsed = parse_shapelike(chunks) + # Same invariant as the Zarr format 3 chunk grid metadata: every chunk edge + # length is at least 1, even on a zero-length axis. + for dim_idx, chunk in enumerate(chunks_parsed): + if chunk < 1: + raise ValueError( + f"Dimension {dim_idx}: chunk edge length must be >= 1, got {chunk}" + ) compressor_parsed = parse_compressor(compressor) order_parsed = parse_indexing_order(order) dimension_separator_parsed = parse_separator(dimension_separator) diff --git a/tests/test_chunk_grids.py b/tests/test_chunk_grids.py index 40133700a8..86383d31ed 100644 --- a/tests/test_chunk_grids.py +++ b/tests/test_chunk_grids.py @@ -367,66 +367,191 @@ def test_create_0d_array_auto_shards_with_target_shard_size() -> None: assert arr.shards == () -@pytest.mark.parametrize("chunks", [-1, False], ids=["minus-one", "false"]) -@pytest.mark.parametrize("shape", [(0,), (0, 4), (4, 0)], ids=["1d", "2d-lead", "2d-trail"]) +# -- Zero-length dimensions -- +# +# One invariant: a chunk edge length is always >= 1, an extent may be 0. Every spelling +# that derives a chunk size from a span (-1, False, "auto", shards="auto") must agree on +# chunk size 1 for a zero-length axis, in both Zarr formats, with or without sharding. +# Historically each spelling clamped (or failed to clamp) on its own; see #4304, #4305, +# #4307, #4328 and, further back, #150, #241, #303, #972, #1977, #2434, #3711. + +ZeroLengthChunkSpelling = Literal["minus-one", "false", "auto", "one", "ones", "rectilinear"] +ZeroLengthShards = Literal["auto", "auto-budget", "explicit"] | None + +# Spellings whose chunk size is derived from the axis span rather than given explicitly. +_SPAN_DERIVED_SPELLINGS: frozenset[ZeroLengthChunkSpelling] = frozenset( + {"minus-one", "false", "auto"} +) + + +def _zero_length_chunks_arg(spelling: ZeroLengthChunkSpelling, shape: tuple[int, ...]) -> Any: + """Translate a chunk-spelling id into the `chunks=` argument for `shape`.""" + match spelling: + case "minus-one": + return -1 + case "false": + return False + case "auto": + return "auto" + case "one": + return 1 + case "ones": + return (1,) * len(shape) + case "rectilinear": + return [[2, 2]] * len(shape) + + +@pytest.mark.parametrize("spelling", ["minus-one", "false", "auto", "one", "ones", "rectilinear"]) @pytest.mark.parametrize( - ("zarr_format", "shards", "target_shard_size_bytes"), - [ - (2, None, None), - (3, None, None), - (3, "auto", None), - (3, "auto", 128 * 1024 * 1024), - ], - ids=["v2", "v3", "v3-auto-shards", "v3-auto-shards-budget"], + "shape", + [(0,), (0, 4), (4, 0), (0, 0), ()], + ids=["1d", "2d-lead", "2d-trail", "2d-both", "0d"], ) -def test_create_zero_length_array_full_span_chunks( - chunks: int | bool, +@pytest.mark.parametrize( + ("zarr_format", "shards"), + [(2, None), (3, None), (3, "auto"), (3, "auto-budget"), (3, "explicit")], + ids=["v2", "v3", "v3-auto-shards", "v3-auto-shards-budget", "v3-explicit-shards"], +) +def test_create_zero_length_array( + spelling: ZeroLengthChunkSpelling, shape: tuple[int, ...], zarr_format: Literal[2, 3], - shards: Literal["auto"] | None, - target_shard_size_bytes: int | None, + shards: ZeroLengthShards, ) -> None: - """`chunks=-1` and `chunks=False` on a zero-length axis must resolve to chunk size 1. - - Both spellings mean "one chunk covering the whole axis". They used to resolve to chunk - size 0 on zero-length axes, which broke every downstream path differently: a ValueError - from the Zarr format 3 chunk grid metadata, a ZeroDivisionError with shards="auto", an - infinite loop with a shard size budget (https://github.com/zarr-developers/zarr-python/issues/4304), - and invalid `chunks: [0]` metadata for Zarr format 2 that silently corrupted reads after - a resize. + """Every chunk spelling produces a valid, usable grid on a zero-length axis. + + Span-derived spellings resolve to chunk size 1 on zero-length axes (and the full span + elsewhere, for these small shapes); explicit spellings are stored verbatim. In every case + the stored metadata matches `arr.chunks` / `arr.shards`, the array can grow along the + empty axis, round-trip data, and shrink back to empty. """ - expected_chunks = tuple(max(s, 1) for s in shape) + ndim = len(shape) + if spelling == "rectilinear": + if zarr_format == 2: + pytest.skip("Zarr format 2 does not support rectilinear chunk grids") + if shards is not None: + pytest.skip("rectilinear chunks with sharding is not supported") + if ndim == 0: + pytest.skip("a 0-d array has no dimension to chunk rectilinearly") + if shards == "explicit" and ndim == 0: + pytest.skip("a 0-d array has no axis to shard explicitly") + + chunks = _zero_length_chunks_arg(spelling, shape) + expected_chunks: tuple[int, ...] | None + if spelling in _SPAN_DERIVED_SPELLINGS: + expected_chunks = tuple(max(s, 1) for s in shape) + elif spelling == "rectilinear": + expected_chunks = None + else: + expected_chunks = (1,) * ndim + + shards_arg: Any + expected_shards: tuple[int, ...] | None + match shards: + case None: + shards_arg, expected_shards = None, None + case "auto" | "auto-budget": + # Axes this short never split, so the guessed shard equals the chunk. + shards_arg, expected_shards = "auto", expected_chunks + case "explicit": + # A shard larger than the (zero) extent is fine: the axis has zero shards. + shards_arg = tuple(2 if s == 0 else s for s in shape) + expected_shards = shards_arg + warns = ( pytest.warns(ZarrUserWarning, match="Automatic shard shape inference is experimental") - if shards == "auto" + if shards_arg == "auto" else contextlib.nullcontext() ) - with zarr.config.set({"array.target_shard_size_bytes": target_shard_size_bytes}), warns: - arr = zarr.create_array( - store={}, - shape=shape, - dtype="int64", - chunks=chunks, - shards=shards, - zarr_format=zarr_format, - ) - assert arr.chunks == expected_chunks - assert arr.shards == (expected_chunks if shards == "auto" else None) + budget = 128 * 1024 * 1024 if shards == "auto-budget" else None + # The rectilinear flag must stay set for the array's whole life, not just creation. + with zarr.config.set( + {"array.rectilinear_chunks": True, "array.target_shard_size_bytes": budget} + ): + with warns: + arr = zarr.create_array( + store={}, + shape=shape, + dtype="int64", + chunks=chunks, + shards=shards_arg, + zarr_format=zarr_format, + ) - # The stored chunk grid must be the clamped shape, whichever format wrote it. - meta = cast(dict[str, Any], arr.metadata.to_dict()) - if zarr_format == 2: - assert meta["chunks"] == expected_chunks - else: - assert meta["chunk_grid"]["configuration"]["chunk_shape"] == expected_chunks - - # The array must remain usable: grow the empty axis and round-trip data through it. - axis = shape.index(0) - grown = tuple(2 if s == 0 else s for s in shape) - arr.append(np.full(grown, 7, dtype="int64"), axis=axis) - assert arr.shape == grown - np.testing.assert_array_equal(arr[...], np.full(grown, 7, dtype="int64")) - resized = tuple(3 if s == 0 else s for s in shape) - arr.resize(resized) - assert arr.shape == resized - assert int(np.asarray(arr[...]).sum()) == 7 * np.prod(grown) + # In-memory view and stored metadata agree with the invariant. + assert arr.shards == expected_shards + meta = cast(dict[str, Any], arr.metadata.to_dict()) + if spelling == "rectilinear": + grid = meta["chunk_grid"] + assert grid["name"] == "rectilinear" + # Stored verbatim on zero-length axes too, run-length encoded as [size, count]. + assert list(grid["configuration"]["chunk_shapes"]) == [[[2, 2]]] * ndim + assert arr.write_chunk_sizes == tuple(() if s == 0 else (2, 2) for s in shape) + else: + assert arr.chunks == expected_chunks + if zarr_format == 2: + assert meta["chunks"] == expected_chunks + else: + stored = meta["chunk_grid"]["configuration"]["chunk_shape"] + assert stored == (expected_chunks if expected_shards is None else expected_shards) + assert all(c >= 1 for c in arr.chunks) + + # The array must remain usable. + if ndim == 0: + arr[...] = 7 + assert arr[...] == 7 + return + axis = shape.index(0) + grown = tuple(2 if i == axis else s for i, s in enumerate(shape)) + data = np.full(grown, 7, dtype="int64") + arr.append(data, axis=axis) + assert arr.shape == grown + np.testing.assert_array_equal(arr[...], data) + arr.resize(shape) + assert arr.shape == shape + assert np.asarray(arr[...]).shape == shape + + +@pytest.mark.parametrize("zarr_format", [2, 3]) +def test_create_zero_chunk_rejected(zarr_format: Literal[2, 3]) -> None: + """An explicit chunk size of 0 is rejected up front, even for a zero-length axis.""" + with pytest.raises(ValueError, match="Chunk size must be positive or -1, got 0"): + zarr.create_array(store={}, shape=(0,), chunks=(0,), dtype="int64", zarr_format=zarr_format) + + +def test_rectilinear_zero_extent_matches_resize() -> None: + """Creating a rectilinear axis at length 0 equals resizing one down to 0. + + Both leave a `VaryingDimension` whose edges lie entirely beyond the extent, so the + stored grids are identical and both grow into the same chunks on append. + """ + with zarr.config.set({"array.rectilinear_chunks": True}): + created = zarr.create_array(store={}, shape=(0,), chunks=[[2, 2]], dtype="int64") + resized = zarr.create_array(store={}, shape=(4,), chunks=[[2, 2]], dtype="int64") + resized.resize((0,)) + created_meta = cast(dict[str, Any], created.metadata.to_dict()) + resized_meta = cast(dict[str, Any], resized.metadata.to_dict()) + assert created_meta["chunk_grid"] == resized_meta["chunk_grid"] + assert created_meta["shape"] == resized_meta["shape"] == (0,) + + created.append(np.arange(3, dtype="int64")) + resized.append(np.arange(3, dtype="int64")) + np.testing.assert_array_equal(created[...], np.arange(3)) + np.testing.assert_array_equal(resized[...], np.arange(3)) + assert created.write_chunk_sizes == resized.write_chunk_sizes == ((2, 1),) + + +def test_normalize_chunks_1d_zero_span_accepts_any_edges() -> None: + """On a zero-length span the explicit edge list is stored verbatim.""" + dim = normalize_chunks_1d([3, 5], span=0) + assert isinstance(dim, VaryingDimension) + assert dim.edges == (3, 5) + assert dim.extent == 0 + assert dim.nchunks == 0 + assert dim.resize(4) == VaryingDimension([3, 5], extent=4) + + +def test_normalize_chunks_1d_nonzero_span_still_requires_exact_sum() -> None: + """Relaxing the sum rule for span 0 must not leak into positive spans.""" + with pytest.raises(ValueError, match="do not sum to span 1"): + normalize_chunks_1d([3, 5], span=1) diff --git a/tests/test_metadata/test_v2.py b/tests/test_metadata/test_v2.py index 1358f458d6..ac3ee4c029 100644 --- a/tests/test_metadata/test_v2.py +++ b/tests/test_metadata/test_v2.py @@ -309,6 +309,18 @@ def test_from_dict_extra_fields() -> None: assert result == expected +@pytest.mark.parametrize(("shape", "chunks"), [((0,), (0,)), ((4, 0), (4, 0)), ((5,), (0,))]) +def test_zero_chunk_edge_rejected(shape: tuple[int, ...], chunks: tuple[int, ...]) -> None: + """A chunk edge length of 0 is invalid metadata, whatever the array shape. + + Older releases could write `chunks: [0]` for a zero-length axis; such documents read + uninitialised memory once resized. The v2 layer now enforces the same `>= 1` rule as + the Zarr format 3 chunk grid. + """ + with pytest.raises(ValueError, match="chunk edge length must be >= 1, got 0"): + ArrayV2Metadata(shape=shape, dtype=Float64(), chunks=chunks, fill_value=0.0, order="C") + + def test_eq_nan_fill_value() -> None: """Two metadata objects with an identical NaN fill_value compare equal. diff --git a/tests/test_unified_chunk_grid.py b/tests/test_unified_chunk_grid.py index b8289d2135..2cf3f7cc10 100644 --- a/tests/test_unified_chunk_grid.py +++ b/tests/test_unified_chunk_grid.py @@ -149,9 +149,9 @@ def test_rectilinear_feature_flag_enabled() -> None: (10, 100, 1, 10, 10, 10, 10), (10, 100, 9, 10, 10, 10, 90), (10, 95, 9, 10, 10, 5, 90), # boundary chunk - (0, 0, None, 0, None, None, None), # zero-size + (10, 0, None, 0, None, None, None), # zero-extent: no chunks, size still >= 1 ], - ids=["start", "middle", "end", "boundary", "zero-size"], + ids=["start", "middle", "end", "boundary", "zero-extent"], ) def test_fixed_dimension( size: int, @@ -190,11 +190,21 @@ def test_fixed_dimension_indices_to_chunks() -> None: @pytest.mark.parametrize( ("size", "extent", "match"), - [(-1, 100, "must be >= 0"), (10, -1, "must be >= 0")], - ids=["negative-size", "negative-extent"], + [ + (-1, 100, "size must be >= 1"), + (0, 100, "size must be >= 1"), + (0, 0, "size must be >= 1"), + (10, -1, "extent must be >= 0"), + ], + ids=["negative-size", "zero-size", "zero-size-zero-extent", "negative-extent"], ) -def test_fixed_dimension_rejects_negative(size: int, extent: int, match: str) -> None: - """FixedDimension raises ValueError for negative size or extent""" +def test_fixed_dimension_rejects_invalid(size: int, extent: int, match: str) -> None: + """FixedDimension raises ValueError for a size below 1 or a negative extent. + + A chunk edge length of 0 is never valid, whatever the extent: the metadata layer + requires every chunk edge length to be >= 1, and the in-memory model enforces the + same invariant so the two can never disagree. + """ with pytest.raises(ValueError, match=match): FixedDimension(size=size, extent=extent) @@ -1421,46 +1431,27 @@ def test_edge_case_chunk_grid_boundary_shape() -> None: # -- Zero-size and zero-extent -- -@pytest.mark.parametrize( - ("size", "extent"), - [(0, 0), (0, 5), (10, 0)], - ids=["zero-size-zero-extent", "zero-size-nonzero-extent", "zero-extent-nonzero-size"], -) -def test_edge_case_zero_size_or_extent(size: int, extent: int) -> None: - """FixedDimension with zero size or extent has zero chunks and getitem returns None""" - d = FixedDimension(size=size, extent=extent) - assert d.nchunks == 0 - g = ChunkGrid(dimensions=(d,)) - assert g[0] is None - - -def test_edge_case_zero_size_data_and_indices() -> None: - """FixedDimension(size=0) handles data_size, index_to_chunk, and indices_to_chunks safely.""" - d = FixedDimension(size=0, extent=0) - # Zero-sized chunks have zero data - assert d.data_size(0) == 0 - # Vectorized lookup maps every index to chunk 0 (avoids division by zero) - indices = np.array([0, 0, 0], dtype=np.intp) - np.testing.assert_array_equal(d.indices_to_chunks(indices), np.zeros(3, dtype=np.intp)) +@pytest.mark.parametrize("size", [1, 10], ids=["size-1", "size-10"]) +def test_fixed_dimension_zero_extent(size: int) -> None: + """A zero-length axis has zero chunks and behaves like an empty grid. - -def test_edge_case_zero_size_nonzero_extent_index() -> None: - """FixedDimension(size=0, extent>0) maps valid indices to chunk 0 without dividing by zero.""" - d = FixedDimension(size=0, extent=5) + The extent may be 0 even though the chunk size may not: `ceildiv(0, size)` is 0, + so there is nothing to look up, and the vectorized index mapping of an empty index + array is empty. + """ + d = FixedDimension(size=size, extent=0) assert d.nchunks == 0 - # index_to_chunk avoids division by zero and returns 0 - assert d.index_to_chunk(0) == 0 - assert d.index_to_chunk(4) == 0 - - -def test_edge_case_zero_size_data_and_index() -> None: - """FixedDimension(size=0) returns zero for data_size and maps indices to chunk 0.""" - d = FixedDimension(size=0, extent=0) - # data_size returns 0 for a zero-sized chunk + assert d.ngridcells == 0 assert d.data_size(0) == 0 - # vectorized indices_to_chunks returns zeros - indices = np.array([0, 0, 0], dtype=np.intp) - np.testing.assert_array_equal(d.indices_to_chunks(indices), np.zeros(3, dtype=np.intp)) + assert d.with_extent(0) == d + assert d.with_extent(3) == FixedDimension(size=size, extent=3) + empty = np.array([], dtype=np.intp) + np.testing.assert_array_equal(d.indices_to_chunks(empty), empty) + with pytest.raises(IndexError): + d.index_to_chunk(0) + g = ChunkGrid(dimensions=(d,)) + assert g[0] is None + assert list(g) == [] # -- 0-d grid -- From a171507980ed67b19cbeaa9e33ed20604b4ba944 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Wed, 9 Sep 2026 20:20:39 +0200 Subject: [PATCH 02/76] fix(metadata): read a legacy v2 zero chunk edge on an empty axis as 1 MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit zarr-python 2.18.7 writes `chunks: [0]` for `zarr.zeros((0,), chunks=False)` and for `chunks=(0,)`, so stores with that document exist. Rejecting them at open would turn a previously-readable array into an error; leaving the 0 in place read uninitialised memory after a resize. Normalize the edge to 1 with a ZarrUserWarning instead — the same grid every other "one chunk spans the axis" spelling produces — and keep rejecting a zero edge on an axis that has data. Assisted-by: ClaudeCode:claude-fable-5-1 --- changes/0000.bugfix.md | 2 +- src/zarr/core/metadata/v2.py | 22 ++++++++++++++++++---- tests/test_metadata/test_v2.py | 32 +++++++++++++++++++++++++------- 3 files changed, 44 insertions(+), 12 deletions(-) diff --git a/changes/0000.bugfix.md b/changes/0000.bugfix.md index 8b224c3f76..dc0cd52e28 100644 --- a/changes/0000.bugfix.md +++ b/changes/0000.bugfix.md @@ -1 +1 @@ -Zero-length dimensions are now handled by a single rule instead of a per-spelling patch: a chunk edge length is always at least 1, while a dimension's extent may be 0 (such a dimension simply has zero chunks). Every way of asking for "one chunk covering the axis" — `chunks=-1`, `chunks=False`, `chunks="auto"`, and `shards="auto"` — now derives the chunk size from the same helper, so they agree on chunk size 1 for a zero-length axis in both Zarr formats. Zarr format 2 metadata now rejects a chunk edge length of 0 with a clear error, matching Zarr format 3, instead of reading uninitialised data after a resize. Rectilinear chunk grids (`chunks=[[...], ...]`) can now be created on a zero-length dimension: since no list of positive edge lengths can sum to 0, the given edge lengths are stored as-is and describe the chunks the dimension will grow into on `append` or `resize`, exactly the state a rectilinear dimension is in after being resized down to 0. The private `FixedDimension(size=0, ...)` model, which previously carried its own zero-size special cases, now raises `ValueError`. +Zero-length dimensions are now handled by a single rule instead of a per-spelling patch: a chunk edge length is always at least 1, while a dimension's extent may be 0 (such a dimension simply has zero chunks). Every way of asking for "one chunk covering the axis" — `chunks=-1`, `chunks=False`, `chunks="auto"`, and `shards="auto"` — now derives the chunk size from the same helper, so they agree on chunk size 1 for a zero-length axis in both Zarr formats. Zarr format 2 metadata now applies the same rule: a stored chunk edge length of 0 on a zero-length axis — which zarr-python 2.x wrote for `chunks=False` and `chunks=(0,)` — is read as 1 (with a `ZarrUserWarning`) so those stores stay readable and no longer read uninitialised data after a resize, while a chunk edge of 0 on an axis that has data is rejected with a clear error. Rectilinear chunk grids (`chunks=[[...], ...]`) can now be created on a zero-length dimension: since no list of positive edge lengths can sum to 0, the given edge lengths are stored as-is and describe the chunks the dimension will grow into on `append` or `resize`, exactly the state a rectilinear dimension is in after being resized down to 0. The private `FixedDimension(size=0, ...)` model, which previously carried its own zero-size special cases, now raises `ValueError`. diff --git a/src/zarr/core/metadata/v2.py b/src/zarr/core/metadata/v2.py index b1fa2d866d..7d6e43fc61 100644 --- a/src/zarr/core/metadata/v2.py +++ b/src/zarr/core/metadata/v2.py @@ -90,12 +90,26 @@ def __init__( shape_parsed = parse_shapelike(shape) chunks_parsed = parse_shapelike(chunks) # Same invariant as the Zarr format 3 chunk grid metadata: every chunk edge - # length is at least 1, even on a zero-length axis. - for dim_idx, chunk in enumerate(chunks_parsed): + # length is at least 1. zarr-python 2.x wrote `chunks: [0]` for a zero-length + # axis created with `chunks=False` or `chunks=(0,)`. Such an axis holds no + # chunks, so those documents are readable; the edge is normalized to 1 so a + # later resize does not divide by zero. On an axis that has data, 0 is invalid. + normalized_chunks: list[int] = [] + for dim_idx, (extent, chunk) in enumerate(zip(shape_parsed, chunks_parsed, strict=False)): if chunk < 1: - raise ValueError( - f"Dimension {dim_idx}: chunk edge length must be >= 1, got {chunk}" + if chunk < 0 or extent != 0: + raise ValueError( + f"Dimension {dim_idx}: chunk edge length must be >= 1, got {chunk}" + ) + warnings.warn( + f"Dimension {dim_idx}: chunk edge length 0 on a zero-length axis " + "(as written by zarr-python 2.x) is treated as 1.", + ZarrUserWarning, + stacklevel=2, ) + chunk = 1 + normalized_chunks.append(chunk) + chunks_parsed = tuple(normalized_chunks) + chunks_parsed[len(shape_parsed) :] compressor_parsed = parse_compressor(compressor) order_parsed = parse_indexing_order(order) dimension_separator_parsed = parse_separator(dimension_separator) diff --git a/tests/test_metadata/test_v2.py b/tests/test_metadata/test_v2.py index ac3ee4c029..39f5961259 100644 --- a/tests/test_metadata/test_v2.py +++ b/tests/test_metadata/test_v2.py @@ -309,15 +309,33 @@ def test_from_dict_extra_fields() -> None: assert result == expected -@pytest.mark.parametrize(("shape", "chunks"), [((0,), (0,)), ((4, 0), (4, 0)), ((5,), (0,))]) -def test_zero_chunk_edge_rejected(shape: tuple[int, ...], chunks: tuple[int, ...]) -> None: - """A chunk edge length of 0 is invalid metadata, whatever the array shape. +@pytest.mark.parametrize( + ("shape", "chunks", "expected"), + [((0,), (0,), (1,)), ((4, 0), (4, 0), (4, 1)), ((0, 0), (0, 0), (1, 1))], +) +def test_zero_chunk_edge_on_empty_axis_normalized( + shape: tuple[int, ...], chunks: tuple[int, ...], expected: tuple[int, ...] +) -> None: + """A stored chunk edge of 0 on a zero-length axis is read as 1, with a warning. - Older releases could write `chunks: [0]` for a zero-length axis; such documents read - uninitialised memory once resized. The v2 layer now enforces the same `>= 1` rule as - the Zarr format 3 chunk grid. + zarr-python 2.x wrote `chunks: [0]` for such an axis (`chunks=False` or + `chunks=(0,)`), and those documents must stay readable. Left at 0, a later resize + read uninitialised memory; normalizing to 1 gives the axis the same grid every other + "one chunk spans the axis" spelling produces. """ - with pytest.raises(ValueError, match="chunk edge length must be >= 1, got 0"): + with pytest.warns(ZarrUserWarning, match="chunk edge length 0 on a zero-length axis"): + meta = ArrayV2Metadata( + shape=shape, dtype=Float64(), chunks=chunks, fill_value=0.0, order="C" + ) + assert meta.chunks == expected + + +@pytest.mark.parametrize(("shape", "chunks"), [((5,), (0,)), ((4, 3), (4, 0))]) +def test_zero_chunk_edge_with_data_rejected( + shape: tuple[int, ...], chunks: tuple[int, ...] +) -> None: + """A chunk edge of 0 on an axis that has data is invalid metadata.""" + with pytest.raises(ValueError, match="chunk edge length must be >= 1"): ArrayV2Metadata(shape=shape, dtype=Float64(), chunks=chunks, fill_value=0.0, order="C") From 921be0d73916abefd50aa1843d4fa35d841ca17c Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Wed, 9 Sep 2026 20:20:47 +0200 Subject: [PATCH 03/76] chore: rename changelog fragment to the PR number Assisted-by: ClaudeCode:claude-fable-5-1 --- changes/{0000.bugfix.md => 4334.bugfix.md} | 0 1 file changed, 0 insertions(+), 0 deletions(-) rename changes/{0000.bugfix.md => 4334.bugfix.md} (100%) diff --git a/changes/0000.bugfix.md b/changes/4334.bugfix.md similarity index 100% rename from changes/0000.bugfix.md rename to changes/4334.bugfix.md From 047a92e3246e9ead0329be04afd5f22e12ba04bf Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Wed, 9 Sep 2026 20:24:28 +0200 Subject: [PATCH 04/76] docs: state what 2.x actually did with a zero chunk edge Measured against zarr 2.18.7: `zeros((0,), chunks=False)`, `chunks=-1` and `chunks=(0,)` all write `chunks: [0]`, after which nchunks, read, write, append, resize and reopen-then-read every raise ZeroDivisionError. There was never a working behaviour to preserve; normalizing the edge to 1 makes such arrays usable for the first time. Say so in the comment and fragment instead of claiming the stores were previously readable. Assisted-by: ClaudeCode:claude-fable-5-1 --- changes/4334.bugfix.md | 2 +- src/zarr/core/metadata/v2.py | 9 ++++++--- 2 files changed, 7 insertions(+), 4 deletions(-) diff --git a/changes/4334.bugfix.md b/changes/4334.bugfix.md index dc0cd52e28..8a04bec7e1 100644 --- a/changes/4334.bugfix.md +++ b/changes/4334.bugfix.md @@ -1 +1 @@ -Zero-length dimensions are now handled by a single rule instead of a per-spelling patch: a chunk edge length is always at least 1, while a dimension's extent may be 0 (such a dimension simply has zero chunks). Every way of asking for "one chunk covering the axis" — `chunks=-1`, `chunks=False`, `chunks="auto"`, and `shards="auto"` — now derives the chunk size from the same helper, so they agree on chunk size 1 for a zero-length axis in both Zarr formats. Zarr format 2 metadata now applies the same rule: a stored chunk edge length of 0 on a zero-length axis — which zarr-python 2.x wrote for `chunks=False` and `chunks=(0,)` — is read as 1 (with a `ZarrUserWarning`) so those stores stay readable and no longer read uninitialised data after a resize, while a chunk edge of 0 on an axis that has data is rejected with a clear error. Rectilinear chunk grids (`chunks=[[...], ...]`) can now be created on a zero-length dimension: since no list of positive edge lengths can sum to 0, the given edge lengths are stored as-is and describe the chunks the dimension will grow into on `append` or `resize`, exactly the state a rectilinear dimension is in after being resized down to 0. The private `FixedDimension(size=0, ...)` model, which previously carried its own zero-size special cases, now raises `ValueError`. +Zero-length dimensions are now handled by a single rule instead of a per-spelling patch: a chunk edge length is always at least 1, while a dimension's extent may be 0 (such a dimension simply has zero chunks). Every way of asking for "one chunk covering the axis" — `chunks=-1`, `chunks=False`, `chunks="auto"`, and `shards="auto"` — now derives the chunk size from the same helper, so they agree on chunk size 1 for a zero-length axis in both Zarr formats. Zarr format 2 metadata now applies the same rule: a stored chunk edge length of 0 on a zero-length axis — which zarr-python 2.x wrote for `chunks=False`, `chunks=-1` and `chunks=(0,)` and then could never read, write, append to or resize (every operation raised `ZeroDivisionError`), and which 3.0–3.3 opened but lost data on append — is read as 1 with a `ZarrUserWarning`, making such arrays usable, while a chunk edge of 0 on an axis that has data is rejected with a clear error. Rectilinear chunk grids (`chunks=[[...], ...]`) can now be created on a zero-length dimension: since no list of positive edge lengths can sum to 0, the given edge lengths are stored as-is and describe the chunks the dimension will grow into on `append` or `resize`, exactly the state a rectilinear dimension is in after being resized down to 0. The private `FixedDimension(size=0, ...)` model, which previously carried its own zero-size special cases, now raises `ValueError`. diff --git a/src/zarr/core/metadata/v2.py b/src/zarr/core/metadata/v2.py index 7d6e43fc61..a098132ab9 100644 --- a/src/zarr/core/metadata/v2.py +++ b/src/zarr/core/metadata/v2.py @@ -91,9 +91,12 @@ def __init__( chunks_parsed = parse_shapelike(chunks) # Same invariant as the Zarr format 3 chunk grid metadata: every chunk edge # length is at least 1. zarr-python 2.x wrote `chunks: [0]` for a zero-length - # axis created with `chunks=False` or `chunks=(0,)`. Such an axis holds no - # chunks, so those documents are readable; the edge is normalized to 1 so a - # later resize does not divide by zero. On an axis that has data, 0 is invalid. + # axis created with `chunks=False`, `-1` or `(0,)`, and then could not read, + # write, append to or resize the array (every operation divided by zero); + # 3.0-3.3 opened such documents but lost data on append. The axis holds no + # chunks, so the edge is normalized to 1 — the grid every other "one chunk + # spans the axis" spelling produces — which makes the array usable at last. + # On an axis that has data, 0 is invalid and any data was never stored. normalized_chunks: list[int] = [] for dim_idx, (extent, chunk) in enumerate(zip(shape_parsed, chunks_parsed, strict=False)): if chunk < 1: From 2297c62745f18ad698f3357a12796ff34ad95ec6 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Sun, 13 Sep 2026 17:41:59 +0200 Subject: [PATCH 05/76] docs: qualify zero-chunk compatibility history Assisted-by: Codex:GPT-6 --- changes/4334.bugfix.md | 2 +- docs/user-guide/arrays.md | 2 +- src/zarr/core/chunk_grids.py | 4 ++-- src/zarr/core/metadata/v2.py | 14 +++++++------- tests/test_metadata/test_v2.py | 11 +++++------ 5 files changed, 16 insertions(+), 17 deletions(-) diff --git a/changes/4334.bugfix.md b/changes/4334.bugfix.md index 8a04bec7e1..5fc5373ee5 100644 --- a/changes/4334.bugfix.md +++ b/changes/4334.bugfix.md @@ -1 +1 @@ -Zero-length dimensions are now handled by a single rule instead of a per-spelling patch: a chunk edge length is always at least 1, while a dimension's extent may be 0 (such a dimension simply has zero chunks). Every way of asking for "one chunk covering the axis" — `chunks=-1`, `chunks=False`, `chunks="auto"`, and `shards="auto"` — now derives the chunk size from the same helper, so they agree on chunk size 1 for a zero-length axis in both Zarr formats. Zarr format 2 metadata now applies the same rule: a stored chunk edge length of 0 on a zero-length axis — which zarr-python 2.x wrote for `chunks=False`, `chunks=-1` and `chunks=(0,)` and then could never read, write, append to or resize (every operation raised `ZeroDivisionError`), and which 3.0–3.3 opened but lost data on append — is read as 1 with a `ZarrUserWarning`, making such arrays usable, while a chunk edge of 0 on an axis that has data is rejected with a clear error. Rectilinear chunk grids (`chunks=[[...], ...]`) can now be created on a zero-length dimension: since no list of positive edge lengths can sum to 0, the given edge lengths are stored as-is and describe the chunks the dimension will grow into on `append` or `resize`, exactly the state a rectilinear dimension is in after being resized down to 0. The private `FixedDimension(size=0, ...)` model, which previously carried its own zero-size special cases, now raises `ValueError`. +Chunk-grid normalization now shares a helper that selects chunk size 1 for a zero-length axis when inferring a full-span chunk size. Chunk edge lengths remain positive while array extents may be zero. Zarr format 2 metadata with a stored chunk edge of 0 on a zero-length axis is interpreted as chunk size 1 with a `ZarrUserWarning`; a zero chunk edge on a positive-length axis is rejected. This supports legacy metadata such as the zero chunk sizes written by zarr-python 2.18.7 for empty arrays created with `chunks=False`, `chunks=-1`, or `chunks=(0,)`, without claiming that every historical reader behaved identically. Rectilinear chunk grids can be created on a zero-length dimension with a non-empty list of positive chunk sizes; these sizes are retained for subsequent growth. `FixedDimension(size=0, ...)` now raises `ValueError`. diff --git a/docs/user-guide/arrays.md b/docs/user-guide/arrays.md index 1fc0670c1b..e8df79a102 100644 --- a/docs/user-guide/arrays.md +++ b/docs/user-guide/arrays.md @@ -708,7 +708,7 @@ print(f"After append: shape={z.shape}, chunk_sizes={z.write_chunk_sizes}") ``` A rectilinear array can also be created with a zero-length dimension: because no -list of positive chunk sizes can sum to 0, the chunk sizes given for such a +non-empty list of positive chunk sizes can sum to 0, the chunk sizes given for such a dimension are stored as-is and describe the chunks the dimension will grow into on `append` or `resize` — the same state as resizing an existing rectilinear dimension down to 0. diff --git a/src/zarr/core/chunk_grids.py b/src/zarr/core/chunk_grids.py index 340d228e15..545aa3c581 100644 --- a/src/zarr/core/chunk_grids.py +++ b/src/zarr/core/chunk_grids.py @@ -762,8 +762,8 @@ def normalize_chunks_1d(chunks: int | Iterable[object], span: int) -> DimensionG which would change how the dimension grows on resize. For scalar sizes the last chunk may overhang the span. - The one exception to the sum rule is a zero-length span: no list of - positive edges can sum to 0, so any non-empty list is accepted verbatim + The one exception to the sum rule is a zero-length span: no non-empty list of + positive edges can sum to 0, so a non-empty list of positive integers is retained and the edges describe the chunks the axis will grow into on `append` / `resize`. This is the same state a rectilinear axis reaches when it is resized down to 0 — `VaryingDimension` allows trailing edges beyond the diff --git a/src/zarr/core/metadata/v2.py b/src/zarr/core/metadata/v2.py index a098132ab9..6389146e7e 100644 --- a/src/zarr/core/metadata/v2.py +++ b/src/zarr/core/metadata/v2.py @@ -90,13 +90,13 @@ def __init__( shape_parsed = parse_shapelike(shape) chunks_parsed = parse_shapelike(chunks) # Same invariant as the Zarr format 3 chunk grid metadata: every chunk edge - # length is at least 1. zarr-python 2.x wrote `chunks: [0]` for a zero-length - # axis created with `chunks=False`, `-1` or `(0,)`, and then could not read, - # write, append to or resize the array (every operation divided by zero); - # 3.0-3.3 opened such documents but lost data on append. The axis holds no - # chunks, so the edge is normalized to 1 — the grid every other "one chunk - # spans the axis" spelling produces — which makes the array usable at last. - # On an axis that has data, 0 is invalid and any data was never stored. + # length is at least 1. zarr-python 2.18.7 can write `chunks: [0]` for + # a zero-length axis created with `chunks=False`, `-1` or `(0,)`. + # Normalize that empty axis to chunk size 1 so it can use the positive-size + # grid model. This is a compatibility policy for legacy metadata, not a + # statement about every historical reader. A zero chunk size on a + # positive-length axis is rejected; metadata alone cannot establish + # whether the store contains chunk payloads. normalized_chunks: list[int] = [] for dim_idx, (extent, chunk) in enumerate(zip(shape_parsed, chunks_parsed, strict=False)): if chunk < 1: diff --git a/tests/test_metadata/test_v2.py b/tests/test_metadata/test_v2.py index 39f5961259..1e578c033c 100644 --- a/tests/test_metadata/test_v2.py +++ b/tests/test_metadata/test_v2.py @@ -318,10 +318,9 @@ def test_zero_chunk_edge_on_empty_axis_normalized( ) -> None: """A stored chunk edge of 0 on a zero-length axis is read as 1, with a warning. - zarr-python 2.x wrote `chunks: [0]` for such an axis (`chunks=False` or - `chunks=(0,)`), and those documents must stay readable. Left at 0, a later resize - read uninitialised memory; normalizing to 1 gives the axis the same grid every other - "one chunk spans the axis" spelling produces. + This checks the compatibility policy for legacy metadata: the in-memory + chunk size becomes positive while the extent remains zero. It does not + test historical reader behavior or perform array I/O. """ with pytest.warns(ZarrUserWarning, match="chunk edge length 0 on a zero-length axis"): meta = ArrayV2Metadata( @@ -331,10 +330,10 @@ def test_zero_chunk_edge_on_empty_axis_normalized( @pytest.mark.parametrize(("shape", "chunks"), [((5,), (0,)), ((4, 3), (4, 0))]) -def test_zero_chunk_edge_with_data_rejected( +def test_zero_chunk_edge_with_positive_extent_rejected( shape: tuple[int, ...], chunks: tuple[int, ...] ) -> None: - """A chunk edge of 0 on an axis that has data is invalid metadata.""" + """A chunk edge of 0 on a positive-length axis is rejected.""" with pytest.raises(ValueError, match="chunk edge length must be >= 1"): ArrayV2Metadata(shape=shape, dtype=Float64(), chunks=chunks, fill_value=0.0, order="C") From 92ead03d6e88b712888d1bfb15c17a4785f6b576 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Fri, 18 Sep 2026 18:11:15 +0200 Subject: [PATCH 06/76] fix: read mixed regular/rectilinear chunk grids written by zarr 3.2.x zarr 3.2.0 and 3.2.1 classified mixed chunk specs like (2, (5, 10, 5)) as regular and stored them as a "regular" grid whose chunk_shape contains an edge list, while laying the chunks out as a rectilinear grid. Newer versions accepted that metadata and failed later with an unrelated TypeError. RegularChunkGridMetadata now rejects edge lists. When reading stored metadata, a "regular" grid with edge lists is read as the rectilinear grid it describes, with a warning explaining how to re-save it; if rectilinear chunks are disabled, the error says what happened and how to enable them. Closes #4374 Assisted-by: ClaudeCode:claude-opus-5 --- src/zarr/core/metadata/v3.py | 61 ++++++++++++++++++++-- tests/test_metadata/test_v3.py | 93 +++++++++++++++++++++++++++++++++- 2 files changed, 148 insertions(+), 6 deletions(-) diff --git a/src/zarr/core/metadata/v3.py b/src/zarr/core/metadata/v3.py index 11f3eb593d..8874097f01 100644 --- a/src/zarr/core/metadata/v3.py +++ b/src/zarr/core/metadata/v3.py @@ -1,6 +1,7 @@ from __future__ import annotations import json +import warnings from collections.abc import Iterable, Mapping, Sequence from dataclasses import dataclass, field, replace from typing import TYPE_CHECKING, Any, Final, Literal, NotRequired, TypeGuard, cast @@ -36,7 +37,7 @@ from zarr.core.dtype.common import check_dtype_spec_v3 from zarr.core.json_parse import parse_field from zarr.core.metadata.common import parse_attributes -from zarr.errors import MetadataValidationError, NodeTypeValidationError +from zarr.errors import MetadataValidationError, NodeTypeValidationError, ZarrUserWarning from zarr.registry import get_codec_class if TYPE_CHECKING: @@ -215,10 +216,19 @@ class RectilinearChunkGridMetadataConfig(TypedDict): def _parse_chunk_shape(chunk_shape: Iterable[int]) -> tuple[int, ...]: """Validate and normalize a regular chunk shape. - Delegates to ``_validate_chunk_shapes`` — a regular chunk shape is just - a sequence of bare ints (one per dimension), each of which must be >= 1. + A regular chunk shape is a sequence of bare ints (one per dimension), + each of which must be >= 1. Per-dimension edge lists belong to a + rectilinear grid and are rejected here. """ - result = _validate_chunk_shapes(tuple(chunk_shape)) + chunk_shape = tuple(chunk_shape) + for dim_idx, dim_spec in enumerate(chunk_shape): + if not isinstance(dim_spec, int): + raise TypeError( + f"Dimension {dim_idx}: a regular chunk grid requires an integer chunk " + f"edge length, got {dim_spec!r}. Per-dimension chunk edge lists " + "require a rectilinear chunk grid." + ) + result = _validate_chunk_shapes(chunk_shape) # Regular grids only have bare ints — cast is safe after validation return cast(tuple[int, ...], result) @@ -429,14 +439,55 @@ def parse_chunk_grid( if isinstance(data, (RegularChunkGridMetadata, RectilinearChunkGridMetadata)): return data - name, _ = parse_named_configuration(data) + name, configuration = parse_named_configuration(data) if name == "regular": + chunk_shape = configuration.get("chunk_shape") + if isinstance(chunk_shape, list | tuple) and any( + isinstance(dim_spec, list | tuple) for dim_spec in chunk_shape + ): + return _parse_mixed_regular_chunk_grid(chunk_shape) return RegularChunkGridMetadata.from_dict(data) # type: ignore[arg-type] if name == "rectilinear": return RectilinearChunkGridMetadata.from_dict(data) # type: ignore[arg-type] raise ValueError(f"Unknown chunk grid name: {name!r}") +def _parse_mixed_regular_chunk_grid( + chunk_shape: Sequence[JSON], +) -> RectilinearChunkGridMetadata: + """Read a "regular" chunk grid whose chunk_shape contains edge lists. + + zarr 3.2.0 and 3.2.1 wrote mixed chunk specs such as ``(2, (5, 10, 5))`` + as ``{"name": "regular", "configuration": {"chunk_shape": [2, [5, 10, 5]]}}`` + while laying the chunks out as a rectilinear grid. That metadata is + invalid, but the data is intact, so it is read as the rectilinear grid + it describes. See https://github.com/zarr-developers/zarr-python/issues/4374. + """ + if not config.get("array.rectilinear_chunks"): + raise ValueError( + f"This array's chunk grid is named 'regular' but its chunk_shape " + f"{list(chunk_shape)!r} lists explicit chunk edges for some dimensions. " + "zarr 3.2.0 and 3.2.1 wrote rectilinear chunk grids this way by mistake. " + "Reading it as a rectilinear chunk grid requires enabling rectilinear chunks: " + "zarr.config.set({'array.rectilinear_chunks': True})" + ) + warnings.warn( + f"This array's chunk grid is named 'regular' but its chunk_shape " + f"{list(chunk_shape)!r} lists explicit chunk edges for some dimensions. " + "zarr 3.2.0 and 3.2.1 wrote rectilinear chunk grids this way by mistake. " + "Reading it as a rectilinear chunk grid. Re-save the array metadata " + "(e.g. with `array.update_attributes({})`) to store a valid rectilinear chunk grid.", + ZarrUserWarning, + stacklevel=2, + ) + return RectilinearChunkGridMetadata.from_dict( + { + "name": "rectilinear", + "configuration": {"kind": "inline", "chunk_shapes": chunk_shape}, # type: ignore[typeddict-item] + } + ) + + class ArrayMetadataJSON_V3(TypedDict, extra_items=AllowedExtraField): # type: ignore[call-arg] """ A typed dictionary model for zarr v3 array metadata. diff --git a/tests/test_metadata/test_v3.py b/tests/test_metadata/test_v3.py index 9f78ed7b70..0284e7262c 100644 --- a/tests/test_metadata/test_v3.py +++ b/tests/test_metadata/test_v3.py @@ -5,11 +5,13 @@ import json from typing import TYPE_CHECKING +import numpy as np import pytest +import zarr from tests.conftest import Expect, ExpectFail from tests.test_metadata.conftest import minimal_metadata_dict_v3 -from zarr.core.buffer import default_buffer_prototype +from zarr.core.buffer import cpu, default_buffer_prototype from zarr.core.chunk_grids import ChunkGrid, is_regular_1d, is_regular_nd from zarr.core.config import config from zarr.core.dtype import Float64, UInt8 @@ -19,6 +21,8 @@ ARRAY_METADATA_KEYS, ArrayMetadataJSON_V3, ArrayV3Metadata, + RectilinearChunkGridMetadata, + RegularChunkGridMetadata, create_chunk_grid_metadata, parse_codecs, parse_dimension_names, @@ -29,6 +33,7 @@ MetadataValidationError, NodeTypeValidationError, UnknownCodecError, + ZarrUserWarning, ) if TYPE_CHECKING: @@ -154,6 +159,92 @@ def test_create_chunk_grid_metadata_unknown_dimension_type() -> None: create_chunk_grid_metadata(grid) +def test_regular_chunk_grid_rejects_edge_lists() -> None: + """A regular chunk grid only accepts integer chunk edge lengths.""" + with pytest.raises(TypeError, match="Dimension 1: a regular chunk grid requires an integer"): + RegularChunkGridMetadata(chunk_shape=(2, (5, 10, 5))) # type: ignore[arg-type] + + +# --------------------------------------------------------------------------- +# Mixed regular/rectilinear chunk grids written by zarr 3.2.x +# https://github.com/zarr-developers/zarr-python/issues/4374 +# --------------------------------------------------------------------------- + + +@pytest.mark.parametrize( + "case", + [ + Expect( + input=((6, 20), [2, [5, 10, 5]]), + output=(2, (5, 10, 5)), + id="edges_in_last_dim", + ), + Expect( + input=((6, 20, 4), [2, [5, 10, 5], 4]), + output=(2, (5, 10, 5), 4), + id="edges_in_middle_dim", + ), + Expect( + input=((6, 20), [[1, 5], [5, 10, 5]]), + output=((1, 5), (5, 10, 5)), + id="edges_in_every_dim", + ), + ], + ids=lambda case: case.id, +) +def test_read_mixed_regular_chunk_grid( + case: Expect[tuple[tuple[int, ...], list[Any]], tuple[Any, ...]], +) -> None: + """A "regular" chunk grid whose chunk_shape lists edges is read as rectilinear.""" + shape, chunk_shape = case.input + d = minimal_metadata_dict_v3( + shape=shape, + chunk_grid={"name": "regular", "configuration": {"chunk_shape": chunk_shape}}, + ) + with config.set({"array.rectilinear_chunks": True}): + with pytest.warns(ZarrUserWarning, match="zarr 3.2.0 and 3.2.1"): + meta = ArrayV3Metadata.from_dict(d) # type: ignore[arg-type] + assert meta.chunk_grid == RectilinearChunkGridMetadata(chunk_shapes=case.output) + + +def test_read_mixed_regular_chunk_grid_requires_rectilinear_chunks() -> None: + """Reading a mixed "regular" chunk grid explains why rectilinear chunks must be enabled.""" + d = minimal_metadata_dict_v3( + shape=(6, 20), + chunk_grid={"name": "regular", "configuration": {"chunk_shape": [2, [5, 10, 5]]}}, + ) + with ( + config.set({"array.rectilinear_chunks": False}), + pytest.raises(ValueError, match="zarr 3.2.0 and 3.2.1 wrote rectilinear chunk grids"), + ): + ArrayV3Metadata.from_dict(d) # type: ignore[arg-type] + + +def test_open_array_with_mixed_regular_chunk_grid() -> None: + """An array stored with zarr 3.2.x's mixed chunk grid reads correctly and re-saves as rectilinear.""" + data = np.arange(120, dtype="float32").reshape(6, 20) + store = zarr.storage.MemoryStore() + with config.set({"array.rectilinear_chunks": True}): + arr = zarr.create_array(store, shape=data.shape, chunks=(2, (5, 10, 5)), dtype=data.dtype) + arr[:] = data + # Rewrite the metadata the way zarr 3.2.0 and 3.2.1 stored it. The chunk + # layout is unchanged: 3.2.x wrote the chunks as a rectilinear grid. + doc = json.loads(store._store_dict["zarr.json"].to_bytes()) + doc["chunk_grid"] = {"name": "regular", "configuration": {"chunk_shape": [2, [5, 10, 5]]}} + store._store_dict["zarr.json"] = cpu.Buffer.from_bytes(json.dumps(doc).encode()) + + with pytest.warns(ZarrUserWarning, match="zarr 3.2.0 and 3.2.1"): + arr = zarr.open_array(store) + np.testing.assert_array_equal(arr[:], data) + + arr.update_attributes({}) + doc = json.loads(store._store_dict["zarr.json"].to_bytes()) + assert doc["chunk_grid"] == { + "name": "rectilinear", + "configuration": {"kind": "inline", "chunk_shapes": [2, [5, 10, 5]]}, + } + + # --------------------------------------------------------------------------- # Types # --------------------------------------------------------------------------- From 7259833ee494b8b5dfc9f051633a1e6e157fc60c Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Fri, 18 Sep 2026 18:11:49 +0200 Subject: [PATCH 07/76] docs: add changelog fragment for #4375 Assisted-by: ClaudeCode:claude-opus-5 --- changes/4375.bugfix.md | 1 + 1 file changed, 1 insertion(+) create mode 100644 changes/4375.bugfix.md diff --git a/changes/4375.bugfix.md b/changes/4375.bugfix.md new file mode 100644 index 0000000000..0a76d24539 --- /dev/null +++ b/changes/4375.bugfix.md @@ -0,0 +1 @@ +Arrays written by zarr 3.2.0 and 3.2.1 with a mixed regular/rectilinear chunk specification (e.g. `chunks=(2, (5, 10, 5))`) can be read again. Their stored "regular" chunk grid is read as the rectilinear grid it describes, with a warning explaining how to re-save the metadata. A regular chunk grid with per-dimension edge lists is now rejected when constructed. From 955fdebde7abd59237725ab36c0301295c4cd974 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Fri, 18 Sep 2026 18:19:18 +0200 Subject: [PATCH 08/76] fix: accept numpy integers in regular chunk grids; normalize tuples in the 3.2.x shim Review follow-ups for #4375: - RegularChunkGridMetadata now accepts numpy integer scalars and stores Python ints, mirroring parse_shapelike. The strict integer check had reported np.int64(2) as if it were a list of chunk edges. - The mixed "regular" grid reader converts tuple entries to lists before delegating to RectilinearChunkGridMetadata.from_dict, so metadata dicts built in Python are handled the same as parsed JSON. - The error and warning share one message prefix. - Tests cover numpy ints, run-length encoded edges, and tuple input, and the test section header says the reader is broader than the 3.2.x bug. Assisted-by: ClaudeCode:claude-fable-5-1 --- src/zarr/core/metadata/v3.py | 31 ++++++++++++++++-------------- tests/test_metadata/test_v3.py | 35 +++++++++++++++++++++++++++++++--- 2 files changed, 49 insertions(+), 17 deletions(-) diff --git a/src/zarr/core/metadata/v3.py b/src/zarr/core/metadata/v3.py index 8874097f01..4f41143c17 100644 --- a/src/zarr/core/metadata/v3.py +++ b/src/zarr/core/metadata/v3.py @@ -6,6 +6,7 @@ from dataclasses import dataclass, field, replace from typing import TYPE_CHECKING, Any, Final, Literal, NotRequired, TypeGuard, cast +import numpy as np from typing_extensions import TypedDict from zarr.abc.codec import ArrayArrayCodec, ArrayBytesCodec, BytesBytesCodec, Codec @@ -220,15 +221,16 @@ def _parse_chunk_shape(chunk_shape: Iterable[int]) -> tuple[int, ...]: each of which must be >= 1. Per-dimension edge lists belong to a rectilinear grid and are rejected here. """ - chunk_shape = tuple(chunk_shape) + normalized: list[int] = [] for dim_idx, dim_spec in enumerate(chunk_shape): - if not isinstance(dim_spec, int): + if not isinstance(dim_spec, int | np.integer): raise TypeError( f"Dimension {dim_idx}: a regular chunk grid requires an integer chunk " - f"edge length, got {dim_spec!r}. Per-dimension chunk edge lists " - "require a rectilinear chunk grid." + f"edge length, got {dim_spec!r}. Lists of chunk edge lengths belong " + "to a rectilinear chunk grid." ) - result = _validate_chunk_shapes(chunk_shape) + normalized.append(int(dim_spec)) + result = _validate_chunk_shapes(tuple(normalized)) # Regular grids only have bare ints — cast is safe after validation return cast(tuple[int, ...], result) @@ -463,19 +465,20 @@ def _parse_mixed_regular_chunk_grid( invalid, but the data is intact, so it is read as the rectilinear grid it describes. See https://github.com/zarr-developers/zarr-python/issues/4374. """ + # Tuples arrive when the metadata dict was built in Python rather than parsed from JSON. + chunk_shapes = [list(d) if isinstance(d, tuple) else d for d in chunk_shape] + msg = ( + f"This array's chunk grid is named 'regular' but its chunk_shape {chunk_shapes!r} " + "lists explicit chunk edges for some dimensions. zarr 3.2.0 and 3.2.1 wrote " + "rectilinear chunk grids this way by mistake. " + ) if not config.get("array.rectilinear_chunks"): raise ValueError( - f"This array's chunk grid is named 'regular' but its chunk_shape " - f"{list(chunk_shape)!r} lists explicit chunk edges for some dimensions. " - "zarr 3.2.0 and 3.2.1 wrote rectilinear chunk grids this way by mistake. " - "Reading it as a rectilinear chunk grid requires enabling rectilinear chunks: " + msg + "Reading it as a rectilinear chunk grid requires enabling rectilinear chunks: " "zarr.config.set({'array.rectilinear_chunks': True})" ) warnings.warn( - f"This array's chunk grid is named 'regular' but its chunk_shape " - f"{list(chunk_shape)!r} lists explicit chunk edges for some dimensions. " - "zarr 3.2.0 and 3.2.1 wrote rectilinear chunk grids this way by mistake. " - "Reading it as a rectilinear chunk grid. Re-save the array metadata " + msg + "Reading it as a rectilinear chunk grid. Re-save the array metadata " "(e.g. with `array.update_attributes({})`) to store a valid rectilinear chunk grid.", ZarrUserWarning, stacklevel=2, @@ -483,7 +486,7 @@ def _parse_mixed_regular_chunk_grid( return RectilinearChunkGridMetadata.from_dict( { "name": "rectilinear", - "configuration": {"kind": "inline", "chunk_shapes": chunk_shape}, # type: ignore[typeddict-item] + "configuration": {"kind": "inline", "chunk_shapes": chunk_shapes}, # type: ignore[typeddict-item] } ) diff --git a/tests/test_metadata/test_v3.py b/tests/test_metadata/test_v3.py index 0284e7262c..f945baf066 100644 --- a/tests/test_metadata/test_v3.py +++ b/tests/test_metadata/test_v3.py @@ -37,6 +37,7 @@ ) if TYPE_CHECKING: + from collections.abc import Sequence from typing import Any @@ -159,6 +160,21 @@ def test_create_chunk_grid_metadata_unknown_dimension_type() -> None: create_chunk_grid_metadata(grid) +@pytest.mark.parametrize( + "case", + [ + Expect(input=(2, 3), output=(2, 3), id="python_ints"), + Expect(input=(np.int64(2), np.uint8(3)), output=(2, 3), id="numpy_ints"), + ], + ids=lambda case: case.id, +) +def test_regular_chunk_grid_normalizes_ints(case: Expect[tuple[Any, ...], tuple[int, ...]]) -> None: + """A regular chunk grid accepts Python and numpy integers and stores Python ints.""" + grid = RegularChunkGridMetadata(chunk_shape=case.input) + assert grid.chunk_shape == case.output + assert all(type(c) is int for c in grid.chunk_shape) + + def test_regular_chunk_grid_rejects_edge_lists() -> None: """A regular chunk grid only accepts integer chunk edge lengths.""" with pytest.raises(TypeError, match="Dimension 1: a regular chunk grid requires an integer"): @@ -166,8 +182,11 @@ def test_regular_chunk_grid_rejects_edge_lists() -> None: # --------------------------------------------------------------------------- -# Mixed regular/rectilinear chunk grids written by zarr 3.2.x -# https://github.com/zarr-developers/zarr-python/issues/4374 +# "regular" chunk grids whose chunk_shape contains edge lists +# +# zarr 3.2.0 and 3.2.1 wrote mixed specs such as (2, (5, 10, 5)) this way +# (https://github.com/zarr-developers/zarr-python/issues/4374). The reader +# accepts any such grid, not only the shapes 3.2.x could produce. # --------------------------------------------------------------------------- @@ -189,11 +208,21 @@ def test_regular_chunk_grid_rejects_edge_lists() -> None: output=((1, 5), (5, 10, 5)), id="edges_in_every_dim", ), + Expect( + input=((6, 20), [2, [[5, 2], 10]]), + output=(2, (5, 5, 10)), + id="run_length_encoded_edges", + ), + Expect( + input=((6, 20), (2, (5, 10, 5))), + output=(2, (5, 10, 5)), + id="tuples_from_python", + ), ], ids=lambda case: case.id, ) def test_read_mixed_regular_chunk_grid( - case: Expect[tuple[tuple[int, ...], list[Any]], tuple[Any, ...]], + case: Expect[tuple[tuple[int, ...], Sequence[Any]], tuple[Any, ...]], ) -> None: """A "regular" chunk grid whose chunk_shape lists edges is read as rectilinear.""" shape, chunk_shape = case.input From f57d01b3fe36e0b2a295e5275bdf9f94f2b965bb Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Sat, 19 Sep 2026 17:10:33 +0200 Subject: [PATCH 09/76] refactor: branch on integer vs sequence when parsing rectilinear chunk shapes The chunk_shapes parsers checked exact types: `from_dict` accepted a dimension only as `int` or `list`, `expand_rle` accepted an RLE pair only as a `list`, and the reader for 3.2.x mixed grids special-cased `tuple` to convert it to a list before handing it to `from_dict`. A metadata dict built in Python holds tuples where parsed JSON holds lists, and a numpy array is as good as either, so these checks rejected valid input. One structural predicate, `declares_chunk_edges`, now answers "is this a sequence of edges rather than a single integer size" for all of them: any non-integer iterable except `str`/`bytes`. It is a `TypeGuard`, not a `TypeIs`, because it is False for strings, which are iterable. The four sites that make that decision use it: `parse_chunk_grid`'s detection of a mixed grid, the mixed-grid reader, `RectilinearChunkGridMetadata.from_dict`, and `expand_rle`. The shim no longer needs its tuple special case. Integers are `int | np.integer` throughout, matching `_parse_chunk_shape`, and every parser stores Python ints: `_validate_chunk_shapes` now coerces, so constructing a rectilinear grid from numpy values works the way it already did for a regular grid. Assisted-by: ClaudeCode:claude-opus-5 --- src/zarr/core/common.py | 34 +++++++++++++++++----- src/zarr/core/metadata/v3.py | 36 ++++++++++++++--------- tests/test_metadata/test_v3.py | 28 ++++++++++++++++++ tests/test_unified_chunk_grid.py | 50 +++++++++++++++++++++++++++++++- 4 files changed, 125 insertions(+), 23 deletions(-) diff --git a/src/zarr/core/common.py b/src/zarr/core/common.py index ba01f6c19f..d6b33dd5c1 100644 --- a/src/zarr/core/common.py +++ b/src/zarr/core/common.py @@ -12,6 +12,7 @@ Literal, NotRequired, TypedDict, + TypeGuard, cast, overload, ) @@ -276,7 +277,21 @@ def _default_zarr_format() -> ZarrFormat: return cast("ZarrFormat", int(zarr_config.get("default_zarr_format", 3))) -def expand_rle(data: Sequence[int | list[int]]) -> list[int]: +def declares_chunk_edges(value: object) -> TypeGuard[Iterable[Any]]: + """Whether a chunk specification element declares explicit chunk edge lengths. + + True for any non-integer iterable, which is the form explicit edge lengths + take, and False for a single integer size. The test is structural rather + than an exact type check: edge lengths reach us as a list parsed from JSON, + as a tuple from a metadata dict built in Python, or as a numpy array, and + all three mean the same thing. + """ + if isinstance(value, int | np.integer): + return False + return isinstance(value, Iterable) and not isinstance(value, str | bytes) + + +def expand_rle(data: Iterable[int | Sequence[int]]) -> list[int]: """Expand a mixed array of bare integers and RLE pairs. Per the rectilinear chunk grid spec, each element can be: @@ -285,18 +300,21 @@ def expand_rle(data: Sequence[int | list[int]]) -> list[int]: """ result: list[int] = [] for item in data: - if isinstance(item, (int, float)) and not isinstance(item, bool): - val = int(item) - if val < 1: - raise ValueError(f"Chunk edge length must be >= 1, got {val}") - result.append(val) - elif isinstance(item, list) and len(item) == 2: - size, count = int(item[0]), int(item[1]) + if declares_chunk_edges(item): + pair = tuple(item) + if len(pair) != 2: + raise ValueError(f"RLE entries must be an integer or [size, count], got {item}") + size, count = int(pair[0]), int(pair[1]) if size < 1: raise ValueError(f"Chunk edge length must be >= 1, got {size}") if count < 1: raise ValueError(f"RLE repeat count must be >= 1, got {count}") result.extend([size] * count) + elif isinstance(item, int | np.integer | float) and not isinstance(item, bool): + val = int(item) + if val < 1: + raise ValueError(f"Chunk edge length must be >= 1, got {val}") + result.append(val) else: raise ValueError(f"RLE entries must be an integer or [size, count], got {item}") return result diff --git a/src/zarr/core/metadata/v3.py b/src/zarr/core/metadata/v3.py index 4f41143c17..5bb2993486 100644 --- a/src/zarr/core/metadata/v3.py +++ b/src/zarr/core/metadata/v3.py @@ -27,6 +27,7 @@ NamedConfig, NamedRequiredConfig, compress_rle, + declares_chunk_edges, expand_rle, parse_named_configuration, parse_shapelike, @@ -245,14 +246,14 @@ def _validate_chunk_shapes( """ result: list[int | tuple[int, ...]] = [] for dim_idx, dim_spec in enumerate(chunk_shapes): - if isinstance(dim_spec, int): + if isinstance(dim_spec, int | np.integer): if dim_spec < 1: raise ValueError( f"Dimension {dim_idx}: integer chunk edge length must be >= 1, got {dim_spec}" ) - result.append(dim_spec) + result.append(int(dim_spec)) else: - edges = tuple(dim_spec) + edges = tuple(int(edge) for edge in dim_spec) if not edges: raise ValueError(f"Dimension {dim_idx} has no chunk edges.") bad = [i for i, e in enumerate(edges) if e < 1] @@ -380,15 +381,16 @@ def from_dict(cls, data: RectilinearChunkGridMetadataJSON) -> Self: # type: ign raw_shapes = configuration["chunk_shapes"] parsed: list[int | tuple[int, ...]] = [] for dim_spec in raw_shapes: - if isinstance(dim_spec, int): + if declares_chunk_edges(dim_spec): + parsed.append(tuple(expand_rle(dim_spec))) + elif isinstance(dim_spec, int | np.integer): if dim_spec < 1: raise ValueError(f"Integer chunk edge length must be >= 1, got {dim_spec}") - parsed.append(dim_spec) - elif isinstance(dim_spec, list): - parsed.append(tuple(expand_rle(dim_spec))) + parsed.append(int(dim_spec)) else: raise TypeError( - f"Invalid chunk_shapes entry: expected int or list, got {type(dim_spec)}" + "Invalid chunk_shapes entry: expected an integer or a sequence of " + f"chunk edge lengths, got {type(dim_spec)}" ) return cls(chunk_shapes=tuple(parsed)) @@ -444,8 +446,10 @@ def parse_chunk_grid( name, configuration = parse_named_configuration(data) if name == "regular": chunk_shape = configuration.get("chunk_shape") - if isinstance(chunk_shape, list | tuple) and any( - isinstance(dim_spec, list | tuple) for dim_spec in chunk_shape + # The outer call asks whether chunk_shape is an iterable at all, so a + # malformed scalar falls through to the regular parser's own error. + if declares_chunk_edges(chunk_shape) and any( + declares_chunk_edges(dim_spec) for dim_spec in chunk_shape ): return _parse_mixed_regular_chunk_grid(chunk_shape) return RegularChunkGridMetadata.from_dict(data) # type: ignore[arg-type] @@ -455,7 +459,7 @@ def parse_chunk_grid( def _parse_mixed_regular_chunk_grid( - chunk_shape: Sequence[JSON], + chunk_shape: Iterable[Any], ) -> RectilinearChunkGridMetadata: """Read a "regular" chunk grid whose chunk_shape contains edge lists. @@ -465,8 +469,12 @@ def _parse_mixed_regular_chunk_grid( invalid, but the data is intact, so it is read as the rectilinear grid it describes. See https://github.com/zarr-developers/zarr-python/issues/4374. """ - # Tuples arrive when the metadata dict was built in Python rather than parsed from JSON. - chunk_shapes = [list(d) if isinstance(d, tuple) else d for d in chunk_shape] + # Put the dimensions in the JSON forms the rectilinear parser reads: a + # metadata dict built in Python holds a tuple where JSON holds a list. + # Elements are left alone, so `from_dict` still reports a bad one. + chunk_shapes: list[Any] = [ + list(dim_spec) if declares_chunk_edges(dim_spec) else dim_spec for dim_spec in chunk_shape + ] msg = ( f"This array's chunk grid is named 'regular' but its chunk_shape {chunk_shapes!r} " "lists explicit chunk edges for some dimensions. zarr 3.2.0 and 3.2.1 wrote " @@ -486,7 +494,7 @@ def _parse_mixed_regular_chunk_grid( return RectilinearChunkGridMetadata.from_dict( { "name": "rectilinear", - "configuration": {"kind": "inline", "chunk_shapes": chunk_shapes}, # type: ignore[typeddict-item] + "configuration": {"kind": "inline", "chunk_shapes": chunk_shapes}, } ) diff --git a/tests/test_metadata/test_v3.py b/tests/test_metadata/test_v3.py index f945baf066..60677bce65 100644 --- a/tests/test_metadata/test_v3.py +++ b/tests/test_metadata/test_v3.py @@ -175,6 +175,29 @@ def test_regular_chunk_grid_normalizes_ints(case: Expect[tuple[Any, ...], tuple[ assert all(type(c) is int for c in grid.chunk_shape) +@pytest.mark.parametrize( + "case", + [ + Expect(input=(2, (5, 10, 5)), output=(2, (5, 10, 5)), id="python_values"), + Expect( + input=(np.int64(2), np.array([5, 10, 5])), output=(2, (5, 10, 5)), id="numpy_values" + ), + Expect(input=(2, [5, 10, 5]), output=(2, (5, 10, 5)), id="list_edges"), + ], + ids=lambda case: case.id, +) +def test_rectilinear_chunk_grid_normalizes_ints( + case: Expect[tuple[Any, ...], tuple[int | tuple[int, ...], ...]], +) -> None: + """A rectilinear chunk grid accepts any sequence of Python or numpy integers + as a dimension's edges and stores Python ints in tuples.""" + with config.set({"array.rectilinear_chunks": True}): + grid = RectilinearChunkGridMetadata(chunk_shapes=case.input) + assert grid.chunk_shapes == case.output + flat = [e for dim in grid.chunk_shapes for e in (dim if isinstance(dim, tuple) else (dim,))] + assert all(type(e) is int for e in flat) + + def test_regular_chunk_grid_rejects_edge_lists() -> None: """A regular chunk grid only accepts integer chunk edge lengths.""" with pytest.raises(TypeError, match="Dimension 1: a regular chunk grid requires an integer"): @@ -218,6 +241,11 @@ def test_regular_chunk_grid_rejects_edge_lists() -> None: output=(2, (5, 10, 5)), id="tuples_from_python", ), + Expect( + input=((6, 20), [2, np.array([5, 10, 5])]), + output=(2, (5, 10, 5)), + id="numpy_array_edges", + ), ], ids=lambda case: case.id, ) diff --git a/tests/test_unified_chunk_grid.py b/tests/test_unified_chunk_grid.py index b8289d2135..1aff4492c1 100644 --- a/tests/test_unified_chunk_grid.py +++ b/tests/test_unified_chunk_grid.py @@ -489,6 +489,8 @@ def test_chunk_grid_iter() -> None: [ ([[10, 3]], [10, 10, 10]), ([[10, 2], [20, 1]], [10, 10, 20]), + pytest.param([(10, 3)], [10, 10, 10], id="tuple-pair"), + pytest.param([np.int64(10), np.array([20, 2])], [10, 20, 20], id="numpy-values"), ], ) def test_rle_expand(compressed: list[Any], expected: list[int]) -> None: @@ -526,6 +528,8 @@ def test_rle_roundtrip() -> None: ([[-10, 2]], "Chunk edge length must be >= 1"), ([[5, 0]], "RLE repeat count must be >= 1"), ([[5, -1]], "RLE repeat count must be >= 1"), + ([(5, 2, 1)], "RLE entries must be an integer or"), + (["5"], "RLE entries must be an integer or"), ], ids=[ "zero-edge", @@ -534,6 +538,8 @@ def test_rle_roundtrip() -> None: "negative-rle-size", "zero-rle-count", "negative-rle-count", + "rle-pair-wrong-length", + "string-entry", ], ) def test_rle_expand_rejects_invalid(rle_input: list[Any], match: str) -> None: @@ -3002,15 +3008,57 @@ def test_iter_chunk_regions_rectilinear() -> None: }, (4, (10, 20)), ), + # A metadata dict built in Python may hold tuples and numpy values + # where parsed JSON holds lists and ints. + pytest.param( + { + "name": "rectilinear", + "configuration": {"kind": "inline", "chunk_shapes": (4, (10, 20))}, + }, + (4, (10, 20)), + id="tuple-dims", + ), + pytest.param( + { + "name": "rectilinear", + "configuration": {"kind": "inline", "chunk_shapes": [((4, 3),), [10, 20]]}, + }, + ((4, 4, 4), (10, 20)), + id="tuple-rle-pair", + ), + pytest.param( + { + "name": "rectilinear", + "configuration": { + "kind": "inline", + "chunk_shapes": [np.int64(4), np.array([10, 20]), [np.array([5, 2])]], + }, + }, + (4, (10, 20), (5, 5)), + id="numpy-values", + ), ], ) def test_rectilinear_from_dict( json_input: RectilinearChunkGridMetadataJSON, expected_chunk_shapes: tuple[int | tuple[int, ...], ...], ) -> None: - """RectilinearChunkGridMetadata.from_dict correctly parses all spec forms.""" + """RectilinearChunkGridMetadata.from_dict correctly parses all spec forms, + whatever sequence type holds a dimension's edges, and stores plain ints.""" grid = RectilinearChunkGridMetadata.from_dict(json_input) assert grid.chunk_shapes == expected_chunk_shapes + flat = [e for dim in grid.chunk_shapes for e in (dim if isinstance(dim, tuple) else (dim,))] + assert all(type(e) is int for e in flat) + + +@pytest.mark.parametrize("dim_spec", [4.5, None, "10"], ids=["float", "none", "string"]) +def test_rectilinear_from_dict_rejects_invalid_dim_spec(dim_spec: Any) -> None: + """A dimension that is neither an integer nor a sequence of edges is rejected. + A string is iterable but is not a sequence of edges.""" + with pytest.raises(TypeError, match="expected an integer or a sequence of chunk edge lengths"): + RectilinearChunkGridMetadata.from_dict( + {"name": "rectilinear", "configuration": {"kind": "inline", "chunk_shapes": [dim_spec]}} + ) @pytest.mark.parametrize( From 38cecdbc488192311deb6dc1ee1119097040a7ad Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Sat, 19 Sep 2026 17:12:19 +0200 Subject: [PATCH 10/76] docs: note the rectilinear parsing change in the changelog fragment Assisted-by: ClaudeCode:claude-opus-5 --- changes/4375.bugfix.md | 2 ++ 1 file changed, 2 insertions(+) diff --git a/changes/4375.bugfix.md b/changes/4375.bugfix.md index 0a76d24539..234c1af012 100644 --- a/changes/4375.bugfix.md +++ b/changes/4375.bugfix.md @@ -1 +1,3 @@ Arrays written by zarr 3.2.0 and 3.2.1 with a mixed regular/rectilinear chunk specification (e.g. `chunks=(2, (5, 10, 5))`) can be read again. Their stored "regular" chunk grid is read as the rectilinear grid it describes, with a warning explaining how to re-save the metadata. A regular chunk grid with per-dimension edge lists is now rejected when constructed. + +Rectilinear chunk grid metadata reads a dimension's chunk edges from any sequence of integers — a list parsed from JSON, a tuple from a metadata document built in Python, or a numpy array — instead of requiring a `list` of `int`, and stores them as Python ints. A dimension that is neither an integer nor a sequence of edge lengths is rejected with a message naming both forms. From 087b62d25d791ee953e5c0a5295fe10a18b4f02b Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Sat, 19 Sep 2026 17:25:55 +0200 Subject: [PATCH 11/76] refactor: validate each chunk grid kind once, in one place Parsing a regular chunk shape repeated its work at two levels: - `_parse_chunk_shape` type-checked and coerced every dimension, then handed the result to `_validate_chunk_shapes`, which re-ran the same isinstance test and coercion and added only the `>= 1` check. Its edge-list branch was unreachable from this caller, which is why a `cast` was needed on the way out. - `RegularChunkGridMetadata.from_dict` parsed the chunk shape and passed it to the constructor, whose `__post_init__` parsed it again. Together that was four passes over the dimensions for `from_dict`. It is now one: `_parse_chunk_shape` checks the range itself and no longer calls the rectilinear validator, and `from_dict` hands the dimensions to the constructor unparsed. Beyond the redundancy, the shared validator was the coupling that let a rectilinear chunk shape be stored as a regular grid (#4374), so the two grid kinds now validate separately. `RectilinearChunkGridMetadata.from_dict` also had its own `>= 1` check for bare-int dimensions, duplicating `_validate_chunk_shapes`, which `__post_init__` runs over the result anyway. It now only puts the JSON into shape (integer vs sequence, RLE expansion), and a bad bare int is reported by the validator, which names the dimension. `expand_rle` keeps its own checks because it is called directly. Assisted-by: ClaudeCode:claude-opus-5 --- src/zarr/core/metadata/v3.py | 26 +++++++++++++++----------- tests/test_metadata/test_v3.py | 25 ++++++++++++++++++++++++- tests/test_unified_chunk_grid.py | 16 ++++++++++++++++ 3 files changed, 55 insertions(+), 12 deletions(-) diff --git a/src/zarr/core/metadata/v3.py b/src/zarr/core/metadata/v3.py index 5bb2993486..2356953cbe 100644 --- a/src/zarr/core/metadata/v3.py +++ b/src/zarr/core/metadata/v3.py @@ -218,11 +218,13 @@ class RectilinearChunkGridMetadataConfig(TypedDict): def _parse_chunk_shape(chunk_shape: Iterable[int]) -> tuple[int, ...]: """Validate and normalize a regular chunk shape. - A regular chunk shape is a sequence of bare ints (one per dimension), - each of which must be >= 1. Per-dimension edge lists belong to a - rectilinear grid and are rejected here. + A regular chunk shape is one bare int per dimension, each >= 1. Lists of + chunk edge lengths belong to a rectilinear chunk grid and are rejected + here; `_validate_chunk_shapes` is the rectilinear counterpart. The two + grid kinds validate separately on purpose — sharing a validator is what + let a rectilinear chunk shape be stored as a regular grid (gh-4374). """ - normalized: list[int] = [] + parsed: list[int] = [] for dim_idx, dim_spec in enumerate(chunk_shape): if not isinstance(dim_spec, int | np.integer): raise TypeError( @@ -230,10 +232,10 @@ def _parse_chunk_shape(chunk_shape: Iterable[int]) -> tuple[int, ...]: f"edge length, got {dim_spec!r}. Lists of chunk edge lengths belong " "to a rectilinear chunk grid." ) - normalized.append(int(dim_spec)) - result = _validate_chunk_shapes(tuple(normalized)) - # Regular grids only have bare ints — cast is safe after validation - return cast(tuple[int, ...], result) + if dim_spec < 1: + raise ValueError(f"Dimension {dim_idx}: chunk size must be >= 1, got {dim_spec}") + parsed.append(int(dim_spec)) + return tuple(parsed) def _validate_chunk_shapes( @@ -294,7 +296,8 @@ def to_dict(self) -> RegularChunkGridMetadataJSON: # type: ignore[override] def from_dict(cls, data: RegularChunkGridMetadataJSON) -> Self: # type: ignore[override] parse_named_configuration(data, "regular") # validate name configuration = data["configuration"] - return cls(chunk_shape=_parse_chunk_shape(configuration["chunk_shape"])) + # `__post_init__` parses, so this only has to hand over the dimensions. + return cls(chunk_shape=tuple(configuration["chunk_shape"])) @dataclass(frozen=True, kw_only=True) @@ -384,8 +387,9 @@ def from_dict(cls, data: RectilinearChunkGridMetadataJSON) -> Self: # type: ign if declares_chunk_edges(dim_spec): parsed.append(tuple(expand_rle(dim_spec))) elif isinstance(dim_spec, int | np.integer): - if dim_spec < 1: - raise ValueError(f"Integer chunk edge length must be >= 1, got {dim_spec}") + # Range checks belong to `_validate_chunk_shapes`, which + # `__post_init__` runs over the result and which names the + # offending dimension. parsed.append(int(dim_spec)) else: raise TypeError( diff --git a/tests/test_metadata/test_v3.py b/tests/test_metadata/test_v3.py index 60677bce65..ef69b9e316 100644 --- a/tests/test_metadata/test_v3.py +++ b/tests/test_metadata/test_v3.py @@ -37,7 +37,7 @@ ) if TYPE_CHECKING: - from collections.abc import Sequence + from collections.abc import Callable, Sequence from typing import Any @@ -198,6 +198,29 @@ def test_rectilinear_chunk_grid_normalizes_ints( assert all(type(e) is int for e in flat) +@pytest.mark.parametrize( + "build", + [ + pytest.param(lambda shape: RegularChunkGridMetadata(chunk_shape=shape), id="constructor"), + pytest.param( + lambda shape: RegularChunkGridMetadata.from_dict( + {"name": "regular", "configuration": {"chunk_shape": list(shape)}} + ), + id="from_dict", + ), + ], +) +@pytest.mark.parametrize("chunk_shape", [(0, 2), (2, -1)], ids=["zero", "negative"]) +def test_regular_chunk_grid_rejects_nonpositive_chunk_size( + build: Callable[[tuple[int, ...]], RegularChunkGridMetadata], chunk_shape: tuple[int, ...] +) -> None: + """A regular chunk size below 1 is rejected, naming the dimension, whether + the grid is built directly or parsed from stored metadata.""" + dim = 0 if chunk_shape[0] < 1 else 1 + with pytest.raises(ValueError, match=f"Dimension {dim}: chunk size must be >= 1"): + build(chunk_shape) + + def test_regular_chunk_grid_rejects_edge_lists() -> None: """A regular chunk grid only accepts integer chunk edge lengths.""" with pytest.raises(TypeError, match="Dimension 1: a regular chunk grid requires an integer"): diff --git a/tests/test_unified_chunk_grid.py b/tests/test_unified_chunk_grid.py index 1aff4492c1..55247d02aa 100644 --- a/tests/test_unified_chunk_grid.py +++ b/tests/test_unified_chunk_grid.py @@ -624,6 +624,22 @@ def test_serialization_error_non_regular_chunk_shape() -> None: grid.chunk_shape # noqa: B018 +@pytest.mark.parametrize("chunk_shapes", [[0, [5, 5]], [[5, 5], -1]], ids=["zero", "negative"]) +def test_rectilinear_from_dict_rejects_nonpositive_bare_int(chunk_shapes: list[Any]) -> None: + """A bare-int dimension below 1 is rejected by the grid's own validator, which + names the dimension, rather than by a duplicate check in `from_dict`.""" + dim = 0 if isinstance(chunk_shapes[0], int) else 1 + with pytest.raises( + ValueError, match=f"Dimension {dim}: integer chunk edge length must be >= 1" + ): + RectilinearChunkGridMetadata.from_dict( + { + "name": "rectilinear", + "configuration": {"kind": "inline", "chunk_shapes": chunk_shapes}, + } + ) + + def test_serialization_error_zero_extent_rectilinear() -> None: """RectilinearChunkGridMetadata rejects empty edge tuples.""" with pytest.raises(ValueError, match="has no chunk edges"): From 49564dc4b59c111d7faecbbcc437eecb8554483c Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Sat, 19 Sep 2026 17:50:20 +0200 Subject: [PATCH 12/76] fix(metadata): read a legacy zero chunk size on an empty axis in Zarr format 3 MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The compatibility policy for a stored chunk size of 0 on a zero-length axis covered Zarr format 2 only, so arrays written by zarr-python 3.0 and 3.1 with `chunk_shape: [0]` — or `[false]`, which 3.0 wrote for `chunks=False` — still could not be opened at all. `ArrayV3Metadata` now applies the same policy to a regular chunk grid: a stored chunk size of 0 on a zero-length axis is read as 1 with a `ZarrUserWarning`, and a zero chunk size on a positive-length axis is left for the chunk grid parser to reject. It runs in `__init__` rather than in the grid parser because the policy needs the array shape, which chunk grid metadata does not carry. Both warnings now say how to store a corrected chunk size — open the array writable and call `array.update_attributes({})`, which rewrites the whole document from the parsed metadata — from one shared constant. The Zarr format 2 warning also named only zarr-python 2.x; measured against real installs, every 3.x release before 3.4 wrote a zero chunk size for an empty array too (3.3.0 for `chunks=-1` and `chunks=False`). Tested against stores written by zarr 3.0.10, 3.1.6 and 3.3.0: they open, append without losing data, and re-save to a chunk size that reopens without a warning. Assisted-by: ClaudeCode:claude-opus-5 --- changes/4334.bugfix.md | 4 ++- src/zarr/core/metadata/common.py | 12 ++++++- src/zarr/core/metadata/v2.py | 9 ++--- src/zarr/core/metadata/v3.py | 55 +++++++++++++++++++++++++++-- tests/test_chunk_grids.py | 59 ++++++++++++++++++++++++++++++++ tests/test_metadata/test_v3.py | 32 +++++++++++++++++ 6 files changed, 162 insertions(+), 9 deletions(-) diff --git a/changes/4334.bugfix.md b/changes/4334.bugfix.md index 5fc5373ee5..610413b42e 100644 --- a/changes/4334.bugfix.md +++ b/changes/4334.bugfix.md @@ -1 +1,3 @@ -Chunk-grid normalization now shares a helper that selects chunk size 1 for a zero-length axis when inferring a full-span chunk size. Chunk edge lengths remain positive while array extents may be zero. Zarr format 2 metadata with a stored chunk edge of 0 on a zero-length axis is interpreted as chunk size 1 with a `ZarrUserWarning`; a zero chunk edge on a positive-length axis is rejected. This supports legacy metadata such as the zero chunk sizes written by zarr-python 2.18.7 for empty arrays created with `chunks=False`, `chunks=-1`, or `chunks=(0,)`, without claiming that every historical reader behaved identically. Rectilinear chunk grids can be created on a zero-length dimension with a non-empty list of positive chunk sizes; these sizes are retained for subsequent growth. `FixedDimension(size=0, ...)` now raises `ValueError`. +Chunk-grid normalization now shares a helper that selects chunk size 1 for a zero-length axis when inferring a full-span chunk size. Chunk edge lengths remain positive while array extents may be zero. `FixedDimension(size=0, ...)` now raises `ValueError`. Rectilinear chunk grids can be created on a zero-length dimension with a non-empty list of positive chunk sizes; these sizes are retained for subsequent growth. + +A stored chunk size of 0 on a zero-length axis is read as 1 with a `ZarrUserWarning`, in both Zarr format 2 (`chunks`) and Zarr format 3 (a regular grid's `chunk_shape`, including the JSON `false` written by zarr-python 3.0 for `chunks=False`); a zero chunk size on a positive-length axis is rejected. zarr-python wrote such metadata for empty arrays until 3.4 — Zarr format 2 through 3.3 and by zarr-python 2.x, Zarr format 3 in 3.0 and 3.1 — and appending to one of these Zarr format 2 arrays previously reported success while writing no chunk, so the appended data read back as the fill value. The warning explains how to store a corrected chunk size: open the array writable and call `array.update_attributes({})`. diff --git a/src/zarr/core/metadata/common.py b/src/zarr/core/metadata/common.py index 6367bdb28a..83267bb16f 100644 --- a/src/zarr/core/metadata/common.py +++ b/src/zarr/core/metadata/common.py @@ -1,10 +1,20 @@ from __future__ import annotations -from typing import TYPE_CHECKING +from typing import TYPE_CHECKING, Final if TYPE_CHECKING: from zarr.core.common import JSON +RESAVE_METADATA_HINT: Final = ( + "Re-save the array metadata to store the corrected value: open the array " + "writable and call `array.update_attributes({})`." +) +"""How to persist metadata that was read under a compatibility policy. + +`update_attributes` rewrites the whole metadata document from the parsed +(corrected) metadata, so an empty update is enough. +""" + def parse_attributes(data: dict[str, JSON] | None) -> dict[str, JSON]: if data is None: diff --git a/src/zarr/core/metadata/v2.py b/src/zarr/core/metadata/v2.py index 6389146e7e..b30109ce47 100644 --- a/src/zarr/core/metadata/v2.py +++ b/src/zarr/core/metadata/v2.py @@ -42,7 +42,7 @@ ) from zarr.core.config import config, parse_indexing_order from zarr.core.json_parse import parse_field -from zarr.core.metadata.common import parse_attributes +from zarr.core.metadata.common import RESAVE_METADATA_HINT, parse_attributes class ArrayV2MetadataDict(TypedDict): @@ -90,8 +90,8 @@ def __init__( shape_parsed = parse_shapelike(shape) chunks_parsed = parse_shapelike(chunks) # Same invariant as the Zarr format 3 chunk grid metadata: every chunk edge - # length is at least 1. zarr-python 2.18.7 can write `chunks: [0]` for - # a zero-length axis created with `chunks=False`, `-1` or `(0,)`. + # length is at least 1. zarr-python 2.18.7, and 3.x before 3.4, can write + # `chunks: [0]` for a zero-length axis (e.g. `chunks=False`, `-1` or `(0,)`). # Normalize that empty axis to chunk size 1 so it can use the positive-size # grid model. This is a compatibility policy for legacy metadata, not a # statement about every historical reader. A zero chunk size on a @@ -106,7 +106,8 @@ def __init__( ) warnings.warn( f"Dimension {dim_idx}: chunk edge length 0 on a zero-length axis " - "(as written by zarr-python 2.x) is treated as 1.", + "(as written by zarr-python 2.x, and by 3.x before 3.4) is treated " + f"as 1. {RESAVE_METADATA_HINT}", ZarrUserWarning, stacklevel=2, ) diff --git a/src/zarr/core/metadata/v3.py b/src/zarr/core/metadata/v3.py index 11f3eb593d..0faf042858 100644 --- a/src/zarr/core/metadata/v3.py +++ b/src/zarr/core/metadata/v3.py @@ -1,10 +1,12 @@ from __future__ import annotations import json +import warnings from collections.abc import Iterable, Mapping, Sequence from dataclasses import dataclass, field, replace from typing import TYPE_CHECKING, Any, Final, Literal, NotRequired, TypeGuard, cast +import numpy as np from typing_extensions import TypedDict from zarr.abc.codec import ArrayArrayCodec, ArrayBytesCodec, BytesBytesCodec, Codec @@ -35,8 +37,8 @@ from zarr.core.dtype import VariableLengthUTF8, ZDType, get_data_type_from_json from zarr.core.dtype.common import check_dtype_spec_v3 from zarr.core.json_parse import parse_field -from zarr.core.metadata.common import parse_attributes -from zarr.errors import MetadataValidationError, NodeTypeValidationError +from zarr.core.metadata.common import RESAVE_METADATA_HINT, parse_attributes +from zarr.errors import MetadataValidationError, NodeTypeValidationError, ZarrUserWarning from zarr.registry import get_codec_class if TYPE_CHECKING: @@ -476,6 +478,51 @@ class ArrayMetadataJSON_V3(TypedDict, extra_items=AllowedExtraField): # type: i } +def _read_legacy_zero_chunk_sizes( + chunk_grid: dict[str, JSON] | ChunkGridMetadata | NamedConfig[str, Any], + shape: tuple[int, ...], +) -> dict[str, JSON] | ChunkGridMetadata | NamedConfig[str, Any]: + """Read a stored regular chunk size of 0 on a zero-length axis as 1. + + zarr-python 3.0 and 3.1 stored `chunk_shape: [0]` for an array created with + a zero-length axis and `chunks=(0,)` or `chunks=False` (3.0 stored `false` + for the latter). Chunk sizes must be at least 1, and a zero-length axis has + no chunks for the size to describe until the array grows, so the size is + read as 1 with a warning. This is the policy `ArrayV2Metadata` applies to + Zarr format 2 metadata. It needs the array shape, which the chunk grid + metadata does not carry, so it runs here rather than in the grid parser. + A zero chunk size on a positive-length axis is left for that parser to + reject. + """ + if not isinstance(chunk_grid, Mapping) or chunk_grid.get("name") != "regular": + return chunk_grid + configuration = chunk_grid.get("configuration") + if not isinstance(configuration, Mapping): + return chunk_grid + chunk_shape = configuration.get("chunk_shape") + if not isinstance(chunk_shape, Sequence) or isinstance(chunk_shape, str): + return chunk_grid + if len(chunk_shape) != len(shape): + return chunk_grid # the dimensionality check reports this + normalized = list(chunk_shape) + changed = False + for dim_idx, (size, extent) in enumerate(zip(chunk_shape, shape, strict=True)): + if isinstance(size, int | np.integer) and size == 0 and extent == 0: + warnings.warn( + f"Dimension {dim_idx}: chunk edge length {size!r} on a zero-length axis " + "(as written by zarr-python 3.0 and 3.1) is treated as 1. " + f"{RESAVE_METADATA_HINT}", + ZarrUserWarning, + stacklevel=2, + ) + normalized[dim_idx] = 1 + changed = True + if not changed: + return chunk_grid + corrected: dict[str, JSON] = {**configuration, "chunk_shape": normalized} + return {"name": "regular", "configuration": corrected} + + @dataclass(frozen=True, kw_only=True) class ArrayV3Metadata(Metadata): shape: tuple[int, ...] @@ -510,7 +557,9 @@ def __init__( """ shape_parsed = parse_shapelike(shape) - chunk_grid_parsed = parse_chunk_grid(chunk_grid) + chunk_grid_parsed = parse_chunk_grid( + _read_legacy_zero_chunk_sizes(chunk_grid, shape_parsed) + ) chunk_key_encoding_parsed = parse_chunk_key_encoding(chunk_key_encoding) dimension_names_parsed = parse_dimension_names(dimension_names) # Note: relying on a type method is numpy-specific diff --git a/tests/test_chunk_grids.py b/tests/test_chunk_grids.py index 86383d31ed..4ba5d19f26 100644 --- a/tests/test_chunk_grids.py +++ b/tests/test_chunk_grids.py @@ -1,4 +1,7 @@ import contextlib +import json +import warnings +from pathlib import Path from typing import Any, Literal, cast import numpy as np @@ -541,6 +544,62 @@ def test_rectilinear_zero_extent_matches_resize() -> None: assert created.write_chunk_sizes == resized.write_chunk_sizes == ((2, 1),) +def _store_legacy_zero_chunk(path: Any, zarr_format: Literal[2, 3], stored: Any) -> None: + """Rewrite an array's stored chunk size to *stored*, as older zarr-python did.""" + doc_name = ".zarray" if zarr_format == 2 else "zarr.json" + doc_path = path / doc_name + doc = json.loads(doc_path.read_text()) + if zarr_format == 2: + doc["chunks"] = [stored] + else: + doc["chunk_grid"]["configuration"]["chunk_shape"] = [stored] + doc_path.write_text(json.dumps(doc)) + + +@pytest.mark.parametrize("zarr_format", [2, 3]) +@pytest.mark.parametrize("stored", [0, False], ids=["zero", "false"]) +def test_legacy_zero_chunk_on_empty_axis_round_trip( + tmp_path: Path, zarr_format: Literal[2, 3], stored: Any +) -> None: + """An empty array whose stored chunk size is 0 opens, appends without losing + data, and re-saves as a valid chunk size. + + zarr-python wrote such metadata for an empty array until 3.4: Zarr format 2 + through 3.3, and Zarr format 3 in 3.0 and 3.1, which also wrote JSON `false` + for `chunks=False`. Appending to one of these arrays used to report success + while writing no chunk, so the data read back as the fill value. + """ + path = tmp_path / "legacy.zarr" + zarr.create_array(store=path, shape=(0,), chunks=(4,), dtype="int64", zarr_format=zarr_format) + _store_legacy_zero_chunk(path, zarr_format, stored) + + with pytest.warns(ZarrUserWarning, match="zero-length axis"): + arr = zarr.open_array(store=path, mode="a") + assert arr.chunks == (1,) + + arr.append(np.arange(3, dtype="int64")) + np.testing.assert_array_equal(zarr.open_array(store=path)[...], np.arange(3)) + + # The warning says to do this; it must leave metadata that reopens cleanly. + arr.update_attributes({}) + with warnings.catch_warnings(): + warnings.simplefilter("error", ZarrUserWarning) + reopened = zarr.open_array(store=path) + assert reopened.chunks == (1,) + + +@pytest.mark.parametrize("zarr_format", [2, 3]) +def test_stored_zero_chunk_on_positive_axis_rejected( + tmp_path: Path, zarr_format: Literal[2, 3] +) -> None: + """A stored chunk size of 0 is only tolerated on a zero-length axis.""" + path = tmp_path / "bad.zarr" + zarr.create_array(store=path, shape=(5,), chunks=(4,), dtype="int64", zarr_format=zarr_format) + _store_legacy_zero_chunk(path, zarr_format, 0) + with pytest.raises(ValueError, match="must be >= 1"): + zarr.open_array(store=path) + + def test_normalize_chunks_1d_zero_span_accepts_any_edges() -> None: """On a zero-length span the explicit edge list is stored verbatim.""" dim = normalize_chunks_1d([3, 5], span=0) diff --git a/tests/test_metadata/test_v3.py b/tests/test_metadata/test_v3.py index 9f78ed7b70..664fb6f092 100644 --- a/tests/test_metadata/test_v3.py +++ b/tests/test_metadata/test_v3.py @@ -463,3 +463,35 @@ def test_group_metadata_to_dict_consolidated(attributes: dict[str, Any] | None) }, }, } + + +@pytest.mark.parametrize( + ("shape", "chunk_shape", "expected"), + [ + ((0,), [0], (1,)), + ((4, 0), [4, 0], (4, 1)), + ((0, 0), [0, 0], (1, 1)), + ((0,), [False], (1,)), + ], + ids=["1d", "one-empty-axis", "all-empty-axes", "json-false"], +) +def test_zero_chunk_size_on_empty_axis_normalized( + shape: tuple[int, ...], chunk_shape: list[Any], expected: tuple[int, ...] +) -> None: + """A stored regular chunk size of 0 on a zero-length axis is read as 1, with a warning. + + zarr-python 3.0 and 3.1 wrote this for an array created with a zero-length + axis; 3.0 wrote JSON `false` for `chunks=False`. This is the Zarr format 3 + counterpart of the policy `ArrayV2Metadata` applies. + """ + from zarr.core.metadata.v3 import RegularChunkGridMetadata + from zarr.errors import ZarrUserWarning + + d = minimal_metadata_dict_v3( + shape=shape, + chunk_grid={"name": "regular", "configuration": {"chunk_shape": chunk_shape}}, + ) + with pytest.warns(ZarrUserWarning, match="zero-length axis"): + meta = ArrayV3Metadata.from_dict(d) # type: ignore[arg-type] + assert isinstance(meta.chunk_grid, RegularChunkGridMetadata) + assert meta.chunk_grid.chunk_shape == expected From 5bb1f97e1ab67093dd6539ca126ecc6b99e520d8 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Sat, 19 Sep 2026 18:05:39 +0200 Subject: [PATCH 13/76] refactor(metadata): check stored chunk shapes against the array shape in one routine The legacy zero-chunk policy was written out twice: inline in `ArrayV2Metadata.__init__`, and again in a Zarr format 3 helper. The V2 copy also zipped with `strict=False` and re-appended any trailing chunk entries only so that a separate length check, `parse_metadata`, could report a dimensionality mismatch after construction. `parse_stored_chunk_shape` in `zarr.core.metadata.common` is now the one place a stored chunk shape is checked against its array's shape, for both formats: one entry per axis, every integer chunk size at least 1, and a size of 0 (or JSON `false`) on a zero-length axis read as 1 with a warning that names the writer and how to re-save. Non-integer entries, such as edge lists, pass through for the caller's own parser. `ArrayV2Metadata.__init__` calls it directly and `parse_metadata` is gone. The Zarr format 3 adapter only locates a regular grid's `chunk_shape` in the stored document and hands it over; it still runs in `ArrayV3Metadata.__init__` because chunk grid metadata has no array shape. The Zarr format 3 import changes that existed only for the old helper are reverted. Tests for the policy now target the routine: one table of valid and legacy inputs, and one test per rejection (dimension mismatch, zero on a non-empty axis, negative). They replace metadata-level tests in test_v2.py and test_v3.py that only re-tested the same rules; the end-to-end tests still cover both formats' wiring against stored arrays. Assisted-by: ClaudeCode:claude-opus-5 --- src/zarr/core/metadata/common.py | 50 ++++++++++++++++++++- src/zarr/core/metadata/v2.py | 44 +++--------------- src/zarr/core/metadata/v3.py | 46 +++++-------------- tests/test_metadata/test_common.py | 72 ++++++++++++++++++++++++++++++ tests/test_metadata/test_v2.py | 29 ------------ tests/test_metadata/test_v3.py | 32 ------------- 6 files changed, 139 insertions(+), 134 deletions(-) create mode 100644 tests/test_metadata/test_common.py diff --git a/src/zarr/core/metadata/common.py b/src/zarr/core/metadata/common.py index 83267bb16f..92fb2b447e 100644 --- a/src/zarr/core/metadata/common.py +++ b/src/zarr/core/metadata/common.py @@ -1,8 +1,14 @@ from __future__ import annotations -from typing import TYPE_CHECKING, Final +import numbers +import warnings +from typing import TYPE_CHECKING, Any, Final + +from zarr.errors import ZarrUserWarning if TYPE_CHECKING: + from collections.abc import Sequence + from zarr.core.common import JSON RESAVE_METADATA_HINT: Final = ( @@ -21,3 +27,45 @@ def parse_attributes(data: dict[str, JSON] | None) -> dict[str, JSON]: return {} return dict(data) + + +def parse_stored_chunk_shape( + chunk_shape: Sequence[Any], shape: Sequence[int], *, legacy_writers: str +) -> tuple[Any, ...]: + """Validate a stored chunk shape against the array shape. + + This is the one place a stored per-axis chunk shape is checked against the + array it belongs to, for both Zarr formats. The chunk shape must have one + entry per array axis, and every integer chunk size must be at least 1. + + The exception is a chunk size of 0 on a zero-length axis, which + `legacy_writers` stored for arrays created empty. An empty axis has no + chunk for the size to describe, so it is read as 1 with a + `ZarrUserWarning` that says how to re-save valid metadata. A chunk size of + 0 on a positive-length axis is rejected, because the metadata cannot say + how the stored chunks were laid out. JSON `false` counts as 0. + + Entries that are not integers, such as explicit chunk edge lists, are + returned unchanged for the caller's own parser. + """ + if len(chunk_shape) != len(shape): + raise ValueError( + f"The chunk shape {tuple(chunk_shape)} and the array shape {tuple(shape)} " + "must have the same number of dimensions." + ) + parsed: list[Any] = [] + for dim_idx, (size, extent) in enumerate(zip(chunk_shape, shape, strict=True)): + if isinstance(size, numbers.Integral) and size < 1: + if size < 0 or extent != 0: + raise ValueError( + f"Dimension {dim_idx}: chunk edge length must be >= 1, got {size!r}" + ) + warnings.warn( + f"Dimension {dim_idx}: chunk edge length {size!r} on a zero-length axis " + f"(as written by {legacy_writers}) is treated as 1. {RESAVE_METADATA_HINT}", + ZarrUserWarning, + stacklevel=3, + ) + size = 1 + parsed.append(size) + return tuple(parsed) diff --git a/src/zarr/core/metadata/v2.py b/src/zarr/core/metadata/v2.py index b30109ce47..5513f1337b 100644 --- a/src/zarr/core/metadata/v2.py +++ b/src/zarr/core/metadata/v2.py @@ -42,7 +42,7 @@ ) from zarr.core.config import config, parse_indexing_order from zarr.core.json_parse import parse_field -from zarr.core.metadata.common import RESAVE_METADATA_HINT, parse_attributes +from zarr.core.metadata.common import parse_attributes, parse_stored_chunk_shape class ArrayV2MetadataDict(TypedDict): @@ -88,32 +88,11 @@ def __init__( Metadata for a Zarr format 2 array. """ shape_parsed = parse_shapelike(shape) - chunks_parsed = parse_shapelike(chunks) - # Same invariant as the Zarr format 3 chunk grid metadata: every chunk edge - # length is at least 1. zarr-python 2.18.7, and 3.x before 3.4, can write - # `chunks: [0]` for a zero-length axis (e.g. `chunks=False`, `-1` or `(0,)`). - # Normalize that empty axis to chunk size 1 so it can use the positive-size - # grid model. This is a compatibility policy for legacy metadata, not a - # statement about every historical reader. A zero chunk size on a - # positive-length axis is rejected; metadata alone cannot establish - # whether the store contains chunk payloads. - normalized_chunks: list[int] = [] - for dim_idx, (extent, chunk) in enumerate(zip(shape_parsed, chunks_parsed, strict=False)): - if chunk < 1: - if chunk < 0 or extent != 0: - raise ValueError( - f"Dimension {dim_idx}: chunk edge length must be >= 1, got {chunk}" - ) - warnings.warn( - f"Dimension {dim_idx}: chunk edge length 0 on a zero-length axis " - "(as written by zarr-python 2.x, and by 3.x before 3.4) is treated " - f"as 1. {RESAVE_METADATA_HINT}", - ZarrUserWarning, - stacklevel=2, - ) - chunk = 1 - normalized_chunks.append(chunk) - chunks_parsed = tuple(normalized_chunks) + chunks_parsed[len(shape_parsed) :] + chunks_parsed = parse_stored_chunk_shape( + parse_shapelike(chunks), + shape_parsed, + legacy_writers="zarr-python 2.x, and by 3.x before 3.4", + ) compressor_parsed = parse_compressor(compressor) order_parsed = parse_indexing_order(order) dimension_separator_parsed = parse_separator(dimension_separator) @@ -136,7 +115,6 @@ def __init__( object.__setattr__(self, "attributes", attributes_parsed) # ensure that the metadata document is consistent - _ = parse_metadata(self) @property def ndim(self) -> int: @@ -348,16 +326,6 @@ def parse_compressor(data: object) -> Numcodec | None: raise ValueError(msg) -def parse_metadata(data: ArrayV2Metadata) -> ArrayV2Metadata: - if (l_chunks := len(data.chunks)) != (l_shape := len(data.shape)): - msg = ( - f"The `shape` and `chunks` attributes must have the same length. " - f"`chunks` has length {l_chunks}, but `shape` has length {l_shape}." - ) - raise ValueError(msg) - return data - - def get_object_codec_id(maybe_object_codecs: Sequence[JSON]) -> str | None: """ Inspect a sequence of codecs / filters for an "object codec", i.e. a codec diff --git a/src/zarr/core/metadata/v3.py b/src/zarr/core/metadata/v3.py index 0faf042858..962564243a 100644 --- a/src/zarr/core/metadata/v3.py +++ b/src/zarr/core/metadata/v3.py @@ -1,12 +1,10 @@ from __future__ import annotations import json -import warnings from collections.abc import Iterable, Mapping, Sequence from dataclasses import dataclass, field, replace from typing import TYPE_CHECKING, Any, Final, Literal, NotRequired, TypeGuard, cast -import numpy as np from typing_extensions import TypedDict from zarr.abc.codec import ArrayArrayCodec, ArrayBytesCodec, BytesBytesCodec, Codec @@ -37,8 +35,8 @@ from zarr.core.dtype import VariableLengthUTF8, ZDType, get_data_type_from_json from zarr.core.dtype.common import check_dtype_spec_v3 from zarr.core.json_parse import parse_field -from zarr.core.metadata.common import RESAVE_METADATA_HINT, parse_attributes -from zarr.errors import MetadataValidationError, NodeTypeValidationError, ZarrUserWarning +from zarr.core.metadata.common import parse_attributes, parse_stored_chunk_shape +from zarr.errors import MetadataValidationError, NodeTypeValidationError from zarr.registry import get_codec_class if TYPE_CHECKING: @@ -482,17 +480,12 @@ def _read_legacy_zero_chunk_sizes( chunk_grid: dict[str, JSON] | ChunkGridMetadata | NamedConfig[str, Any], shape: tuple[int, ...], ) -> dict[str, JSON] | ChunkGridMetadata | NamedConfig[str, Any]: - """Read a stored regular chunk size of 0 on a zero-length axis as 1. - - zarr-python 3.0 and 3.1 stored `chunk_shape: [0]` for an array created with - a zero-length axis and `chunks=(0,)` or `chunks=False` (3.0 stored `false` - for the latter). Chunk sizes must be at least 1, and a zero-length axis has - no chunks for the size to describe until the array grows, so the size is - read as 1 with a warning. This is the policy `ArrayV2Metadata` applies to - Zarr format 2 metadata. It needs the array shape, which the chunk grid - metadata does not carry, so it runs here rather than in the grid parser. - A zero chunk size on a positive-length axis is left for that parser to - reject. + """Check a stored regular grid's chunk shape against the array shape. + + zarr-python 3.0 and 3.1 stored `chunk_shape: [0]` (and 3.0 `[false]`) for + an array created with a zero-length axis; `parse_stored_chunk_shape` + decides what that means. It needs the array shape, which chunk grid + metadata does not carry, so this runs here rather than in the grid parser. """ if not isinstance(chunk_grid, Mapping) or chunk_grid.get("name") != "regular": return chunk_grid @@ -502,25 +495,10 @@ def _read_legacy_zero_chunk_sizes( chunk_shape = configuration.get("chunk_shape") if not isinstance(chunk_shape, Sequence) or isinstance(chunk_shape, str): return chunk_grid - if len(chunk_shape) != len(shape): - return chunk_grid # the dimensionality check reports this - normalized = list(chunk_shape) - changed = False - for dim_idx, (size, extent) in enumerate(zip(chunk_shape, shape, strict=True)): - if isinstance(size, int | np.integer) and size == 0 and extent == 0: - warnings.warn( - f"Dimension {dim_idx}: chunk edge length {size!r} on a zero-length axis " - "(as written by zarr-python 3.0 and 3.1) is treated as 1. " - f"{RESAVE_METADATA_HINT}", - ZarrUserWarning, - stacklevel=2, - ) - normalized[dim_idx] = 1 - changed = True - if not changed: - return chunk_grid - corrected: dict[str, JSON] = {**configuration, "chunk_shape": normalized} - return {"name": "regular", "configuration": corrected} + parsed = parse_stored_chunk_shape(chunk_shape, shape, legacy_writers="zarr-python 3.0 and 3.1") + corrected: dict[str, Any] = dict(chunk_grid) + corrected["configuration"] = {**configuration, "chunk_shape": list(parsed)} + return corrected @dataclass(frozen=True, kw_only=True) diff --git a/tests/test_metadata/test_common.py b/tests/test_metadata/test_common.py new file mode 100644 index 0000000000..18e6a47e1c --- /dev/null +++ b/tests/test_metadata/test_common.py @@ -0,0 +1,72 @@ +"""Tests for metadata helpers shared by both Zarr formats.""" + +from __future__ import annotations + +import warnings +from typing import Any + +import numpy as np +import pytest + +from zarr.core.metadata.common import parse_stored_chunk_shape +from zarr.errors import ZarrUserWarning + + +@pytest.mark.parametrize( + ("chunk_shape", "shape", "expected", "warns"), + [ + ((4, 5), (10, 10), (4, 5), False), + ((1, 1), (0, 0), (1, 1), False), + ((0,), (0,), (1,), True), + ((False,), (0,), (1,), True), + ((np.int64(0),), (0,), (1,), True), + ((4, 0), (4, 0), (4, 1), True), + ((0, 0), (0, 0), (1, 1), True), + ((2, [5, 10, 5]), (6, 20), (2, [5, 10, 5]), False), + ], + ids=[ + "valid", + "valid-on-empty-axes", + "legacy-zero", + "legacy-json-false", + "legacy-numpy-zero", + "only-empty-axis-corrected", + "every-empty-axis-corrected", + "edge-list-passed-through", + ], +) +def test_parse_stored_chunk_shape( + chunk_shape: tuple[Any, ...], shape: tuple[int, ...], expected: tuple[Any, ...], warns: bool +) -> None: + """A valid chunk shape is returned as is; a chunk size of 0 on a zero-length + axis is read as 1, with a warning naming the writer and how to re-save.""" + with warnings.catch_warnings(record=True) as record: + warnings.simplefilter("always") + parsed = parse_stored_chunk_shape(chunk_shape, shape, legacy_writers="an old writer") + assert parsed == expected + messages = [str(w.message) for w in record if issubclass(w.category, ZarrUserWarning)] + assert bool(messages) is warns + for message in messages: + assert "an old writer" in message + assert "update_attributes({})" in message + + +def test_parse_stored_chunk_shape_rejects_dimension_mismatch() -> None: + """The chunk shape needs one entry per array axis.""" + with pytest.raises(ValueError, match="same number of dimensions"): + parse_stored_chunk_shape((4,), (10, 10), legacy_writers="an old writer") + + +@pytest.mark.parametrize(("chunk_shape", "shape"), [((0,), (5,)), ((4, 0), (4, 3))]) +def test_parse_stored_chunk_shape_rejects_zero_on_nonempty_axis( + chunk_shape: tuple[int, ...], shape: tuple[int, ...] +) -> None: + """A chunk size of 0 is only tolerated on a zero-length axis.""" + with pytest.raises(ValueError, match="chunk edge length must be >= 1, got 0"): + parse_stored_chunk_shape(chunk_shape, shape, legacy_writers="an old writer") + + +def test_parse_stored_chunk_shape_rejects_negative() -> None: + """A negative chunk size is rejected even on a zero-length axis.""" + with pytest.raises(ValueError, match="chunk edge length must be >= 1, got -1"): + parse_stored_chunk_shape((-1,), (0,), legacy_writers="an old writer") diff --git a/tests/test_metadata/test_v2.py b/tests/test_metadata/test_v2.py index 1e578c033c..1358f458d6 100644 --- a/tests/test_metadata/test_v2.py +++ b/tests/test_metadata/test_v2.py @@ -309,35 +309,6 @@ def test_from_dict_extra_fields() -> None: assert result == expected -@pytest.mark.parametrize( - ("shape", "chunks", "expected"), - [((0,), (0,), (1,)), ((4, 0), (4, 0), (4, 1)), ((0, 0), (0, 0), (1, 1))], -) -def test_zero_chunk_edge_on_empty_axis_normalized( - shape: tuple[int, ...], chunks: tuple[int, ...], expected: tuple[int, ...] -) -> None: - """A stored chunk edge of 0 on a zero-length axis is read as 1, with a warning. - - This checks the compatibility policy for legacy metadata: the in-memory - chunk size becomes positive while the extent remains zero. It does not - test historical reader behavior or perform array I/O. - """ - with pytest.warns(ZarrUserWarning, match="chunk edge length 0 on a zero-length axis"): - meta = ArrayV2Metadata( - shape=shape, dtype=Float64(), chunks=chunks, fill_value=0.0, order="C" - ) - assert meta.chunks == expected - - -@pytest.mark.parametrize(("shape", "chunks"), [((5,), (0,)), ((4, 3), (4, 0))]) -def test_zero_chunk_edge_with_positive_extent_rejected( - shape: tuple[int, ...], chunks: tuple[int, ...] -) -> None: - """A chunk edge of 0 on a positive-length axis is rejected.""" - with pytest.raises(ValueError, match="chunk edge length must be >= 1"): - ArrayV2Metadata(shape=shape, dtype=Float64(), chunks=chunks, fill_value=0.0, order="C") - - def test_eq_nan_fill_value() -> None: """Two metadata objects with an identical NaN fill_value compare equal. diff --git a/tests/test_metadata/test_v3.py b/tests/test_metadata/test_v3.py index 664fb6f092..9f78ed7b70 100644 --- a/tests/test_metadata/test_v3.py +++ b/tests/test_metadata/test_v3.py @@ -463,35 +463,3 @@ def test_group_metadata_to_dict_consolidated(attributes: dict[str, Any] | None) }, }, } - - -@pytest.mark.parametrize( - ("shape", "chunk_shape", "expected"), - [ - ((0,), [0], (1,)), - ((4, 0), [4, 0], (4, 1)), - ((0, 0), [0, 0], (1, 1)), - ((0,), [False], (1,)), - ], - ids=["1d", "one-empty-axis", "all-empty-axes", "json-false"], -) -def test_zero_chunk_size_on_empty_axis_normalized( - shape: tuple[int, ...], chunk_shape: list[Any], expected: tuple[int, ...] -) -> None: - """A stored regular chunk size of 0 on a zero-length axis is read as 1, with a warning. - - zarr-python 3.0 and 3.1 wrote this for an array created with a zero-length - axis; 3.0 wrote JSON `false` for `chunks=False`. This is the Zarr format 3 - counterpart of the policy `ArrayV2Metadata` applies. - """ - from zarr.core.metadata.v3 import RegularChunkGridMetadata - from zarr.errors import ZarrUserWarning - - d = minimal_metadata_dict_v3( - shape=shape, - chunk_grid={"name": "regular", "configuration": {"chunk_shape": chunk_shape}}, - ) - with pytest.warns(ZarrUserWarning, match="zero-length axis"): - meta = ArrayV3Metadata.from_dict(d) # type: ignore[arg-type] - assert isinstance(meta.chunk_grid, RegularChunkGridMetadata) - assert meta.chunk_grid.chunk_shape == expected From f93193de1eead0419c4e3e53f4ae3d93a064e45c Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Sat, 19 Sep 2026 18:17:02 +0200 Subject: [PATCH 14/76] refactor(metadata): scope the stored chunk shape check to regular chunk grids `parse_stored_chunk_shape` passed non-integer entries through "for the caller's own parser", which made a regular-grid policy look like a general chunk shape routine and let it decide what a 0-length chunk means for grids it does not own. A rectilinear grid, or any other grid, is free to define its own semantics for 0-length chunks. It is now `parse_stored_regular_chunk_shape`, typed `Sequence[int]`, with no pass-through, and its docstring says it applies to Zarr format 2 `chunks` and Zarr format 3 `regular` grids only. The Zarr format 3 caller hands it a chunk shape only when the grid is named `regular` and every entry is an integer (`_is_regular_chunk_shape`); anything else is not a regular chunk shape and goes to the chunk grid parser untouched. Assisted-by: ClaudeCode:claude-opus-5 --- src/zarr/core/metadata/common.py | 32 +++++++++++------------- src/zarr/core/metadata/v2.py | 4 +-- src/zarr/core/metadata/v3.py | 40 ++++++++++++++++++++++-------- tests/test_metadata/test_common.py | 22 ++++++++-------- 4 files changed, 57 insertions(+), 41 deletions(-) diff --git a/src/zarr/core/metadata/common.py b/src/zarr/core/metadata/common.py index 92fb2b447e..6e6bab48a5 100644 --- a/src/zarr/core/metadata/common.py +++ b/src/zarr/core/metadata/common.py @@ -1,8 +1,7 @@ from __future__ import annotations -import numbers import warnings -from typing import TYPE_CHECKING, Any, Final +from typing import TYPE_CHECKING, Final from zarr.errors import ZarrUserWarning @@ -29,33 +28,32 @@ def parse_attributes(data: dict[str, JSON] | None) -> dict[str, JSON]: return dict(data) -def parse_stored_chunk_shape( - chunk_shape: Sequence[Any], shape: Sequence[int], *, legacy_writers: str -) -> tuple[Any, ...]: - """Validate a stored chunk shape against the array shape. +def parse_stored_regular_chunk_shape( + chunk_shape: Sequence[int], shape: Sequence[int], *, legacy_writers: str +) -> tuple[int, ...]: + """Validate a stored regular chunk grid's chunk shape against the array shape. - This is the one place a stored per-axis chunk shape is checked against the - array it belongs to, for both Zarr formats. The chunk shape must have one - entry per array axis, and every integer chunk size must be at least 1. + This is for regular chunk grids only: Zarr format 2 `chunks`, and the + `chunk_shape` of a Zarr format 3 `regular` grid. Another chunk grid, such + as the rectilinear grid, is free to define its own meaning for a chunk of + length 0, so its chunk sizes must not be passed here. - The exception is a chunk size of 0 on a zero-length axis, which - `legacy_writers` stored for arrays created empty. An empty axis has no - chunk for the size to describe, so it is read as 1 with a + The chunk shape must have one entry per array axis, and every chunk size + must be at least 1. The exception is a chunk size of 0 on a zero-length + axis, which `legacy_writers` stored for arrays created empty. An empty axis + has no chunk for the size to describe, so it is read as 1 with a `ZarrUserWarning` that says how to re-save valid metadata. A chunk size of 0 on a positive-length axis is rejected, because the metadata cannot say how the stored chunks were laid out. JSON `false` counts as 0. - - Entries that are not integers, such as explicit chunk edge lists, are - returned unchanged for the caller's own parser. """ if len(chunk_shape) != len(shape): raise ValueError( f"The chunk shape {tuple(chunk_shape)} and the array shape {tuple(shape)} " "must have the same number of dimensions." ) - parsed: list[Any] = [] + parsed: list[int] = [] for dim_idx, (size, extent) in enumerate(zip(chunk_shape, shape, strict=True)): - if isinstance(size, numbers.Integral) and size < 1: + if size < 1: if size < 0 or extent != 0: raise ValueError( f"Dimension {dim_idx}: chunk edge length must be >= 1, got {size!r}" diff --git a/src/zarr/core/metadata/v2.py b/src/zarr/core/metadata/v2.py index 5513f1337b..c213bc98b6 100644 --- a/src/zarr/core/metadata/v2.py +++ b/src/zarr/core/metadata/v2.py @@ -42,7 +42,7 @@ ) from zarr.core.config import config, parse_indexing_order from zarr.core.json_parse import parse_field -from zarr.core.metadata.common import parse_attributes, parse_stored_chunk_shape +from zarr.core.metadata.common import parse_attributes, parse_stored_regular_chunk_shape class ArrayV2MetadataDict(TypedDict): @@ -88,7 +88,7 @@ def __init__( Metadata for a Zarr format 2 array. """ shape_parsed = parse_shapelike(shape) - chunks_parsed = parse_stored_chunk_shape( + chunks_parsed = parse_stored_regular_chunk_shape( parse_shapelike(chunks), shape_parsed, legacy_writers="zarr-python 2.x, and by 3.x before 3.4", diff --git a/src/zarr/core/metadata/v3.py b/src/zarr/core/metadata/v3.py index 962564243a..c5a70e2be4 100644 --- a/src/zarr/core/metadata/v3.py +++ b/src/zarr/core/metadata/v3.py @@ -5,6 +5,7 @@ from dataclasses import dataclass, field, replace from typing import TYPE_CHECKING, Any, Final, Literal, NotRequired, TypeGuard, cast +import numpy as np from typing_extensions import TypedDict from zarr.abc.codec import ArrayArrayCodec, ArrayBytesCodec, BytesBytesCodec, Codec @@ -35,7 +36,7 @@ from zarr.core.dtype import VariableLengthUTF8, ZDType, get_data_type_from_json from zarr.core.dtype.common import check_dtype_spec_v3 from zarr.core.json_parse import parse_field -from zarr.core.metadata.common import parse_attributes, parse_stored_chunk_shape +from zarr.core.metadata.common import parse_attributes, parse_stored_regular_chunk_shape from zarr.errors import MetadataValidationError, NodeTypeValidationError from zarr.registry import get_codec_class @@ -476,16 +477,31 @@ class ArrayMetadataJSON_V3(TypedDict, extra_items=AllowedExtraField): # type: i } -def _read_legacy_zero_chunk_sizes( +def _is_regular_chunk_shape(value: object) -> TypeGuard[Sequence[int]]: + """Whether a stored `chunk_shape` is a regular chunk shape: a sequence of integers. + + JSON `false` counts, because `bool` is an integer type. + """ + return ( + isinstance(value, Sequence) + and not isinstance(value, str) + and all(isinstance(size, int | np.integer) for size in value) + ) + + +def _parse_stored_regular_chunk_grid( chunk_grid: dict[str, JSON] | ChunkGridMetadata | NamedConfig[str, Any], shape: tuple[int, ...], ) -> dict[str, JSON] | ChunkGridMetadata | NamedConfig[str, Any]: - """Check a stored regular grid's chunk shape against the array shape. - - zarr-python 3.0 and 3.1 stored `chunk_shape: [0]` (and 3.0 `[false]`) for - an array created with a zero-length axis; `parse_stored_chunk_shape` - decides what that means. It needs the array shape, which chunk grid - metadata does not carry, so this runs here rather than in the grid parser. + """Check a stored regular chunk grid's chunk shape against the array shape. + + Only a `regular` grid whose `chunk_shape` is all integers is a regular chunk + shape, and only that is handed to `parse_stored_regular_chunk_shape`. + Anything else is not a regular chunk shape and is left for the chunk grid + parser: other grids define their own chunk semantics. zarr-python 3.0 and + 3.1 stored `chunk_shape: [0]` (and 3.0 `[false]`) for an array created with + a zero-length axis. This runs here rather than in the grid parser because + it needs the array shape, which chunk grid metadata does not carry. """ if not isinstance(chunk_grid, Mapping) or chunk_grid.get("name") != "regular": return chunk_grid @@ -493,9 +509,11 @@ def _read_legacy_zero_chunk_sizes( if not isinstance(configuration, Mapping): return chunk_grid chunk_shape = configuration.get("chunk_shape") - if not isinstance(chunk_shape, Sequence) or isinstance(chunk_shape, str): + if not _is_regular_chunk_shape(chunk_shape): return chunk_grid - parsed = parse_stored_chunk_shape(chunk_shape, shape, legacy_writers="zarr-python 3.0 and 3.1") + parsed = parse_stored_regular_chunk_shape( + chunk_shape, shape, legacy_writers="zarr-python 3.0 and 3.1" + ) corrected: dict[str, Any] = dict(chunk_grid) corrected["configuration"] = {**configuration, "chunk_shape": list(parsed)} return corrected @@ -536,7 +554,7 @@ def __init__( shape_parsed = parse_shapelike(shape) chunk_grid_parsed = parse_chunk_grid( - _read_legacy_zero_chunk_sizes(chunk_grid, shape_parsed) + _parse_stored_regular_chunk_grid(chunk_grid, shape_parsed) ) chunk_key_encoding_parsed = parse_chunk_key_encoding(chunk_key_encoding) dimension_names_parsed = parse_dimension_names(dimension_names) diff --git a/tests/test_metadata/test_common.py b/tests/test_metadata/test_common.py index 18e6a47e1c..3ebaaac9ab 100644 --- a/tests/test_metadata/test_common.py +++ b/tests/test_metadata/test_common.py @@ -8,7 +8,7 @@ import numpy as np import pytest -from zarr.core.metadata.common import parse_stored_chunk_shape +from zarr.core.metadata.common import parse_stored_regular_chunk_shape from zarr.errors import ZarrUserWarning @@ -22,7 +22,6 @@ ((np.int64(0),), (0,), (1,), True), ((4, 0), (4, 0), (4, 1), True), ((0, 0), (0, 0), (1, 1), True), - ((2, [5, 10, 5]), (6, 20), (2, [5, 10, 5]), False), ], ids=[ "valid", @@ -32,17 +31,18 @@ "legacy-numpy-zero", "only-empty-axis-corrected", "every-empty-axis-corrected", - "edge-list-passed-through", ], ) -def test_parse_stored_chunk_shape( +def test_parse_stored_regular_chunk_shape( chunk_shape: tuple[Any, ...], shape: tuple[int, ...], expected: tuple[Any, ...], warns: bool ) -> None: """A valid chunk shape is returned as is; a chunk size of 0 on a zero-length axis is read as 1, with a warning naming the writer and how to re-save.""" with warnings.catch_warnings(record=True) as record: warnings.simplefilter("always") - parsed = parse_stored_chunk_shape(chunk_shape, shape, legacy_writers="an old writer") + parsed = parse_stored_regular_chunk_shape( + chunk_shape, shape, legacy_writers="an old writer" + ) assert parsed == expected messages = [str(w.message) for w in record if issubclass(w.category, ZarrUserWarning)] assert bool(messages) is warns @@ -51,22 +51,22 @@ def test_parse_stored_chunk_shape( assert "update_attributes({})" in message -def test_parse_stored_chunk_shape_rejects_dimension_mismatch() -> None: +def test_parse_stored_regular_chunk_shape_rejects_dimension_mismatch() -> None: """The chunk shape needs one entry per array axis.""" with pytest.raises(ValueError, match="same number of dimensions"): - parse_stored_chunk_shape((4,), (10, 10), legacy_writers="an old writer") + parse_stored_regular_chunk_shape((4,), (10, 10), legacy_writers="an old writer") @pytest.mark.parametrize(("chunk_shape", "shape"), [((0,), (5,)), ((4, 0), (4, 3))]) -def test_parse_stored_chunk_shape_rejects_zero_on_nonempty_axis( +def test_parse_stored_regular_chunk_shape_rejects_zero_on_nonempty_axis( chunk_shape: tuple[int, ...], shape: tuple[int, ...] ) -> None: """A chunk size of 0 is only tolerated on a zero-length axis.""" with pytest.raises(ValueError, match="chunk edge length must be >= 1, got 0"): - parse_stored_chunk_shape(chunk_shape, shape, legacy_writers="an old writer") + parse_stored_regular_chunk_shape(chunk_shape, shape, legacy_writers="an old writer") -def test_parse_stored_chunk_shape_rejects_negative() -> None: +def test_parse_stored_regular_chunk_shape_rejects_negative() -> None: """A negative chunk size is rejected even on a zero-length axis.""" with pytest.raises(ValueError, match="chunk edge length must be >= 1, got -1"): - parse_stored_chunk_shape((-1,), (0,), legacy_writers="an old writer") + parse_stored_regular_chunk_shape((-1,), (0,), legacy_writers="an old writer") From e324baa7f61aefc5d0222ad318573b93938289f0 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Fri, 25 Sep 2026 12:04:40 +0200 Subject: [PATCH 15/76] refactor(metadata): read both legacy regular chunk grid forms in one place #4334 read a zero chunk size on an empty axis in `ArrayV3Metadata.__init__` and #4375 read a 3.2.x mixed grid inside `parse_chunk_grid`, each with its own predicate, warning text and re-save advice. Both are compatibility readings of a stored `regular` grid, so `_read_stored_regular_chunk_grid` now dispatches to both, `parse_chunk_grid` accepts only what the spec allows, and both warnings use `RESAVE_METADATA_HINT`. Assisted-by: ClaudeCode:claude-opus-5-5 Co-Authored-By: Claude Opus 5.5 --- src/zarr/core/metadata/v2.py | 4 +- src/zarr/core/metadata/v3.py | 73 +++++++++++++++++------------------- 2 files changed, 36 insertions(+), 41 deletions(-) diff --git a/src/zarr/core/metadata/v2.py b/src/zarr/core/metadata/v2.py index c213bc98b6..8655ab7f82 100644 --- a/src/zarr/core/metadata/v2.py +++ b/src/zarr/core/metadata/v2.py @@ -91,7 +91,7 @@ def __init__( chunks_parsed = parse_stored_regular_chunk_shape( parse_shapelike(chunks), shape_parsed, - legacy_writers="zarr-python 2.x, and by 3.x before 3.4", + legacy_writers="zarr 2.x, and by 3.x before 3.4", ) compressor_parsed = parse_compressor(compressor) order_parsed = parse_indexing_order(order) @@ -114,8 +114,6 @@ def __init__( object.__setattr__(self, "fill_value", fill_value_parsed) object.__setattr__(self, "attributes", attributes_parsed) - # ensure that the metadata document is consistent - @property def ndim(self) -> int: return len(self.shape) diff --git a/src/zarr/core/metadata/v3.py b/src/zarr/core/metadata/v3.py index fa3778a915..1aa280d558 100644 --- a/src/zarr/core/metadata/v3.py +++ b/src/zarr/core/metadata/v3.py @@ -38,7 +38,11 @@ from zarr.core.dtype import VariableLengthUTF8, ZDType, get_data_type_from_json from zarr.core.dtype.common import check_dtype_spec_v3 from zarr.core.json_parse import parse_field -from zarr.core.metadata.common import parse_attributes, parse_stored_regular_chunk_shape +from zarr.core.metadata.common import ( + RESAVE_METADATA_HINT, + parse_attributes, + parse_stored_regular_chunk_shape, +) from zarr.errors import MetadataValidationError, NodeTypeValidationError, ZarrUserWarning from zarr.registry import get_codec_class @@ -447,15 +451,8 @@ def parse_chunk_grid( if isinstance(data, (RegularChunkGridMetadata, RectilinearChunkGridMetadata)): return data - name, configuration = parse_named_configuration(data) + name, _ = parse_named_configuration(data) if name == "regular": - chunk_shape = configuration.get("chunk_shape") - # The outer call asks whether chunk_shape is an iterable at all, so a - # malformed scalar falls through to the regular parser's own error. - if declares_chunk_edges(chunk_shape) and any( - declares_chunk_edges(dim_spec) for dim_spec in chunk_shape - ): - return _parse_mixed_regular_chunk_grid(chunk_shape) return RegularChunkGridMetadata.from_dict(data) # type: ignore[arg-type] if name == "rectilinear": return RectilinearChunkGridMetadata.from_dict(data) # type: ignore[arg-type] @@ -490,8 +487,7 @@ def _parse_mixed_regular_chunk_grid( "zarr.config.set({'array.rectilinear_chunks': True})" ) warnings.warn( - msg + "Reading it as a rectilinear chunk grid. Re-save the array metadata " - "(e.g. with `array.update_attributes({})`) to store a valid rectilinear chunk grid.", + msg + f"Reading it as a rectilinear chunk grid. {RESAVE_METADATA_HINT}", ZarrUserWarning, stacklevel=2, ) @@ -542,31 +538,27 @@ class ArrayMetadataJSON_V3(TypedDict, extra_items=AllowedExtraField): # type: i } -def _is_regular_chunk_shape(value: object) -> TypeGuard[Sequence[int]]: - """Whether a stored `chunk_shape` is a regular chunk shape: a sequence of integers. - - JSON `false` counts, because `bool` is an integer type. - """ - return ( - isinstance(value, Sequence) - and not isinstance(value, str) - and all(isinstance(size, int | np.integer) for size in value) - ) - - -def _parse_stored_regular_chunk_grid( +def _read_stored_regular_chunk_grid( chunk_grid: dict[str, JSON] | ChunkGridMetadata | NamedConfig[str, Any], shape: tuple[int, ...], ) -> dict[str, JSON] | ChunkGridMetadata | NamedConfig[str, Any]: - """Check a stored regular chunk grid's chunk shape against the array shape. - - Only a `regular` grid whose `chunk_shape` is all integers is a regular chunk - shape, and only that is handed to `parse_stored_regular_chunk_shape`. - Anything else is not a regular chunk shape and is left for the chunk grid - parser: other grids define their own chunk semantics. zarr-python 3.0 and - 3.1 stored `chunk_shape: [0]` (and 3.0 `[false]`) for an array created with - a zero-length axis. This runs here rather than in the grid parser because - it needs the array shape, which chunk grid metadata does not carry. + """Read a stored `regular` chunk grid, including the invalid forms earlier releases wrote. + + This is the one place a stored regular grid is checked against the array + shape and read under a compatibility policy; `parse_chunk_grid` accepts + only what the spec allows. It runs here because both policies need to see + the whole grid, and one of them needs the array shape, which chunk grid + metadata does not carry. Two invalid forms are read, each with a warning: + + - A `chunk_shape` that lists chunk edges for some dimensions, which zarr + 3.2.0 and 3.2.1 wrote for mixed specifications such as `(2, (5, 10, 5))`, + is read as the rectilinear grid it describes. + - An all-integer `chunk_shape` is handed to + `parse_stored_regular_chunk_shape`, which reads a chunk size of 0 (or + JSON `false`) on a zero-length axis as 1. zarr 3.0 and 3.1 wrote these. + + Any other grid is returned unchanged for `parse_chunk_grid`; other grids + define their own chunk semantics. """ if not isinstance(chunk_grid, Mapping) or chunk_grid.get("name") != "regular": return chunk_grid @@ -574,11 +566,16 @@ def _parse_stored_regular_chunk_grid( if not isinstance(configuration, Mapping): return chunk_grid chunk_shape = configuration.get("chunk_shape") - if not _is_regular_chunk_shape(chunk_shape): + # The outer check asks whether chunk_shape is an iterable at all, so a + # malformed scalar falls through to the regular parser's own error. + if not declares_chunk_edges(chunk_shape): return chunk_grid - parsed = parse_stored_regular_chunk_shape( - chunk_shape, shape, legacy_writers="zarr-python 3.0 and 3.1" - ) + dims = list(chunk_shape) + if any(declares_chunk_edges(dim) for dim in dims): + return _parse_mixed_regular_chunk_grid(dims) + if not all(isinstance(dim, int | np.integer) for dim in dims): + return chunk_grid + parsed = parse_stored_regular_chunk_shape(dims, shape, legacy_writers="zarr 3.0 and 3.1") corrected: dict[str, Any] = dict(chunk_grid) corrected["configuration"] = {**configuration, "chunk_shape": list(parsed)} return corrected @@ -619,7 +616,7 @@ def __init__( shape_parsed = parse_shapelike(shape) chunk_grid_parsed = parse_chunk_grid( - _parse_stored_regular_chunk_grid(chunk_grid, shape_parsed) + _read_stored_regular_chunk_grid(chunk_grid, shape_parsed) ) chunk_key_encoding_parsed = parse_chunk_key_encoding(chunk_key_encoding) dimension_names_parsed = parse_dimension_names(dimension_names) From 8414ea3deda8d314ec85160a6203cb08f29192e9 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Fri, 25 Sep 2026 12:35:42 +0200 Subject: [PATCH 16/76] fix(metadata): read a stored zero chunk size on a grown axis as one spanning chunk A stored chunk size of 0 was tolerated only on a zero-length axis and rejected otherwise, because the metadata supposedly could not say how the stored chunks were laid out. But a chunk size of 0 gives a grid of zero chunks, so no release could store a chunk under it, and the writers of that metadata let the axis grow: 3.4.0 appends to a Zarr format 2 array created empty by 3.3.0 (shape grows, no chunk written), and 3.1.6 records a Zarr format 3 resize and the resize half of a failed append. Measured with real installs. Those arrays open in 3.4.0, attributes included; the rejection would have made them unopenable. `parse_stored_regular_chunk_shape` now reads a stored 0 (or JSON `false`) on any axis as one chunk spanning it, `max(extent, 1)`, which is what the `-1`/`False` spec that wrote it meant. On a grown axis the warning also says that data written to it was not saved. Negative sizes are still rejected. Assisted-by: ClaudeCode:claude-opus-5-5 Co-Authored-By: Claude Opus 5.5 --- changes/4334.bugfix.md | 2 +- src/zarr/core/metadata/common.py | 42 ++++++++++++++----------- src/zarr/core/metadata/v3.py | 2 +- tests/test_chunk_grids.py | 46 ++++++++++++---------------- tests/test_metadata/test_common.py | 49 ++++++++++++++++-------------- 5 files changed, 73 insertions(+), 68 deletions(-) diff --git a/changes/4334.bugfix.md b/changes/4334.bugfix.md index 610413b42e..68ae5ade9e 100644 --- a/changes/4334.bugfix.md +++ b/changes/4334.bugfix.md @@ -1,3 +1,3 @@ Chunk-grid normalization now shares a helper that selects chunk size 1 for a zero-length axis when inferring a full-span chunk size. Chunk edge lengths remain positive while array extents may be zero. `FixedDimension(size=0, ...)` now raises `ValueError`. Rectilinear chunk grids can be created on a zero-length dimension with a non-empty list of positive chunk sizes; these sizes are retained for subsequent growth. -A stored chunk size of 0 on a zero-length axis is read as 1 with a `ZarrUserWarning`, in both Zarr format 2 (`chunks`) and Zarr format 3 (a regular grid's `chunk_shape`, including the JSON `false` written by zarr-python 3.0 for `chunks=False`); a zero chunk size on a positive-length axis is rejected. zarr-python wrote such metadata for empty arrays until 3.4 — Zarr format 2 through 3.3 and by zarr-python 2.x, Zarr format 3 in 3.0 and 3.1 — and appending to one of these Zarr format 2 arrays previously reported success while writing no chunk, so the appended data read back as the fill value. The warning explains how to store a corrected chunk size: open the array writable and call `array.update_attributes({})`. +A stored chunk size of 0 is read as one chunk spanning the axis, with a `ZarrUserWarning`, in both Zarr format 2 (`chunks`) and Zarr format 3 (a regular grid's `chunk_shape`, including the JSON `false` written by zarr-python 3.0 for `chunks=False`). zarr-python wrote such metadata for arrays created with a zero-length axis until 3.4 — Zarr format 2 through 3.3 and by zarr-python 2.x, Zarr format 3 in 3.0 and 3.1. Those versions could then grow the axis without storing any chunk: appending to one of these Zarr format 2 arrays in 3.4.0 reported success while the appended data read back as the fill value, and 3.1 recorded a Zarr format 3 resize or failed append. Such grown arrays open too, and the warning says that data written to the grown axis was not saved. A negative chunk size is rejected. The warning explains how to store a corrected chunk size: open the array writable and call `array.update_attributes({})`. diff --git a/src/zarr/core/metadata/common.py b/src/zarr/core/metadata/common.py index 6e6bab48a5..14c6b8dd66 100644 --- a/src/zarr/core/metadata/common.py +++ b/src/zarr/core/metadata/common.py @@ -39,12 +39,15 @@ def parse_stored_regular_chunk_shape( length 0, so its chunk sizes must not be passed here. The chunk shape must have one entry per array axis, and every chunk size - must be at least 1. The exception is a chunk size of 0 on a zero-length - axis, which `legacy_writers` stored for arrays created empty. An empty axis - has no chunk for the size to describe, so it is read as 1 with a - `ZarrUserWarning` that says how to re-save valid metadata. A chunk size of - 0 on a positive-length axis is rejected, because the metadata cannot say - how the stored chunks were laid out. JSON `false` counts as 0. + must be at least 1, with one exception. `legacy_writers` stored a chunk + size of 0 (or JSON `false`) for an array created with a zero-length axis + and a chunk spec meaning "one chunk spanning the axis". That is how the + size is read, `max(extent, 1)`, with a `ZarrUserWarning` that says how to + re-save valid metadata. The axis may have grown since: those writers let + the array be resized or appended to, but a chunk size of 0 gives a grid of + zero chunks, so no chunk was ever stored for it and any chunk size reads + the store correctly. The warning then says that the appended data was not + saved. A negative chunk size is rejected. """ if len(chunk_shape) != len(shape): raise ValueError( @@ -53,17 +56,22 @@ def parse_stored_regular_chunk_shape( ) parsed: list[int] = [] for dim_idx, (size, extent) in enumerate(zip(chunk_shape, shape, strict=True)): - if size < 1: - if size < 0 or extent != 0: - raise ValueError( - f"Dimension {dim_idx}: chunk edge length must be >= 1, got {size!r}" - ) - warnings.warn( - f"Dimension {dim_idx}: chunk edge length {size!r} on a zero-length axis " - f"(as written by {legacy_writers}) is treated as 1. {RESAVE_METADATA_HINT}", - ZarrUserWarning, - stacklevel=3, + if size < 0: + raise ValueError(f"Dimension {dim_idx}: chunk edge length must be >= 1, got {size!r}") + if size == 0: + corrected = max(extent, 1) + msg = ( + f"Dimension {dim_idx}: chunk edge length {size!r} (as written by " + f"{legacy_writers} for an array created with a zero-length axis) is read " + f"as one chunk spanning the axis, of size {corrected}." ) - size = 1 + if extent > 0: + msg += ( + f" The axis has since grown to {extent}, but no chunk can be stored " + "under a chunk size of 0, so data written to it before now was not " + "saved and reads as the fill value." + ) + warnings.warn(f"{msg} {RESAVE_METADATA_HINT}", ZarrUserWarning, stacklevel=3) + size = corrected parsed.append(size) return tuple(parsed) diff --git a/src/zarr/core/metadata/v3.py b/src/zarr/core/metadata/v3.py index c5a70e2be4..ef7a5bf0e2 100644 --- a/src/zarr/core/metadata/v3.py +++ b/src/zarr/core/metadata/v3.py @@ -500,7 +500,7 @@ def _parse_stored_regular_chunk_grid( Anything else is not a regular chunk shape and is left for the chunk grid parser: other grids define their own chunk semantics. zarr-python 3.0 and 3.1 stored `chunk_shape: [0]` (and 3.0 `[false]`) for an array created with - a zero-length axis. This runs here rather than in the grid parser because + a zero-length axis, and 3.1 kept it when the axis grew. This runs here rather than in the grid parser because it needs the array shape, which chunk grid metadata does not carry. """ if not isinstance(chunk_grid, Mapping) or chunk_grid.get("name") != "regular": diff --git a/tests/test_chunk_grids.py b/tests/test_chunk_grids.py index 4ba5d19f26..cfd2e42258 100644 --- a/tests/test_chunk_grids.py +++ b/tests/test_chunk_grids.py @@ -558,46 +558,40 @@ def _store_legacy_zero_chunk(path: Any, zarr_format: Literal[2, 3], stored: Any) @pytest.mark.parametrize("zarr_format", [2, 3]) @pytest.mark.parametrize("stored", [0, False], ids=["zero", "false"]) -def test_legacy_zero_chunk_on_empty_axis_round_trip( - tmp_path: Path, zarr_format: Literal[2, 3], stored: Any +@pytest.mark.parametrize("extent", [0, 3], ids=["empty-axis", "grown-axis"]) +def test_legacy_zero_chunk_round_trip( + tmp_path: Path, zarr_format: Literal[2, 3], stored: Any, extent: int ) -> None: - """An empty array whose stored chunk size is 0 opens, appends without losing - data, and re-saves as a valid chunk size. - - zarr-python wrote such metadata for an empty array until 3.4: Zarr format 2 - through 3.3, and Zarr format 3 in 3.0 and 3.1, which also wrote JSON `false` - for `chunks=False`. Appending to one of these arrays used to report success - while writing no chunk, so the data read back as the fill value. + """An array whose stored chunk size is 0 opens with one chunk spanning the + axis, appends without losing data, and re-saves as a valid chunk size. + + zarr-python wrote such metadata for an array created with a zero-length axis + until 3.4: Zarr format 2 through 3.3, and Zarr format 3 in 3.0 and 3.1, which + also wrote JSON `false` for `chunks=False`. Those versions could then grow + the axis (a 3.4.0 append, a 3.1 resize) while storing no chunk, so the + grown axis holds only the fill value. """ path = tmp_path / "legacy.zarr" - zarr.create_array(store=path, shape=(0,), chunks=(4,), dtype="int64", zarr_format=zarr_format) + zarr.create_array( + store=path, shape=(extent,), chunks=(4,), dtype="int64", zarr_format=zarr_format + ) _store_legacy_zero_chunk(path, zarr_format, stored) - with pytest.warns(ZarrUserWarning, match="zero-length axis"): + with pytest.warns(ZarrUserWarning, match="one chunk spanning the axis"): arr = zarr.open_array(store=path, mode="a") - assert arr.chunks == (1,) + assert arr.chunks == (max(extent, 1),) + np.testing.assert_array_equal(arr[...], np.zeros(extent, dtype="int64")) arr.append(np.arange(3, dtype="int64")) - np.testing.assert_array_equal(zarr.open_array(store=path)[...], np.arange(3)) + expected = np.concatenate([np.zeros(extent, dtype="int64"), np.arange(3)]) + np.testing.assert_array_equal(zarr.open_array(store=path)[...], expected) # The warning says to do this; it must leave metadata that reopens cleanly. arr.update_attributes({}) with warnings.catch_warnings(): warnings.simplefilter("error", ZarrUserWarning) reopened = zarr.open_array(store=path) - assert reopened.chunks == (1,) - - -@pytest.mark.parametrize("zarr_format", [2, 3]) -def test_stored_zero_chunk_on_positive_axis_rejected( - tmp_path: Path, zarr_format: Literal[2, 3] -) -> None: - """A stored chunk size of 0 is only tolerated on a zero-length axis.""" - path = tmp_path / "bad.zarr" - zarr.create_array(store=path, shape=(5,), chunks=(4,), dtype="int64", zarr_format=zarr_format) - _store_legacy_zero_chunk(path, zarr_format, 0) - with pytest.raises(ValueError, match="must be >= 1"): - zarr.open_array(store=path) + assert reopened.chunks == (max(extent, 1),) def test_normalize_chunks_1d_zero_span_accepts_any_edges() -> None: diff --git a/tests/test_metadata/test_common.py b/tests/test_metadata/test_common.py index 3ebaaac9ab..f35f2fa954 100644 --- a/tests/test_metadata/test_common.py +++ b/tests/test_metadata/test_common.py @@ -2,6 +2,7 @@ from __future__ import annotations +import re import warnings from typing import Any @@ -13,15 +14,17 @@ @pytest.mark.parametrize( - ("chunk_shape", "shape", "expected", "warns"), + ("chunk_shape", "shape", "expected", "warning"), [ - ((4, 5), (10, 10), (4, 5), False), - ((1, 1), (0, 0), (1, 1), False), - ((0,), (0,), (1,), True), - ((False,), (0,), (1,), True), - ((np.int64(0),), (0,), (1,), True), - ((4, 0), (4, 0), (4, 1), True), - ((0, 0), (0, 0), (1, 1), True), + ((4, 5), (10, 10), (4, 5), None), + ((1, 1), (0, 0), (1, 1), None), + ((0,), (0,), (1,), "of size 1"), + ((False,), (0,), (1,), "of size 1"), + ((np.int64(0),), (0,), (1,), "of size 1"), + ((4, 0), (4, 0), (4, 1), "Dimension 1"), + ((0, 0), (0, 0), (1, 1), "Dimension 0"), + ((0,), (5,), (5,), "grown to 5.*was not saved"), + ((4, 0), (4, 3), (4, 3), "grown to 3.*was not saved"), ], ids=[ "valid", @@ -29,15 +32,21 @@ "legacy-zero", "legacy-json-false", "legacy-numpy-zero", - "only-empty-axis-corrected", - "every-empty-axis-corrected", + "only-zero-size-corrected", + "every-zero-size-corrected", + "legacy-zero-on-grown-axis", + "legacy-zero-on-grown-axis-2d", ], ) def test_parse_stored_regular_chunk_shape( - chunk_shape: tuple[Any, ...], shape: tuple[int, ...], expected: tuple[Any, ...], warns: bool + chunk_shape: tuple[Any, ...], + shape: tuple[int, ...], + expected: tuple[Any, ...], + warning: str | None, ) -> None: - """A valid chunk shape is returned as is; a chunk size of 0 on a zero-length - axis is read as 1, with a warning naming the writer and how to re-save.""" + """A valid chunk shape is returned as is; a chunk size of 0 is read as one + chunk spanning the axis, with a warning naming the writer and how to re-save, + and, if the axis has grown, that data written to it was not saved.""" with warnings.catch_warnings(record=True) as record: warnings.simplefilter("always") parsed = parse_stored_regular_chunk_shape( @@ -45,7 +54,10 @@ def test_parse_stored_regular_chunk_shape( ) assert parsed == expected messages = [str(w.message) for w in record if issubclass(w.category, ZarrUserWarning)] - assert bool(messages) is warns + if warning is None: + assert messages == [] + else: + assert any(re.search(warning, message) for message in messages) for message in messages: assert "an old writer" in message assert "update_attributes({})" in message @@ -57,15 +69,6 @@ def test_parse_stored_regular_chunk_shape_rejects_dimension_mismatch() -> None: parse_stored_regular_chunk_shape((4,), (10, 10), legacy_writers="an old writer") -@pytest.mark.parametrize(("chunk_shape", "shape"), [((0,), (5,)), ((4, 0), (4, 3))]) -def test_parse_stored_regular_chunk_shape_rejects_zero_on_nonempty_axis( - chunk_shape: tuple[int, ...], shape: tuple[int, ...] -) -> None: - """A chunk size of 0 is only tolerated on a zero-length axis.""" - with pytest.raises(ValueError, match="chunk edge length must be >= 1, got 0"): - parse_stored_regular_chunk_shape(chunk_shape, shape, legacy_writers="an old writer") - - def test_parse_stored_regular_chunk_shape_rejects_negative() -> None: """A negative chunk size is rejected even on a zero-length axis.""" with pytest.raises(ValueError, match="chunk edge length must be >= 1, got -1"): From 2bf32df2294270de30ab6f8c85506c0ed8d589ce Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Fri, 25 Sep 2026 12:43:38 +0200 Subject: [PATCH 17/76] test: a stateful test of one array's create/append/resize/write life The only state machine that touched arrays compared zarr on one store with zarr on a MemoryStore, so a chunk grid bug showed up identically on both sides; it had no append rule, covered Zarr format 3 only, and kept every empty axis at 0 when resizing, which is where the zero-length bugs live. `ArrayLifecycle` checks one array against a NumPy model across both formats, every chunk spelling (-1, False, "auto", ints, sharded, rectilinear) and the stored chunk size of 0 that releases before 3.4 wrote, including on an axis those releases grew. Rules append, resize (growing and shrinking to and from 0), write and re-save the metadata; the invariant reopens the array and compares shape, values and whether the legacy warning is due. Deliberately breaking the grown-axis policy, the legacy warning, or append on an empty axis each fails it. `resize` keeps partly retained chunks whole, so cells cut off by a shrink can come back with their old values when the axis grows (as in 2.x); the model marks such cells unknown until written instead of encoding that. Assisted-by: ClaudeCode:claude-opus-5-5 Co-Authored-By: Claude Opus 5.5 --- tests/test_array_stateful.py | 252 +++++++++++++++++++++++++++++++++++ 1 file changed, 252 insertions(+) create mode 100644 tests/test_array_stateful.py diff --git a/tests/test_array_stateful.py b/tests/test_array_stateful.py new file mode 100644 index 0000000000..75a6b12d59 --- /dev/null +++ b/tests/test_array_stateful.py @@ -0,0 +1,252 @@ +"""A stateful test of one array's life: create, append, resize, write, reopen. + +The model is a NumPy array, not a second zarr array, so a bug in zarr's chunk +grid logic cannot hide by being made on both sides. Zero-length axes are drawn +on purpose, both at creation and by resizing and appending, and so are the +stored chunk sizes of 0 that zarr-python wrote for empty arrays before 3.4. + +`resize` deletes only the chunks that fall entirely outside the new shape, so +cells cut off by a shrink can come back with their old values when the axis +grows again (as in zarr-python 2.x). The model does not encode that chunk-level +behaviour: a cell cut off and brought back is unknown until it is written. +""" + +from __future__ import annotations + +import json +import warnings +from typing import Any, Literal + +import hypothesis.extra.numpy as npst +import hypothesis.strategies as st +import numpy as np +import pytest +from hypothesis import event, note, settings +from hypothesis.stateful import ( + RuleBasedStateMachine, + initialize, + invariant, + rule, +) + +import zarr +from zarr.core.buffer import cpu, default_buffer_prototype +from zarr.core.sync import sync +from zarr.errors import ZarrUserWarning +from zarr.storage import MemoryStore + +pytestmark = pytest.mark.filterwarnings( + "ignore::zarr.core.dtype.common.UnstableSpecificationWarning" +) + +DTYPE = np.dtype("int16") +MAX_SIDE = 6 + + +def _rectilinear_dim(extent: int) -> st.SearchStrategy[int | list[int]]: + """A bare step, or an edge list covering `extent` (any edges for extent 0).""" + steps = st.integers(min_value=1, max_value=MAX_SIDE) + if extent == 0: + return steps | st.lists(steps, min_size=1, max_size=3) + if extent == 1: + return steps | st.just([1]) + cuts = st.lists(st.integers(min_value=1, max_value=extent - 1), unique=True, max_size=3) + edges = cuts.map( + lambda c: [b - a for a, b in zip([0, *sorted(c)], [*sorted(c), extent], strict=True)] + ) + return steps | edges + + +class ArrayLifecycle(RuleBasedStateMachine): + def __init__(self) -> None: + super().__init__() + self._rectilinear = zarr.config.set({"array.rectilinear_chunks": True}) + self._rectilinear.__enter__() + self.store = MemoryStore() + self.path = "a" + self.model: np.ndarray[Any, np.dtype[np.int16]] = np.zeros((0,), dtype=DTYPE) + # Cells whose value the model knows; see the module docstring. + self.known: np.ndarray[Any, np.dtype[np.bool_]] = np.ones((0,), dtype=bool) + # Every shape the array has had, to find cells a resize brings back. + self.past_shapes: list[tuple[int, ...]] = [] + self.fill = 0 + # A legacy store warns until its metadata is re-saved. + self.expect_open_warning = False + + # -------------------------------------------------------------- creation + @initialize(data=st.data()) + def create(self, data: st.DataObject) -> None: + zarr_format: Literal[2, 3] = data.draw(st.sampled_from([2, 3]), label="zarr_format") + shape = data.draw( + npst.array_shapes(min_dims=1, max_dims=3, min_side=0, max_side=MAX_SIDE), + label="shape", + ) + self.fill = data.draw(st.integers(-3, 3), label="fill_value") + # sampled_from favours early entries; the less common spellings go first. + spellings = ["ints", "legacy-zero", "-1", "False", "auto"] + if zarr_format == 3: + spellings.insert(1, "rectilinear") + spelling = data.draw(st.sampled_from(spellings), label="chunk spelling") + event(f"chunks: {spelling}") + if any(s == 0 for s in shape): + event("created with a zero-length axis") + + chunks: Any + shards: Any = None + if spelling == "-1": + chunks = -1 + elif spelling == "False": + chunks = False + elif spelling == "auto": + chunks = "auto" + elif spelling == "rectilinear": + chunks = [data.draw(_rectilinear_dim(s)) for s in shape] + if not any(isinstance(c, list) for c in chunks): + chunks[0] = [chunks[0]] if shape[0] == 0 else [shape[0]] + else: + chunks = tuple(data.draw(st.integers(1, 4)) for _ in shape) + if ( + spelling == "ints" + and zarr_format == 3 + and data.draw(st.booleans(), label="sharded") + ): + shards = tuple(c * data.draw(st.integers(1, 2)) for c in chunks) + event("sharded") + note(f"create {shape=} {chunks=} {shards=} {zarr_format=} fill={self.fill}") + zarr.create_array( + self.store, + name=self.path, + shape=shape, + chunks=chunks, + shards=shards, + dtype=DTYPE, + fill_value=self.fill, + zarr_format=zarr_format, + ) + self.model = np.full(shape, self.fill, dtype=DTYPE) + self.known = np.ones(shape, dtype=bool) + self.past_shapes = [shape] + + if spelling == "legacy-zero": + # What zarr-python wrote before 3.4 for an array created with a + # zero-length axis and one chunk spanning it; older releases could + # grow that axis without storing a chunk, so any extent is possible. + zero_axes = data.draw( + st.lists(st.integers(0, len(shape) - 1), min_size=1, unique=True), + label="axes stored with chunk size 0", + ) + stored_zero = data.draw(st.sampled_from([0, False]), label="stored zero") + self._rewrite_stored_chunks(zarr_format, zero_axes, stored_zero) + self.expect_open_warning = True + if any(shape[i] > 0 for i in zero_axes): + event("legacy zero chunk on a grown axis") + + def _rewrite_stored_chunks( + self, zarr_format: Literal[2, 3], axes: list[int], value: Any + ) -> None: + key = f"{self.path}/{'.zarray' if zarr_format == 2 else 'zarr.json'}" + buf = sync(self.store.get(key, prototype=default_buffer_prototype())) + assert buf is not None + doc = json.loads(buf.to_bytes()) + sizes = ( + doc["chunks"] if zarr_format == 2 else doc["chunk_grid"]["configuration"]["chunk_shape"] + ) + for axis in axes: + sizes[axis] = value + sync(self.store.set(key, cpu.Buffer.from_bytes(json.dumps(doc).encode()))) + + def _open(self) -> zarr.Array[Any]: + with warnings.catch_warnings(record=True) as record: + warnings.simplefilter("always", ZarrUserWarning) + arr = zarr.open_array(self.store, path=self.path, mode="r+") + warned = any(issubclass(w.category, ZarrUserWarning) for w in record) + assert warned is self.expect_open_warning, [str(w.message) for w in record] + return arr + + # ----------------------------------------------------------------- rules + @rule(data=st.data()) + def append(self, data: st.DataObject) -> None: + arr = self._open() + axis = data.draw(st.integers(0, self.model.ndim - 1), label="axis") + block_shape = list(self.model.shape) + block_shape[axis] = data.draw(st.integers(0, 4), label="rows") + block = data.draw(npst.arrays(DTYPE, tuple(block_shape)), label="block") + note(f"append {block.shape} along {axis} to {self.model.shape}") + if self.model.shape[axis] == 0: + event("append to a zero-length axis") + arr.append(block, axis=axis) + self._reshape_model(arr.shape) + tail = tuple( + slice(-block.shape[axis], None) if i == axis and block.shape[axis] else slice(None) + for i in range(self.model.ndim) + ) + if block.shape[axis]: + self.model[tail] = block + self.known[tail] = True + # Growing the array rewrites its metadata, which stores any correction. + self.expect_open_warning = False + + @rule(data=st.data()) + def resize(self, data: st.DataObject) -> None: + arr = self._open() + new_shape = data.draw( + st.tuples(*(st.integers(0, MAX_SIDE) for _ in self.model.shape)), label="new shape" + ) + note(f"resize {self.model.shape} -> {new_shape}") + if any(o == 0 and n > 0 for o, n in zip(self.model.shape, new_shape, strict=True)): + event("resize grows a zero-length axis") + arr.resize(new_shape) + self._reshape_model(new_shape) + self.expect_open_warning = False + + def _reshape_model(self, new_shape: tuple[int, ...]) -> None: + """Resize the model: kept cells keep their values, new cells hold the fill + value, and cells that an earlier shape held but the current one cut off + become unknown.""" + overlap = tuple( + slice(0, min(o, n)) for o, n in zip(self.model.shape, new_shape, strict=True) + ) + model = np.full(new_shape, self.fill, dtype=DTYPE) + known = np.ones(new_shape, dtype=bool) + for past in self.past_shapes: + known[tuple(slice(0, min(p, n)) for p, n in zip(past, new_shape, strict=True))] = False + model[overlap] = self.model[overlap] + known[overlap] = self.known[overlap] + if not known.all(): + event("resize brings back cells cut off earlier") + self.model, self.known = model, known + self.past_shapes.append(tuple(new_shape)) + + @rule(data=st.data()) + def write(self, data: st.DataObject) -> None: + arr = self._open() + region = tuple( + slice(*sorted(data.draw(st.tuples(st.integers(0, s), st.integers(0, s))))) + for s in self.model.shape + ) + values = data.draw(npst.arrays(DTYPE, self.model[region].shape), label="values") + note(f"write {region}") + arr[region] = values + self.model[region] = values + self.known[region] = True + + @rule() + def resave_metadata(self) -> None: + """What the legacy warning tells users to do.""" + self._open().update_attributes({}) + self.expect_open_warning = False + + def teardown(self) -> None: + self._rectilinear.__exit__(None, None, None) + + # ------------------------------------------------------------ invariants + @invariant() + def matches_model(self) -> None: + arr = self._open() + assert arr.shape == self.model.shape + actual = np.asarray(arr[...]) + np.testing.assert_array_equal(actual[self.known], self.model[self.known]) + + +ArrayLifecycle.TestCase.settings = settings(max_examples=200, stateful_step_count=12, deadline=None) +TestArrayLifecycle = ArrayLifecycle.TestCase From 1723f0b43f0b14b44aa76414dd7ce3111564a444 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Fri, 25 Sep 2026 15:26:25 +0200 Subject: [PATCH 18/76] fix(metadata): stop the zero chunk size warning from naming writers The warning for a stored chunk size of 0 said which zarr-python releases wrote it. The check runs on every metadata construction, so metadata built in code, such as VirtualiZarr's kerchunk writer passing an empty array's shape as `chunks`, was told it came from zarr-python 2.x. The warning now says what the chunk size is read as and, on an axis of positive length, that the axis holds only the fill value. `legacy_writers` is gone, and the docstrings no longer narrate release history; the changelog keeps it. Assisted-by: ClaudeCode:claude-opus-5-5 Co-Authored-By: Claude Opus 5.5 --- src/zarr/core/metadata/common.py | 26 +++++++++++--------------- src/zarr/core/metadata/v2.py | 6 +----- src/zarr/core/metadata/v3.py | 11 ++++------- tests/test_array_stateful.py | 4 ++-- tests/test_metadata/test_common.py | 17 +++++++---------- 5 files changed, 25 insertions(+), 39 deletions(-) diff --git a/src/zarr/core/metadata/common.py b/src/zarr/core/metadata/common.py index 14c6b8dd66..00f1e96254 100644 --- a/src/zarr/core/metadata/common.py +++ b/src/zarr/core/metadata/common.py @@ -29,7 +29,7 @@ def parse_attributes(data: dict[str, JSON] | None) -> dict[str, JSON]: def parse_stored_regular_chunk_shape( - chunk_shape: Sequence[int], shape: Sequence[int], *, legacy_writers: str + chunk_shape: Sequence[int], shape: Sequence[int] ) -> tuple[int, ...]: """Validate a stored regular chunk grid's chunk shape against the array shape. @@ -39,15 +39,13 @@ def parse_stored_regular_chunk_shape( length 0, so its chunk sizes must not be passed here. The chunk shape must have one entry per array axis, and every chunk size - must be at least 1, with one exception. `legacy_writers` stored a chunk - size of 0 (or JSON `false`) for an array created with a zero-length axis - and a chunk spec meaning "one chunk spanning the axis". That is how the - size is read, `max(extent, 1)`, with a `ZarrUserWarning` that says how to - re-save valid metadata. The axis may have grown since: those writers let - the array be resized or appended to, but a chunk size of 0 gives a grid of - zero chunks, so no chunk was ever stored for it and any chunk size reads - the store correctly. The warning then says that the appended data was not - saved. A negative chunk size is rejected. + must be at least 1, with one exception: a chunk size of 0 (or JSON + `false`) is read as one chunk spanning its axis, `max(extent, 1)`, with a + `ZarrUserWarning` that says how to re-save valid metadata. A chunk size of + 0 gives a grid of zero chunks, so no chunk can be stored under it and any + positive chunk size reads the store correctly; if the axis has positive + length, the warning also says that it holds only the fill value. A + negative chunk size is rejected. """ if len(chunk_shape) != len(shape): raise ValueError( @@ -61,15 +59,13 @@ def parse_stored_regular_chunk_shape( if size == 0: corrected = max(extent, 1) msg = ( - f"Dimension {dim_idx}: chunk edge length {size!r} (as written by " - f"{legacy_writers} for an array created with a zero-length axis) is read " + f"Dimension {dim_idx}: chunk edge length {size!r} is invalid and is read " f"as one chunk spanning the axis, of size {corrected}." ) if extent > 0: msg += ( - f" The axis has since grown to {extent}, but no chunk can be stored " - "under a chunk size of 0, so data written to it before now was not " - "saved and reads as the fill value." + f" No chunk can be stored under a chunk size of 0, so the {extent} " + "elements along this axis hold only the fill value." ) warnings.warn(f"{msg} {RESAVE_METADATA_HINT}", ZarrUserWarning, stacklevel=3) size = corrected diff --git a/src/zarr/core/metadata/v2.py b/src/zarr/core/metadata/v2.py index c213bc98b6..08c290fab2 100644 --- a/src/zarr/core/metadata/v2.py +++ b/src/zarr/core/metadata/v2.py @@ -88,11 +88,7 @@ def __init__( Metadata for a Zarr format 2 array. """ shape_parsed = parse_shapelike(shape) - chunks_parsed = parse_stored_regular_chunk_shape( - parse_shapelike(chunks), - shape_parsed, - legacy_writers="zarr-python 2.x, and by 3.x before 3.4", - ) + chunks_parsed = parse_stored_regular_chunk_shape(parse_shapelike(chunks), shape_parsed) compressor_parsed = parse_compressor(compressor) order_parsed = parse_indexing_order(order) dimension_separator_parsed = parse_separator(dimension_separator) diff --git a/src/zarr/core/metadata/v3.py b/src/zarr/core/metadata/v3.py index ef7a5bf0e2..1db659beb8 100644 --- a/src/zarr/core/metadata/v3.py +++ b/src/zarr/core/metadata/v3.py @@ -498,10 +498,9 @@ def _parse_stored_regular_chunk_grid( Only a `regular` grid whose `chunk_shape` is all integers is a regular chunk shape, and only that is handed to `parse_stored_regular_chunk_shape`. Anything else is not a regular chunk shape and is left for the chunk grid - parser: other grids define their own chunk semantics. zarr-python 3.0 and - 3.1 stored `chunk_shape: [0]` (and 3.0 `[false]`) for an array created with - a zero-length axis, and 3.1 kept it when the axis grew. This runs here rather than in the grid parser because - it needs the array shape, which chunk grid metadata does not carry. + parser: other grids define their own chunk semantics. This runs here rather + than in the grid parser because it needs the array shape, which chunk grid + metadata does not carry. """ if not isinstance(chunk_grid, Mapping) or chunk_grid.get("name") != "regular": return chunk_grid @@ -511,9 +510,7 @@ def _parse_stored_regular_chunk_grid( chunk_shape = configuration.get("chunk_shape") if not _is_regular_chunk_shape(chunk_shape): return chunk_grid - parsed = parse_stored_regular_chunk_shape( - chunk_shape, shape, legacy_writers="zarr-python 3.0 and 3.1" - ) + parsed = parse_stored_regular_chunk_shape(chunk_shape, shape) corrected: dict[str, Any] = dict(chunk_grid) corrected["configuration"] = {**configuration, "chunk_shape": list(parsed)} return corrected diff --git a/tests/test_array_stateful.py b/tests/test_array_stateful.py index 75a6b12d59..a5a6627a5d 100644 --- a/tests/test_array_stateful.py +++ b/tests/test_array_stateful.py @@ -7,8 +7,8 @@ `resize` deletes only the chunks that fall entirely outside the new shape, so cells cut off by a shrink can come back with their old values when the axis -grows again (as in zarr-python 2.x). The model does not encode that chunk-level -behaviour: a cell cut off and brought back is unknown until it is written. +grows again. The model does not encode that chunk-level behaviour: a cell cut +off and brought back is unknown until it is written. """ from __future__ import annotations diff --git a/tests/test_metadata/test_common.py b/tests/test_metadata/test_common.py index f35f2fa954..c3ed9416eb 100644 --- a/tests/test_metadata/test_common.py +++ b/tests/test_metadata/test_common.py @@ -23,8 +23,8 @@ ((np.int64(0),), (0,), (1,), "of size 1"), ((4, 0), (4, 0), (4, 1), "Dimension 1"), ((0, 0), (0, 0), (1, 1), "Dimension 0"), - ((0,), (5,), (5,), "grown to 5.*was not saved"), - ((4, 0), (4, 3), (4, 3), "grown to 3.*was not saved"), + ((0,), (5,), (5,), "5 elements along this axis hold only the fill value"), + ((4, 0), (4, 3), (4, 3), "3 elements along this axis hold only the fill value"), ], ids=[ "valid", @@ -45,13 +45,11 @@ def test_parse_stored_regular_chunk_shape( warning: str | None, ) -> None: """A valid chunk shape is returned as is; a chunk size of 0 is read as one - chunk spanning the axis, with a warning naming the writer and how to re-save, - and, if the axis has grown, that data written to it was not saved.""" + chunk spanning the axis, with a warning saying how to re-save and, on an axis + of positive length, that it holds only the fill value.""" with warnings.catch_warnings(record=True) as record: warnings.simplefilter("always") - parsed = parse_stored_regular_chunk_shape( - chunk_shape, shape, legacy_writers="an old writer" - ) + parsed = parse_stored_regular_chunk_shape(chunk_shape, shape) assert parsed == expected messages = [str(w.message) for w in record if issubclass(w.category, ZarrUserWarning)] if warning is None: @@ -59,17 +57,16 @@ def test_parse_stored_regular_chunk_shape( else: assert any(re.search(warning, message) for message in messages) for message in messages: - assert "an old writer" in message assert "update_attributes({})" in message def test_parse_stored_regular_chunk_shape_rejects_dimension_mismatch() -> None: """The chunk shape needs one entry per array axis.""" with pytest.raises(ValueError, match="same number of dimensions"): - parse_stored_regular_chunk_shape((4,), (10, 10), legacy_writers="an old writer") + parse_stored_regular_chunk_shape((4,), (10, 10)) def test_parse_stored_regular_chunk_shape_rejects_negative() -> None: """A negative chunk size is rejected even on a zero-length axis.""" with pytest.raises(ValueError, match="chunk edge length must be >= 1, got -1"): - parse_stored_regular_chunk_shape((-1,), (0,), legacy_writers="an old writer") + parse_stored_regular_chunk_shape((-1,), (0,)) From 09639f482600afa8c334c8ad4ef536c66f712599 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Fri, 25 Sep 2026 15:35:41 +0200 Subject: [PATCH 19/76] fix(metadata): stop the mixed chunk grid messages from naming releases The warning and error for a stored `regular` grid that lists chunk edges said which zarr releases wrote it. They now say what is wrong with the metadata, that it lists chunk edges only a rectilinear grid can declare, and how it is read. The reader's docstrings no longer narrate release history either; the changelog fragment keeps it. Assisted-by: ClaudeCode:claude-opus-5-5 Co-Authored-By: Claude Opus 5.5 --- src/zarr/core/metadata/v3.py | 13 ++++++------- tests/test_metadata/test_v3.py | 6 +++--- 2 files changed, 9 insertions(+), 10 deletions(-) diff --git a/src/zarr/core/metadata/v3.py b/src/zarr/core/metadata/v3.py index d877df192a..28fc755d82 100644 --- a/src/zarr/core/metadata/v3.py +++ b/src/zarr/core/metadata/v3.py @@ -464,11 +464,10 @@ def _parse_mixed_regular_chunk_grid( ) -> RectilinearChunkGridMetadata: """Read a "regular" chunk grid whose chunk_shape contains edge lists. - zarr 3.2.0 and 3.2.1 wrote mixed chunk specs such as ``(2, (5, 10, 5))`` - as ``{"name": "regular", "configuration": {"chunk_shape": [2, [5, 10, 5]]}}`` - while laying the chunks out as a rectilinear grid. That metadata is - invalid, but the data is intact, so it is read as the rectilinear grid - it describes. See https://github.com/zarr-developers/zarr-python/issues/4374. + A regular grid cannot list chunk edges, so metadata such as + ``{"name": "regular", "configuration": {"chunk_shape": [2, [5, 10, 5]]}}`` + is invalid. It does describe one rectilinear grid unambiguously, so it is + read as that grid. See https://github.com/zarr-developers/zarr-python/issues/4374. """ # Put the dimensions in the JSON forms the rectilinear parser reads: a # metadata dict built in Python holds a tuple where JSON holds a list. @@ -478,8 +477,8 @@ def _parse_mixed_regular_chunk_grid( ] msg = ( f"This array's chunk grid is named 'regular' but its chunk_shape {chunk_shapes!r} " - "lists explicit chunk edges for some dimensions. zarr 3.2.0 and 3.2.1 wrote " - "rectilinear chunk grids this way by mistake. " + "lists explicit chunk edges for some dimensions, which only a rectilinear chunk " + "grid can declare. " ) if not config.get("array.rectilinear_chunks"): raise ValueError( diff --git a/tests/test_metadata/test_v3.py b/tests/test_metadata/test_v3.py index ef69b9e316..f0d3541de0 100644 --- a/tests/test_metadata/test_v3.py +++ b/tests/test_metadata/test_v3.py @@ -282,7 +282,7 @@ def test_read_mixed_regular_chunk_grid( chunk_grid={"name": "regular", "configuration": {"chunk_shape": chunk_shape}}, ) with config.set({"array.rectilinear_chunks": True}): - with pytest.warns(ZarrUserWarning, match="zarr 3.2.0 and 3.2.1"): + with pytest.warns(ZarrUserWarning, match="only a rectilinear chunk grid can declare"): meta = ArrayV3Metadata.from_dict(d) # type: ignore[arg-type] assert meta.chunk_grid == RectilinearChunkGridMetadata(chunk_shapes=case.output) @@ -295,7 +295,7 @@ def test_read_mixed_regular_chunk_grid_requires_rectilinear_chunks() -> None: ) with ( config.set({"array.rectilinear_chunks": False}), - pytest.raises(ValueError, match="zarr 3.2.0 and 3.2.1 wrote rectilinear chunk grids"), + pytest.raises(ValueError, match="only a rectilinear chunk grid can declare"), ): ArrayV3Metadata.from_dict(d) # type: ignore[arg-type] @@ -313,7 +313,7 @@ def test_open_array_with_mixed_regular_chunk_grid() -> None: doc["chunk_grid"] = {"name": "regular", "configuration": {"chunk_shape": [2, [5, 10, 5]]}} store._store_dict["zarr.json"] = cpu.Buffer.from_bytes(json.dumps(doc).encode()) - with pytest.warns(ZarrUserWarning, match="zarr 3.2.0 and 3.2.1"): + with pytest.warns(ZarrUserWarning, match="only a rectilinear chunk grid can declare"): arr = zarr.open_array(store) np.testing.assert_array_equal(arr[:], data) From 3d223f18f4035ae25eb2d88d2d352ec3dbb3fafb Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Fri, 25 Sep 2026 20:56:13 +0200 Subject: [PATCH 20/76] chore(metadata): drop an orphaned comment in ArrayV2Metadata.__init__ The comment introduced a consistency check (`parse_metadata`) that this branch removed; the check now lives in `parse_stored_regular_chunk_shape`. Assisted-by: ClaudeCode:claude-opus-5-5 Co-Authored-By: Claude Opus 5.5 --- src/zarr/core/metadata/v2.py | 2 -- 1 file changed, 2 deletions(-) diff --git a/src/zarr/core/metadata/v2.py b/src/zarr/core/metadata/v2.py index 08c290fab2..3ae2a8bd96 100644 --- a/src/zarr/core/metadata/v2.py +++ b/src/zarr/core/metadata/v2.py @@ -110,8 +110,6 @@ def __init__( object.__setattr__(self, "fill_value", fill_value_parsed) object.__setattr__(self, "attributes", attributes_parsed) - # ensure that the metadata document is consistent - @property def ndim(self) -> int: return len(self.shape) From 339c514cac483c6555619fd2741a772d95965982 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Fri, 25 Sep 2026 21:54:57 +0200 Subject: [PATCH 21/76] fix(chunk-grids): size a full-span shard as a multiple of the inner chunk One definition of "one chunk spanning the axis", full_span_chunk_size(span, unit), is the smallest positive multiple of unit covering span. shards=-1 and shards=False now use the inner chunk size as unit, so a zero-length axis, or one whose length is not a multiple of the inner chunk, gets a valid shard instead of a divisibility error. The zero-length array test drops the explicit-size spellings and keeps the spellings that derive a chunk from the span. Assisted-by: ClaudeCode:claude-opus-5-5 Co-Authored-By: Claude Opus 5.5 --- src/zarr/core/chunk_grids.py | 66 +++++----- tests/test_chunk_grids.py | 247 ++++++++++------------------------- 2 files changed, 96 insertions(+), 217 deletions(-) diff --git a/src/zarr/core/chunk_grids.py b/src/zarr/core/chunk_grids.py index 545aa3c581..d5d61ee010 100644 --- a/src/zarr/core/chunk_grids.py +++ b/src/zarr/core/chunk_grids.py @@ -47,12 +47,7 @@ @dataclass(frozen=True) class FixedDimension: """Uniform chunk size. Boundary chunks contain less data but are - encoded at full size by the codec pipeline. - - The chunk edge length is always at least 1, matching the invariant the - metadata layer enforces for every stored chunk grid. The extent may be 0: - a zero-length axis simply has zero chunks (``ceildiv(0, size) == 0``). - """ + encoded at full size by the codec pipeline.""" size: int # chunk edge length (>= 1) extent: int # array dimension length (>= 0) @@ -656,18 +651,15 @@ class ChunkLayout(NamedTuple): inner: ChunkLayout | None = None -def _full_span_chunk_size(span: int) -> int: - """The edge length of one chunk covering an entire axis of length *span*. +def full_span_chunk_size(span: int, unit: int = 1) -> int: + """The edge length of one chunk spanning an axis of length `span`. - This is *the* definition of "one chunk spans the axis" for a possibly - zero-length axis. Chunk edge lengths must be at least 1 (the invariant - shared by `FixedDimension`, `VaryingDimension` and the stored chunk grid - metadata), so a zero-length axis gets chunk size 1 and zero chunks. Every - spelling that derives a chunk size from a span — ``chunks=-1``, - ``chunks=False``, ``chunks="auto"``, ``shards="auto"`` — must route - through this helper rather than clamping on its own. + This is the smallest positive multiple of `unit` that covers `span`, so a + zero-length axis gets a chunk of size `unit` and zero chunks. `unit` is the + size the chunk must be a multiple of: the inner chunk size for a shard, 1 + otherwise. """ - return max(span, 1) + return unit * max(1, ceildiv(span, unit)) def _guess_regular_chunks( @@ -707,10 +699,11 @@ def _guess_regular_chunks( shape = (shape,) if typesize == 0: - return tuple(_full_span_chunk_size(s) for s in shape) + return tuple(full_span_chunk_size(s) for s in shape) ndims = len(shape) - chunks = np.array([_full_span_chunk_size(s) for s in shape], dtype="=f8") + # require chunks to have non-zero length for all dimensions + chunks = np.maximum(np.array(shape, dtype="=f8"), 1) # Determine the optimal chunk size in bytes using a PyTables expression. # This is kept as a float. @@ -745,7 +738,7 @@ def _guess_regular_chunks( return tuple(int(x) for x in chunks) -def normalize_chunks_1d(chunks: int | Iterable[object], span: int) -> DimensionGrid: +def normalize_chunks_1d(chunks: int | Iterable[object], span: int, unit: int = 1) -> DimensionGrid: """ Normalize a one-dimensional chunk specification into a dimension grid: `FixedDimension` for scalar chunk sizes, `VaryingDimension` for explicit @@ -753,21 +746,15 @@ def normalize_chunks_1d(chunks: int | Iterable[object], span: int) -> DimensionG the span, and the uniform form is O(1) in the number of chunks — a dimension with `2**62` chunks must not materialize one entry per chunk. - `-1` means "one chunk covering the entire span" (see `_full_span_chunk_size` - for what that means on a zero-length span). + `-1` means "one chunk covering the entire span", sized by + `full_span_chunk_size(span, unit)`. Explicit chunk size lists must sum to the span exactly and always produce `VaryingDimension`, even when the sizes happen to be uniform: the input syntax declares the grid kind, so a per-chunk list is preserved as a rectilinear dimension rather than silently collapsed to a regular one, which would change how the dimension grows on resize. For scalar sizes - the last chunk may overhang the span. - - The one exception to the sum rule is a zero-length span: no non-empty list of - positive edges can sum to 0, so a non-empty list of positive integers is retained - and the edges describe the chunks the axis will grow into on `append` / - `resize`. This is the same state a rectilinear axis reaches when it is - resized down to 0 — `VaryingDimension` allows trailing edges beyond the - extent — so creating at length 0 and shrinking to 0 are indistinguishable. + the last chunk may overhang the span. On a zero-length span any non-empty + list of positive sizes is kept: the chunks the axis grows into. """ # `numbers.Integral` rather than `int` so that numpy integer scalars (which are not # `int` subclasses) take the uniform-chunk path instead of being treated as a sequence. @@ -778,7 +765,7 @@ def normalize_chunks_1d(chunks: int | Iterable[object], span: int) -> DimensionG if chunk_size < -1 or chunk_size == 0: raise ValueError(f"Chunk size must be positive or -1, got {chunk_size}") if chunk_size == -1: - return FixedDimension(size=_full_span_chunk_size(span), extent=span) + return FixedDimension(size=full_span_chunk_size(span, unit), extent=span) return FixedDimension(size=chunk_size, extent=span) else: try: @@ -803,8 +790,6 @@ def normalize_chunks_1d(chunks: int | Iterable[object], span: int) -> DimensionG ints: list[int] = [int(c) for c in chunk_list] # type: ignore[call-overload] if any(c <= 0 for c in ints): raise ValueError(f"All chunk sizes must be positive, got {ints}") - # A zero-length span cannot be covered by positive edges; the edges are the - # chunks the axis will grow into, exactly as after ``resize(0)``. if span > 0 and sum(ints) != span: raise ValueError(f"Chunk sizes {ints} do not sum to span {span}") return VaryingDimension(ints, extent=span) @@ -813,6 +798,7 @@ def normalize_chunks_1d(chunks: int | Iterable[object], span: int) -> DimensionG def normalize_chunks_nd( chunks: Any, shape: tuple[int, ...], + unit: tuple[int, ...] | None = None, ) -> ChunkGrid: """ Normalize a chunk specification into a `ChunkGrid`. @@ -832,6 +818,10 @@ def normalize_chunks_nd( `ChunkGrid` directly. `chunks=None` and `chunks=True` are rejected here — the caller is responsible for choosing between explicit sizes and auto-chunking. + + `unit` gives, per axis, the size a chunk must be a multiple of (the inner + chunk shape, when normalizing a shard shape); it only affects the chunks + that `-1` and `False` derive from the span. """ from zarr.core.metadata.v3 import RectilinearChunkGridMetadata, RegularChunkGridMetadata @@ -845,8 +835,7 @@ def normalize_chunks_nd( f'{chunks!r} is not a valid chunk input. Use chunks=None or chunks="auto" from the top-level API for auto-chunking, or pass an int / tuple of ints.' ) - # handle no chunking: one chunk covering every axis. Routed through the -1 sentinel so - # the zero-length-axis rule lives in one place (_full_span_chunk_size). + # handle no chunking: one chunk covering every axis. if chunks is False: chunks = -1 @@ -860,8 +849,13 @@ def normalize_chunks_nd( f"chunks has {len(chunks)} dimensions but shape has {len(shape)} dimensions" ) + if unit is None: + unit = (1,) * len(shape) return ChunkGrid( - dimensions=tuple(normalize_chunks_1d(c, span=s) for c, s in zip(chunks, shape, strict=True)) + dimensions=tuple( + normalize_chunks_1d(c, span=s, unit=u) + for c, s, u in zip(chunks, shape, unit, strict=True) + ) ) @@ -1009,5 +1003,5 @@ def resolve_outer_and_inner_chunks( else: shard_flat = cast("tuple[int, ...]", shard_shape) - outer = normalize_chunks_nd(shard_flat, array_shape) + outer = normalize_chunks_nd(shard_flat, array_shape, unit=chunk_shape_flat) return ChunkLayout(outer_chunks=outer, inner=ChunkLayout(outer_chunks=chunks)) diff --git a/tests/test_chunk_grids.py b/tests/test_chunk_grids.py index cfd2e42258..d171937898 100644 --- a/tests/test_chunk_grids.py +++ b/tests/test_chunk_grids.py @@ -1,7 +1,4 @@ import contextlib -import json -import warnings -from pathlib import Path from typing import Any, Literal, cast import numpy as np @@ -16,6 +13,7 @@ VaryingDimension, _guess_num_chunks_per_axis_shard, _guess_regular_chunks, + full_span_chunk_size, normalize_chunks_1d, normalize_chunks_nd, resolve_outer_and_inner_chunks, @@ -371,148 +369,89 @@ def test_create_0d_array_auto_shards_with_target_shard_size() -> None: # -- Zero-length dimensions -- -# -# One invariant: a chunk edge length is always >= 1, an extent may be 0. Every spelling -# that derives a chunk size from a span (-1, False, "auto", shards="auto") must agree on -# chunk size 1 for a zero-length axis, in both Zarr formats, with or without sharding. -# Historically each spelling clamped (or failed to clamp) on its own; see #4304, #4305, -# #4307, #4328 and, further back, #150, #241, #303, #972, #1977, #2434, #3711. - -ZeroLengthChunkSpelling = Literal["minus-one", "false", "auto", "one", "ones", "rectilinear"] -ZeroLengthShards = Literal["auto", "auto-budget", "explicit"] | None - -# Spellings whose chunk size is derived from the axis span rather than given explicitly. -_SPAN_DERIVED_SPELLINGS: frozenset[ZeroLengthChunkSpelling] = frozenset( - {"minus-one", "false", "auto"} + + +@pytest.mark.parametrize( + ("span", "unit", "expected"), + [(0, 1, 1), (5, 1, 5), (0, 4, 4), (8, 4, 8), (10, 4, 12)], ) +def test_full_span_chunk_size(span: int, unit: int, expected: int) -> None: + """One chunk spanning an axis is the smallest positive multiple of `unit` covering it.""" + assert full_span_chunk_size(span, unit) == expected -def _zero_length_chunks_arg(spelling: ZeroLengthChunkSpelling, shape: tuple[int, ...]) -> Any: - """Translate a chunk-spelling id into the `chunks=` argument for `shape`.""" - match spelling: - case "minus-one": - return -1 - case "false": - return False - case "auto": - return "auto" - case "one": - return 1 - case "ones": - return (1,) * len(shape) - case "rectilinear": - return [[2, 2]] * len(shape) - - -@pytest.mark.parametrize("spelling", ["minus-one", "false", "auto", "one", "ones", "rectilinear"]) +@pytest.mark.parametrize("chunks", [-1, False, "auto"]) @pytest.mark.parametrize( "shape", [(0,), (0, 4), (4, 0), (0, 0), ()], ids=["1d", "2d-lead", "2d-trail", "2d-both", "0d"], ) @pytest.mark.parametrize( - ("zarr_format", "shards"), - [(2, None), (3, None), (3, "auto"), (3, "auto-budget"), (3, "explicit")], - ids=["v2", "v3", "v3-auto-shards", "v3-auto-shards-budget", "v3-explicit-shards"], + ("zarr_format", "shards", "target_shard_size_bytes"), + [(2, None, None), (3, None, None), (3, "auto", None), (3, "auto", 128 * 1024 * 1024)], + ids=["v2", "v3", "v3-auto-shards", "v3-auto-shards-budget"], ) def test_create_zero_length_array( - spelling: ZeroLengthChunkSpelling, + chunks: Any, shape: tuple[int, ...], zarr_format: Literal[2, 3], - shards: ZeroLengthShards, + shards: Literal["auto"] | None, + target_shard_size_bytes: int | None, ) -> None: - """Every chunk spelling produces a valid, usable grid on a zero-length axis. - - Span-derived spellings resolve to chunk size 1 on zero-length axes (and the full span - elsewhere, for these small shapes); explicit spellings are stored verbatim. In every case - the stored metadata matches `arr.chunks` / `arr.shards`, the array can grow along the - empty axis, round-trip data, and shrink back to empty. - """ - ndim = len(shape) - if spelling == "rectilinear": - if zarr_format == 2: - pytest.skip("Zarr format 2 does not support rectilinear chunk grids") - if shards is not None: - pytest.skip("rectilinear chunks with sharding is not supported") - if ndim == 0: - pytest.skip("a 0-d array has no dimension to chunk rectilinearly") - if shards == "explicit" and ndim == 0: - pytest.skip("a 0-d array has no axis to shard explicitly") - - chunks = _zero_length_chunks_arg(spelling, shape) - expected_chunks: tuple[int, ...] | None - if spelling in _SPAN_DERIVED_SPELLINGS: - expected_chunks = tuple(max(s, 1) for s in shape) - elif spelling == "rectilinear": - expected_chunks = None - else: - expected_chunks = (1,) * ndim - - shards_arg: Any - expected_shards: tuple[int, ...] | None - match shards: - case None: - shards_arg, expected_shards = None, None - case "auto" | "auto-budget": - # Axes this short never split, so the guessed shard equals the chunk. - shards_arg, expected_shards = "auto", expected_chunks - case "explicit": - # A shard larger than the (zero) extent is fine: the axis has zero shards. - shards_arg = tuple(2 if s == 0 else s for s in shape) - expected_shards = shards_arg - + """Every spelling of one chunk spanning the axis gives chunk size 1 on a + zero-length axis, and the array can grow along that axis and shrink back.""" + expected = tuple(max(s, 1) for s in shape) warns = ( pytest.warns(ZarrUserWarning, match="Automatic shard shape inference is experimental") - if shards_arg == "auto" + if shards == "auto" else contextlib.nullcontext() ) - budget = 128 * 1024 * 1024 if shards == "auto-budget" else None - # The rectilinear flag must stay set for the array's whole life, not just creation. - with zarr.config.set( - {"array.rectilinear_chunks": True, "array.target_shard_size_bytes": budget} - ): - with warns: - arr = zarr.create_array( - store={}, - shape=shape, - dtype="int64", - chunks=chunks, - shards=shards_arg, - zarr_format=zarr_format, - ) + with zarr.config.set({"array.target_shard_size_bytes": target_shard_size_bytes}), warns: + arr = zarr.create_array( + store={}, + shape=shape, + dtype="int64", + chunks=chunks, + shards=shards, + zarr_format=zarr_format, + ) + assert arr.chunks == expected + assert arr.shards == (None if shards is None else expected) + meta = cast(dict[str, Any], arr.metadata.to_dict()) + stored = ( + meta["chunks"] if zarr_format == 2 else meta["chunk_grid"]["configuration"]["chunk_shape"] + ) + assert tuple(stored) == expected - # In-memory view and stored metadata agree with the invariant. - assert arr.shards == expected_shards - meta = cast(dict[str, Any], arr.metadata.to_dict()) - if spelling == "rectilinear": - grid = meta["chunk_grid"] - assert grid["name"] == "rectilinear" - # Stored verbatim on zero-length axes too, run-length encoded as [size, count]. - assert list(grid["configuration"]["chunk_shapes"]) == [[[2, 2]]] * ndim - assert arr.write_chunk_sizes == tuple(() if s == 0 else (2, 2) for s in shape) - else: - assert arr.chunks == expected_chunks - if zarr_format == 2: - assert meta["chunks"] == expected_chunks - else: - stored = meta["chunk_grid"]["configuration"]["chunk_shape"] - assert stored == (expected_chunks if expected_shards is None else expected_shards) - assert all(c >= 1 for c in arr.chunks) - - # The array must remain usable. - if ndim == 0: - arr[...] = 7 - assert arr[...] == 7 - return - axis = shape.index(0) - grown = tuple(2 if i == axis else s for i, s in enumerate(shape)) - data = np.full(grown, 7, dtype="int64") - arr.append(data, axis=axis) - assert arr.shape == grown - np.testing.assert_array_equal(arr[...], data) - arr.resize(shape) - assert arr.shape == shape - assert np.asarray(arr[...]).shape == shape + if not shape: + arr[...] = 7 + assert arr[...] == 7 + return + axis = shape.index(0) + grown = tuple(2 if i == axis else s for i, s in enumerate(shape)) + data = np.full(grown, 7, dtype="int64") + arr.append(data, axis=axis) + np.testing.assert_array_equal(arr[...], data) + arr.resize(shape) + assert np.asarray(arr[...]).shape == shape + + +@pytest.mark.parametrize( + ("shape", "chunks", "shards", "expected"), + [ + ((0, 20), (5, 5), -1, (5, 20)), + ((0,), (4,), False, (4,)), + ((10,), (4,), -1, (12,)), + ((8, 0), (4, 3), (-1, 6), (8, 6)), + ], +) +def test_create_full_span_shards( + shape: tuple[int, ...], chunks: tuple[int, ...], shards: Any, expected: tuple[int, ...] +) -> None: + """A shard spanning the axis is a multiple of the inner chunk, even on a zero-length axis.""" + arr = zarr.create_array(store={}, shape=shape, chunks=chunks, shards=shards, dtype="int8") + assert arr.shards == expected + assert arr.chunks == chunks @pytest.mark.parametrize("zarr_format", [2, 3]) @@ -523,11 +462,7 @@ def test_create_zero_chunk_rejected(zarr_format: Literal[2, 3]) -> None: def test_rectilinear_zero_extent_matches_resize() -> None: - """Creating a rectilinear axis at length 0 equals resizing one down to 0. - - Both leave a `VaryingDimension` whose edges lie entirely beyond the extent, so the - stored grids are identical and both grow into the same chunks on append. - """ + """Creating a rectilinear axis at length 0 equals resizing one down to 0.""" with zarr.config.set({"array.rectilinear_chunks": True}): created = zarr.create_array(store={}, shape=(0,), chunks=[[2, 2]], dtype="int64") resized = zarr.create_array(store={}, shape=(4,), chunks=[[2, 2]], dtype="int64") @@ -544,56 +479,6 @@ def test_rectilinear_zero_extent_matches_resize() -> None: assert created.write_chunk_sizes == resized.write_chunk_sizes == ((2, 1),) -def _store_legacy_zero_chunk(path: Any, zarr_format: Literal[2, 3], stored: Any) -> None: - """Rewrite an array's stored chunk size to *stored*, as older zarr-python did.""" - doc_name = ".zarray" if zarr_format == 2 else "zarr.json" - doc_path = path / doc_name - doc = json.loads(doc_path.read_text()) - if zarr_format == 2: - doc["chunks"] = [stored] - else: - doc["chunk_grid"]["configuration"]["chunk_shape"] = [stored] - doc_path.write_text(json.dumps(doc)) - - -@pytest.mark.parametrize("zarr_format", [2, 3]) -@pytest.mark.parametrize("stored", [0, False], ids=["zero", "false"]) -@pytest.mark.parametrize("extent", [0, 3], ids=["empty-axis", "grown-axis"]) -def test_legacy_zero_chunk_round_trip( - tmp_path: Path, zarr_format: Literal[2, 3], stored: Any, extent: int -) -> None: - """An array whose stored chunk size is 0 opens with one chunk spanning the - axis, appends without losing data, and re-saves as a valid chunk size. - - zarr-python wrote such metadata for an array created with a zero-length axis - until 3.4: Zarr format 2 through 3.3, and Zarr format 3 in 3.0 and 3.1, which - also wrote JSON `false` for `chunks=False`. Those versions could then grow - the axis (a 3.4.0 append, a 3.1 resize) while storing no chunk, so the - grown axis holds only the fill value. - """ - path = tmp_path / "legacy.zarr" - zarr.create_array( - store=path, shape=(extent,), chunks=(4,), dtype="int64", zarr_format=zarr_format - ) - _store_legacy_zero_chunk(path, zarr_format, stored) - - with pytest.warns(ZarrUserWarning, match="one chunk spanning the axis"): - arr = zarr.open_array(store=path, mode="a") - assert arr.chunks == (max(extent, 1),) - np.testing.assert_array_equal(arr[...], np.zeros(extent, dtype="int64")) - - arr.append(np.arange(3, dtype="int64")) - expected = np.concatenate([np.zeros(extent, dtype="int64"), np.arange(3)]) - np.testing.assert_array_equal(zarr.open_array(store=path)[...], expected) - - # The warning says to do this; it must leave metadata that reopens cleanly. - arr.update_attributes({}) - with warnings.catch_warnings(): - warnings.simplefilter("error", ZarrUserWarning) - reopened = zarr.open_array(store=path) - assert reopened.chunks == (max(extent, 1),) - - def test_normalize_chunks_1d_zero_span_accepts_any_edges() -> None: """On a zero-length span the explicit edge list is stored verbatim.""" dim = normalize_chunks_1d([3, 5], span=0) From e49eb7e7b4c1f66dab6bf5a803f086199c572922 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Fri, 25 Sep 2026 21:55:20 +0200 Subject: [PATCH 22/76] refactor(metadata): read invalid stored chunk sizes in one module of document upgrades The metadata constructors are strict again: a chunk edge length is an int of at least 1, so metadata built in code with 0 or True raises, without a warning. Stored documents are read leniently only in zarr.core.metadata.upgrades, whose upgrades map a stored array document to a valid one before the constructors run. ArrayV2Metadata.from_dict and ArrayV3Metadata.from_dict apply them, so opening an array and parsing consolidated metadata take the same path. The first upgrade reads a regular chunk size of 0 or JSON false as one chunk spanning the axis, a multiple of the inner chunk size when the array is sharded (this opens sharded stores with an outer chunk size of 0), and JSON true, which real releases wrote for chunks=(True,), as 1. Its one warning says what is invalid, how it is read, and how to re-save, including zarr.consolidate_metadata. Assisted-by: ClaudeCode:claude-opus-5-5 Co-Authored-By: Claude Opus 5.5 --- src/zarr/core/metadata/common.py | 63 +----- src/zarr/core/metadata/upgrades.py | 135 ++++++++++++ src/zarr/core/metadata/v2.py | 18 +- src/zarr/core/metadata/v3.py | 59 +----- tests/test_metadata/test_common.py | 72 ------- tests/test_metadata/test_upgrades.py | 293 +++++++++++++++++++++++++++ 6 files changed, 456 insertions(+), 184 deletions(-) create mode 100644 src/zarr/core/metadata/upgrades.py delete mode 100644 tests/test_metadata/test_common.py create mode 100644 tests/test_metadata/test_upgrades.py diff --git a/src/zarr/core/metadata/common.py b/src/zarr/core/metadata/common.py index 00f1e96254..b9e86f537a 100644 --- a/src/zarr/core/metadata/common.py +++ b/src/zarr/core/metadata/common.py @@ -1,25 +1,10 @@ from __future__ import annotations -import warnings -from typing import TYPE_CHECKING, Final - -from zarr.errors import ZarrUserWarning +from typing import TYPE_CHECKING if TYPE_CHECKING: - from collections.abc import Sequence - from zarr.core.common import JSON -RESAVE_METADATA_HINT: Final = ( - "Re-save the array metadata to store the corrected value: open the array " - "writable and call `array.update_attributes({})`." -) -"""How to persist metadata that was read under a compatibility policy. - -`update_attributes` rewrites the whole metadata document from the parsed -(corrected) metadata, so an empty update is enough. -""" - def parse_attributes(data: dict[str, JSON] | None) -> dict[str, JSON]: if data is None: @@ -28,46 +13,10 @@ def parse_attributes(data: dict[str, JSON] | None) -> dict[str, JSON]: return dict(data) -def parse_stored_regular_chunk_shape( - chunk_shape: Sequence[int], shape: Sequence[int] -) -> tuple[int, ...]: - """Validate a stored regular chunk grid's chunk shape against the array shape. - - This is for regular chunk grids only: Zarr format 2 `chunks`, and the - `chunk_shape` of a Zarr format 3 `regular` grid. Another chunk grid, such - as the rectilinear grid, is free to define its own meaning for a chunk of - length 0, so its chunk sizes must not be passed here. - - The chunk shape must have one entry per array axis, and every chunk size - must be at least 1, with one exception: a chunk size of 0 (or JSON - `false`) is read as one chunk spanning its axis, `max(extent, 1)`, with a - `ZarrUserWarning` that says how to re-save valid metadata. A chunk size of - 0 gives a grid of zero chunks, so no chunk can be stored under it and any - positive chunk size reads the store correctly; if the axis has positive - length, the warning also says that it holds only the fill value. A - negative chunk size is rejected. - """ - if len(chunk_shape) != len(shape): +def parse_chunk_edge(size: object, axis: int) -> int: + """Check that `size` is a chunk edge length: an `int` (not a `bool`) of at least 1.""" + if isinstance(size, bool) or not isinstance(size, int) or size < 1: raise ValueError( - f"The chunk shape {tuple(chunk_shape)} and the array shape {tuple(shape)} " - "must have the same number of dimensions." + f"Dimension {axis}: chunk edge length must be an integer >= 1, got {size!r}" ) - parsed: list[int] = [] - for dim_idx, (size, extent) in enumerate(zip(chunk_shape, shape, strict=True)): - if size < 0: - raise ValueError(f"Dimension {dim_idx}: chunk edge length must be >= 1, got {size!r}") - if size == 0: - corrected = max(extent, 1) - msg = ( - f"Dimension {dim_idx}: chunk edge length {size!r} is invalid and is read " - f"as one chunk spanning the axis, of size {corrected}." - ) - if extent > 0: - msg += ( - f" No chunk can be stored under a chunk size of 0, so the {extent} " - "elements along this axis hold only the fill value." - ) - warnings.warn(f"{msg} {RESAVE_METADATA_HINT}", ZarrUserWarning, stacklevel=3) - size = corrected - parsed.append(size) - return tuple(parsed) + return size diff --git a/src/zarr/core/metadata/upgrades.py b/src/zarr/core/metadata/upgrades.py new file mode 100644 index 0000000000..933b4c53e2 --- /dev/null +++ b/src/zarr/core/metadata/upgrades.py @@ -0,0 +1,135 @@ +"""Upgrades that read invalid stored array metadata documents written by older software. + +This is the only place invalid metadata is read leniently; the metadata constructors +are strict. An upgrade maps a stored array metadata document (parsed JSON) to a valid +one and warns once when it changes anything. `ArrayV2Metadata.from_dict` and +`ArrayV3Metadata.from_dict` apply the upgrades for their Zarr format, so every path +that parses a stored document, including consolidated metadata, goes through them. + +To read another kind of invalid document, add an upgrade to `V2_ARRAY_UPGRADES` or +`V3_ARRAY_UPGRADES`. +""" + +from __future__ import annotations + +import json +import warnings +from collections.abc import Callable, Mapping, Sequence +from typing import TYPE_CHECKING, Final + +from typing_extensions import TypeIs + +from zarr.core.chunk_grids import full_span_chunk_size +from zarr.errors import ZarrUserWarning + +if TYPE_CHECKING: + from zarr.core.common import JSON + +type ArrayDocument = Mapping[str, JSON] +type Upgrade = Callable[[ArrayDocument], ArrayDocument] + +RESAVE_HINT: Final = ( + "To store valid metadata, open the array writable and call `array.update_attributes({})`; " + "if a group holds consolidated metadata for the array, then also call " + "`zarr.consolidate_metadata` on that group." +) + + +def _warn(message: str) -> None: + # The synchronous API parses metadata on zarr's IO thread, whose stack holds no + # user code, so the warning points at the upgrade on every path. + warnings.warn(f"{message} {RESAVE_HINT}", ZarrUserWarning, stacklevel=2) + + +def _is_int_list(value: object) -> TypeIs[list[int] | tuple[int, ...]]: + """Whether `value` is a JSON array of integers (JSON `false` and `true` count).""" + return isinstance(value, list | tuple) and all(isinstance(v, int) for v in value) + + +def _read_invalid_chunk_sizes(chunk_shape: JSON, shape: JSON, unit: JSON) -> list[int] | None: + """Read a regular chunk shape whose chunk sizes include 0, JSON `false` or JSON `true`. + + 0 and `false` are read as one chunk spanning the axis, a multiple of `unit` on that + axis (the inner chunk shape of a sharded array); `true` is read as 1. Returns `None` + when there is nothing to read this way: anything else that is invalid is left for + the metadata constructors to reject. + """ + if not (_is_int_list(chunk_shape) and _is_int_list(shape)) or len(chunk_shape) != len(shape): + return None + invalid_axes = [ + axis for axis, size in enumerate(chunk_shape) if size == 0 or isinstance(size, bool) + ] + if not invalid_axes: + return None + units = ( + list(unit) + if _is_int_list(unit) and len(unit) == len(shape) and all(u >= 1 for u in unit) + else [1] * len(shape) + ) + upgraded = [ + full_span_chunk_size(extent, int(u)) if size == 0 else int(size) + for size, extent, u in zip(chunk_shape, shape, units, strict=True) + ] + readings = "; ".join( + f"{json.dumps(chunk_shape[axis])} on axis {axis} as " + + ("1" if chunk_shape[axis] else f"one chunk spanning the axis ({upgraded[axis]})") + for axis in invalid_axes + ) + message = ( + f"The stored chunk shape {json.dumps(list(chunk_shape))} is invalid: chunk sizes " + f"must be integers of at least 1. It is read as {upgraded}, reading {readings}." + ) + if any(chunk_shape[axis] == 0 and shape[axis] > 0 for axis in invalid_axes): + message += ( + " No chunk can be stored under a chunk size of 0, so the array holds only its " + "fill value." + ) + _warn(message) + return upgraded + + +def _invalid_chunk_sizes_v2(doc: ArrayDocument) -> ArrayDocument: + chunks = _read_invalid_chunk_sizes(doc.get("chunks"), doc.get("shape"), None) + return doc if chunks is None else {**doc, "chunks": chunks} + + +def _sharding_chunk_shape(codecs: JSON) -> JSON: + """The inner chunk shape of a sharding codec in a Zarr format 3 codec list, if any.""" + if isinstance(codecs, Sequence) and not isinstance(codecs, str): + for codec in codecs: + if isinstance(codec, Mapping) and codec.get("name") == "sharding_indexed": + configuration = codec.get("configuration") + if isinstance(configuration, Mapping): + return configuration.get("chunk_shape") + return None + + +def _invalid_chunk_sizes_v3(doc: ArrayDocument) -> ArrayDocument: + grid = doc.get("chunk_grid") + if not isinstance(grid, Mapping) or grid.get("name") != "regular": + return doc + configuration = grid.get("configuration") + if not isinstance(configuration, Mapping): + return doc + chunk_shape = _read_invalid_chunk_sizes( + configuration.get("chunk_shape"), + doc.get("shape"), + _sharding_chunk_shape(doc.get("codecs")), + ) + if chunk_shape is None: + return doc + return { + **doc, + "chunk_grid": {**grid, "configuration": {**configuration, "chunk_shape": chunk_shape}}, + } + + +V2_ARRAY_UPGRADES: Final[tuple[Upgrade, ...]] = (_invalid_chunk_sizes_v2,) +V3_ARRAY_UPGRADES: Final[tuple[Upgrade, ...]] = (_invalid_chunk_sizes_v3,) + + +def upgrade_array_document(doc: ArrayDocument, upgrades: Sequence[Upgrade]) -> ArrayDocument: + """Apply `upgrades` to a stored array metadata document, in order.""" + for upgrade in upgrades: + doc = upgrade(doc) + return doc diff --git a/src/zarr/core/metadata/v2.py b/src/zarr/core/metadata/v2.py index 3ae2a8bd96..0eb28efb11 100644 --- a/src/zarr/core/metadata/v2.py +++ b/src/zarr/core/metadata/v2.py @@ -42,7 +42,8 @@ ) from zarr.core.config import config, parse_indexing_order from zarr.core.json_parse import parse_field -from zarr.core.metadata.common import parse_attributes, parse_stored_regular_chunk_shape +from zarr.core.metadata.common import parse_attributes, parse_chunk_edge +from zarr.core.metadata.upgrades import V2_ARRAY_UPGRADES, upgrade_array_document class ArrayV2MetadataDict(TypedDict): @@ -88,7 +89,7 @@ def __init__( Metadata for a Zarr format 2 array. """ shape_parsed = parse_shapelike(shape) - chunks_parsed = parse_stored_regular_chunk_shape(parse_shapelike(chunks), shape_parsed) + chunks_parsed = parse_chunks(chunks, shape_parsed) compressor_parsed = parse_compressor(compressor) order_parsed = parse_indexing_order(order) dimension_separator_parsed = parse_separator(dimension_separator) @@ -147,7 +148,7 @@ def to_buffer_dict(self, prototype: BufferPrototype) -> dict[str, Buffer]: @classmethod def from_dict(cls, data: dict[str, Any]) -> ArrayV2Metadata: - # Make a copy to protect the original from modification. + data = dict(upgrade_array_document(data, V2_ARRAY_UPGRADES)) _data = data.copy() # Check that the zarr_format attribute is correct. _ = parse_zarr_format(_data.pop("zarr_format")) @@ -320,6 +321,17 @@ def parse_compressor(data: object) -> Numcodec | None: raise ValueError(msg) +def parse_chunks(chunks: Iterable[int], shape: tuple[int, ...]) -> tuple[int, ...]: + """Check a chunk shape: one chunk edge length (an integer >= 1) per array axis.""" + chunks_parsed = tuple(parse_chunk_edge(size, axis) for axis, size in enumerate(chunks)) + if len(chunks_parsed) != len(shape): + raise ValueError( + f"The `shape` and `chunks` attributes must have the same length. " + f"`chunks` has length {len(chunks_parsed)}, but `shape` has length {len(shape)}." + ) + return chunks_parsed + + def get_object_codec_id(maybe_object_codecs: Sequence[JSON]) -> str | None: """ Inspect a sequence of codecs / filters for an "object codec", i.e. a codec diff --git a/src/zarr/core/metadata/v3.py b/src/zarr/core/metadata/v3.py index 1db659beb8..4a27d495b3 100644 --- a/src/zarr/core/metadata/v3.py +++ b/src/zarr/core/metadata/v3.py @@ -5,7 +5,6 @@ from dataclasses import dataclass, field, replace from typing import TYPE_CHECKING, Any, Final, Literal, NotRequired, TypeGuard, cast -import numpy as np from typing_extensions import TypedDict from zarr.abc.codec import ArrayArrayCodec, ArrayBytesCodec, BytesBytesCodec, Codec @@ -36,7 +35,8 @@ from zarr.core.dtype import VariableLengthUTF8, ZDType, get_data_type_from_json from zarr.core.dtype.common import check_dtype_spec_v3 from zarr.core.json_parse import parse_field -from zarr.core.metadata.common import parse_attributes, parse_stored_regular_chunk_shape +from zarr.core.metadata.common import parse_attributes, parse_chunk_edge +from zarr.core.metadata.upgrades import V3_ARRAY_UPGRADES, upgrade_array_document from zarr.errors import MetadataValidationError, NodeTypeValidationError from zarr.registry import get_codec_class @@ -235,16 +235,12 @@ def _validate_chunk_shapes( result: list[int | tuple[int, ...]] = [] for dim_idx, dim_spec in enumerate(chunk_shapes): if isinstance(dim_spec, int): - if dim_spec < 1: - raise ValueError( - f"Dimension {dim_idx}: integer chunk edge length must be >= 1, got {dim_spec}" - ) - result.append(dim_spec) + result.append(parse_chunk_edge(dim_spec, dim_idx)) else: edges = tuple(dim_spec) if not edges: raise ValueError(f"Dimension {dim_idx} has no chunk edges.") - bad = [i for i, e in enumerate(edges) if e < 1] + bad = [i for i, e in enumerate(edges) if isinstance(e, bool) or e < 1] if bad: raise ValueError( f"Dimension {dim_idx} has invalid edge lengths at indices {bad}: " @@ -477,45 +473,6 @@ class ArrayMetadataJSON_V3(TypedDict, extra_items=AllowedExtraField): # type: i } -def _is_regular_chunk_shape(value: object) -> TypeGuard[Sequence[int]]: - """Whether a stored `chunk_shape` is a regular chunk shape: a sequence of integers. - - JSON `false` counts, because `bool` is an integer type. - """ - return ( - isinstance(value, Sequence) - and not isinstance(value, str) - and all(isinstance(size, int | np.integer) for size in value) - ) - - -def _parse_stored_regular_chunk_grid( - chunk_grid: dict[str, JSON] | ChunkGridMetadata | NamedConfig[str, Any], - shape: tuple[int, ...], -) -> dict[str, JSON] | ChunkGridMetadata | NamedConfig[str, Any]: - """Check a stored regular chunk grid's chunk shape against the array shape. - - Only a `regular` grid whose `chunk_shape` is all integers is a regular chunk - shape, and only that is handed to `parse_stored_regular_chunk_shape`. - Anything else is not a regular chunk shape and is left for the chunk grid - parser: other grids define their own chunk semantics. This runs here rather - than in the grid parser because it needs the array shape, which chunk grid - metadata does not carry. - """ - if not isinstance(chunk_grid, Mapping) or chunk_grid.get("name") != "regular": - return chunk_grid - configuration = chunk_grid.get("configuration") - if not isinstance(configuration, Mapping): - return chunk_grid - chunk_shape = configuration.get("chunk_shape") - if not _is_regular_chunk_shape(chunk_shape): - return chunk_grid - parsed = parse_stored_regular_chunk_shape(chunk_shape, shape) - corrected: dict[str, Any] = dict(chunk_grid) - corrected["configuration"] = {**configuration, "chunk_shape": list(parsed)} - return corrected - - @dataclass(frozen=True, kw_only=True) class ArrayV3Metadata(Metadata): shape: tuple[int, ...] @@ -550,9 +507,7 @@ def __init__( """ shape_parsed = parse_shapelike(shape) - chunk_grid_parsed = parse_chunk_grid( - _parse_stored_regular_chunk_grid(chunk_grid, shape_parsed) - ) + chunk_grid_parsed = parse_chunk_grid(chunk_grid) chunk_key_encoding_parsed = parse_chunk_key_encoding(chunk_key_encoding) dimension_names_parsed = parse_dimension_names(dimension_names) # Note: relying on a type method is numpy-specific @@ -673,8 +628,8 @@ def to_buffer_dict(self, prototype: BufferPrototype) -> dict[str, Buffer]: @classmethod def from_dict(cls, data: dict[str, JSON]) -> Self: - # make a copy because we are modifying the dict - _data = data.copy() + # a new dict, because we are modifying it + _data = dict(upgrade_array_document(data, V3_ARRAY_UPGRADES)) # check that the zarr_format attribute is correct _ = parse_zarr_format(_data.pop("zarr_format")) diff --git a/tests/test_metadata/test_common.py b/tests/test_metadata/test_common.py deleted file mode 100644 index c3ed9416eb..0000000000 --- a/tests/test_metadata/test_common.py +++ /dev/null @@ -1,72 +0,0 @@ -"""Tests for metadata helpers shared by both Zarr formats.""" - -from __future__ import annotations - -import re -import warnings -from typing import Any - -import numpy as np -import pytest - -from zarr.core.metadata.common import parse_stored_regular_chunk_shape -from zarr.errors import ZarrUserWarning - - -@pytest.mark.parametrize( - ("chunk_shape", "shape", "expected", "warning"), - [ - ((4, 5), (10, 10), (4, 5), None), - ((1, 1), (0, 0), (1, 1), None), - ((0,), (0,), (1,), "of size 1"), - ((False,), (0,), (1,), "of size 1"), - ((np.int64(0),), (0,), (1,), "of size 1"), - ((4, 0), (4, 0), (4, 1), "Dimension 1"), - ((0, 0), (0, 0), (1, 1), "Dimension 0"), - ((0,), (5,), (5,), "5 elements along this axis hold only the fill value"), - ((4, 0), (4, 3), (4, 3), "3 elements along this axis hold only the fill value"), - ], - ids=[ - "valid", - "valid-on-empty-axes", - "legacy-zero", - "legacy-json-false", - "legacy-numpy-zero", - "only-zero-size-corrected", - "every-zero-size-corrected", - "legacy-zero-on-grown-axis", - "legacy-zero-on-grown-axis-2d", - ], -) -def test_parse_stored_regular_chunk_shape( - chunk_shape: tuple[Any, ...], - shape: tuple[int, ...], - expected: tuple[Any, ...], - warning: str | None, -) -> None: - """A valid chunk shape is returned as is; a chunk size of 0 is read as one - chunk spanning the axis, with a warning saying how to re-save and, on an axis - of positive length, that it holds only the fill value.""" - with warnings.catch_warnings(record=True) as record: - warnings.simplefilter("always") - parsed = parse_stored_regular_chunk_shape(chunk_shape, shape) - assert parsed == expected - messages = [str(w.message) for w in record if issubclass(w.category, ZarrUserWarning)] - if warning is None: - assert messages == [] - else: - assert any(re.search(warning, message) for message in messages) - for message in messages: - assert "update_attributes({})" in message - - -def test_parse_stored_regular_chunk_shape_rejects_dimension_mismatch() -> None: - """The chunk shape needs one entry per array axis.""" - with pytest.raises(ValueError, match="same number of dimensions"): - parse_stored_regular_chunk_shape((4,), (10, 10)) - - -def test_parse_stored_regular_chunk_shape_rejects_negative() -> None: - """A negative chunk size is rejected even on a zero-length axis.""" - with pytest.raises(ValueError, match="chunk edge length must be >= 1, got -1"): - parse_stored_regular_chunk_shape((-1,), (0,)) diff --git a/tests/test_metadata/test_upgrades.py b/tests/test_metadata/test_upgrades.py new file mode 100644 index 0000000000..7506bb6875 --- /dev/null +++ b/tests/test_metadata/test_upgrades.py @@ -0,0 +1,293 @@ +"""Tests for the upgrades that read invalid stored array metadata documents.""" + +from __future__ import annotations + +import json +import re +import warnings +from typing import TYPE_CHECKING, Any, Literal + +import numpy as np +import pytest + +import zarr +from zarr.core.metadata import ArrayV2Metadata, ArrayV3Metadata +from zarr.core.metadata.upgrades import ( + V2_ARRAY_UPGRADES, + V3_ARRAY_UPGRADES, + upgrade_array_document, +) +from zarr.core.metadata.v3 import RegularChunkGridMetadata +from zarr.dtype import Int16 +from zarr.errors import ZarrUserWarning + +if TYPE_CHECKING: + from pathlib import Path + + from zarr.core.common import JSON + + +def _v2_doc(shape: list[int], chunks: list[Any]) -> dict[str, JSON]: + return { + "zarr_format": 2, + "shape": shape, + "chunks": chunks, + "dtype": " dict[str, JSON]: + bytes_codec: dict[str, JSON] = {"name": "bytes", "configuration": {"endian": "little"}} + codecs: list[JSON] = [bytes_codec] + if inner is not None: + codecs = [ + { + "name": "sharding_indexed", + "configuration": { + "chunk_shape": inner, + "codecs": [bytes_codec], + "index_codecs": [bytes_codec, {"name": "crc32c"}], + "index_location": "end", + }, + } + ] + return { + "zarr_format": 3, + "node_type": "array", + "shape": shape, + "data_type": "int16", + "chunk_grid": {"name": "regular", "configuration": {"chunk_shape": chunk_shape}}, + "chunk_key_encoding": {"name": "default", "configuration": {"separator": "/"}}, + "fill_value": 0, + "codecs": codecs, + } + + +def _stored_chunks(doc: dict[str, Any]) -> Any: + return ( + doc["chunks"] + if doc["zarr_format"] == 2 + else doc["chunk_grid"]["configuration"]["chunk_shape"] + ) + + +@pytest.mark.parametrize( + ("doc", "expected", "warning"), + [ + (_v2_doc([10, 10], [4, 5]), [4, 5], None), + (_v3_doc([0, 0], [1, 1]), [1, 1], None), + (_v3_doc([10], [4], inner=[2]), [4], None), + (_v2_doc([0, 4], [0, 4]), [1, 4], "0 on axis 0 as one chunk spanning the axis"), + (_v3_doc([0], [False]), [1], "false on axis 0 as one chunk spanning the axis"), + (_v2_doc([5], [True]), [1], "true on axis 0 as 1"), + (_v3_doc([5, 4], [True, 4]), [1, 4], "true on axis 0 as 1"), + (_v2_doc([3], [0]), [3], "holds only its fill value"), + (_v3_doc([4, 3], [4, 0]), [4, 3], "holds only its fill value"), + (_v3_doc([0], [0], inner=[4]), [4], "spanning the axis \\(4\\)"), + (_v3_doc([10], [0], inner=[4]), [12], "spanning the axis \\(12\\)"), + (_v3_doc([0, 3], [0, 3], inner=[2, 3]), [2, 3], "spanning the axis \\(2\\)"), + ], + ids=[ + "v2-valid", + "v3-valid-empty-axes", + "v3-valid-sharded", + "v2-zero-empty-axis", + "v3-false-empty-axis", + "v2-true", + "v3-true", + "v2-zero-grown-axis", + "v3-zero-grown-axis", + "v3-sharded-zero-empty-axis", + "v3-sharded-zero-grown-axis", + "v3-sharded-zero-2d", + ], +) +def test_upgrade_array_document( + doc: dict[str, JSON], expected: list[int], warning: str | None +) -> None: + """Valid documents pass unchanged and silently. A stored chunk size of 0 or `false` + is read as one chunk spanning the axis (a multiple of the inner chunk when sharded) + and `true` as 1, with one warning that says how to re-save.""" + upgrades = V2_ARRAY_UPGRADES if doc["zarr_format"] == 2 else V3_ARRAY_UPGRADES + with warnings.catch_warnings(record=True) as record: + warnings.simplefilter("always") + upgraded = upgrade_array_document(doc, upgrades) + assert _stored_chunks(dict(upgraded)) == expected + assert {k: v for k, v in upgraded.items() if k not in ("chunks", "chunk_grid")} == { + k: v for k, v in doc.items() if k not in ("chunks", "chunk_grid") + } + messages = [str(w.message) for w in record] + if warning is None: + assert upgraded is doc + assert messages == [] + else: + assert len(messages) == 1 + assert re.search(warning, messages[0]) + assert "update_attributes({})" in messages[0] + assert "zarr.consolidate_metadata" in messages[0] + metadata_cls = ArrayV2Metadata if doc["zarr_format"] == 2 else ArrayV3Metadata + with warnings.catch_warnings(): + warnings.simplefilter("ignore", ZarrUserWarning) + metadata_cls.from_dict(dict(doc)) + + +@pytest.mark.parametrize("doc", [_v2_doc([4], [-1]), _v3_doc([4], [-1])], ids=["v2", "v3"]) +def test_stored_negative_chunk_size_rejected(doc: dict[str, JSON]) -> None: + metadata_cls = ArrayV2Metadata if doc["zarr_format"] == 2 else ArrayV3Metadata + with pytest.raises(ValueError, match="chunk edge length must be an integer >= 1, got -1"): + metadata_cls.from_dict(doc) + + +@pytest.mark.parametrize("doc", [_v2_doc([4, 4], [0]), _v3_doc([4, 4], [0])], ids=["v2", "v3"]) +def test_stored_chunk_shape_ndim_mismatch_rejected(doc: dict[str, JSON]) -> None: + """A chunk shape with the wrong number of axes is not upgraded, only rejected.""" + metadata_cls = ArrayV2Metadata if doc["zarr_format"] == 2 else ArrayV3Metadata + with warnings.catch_warnings(): + warnings.simplefilter("error", ZarrUserWarning) + with pytest.raises(ValueError, match="chunk edge length|same length|same number"): + metadata_cls.from_dict(doc) + + +@pytest.mark.parametrize("size", [0, True], ids=["zero", "true"]) +def test_constructor_rejects_invalid_chunk_size(size: int) -> None: + """Metadata built in code is strict: 0 and `True` are rejected, without a warning.""" + with warnings.catch_warnings(): + warnings.simplefilter("error", ZarrUserWarning) + with pytest.raises(ValueError, match=f"got {size!r}"): + RegularChunkGridMetadata(chunk_shape=(size,)) + with pytest.raises(ValueError, match=f"got {size!r}"): + ArrayV2Metadata( + shape=(0,), + chunks=(size,), + dtype=Int16(), + fill_value=0, + order="C", + ) + + +def _rewrite_doc(path: Path, zarr_format: Literal[2, 3], edit: Any) -> None: + doc_path = path / (".zarray" if zarr_format == 2 else "zarr.json") + doc = json.loads(doc_path.read_text()) + edit(doc) + doc_path.write_text(json.dumps(doc)) + + +@pytest.mark.parametrize( + ("zarr_format", "shape", "stored", "inner", "expected"), + [ + (2, (0, 4), [0, 4], None, (1, 4)), + (2, (3,), [0], None, (3,)), + (2, (5,), [True], None, (1,)), + (3, (0,), [False], None, (1,)), + (3, (3,), [0], None, (3,)), + (3, (5,), [True], None, (1,)), + (3, (0,), [0], (4,), (4,)), + (3, (10,), [0], (4,), (12,)), + (3, (0, 3), [0, 3], (2, 3), (2, 3)), + ], + ids=[ + "v2-empty-2d", + "v2-grown", + "v2-true", + "v3-false-empty", + "v3-grown", + "v3-true", + "v3-sharded-empty", + "v3-sharded-grown", + "v3-sharded-2d", + ], +) +def test_legacy_chunk_size_round_trip( + tmp_path: Path, + zarr_format: Literal[2, 3], + shape: tuple[int, ...], + stored: list[Any], + inner: tuple[int, ...] | None, + expected: tuple[int, ...], +) -> None: + """A store whose metadata holds a chunk size written by older software opens with a + warning, reads and appends under the upgraded grid, and re-saves valid metadata.""" + path = tmp_path / "legacy.zarr" + arr = zarr.create_array( + store=path, + shape=shape, + chunks=inner or expected, + shards=expected if inner else None, + dtype="int16", + fill_value=0, + zarr_format=zarr_format, + ) + data = np.arange(np.prod(shape), dtype="int16").reshape(shape) + arr[...] = data + + def store_legacy(doc: dict[str, Any]) -> None: + if zarr_format == 2: + doc["chunks"] = stored + else: + doc["chunk_grid"]["configuration"]["chunk_shape"] = stored + + _rewrite_doc(path, zarr_format, store_legacy) + + with pytest.warns(ZarrUserWarning, match="is read as"): + arr = zarr.open_array(store=path, mode="a") + assert (arr.shards or arr.chunks) == expected + np.testing.assert_array_equal(arr[...], data) + + block = np.full((2, *shape[1:]), 7, dtype="int16") + arr.append(block) + arr.update_attributes({}) + with warnings.catch_warnings(): + warnings.simplefilter("error", ZarrUserWarning) + reopened = zarr.open_array(store=path) + np.testing.assert_array_equal(reopened[...], np.concatenate([data, block])) + + +@pytest.mark.filterwarnings("ignore:Consolidated metadata is currently not part:UserWarning") +@pytest.mark.parametrize("zarr_format", [2, 3]) +def test_legacy_chunk_size_consolidated(tmp_path: Path, zarr_format: Literal[2, 3]) -> None: + """Consolidated metadata goes through the same upgrade; re-saving the array and + consolidating again leaves a group that opens without a warning.""" + path = tmp_path / "group.zarr" + group = zarr.open_group(path, mode="w", zarr_format=zarr_format) + group.create_array("a", shape=(0,), chunks=(1,), dtype="int32") + zarr.consolidate_metadata(path) + + # What zarr-python wrote for `chunks=(0,)` on an empty array, in both copies. + if zarr_format == 2: + _rewrite_doc(path / "a", 2, lambda doc: doc.update(chunks=[0])) + zmetadata = json.loads((path / ".zmetadata").read_text()) + zmetadata["metadata"]["a/.zarray"]["chunks"] = [0] + (path / ".zmetadata").write_text(json.dumps(zmetadata)) + else: + _rewrite_doc( + path / "a", 3, lambda doc: doc["chunk_grid"]["configuration"].update(chunk_shape=[0]) + ) + _rewrite_doc( + path, + 3, + lambda doc: doc["consolidated_metadata"]["metadata"]["a"]["chunk_grid"][ + "configuration" + ].update(chunk_shape=[0]), + ) + + with pytest.warns(ZarrUserWarning, match="zarr.consolidate_metadata"): + group = zarr.open_group(path, mode="r+") + array = group["a"] + assert isinstance(array, zarr.Array) + assert array.chunks == (1,) + + array.update_attributes({}) + zarr.consolidate_metadata(path) + with warnings.catch_warnings(): + warnings.simplefilter("error", ZarrUserWarning) + warnings.filterwarnings("ignore", "Consolidated metadata is currently not part") + for use_consolidated in (True, False): + reopened = zarr.open_group(path, mode="r", use_consolidated=use_consolidated)["a"] + assert isinstance(reopened, zarr.Array) + assert reopened.chunks == (1,) From f6f5c7b5cd97500987220f74bd20032b00e4e0f0 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Fri, 25 Sep 2026 22:22:48 +0200 Subject: [PATCH 23/76] docs: rewrite the 4334 changelog fragment and trim restating docstrings The fragment now covers the full-span shard rule, strict constructors, and the stored-document upgrade (including JSON true) in two short paragraphs. Assisted-by: ClaudeCode:claude-opus-5-5 Co-Authored-By: Claude Opus 5.5 --- changes/4334.bugfix.md | 4 ++-- src/zarr/core/chunk_grids.py | 6 ++---- tests/test_unified_chunk_grid.py | 14 ++------------ 3 files changed, 6 insertions(+), 18 deletions(-) diff --git a/changes/4334.bugfix.md b/changes/4334.bugfix.md index 68ae5ade9e..26228e668f 100644 --- a/changes/4334.bugfix.md +++ b/changes/4334.bugfix.md @@ -1,3 +1,3 @@ -Chunk-grid normalization now shares a helper that selects chunk size 1 for a zero-length axis when inferring a full-span chunk size. Chunk edge lengths remain positive while array extents may be zero. `FixedDimension(size=0, ...)` now raises `ValueError`. Rectilinear chunk grids can be created on a zero-length dimension with a non-empty list of positive chunk sizes; these sizes are retained for subsequent growth. +A chunk edge length is now always at least 1, while an array extent may be 0. Every spelling of one chunk spanning an axis (`chunks=-1`, `chunks=False`, `chunks="auto"`, `shards=-1`, `shards=False`) gives chunk size 1 on a zero-length axis, and a shard spanning an axis is a multiple of the inner chunk size, so `shards=-1` no longer fails when the axis length is not a multiple of it. Array metadata built in code with a chunk size of 0 or `True` now raises, as does `FixedDimension(size=0, ...)`. Rectilinear chunk grids can be created on a zero-length dimension with a non-empty list of positive chunk sizes, which are kept for later growth. -A stored chunk size of 0 is read as one chunk spanning the axis, with a `ZarrUserWarning`, in both Zarr format 2 (`chunks`) and Zarr format 3 (a regular grid's `chunk_shape`, including the JSON `false` written by zarr-python 3.0 for `chunks=False`). zarr-python wrote such metadata for arrays created with a zero-length axis until 3.4 — Zarr format 2 through 3.3 and by zarr-python 2.x, Zarr format 3 in 3.0 and 3.1. Those versions could then grow the axis without storing any chunk: appending to one of these Zarr format 2 arrays in 3.4.0 reported success while the appended data read back as the fill value, and 3.1 recorded a Zarr format 3 resize or failed append. Such grown arrays open too, and the warning says that data written to the grown axis was not saved. A negative chunk size is rejected. The warning explains how to store a corrected chunk size: open the array writable and call `array.update_attributes({})`. +Stored metadata with a regular chunk size of 0 or JSON `false`, as zarr-python wrote for arrays created with a zero-length axis until 3.4, now opens with a `ZarrUserWarning` and is read as one chunk spanning the axis (a multiple of the inner chunk size for sharded arrays, which previously failed to open), and JSON `true`, as zarr-python 3.0 and 3.2 wrote for `chunks=(True,)`, is read as 1. The warning says how to store valid metadata: `array.update_attributes({})`, then `zarr.consolidate_metadata` if the metadata is consolidated. diff --git a/src/zarr/core/chunk_grids.py b/src/zarr/core/chunk_grids.py index d5d61ee010..17b3225725 100644 --- a/src/zarr/core/chunk_grids.py +++ b/src/zarr/core/chunk_grids.py @@ -895,10 +895,8 @@ def _guess_num_chunks_per_axis_shard( In other words the shard would be a (2,2,2) grid of (2,2,2) chunks i.e., prod(chunk_shape) * (returned_val ** len(chunk_shape)) * item_size = 256 bytes. - Degenerate inputs — a 0-dimensional chunk shape, or a zero-byte chunk (``item_size`` - of 0; chunk edge lengths themselves are always at least 1) — return 1, as the - search loop's stopping conditions can never be met. A zero-length *array* axis - needs no special case: the array-bound check fails immediately for it. + Degenerate inputs — a 0-dimensional chunk shape, or a zero-byte chunk — return 1, + as the search loop's stopping conditions can never be met. Parameters ---------- diff --git a/tests/test_unified_chunk_grid.py b/tests/test_unified_chunk_grid.py index 2cf3f7cc10..6e2379e2fd 100644 --- a/tests/test_unified_chunk_grid.py +++ b/tests/test_unified_chunk_grid.py @@ -199,12 +199,7 @@ def test_fixed_dimension_indices_to_chunks() -> None: ids=["negative-size", "zero-size", "zero-size-zero-extent", "negative-extent"], ) def test_fixed_dimension_rejects_invalid(size: int, extent: int, match: str) -> None: - """FixedDimension raises ValueError for a size below 1 or a negative extent. - - A chunk edge length of 0 is never valid, whatever the extent: the metadata layer - requires every chunk edge length to be >= 1, and the in-memory model enforces the - same invariant so the two can never disagree. - """ + """FixedDimension raises ValueError for a size below 1 or a negative extent.""" with pytest.raises(ValueError, match=match): FixedDimension(size=size, extent=extent) @@ -1433,12 +1428,7 @@ def test_edge_case_chunk_grid_boundary_shape() -> None: @pytest.mark.parametrize("size", [1, 10], ids=["size-1", "size-10"]) def test_fixed_dimension_zero_extent(size: int) -> None: - """A zero-length axis has zero chunks and behaves like an empty grid. - - The extent may be 0 even though the chunk size may not: `ceildiv(0, size)` is 0, - so there is nothing to look up, and the vectorized index mapping of an empty index - array is empty. - """ + """A zero-length axis has zero chunks and behaves like an empty grid.""" d = FixedDimension(size=size, extent=0) assert d.nchunks == 0 assert d.ngridcells == 0 From f87335852b1677bf78bcbb70ad767a9e3112621a Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Fri, 25 Sep 2026 22:28:18 +0200 Subject: [PATCH 24/76] test: model exactly what the store holds in the array lifecycle state machine The model now tracks cells beyond the array's shape: a shrinking resize deletes exactly the chunks outside the new grid, kept chunks keep their out-of-bounds cells, and a write covering every in-bounds cell of an unsharded chunk resets its out-of-bounds cells to the fill value. A resize that never deletes chunks now fails the test. Sharding is its own chunk spelling, a stored chunk size of 0 applies to sharded arrays too (in the outer grid), re-saving metadata only runs when the store warns, and the test takes its settings from the repository's hypothesis profiles under the slow_hypothesis marker, like the other stateful tests. Assisted-by: ClaudeCode:claude-opus-5-5 Co-Authored-By: Claude Opus 5.5 --- tests/test_array_stateful.py | 201 +++++++++++++++++++---------------- 1 file changed, 111 insertions(+), 90 deletions(-) diff --git a/tests/test_array_stateful.py b/tests/test_array_stateful.py index a5a6627a5d..6e9bea8f8b 100644 --- a/tests/test_array_stateful.py +++ b/tests/test_array_stateful.py @@ -1,18 +1,22 @@ """A stateful test of one array's life: create, append, resize, write, reopen. -The model is a NumPy array, not a second zarr array, so a bug in zarr's chunk -grid logic cannot hide by being made on both sides. Zero-length axes are drawn -on purpose, both at creation and by resizing and appending, and so are the -stored chunk sizes of 0 that zarr-python wrote for empty arrays before 3.4. - -`resize` deletes only the chunks that fall entirely outside the new shape, so -cells cut off by a shrink can come back with their old values when the axis -grows again. The model does not encode that chunk-level behaviour: a cell cut -off and brought back is unknown until it is written. +The model is a NumPy array of what the store holds, not a second zarr array, so a bug +in zarr's chunk grid logic cannot hide by being made on both sides. Zero-length axes +are drawn on purpose, both at creation and by resizing and appending, and so are +stored chunk sizes of 0. + +The model tracks cells beyond the array's shape too, because chunks do: a shrinking +`resize` deletes exactly the chunks outside the new grid, a chunk it keeps keeps its +cells beyond the new shape (and they come back if the array grows again), and a write +that covers every in-bounds cell of an unsharded chunk rewrites the whole chunk, +resetting its cells beyond the shape to the fill value. A sharded array rewrites a +shard through its inner chunks, each judged against the shard rather than the array +shape, so a write never resets cells beyond the shape. """ from __future__ import annotations +import itertools import json import warnings from typing import Any, Literal @@ -21,30 +25,37 @@ import hypothesis.strategies as st import numpy as np import pytest -from hypothesis import event, note, settings +from hypothesis import event, note from hypothesis.stateful import ( RuleBasedStateMachine, initialize, invariant, + precondition, rule, ) import zarr from zarr.core.buffer import cpu, default_buffer_prototype +from zarr.core.chunk_grids import ChunkGrid from zarr.core.sync import sync from zarr.errors import ZarrUserWarning from zarr.storage import MemoryStore -pytestmark = pytest.mark.filterwarnings( - "ignore::zarr.core.dtype.common.UnstableSpecificationWarning" -) +pytestmark = [ + pytest.mark.slow_hypothesis, + pytest.mark.filterwarnings("ignore::zarr.core.dtype.common.UnstableSpecificationWarning"), +] DTYPE = np.dtype("int16") MAX_SIDE = 6 def _rectilinear_dim(extent: int) -> st.SearchStrategy[int | list[int]]: - """A bare step, or an edge list covering `extent` (any edges for extent 0).""" + """A bare step, or an edge list covering `extent` (any edges for extent 0). + + A small local copy of what `zarr.testing.strategies` draws for rectilinear + declarations, so this test does not depend on that module's experimental API. + """ steps = st.integers(min_value=1, max_value=MAX_SIDE) if extent == 0: return steps | st.lists(steps, min_size=1, max_size=3) @@ -64,39 +75,33 @@ def __init__(self) -> None: self._rectilinear.__enter__() self.store = MemoryStore() self.path = "a" - self.model: np.ndarray[Any, np.dtype[np.int16]] = np.zeros((0,), dtype=DTYPE) - # Cells whose value the model knows; see the module docstring. - self.known: np.ndarray[Any, np.dtype[np.bool_]] = np.ones((0,), dtype=bool) - # Every shape the array has had, to find cells a resize brings back. - self.past_shapes: list[tuple[int, ...]] = [] + self.shape: tuple[int, ...] = (0,) self.fill = 0 - # A legacy store warns until its metadata is re-saved. + # What the store holds, indexed like the array and extending past its shape. + self.stored: np.ndarray[Any, np.dtype[np.int16]] = np.zeros((0,), dtype=DTYPE) + # A store with an invalid stored chunk size warns until its metadata is re-saved. self.expect_open_warning = False # -------------------------------------------------------------- creation @initialize(data=st.data()) def create(self, data: st.DataObject) -> None: - zarr_format: Literal[2, 3] = data.draw(st.sampled_from([2, 3]), label="zarr_format") + zarr_format: Literal[2, 3] = data.draw(st.sampled_from([3, 2]), label="zarr_format") shape = data.draw( npst.array_shapes(min_dims=1, max_dims=3, min_side=0, max_side=MAX_SIDE), label="shape", ) self.fill = data.draw(st.integers(-3, 3), label="fill_value") # sampled_from favours early entries; the less common spellings go first. - spellings = ["ints", "legacy-zero", "-1", "False", "auto"] + spellings = ["ints", "-1", "False", "auto"] if zarr_format == 3: - spellings.insert(1, "rectilinear") + spellings[:0] = ["sharded", "rectilinear"] spelling = data.draw(st.sampled_from(spellings), label="chunk spelling") event(f"chunks: {spelling}") - if any(s == 0 for s in shape): - event("created with a zero-length axis") chunks: Any shards: Any = None - if spelling == "-1": - chunks = -1 - elif spelling == "False": - chunks = False + if spelling in ("-1", "False"): + chunks = {"-1": -1, "False": False}[spelling] elif spelling == "auto": chunks = "auto" elif spelling == "rectilinear": @@ -104,14 +109,9 @@ def create(self, data: st.DataObject) -> None: if not any(isinstance(c, list) for c in chunks): chunks[0] = [chunks[0]] if shape[0] == 0 else [shape[0]] else: - chunks = tuple(data.draw(st.integers(1, 4)) for _ in shape) - if ( - spelling == "ints" - and zarr_format == 3 - and data.draw(st.booleans(), label="sharded") - ): - shards = tuple(c * data.draw(st.integers(1, 2)) for c in chunks) - event("sharded") + chunks = tuple(data.draw(st.integers(1, 3)) for _ in shape) + if spelling == "sharded": + shards = tuple(c * data.draw(st.integers(1, 3)) for c in chunks) note(f"create {shape=} {chunks=} {shards=} {zarr_format=} fill={self.fill}") zarr.create_array( self.store, @@ -123,14 +123,13 @@ def create(self, data: st.DataObject) -> None: fill_value=self.fill, zarr_format=zarr_format, ) - self.model = np.full(shape, self.fill, dtype=DTYPE) - self.known = np.ones(shape, dtype=bool) - self.past_shapes = [shape] - - if spelling == "legacy-zero": - # What zarr-python wrote before 3.4 for an array created with a - # zero-length axis and one chunk spanning it; older releases could - # grow that axis without storing a chunk, so any extent is possible. + self.shape = shape + self.stored = np.full(shape, self.fill, dtype=DTYPE) + + if spelling != "rectilinear" and data.draw(st.booleans(), label="legacy zero"): + # A stored chunk size of 0, as zarr-python wrote for arrays created with a + # zero-length axis; older releases could then grow the axis without storing + # a chunk, so any extent is possible. Sharded arrays store it in the outer grid. zero_axes = data.draw( st.lists(st.integers(0, len(shape) - 1), min_size=1, unique=True), label="axes stored with chunk size 0", @@ -138,8 +137,7 @@ def create(self, data: st.DataObject) -> None: stored_zero = data.draw(st.sampled_from([0, False]), label="stored zero") self._rewrite_stored_chunks(zarr_format, zero_axes, stored_zero) self.expect_open_warning = True - if any(shape[i] > 0 for i in zero_axes): - event("legacy zero chunk on a grown axis") + event("legacy zero chunk size") def _rewrite_stored_chunks( self, zarr_format: Literal[2, 3], axes: list[int], value: Any @@ -163,26 +161,68 @@ def _open(self) -> zarr.Array[Any]: assert warned is self.expect_open_warning, [str(w.message) for w in record] return arr + # ----------------------------------------------------------------- model + def _cover(self, shape: tuple[int, ...]) -> None: + """Grow `stored` with the fill value so that it covers `shape`.""" + pad = [(0, max(0, n - s)) for n, s in zip(shape, self.stored.shape, strict=True)] + if any(after for _, after in pad): + self.stored = np.pad(self.stored, pad, constant_values=self.fill) + + def _model_resize(self, grid: ChunkGrid, new_shape: tuple[int, ...]) -> None: + """Delete exactly the chunks of `grid` outside the grid for `new_shape`.""" + kept = [] + for dim, old, new in zip(grid.dimensions, self.shape, new_shape, strict=True): + if new >= old: + kept.append(slice(None)) + elif new == 0: + kept.append(slice(0, 0)) + else: + last = dim.index_to_chunk(new - 1) + kept.append(slice(0, dim.chunk_offset(last) + dim.chunk_size(last))) + stored = np.full_like(self.stored, self.fill) + stored[tuple(kept)] = self.stored[tuple(kept)] + self.stored = stored + self.shape = new_shape + self._cover(new_shape) + + def _model_write(self, arr: zarr.Array[Any], region: tuple[slice, ...], values: Any) -> None: + """Write `values`; an unsharded chunk whose in-bounds cells are all written is + rewritten whole.""" + grid = ChunkGrid.from_metadata(arr.metadata) + if arr.shards is None and all(r.stop > r.start for r in region): + # Per axis, each chunk the region touches: (start, stop, all in-bounds cells written). + per_axis: list[list[tuple[int, int, bool]]] = [] + for dim, r, extent in zip(grid.dimensions, region, self.shape, strict=True): + spans = [] + for c in range(dim.index_to_chunk(r.start), dim.index_to_chunk(r.stop - 1) + 1): + lo = dim.chunk_offset(c) + hi = lo + dim.chunk_size(c) + spans.append((lo, hi, r.start <= lo and r.stop >= min(hi, extent))) + per_axis.append(spans) + self._cover(tuple(max(hi for _, hi, _ in axis) for axis in per_axis)) + for combo in itertools.product(*per_axis): + if all(complete for *_, complete in combo): + self.stored[tuple(slice(lo, hi) for lo, hi, _ in combo)] = self.fill + self.stored[region] = values + # ----------------------------------------------------------------- rules @rule(data=st.data()) def append(self, data: st.DataObject) -> None: arr = self._open() - axis = data.draw(st.integers(0, self.model.ndim - 1), label="axis") - block_shape = list(self.model.shape) + axis = data.draw(st.integers(0, len(self.shape) - 1), label="axis") + block_shape = list(self.shape) block_shape[axis] = data.draw(st.integers(0, 4), label="rows") block = data.draw(npst.arrays(DTYPE, tuple(block_shape)), label="block") - note(f"append {block.shape} along {axis} to {self.model.shape}") - if self.model.shape[axis] == 0: + note(f"append {block.shape} along {axis} to {self.shape}") + if self.shape[axis] == 0 and block.shape[axis]: event("append to a zero-length axis") + old_extent = self.shape[axis] arr.append(block, axis=axis) - self._reshape_model(arr.shape) - tail = tuple( - slice(-block.shape[axis], None) if i == axis and block.shape[axis] else slice(None) - for i in range(self.model.ndim) + self._model_resize(ChunkGrid.from_metadata(arr.metadata), arr.shape) + region = tuple( + slice(old_extent, s) if i == axis else slice(0, s) for i, s in enumerate(arr.shape) ) - if block.shape[axis]: - self.model[tail] = block - self.known[tail] = True + self._model_write(arr, region, block) # Growing the array rewrites its metadata, which stores any correction. self.expect_open_warning = False @@ -190,49 +230,31 @@ def append(self, data: st.DataObject) -> None: def resize(self, data: st.DataObject) -> None: arr = self._open() new_shape = data.draw( - st.tuples(*(st.integers(0, MAX_SIDE) for _ in self.model.shape)), label="new shape" + st.tuples(*(st.integers(0, MAX_SIDE) for _ in self.shape)), label="new shape" ) - note(f"resize {self.model.shape} -> {new_shape}") - if any(o == 0 and n > 0 for o, n in zip(self.model.shape, new_shape, strict=True)): - event("resize grows a zero-length axis") + note(f"resize {self.shape} -> {new_shape}") + grid = ChunkGrid.from_metadata(arr.metadata) arr.resize(new_shape) - self._reshape_model(new_shape) + self._model_resize(grid, new_shape) self.expect_open_warning = False - def _reshape_model(self, new_shape: tuple[int, ...]) -> None: - """Resize the model: kept cells keep their values, new cells hold the fill - value, and cells that an earlier shape held but the current one cut off - become unknown.""" - overlap = tuple( - slice(0, min(o, n)) for o, n in zip(self.model.shape, new_shape, strict=True) - ) - model = np.full(new_shape, self.fill, dtype=DTYPE) - known = np.ones(new_shape, dtype=bool) - for past in self.past_shapes: - known[tuple(slice(0, min(p, n)) for p, n in zip(past, new_shape, strict=True))] = False - model[overlap] = self.model[overlap] - known[overlap] = self.known[overlap] - if not known.all(): - event("resize brings back cells cut off earlier") - self.model, self.known = model, known - self.past_shapes.append(tuple(new_shape)) - @rule(data=st.data()) def write(self, data: st.DataObject) -> None: arr = self._open() region = tuple( slice(*sorted(data.draw(st.tuples(st.integers(0, s), st.integers(0, s))))) - for s in self.model.shape + for s in self.shape ) - values = data.draw(npst.arrays(DTYPE, self.model[region].shape), label="values") + shape = tuple(r.stop - r.start for r in region) + values = data.draw(npst.arrays(DTYPE, shape), label="values") note(f"write {region}") arr[region] = values - self.model[region] = values - self.known[region] = True + self._model_write(arr, region, values) + @precondition(lambda self: self.expect_open_warning) @rule() def resave_metadata(self) -> None: - """What the legacy warning tells users to do.""" + """What the warning for an invalid stored chunk size tells users to do.""" self._open().update_attributes({}) self.expect_open_warning = False @@ -243,10 +265,9 @@ def teardown(self) -> None: @invariant() def matches_model(self) -> None: arr = self._open() - assert arr.shape == self.model.shape - actual = np.asarray(arr[...]) - np.testing.assert_array_equal(actual[self.known], self.model[self.known]) + assert arr.shape == self.shape + expected = self.stored[tuple(slice(0, s) for s in self.shape)] + np.testing.assert_array_equal(np.asarray(arr[...]), expected) -ArrayLifecycle.TestCase.settings = settings(max_examples=200, stateful_step_count=12, deadline=None) TestArrayLifecycle = ArrayLifecycle.TestCase From ae7c2951a4a3e713634766f090fa1799bf63aa3f Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Fri, 25 Sep 2026 22:46:45 +0200 Subject: [PATCH 25/76] refactor(metadata): warn about an upgraded document only once it validates An upgrade now returns the upgraded document and how it read it, and `from_dict` warns with those readings after the metadata constructor accepts the upgraded document. A stored document that is still invalid after an upgrade raises its own error instead of first warning how it was read. Assisted-by: ClaudeCode:claude-opus-5-5 Co-Authored-By: Claude Opus 5.5 --- src/zarr/core/metadata/upgrades.py | 76 ++++++++++++++++++---------- src/zarr/core/metadata/v2.py | 9 ++-- src/zarr/core/metadata/v3.py | 14 +++-- tests/test_metadata/test_upgrades.py | 30 ++++++++--- 4 files changed, 89 insertions(+), 40 deletions(-) diff --git a/src/zarr/core/metadata/upgrades.py b/src/zarr/core/metadata/upgrades.py index 933b4c53e2..e15ffd30fe 100644 --- a/src/zarr/core/metadata/upgrades.py +++ b/src/zarr/core/metadata/upgrades.py @@ -2,9 +2,11 @@ This is the only place invalid metadata is read leniently; the metadata constructors are strict. An upgrade maps a stored array metadata document (parsed JSON) to a valid -one and warns once when it changes anything. `ArrayV2Metadata.from_dict` and +one and says how it read the document. `ArrayV2Metadata.from_dict` and `ArrayV3Metadata.from_dict` apply the upgrades for their Zarr format, so every path -that parses a stored document, including consolidated metadata, goes through them. +that parses a stored document, including consolidated metadata, goes through them, and +warn with each reading once the upgraded document has passed the metadata constructor. +An invalid document therefore raises its own error, not a warning about how it was read. To read another kind of invalid document, add an upgrade to `V2_ARRAY_UPGRADES` or `V3_ARRAY_UPGRADES`. @@ -14,7 +16,7 @@ import json import warnings -from collections.abc import Callable, Mapping, Sequence +from collections.abc import Callable, Iterable, Mapping, Sequence from typing import TYPE_CHECKING, Final from typing_extensions import TypeIs @@ -26,7 +28,9 @@ from zarr.core.common import JSON type ArrayDocument = Mapping[str, JSON] -type Upgrade = Callable[[ArrayDocument], ArrayDocument] +type Upgrade = Callable[[ArrayDocument], tuple[ArrayDocument, str] | None] +"""Returns `None` if the document needs no upgrade, else the upgraded document and a +sentence saying how it was read.""" RESAVE_HINT: Final = ( "To store valid metadata, open the array writable and call `array.update_attributes({})`; " @@ -35,10 +39,12 @@ ) -def _warn(message: str) -> None: - # The synchronous API parses metadata on zarr's IO thread, whose stack holds no - # user code, so the warning points at the upgrade on every path. - warnings.warn(f"{message} {RESAVE_HINT}", ZarrUserWarning, stacklevel=2) +def warn_readings(readings: Iterable[str]) -> None: + """Warn once for each reading returned by `upgrade_array_document`.""" + for reading in readings: + # The synchronous API parses metadata on zarr's IO thread, whose stack holds no + # user code, so the warning points at the `from_dict` that read the document. + warnings.warn(f"{reading} {RESAVE_HINT}", ZarrUserWarning, stacklevel=2) def _is_int_list(value: object) -> TypeIs[list[int] | tuple[int, ...]]: @@ -46,12 +52,15 @@ def _is_int_list(value: object) -> TypeIs[list[int] | tuple[int, ...]]: return isinstance(value, list | tuple) and all(isinstance(v, int) for v in value) -def _read_invalid_chunk_sizes(chunk_shape: JSON, shape: JSON, unit: JSON) -> list[int] | None: +def _read_invalid_chunk_sizes( + chunk_shape: JSON, shape: JSON, unit: JSON +) -> tuple[list[int], str] | None: """Read a regular chunk shape whose chunk sizes include 0, JSON `false` or JSON `true`. 0 and `false` are read as one chunk spanning the axis, a multiple of `unit` on that - axis (the inner chunk shape of a sharded array); `true` is read as 1. Returns `None` - when there is nothing to read this way: anything else that is invalid is left for + axis (the inner chunk shape of a sharded array); `true` is read as 1. Returns the + upgraded chunk shape and how it was read, or `None` when there is nothing to read + this way: anything else that is invalid is left for the metadata constructors to reject. """ if not (_is_int_list(chunk_shape) and _is_int_list(shape)) or len(chunk_shape) != len(shape): @@ -84,13 +93,15 @@ def _read_invalid_chunk_sizes(chunk_shape: JSON, shape: JSON, unit: JSON) -> lis " No chunk can be stored under a chunk size of 0, so the array holds only its " "fill value." ) - _warn(message) - return upgraded + return upgraded, message -def _invalid_chunk_sizes_v2(doc: ArrayDocument) -> ArrayDocument: - chunks = _read_invalid_chunk_sizes(doc.get("chunks"), doc.get("shape"), None) - return doc if chunks is None else {**doc, "chunks": chunks} +def _invalid_chunk_sizes_v2(doc: ArrayDocument) -> tuple[ArrayDocument, str] | None: + read = _read_invalid_chunk_sizes(doc.get("chunks"), doc.get("shape"), None) + if read is None: + return None + chunks, reading = read + return {**doc, "chunks": chunks}, reading def _sharding_chunk_shape(codecs: JSON) -> JSON: @@ -104,32 +115,43 @@ def _sharding_chunk_shape(codecs: JSON) -> JSON: return None -def _invalid_chunk_sizes_v3(doc: ArrayDocument) -> ArrayDocument: +def _invalid_chunk_sizes_v3(doc: ArrayDocument) -> tuple[ArrayDocument, str] | None: grid = doc.get("chunk_grid") if not isinstance(grid, Mapping) or grid.get("name") != "regular": - return doc + return None configuration = grid.get("configuration") if not isinstance(configuration, Mapping): - return doc - chunk_shape = _read_invalid_chunk_sizes( + return None + read = _read_invalid_chunk_sizes( configuration.get("chunk_shape"), doc.get("shape"), _sharding_chunk_shape(doc.get("codecs")), ) - if chunk_shape is None: - return doc + if read is None: + return None + chunk_shape, reading = read return { **doc, "chunk_grid": {**grid, "configuration": {**configuration, "chunk_shape": chunk_shape}}, - } + }, reading V2_ARRAY_UPGRADES: Final[tuple[Upgrade, ...]] = (_invalid_chunk_sizes_v2,) V3_ARRAY_UPGRADES: Final[tuple[Upgrade, ...]] = (_invalid_chunk_sizes_v3,) -def upgrade_array_document(doc: ArrayDocument, upgrades: Sequence[Upgrade]) -> ArrayDocument: - """Apply `upgrades` to a stored array metadata document, in order.""" +def upgrade_array_document( + doc: ArrayDocument, upgrades: Sequence[Upgrade] +) -> tuple[ArrayDocument, list[str]]: + """Apply `upgrades` to a stored array metadata document, in order. + + Returns the upgraded document and the readings of the upgrades that changed it, + for `warn_readings` once the document has been validated. + """ + readings: list[str] = [] for upgrade in upgrades: - doc = upgrade(doc) - return doc + upgraded = upgrade(doc) + if upgraded is not None: + doc, reading = upgraded + readings.append(reading) + return doc, readings diff --git a/src/zarr/core/metadata/v2.py b/src/zarr/core/metadata/v2.py index 0eb28efb11..c648f271e6 100644 --- a/src/zarr/core/metadata/v2.py +++ b/src/zarr/core/metadata/v2.py @@ -43,7 +43,7 @@ from zarr.core.config import config, parse_indexing_order from zarr.core.json_parse import parse_field from zarr.core.metadata.common import parse_attributes, parse_chunk_edge -from zarr.core.metadata.upgrades import V2_ARRAY_UPGRADES, upgrade_array_document +from zarr.core.metadata.upgrades import V2_ARRAY_UPGRADES, upgrade_array_document, warn_readings class ArrayV2MetadataDict(TypedDict): @@ -148,7 +148,8 @@ def to_buffer_dict(self, prototype: BufferPrototype) -> dict[str, Buffer]: @classmethod def from_dict(cls, data: dict[str, Any]) -> ArrayV2Metadata: - data = dict(upgrade_array_document(data, V2_ARRAY_UPGRADES)) + upgraded, readings = upgrade_array_document(data, V2_ARRAY_UPGRADES) + data = dict(upgraded) _data = data.copy() # Check that the zarr_format attribute is correct. _ = parse_zarr_format(_data.pop("zarr_format")) @@ -200,7 +201,9 @@ def from_dict(cls, data: dict[str, Any]) -> ArrayV2Metadata: _data = {k: v for k, v in _data.items() if k in expected} - return cls(**_data) + metadata = cls(**_data) + warn_readings(readings) + return metadata def to_dict(self) -> dict[str, JSON]: zarray_dict = super().to_dict() diff --git a/src/zarr/core/metadata/v3.py b/src/zarr/core/metadata/v3.py index 79889ff893..889fcbd822 100644 --- a/src/zarr/core/metadata/v3.py +++ b/src/zarr/core/metadata/v3.py @@ -39,7 +39,12 @@ from zarr.core.dtype.common import check_dtype_spec_v3 from zarr.core.json_parse import parse_field from zarr.core.metadata.common import parse_attributes, parse_chunk_edge -from zarr.core.metadata.upgrades import RESAVE_HINT, V3_ARRAY_UPGRADES, upgrade_array_document +from zarr.core.metadata.upgrades import ( + RESAVE_HINT, + V3_ARRAY_UPGRADES, + upgrade_array_document, + warn_readings, +) from zarr.errors import MetadataValidationError, NodeTypeValidationError, ZarrUserWarning from zarr.registry import get_codec_class @@ -709,8 +714,9 @@ def to_buffer_dict(self, prototype: BufferPrototype) -> dict[str, Buffer]: @classmethod def from_dict(cls, data: dict[str, JSON]) -> Self: + upgraded, readings = upgrade_array_document(data, V3_ARRAY_UPGRADES) # a new dict, because we are modifying it - _data = dict(upgrade_array_document(data, V3_ARRAY_UPGRADES)) + _data = dict(upgraded) # check that the zarr_format attribute is correct _ = parse_zarr_format(_data.pop("zarr_format")) @@ -750,7 +756,7 @@ def from_dict(cls, data: dict[str, JSON]) -> Self: # TODO: replace this with a real type check! _data_typed = cast(ArrayMetadataJSON_V3, _data) - return cls( + metadata = cls( shape=_data_typed["shape"], chunk_grid=_data_typed["chunk_grid"], # type: ignore[arg-type] chunk_key_encoding=_data_typed["chunk_key_encoding"], # type: ignore[arg-type] @@ -764,6 +770,8 @@ def from_dict(cls, data: dict[str, JSON]) -> Self: extra_fields=allowed_extra_fields, storage_transformers=_data_typed.get("storage_transformers", ()), # type: ignore[arg-type] ) + warn_readings(readings) + return metadata def to_dict(self) -> dict[str, JSON]: out_dict = super().to_dict() diff --git a/tests/test_metadata/test_upgrades.py b/tests/test_metadata/test_upgrades.py index 7506bb6875..a82a8a7717 100644 --- a/tests/test_metadata/test_upgrades.py +++ b/tests/test_metadata/test_upgrades.py @@ -113,28 +113,44 @@ def test_upgrade_array_document( ) -> None: """Valid documents pass unchanged and silently. A stored chunk size of 0 or `false` is read as one chunk spanning the axis (a multiple of the inner chunk when sharded) - and `true` as 1, with one warning that says how to re-save.""" + and `true` as 1; `from_dict` warns once with the reading and how to re-save.""" upgrades = V2_ARRAY_UPGRADES if doc["zarr_format"] == 2 else V3_ARRAY_UPGRADES - with warnings.catch_warnings(record=True) as record: - warnings.simplefilter("always") - upgraded = upgrade_array_document(doc, upgrades) + upgraded, readings = upgrade_array_document(doc, upgrades) assert _stored_chunks(dict(upgraded)) == expected assert {k: v for k, v in upgraded.items() if k not in ("chunks", "chunk_grid")} == { k: v for k, v in doc.items() if k not in ("chunks", "chunk_grid") } + metadata_cls = ArrayV2Metadata if doc["zarr_format"] == 2 else ArrayV3Metadata + with warnings.catch_warnings(record=True) as record: + warnings.simplefilter("always") + metadata_cls.from_dict(dict(doc)) messages = [str(w.message) for w in record] if warning is None: assert upgraded is doc + assert readings == [] assert messages == [] else: + assert len(readings) == 1 + assert re.search(warning, readings[0]) assert len(messages) == 1 - assert re.search(warning, messages[0]) + assert messages[0].startswith(readings[0]) assert "update_attributes({})" in messages[0] assert "zarr.consolidate_metadata" in messages[0] + + +@pytest.mark.parametrize( + "doc", + [_v2_doc([4], [0]) | {"dtype": " None: + """A document that is invalid after its upgrade raises its own error, without first + warning how it was read.""" metadata_cls = ArrayV2Metadata if doc["zarr_format"] == 2 else ArrayV3Metadata with warnings.catch_warnings(): - warnings.simplefilter("ignore", ZarrUserWarning) - metadata_cls.from_dict(dict(doc)) + warnings.simplefilter("error", ZarrUserWarning) + with pytest.raises(ValueError): + metadata_cls.from_dict(doc) @pytest.mark.parametrize("doc", [_v2_doc([4], [-1]), _v3_doc([4], [-1])], ids=["v2", "v3"]) From d17a848fa18a90471224526929eedf30369d8615 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Fri, 25 Sep 2026 23:01:09 +0200 Subject: [PATCH 26/76] fix(metadata): read a regular grid listing chunk edges as a document upgrade A `regular` chunk grid whose `chunk_shape` holds flat lists of integer edge lengths, e.g. `[2, [5, 10, 5]]`, is now read by one more entry in `V3_ARRAY_UPGRADES` instead of by a reader in `ArrayV3Metadata.__init__`. It applies on every path that parses a stored document, including consolidated metadata, and does not require `array.rectilinear_chunks`. Only what such documents hold is upgraded (integer sizes and flat integer edge lists); run-length encoded or non-integer entries are left for the regular grid to reject. The warning quotes a bounded prefix of the chunk shape and is emitted only after the upgraded document validates. The rectilinear flag moves off the chunk grid metadata constructor to the stored-document boundary: `ArrayV3Metadata.from_dict` checks the document as stored (before upgrades) and `ArrayV3Metadata.to_buffer_dict` checks the document about to be stored. Reading or creating a genuine rectilinear array still requires the flag, as before; re-saving an upgraded array stores a rectilinear grid and so requires it too, which the warning says. Tests: the upgrade's table test and one test per rejected form, using a document copied verbatim from a store zarr 3.2.1 wrote. Assisted-by: ClaudeCode:claude-opus-5-5 Co-Authored-By: Claude Opus 5.5 --- src/zarr/core/metadata/upgrades.py | 55 +++++++++++- src/zarr/core/metadata/v3.py | 93 +++++--------------- tests/test_metadata/test_upgrades.py | 123 ++++++++++++++++++++++++++- tests/test_metadata/test_v3.py | 104 +--------------------- tests/test_unified_chunk_grid.py | 42 ++++++--- 5 files changed, 230 insertions(+), 187 deletions(-) diff --git a/src/zarr/core/metadata/upgrades.py b/src/zarr/core/metadata/upgrades.py index e15ffd30fe..93b90a0662 100644 --- a/src/zarr/core/metadata/upgrades.py +++ b/src/zarr/core/metadata/upgrades.py @@ -136,8 +136,61 @@ def _invalid_chunk_sizes_v3(doc: ArrayDocument) -> tuple[ArrayDocument, str] | N }, reading +def _abbreviate(value: JSON, limit: int = 60) -> str: + """`value` as JSON, cut to at most `limit` characters.""" + text = json.dumps(value) + return text if len(text) <= limit else f"{text[: limit - 3]}..." + + +def _is_json_int(value: object) -> TypeIs[int]: + """Whether `value` is a JSON integer (not `true` or `false`).""" + return isinstance(value, int) and not isinstance(value, bool) + + +def _is_edge_list(value: object) -> TypeIs[list[int]]: + """Whether `value` is a JSON array of integers (not `true` or `false`).""" + return isinstance(value, list) and all(_is_json_int(edge) for edge in value) + + +def _edge_lists_in_regular_grid(doc: ArrayDocument) -> tuple[ArrayDocument, str] | None: + """Read a `regular` chunk grid whose `chunk_shape` lists chunk edges as rectilinear. + + Such a `chunk_shape` holds, per axis, an integer chunk size or a flat list of + integer chunk edge lengths, e.g. `[2, [5, 10, 5]]`. That is the `chunk_shapes` of + the rectilinear chunk grid it describes. Any other `chunk_shape` is left for the + metadata constructors to reject. + """ + grid = doc.get("chunk_grid") + if not isinstance(grid, Mapping) or grid.get("name") != "regular": + return None + configuration = grid.get("configuration") + if not isinstance(configuration, Mapping): + return None + chunk_shape = configuration.get("chunk_shape") + if not isinstance(chunk_shape, list): + return None + edge_axes = [axis for axis, dim in enumerate(chunk_shape) if isinstance(dim, list)] + if not edge_axes or not all(_is_json_int(dim) or _is_edge_list(dim) for dim in chunk_shape): + return None + reading = ( + f"The stored chunk grid is named 'regular', but its chunk_shape " + f"{_abbreviate(chunk_shape)} lists chunk edge lengths on axes {edge_axes}, which " + "only a rectilinear chunk grid can declare. It is read as that rectilinear chunk " + "grid. Storing or reading a rectilinear chunk grid requires " + "`zarr.config.set({'array.rectilinear_chunks': True})`." + ) + rectilinear: JSON = { + "name": "rectilinear", + "configuration": {"kind": "inline", "chunk_shapes": chunk_shape}, + } + return {**doc, "chunk_grid": rectilinear}, reading + + V2_ARRAY_UPGRADES: Final[tuple[Upgrade, ...]] = (_invalid_chunk_sizes_v2,) -V3_ARRAY_UPGRADES: Final[tuple[Upgrade, ...]] = (_invalid_chunk_sizes_v3,) +V3_ARRAY_UPGRADES: Final[tuple[Upgrade, ...]] = ( + _invalid_chunk_sizes_v3, + _edge_lists_in_regular_grid, +) def upgrade_array_document( diff --git a/src/zarr/core/metadata/v3.py b/src/zarr/core/metadata/v3.py index 889fcbd822..2fa41938fb 100644 --- a/src/zarr/core/metadata/v3.py +++ b/src/zarr/core/metadata/v3.py @@ -1,7 +1,6 @@ from __future__ import annotations import json -import warnings from collections.abc import Iterable, Mapping, Sequence from dataclasses import dataclass, field, replace from typing import TYPE_CHECKING, Any, Final, Literal, NotRequired, TypeGuard, cast @@ -40,12 +39,11 @@ from zarr.core.json_parse import parse_field from zarr.core.metadata.common import parse_attributes, parse_chunk_edge from zarr.core.metadata.upgrades import ( - RESAVE_HINT, V3_ARRAY_UPGRADES, upgrade_array_document, warn_readings, ) -from zarr.errors import MetadataValidationError, NodeTypeValidationError, ZarrUserWarning +from zarr.errors import MetadataValidationError, NodeTypeValidationError from zarr.registry import get_codec_class if TYPE_CHECKING: @@ -323,12 +321,6 @@ class RectilinearChunkGridMetadata(Metadata): chunk_shapes: tuple[int | tuple[int, ...], ...] def __post_init__(self) -> None: - if not config.get("array.rectilinear_chunks"): - raise ValueError( - "Rectilinear chunk grids are experimental and disabled by default. " - "Enable them with: zarr.config.set({'array.rectilinear_chunks': True}) " - "or set the environment variable ZARR_ARRAY__RECTILINEAR_CHUNKS=True" - ) object.__setattr__(self, "chunk_shapes", _validate_chunk_shapes(self.chunk_shapes)) @property @@ -406,6 +398,20 @@ def from_dict(cls, data: RectilinearChunkGridMetadataJSON) -> Self: # type: ign ChunkGridMetadata = RegularChunkGridMetadata | RectilinearChunkGridMetadata +def _check_rectilinear_chunks_enabled() -> None: + """Raise unless rectilinear chunks are enabled. + + The flag gates storing and reading array metadata documents that declare a + rectilinear chunk grid; the chunk grid metadata classes themselves are not gated. + """ + if not config.get("array.rectilinear_chunks"): + raise ValueError( + "Rectilinear chunk grids are experimental and disabled by default. " + "Enable them with: zarr.config.set({'array.rectilinear_chunks': True}) " + "or set the environment variable ZARR_ARRAY__RECTILINEAR_CHUNKS=True" + ) + + def create_chunk_grid_metadata( chunks: ChunkGrid, ) -> ChunkGridMetadata: @@ -459,45 +465,6 @@ def parse_chunk_grid( raise ValueError(f"Unknown chunk grid name: {name!r}") -def _parse_mixed_regular_chunk_grid( - chunk_shape: Iterable[Any], -) -> RectilinearChunkGridMetadata: - """Read a "regular" chunk grid whose chunk_shape contains edge lists. - - A regular grid cannot list chunk edges, so metadata such as - ``{"name": "regular", "configuration": {"chunk_shape": [2, [5, 10, 5]]}}`` - is invalid. It does describe one rectilinear grid unambiguously, so it is - read as that grid. See https://github.com/zarr-developers/zarr-python/issues/4374. - """ - # Put the dimensions in the JSON forms the rectilinear parser reads: a - # metadata dict built in Python holds a tuple where JSON holds a list. - # Elements are left alone, so `from_dict` still reports a bad one. - chunk_shapes: list[Any] = [ - list(dim_spec) if declares_chunk_edges(dim_spec) else dim_spec for dim_spec in chunk_shape - ] - msg = ( - f"This array's chunk grid is named 'regular' but its chunk_shape {chunk_shapes!r} " - "lists explicit chunk edges for some dimensions, which only a rectilinear chunk " - "grid can declare. " - ) - if not config.get("array.rectilinear_chunks"): - raise ValueError( - msg + "Reading it as a rectilinear chunk grid requires enabling rectilinear chunks: " - "zarr.config.set({'array.rectilinear_chunks': True})" - ) - warnings.warn( - msg + f"Reading it as a rectilinear chunk grid. {RESAVE_HINT}", - ZarrUserWarning, - stacklevel=2, - ) - return RectilinearChunkGridMetadata.from_dict( - { - "name": "rectilinear", - "configuration": {"kind": "inline", "chunk_shapes": chunk_shapes}, - } - ) - - class ArrayMetadataJSON_V3(TypedDict, extra_items=AllowedExtraField): # type: ignore[call-arg] """ A typed dictionary model for zarr v3 array metadata. @@ -537,28 +504,6 @@ class ArrayMetadataJSON_V3(TypedDict, extra_items=AllowedExtraField): # type: i } -def _read_stored_regular_chunk_grid( - chunk_grid: dict[str, JSON] | ChunkGridMetadata | NamedConfig[str, Any], -) -> dict[str, JSON] | ChunkGridMetadata | NamedConfig[str, Any]: - """Read a stored `regular` chunk grid whose `chunk_shape` lists chunk edges. - - Such a grid, e.g. `[2, [5, 10, 5]]`, is read as the rectilinear grid it - describes. Any other grid is returned unchanged for `parse_chunk_grid`. - """ - if not isinstance(chunk_grid, Mapping) or chunk_grid.get("name") != "regular": - return chunk_grid - configuration = chunk_grid.get("configuration") - if not isinstance(configuration, Mapping): - return chunk_grid - chunk_shape = configuration.get("chunk_shape") - if not declares_chunk_edges(chunk_shape): - return chunk_grid - dims = list(chunk_shape) - if any(declares_chunk_edges(dim) for dim in dims): - return _parse_mixed_regular_chunk_grid(dims) - return chunk_grid - - @dataclass(frozen=True, kw_only=True) class ArrayV3Metadata(Metadata): shape: tuple[int, ...] @@ -593,7 +538,7 @@ def __init__( """ shape_parsed = parse_shapelike(shape) - chunk_grid_parsed = parse_chunk_grid(_read_stored_regular_chunk_grid(chunk_grid)) + chunk_grid_parsed = parse_chunk_grid(chunk_grid) chunk_key_encoding_parsed = parse_chunk_key_encoding(chunk_key_encoding) dimension_names_parsed = parse_dimension_names(dimension_names) # Note: relying on a type method is numpy-specific @@ -709,11 +654,17 @@ def encode_chunk_key(self, chunk_coords: tuple[int, ...]) -> str: return self.chunk_key_encoding.encode_chunk_key(chunk_coords) def to_buffer_dict(self, prototype: BufferPrototype) -> dict[str, Buffer]: + if isinstance(self.chunk_grid, RectilinearChunkGridMetadata): + _check_rectilinear_chunks_enabled() indent = config.get("json_indent") return {ZARR_JSON: json_to_buffer(self.to_dict(), prototype=prototype, indent=indent)} @classmethod def from_dict(cls, data: dict[str, JSON]) -> Self: + # The flag gates what the document declares, so it is checked before upgrades. + chunk_grid = data.get("chunk_grid") + if isinstance(chunk_grid, Mapping) and chunk_grid.get("name") == "rectilinear": + _check_rectilinear_chunks_enabled() upgraded, readings = upgrade_array_document(data, V3_ARRAY_UPGRADES) # a new dict, because we are modifying it _data = dict(upgraded) diff --git a/tests/test_metadata/test_upgrades.py b/tests/test_metadata/test_upgrades.py index a82a8a7717..2cc9d40c2a 100644 --- a/tests/test_metadata/test_upgrades.py +++ b/tests/test_metadata/test_upgrades.py @@ -17,7 +17,7 @@ V3_ARRAY_UPGRADES, upgrade_array_document, ) -from zarr.core.metadata.v3 import RegularChunkGridMetadata +from zarr.core.metadata.v3 import RectilinearChunkGridMetadata, RegularChunkGridMetadata from zarr.dtype import Int16 from zarr.errors import ZarrUserWarning @@ -307,3 +307,124 @@ def test_legacy_chunk_size_consolidated(tmp_path: Path, zarr_format: Literal[2, reopened = zarr.open_group(path, mode="r", use_consolidated=use_consolidated)["a"] assert isinstance(reopened, zarr.Array) assert reopened.chunks == (1,) + + +# A document copied verbatim from a store that zarr 3.2.1 wrote for +# `create_array(shape=(6, 20), chunks=(2, (5, 10, 5)), dtype="float32")`. +MIXED_REGULAR_GRID_DOC = """{ + "shape": [6, 20], + "data_type": "float32", + "chunk_grid": {"name": "regular", "configuration": {"chunk_shape": [2, [5, 10, 5]]}}, + "chunk_key_encoding": {"name": "default", "configuration": {"separator": "/"}}, + "fill_value": 0.0, + "codecs": [ + {"name": "bytes", "configuration": {"endian": "little"}}, + {"name": "zstd", "configuration": {"level": 0, "checksum": false}} + ], + "attributes": {}, + "zarr_format": 3, + "node_type": "array", + "storage_transformers": [] +}""" + + +def _mixed_doc(shape: list[int], chunk_shape: list[Any]) -> dict[str, JSON]: + doc: dict[str, JSON] = json.loads(MIXED_REGULAR_GRID_DOC) + doc["shape"] = shape + doc["chunk_grid"] = {"name": "regular", "configuration": {"chunk_shape": chunk_shape}} + return doc + + +@pytest.mark.parametrize( + ("doc", "expected", "axes"), + [ + (json.loads(MIXED_REGULAR_GRID_DOC), (2, (5, 10, 5)), [1]), + (_mixed_doc([6, 20, 4], [2, [5, 10, 5], [1, 3]]), (2, (5, 10, 5), (1, 3)), [1, 2]), + (_mixed_doc([6, 20], [[1, 5], [5, 10, 5]]), ((1, 5), (5, 10, 5)), [0, 1]), + (_mixed_doc([6, 12], [2, [5, 10, 5]]), (2, (5, 10, 5)), [1]), + (_mixed_doc([0, 20], [2, [5, 10, 5]]), (2, (5, 10, 5)), [1]), + (_mixed_doc([6, 0], [2, [5, 10, 5]]), (2, (5, 10, 5)), [1]), + (_mixed_doc([4, 10_000], [2, [10] * 1000]), (2, (10,) * 1000), [1]), + ], + ids=["written", "3d", "every-axis", "shrunk", "empty-regular-axis", "empty-edge-axis", "long"], +) +def test_read_edge_lists_in_regular_grid( + doc: dict[str, JSON], expected: tuple[int | tuple[int, ...], ...], axes: list[int] +) -> None: + """A `regular` chunk grid whose `chunk_shape` lists chunk edges is read as the + rectilinear grid it describes, without the rectilinear chunks flag, with one + warning that names the axes, quotes at most a bounded part of the chunk shape and + says how to re-save.""" + with ( + zarr.config.set({"array.rectilinear_chunks": False}), + warnings.catch_warnings(record=True) as record, + ): + warnings.simplefilter("always") + metadata = ArrayV3Metadata.from_dict(doc) + assert metadata.chunk_grid == RectilinearChunkGridMetadata(chunk_shapes=expected) + [message] = [str(w.message) for w in record] + assert f"lists chunk edge lengths on axes {axes}" in message + assert "array.rectilinear_chunks" in message + assert "update_attributes({})" in message + assert len(message) < 1000 + + +def test_edge_lists_in_regular_grid_rle_rejected() -> None: + """Run-length encoded edges were never written inside a `regular` chunk_shape, so + such a document is not read as rectilinear.""" + with warnings.catch_warnings(): + warnings.simplefilter("error", ZarrUserWarning) + with pytest.raises(TypeError, match="Dimension 1: a regular chunk grid requires"): + ArrayV3Metadata.from_dict(_mixed_doc([6, 20], [2, [[5, 2], 10]])) + + +@pytest.mark.parametrize("edges", [[10.0, 10.0], [True, True]], ids=["float", "bool"]) +def test_edge_lists_in_regular_grid_non_integer_edges_rejected(edges: list[Any]) -> None: + with warnings.catch_warnings(): + warnings.simplefilter("error", ZarrUserWarning) + with pytest.raises(TypeError, match="Dimension 1: a regular chunk grid requires"): + ArrayV3Metadata.from_dict(_mixed_doc([6, 20], [2, edges])) + + +def test_edge_lists_in_regular_grid_nonpositive_edge_rejected() -> None: + """The upgraded grid is validated before any warning, so the real error surfaces.""" + with warnings.catch_warnings(): + warnings.simplefilter("error", ZarrUserWarning) + with pytest.raises(ValueError, match="must be >= 1, got 0"): + ArrayV3Metadata.from_dict(_mixed_doc([6, 20], [2, [0, 20]])) + + +def test_edge_lists_in_regular_grid_short_edges_rejected() -> None: + with warnings.catch_warnings(): + warnings.simplefilter("error", ZarrUserWarning) + with pytest.raises(ValueError, match="sum to 15 but array shape extent is 20"): + ArrayV3Metadata.from_dict(_mixed_doc([6, 20], [2, [5, 10]])) + + +def test_edge_lists_in_regular_grid_round_trip(tmp_path: Path) -> None: + """A store holding the verbatim document opens without the rectilinear chunks flag, + reads its data, re-saves as rectilinear with the flag, and then opens cleanly.""" + path = tmp_path / "mixed.zarr" + data = np.arange(120, dtype="float32").reshape(6, 20) + with zarr.config.set({"array.rectilinear_chunks": True}): + arr = zarr.create_array(path, shape=data.shape, chunks=(2, (5, 10, 5)), dtype="float32") + arr[...] = data + (path / "zarr.json").write_text(MIXED_REGULAR_GRID_DOC) + + with ( + zarr.config.set({"array.rectilinear_chunks": False}), + pytest.warns(ZarrUserWarning, match="read as that rectilinear chunk grid"), + ): + arr = zarr.open_array(path, mode="a") + np.testing.assert_array_equal(arr[...], data) + + with zarr.config.set({"array.rectilinear_chunks": True}): + arr.update_attributes({}) + assert json.loads((path / "zarr.json").read_text())["chunk_grid"] == { + "name": "rectilinear", + "configuration": {"kind": "inline", "chunk_shapes": [2, [5, 10, 5]]}, + } + with warnings.catch_warnings(): + warnings.simplefilter("error", ZarrUserWarning) + reopened = zarr.open_array(path) + np.testing.assert_array_equal(reopened[...], data) diff --git a/tests/test_metadata/test_v3.py b/tests/test_metadata/test_v3.py index 6104815e2a..8a949a32b4 100644 --- a/tests/test_metadata/test_v3.py +++ b/tests/test_metadata/test_v3.py @@ -8,10 +8,9 @@ import numpy as np import pytest -import zarr from tests.conftest import Expect, ExpectFail from tests.test_metadata.conftest import minimal_metadata_dict_v3 -from zarr.core.buffer import cpu, default_buffer_prototype +from zarr.core.buffer import default_buffer_prototype from zarr.core.chunk_grids import ChunkGrid, is_regular_1d, is_regular_nd from zarr.core.config import config from zarr.core.dtype import Float64, UInt8 @@ -37,7 +36,7 @@ ) if TYPE_CHECKING: - from collections.abc import Callable, Sequence + from collections.abc import Callable from typing import Any @@ -229,104 +228,6 @@ def test_regular_chunk_grid_rejects_edge_lists() -> None: RegularChunkGridMetadata(chunk_shape=(2, (5, 10, 5))) # type: ignore[arg-type] -# --------------------------------------------------------------------------- -# "regular" chunk grids whose chunk_shape contains edge lists -# -# zarr 3.2.0 and 3.2.1 wrote mixed specs such as (2, (5, 10, 5)) this way -# (https://github.com/zarr-developers/zarr-python/issues/4374). The reader -# accepts any such grid, not only the shapes 3.2.x could produce. -# --------------------------------------------------------------------------- - - -@pytest.mark.parametrize( - "case", - [ - Expect( - input=((6, 20), [2, [5, 10, 5]]), - output=(2, (5, 10, 5)), - id="edges_in_last_dim", - ), - Expect( - input=((6, 20, 4), [2, [5, 10, 5], 4]), - output=(2, (5, 10, 5), 4), - id="edges_in_middle_dim", - ), - Expect( - input=((6, 20), [[1, 5], [5, 10, 5]]), - output=((1, 5), (5, 10, 5)), - id="edges_in_every_dim", - ), - Expect( - input=((6, 20), [2, [[5, 2], 10]]), - output=(2, (5, 5, 10)), - id="run_length_encoded_edges", - ), - Expect( - input=((6, 20), (2, (5, 10, 5))), - output=(2, (5, 10, 5)), - id="tuples_from_python", - ), - Expect( - input=((6, 20), [2, np.array([5, 10, 5])]), - output=(2, (5, 10, 5)), - id="numpy_array_edges", - ), - ], - ids=lambda case: case.id, -) -def test_read_mixed_regular_chunk_grid( - case: Expect[tuple[tuple[int, ...], Sequence[Any]], tuple[Any, ...]], -) -> None: - """A "regular" chunk grid whose chunk_shape lists edges is read as rectilinear.""" - shape, chunk_shape = case.input - d = minimal_metadata_dict_v3( - shape=shape, - chunk_grid={"name": "regular", "configuration": {"chunk_shape": chunk_shape}}, - ) - with config.set({"array.rectilinear_chunks": True}): - with pytest.warns(ZarrUserWarning, match="only a rectilinear chunk grid can declare"): - meta = ArrayV3Metadata.from_dict(d) # type: ignore[arg-type] - assert meta.chunk_grid == RectilinearChunkGridMetadata(chunk_shapes=case.output) - - -def test_read_mixed_regular_chunk_grid_requires_rectilinear_chunks() -> None: - """Reading a mixed "regular" chunk grid explains why rectilinear chunks must be enabled.""" - d = minimal_metadata_dict_v3( - shape=(6, 20), - chunk_grid={"name": "regular", "configuration": {"chunk_shape": [2, [5, 10, 5]]}}, - ) - with ( - config.set({"array.rectilinear_chunks": False}), - pytest.raises(ValueError, match="only a rectilinear chunk grid can declare"), - ): - ArrayV3Metadata.from_dict(d) # type: ignore[arg-type] - - -def test_open_array_with_mixed_regular_chunk_grid() -> None: - """An array stored with zarr 3.2.x's mixed chunk grid reads correctly and re-saves as rectilinear.""" - data = np.arange(120, dtype="float32").reshape(6, 20) - store = zarr.storage.MemoryStore() - with config.set({"array.rectilinear_chunks": True}): - arr = zarr.create_array(store, shape=data.shape, chunks=(2, (5, 10, 5)), dtype=data.dtype) - arr[:] = data - # Rewrite the metadata the way zarr 3.2.0 and 3.2.1 stored it. The chunk - # layout is unchanged: 3.2.x wrote the chunks as a rectilinear grid. - doc = json.loads(store._store_dict["zarr.json"].to_bytes()) - doc["chunk_grid"] = {"name": "regular", "configuration": {"chunk_shape": [2, [5, 10, 5]]}} - store._store_dict["zarr.json"] = cpu.Buffer.from_bytes(json.dumps(doc).encode()) - - with pytest.warns(ZarrUserWarning, match="only a rectilinear chunk grid can declare"): - arr = zarr.open_array(store) - np.testing.assert_array_equal(arr[:], data) - - arr.update_attributes({}) - doc = json.loads(store._store_dict["zarr.json"].to_bytes()) - assert doc["chunk_grid"] == { - "name": "rectilinear", - "configuration": {"kind": "inline", "chunk_shapes": [2, [5, 10, 5]]}, - } - - # --------------------------------------------------------------------------- # Types # --------------------------------------------------------------------------- @@ -604,7 +505,6 @@ def test_group_metadata_to_dict(attributes: dict[str, Any] | None) -> None: def test_group_metadata_to_dict_consolidated(attributes: dict[str, Any] | None) -> None: """GroupMetadata.to_dict includes consolidated_metadata when present.""" from zarr import consolidate_metadata, create_group - from zarr.errors import ZarrUserWarning store: dict[str, object] = {} group = create_group(store, attributes=attributes, zarr_format=3) diff --git a/tests/test_unified_chunk_grid.py b/tests/test_unified_chunk_grid.py index 0b9ba196bc..ac54c62756 100644 --- a/tests/test_unified_chunk_grid.py +++ b/tests/test_unified_chunk_grid.py @@ -14,6 +14,8 @@ import pytest import zarr +from tests.test_metadata.conftest import minimal_metadata_dict_v3 +from zarr.core.buffer import default_buffer_prototype from zarr.core.chunk_grids import ( ChunkGrid, ChunkSpec, @@ -22,7 +24,9 @@ _is_rectilinear_chunks, ) from zarr.core.common import compress_rle, expand_rle +from zarr.core.dtype import UInt8 from zarr.core.metadata.v3 import ( + ArrayV3Metadata, RectilinearChunkGridMetadata, RectilinearChunkGridMetadataJSON, RegularChunkGridMetadata, @@ -101,30 +105,44 @@ def test_dimension_index_to_chunk_last_valid( # --------------------------------------------------------------------------- +_RECTILINEAR_DOC = minimal_metadata_dict_v3( + shape=(30, 50), + chunk_grid={ + "name": "rectilinear", + "configuration": {"kind": "inline", "chunk_shapes": [[10, 20], [25, 25]]}, + }, +) + + @pytest.mark.parametrize( "action", [ - lambda: RectilinearChunkGridMetadata(chunk_shapes=((10, 20), (25, 25))), - lambda: RectilinearChunkGridMetadata.from_dict( - { - "name": "rectilinear", - "configuration": {"kind": "inline", "chunk_shapes": [[10, 20, 30], [50, 50]]}, - } - ), + lambda: ArrayV3Metadata.from_dict(dict(_RECTILINEAR_DOC)), # type: ignore[arg-type] + lambda: ArrayV3Metadata( + shape=(30, 50), + data_type=UInt8(), + chunk_grid=RectilinearChunkGridMetadata(chunk_shapes=((10, 20), (25, 25))), + chunk_key_encoding={"name": "default"}, + fill_value=0, + codecs=[{"name": "bytes"}], + attributes=None, + dimension_names=None, + ).to_buffer_dict(default_buffer_prototype()), lambda: zarr.create_array(MemoryStore(), shape=(30,), chunks=[[10, 20]], dtype="int32"), ], - ids=["constructor", "from_dict", "create_array"], + ids=["read", "store", "create_array"], ) def test_rectilinear_feature_flag_blocked(action: Any) -> None: - """Rectilinear chunk operations raise ValueError when the feature flag is disabled""" + """Reading or storing an array metadata document that declares a rectilinear chunk + grid raises ValueError when the feature flag is disabled.""" with zarr.config.set({"array.rectilinear_chunks": False}): with pytest.raises(ValueError, match="experimental and disabled by default"): action() -def test_rectilinear_feature_flag_enabled() -> None: - """Rectilinear chunk grid construction succeeds when the feature flag is enabled""" - with zarr.config.set({"array.rectilinear_chunks": True}): +def test_rectilinear_metadata_classes_not_gated() -> None: + """The flag gates stored documents, not the chunk grid metadata classes.""" + with zarr.config.set({"array.rectilinear_chunks": False}): grid = RectilinearChunkGridMetadata(chunk_shapes=((10, 20), (25, 25))) assert grid.ndim == 2 From 8aef3197d7d22473ae4b61b0bfb94cc61ea6b039 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Fri, 25 Sep 2026 23:06:49 +0200 Subject: [PATCH 27/76] refactor(metadata): drop numpy and tuple widening from chunk grid metadata The chunk grid metadata classes take typed Python values again: a regular chunk shape of `int`s, rectilinear dimensions of `int` or `list` in stored documents. `declares_chunk_edges` and the numpy and tuple handling in `expand_rle`, `_validate_chunk_shapes` and `RectilinearChunkGridMetadata.from_dict` are removed with their tests; numpy input belongs to the `chunks=` normalizer. The regular grid still rejects edge lists (gh-4374), now naming the offending type instead of echoing the value. Assisted-by: ClaudeCode:claude-opus-5-5 Co-Authored-By: Claude Opus 5.5 --- src/zarr/core/common.py | 34 ++++++------------------ src/zarr/core/metadata/v3.py | 37 ++++++++++---------------- tests/test_metadata/test_v3.py | 42 +---------------------------- tests/test_unified_chunk_grid.py | 45 +++----------------------------- 4 files changed, 26 insertions(+), 132 deletions(-) diff --git a/src/zarr/core/common.py b/src/zarr/core/common.py index d6b33dd5c1..ba01f6c19f 100644 --- a/src/zarr/core/common.py +++ b/src/zarr/core/common.py @@ -12,7 +12,6 @@ Literal, NotRequired, TypedDict, - TypeGuard, cast, overload, ) @@ -277,21 +276,7 @@ def _default_zarr_format() -> ZarrFormat: return cast("ZarrFormat", int(zarr_config.get("default_zarr_format", 3))) -def declares_chunk_edges(value: object) -> TypeGuard[Iterable[Any]]: - """Whether a chunk specification element declares explicit chunk edge lengths. - - True for any non-integer iterable, which is the form explicit edge lengths - take, and False for a single integer size. The test is structural rather - than an exact type check: edge lengths reach us as a list parsed from JSON, - as a tuple from a metadata dict built in Python, or as a numpy array, and - all three mean the same thing. - """ - if isinstance(value, int | np.integer): - return False - return isinstance(value, Iterable) and not isinstance(value, str | bytes) - - -def expand_rle(data: Iterable[int | Sequence[int]]) -> list[int]: +def expand_rle(data: Sequence[int | list[int]]) -> list[int]: """Expand a mixed array of bare integers and RLE pairs. Per the rectilinear chunk grid spec, each element can be: @@ -300,21 +285,18 @@ def expand_rle(data: Iterable[int | Sequence[int]]) -> list[int]: """ result: list[int] = [] for item in data: - if declares_chunk_edges(item): - pair = tuple(item) - if len(pair) != 2: - raise ValueError(f"RLE entries must be an integer or [size, count], got {item}") - size, count = int(pair[0]), int(pair[1]) + if isinstance(item, (int, float)) and not isinstance(item, bool): + val = int(item) + if val < 1: + raise ValueError(f"Chunk edge length must be >= 1, got {val}") + result.append(val) + elif isinstance(item, list) and len(item) == 2: + size, count = int(item[0]), int(item[1]) if size < 1: raise ValueError(f"Chunk edge length must be >= 1, got {size}") if count < 1: raise ValueError(f"RLE repeat count must be >= 1, got {count}") result.extend([size] * count) - elif isinstance(item, int | np.integer | float) and not isinstance(item, bool): - val = int(item) - if val < 1: - raise ValueError(f"Chunk edge length must be >= 1, got {val}") - result.append(val) else: raise ValueError(f"RLE entries must be an integer or [size, count], got {item}") return result diff --git a/src/zarr/core/metadata/v3.py b/src/zarr/core/metadata/v3.py index 2fa41938fb..7729ed4f75 100644 --- a/src/zarr/core/metadata/v3.py +++ b/src/zarr/core/metadata/v3.py @@ -5,7 +5,6 @@ from dataclasses import dataclass, field, replace from typing import TYPE_CHECKING, Any, Final, Literal, NotRequired, TypeGuard, cast -import numpy as np from typing_extensions import TypedDict from zarr.abc.codec import ArrayArrayCodec, ArrayBytesCodec, BytesBytesCodec, Codec @@ -26,7 +25,6 @@ NamedConfig, NamedRequiredConfig, compress_rle, - declares_chunk_edges, expand_rle, parse_named_configuration, parse_shapelike, @@ -229,18 +227,15 @@ def _parse_chunk_shape(chunk_shape: Iterable[int]) -> tuple[int, ...]: let a rectilinear chunk shape be stored as a regular grid (gh-4374). """ parsed: list[int] = [] + # Typed as ints, but a stored document can hold anything here. for dim_idx, dim_spec in enumerate(cast("Iterable[object]", chunk_shape)): - if not isinstance(dim_spec, int | np.integer): + if not isinstance(dim_spec, int): raise TypeError( f"Dimension {dim_idx}: a regular chunk grid requires an integer chunk " - f"edge length, got {dim_spec!r}. Lists of chunk edge lengths belong " - "to a rectilinear chunk grid." + f"edge length, got a {type(dim_spec).__name__}. Lists of chunk edge " + "lengths belong to a rectilinear chunk grid." ) - parsed.append( - parse_chunk_edge( - int(dim_spec) if isinstance(dim_spec, np.integer) else dim_spec, dim_idx - ) - ) + parsed.append(parse_chunk_edge(dim_spec, dim_idx)) return tuple(parsed) @@ -254,10 +249,10 @@ def _validate_chunk_shapes( """ result: list[int | tuple[int, ...]] = [] for dim_idx, dim_spec in enumerate(chunk_shapes): - if isinstance(dim_spec, int | np.integer): - result.append(parse_chunk_edge(int(dim_spec), dim_idx)) + if isinstance(dim_spec, int): + result.append(parse_chunk_edge(dim_spec, dim_idx)) else: - edges = tuple(int(edge) for edge in dim_spec) + edges = tuple(dim_spec) if not edges: raise ValueError(f"Dimension {dim_idx} has no chunk edges.") bad = [i for i, e in enumerate(edges) if isinstance(e, bool) or e < 1] @@ -298,8 +293,7 @@ def to_dict(self) -> RegularChunkGridMetadataJSON: # type: ignore[override] def from_dict(cls, data: RegularChunkGridMetadataJSON) -> Self: # type: ignore[override] parse_named_configuration(data, "regular") # validate name configuration = data["configuration"] - # `__post_init__` parses, so this only has to hand over the dimensions. - return cls(chunk_shape=tuple(configuration["chunk_shape"])) + return cls(chunk_shape=_parse_chunk_shape(configuration["chunk_shape"])) @dataclass(frozen=True, kw_only=True) @@ -380,17 +374,14 @@ def from_dict(cls, data: RectilinearChunkGridMetadataJSON) -> Self: # type: ign raw_shapes = configuration["chunk_shapes"] parsed: list[int | tuple[int, ...]] = [] for dim_spec in raw_shapes: - if declares_chunk_edges(dim_spec): + if isinstance(dim_spec, int): + # `__post_init__` range-checks it, naming the dimension. + parsed.append(dim_spec) + elif isinstance(dim_spec, list): parsed.append(tuple(expand_rle(dim_spec))) - elif isinstance(dim_spec, int | np.integer): - # Range checks belong to `_validate_chunk_shapes`, which - # `__post_init__` runs over the result and which names the - # offending dimension. - parsed.append(int(dim_spec)) else: raise TypeError( - "Invalid chunk_shapes entry: expected an integer or a sequence of " - f"chunk edge lengths, got {type(dim_spec)}" + f"Invalid chunk_shapes entry: expected int or list, got {type(dim_spec)}" ) return cls(chunk_shapes=tuple(parsed)) diff --git a/tests/test_metadata/test_v3.py b/tests/test_metadata/test_v3.py index 8a949a32b4..96e44ce230 100644 --- a/tests/test_metadata/test_v3.py +++ b/tests/test_metadata/test_v3.py @@ -5,7 +5,6 @@ import json from typing import TYPE_CHECKING -import numpy as np import pytest from tests.conftest import Expect, ExpectFail @@ -20,7 +19,6 @@ ARRAY_METADATA_KEYS, ArrayMetadataJSON_V3, ArrayV3Metadata, - RectilinearChunkGridMetadata, RegularChunkGridMetadata, create_chunk_grid_metadata, parse_codecs, @@ -32,7 +30,6 @@ MetadataValidationError, NodeTypeValidationError, UnknownCodecError, - ZarrUserWarning, ) if TYPE_CHECKING: @@ -159,44 +156,6 @@ def test_create_chunk_grid_metadata_unknown_dimension_type() -> None: create_chunk_grid_metadata(grid) -@pytest.mark.parametrize( - "case", - [ - Expect(input=(2, 3), output=(2, 3), id="python_ints"), - Expect(input=(np.int64(2), np.uint8(3)), output=(2, 3), id="numpy_ints"), - ], - ids=lambda case: case.id, -) -def test_regular_chunk_grid_normalizes_ints(case: Expect[tuple[Any, ...], tuple[int, ...]]) -> None: - """A regular chunk grid accepts Python and numpy integers and stores Python ints.""" - grid = RegularChunkGridMetadata(chunk_shape=case.input) - assert grid.chunk_shape == case.output - assert all(type(c) is int for c in grid.chunk_shape) - - -@pytest.mark.parametrize( - "case", - [ - Expect(input=(2, (5, 10, 5)), output=(2, (5, 10, 5)), id="python_values"), - Expect( - input=(np.int64(2), np.array([5, 10, 5])), output=(2, (5, 10, 5)), id="numpy_values" - ), - Expect(input=(2, [5, 10, 5]), output=(2, (5, 10, 5)), id="list_edges"), - ], - ids=lambda case: case.id, -) -def test_rectilinear_chunk_grid_normalizes_ints( - case: Expect[tuple[Any, ...], tuple[int | tuple[int, ...], ...]], -) -> None: - """A rectilinear chunk grid accepts any sequence of Python or numpy integers - as a dimension's edges and stores Python ints in tuples.""" - with config.set({"array.rectilinear_chunks": True}): - grid = RectilinearChunkGridMetadata(chunk_shapes=case.input) - assert grid.chunk_shapes == case.output - flat = [e for dim in grid.chunk_shapes for e in (dim if isinstance(dim, tuple) else (dim,))] - assert all(type(e) is int for e in flat) - - @pytest.mark.parametrize( "build", [ @@ -505,6 +464,7 @@ def test_group_metadata_to_dict(attributes: dict[str, Any] | None) -> None: def test_group_metadata_to_dict_consolidated(attributes: dict[str, Any] | None) -> None: """GroupMetadata.to_dict includes consolidated_metadata when present.""" from zarr import consolidate_metadata, create_group + from zarr.errors import ZarrUserWarning store: dict[str, object] = {} group = create_group(store, attributes=attributes, zarr_format=3) diff --git a/tests/test_unified_chunk_grid.py b/tests/test_unified_chunk_grid.py index ac54c62756..69665d4778 100644 --- a/tests/test_unified_chunk_grid.py +++ b/tests/test_unified_chunk_grid.py @@ -512,8 +512,6 @@ def test_chunk_grid_iter() -> None: [ ([[10, 3]], [10, 10, 10]), ([[10, 2], [20, 1]], [10, 10, 20]), - pytest.param([(10, 3)], [10, 10, 10], id="tuple-pair"), - pytest.param([np.int64(10), np.array([20, 2])], [10, 20, 20], id="numpy-values"), ], ) def test_rle_expand(compressed: list[Any], expected: list[int]) -> None: @@ -551,8 +549,6 @@ def test_rle_roundtrip() -> None: ([[-10, 2]], "Chunk edge length must be >= 1"), ([[5, 0]], "RLE repeat count must be >= 1"), ([[5, -1]], "RLE repeat count must be >= 1"), - ([(5, 2, 1)], "RLE entries must be an integer or"), - (["5"], "RLE entries must be an integer or"), ], ids=[ "zero-edge", @@ -561,8 +557,6 @@ def test_rle_roundtrip() -> None: "negative-rle-size", "zero-rle-count", "negative-rle-count", - "rle-pair-wrong-length", - "string-entry", ], ) def test_rle_expand_rejects_invalid(rle_input: list[Any], match: str) -> None: @@ -3023,54 +3017,21 @@ def test_iter_chunk_regions_rectilinear() -> None: }, (4, (10, 20)), ), - # A metadata dict built in Python may hold tuples and numpy values - # where parsed JSON holds lists and ints. - pytest.param( - { - "name": "rectilinear", - "configuration": {"kind": "inline", "chunk_shapes": (4, (10, 20))}, - }, - (4, (10, 20)), - id="tuple-dims", - ), - pytest.param( - { - "name": "rectilinear", - "configuration": {"kind": "inline", "chunk_shapes": [((4, 3),), [10, 20]]}, - }, - ((4, 4, 4), (10, 20)), - id="tuple-rle-pair", - ), - pytest.param( - { - "name": "rectilinear", - "configuration": { - "kind": "inline", - "chunk_shapes": [np.int64(4), np.array([10, 20]), [np.array([5, 2])]], - }, - }, - (4, (10, 20), (5, 5)), - id="numpy-values", - ), ], ) def test_rectilinear_from_dict( json_input: RectilinearChunkGridMetadataJSON, expected_chunk_shapes: tuple[int | tuple[int, ...], ...], ) -> None: - """RectilinearChunkGridMetadata.from_dict correctly parses all spec forms, - whatever sequence type holds a dimension's edges, and stores plain ints.""" + """RectilinearChunkGridMetadata.from_dict correctly parses all spec forms.""" grid = RectilinearChunkGridMetadata.from_dict(json_input) assert grid.chunk_shapes == expected_chunk_shapes - flat = [e for dim in grid.chunk_shapes for e in (dim if isinstance(dim, tuple) else (dim,))] - assert all(type(e) is int for e in flat) @pytest.mark.parametrize("dim_spec", [4.5, None, "10"], ids=["float", "none", "string"]) def test_rectilinear_from_dict_rejects_invalid_dim_spec(dim_spec: Any) -> None: - """A dimension that is neither an integer nor a sequence of edges is rejected. - A string is iterable but is not a sequence of edges.""" - with pytest.raises(TypeError, match="expected an integer or a sequence of chunk edge lengths"): + """A dimension that is neither an integer nor a list of edges is rejected.""" + with pytest.raises(TypeError, match="expected int or list"): RectilinearChunkGridMetadata.from_dict( {"name": "rectilinear", "configuration": {"kind": "inline", "chunk_shapes": [dim_spec]}} ) From d56ac403b610c7e413969d20056433a0efa87adb Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Fri, 25 Sep 2026 23:07:19 +0200 Subject: [PATCH 28/76] docs: describe reading mixed regular grids without the rectilinear flag Rewrite the 4375 changelog fragment for the upgrade and the moved flag gate, and drop the numpy/tuple parsing claims that no longer hold. Trim the history from the regular chunk shape docstring. Assisted-by: ClaudeCode:claude-opus-5-5 Co-Authored-By: Claude Opus 5.5 --- changes/4375.bugfix.md | 4 ++-- src/zarr/core/metadata/v3.py | 7 ++----- 2 files changed, 4 insertions(+), 7 deletions(-) diff --git a/changes/4375.bugfix.md b/changes/4375.bugfix.md index 234c1af012..73c4caa4c1 100644 --- a/changes/4375.bugfix.md +++ b/changes/4375.bugfix.md @@ -1,3 +1,3 @@ -Arrays written by zarr 3.2.0 and 3.2.1 with a mixed regular/rectilinear chunk specification (e.g. `chunks=(2, (5, 10, 5))`) can be read again. Their stored "regular" chunk grid is read as the rectilinear grid it describes, with a warning explaining how to re-save the metadata. A regular chunk grid with per-dimension edge lists is now rejected when constructed. +Arrays whose stored `regular` chunk grid lists chunk edge lengths for some dimensions, such as `"chunk_shape": [2, [5, 10, 5]]` (written by zarr 3.2.0 and 3.2.1 for `chunks=(2, (5, 10, 5))`), can be read again, without enabling `array.rectilinear_chunks`. The grid is read as the rectilinear chunk grid it describes, with a `ZarrUserWarning`; re-saving the metadata stores that rectilinear grid, which requires the flag. A regular chunk grid given edge lists is now rejected, so this metadata is no longer written. -Rectilinear chunk grid metadata reads a dimension's chunk edges from any sequence of integers — a list parsed from JSON, a tuple from a metadata document built in Python, or a numpy array — instead of requiring a `list` of `int`, and stores them as Python ints. A dimension that is neither an integer nor a sequence of edge lengths is rejected with a message naming both forms. +The `array.rectilinear_chunks` flag now gates reading and storing array metadata that declares a rectilinear chunk grid, instead of constructing `RectilinearChunkGridMetadata`. diff --git a/src/zarr/core/metadata/v3.py b/src/zarr/core/metadata/v3.py index 7729ed4f75..22108f8c78 100644 --- a/src/zarr/core/metadata/v3.py +++ b/src/zarr/core/metadata/v3.py @@ -220,11 +220,8 @@ class RectilinearChunkGridMetadataConfig(TypedDict): def _parse_chunk_shape(chunk_shape: Iterable[int]) -> tuple[int, ...]: """Validate and normalize a regular chunk shape. - A regular chunk shape is one bare int per dimension, each >= 1. Lists of - chunk edge lengths belong to a rectilinear chunk grid and are rejected - here; `_validate_chunk_shapes` is the rectilinear counterpart. The two - grid kinds validate separately on purpose — sharing a validator is what - let a rectilinear chunk shape be stored as a regular grid (gh-4374). + A regular chunk shape is one int per dimension, each >= 1. Lists of chunk + edge lengths belong to a rectilinear chunk grid and are rejected. """ parsed: list[int] = [] # Typed as ints, but a stored document can hold anything here. From d9549b8710d638e96bd2d6fda9345b54feee63e4 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Fri, 25 Sep 2026 22:46:45 +0200 Subject: [PATCH 29/76] refactor(metadata): warn about an upgraded document only once it validates An upgrade now returns the upgraded document and how it read it, and `from_dict` warns with those readings after the metadata constructor accepts the upgraded document. A stored document that is still invalid after an upgrade raises its own error instead of first warning how it was read. Assisted-by: ClaudeCode:claude-opus-5-5 Co-Authored-By: Claude Opus 5.5 --- src/zarr/core/metadata/upgrades.py | 76 ++++++++++++++++++---------- src/zarr/core/metadata/v2.py | 9 ++-- src/zarr/core/metadata/v3.py | 13 +++-- tests/test_metadata/test_upgrades.py | 30 ++++++++--- 4 files changed, 88 insertions(+), 40 deletions(-) diff --git a/src/zarr/core/metadata/upgrades.py b/src/zarr/core/metadata/upgrades.py index 933b4c53e2..e15ffd30fe 100644 --- a/src/zarr/core/metadata/upgrades.py +++ b/src/zarr/core/metadata/upgrades.py @@ -2,9 +2,11 @@ This is the only place invalid metadata is read leniently; the metadata constructors are strict. An upgrade maps a stored array metadata document (parsed JSON) to a valid -one and warns once when it changes anything. `ArrayV2Metadata.from_dict` and +one and says how it read the document. `ArrayV2Metadata.from_dict` and `ArrayV3Metadata.from_dict` apply the upgrades for their Zarr format, so every path -that parses a stored document, including consolidated metadata, goes through them. +that parses a stored document, including consolidated metadata, goes through them, and +warn with each reading once the upgraded document has passed the metadata constructor. +An invalid document therefore raises its own error, not a warning about how it was read. To read another kind of invalid document, add an upgrade to `V2_ARRAY_UPGRADES` or `V3_ARRAY_UPGRADES`. @@ -14,7 +16,7 @@ import json import warnings -from collections.abc import Callable, Mapping, Sequence +from collections.abc import Callable, Iterable, Mapping, Sequence from typing import TYPE_CHECKING, Final from typing_extensions import TypeIs @@ -26,7 +28,9 @@ from zarr.core.common import JSON type ArrayDocument = Mapping[str, JSON] -type Upgrade = Callable[[ArrayDocument], ArrayDocument] +type Upgrade = Callable[[ArrayDocument], tuple[ArrayDocument, str] | None] +"""Returns `None` if the document needs no upgrade, else the upgraded document and a +sentence saying how it was read.""" RESAVE_HINT: Final = ( "To store valid metadata, open the array writable and call `array.update_attributes({})`; " @@ -35,10 +39,12 @@ ) -def _warn(message: str) -> None: - # The synchronous API parses metadata on zarr's IO thread, whose stack holds no - # user code, so the warning points at the upgrade on every path. - warnings.warn(f"{message} {RESAVE_HINT}", ZarrUserWarning, stacklevel=2) +def warn_readings(readings: Iterable[str]) -> None: + """Warn once for each reading returned by `upgrade_array_document`.""" + for reading in readings: + # The synchronous API parses metadata on zarr's IO thread, whose stack holds no + # user code, so the warning points at the `from_dict` that read the document. + warnings.warn(f"{reading} {RESAVE_HINT}", ZarrUserWarning, stacklevel=2) def _is_int_list(value: object) -> TypeIs[list[int] | tuple[int, ...]]: @@ -46,12 +52,15 @@ def _is_int_list(value: object) -> TypeIs[list[int] | tuple[int, ...]]: return isinstance(value, list | tuple) and all(isinstance(v, int) for v in value) -def _read_invalid_chunk_sizes(chunk_shape: JSON, shape: JSON, unit: JSON) -> list[int] | None: +def _read_invalid_chunk_sizes( + chunk_shape: JSON, shape: JSON, unit: JSON +) -> tuple[list[int], str] | None: """Read a regular chunk shape whose chunk sizes include 0, JSON `false` or JSON `true`. 0 and `false` are read as one chunk spanning the axis, a multiple of `unit` on that - axis (the inner chunk shape of a sharded array); `true` is read as 1. Returns `None` - when there is nothing to read this way: anything else that is invalid is left for + axis (the inner chunk shape of a sharded array); `true` is read as 1. Returns the + upgraded chunk shape and how it was read, or `None` when there is nothing to read + this way: anything else that is invalid is left for the metadata constructors to reject. """ if not (_is_int_list(chunk_shape) and _is_int_list(shape)) or len(chunk_shape) != len(shape): @@ -84,13 +93,15 @@ def _read_invalid_chunk_sizes(chunk_shape: JSON, shape: JSON, unit: JSON) -> lis " No chunk can be stored under a chunk size of 0, so the array holds only its " "fill value." ) - _warn(message) - return upgraded + return upgraded, message -def _invalid_chunk_sizes_v2(doc: ArrayDocument) -> ArrayDocument: - chunks = _read_invalid_chunk_sizes(doc.get("chunks"), doc.get("shape"), None) - return doc if chunks is None else {**doc, "chunks": chunks} +def _invalid_chunk_sizes_v2(doc: ArrayDocument) -> tuple[ArrayDocument, str] | None: + read = _read_invalid_chunk_sizes(doc.get("chunks"), doc.get("shape"), None) + if read is None: + return None + chunks, reading = read + return {**doc, "chunks": chunks}, reading def _sharding_chunk_shape(codecs: JSON) -> JSON: @@ -104,32 +115,43 @@ def _sharding_chunk_shape(codecs: JSON) -> JSON: return None -def _invalid_chunk_sizes_v3(doc: ArrayDocument) -> ArrayDocument: +def _invalid_chunk_sizes_v3(doc: ArrayDocument) -> tuple[ArrayDocument, str] | None: grid = doc.get("chunk_grid") if not isinstance(grid, Mapping) or grid.get("name") != "regular": - return doc + return None configuration = grid.get("configuration") if not isinstance(configuration, Mapping): - return doc - chunk_shape = _read_invalid_chunk_sizes( + return None + read = _read_invalid_chunk_sizes( configuration.get("chunk_shape"), doc.get("shape"), _sharding_chunk_shape(doc.get("codecs")), ) - if chunk_shape is None: - return doc + if read is None: + return None + chunk_shape, reading = read return { **doc, "chunk_grid": {**grid, "configuration": {**configuration, "chunk_shape": chunk_shape}}, - } + }, reading V2_ARRAY_UPGRADES: Final[tuple[Upgrade, ...]] = (_invalid_chunk_sizes_v2,) V3_ARRAY_UPGRADES: Final[tuple[Upgrade, ...]] = (_invalid_chunk_sizes_v3,) -def upgrade_array_document(doc: ArrayDocument, upgrades: Sequence[Upgrade]) -> ArrayDocument: - """Apply `upgrades` to a stored array metadata document, in order.""" +def upgrade_array_document( + doc: ArrayDocument, upgrades: Sequence[Upgrade] +) -> tuple[ArrayDocument, list[str]]: + """Apply `upgrades` to a stored array metadata document, in order. + + Returns the upgraded document and the readings of the upgrades that changed it, + for `warn_readings` once the document has been validated. + """ + readings: list[str] = [] for upgrade in upgrades: - doc = upgrade(doc) - return doc + upgraded = upgrade(doc) + if upgraded is not None: + doc, reading = upgraded + readings.append(reading) + return doc, readings diff --git a/src/zarr/core/metadata/v2.py b/src/zarr/core/metadata/v2.py index 0eb28efb11..c648f271e6 100644 --- a/src/zarr/core/metadata/v2.py +++ b/src/zarr/core/metadata/v2.py @@ -43,7 +43,7 @@ from zarr.core.config import config, parse_indexing_order from zarr.core.json_parse import parse_field from zarr.core.metadata.common import parse_attributes, parse_chunk_edge -from zarr.core.metadata.upgrades import V2_ARRAY_UPGRADES, upgrade_array_document +from zarr.core.metadata.upgrades import V2_ARRAY_UPGRADES, upgrade_array_document, warn_readings class ArrayV2MetadataDict(TypedDict): @@ -148,7 +148,8 @@ def to_buffer_dict(self, prototype: BufferPrototype) -> dict[str, Buffer]: @classmethod def from_dict(cls, data: dict[str, Any]) -> ArrayV2Metadata: - data = dict(upgrade_array_document(data, V2_ARRAY_UPGRADES)) + upgraded, readings = upgrade_array_document(data, V2_ARRAY_UPGRADES) + data = dict(upgraded) _data = data.copy() # Check that the zarr_format attribute is correct. _ = parse_zarr_format(_data.pop("zarr_format")) @@ -200,7 +201,9 @@ def from_dict(cls, data: dict[str, Any]) -> ArrayV2Metadata: _data = {k: v for k, v in _data.items() if k in expected} - return cls(**_data) + metadata = cls(**_data) + warn_readings(readings) + return metadata def to_dict(self) -> dict[str, JSON]: zarray_dict = super().to_dict() diff --git a/src/zarr/core/metadata/v3.py b/src/zarr/core/metadata/v3.py index 4a27d495b3..b6929220ce 100644 --- a/src/zarr/core/metadata/v3.py +++ b/src/zarr/core/metadata/v3.py @@ -36,7 +36,11 @@ from zarr.core.dtype.common import check_dtype_spec_v3 from zarr.core.json_parse import parse_field from zarr.core.metadata.common import parse_attributes, parse_chunk_edge -from zarr.core.metadata.upgrades import V3_ARRAY_UPGRADES, upgrade_array_document +from zarr.core.metadata.upgrades import ( + V3_ARRAY_UPGRADES, + upgrade_array_document, + warn_readings, +) from zarr.errors import MetadataValidationError, NodeTypeValidationError from zarr.registry import get_codec_class @@ -628,8 +632,9 @@ def to_buffer_dict(self, prototype: BufferPrototype) -> dict[str, Buffer]: @classmethod def from_dict(cls, data: dict[str, JSON]) -> Self: + upgraded, readings = upgrade_array_document(data, V3_ARRAY_UPGRADES) # a new dict, because we are modifying it - _data = dict(upgrade_array_document(data, V3_ARRAY_UPGRADES)) + _data = dict(upgraded) # check that the zarr_format attribute is correct _ = parse_zarr_format(_data.pop("zarr_format")) @@ -669,7 +674,7 @@ def from_dict(cls, data: dict[str, JSON]) -> Self: # TODO: replace this with a real type check! _data_typed = cast(ArrayMetadataJSON_V3, _data) - return cls( + metadata = cls( shape=_data_typed["shape"], chunk_grid=_data_typed["chunk_grid"], # type: ignore[arg-type] chunk_key_encoding=_data_typed["chunk_key_encoding"], # type: ignore[arg-type] @@ -683,6 +688,8 @@ def from_dict(cls, data: dict[str, JSON]) -> Self: extra_fields=allowed_extra_fields, storage_transformers=_data_typed.get("storage_transformers", ()), # type: ignore[arg-type] ) + warn_readings(readings) + return metadata def to_dict(self) -> dict[str, JSON]: out_dict = super().to_dict() diff --git a/tests/test_metadata/test_upgrades.py b/tests/test_metadata/test_upgrades.py index 7506bb6875..a82a8a7717 100644 --- a/tests/test_metadata/test_upgrades.py +++ b/tests/test_metadata/test_upgrades.py @@ -113,28 +113,44 @@ def test_upgrade_array_document( ) -> None: """Valid documents pass unchanged and silently. A stored chunk size of 0 or `false` is read as one chunk spanning the axis (a multiple of the inner chunk when sharded) - and `true` as 1, with one warning that says how to re-save.""" + and `true` as 1; `from_dict` warns once with the reading and how to re-save.""" upgrades = V2_ARRAY_UPGRADES if doc["zarr_format"] == 2 else V3_ARRAY_UPGRADES - with warnings.catch_warnings(record=True) as record: - warnings.simplefilter("always") - upgraded = upgrade_array_document(doc, upgrades) + upgraded, readings = upgrade_array_document(doc, upgrades) assert _stored_chunks(dict(upgraded)) == expected assert {k: v for k, v in upgraded.items() if k not in ("chunks", "chunk_grid")} == { k: v for k, v in doc.items() if k not in ("chunks", "chunk_grid") } + metadata_cls = ArrayV2Metadata if doc["zarr_format"] == 2 else ArrayV3Metadata + with warnings.catch_warnings(record=True) as record: + warnings.simplefilter("always") + metadata_cls.from_dict(dict(doc)) messages = [str(w.message) for w in record] if warning is None: assert upgraded is doc + assert readings == [] assert messages == [] else: + assert len(readings) == 1 + assert re.search(warning, readings[0]) assert len(messages) == 1 - assert re.search(warning, messages[0]) + assert messages[0].startswith(readings[0]) assert "update_attributes({})" in messages[0] assert "zarr.consolidate_metadata" in messages[0] + + +@pytest.mark.parametrize( + "doc", + [_v2_doc([4], [0]) | {"dtype": " None: + """A document that is invalid after its upgrade raises its own error, without first + warning how it was read.""" metadata_cls = ArrayV2Metadata if doc["zarr_format"] == 2 else ArrayV3Metadata with warnings.catch_warnings(): - warnings.simplefilter("ignore", ZarrUserWarning) - metadata_cls.from_dict(dict(doc)) + warnings.simplefilter("error", ZarrUserWarning) + with pytest.raises(ValueError): + metadata_cls.from_dict(doc) @pytest.mark.parametrize("doc", [_v2_doc([4], [-1]), _v3_doc([4], [-1])], ids=["v2", "v3"]) From 5ef0f51f787fc523332d99a4c2d4c8cf4dc65d2c Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Sat, 26 Sep 2026 02:08:34 +0200 Subject: [PATCH 30/76] refactor(metadata): one per-axis rule for stored chunk sizes, one rule for edges Stored regular chunk shapes, in both formats and in a sharding codec's inner chunk shape, are read entry by entry by one rule in upgrades.py: an int >= 1 is kept, `true` is read as 1 (zarr 3.0.10 also wrote it as the inner chunk size of a shard, which the sharding codec used to read silently), and 0 or `false` as one chunk spanning the axis. Integral floats are not read: no release wrote a regular grid with them. A document warns once, naming the array where the caller knows its path. Metadata constructors check every chunk edge length with one function, `parse_chunk_edge`, which tells a wrong type (TypeError) from a wrong value (ValueError); this covers bare sizes, rectilinear edges, RLE sizes and the sharding codec's inner chunk shape. Assisted-by: ClaudeCode:claude-opus-5-5 Co-Authored-By: Claude Opus 5.5 --- changes/4334.bugfix.md | 6 +- src/zarr/codecs/sharding.py | 9 +- src/zarr/core/array.py | 8 +- src/zarr/core/common.py | 46 +++-- src/zarr/core/group.py | 20 +- src/zarr/core/metadata/common.py | 9 - src/zarr/core/metadata/upgrades.py | 185 ++++++++++------- src/zarr/core/metadata/v2.py | 13 +- src/zarr/core/metadata/v3.py | 39 ++-- tests/test_metadata/test_upgrades.py | 292 ++++++++++++++++++--------- tests/test_unified_chunk_grid.py | 26 +-- 11 files changed, 396 insertions(+), 257 deletions(-) diff --git a/changes/4334.bugfix.md b/changes/4334.bugfix.md index 26228e668f..77c211643e 100644 --- a/changes/4334.bugfix.md +++ b/changes/4334.bugfix.md @@ -1,3 +1,5 @@ -A chunk edge length is now always at least 1, while an array extent may be 0. Every spelling of one chunk spanning an axis (`chunks=-1`, `chunks=False`, `chunks="auto"`, `shards=-1`, `shards=False`) gives chunk size 1 on a zero-length axis, and a shard spanning an axis is a multiple of the inner chunk size, so `shards=-1` no longer fails when the axis length is not a multiple of it. Array metadata built in code with a chunk size of 0 or `True` now raises, as does `FixedDimension(size=0, ...)`. Rectilinear chunk grids can be created on a zero-length dimension with a non-empty list of positive chunk sizes, which are kept for later growth. +A chunk edge length is now always at least 1, while an array extent may be 0. Every spelling of one chunk spanning an axis (`chunks=-1`, `chunks=False`, `chunks="auto"`, `shards=-1`, `shards=False`) gives chunk size 1 on a zero-length axis, and a shard spanning an axis is a multiple of the inner chunk size, so `shards=-1` no longer fails when the axis length is not a multiple of it. Rectilinear chunk grids can be created on a zero-length dimension with a non-empty list of positive chunk sizes, which are kept for later growth. -Stored metadata with a regular chunk size of 0 or JSON `false`, as zarr-python wrote for arrays created with a zero-length axis until 3.4, now opens with a `ZarrUserWarning` and is read as one chunk spanning the axis (a multiple of the inner chunk size for sharded arrays, which previously failed to open), and JSON `true`, as zarr-python 3.0 and 3.2 wrote for `chunks=(True,)`, is read as 1. The warning says how to store valid metadata: `array.update_attributes({})`, then `zarr.consolidate_metadata` if the metadata is consolidated. +Metadata built in code is strict about chunk edge lengths: `ArrayV2Metadata`, `RegularChunkGridMetadata`, `RectilinearChunkGridMetadata` and `ShardingCodec` take them as Python `int`s of at least 1, so a size of 0 now raises a `ValueError`, and a `bool`, a float or a NumPy integer (including a scalar `chunks=np.int64(5)` for `ArrayV2Metadata`) raises a `TypeError`; `FixedDimension(size=0, ...)` raises a `ValueError`. Stored rectilinear chunk grids with integral floats (`10.0`) are rejected too. The array creation functions still accept NumPy integers in `chunks=` and `shards=`. + +Stored metadata with a regular chunk size of 0 or JSON `false`, as zarr-python wrote for arrays created with a zero-length axis until 3.4, now opens with a `ZarrUserWarning` and is read as one chunk spanning the axis (a multiple of the inner chunk size for sharded arrays, which previously failed to open). JSON `true`, as zarr-python 3.0 and 3.2 wrote for `chunks=(True,)`, is read as 1, in the chunk grid and in the inner chunk shape of a sharding codec. The warning names the array and says how to store valid metadata: `array.update_attributes({})`, then `zarr.consolidate_metadata` if the metadata is consolidated. diff --git a/src/zarr/codecs/sharding.py b/src/zarr/codecs/sharding.py index 7a053867d6..37fe52c1a6 100644 --- a/src/zarr/codecs/sharding.py +++ b/src/zarr/codecs/sharding.py @@ -45,9 +45,8 @@ merge_and_encode_chunk, ) from zarr.core.common import ( - ShapeLike, + parse_chunk_shape, parse_named_configuration, - parse_shapelike, product, ) from zarr.core.config import config as zarr_config @@ -459,13 +458,13 @@ class ShardingCodec( def __init__( self, *, - chunk_shape: ShapeLike, + chunk_shape: Iterable[int], codecs: Iterable[Codec | dict[str, JSON]] = (BytesCodec(),), index_codecs: Iterable[Codec | dict[str, JSON]] = (BytesCodec(), Crc32cCodec()), index_location: ShardingCodecIndexLocation | IndexLocation = "end", subchunk_write_order: SubchunkWriteOrder = "morton", ) -> None: - chunk_shape_parsed = parse_shapelike(chunk_shape) + chunk_shape_parsed = parse_chunk_shape(chunk_shape) codecs_parsed = parse_codecs(codecs) index_codecs_parsed = parse_codecs(index_codecs) _check_index_codecs_fixed_size(index_codecs_parsed) @@ -504,7 +503,7 @@ def __getstate__(self) -> dict[str, Any]: def __setstate__(self, state: dict[str, Any]) -> None: config = state["configuration"] - object.__setattr__(self, "chunk_shape", parse_shapelike(config["chunk_shape"])) + object.__setattr__(self, "chunk_shape", parse_chunk_shape(config["chunk_shape"])) object.__setattr__(self, "codecs", parse_codecs(config["codecs"])) object.__setattr__(self, "index_codecs", parse_codecs(config["index_codecs"])) object.__setattr__(self, "index_location", _parse_index_location(config["index_location"])) diff --git a/src/zarr/core/array.py b/src/zarr/core/array.py index 5a8d6bf57e..e39080b668 100644 --- a/src/zarr/core/array.py +++ b/src/zarr/core/array.py @@ -201,13 +201,13 @@ def _chunk_sizes_from_shape( return tuple(result) -def parse_array_metadata(data: Any) -> ArrayMetadata: +def parse_array_metadata(data: Any, path: str | None = None) -> ArrayMetadata: if isinstance(data, ArrayMetadata): return data elif isinstance(data, dict): zarr_format = data.get("zarr_format") if zarr_format == 3: - meta_out = ArrayV3Metadata.from_dict(data) + meta_out = ArrayV3Metadata.from_dict(data, path=path) if len(meta_out.storage_transformers) > 0: msg = ( f"Array metadata contains storage transformers: {meta_out.storage_transformers}." @@ -216,7 +216,7 @@ def parse_array_metadata(data: Any) -> ArrayMetadata: raise ValueError(msg) return meta_out elif zarr_format == 2: - return ArrayV2Metadata.from_dict(data) + return ArrayV2Metadata.from_dict(data, path=path) else: raise ValueError(f"Invalid zarr_format: {zarr_format}. Expected 2 or 3") raise TypeError # pragma: no cover @@ -404,7 +404,7 @@ def __init__( store_path: StorePath, config: ArrayConfigLike | None = None, ) -> None: - metadata_parsed = parse_array_metadata(metadata) + metadata_parsed = parse_array_metadata(metadata, str(store_path)) config_parsed = parse_array_config(config) object.__setattr__(self, "metadata", metadata_parsed) diff --git a/src/zarr/core/common.py b/src/zarr/core/common.py index ba01f6c19f..cddbf91912 100644 --- a/src/zarr/core/common.py +++ b/src/zarr/core/common.py @@ -276,7 +276,32 @@ def _default_zarr_format() -> ZarrFormat: return cast("ZarrFormat", int(zarr_config.get("default_zarr_format", 3))) -def expand_rle(data: Sequence[int | list[int]]) -> list[int]: +def _parse_positive_int(value: object, name: str) -> int: + if isinstance(value, bool) or not isinstance(value, int): + raise TypeError(f"{name} must be an int, got {value!r}") + if value < 1: + raise ValueError(f"{name} must be >= 1, got {value!r}") + return value + + +def parse_chunk_edge(size: object, axis: int | None = None) -> int: + """Check that `size` is a chunk edge length: an `int` (not a `bool`) of at least 1. + + This is the one rule for chunk edge lengths in metadata: bare chunk sizes, explicit + edges and run-length encoded sizes. `axis`, when given, is named in the error. + """ + where = "" if axis is None else f"Dimension {axis}: " + return _parse_positive_int(size, f"{where}Chunk edge length") + + +def parse_chunk_shape(data: object) -> tuple[int, ...]: + """Check a regular chunk shape: one chunk edge length per axis (see `parse_chunk_edge`).""" + if not isinstance(data, Iterable): + raise TypeError(f"A chunk shape must be a sequence of chunk edge lengths, got {data!r}") + return tuple(parse_chunk_edge(size, axis) for axis, size in enumerate(data)) + + +def expand_rle(data: Sequence[object]) -> list[int]: """Expand a mixed array of bare integers and RLE pairs. Per the rectilinear chunk grid spec, each element can be: @@ -285,20 +310,13 @@ def expand_rle(data: Sequence[int | list[int]]) -> list[int]: """ result: list[int] = [] for item in data: - if isinstance(item, (int, float)) and not isinstance(item, bool): - val = int(item) - if val < 1: - raise ValueError(f"Chunk edge length must be >= 1, got {val}") - result.append(val) - elif isinstance(item, list) and len(item) == 2: - size, count = int(item[0]), int(item[1]) - if size < 1: - raise ValueError(f"Chunk edge length must be >= 1, got {size}") - if count < 1: - raise ValueError(f"RLE repeat count must be >= 1, got {count}") - result.extend([size] * count) + if isinstance(item, list): + if len(item) != 2: + raise ValueError(f"RLE entries must be an integer or [size, count], got {item}") + size, count = item + result.extend([parse_chunk_edge(size)] * _parse_positive_int(count, "RLE repeat count")) else: - raise ValueError(f"RLE entries must be an integer or [size, count], got {item}") + result.append(parse_chunk_edge(item)) return result diff --git a/src/zarr/core/group.py b/src/zarr/core/group.py index d734e6b7cd..78071300f0 100644 --- a/src/zarr/core/group.py +++ b/src/zarr/core/group.py @@ -190,12 +190,12 @@ def from_dict(cls, data: dict[str, JSON]) -> ConsolidatedMetadata: if node_type == "group": metadata[k] = GroupMetadata.from_dict(v) elif node_type == "array": - metadata[k] = ArrayV3Metadata.from_dict(v) + metadata[k] = ArrayV3Metadata.from_dict(v, path=k) else: assert_never(node_type) elif zarr_format == 2: if "shape" in v: - metadata[k] = ArrayV2Metadata.from_dict(v) + metadata[k] = ArrayV2Metadata.from_dict(v, path=k) else: metadata[k] = GroupMetadata.from_dict(v) else: @@ -3504,7 +3504,9 @@ async def _read_metadata_v3(store: Store, path: str) -> ArrayV3Metadata | GroupM ) if zarr_json_bytes is None: raise FileNotFoundError(path) - return _build_metadata_v3(buffer_to_json_object(zarr_json_bytes)) + return _build_metadata_v3( + buffer_to_json_object(zarr_json_bytes), path=_join_paths([str(store), path]) + ) async def _read_metadata_v2(store: Store, path: str) -> ArrayV2Metadata | GroupMetadata: @@ -3539,7 +3541,7 @@ async def _read_metadata_v2(store: Store, path: str) -> ArrayV2Metadata | GroupM else: zmeta = buffer_to_json_object(zgroup_bytes) - return _build_metadata_v2(zmeta, zattrs) + return _build_metadata_v2(zmeta, zattrs, path=_join_paths([str(store), path])) async def _read_group_metadata_v2(store: Store, path: str) -> GroupMetadata: @@ -3570,7 +3572,9 @@ async def _read_group_metadata( return await _read_group_metadata_v3(store=store, path=path) -def _build_metadata_v3(zarr_json: dict[str, JSON]) -> ArrayV3Metadata | GroupMetadata: +def _build_metadata_v3( + zarr_json: dict[str, JSON], *, path: str | None = None +) -> ArrayV3Metadata | GroupMetadata: """ Convert a dict representation of Zarr V3 metadata into the corresponding metadata class. """ @@ -3579,7 +3583,7 @@ def _build_metadata_v3(zarr_json: dict[str, JSON]) -> ArrayV3Metadata | GroupMet raise MetadataValidationError(msg) match zarr_json: case {"node_type": "array"}: - return ArrayV3Metadata.from_dict(zarr_json) + return ArrayV3Metadata.from_dict(zarr_json, path=path) case {"node_type": "group"}: return GroupMetadata.from_dict(zarr_json) case _: # pragma: no cover @@ -3589,14 +3593,14 @@ def _build_metadata_v3(zarr_json: dict[str, JSON]) -> ArrayV3Metadata | GroupMet def _build_metadata_v2( - zarr_json: dict[str, JSON], attrs_json: dict[str, JSON] + zarr_json: dict[str, JSON], attrs_json: dict[str, JSON], *, path: str | None = None ) -> ArrayV2Metadata | GroupMetadata: """ Convert a dict representation of Zarr V2 metadata into the corresponding metadata class. """ match zarr_json: case {"shape": _}: - return ArrayV2Metadata.from_dict(zarr_json | {"attributes": attrs_json}) + return ArrayV2Metadata.from_dict(zarr_json | {"attributes": attrs_json}, path=path) case _: # pragma: no cover return GroupMetadata.from_dict(zarr_json | {"attributes": attrs_json}) diff --git a/src/zarr/core/metadata/common.py b/src/zarr/core/metadata/common.py index b9e86f537a..6367bdb28a 100644 --- a/src/zarr/core/metadata/common.py +++ b/src/zarr/core/metadata/common.py @@ -11,12 +11,3 @@ def parse_attributes(data: dict[str, JSON] | None) -> dict[str, JSON]: return {} return dict(data) - - -def parse_chunk_edge(size: object, axis: int) -> int: - """Check that `size` is a chunk edge length: an `int` (not a `bool`) of at least 1.""" - if isinstance(size, bool) or not isinstance(size, int) or size < 1: - raise ValueError( - f"Dimension {axis}: chunk edge length must be an integer >= 1, got {size!r}" - ) - return size diff --git a/src/zarr/core/metadata/upgrades.py b/src/zarr/core/metadata/upgrades.py index e15ffd30fe..0cdac3151d 100644 --- a/src/zarr/core/metadata/upgrades.py +++ b/src/zarr/core/metadata/upgrades.py @@ -5,8 +5,9 @@ one and says how it read the document. `ArrayV2Metadata.from_dict` and `ArrayV3Metadata.from_dict` apply the upgrades for their Zarr format, so every path that parses a stored document, including consolidated metadata, goes through them, and -warn with each reading once the upgraded document has passed the metadata constructor. -An invalid document therefore raises its own error, not a warning about how it was read. +warn once, with every reading, after the upgraded document has passed the metadata +constructor. An invalid document therefore raises its own error, not a warning about +how it was read. To read another kind of invalid document, add an upgrade to `V2_ARRAY_UPGRADES` or `V3_ARRAY_UPGRADES`. @@ -17,9 +18,8 @@ import json import warnings from collections.abc import Callable, Iterable, Mapping, Sequence -from typing import TYPE_CHECKING, Final - -from typing_extensions import TypeIs +from itertools import chain, repeat +from typing import TYPE_CHECKING, Final, TypeGuard, cast from zarr.core.chunk_grids import full_span_chunk_size from zarr.errors import ZarrUserWarning @@ -39,105 +39,148 @@ ) -def warn_readings(readings: Iterable[str]) -> None: - """Warn once for each reading returned by `upgrade_array_document`.""" - for reading in readings: +def warn_readings(readings: Sequence[str], path: str | None) -> None: + """Warn once with the readings returned by `upgrade_array_document`, naming the + array at `path` when the caller knows it.""" + if readings: + subject = "" if path is None else f"Array {path!r}: " # The synchronous API parses metadata on zarr's IO thread, whose stack holds no # user code, so the warning points at the `from_dict` that read the document. - warnings.warn(f"{reading} {RESAVE_HINT}", ZarrUserWarning, stacklevel=2) + warnings.warn(f"{subject}{' '.join(readings)} {RESAVE_HINT}", ZarrUserWarning, stacklevel=2) -def _is_int_list(value: object) -> TypeIs[list[int] | tuple[int, ...]]: +def _is_int_list(value: object) -> TypeGuard[list[int] | tuple[int, ...]]: """Whether `value` is a JSON array of integers (JSON `false` and `true` count).""" return isinstance(value, list | tuple) and all(isinstance(v, int) for v in value) -def _read_invalid_chunk_sizes( - chunk_shape: JSON, shape: JSON, unit: JSON -) -> tuple[list[int], str] | None: - """Read a regular chunk shape whose chunk sizes include 0, JSON `false` or JSON `true`. +def _read_chunk_size(size: JSON, span: int | None, unit: int) -> tuple[int, str | None] | None: + """Read one entry of a stored regular chunk shape as a chunk edge length. - 0 and `false` are read as one chunk spanning the axis, a multiple of `unit` on that - axis (the inner chunk shape of a sharded array); `true` is read as 1. Returns the - upgraded chunk shape and how it was read, or `None` when there is nothing to read - this way: anything else that is invalid is left for - the metadata constructors to reject. + Returns the edge length and, for an invalid entry, how it was read; `None` if the + entry cannot be read, which leaves it for the metadata constructors to reject. A + JSON int >= 1 is kept, JSON `true` is read as 1, and 0 or JSON `false` is read as + one chunk spanning the axis of length `span`, a multiple of `unit` (the inner chunk + size of a shard), when the span is known. """ - if not (_is_int_list(chunk_shape) and _is_int_list(shape)) or len(chunk_shape) != len(shape): - return None - invalid_axes = [ - axis for axis, size in enumerate(chunk_shape) if size == 0 or isinstance(size, bool) - ] - if not invalid_axes: + match size: + case True: + return 1, "1" + case int() if size >= 1: + return size, None + case int() if size == 0 and span is not None: + edge = full_span_chunk_size(span, unit) + how = f"one chunk spanning the axis ({edge})" + if span > 0: + how += ( + ", and as no chunk can be stored under a chunk size of 0, the array " + "holds only its fill value" + ) + return edge, how + return None + + +def _read_chunk_shape( + stored: JSON, spans: Sequence[int | None], units: Iterable[int], name: str +) -> tuple[list[int], str | None] | None: + """Read a stored regular chunk shape, entry by entry (see `_read_chunk_size`), for + axes of lengths `spans` whose chunks are multiples of `units` (1 where not given). + + Returns the chunk shape and, if an entry is invalid, a sentence saying how the + `name` was read; `None` if it cannot be read. + """ + if not (isinstance(stored, list | tuple) and len(stored) == len(spans)): return None - units = ( - list(unit) - if _is_int_list(unit) and len(unit) == len(shape) and all(u >= 1 for u in unit) - else [1] * len(shape) - ) - upgraded = [ - full_span_chunk_size(extent, int(u)) if size == 0 else int(size) - for size, extent, u in zip(chunk_shape, shape, units, strict=True) - ] - readings = "; ".join( - f"{json.dumps(chunk_shape[axis])} on axis {axis} as " - + ("1" if chunk_shape[axis] else f"one chunk spanning the axis ({upgraded[axis]})") - for axis in invalid_axes - ) - message = ( - f"The stored chunk shape {json.dumps(list(chunk_shape))} is invalid: chunk sizes " - f"must be integers of at least 1. It is read as {upgraded}, reading {readings}." + edges: list[int] = [] + readings: list[str] = [] + axes = zip(stored, spans, chain(units, repeat(1)), strict=False) + for axis, (size, span, unit) in enumerate(axes): + read = _read_chunk_size(size, span, unit) + if read is None: + return None + edge, how = read + edges.append(edge) + if how is not None: + readings.append(f"{json.dumps(size)} on axis {axis} as {how}") + if not readings: + return edges, None + return edges, ( + f"The stored {name} {json.dumps(list(stored))} is invalid: chunk sizes must be " + f"integers of at least 1. It is read as {edges}, reading {'; '.join(readings)}." ) - if any(chunk_shape[axis] == 0 and shape[axis] > 0 for axis in invalid_axes): - message += ( - " No chunk can be stored under a chunk size of 0, so the array holds only its " - "fill value." - ) - return upgraded, message def _invalid_chunk_sizes_v2(doc: ArrayDocument) -> tuple[ArrayDocument, str] | None: - read = _read_invalid_chunk_sizes(doc.get("chunks"), doc.get("shape"), None) - if read is None: + shape = doc.get("shape") + if not _is_int_list(shape): return None - chunks, reading = read - return {**doc, "chunks": chunks}, reading + match _read_chunk_shape(doc.get("chunks"), shape, (), "chunk shape"): + case chunks, str(reading): + return {**doc, "chunks": chunks}, reading + return None -def _sharding_chunk_shape(codecs: JSON) -> JSON: - """The inner chunk shape of a sharding codec in a Zarr format 3 codec list, if any.""" - if isinstance(codecs, Sequence) and not isinstance(codecs, str): - for codec in codecs: +def _sharding_codec(doc: ArrayDocument) -> tuple[Sequence[JSON], int, Mapping[str, JSON]] | None: + """The codec list of a Zarr format 3 array document, with the position and the + configuration of its sharding codec, if it has one.""" + codecs = doc.get("codecs") + if isinstance(codecs, list | tuple): + for index, codec in enumerate(codecs): if isinstance(codec, Mapping) and codec.get("name") == "sharding_indexed": configuration = codec.get("configuration") if isinstance(configuration, Mapping): - return configuration.get("chunk_shape") + return codecs, index, configuration + return None + + +def _read_inner_chunk_shape(doc: ArrayDocument) -> tuple[list[int], str | None] | None: + """Read the inner chunk shape of a sharded array. No stored inner chunk size of 0 + or `false` is known, so the spans of its axes are not given.""" + shape = doc.get("shape") + sharding = _sharding_codec(doc) + if sharding is None or not _is_int_list(shape): + return None + _, _, configuration = sharding + return _read_chunk_shape( + configuration.get("chunk_shape"), + [None] * len(shape), + (), + "inner chunk shape of the sharding codec", + ) + + +def _invalid_inner_chunk_sizes_v3(doc: ArrayDocument) -> tuple[ArrayDocument, str] | None: + match _read_inner_chunk_shape(doc), _sharding_codec(doc): + case (inner, str(reading)), (codecs, index, configuration): + # `_sharding_codec` found a mapping at `index`. + codec = cast("Mapping[str, JSON]", codecs[index]) + upgraded = {**codec, "configuration": {**configuration, "chunk_shape": inner}} + return {**doc, "codecs": [*codecs[:index], upgraded, *codecs[index + 1 :]]}, reading return None def _invalid_chunk_sizes_v3(doc: ArrayDocument) -> tuple[ArrayDocument, str] | None: grid = doc.get("chunk_grid") - if not isinstance(grid, Mapping) or grid.get("name") != "regular": + shape = doc.get("shape") + if not (isinstance(grid, Mapping) and grid.get("name") == "regular" and _is_int_list(shape)): return None configuration = grid.get("configuration") if not isinstance(configuration, Mapping): return None - read = _read_invalid_chunk_sizes( - configuration.get("chunk_shape"), - doc.get("shape"), - _sharding_chunk_shape(doc.get("codecs")), - ) - if read is None: - return None - chunk_shape, reading = read - return { - **doc, - "chunk_grid": {**grid, "configuration": {**configuration, "chunk_shape": chunk_shape}}, - }, reading + inner = _read_inner_chunk_shape(doc) + units = () if inner is None else inner[0] + match _read_chunk_shape(configuration.get("chunk_shape"), shape, units, "chunk shape"): + case chunk_shape, str(reading): + upgraded = {**configuration, "chunk_shape": chunk_shape} + return {**doc, "chunk_grid": {**grid, "configuration": upgraded}}, reading + return None V2_ARRAY_UPGRADES: Final[tuple[Upgrade, ...]] = (_invalid_chunk_sizes_v2,) -V3_ARRAY_UPGRADES: Final[tuple[Upgrade, ...]] = (_invalid_chunk_sizes_v3,) +V3_ARRAY_UPGRADES: Final[tuple[Upgrade, ...]] = ( + _invalid_inner_chunk_sizes_v3, + _invalid_chunk_sizes_v3, +) def upgrade_array_document( diff --git a/src/zarr/core/metadata/v2.py b/src/zarr/core/metadata/v2.py index c648f271e6..b99ee442b2 100644 --- a/src/zarr/core/metadata/v2.py +++ b/src/zarr/core/metadata/v2.py @@ -38,11 +38,12 @@ ZARRAY_JSON, ZATTRS_JSON, MemoryOrder, + parse_chunk_shape, parse_shapelike, ) from zarr.core.config import config, parse_indexing_order from zarr.core.json_parse import parse_field -from zarr.core.metadata.common import parse_attributes, parse_chunk_edge +from zarr.core.metadata.common import parse_attributes from zarr.core.metadata.upgrades import V2_ARRAY_UPGRADES, upgrade_array_document, warn_readings @@ -147,7 +148,9 @@ def to_buffer_dict(self, prototype: BufferPrototype) -> dict[str, Buffer]: } @classmethod - def from_dict(cls, data: dict[str, Any]) -> ArrayV2Metadata: + def from_dict(cls, data: dict[str, Any], *, path: str | None = None) -> ArrayV2Metadata: + """Read a stored `.zarray` document (with its attributes). An invalid document + that `zarr.core.metadata.upgrades` can read warns, naming the array at `path`.""" upgraded, readings = upgrade_array_document(data, V2_ARRAY_UPGRADES) data = dict(upgraded) _data = data.copy() @@ -202,7 +205,7 @@ def from_dict(cls, data: dict[str, Any]) -> ArrayV2Metadata: _data = {k: v for k, v in _data.items() if k in expected} metadata = cls(**_data) - warn_readings(readings) + warn_readings(readings, path) return metadata def to_dict(self) -> dict[str, JSON]: @@ -325,8 +328,8 @@ def parse_compressor(data: object) -> Numcodec | None: def parse_chunks(chunks: Iterable[int], shape: tuple[int, ...]) -> tuple[int, ...]: - """Check a chunk shape: one chunk edge length (an integer >= 1) per array axis.""" - chunks_parsed = tuple(parse_chunk_edge(size, axis) for axis, size in enumerate(chunks)) + """Check a chunk shape: one chunk edge length (an `int` >= 1) per array axis.""" + chunks_parsed = parse_chunk_shape(chunks) if len(chunks_parsed) != len(shape): raise ValueError( f"The `shape` and `chunks` attributes must have the same length. " diff --git a/src/zarr/core/metadata/v3.py b/src/zarr/core/metadata/v3.py index b6929220ce..f1cbfcbe63 100644 --- a/src/zarr/core/metadata/v3.py +++ b/src/zarr/core/metadata/v3.py @@ -26,6 +26,7 @@ NamedRequiredConfig, compress_rle, expand_rle, + parse_chunk_edge, parse_named_configuration, parse_shapelike, validate_rectilinear_edges, @@ -35,7 +36,7 @@ from zarr.core.dtype import VariableLengthUTF8, ZDType, get_data_type_from_json from zarr.core.dtype.common import check_dtype_spec_v3 from zarr.core.json_parse import parse_field -from zarr.core.metadata.common import parse_attributes, parse_chunk_edge +from zarr.core.metadata.common import parse_attributes from zarr.core.metadata.upgrades import ( V3_ARRAY_UPGRADES, upgrade_array_document, @@ -238,19 +239,13 @@ def _validate_chunk_shapes( """ result: list[int | tuple[int, ...]] = [] for dim_idx, dim_spec in enumerate(chunk_shapes): - if isinstance(dim_spec, int): - result.append(parse_chunk_edge(dim_spec, dim_idx)) - else: - edges = tuple(dim_spec) + if isinstance(dim_spec, Iterable): + edges = tuple(parse_chunk_edge(edge, dim_idx) for edge in dim_spec) if not edges: raise ValueError(f"Dimension {dim_idx} has no chunk edges.") - bad = [i for i, e in enumerate(edges) if isinstance(e, bool) or e < 1] - if bad: - raise ValueError( - f"Dimension {dim_idx} has invalid edge lengths at indices {bad}: " - f"{[edges[i] for i in bad]}" - ) result.append(edges) + else: + result.append(parse_chunk_edge(dim_spec, dim_idx)) return tuple(result) @@ -367,18 +362,10 @@ def from_dict(cls, data: RectilinearChunkGridMetadataJSON) -> Self: # type: ign configuration = data["configuration"] validate_rectilinear_kind(configuration.get("kind")) raw_shapes = configuration["chunk_shapes"] - parsed: list[int | tuple[int, ...]] = [] - for dim_spec in raw_shapes: - if isinstance(dim_spec, int): - if dim_spec < 1: - raise ValueError(f"Integer chunk edge length must be >= 1, got {dim_spec}") - parsed.append(dim_spec) - elif isinstance(dim_spec, list): - parsed.append(tuple(expand_rle(dim_spec))) - else: - raise TypeError( - f"Invalid chunk_shapes entry: expected int or list, got {type(dim_spec)}" - ) + parsed = [ + tuple(expand_rle(dim_spec)) if isinstance(dim_spec, list) else dim_spec + for dim_spec in raw_shapes + ] return cls(chunk_shapes=tuple(parsed)) @@ -631,7 +618,9 @@ def to_buffer_dict(self, prototype: BufferPrototype) -> dict[str, Buffer]: return {ZARR_JSON: json_to_buffer(self.to_dict(), prototype=prototype, indent=indent)} @classmethod - def from_dict(cls, data: dict[str, JSON]) -> Self: + def from_dict(cls, data: dict[str, JSON], *, path: str | None = None) -> Self: + """Read a stored `zarr.json` array document. An invalid document that + `zarr.core.metadata.upgrades` can read warns, naming the array at `path`.""" upgraded, readings = upgrade_array_document(data, V3_ARRAY_UPGRADES) # a new dict, because we are modifying it _data = dict(upgraded) @@ -688,7 +677,7 @@ def from_dict(cls, data: dict[str, JSON]) -> Self: extra_fields=allowed_extra_fields, storage_transformers=_data_typed.get("storage_transformers", ()), # type: ignore[arg-type] ) - warn_readings(readings) + warn_readings(readings, path) return metadata def to_dict(self) -> dict[str, JSON]: diff --git a/tests/test_metadata/test_upgrades.py b/tests/test_metadata/test_upgrades.py index a82a8a7717..fabf22b1b8 100644 --- a/tests/test_metadata/test_upgrades.py +++ b/tests/test_metadata/test_upgrades.py @@ -11,17 +11,20 @@ import pytest import zarr +from zarr.codecs import ShardingCodec from zarr.core.metadata import ArrayV2Metadata, ArrayV3Metadata from zarr.core.metadata.upgrades import ( + RESAVE_HINT, V2_ARRAY_UPGRADES, V3_ARRAY_UPGRADES, upgrade_array_document, ) -from zarr.core.metadata.v3 import RegularChunkGridMetadata +from zarr.core.metadata.v3 import RectilinearChunkGridMetadata, RegularChunkGridMetadata from zarr.dtype import Int16 from zarr.errors import ZarrUserWarning if TYPE_CHECKING: + from collections.abc import Callable from pathlib import Path from zarr.core.common import JSON @@ -77,21 +80,58 @@ def _stored_chunks(doc: dict[str, Any]) -> Any: ) +def _chunk_shapes(metadata: ArrayV2Metadata | ArrayV3Metadata) -> tuple[Any, Any]: + """The chunk shape of `metadata` and, if it is sharded, its inner chunk shape.""" + if isinstance(metadata, ArrayV2Metadata): + return metadata.chunks, None + assert isinstance(metadata.chunk_grid, RegularChunkGridMetadata) + inner = next((c.chunk_shape for c in metadata.codecs if isinstance(c, ShardingCodec)), None) + return metadata.chunk_grid.chunk_shape, inner + + @pytest.mark.parametrize( ("doc", "expected", "warning"), [ - (_v2_doc([10, 10], [4, 5]), [4, 5], None), - (_v3_doc([0, 0], [1, 1]), [1, 1], None), - (_v3_doc([10], [4], inner=[2]), [4], None), - (_v2_doc([0, 4], [0, 4]), [1, 4], "0 on axis 0 as one chunk spanning the axis"), - (_v3_doc([0], [False]), [1], "false on axis 0 as one chunk spanning the axis"), - (_v2_doc([5], [True]), [1], "true on axis 0 as 1"), - (_v3_doc([5, 4], [True, 4]), [1, 4], "true on axis 0 as 1"), - (_v2_doc([3], [0]), [3], "holds only its fill value"), - (_v3_doc([4, 3], [4, 0]), [4, 3], "holds only its fill value"), - (_v3_doc([0], [0], inner=[4]), [4], "spanning the axis \\(4\\)"), - (_v3_doc([10], [0], inner=[4]), [12], "spanning the axis \\(12\\)"), - (_v3_doc([0, 3], [0, 3], inner=[2, 3]), [2, 3], "spanning the axis \\(2\\)"), + (_v2_doc([10, 10], [4, 5]), ((4, 5), None), None), + (_v3_doc([0, 0], [1, 1]), ((1, 1), None), None), + (_v3_doc([10], [4], inner=[2]), ((4,), (2,)), None), + ( + _v2_doc([0, 4], [0, 4]), + ((1, 4), None), + r"0 on axis 0 as one chunk spanning the axis \(1\)\.", + ), + ( + _v3_doc([0], [False]), + ((1,), None), + r"false on axis 0 as one chunk spanning the axis \(1\)\.", + ), + (_v2_doc([5], [True]), ((1,), None), "true on axis 0 as 1"), + ( + _v3_doc([5, 4], [True, 4]), + ((1, 4), None), + r"\[true, 4\] is invalid.*true on axis 0 as 1", + ), + ( + _v2_doc([3], [0]), + ((3,), None), + r"spanning the axis \(3\), and .* holds only its fill value", + ), + (_v3_doc([4, 3], [4, 0]), ((4, 3), None), "0 on axis 1 as .* holds only its fill value"), + (_v3_doc([0], [0], inner=[4]), ((4,), (4,)), r"spanning the axis \(4\)\."), + ( + _v3_doc([10], [0], inner=[4]), + ((12,), (4,)), + r"spanning the axis \(12\), and .* holds only its fill value", + ), + (_v3_doc([0, 3], [0, 3], inner=[2, 3]), ((2, 3), (2, 3)), r"spanning the axis \(2\)\."), + ( + _v3_doc([5], [True], inner=[True]), + ((1,), (1,)), + ( + r"^The stored inner chunk shape of the sharding codec \[true\] is invalid: .* " + r"The stored chunk shape \[true\] is invalid: " + ), + ), ], ids=[ "v2-valid", @@ -106,85 +146,142 @@ def _stored_chunks(doc: dict[str, Any]) -> Any: "v3-sharded-zero-empty-axis", "v3-sharded-zero-grown-axis", "v3-sharded-zero-2d", + "v3-sharded-true-inner-and-outer", ], ) def test_upgrade_array_document( - doc: dict[str, JSON], expected: list[int], warning: str | None + doc: dict[str, JSON], expected: tuple[Any, Any], warning: str | None ) -> None: """Valid documents pass unchanged and silently. A stored chunk size of 0 or `false` is read as one chunk spanning the axis (a multiple of the inner chunk when sharded) - and `true` as 1; `from_dict` warns once with the reading and how to re-save.""" + and `true` as 1, in the chunk shape and in a sharding codec's inner chunk shape; + `from_dict` warns once for the document, naming the array, saying how each part was + read (and that the array holds only its fill value where a chunk size of 0 was + stored for a non-empty axis) and how to re-save.""" upgrades = V2_ARRAY_UPGRADES if doc["zarr_format"] == 2 else V3_ARRAY_UPGRADES upgraded, readings = upgrade_array_document(doc, upgrades) - assert _stored_chunks(dict(upgraded)) == expected - assert {k: v for k, v in upgraded.items() if k not in ("chunks", "chunk_grid")} == { - k: v for k, v in doc.items() if k not in ("chunks", "chunk_grid") + assert {k: v for k, v in upgraded.items() if k not in ("chunks", "chunk_grid", "codecs")} == { + k: v for k, v in doc.items() if k not in ("chunks", "chunk_grid", "codecs") } metadata_cls = ArrayV2Metadata if doc["zarr_format"] == 2 else ArrayV3Metadata with warnings.catch_warnings(record=True) as record: warnings.simplefilter("always") - metadata_cls.from_dict(dict(doc)) + metadata = metadata_cls.from_dict(dict(doc), path="group/array") + assert _chunk_shapes(metadata) == expected messages = [str(w.message) for w in record] if warning is None: assert upgraded is doc assert readings == [] assert messages == [] else: - assert len(readings) == 1 - assert re.search(warning, readings[0]) assert len(messages) == 1 - assert messages[0].startswith(readings[0]) - assert "update_attributes({})" in messages[0] - assert "zarr.consolidate_metadata" in messages[0] + message = messages[0] + assert message.startswith("Array 'group/array': ") + assert re.search(warning, message.removeprefix("Array 'group/array': ")) + assert ("holds only its fill value" in message) == ("fill value" in warning) + assert message.endswith(RESAVE_HINT) @pytest.mark.parametrize( - "doc", - [_v2_doc([4], [0]) | {"dtype": " None: - """A document that is invalid after its upgrade raises its own error, without first - warning how it was read.""" +def test_invalid_upgraded_document_raises_without_warning(doc: dict[str, JSON], error: str) -> None: + """A document the upgrades read that the metadata constructor then rejects raises + that error, without first warning how it was read.""" metadata_cls = ArrayV2Metadata if doc["zarr_format"] == 2 else ArrayV3Metadata + assert upgrade_array_document(doc, V2_ARRAY_UPGRADES + V3_ARRAY_UPGRADES)[1] with warnings.catch_warnings(): warnings.simplefilter("error", ZarrUserWarning) - with pytest.raises(ValueError): + with pytest.raises(ValueError, match=error): metadata_cls.from_dict(doc) -@pytest.mark.parametrize("doc", [_v2_doc([4], [-1]), _v3_doc([4], [-1])], ids=["v2", "v3"]) -def test_stored_negative_chunk_size_rejected(doc: dict[str, JSON]) -> None: - metadata_cls = ArrayV2Metadata if doc["zarr_format"] == 2 else ArrayV3Metadata - with pytest.raises(ValueError, match="chunk edge length must be an integer >= 1, got -1"): - metadata_cls.from_dict(doc) - - -@pytest.mark.parametrize("doc", [_v2_doc([4, 4], [0]), _v3_doc([4, 4], [0])], ids=["v2", "v3"]) -def test_stored_chunk_shape_ndim_mismatch_rejected(doc: dict[str, JSON]) -> None: - """A chunk shape with the wrong number of axes is not upgraded, only rejected.""" +@pytest.mark.parametrize( + ("doc", "error"), + [ + (_v2_doc([4], [-1]), "Dimension 0: Chunk edge length must be >= 1, got -1"), + (_v3_doc([4, 4], [0]), "Dimension 0: Chunk edge length must be >= 1, got 0"), + (_v3_doc([4], [4.0]), "Dimension 0: Chunk edge length must be an int, got 4.0"), + (_v3_doc([4], [0], inner=[0]), "Dimension 0: Chunk edge length must be >= 1, got 0"), + ], + ids=["negative", "ndim-mismatch", "float", "sharded-inner-zero"], +) +def test_stored_chunk_shape_not_upgraded(doc: dict[str, JSON], error: str) -> None: + """Invalid chunk sizes no known writer stored, and chunk shapes with the wrong + number of axes, are not upgraded, only rejected.""" metadata_cls = ArrayV2Metadata if doc["zarr_format"] == 2 else ArrayV3Metadata with warnings.catch_warnings(): warnings.simplefilter("error", ZarrUserWarning) - with pytest.raises(ValueError, match="chunk edge length|same length|same number"): + with pytest.raises((TypeError, ValueError), match=re.escape(error)): metadata_cls.from_dict(doc) -@pytest.mark.parametrize("size", [0, True], ids=["zero", "true"]) -def test_constructor_rejects_invalid_chunk_size(size: int) -> None: - """Metadata built in code is strict: 0 and `True` are rejected, without a warning.""" +def test_v2_constructor_rejects_chunks_of_wrong_length() -> None: + with pytest.raises(ValueError, match="`chunks` has length 1, but `shape` has length 2"): + ArrayV2Metadata(shape=(4, 4), chunks=(2,), dtype=Int16(), fill_value=0, order="C") + + +def _v2_metadata(chunks: Any) -> ArrayV2Metadata: + return ArrayV2Metadata(shape=(4,), chunks=chunks, dtype=Int16(), fill_value=0, order="C") + + +def _rectilinear(chunk_shapes: tuple[Any, ...]) -> RectilinearChunkGridMetadata: + with zarr.config.set({"array.rectilinear_chunks": True}): + return RectilinearChunkGridMetadata(chunk_shapes=chunk_shapes) + + +def _rectilinear_from_dict(chunk_shapes: list[Any]) -> RectilinearChunkGridMetadata: + with zarr.config.set({"array.rectilinear_chunks": True}): + return RectilinearChunkGridMetadata.from_dict( + { + "name": "rectilinear", + "configuration": {"kind": "inline", "chunk_shapes": chunk_shapes}, + } + ) + + +CHUNK_EDGE_SITES: dict[str, Callable[[Any], object]] = { + "regular": lambda size: RegularChunkGridMetadata(chunk_shape=(size,)), + "v2": lambda size: _v2_metadata((size,)), + "rectilinear-bare": lambda size: _rectilinear((size,)), + "rectilinear-edge": lambda size: _rectilinear(((4, size),)), + "rectilinear-bare-json": lambda size: _rectilinear_from_dict([size]), + "rectilinear-edge-json": lambda size: _rectilinear_from_dict([[4, size]]), + "rectilinear-rle-json": lambda size: _rectilinear_from_dict([[[size, 2]]]), + "sharding-inner": lambda size: ShardingCodec(chunk_shape=(size,)), +} + + +@pytest.mark.parametrize("site", CHUNK_EDGE_SITES) +@pytest.mark.parametrize("size", [True, False, 4.0, np.int64(4), "4"]) +def test_metadata_rejects_non_int_chunk_edge(site: str, size: object) -> None: + """Metadata built in code takes chunk edge lengths as `int`s only, everywhere.""" + with pytest.raises( + TypeError, match=re.escape(f"Chunk edge length must be an int, got {size!r}") + ): + CHUNK_EDGE_SITES[site](size) + + +@pytest.mark.parametrize("site", CHUNK_EDGE_SITES) +@pytest.mark.parametrize("size", [0, -1]) +def test_metadata_rejects_chunk_edge_below_one(site: str, size: int) -> None: + """Metadata built in code is strict: a chunk edge length below 1 is rejected, + without a warning.""" with warnings.catch_warnings(): warnings.simplefilter("error", ZarrUserWarning) - with pytest.raises(ValueError, match=f"got {size!r}"): - RegularChunkGridMetadata(chunk_shape=(size,)) - with pytest.raises(ValueError, match=f"got {size!r}"): - ArrayV2Metadata( - shape=(0,), - chunks=(size,), - dtype=Int16(), - fill_value=0, - order="C", - ) + with pytest.raises(ValueError, match=f"Chunk edge length must be >= 1, got {size}"): + CHUNK_EDGE_SITES[site](size) + + +@pytest.mark.parametrize("chunks", [4, np.int64(4)]) +def test_v2_constructor_rejects_scalar_chunks(chunks: object) -> None: + with pytest.raises(TypeError, match="A chunk shape must be a sequence of chunk edge lengths"): + _v2_metadata(chunks) def _rewrite_doc(path: Path, zarr_format: Literal[2, 3], edit: Any) -> None: @@ -198,26 +295,10 @@ def _rewrite_doc(path: Path, zarr_format: Literal[2, 3], edit: Any) -> None: ("zarr_format", "shape", "stored", "inner", "expected"), [ (2, (0, 4), [0, 4], None, (1, 4)), - (2, (3,), [0], None, (3,)), - (2, (5,), [True], None, (1,)), - (3, (0,), [False], None, (1,)), - (3, (3,), [0], None, (3,)), (3, (5,), [True], None, (1,)), - (3, (0,), [0], (4,), (4,)), (3, (10,), [0], (4,), (12,)), - (3, (0, 3), [0, 3], (2, 3), (2, 3)), - ], - ids=[ - "v2-empty-2d", - "v2-grown", - "v2-true", - "v3-false-empty", - "v3-grown", - "v3-true", - "v3-sharded-empty", - "v3-sharded-grown", - "v3-sharded-2d", ], + ids=["v2-empty-2d", "v3-true", "v3-sharded-grown"], ) def test_legacy_chunk_size_round_trip( tmp_path: Path, @@ -250,7 +331,7 @@ def store_legacy(doc: dict[str, Any]) -> None: _rewrite_doc(path, zarr_format, store_legacy) - with pytest.warns(ZarrUserWarning, match="is read as"): + with pytest.warns(ZarrUserWarning, match=r"^Array '.*legacy\.zarr': .* is read as"): arr = zarr.open_array(store=path, mode="a") assert (arr.shards or arr.chunks) == expected np.testing.assert_array_equal(arr[...], data) @@ -267,43 +348,52 @@ def store_legacy(doc: dict[str, Any]) -> None: @pytest.mark.filterwarnings("ignore:Consolidated metadata is currently not part:UserWarning") @pytest.mark.parametrize("zarr_format", [2, 3]) def test_legacy_chunk_size_consolidated(tmp_path: Path, zarr_format: Literal[2, 3]) -> None: - """Consolidated metadata goes through the same upgrade; re-saving the array and - consolidating again leaves a group that opens without a warning.""" + """Consolidated metadata goes through the same upgrade, with one warning naming each + array; re-saving the arrays and consolidating again leaves a group that opens + without a warning.""" path = tmp_path / "group.zarr" group = zarr.open_group(path, mode="w", zarr_format=zarr_format) - group.create_array("a", shape=(0,), chunks=(1,), dtype="int32") + names = ("a", "b") + for name in names: + group.create_array(name, shape=(0,), chunks=(1,), dtype="int32") zarr.consolidate_metadata(path) # What zarr-python wrote for `chunks=(0,)` on an empty array, in both copies. - if zarr_format == 2: - _rewrite_doc(path / "a", 2, lambda doc: doc.update(chunks=[0])) - zmetadata = json.loads((path / ".zmetadata").read_text()) - zmetadata["metadata"]["a/.zarray"]["chunks"] = [0] - (path / ".zmetadata").write_text(json.dumps(zmetadata)) - else: - _rewrite_doc( - path / "a", 3, lambda doc: doc["chunk_grid"]["configuration"].update(chunk_shape=[0]) - ) - _rewrite_doc( - path, - 3, - lambda doc: doc["consolidated_metadata"]["metadata"]["a"]["chunk_grid"][ - "configuration" - ].update(chunk_shape=[0]), - ) + for name in names: + if zarr_format == 2: + _rewrite_doc(path / name, 2, lambda doc: doc.update(chunks=[0])) + zmetadata = json.loads((path / ".zmetadata").read_text()) + zmetadata["metadata"][f"{name}/.zarray"]["chunks"] = [0] + (path / ".zmetadata").write_text(json.dumps(zmetadata)) + else: + _rewrite_doc( + path / name, + 3, + lambda doc: doc["chunk_grid"]["configuration"].update(chunk_shape=[0]), + ) + _rewrite_doc( + path, + 3, + lambda doc, name=name: doc["consolidated_metadata"]["metadata"][name]["chunk_grid"][ + "configuration" + ].update(chunk_shape=[0]), + ) - with pytest.warns(ZarrUserWarning, match="zarr.consolidate_metadata"): + with pytest.warns(ZarrUserWarning, match="zarr.consolidate_metadata") as record: group = zarr.open_group(path, mode="r+") - array = group["a"] - assert isinstance(array, zarr.Array) - assert array.chunks == (1,) - - array.update_attributes({}) + assert sorted(str(w.message).split(":")[0] for w in record) == ["Array 'a'", "Array 'b'"] + for name in names: + array = group[name] + assert isinstance(array, zarr.Array) + assert array.chunks == (1,) + array.update_attributes({}) zarr.consolidate_metadata(path) with warnings.catch_warnings(): warnings.simplefilter("error", ZarrUserWarning) warnings.filterwarnings("ignore", "Consolidated metadata is currently not part") for use_consolidated in (True, False): - reopened = zarr.open_group(path, mode="r", use_consolidated=use_consolidated)["a"] - assert isinstance(reopened, zarr.Array) - assert reopened.chunks == (1,) + reopened = zarr.open_group(path, mode="r", use_consolidated=use_consolidated) + for name in names: + array = reopened[name] + assert isinstance(array, zarr.Array) + assert array.chunks == (1,) diff --git a/tests/test_unified_chunk_grid.py b/tests/test_unified_chunk_grid.py index 6e2379e2fd..4a8d6f791e 100644 --- a/tests/test_unified_chunk_grid.py +++ b/tests/test_unified_chunk_grid.py @@ -547,19 +547,19 @@ def test_rle_expand_rejects_invalid(rle_input: list[Any], match: str) -> None: expand_rle(rle_input) -# -- expand_rle handles JSON floats -- - - -def test_expand_rle_bare_integer_floats_accepted() -> None: - """JSON parsers may emit 10.0 for the integer 10; expand_rle should handle it.""" - result = expand_rle([10.0, 20.0]) # type: ignore[list-item] - assert result == [10, 20] - - -def test_expand_rle_pair_with_float_count() -> None: - """expand_rle accepts float repeat counts that are integer-valued""" - result = expand_rle([[10, 3.0]]) # type: ignore[list-item] - assert result == [10, 10, 10] +@pytest.mark.parametrize( + ("rle_input", "match"), + [ + ([10.0], "Chunk edge length must be an int, got 10.0"), + ([True], "Chunk edge length must be an int, got True"), + ([[10, 3.0]], "RLE repeat count must be an int, got 3.0"), + ], + ids=["float-edge", "bool-edge", "float-count"], +) +def test_rle_expand_rejects_non_int(rle_input: list[Any], match: str) -> None: + """expand_rle takes JSON integers only: no stored document holds integral floats.""" + with pytest.raises(TypeError, match=match): + expand_rle(rle_input) # --------------------------------------------------------------------------- From 192f86c446257ab0255976cae068c3218284f432 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Sat, 26 Sep 2026 02:08:38 +0200 Subject: [PATCH 31/76] refactor(chunk-grids): start chunk guessing from the one full-span rule `_guess_regular_chunks` clamped zero-length axes with `np.maximum`, restating `full_span_chunk_size` with unit 1. Fold the zero-span normalizer tests into the `normalize_chunks_1d` tables. Assisted-by: ClaudeCode:claude-opus-5-5 Co-Authored-By: Claude Opus 5.5 --- src/zarr/core/chunk_grids.py | 6 +++--- tests/test_chunk_grids.py | 25 +++++++++---------------- 2 files changed, 12 insertions(+), 19 deletions(-) diff --git a/src/zarr/core/chunk_grids.py b/src/zarr/core/chunk_grids.py index 17b3225725..399943f72d 100644 --- a/src/zarr/core/chunk_grids.py +++ b/src/zarr/core/chunk_grids.py @@ -698,12 +698,12 @@ def _guess_regular_chunks( if isinstance(shape, int): shape = (shape,) + # Start from one chunk spanning each axis, then halve axes until the chunk is small enough. + chunks = np.array([full_span_chunk_size(s) for s in shape], dtype="=f8") if typesize == 0: - return tuple(full_span_chunk_size(s) for s in shape) + return tuple(int(x) for x in chunks) ndims = len(shape) - # require chunks to have non-zero length for all dimensions - chunks = np.maximum(np.array(shape, dtype="=f8"), 1) # Determine the optimal chunk size in bytes using a PyTables expression. # This is kept as a float. diff --git a/tests/test_chunk_grids.py b/tests/test_chunk_grids.py index d171937898..207f121d11 100644 --- a/tests/test_chunk_grids.py +++ b/tests/test_chunk_grids.py @@ -210,6 +210,10 @@ def test_chunk_layout_nested() -> None: ExpectFail( input=([10, 20], 100), exception=ValueError, id="wrong-sum", msg="do not sum to span" ), + # Only a zero-length span keeps edges that do not sum to it. + ExpectFail( + input=([3, 5], 1), exception=ValueError, id="over-sum", msg="do not sum to span 1" + ), # Nested/RLE form for a single dim is rejected with offending indices. ExpectFail( input=([[3, 3], 1], 7), @@ -319,6 +323,11 @@ def test_normalize_chunks_nd_errors(case: ExpectFail[tuple[Any, tuple[int, ...]] output=VaryingDimension([10, 20, 70], extent=100), id="explicit-irregular", ), + # on a zero-length span any non-empty list of positive edges is kept: the + # chunks the axis grows into. + Expect( + input=([3, 5], 0), output=VaryingDimension([3, 5], extent=0), id="explicit-zero-span" + ), ], ids=lambda c: c.id, ) @@ -477,19 +486,3 @@ def test_rectilinear_zero_extent_matches_resize() -> None: np.testing.assert_array_equal(created[...], np.arange(3)) np.testing.assert_array_equal(resized[...], np.arange(3)) assert created.write_chunk_sizes == resized.write_chunk_sizes == ((2, 1),) - - -def test_normalize_chunks_1d_zero_span_accepts_any_edges() -> None: - """On a zero-length span the explicit edge list is stored verbatim.""" - dim = normalize_chunks_1d([3, 5], span=0) - assert isinstance(dim, VaryingDimension) - assert dim.edges == (3, 5) - assert dim.extent == 0 - assert dim.nchunks == 0 - assert dim.resize(4) == VaryingDimension([3, 5], extent=4) - - -def test_normalize_chunks_1d_nonzero_span_still_requires_exact_sum() -> None: - """Relaxing the sum rule for span 0 must not leak into positive spans.""" - with pytest.raises(ValueError, match="do not sum to span 1"): - normalize_chunks_1d([3, 5], span=1) From 2e878cf41e346857a40ec03458340e94ca249d80 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Sat, 26 Sep 2026 02:08:39 +0200 Subject: [PATCH 32/76] test: run the array lifecycle state machine in the slow Hypothesis job Add it to `just hypothesis`. Re-saving metadata is always enabled and must leave valid metadata unchanged; appends favour an axis stored with chunk size 0, which raises how often a legacy axis grows before it is re-saved. A fixed example pins that a write to a shard kept by a shrink leaves its cells beyond the shape alone. Assisted-by: ClaudeCode:claude-opus-5-5 Co-Authored-By: Claude Opus 5.5 --- Justfile | 2 +- tests/test_array.py | 12 ++++++++++++ tests/test_array_stateful.py | 37 ++++++++++++++++++++++-------------- 3 files changed, 36 insertions(+), 15 deletions(-) diff --git a/Justfile b/Justfile index 6e56644207..b3f187e231 100644 --- a/Justfile +++ b/Justfile @@ -57,7 +57,7 @@ coverage-serve *args: # Run slow Hypothesis tests and write coverage.xml hypothesis *args: - hatch run {{ quote(hatch_env) }}:coverage run --source=src -m pytest -nauto --run-slow-hypothesis tests/test_properties.py tests/test_store/test_stateful* "$@" + hatch run {{ quote(hatch_env) }}:coverage run --source=src -m pytest -nauto --run-slow-hypothesis tests/test_properties.py tests/test_store/test_stateful* tests/test_array_stateful.py "$@" hatch run {{ quote(hatch_env) }}:coverage xml # Validate executable documentation code blocks diff --git a/tests/test_array.py b/tests/test_array.py index 3ae502a4e8..b153f24c0f 100644 --- a/tests/test_array.py +++ b/tests/test_array.py @@ -732,6 +732,18 @@ def test_resize_1d(store: MemoryStore, zarr_format: ZarrFormat) -> None: assert new_shape == result.shape +@pytest.mark.parametrize("chunks", [(1,), (2,), (4,)]) +def test_resize_sharded_keeps_cells_beyond_shape(chunks: tuple[int, ...]) -> None: + """A shard kept by a shrinking resize keeps its cells beyond the new shape, and a + later write to the shard leaves them alone, so they come back when the array grows.""" + arr = zarr.create_array({}, shape=(4,), chunks=chunks, shards=(4,), dtype="int16", fill_value=0) + arr[:] = [1, 2, 3, 4] + arr.resize((2,)) + arr[:] = [9, 9] + arr.resize((4,)) + np.testing.assert_array_equal(arr[:], [9, 9, 3, 4]) + + @pytest.mark.parametrize("store", ["memory"], indirect=True) def test_resize_2d(store: MemoryStore, zarr_format: ZarrFormat) -> None: z = zarr.create( diff --git a/tests/test_array_stateful.py b/tests/test_array_stateful.py index 6e9bea8f8b..0ea4da0167 100644 --- a/tests/test_array_stateful.py +++ b/tests/test_array_stateful.py @@ -30,7 +30,6 @@ RuleBasedStateMachine, initialize, invariant, - precondition, rule, ) @@ -79,8 +78,8 @@ def __init__(self) -> None: self.fill = 0 # What the store holds, indexed like the array and extending past its shape. self.stored: np.ndarray[Any, np.dtype[np.int16]] = np.zeros((0,), dtype=DTYPE) - # A store with an invalid stored chunk size warns until its metadata is re-saved. - self.expect_open_warning = False + # Axes stored with chunk size 0; the store warns until its metadata is re-saved. + self.legacy_axes: list[int] = [] # -------------------------------------------------------------- creation @initialize(data=st.data()) @@ -130,13 +129,12 @@ def create(self, data: st.DataObject) -> None: # A stored chunk size of 0, as zarr-python wrote for arrays created with a # zero-length axis; older releases could then grow the axis without storing # a chunk, so any extent is possible. Sharded arrays store it in the outer grid. - zero_axes = data.draw( + self.legacy_axes = data.draw( st.lists(st.integers(0, len(shape) - 1), min_size=1, unique=True), label="axes stored with chunk size 0", ) stored_zero = data.draw(st.sampled_from([0, False]), label="stored zero") - self._rewrite_stored_chunks(zarr_format, zero_axes, stored_zero) - self.expect_open_warning = True + self._rewrite_stored_chunks(zarr_format, self.legacy_axes, stored_zero) event("legacy zero chunk size") def _rewrite_stored_chunks( @@ -158,7 +156,7 @@ def _open(self) -> zarr.Array[Any]: warnings.simplefilter("always", ZarrUserWarning) arr = zarr.open_array(self.store, path=self.path, mode="r+") warned = any(issubclass(w.category, ZarrUserWarning) for w in record) - assert warned is self.expect_open_warning, [str(w.message) for w in record] + assert warned is bool(self.legacy_axes), [str(w.message) for w in record] return arr # ----------------------------------------------------------------- model @@ -209,13 +207,19 @@ def _model_write(self, arr: zarr.Array[Any], region: tuple[slice, ...], values: @rule(data=st.data()) def append(self, data: st.DataObject) -> None: arr = self._open() - axis = data.draw(st.integers(0, len(self.shape) - 1), label="axis") + axes = st.integers(0, len(self.shape) - 1) + if self.legacy_axes: + # What a user of an older release did next: grow an axis stored with chunk size 0. + axes = st.sampled_from(self.legacy_axes) | axes + axis = data.draw(axes, label="axis") block_shape = list(self.shape) block_shape[axis] = data.draw(st.integers(0, 4), label="rows") block = data.draw(npst.arrays(DTYPE, tuple(block_shape)), label="block") note(f"append {block.shape} along {axis} to {self.shape}") if self.shape[axis] == 0 and block.shape[axis]: event("append to a zero-length axis") + if axis in self.legacy_axes and block.shape[axis]: + event("grow an axis stored with chunk size 0") old_extent = self.shape[axis] arr.append(block, axis=axis) self._model_resize(ChunkGrid.from_metadata(arr.metadata), arr.shape) @@ -224,7 +228,7 @@ def append(self, data: st.DataObject) -> None: ) self._model_write(arr, region, block) # Growing the array rewrites its metadata, which stores any correction. - self.expect_open_warning = False + self.legacy_axes = [] @rule(data=st.data()) def resize(self, data: st.DataObject) -> None: @@ -233,10 +237,12 @@ def resize(self, data: st.DataObject) -> None: st.tuples(*(st.integers(0, MAX_SIDE) for _ in self.shape)), label="new shape" ) note(f"resize {self.shape} -> {new_shape}") + if any(new_shape[axis] > self.shape[axis] for axis in self.legacy_axes): + event("grow an axis stored with chunk size 0") grid = ChunkGrid.from_metadata(arr.metadata) arr.resize(new_shape) self._model_resize(grid, new_shape) - self.expect_open_warning = False + self.legacy_axes = [] @rule(data=st.data()) def write(self, data: st.DataObject) -> None: @@ -251,12 +257,15 @@ def write(self, data: st.DataObject) -> None: arr[region] = values self._model_write(arr, region, values) - @precondition(lambda self: self.expect_open_warning) @rule() def resave_metadata(self) -> None: - """What the warning for an invalid stored chunk size tells users to do.""" - self._open().update_attributes({}) - self.expect_open_warning = False + """What the warning for an invalid stored chunk size tells users to do. It stores + the metadata as read, which is a no-op for valid metadata.""" + arr = self._open() + read = arr.metadata + arr.update_attributes({}) + self.legacy_axes = [] + assert self._open().metadata == read def teardown(self) -> None: self._rectilinear.__exit__(None, None, None) From 8e0cfc2baaf47a38761851be154aa02ad9a8ea8f Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Sat, 26 Sep 2026 02:21:43 +0200 Subject: [PATCH 33/76] fix(metadata): read a regular grid mixing sizes and edge lists per axis A stored regular chunk shape is read by the one per-axis rule of upgrades.py; a flat list is one more case of it, kept as that axis's chunk edges. A shape mixing chunk sizes and edge lists is read as the rectilinear grid it describes, so `[true, [5, 10, 5]]`, which zarr 3.2.x wrote and read, reads as `[1, [5, 10, 5]]`. A shape of only lists was never stored in a regular grid and is rejected. The rectilinear grid's constructor checks each edge with the one edge rule, so a float or bool in an edge list is reported as itself. The warning says that re-saving stores the rectilinear grid and so needs the flag, before the re-save steps it lists. Assisted-by: ClaudeCode:claude-opus-5-5 Co-Authored-By: Claude Opus 5.5 --- src/zarr/core/metadata/upgrades.py | 71 +++++++++++---- tests/test_metadata/test_upgrades.py | 131 ++++++++++++++++++--------- tests/test_metadata/test_v3.py | 31 +------ tests/test_unified_chunk_grid.py | 25 ----- 4 files changed, 145 insertions(+), 113 deletions(-) diff --git a/src/zarr/core/metadata/upgrades.py b/src/zarr/core/metadata/upgrades.py index 0cdac3151d..8ef20833e0 100644 --- a/src/zarr/core/metadata/upgrades.py +++ b/src/zarr/core/metadata/upgrades.py @@ -54,14 +54,25 @@ def _is_int_list(value: object) -> TypeGuard[list[int] | tuple[int, ...]]: return isinstance(value, list | tuple) and all(isinstance(v, int) for v in value) -def _read_chunk_size(size: JSON, span: int | None, unit: int) -> tuple[int, str | None] | None: - """Read one entry of a stored regular chunk shape as a chunk edge length. +def _abbreviate(value: JSON, limit: int = 60) -> str: + """`value` as JSON, cut to at most `limit` characters.""" + text = json.dumps(value) + return text if len(text) <= limit else f"{text[: limit - 3]}..." - Returns the edge length and, for an invalid entry, how it was read; `None` if the + +def _read_chunk_size( + size: JSON, span: int | None, unit: int +) -> tuple[int | list[JSON], str | None] | None: + """Read one entry of a stored regular chunk shape as a chunk edge length, or as the + chunk edge lengths of its axis. + + Returns the reading and, for an invalid entry, how it was read; `None` if the entry cannot be read, which leaves it for the metadata constructors to reject. A JSON int >= 1 is kept, JSON `true` is read as 1, and 0 or JSON `false` is read as one chunk spanning the axis of length `span`, a multiple of `unit` (the inner chunk - size of a shard), when the span is known. + size of a shard), when the span is known. A flat list is kept as the chunk edge + lengths of its axis, which only a rectilinear chunk grid can declare (see + `_invalid_chunk_sizes_v3`); the rectilinear chunk grid checks each edge. """ match size: case True: @@ -77,12 +88,14 @@ def _read_chunk_size(size: JSON, span: int | None, unit: int) -> tuple[int, str "holds only its fill value" ) return edge, how + case list() if not any(isinstance(edge, list) for edge in size): + return size, None return None def _read_chunk_shape( stored: JSON, spans: Sequence[int | None], units: Iterable[int], name: str -) -> tuple[list[int], str | None] | None: +) -> tuple[list[int | list[JSON]], str | None] | None: """Read a stored regular chunk shape, entry by entry (see `_read_chunk_size`), for axes of lengths `spans` whose chunks are multiples of `units` (1 where not given). @@ -91,7 +104,7 @@ def _read_chunk_shape( """ if not (isinstance(stored, list | tuple) and len(stored) == len(spans)): return None - edges: list[int] = [] + edges: list[int | list[JSON]] = [] readings: list[str] = [] axes = zip(stored, spans, chain(units, repeat(1)), strict=False) for axis, (size, span, unit) in enumerate(axes): @@ -105,8 +118,9 @@ def _read_chunk_shape( if not readings: return edges, None return edges, ( - f"The stored {name} {json.dumps(list(stored))} is invalid: chunk sizes must be " - f"integers of at least 1. It is read as {edges}, reading {'; '.join(readings)}." + f"The stored {name} {_abbreviate(stored)} is invalid: chunk sizes must be " + f"integers of at least 1. It is read as {_abbreviate(edges)}, reading " + f"{'; '.join(readings)}." ) @@ -133,7 +147,9 @@ def _sharding_codec(doc: ArrayDocument) -> tuple[Sequence[JSON], int, Mapping[st return None -def _read_inner_chunk_shape(doc: ArrayDocument) -> tuple[list[int], str | None] | None: +def _read_inner_chunk_shape( + doc: ArrayDocument, +) -> tuple[Sequence[int], str | None] | None: """Read the inner chunk shape of a sharded array. No stored inner chunk size of 0 or `false` is known, so the spans of its axes are not given.""" shape = doc.get("shape") @@ -141,12 +157,15 @@ def _read_inner_chunk_shape(doc: ArrayDocument) -> tuple[list[int], str | None] if sharding is None or not _is_int_list(shape): return None _, _, configuration = sharding - return _read_chunk_shape( + match _read_chunk_shape( configuration.get("chunk_shape"), [None] * len(shape), (), "inner chunk shape of the sharding codec", - ) + ): + case inner, reading if _is_int_list(inner): + return inner, reading + return None def _invalid_inner_chunk_sizes_v3(doc: ArrayDocument) -> tuple[ArrayDocument, str] | None: @@ -169,11 +188,31 @@ def _invalid_chunk_sizes_v3(doc: ArrayDocument) -> tuple[ArrayDocument, str] | N return None inner = _read_inner_chunk_shape(doc) units = () if inner is None else inner[0] - match _read_chunk_shape(configuration.get("chunk_shape"), shape, units, "chunk shape"): - case chunk_shape, str(reading): - upgraded = {**configuration, "chunk_shape": chunk_shape} - return {**doc, "chunk_grid": {**grid, "configuration": upgraded}}, reading - return None + read = _read_chunk_shape(configuration.get("chunk_shape"), shape, units, "chunk shape") + if read is None: + return None + chunk_shape, reading = read + edge_axes = [axis for axis, size in enumerate(chunk_shape) if isinstance(size, list)] + if not edge_axes: + if reading is None: + return None + upgraded = {**configuration, "chunk_shape": chunk_shape} + return {**doc, "chunk_grid": {**grid, "configuration": upgraded}}, reading + if len(edge_axes) == len(chunk_shape): + # Only a mix of chunk sizes and edge lists was ever stored in a regular grid. + return None + as_rectilinear = ( + f"The stored chunk grid is named 'regular', but its chunk shape lists chunk edge " + f"lengths on axes {edge_axes}, which only a rectilinear chunk grid can declare. " + "It is read as that rectilinear chunk grid. Re-saving the metadata stores that " + "rectilinear chunk grid, so each step that follows requires " + "`zarr.config.set({'array.rectilinear_chunks': True})`." + ) + rectilinear: JSON = { + "name": "rectilinear", + "configuration": {"kind": "inline", "chunk_shapes": chunk_shape}, + } + return {**doc, "chunk_grid": rectilinear}, " ".join(filter(None, (reading, as_rectilinear))) V2_ARRAY_UPGRADES: Final[tuple[Upgrade, ...]] = (_invalid_chunk_sizes_v2,) diff --git a/tests/test_metadata/test_upgrades.py b/tests/test_metadata/test_upgrades.py index 0777f0bdae..7cbdca9cb7 100644 --- a/tests/test_metadata/test_upgrades.py +++ b/tests/test_metadata/test_upgrades.py @@ -426,80 +426,121 @@ def _mixed_doc(shape: list[int], chunk_shape: list[Any]) -> dict[str, JSON]: @pytest.mark.parametrize( - ("doc", "expected", "axes"), + ("doc", "expected", "warning"), [ - (json.loads(MIXED_REGULAR_GRID_DOC), (2, (5, 10, 5)), [1]), - (_mixed_doc([6, 20, 4], [2, [5, 10, 5], [1, 3]]), (2, (5, 10, 5), (1, 3)), [1, 2]), - (_mixed_doc([6, 20], [[1, 5], [5, 10, 5]]), ((1, 5), (5, 10, 5)), [0, 1]), - (_mixed_doc([6, 12], [2, [5, 10, 5]]), (2, (5, 10, 5)), [1]), - (_mixed_doc([0, 20], [2, [5, 10, 5]]), (2, (5, 10, 5)), [1]), - (_mixed_doc([6, 0], [2, [5, 10, 5]]), (2, (5, 10, 5)), [1]), - (_mixed_doc([4, 10_000], [2, [10] * 1000]), (2, (10,) * 1000), [1]), + (json.loads(MIXED_REGULAR_GRID_DOC), (2, (5, 10, 5)), r"^The stored chunk grid .* \[1\]"), + ( + _mixed_doc([6, 20, 4], [2, [5, 10, 5], [1, 3]]), + (2, (5, 10, 5), (1, 3)), + r"^The stored chunk grid .* on axes \[1, 2\]", + ), + (_mixed_doc([6, 20], [2, [20]]), (2, (20,)), r"^The stored chunk grid .* \[1\]"), + (_mixed_doc([6, 12], [2, [5, 10, 5]]), (2, (5, 10, 5)), r"^The stored chunk grid"), + (_mixed_doc([0, 20], [2, [5, 10, 5]]), (2, (5, 10, 5)), r"^The stored chunk grid"), + (_mixed_doc([6, 0], [2, [5, 10, 5]]), (2, (5, 10, 5)), r"^The stored chunk grid"), + ( + _mixed_doc([6, 20], [True, [5, 10, 5]]), + (1, (5, 10, 5)), + ( + r"^The stored chunk shape \[true, \[5, 10, 5\]\] is invalid: .* read as " + r"\[1, \[5, 10, 5\]\], reading true on axis 0 as 1\. The stored chunk grid" + ), + ), + ( + _mixed_doc([4, 10_000], [2, [10] * 1000]), + (2, (10,) * 1000), + r"^The stored chunk grid .* \[1\]", + ), + ], + ids=[ + "written", + "3d", + "one-edge", + "shrunk", + "empty-int-axis", + "empty-edge-axis", + "true", + "long", ], - ids=["written", "3d", "every-axis", "shrunk", "empty-regular-axis", "empty-edge-axis", "long"], ) def test_read_edge_lists_in_regular_grid( - doc: dict[str, JSON], expected: tuple[int | tuple[int, ...], ...], axes: list[int] + doc: dict[str, JSON], expected: tuple[int | tuple[int, ...], ...], warning: str ) -> None: - """A `regular` chunk grid whose `chunk_shape` lists chunk edges is read as the - rectilinear grid it describes, without the rectilinear chunks flag, with one - warning that names the axes, quotes at most a bounded part of the chunk shape and - says how to re-save.""" + """A `regular` chunk grid whose chunk shape mixes chunk sizes with lists of chunk + edge lengths is read as the rectilinear chunk grid it describes, without the + rectilinear chunks flag. `from_dict` warns once, naming the array and the axes, + quoting a bounded part of the chunk shape, and saying that re-saving requires the + flag.""" with ( zarr.config.set({"array.rectilinear_chunks": False}), warnings.catch_warnings(record=True) as record, ): warnings.simplefilter("always") - metadata = ArrayV3Metadata.from_dict(doc) + metadata = ArrayV3Metadata.from_dict(doc, path="group/array") assert metadata.chunk_grid == RectilinearChunkGridMetadata(chunk_shapes=expected) [message] = [str(w.message) for w in record] - assert f"lists chunk edge lengths on axes {axes}" in message - assert "array.rectilinear_chunks" in message - assert "update_attributes({})" in message + assert re.search(warning, message.removeprefix("Array 'group/array': ")) + assert message.startswith("Array 'group/array': ") + assert ( + "Re-saving the metadata stores that rectilinear chunk grid, so each step that " + "follows requires `zarr.config.set({'array.rectilinear_chunks': True})`. " + RESAVE_HINT + ) in message assert len(message) < 1000 -def test_edge_lists_in_regular_grid_rle_rejected() -> None: - """Run-length encoded edges were never written inside a `regular` chunk_shape, so - such a document is not read as rectilinear.""" +def _rejected_without_warning(doc: dict[str, JSON]) -> pytest.ExceptionInfo[Exception]: with warnings.catch_warnings(): warnings.simplefilter("error", ZarrUserWarning) - with pytest.raises(TypeError, match="Dimension 1: a regular chunk grid requires"): - ArrayV3Metadata.from_dict(_mixed_doc([6, 20], [2, [[5, 2], 10]])) + with pytest.raises((TypeError, ValueError)) as info: + ArrayV3Metadata.from_dict(doc) + return info -@pytest.mark.parametrize("edges", [[10.0, 10.0], [True, True]], ids=["float", "bool"]) -def test_edge_lists_in_regular_grid_non_integer_edges_rejected(edges: list[Any]) -> None: - with warnings.catch_warnings(): - warnings.simplefilter("error", ZarrUserWarning) - with pytest.raises(TypeError, match="Dimension 1: a regular chunk grid requires"): - ArrayV3Metadata.from_dict(_mixed_doc([6, 20], [2, edges])) +def test_regular_grid_of_only_edge_lists_rejected() -> None: + """A regular chunk shape made only of edge lists was never stored (a rectilinear + chunk grid was), so it is not read as rectilinear.""" + info = _rejected_without_warning(_mixed_doc([6, 20], [[1, 5], [5, 10, 5]])) + assert info.match(re.escape("Dimension 0: Chunk edge length must be an int, got [1, 5]")) -def test_edge_lists_in_regular_grid_nonpositive_edge_rejected() -> None: - """The upgraded grid is validated before any warning, so the real error surfaces.""" - with warnings.catch_warnings(): - warnings.simplefilter("error", ZarrUserWarning) - with pytest.raises(ValueError, match="must be >= 1, got 0"): - ArrayV3Metadata.from_dict(_mixed_doc([6, 20], [2, [0, 20]])) +def test_run_length_encoded_edges_in_regular_grid_rejected() -> None: + """Run-length encoded edges were never stored in a regular chunk shape.""" + info = _rejected_without_warning(_mixed_doc([6, 20], [2, [[5, 2], 10]])) + assert info.match(re.escape("Dimension 1: Chunk edge length must be an int, got [[5, 2], 10]")) -def test_edge_lists_in_regular_grid_short_edges_rejected() -> None: - with warnings.catch_warnings(): - warnings.simplefilter("error", ZarrUserWarning) - with pytest.raises(ValueError, match="sum to 15 but array shape extent is 20"): - ArrayV3Metadata.from_dict(_mixed_doc([6, 20], [2, [5, 10]])) +@pytest.mark.parametrize("edge", [5.0, True], ids=["float", "bool"]) +def test_non_int_edge_in_regular_grid_rejected(edge: object) -> None: + """An edge that is not an int is reported as such, not blamed on its list.""" + info = _rejected_without_warning(_mixed_doc([6, 20], [2, [edge, 15]])) + assert info.match(re.escape(f"Chunk edge length must be an int, got {edge!r}")) -def test_edge_lists_in_regular_grid_round_trip(tmp_path: Path) -> None: - """A store holding the verbatim document opens without the rectilinear chunks flag, - reads its data, re-saves as rectilinear with the flag, and then opens cleanly.""" - path = tmp_path / "mixed.zarr" +def test_edge_below_one_in_regular_grid_rejected() -> None: + info = _rejected_without_warning(_mixed_doc([6, 20], [2, [0, 20]])) + assert info.match("Chunk edge length must be >= 1, got 0") + + +def test_short_edges_in_regular_grid_rejected() -> None: + info = _rejected_without_warning(_mixed_doc([6, 20], [2, [5, 10]])) + assert info.match("sum to 15 but array shape extent is 20") + + +def _store_mixed_array(path: Path) -> np.ndarray[Any, np.dtype[np.float32]]: + """Write the chunks of the verbatim document and the document itself at `path`.""" data = np.arange(120, dtype="float32").reshape(6, 20) with zarr.config.set({"array.rectilinear_chunks": True}): arr = zarr.create_array(path, shape=data.shape, chunks=(2, (5, 10, 5)), dtype="float32") - arr[...] = data + arr[...] = data (path / "zarr.json").write_text(MIXED_REGULAR_GRID_DOC) + return data + + +def test_edge_lists_in_regular_grid_round_trip(tmp_path: Path) -> None: + """A store holding the verbatim document opens without the rectilinear chunks flag, + reads its data, re-saves as rectilinear with the flag, and then opens cleanly.""" + path = tmp_path / "mixed.zarr" + data = _store_mixed_array(path) with ( zarr.config.set({"array.rectilinear_chunks": False}), diff --git a/tests/test_metadata/test_v3.py b/tests/test_metadata/test_v3.py index 96e44ce230..fb9632f151 100644 --- a/tests/test_metadata/test_v3.py +++ b/tests/test_metadata/test_v3.py @@ -3,6 +3,7 @@ from __future__ import annotations import json +import re from typing import TYPE_CHECKING import pytest @@ -33,7 +34,6 @@ ) if TYPE_CHECKING: - from collections.abc import Callable from typing import Any @@ -156,34 +156,11 @@ def test_create_chunk_grid_metadata_unknown_dimension_type() -> None: create_chunk_grid_metadata(grid) -@pytest.mark.parametrize( - "build", - [ - pytest.param(lambda shape: RegularChunkGridMetadata(chunk_shape=shape), id="constructor"), - pytest.param( - lambda shape: RegularChunkGridMetadata.from_dict( - {"name": "regular", "configuration": {"chunk_shape": list(shape)}} - ), - id="from_dict", - ), - ], -) -@pytest.mark.parametrize("chunk_shape", [(0, 2), (2, -1)], ids=["zero", "negative"]) -def test_regular_chunk_grid_rejects_nonpositive_chunk_size( - build: Callable[[tuple[int, ...]], RegularChunkGridMetadata], chunk_shape: tuple[int, ...] -) -> None: - """A regular chunk size below 1 is rejected, naming the dimension, whether - the grid is built directly or parsed from stored metadata.""" - dim = 0 if chunk_shape[0] < 1 else 1 - with pytest.raises( - ValueError, match=f"Dimension {dim}: chunk edge length must be an integer >= 1" - ): - build(chunk_shape) - - def test_regular_chunk_grid_rejects_edge_lists() -> None: """A regular chunk grid only accepts integer chunk edge lengths.""" - with pytest.raises(TypeError, match="Dimension 1: a regular chunk grid requires an integer"): + with pytest.raises( + TypeError, match=re.escape("Dimension 1: Chunk edge length must be an int, got (5, 10, 5)") + ): RegularChunkGridMetadata(chunk_shape=(2, (5, 10, 5))) # type: ignore[arg-type] diff --git a/tests/test_unified_chunk_grid.py b/tests/test_unified_chunk_grid.py index dea7a6352d..513e78a8ab 100644 --- a/tests/test_unified_chunk_grid.py +++ b/tests/test_unified_chunk_grid.py @@ -641,22 +641,6 @@ def test_serialization_error_non_regular_chunk_shape() -> None: grid.chunk_shape # noqa: B018 -@pytest.mark.parametrize("chunk_shapes", [[0, [5, 5]], [[5, 5], -1]], ids=["zero", "negative"]) -def test_rectilinear_from_dict_rejects_nonpositive_bare_int(chunk_shapes: list[Any]) -> None: - """A bare-int dimension below 1 is rejected by the grid's own validator, which - names the dimension, rather than by a duplicate check in `from_dict`.""" - dim = 0 if isinstance(chunk_shapes[0], int) else 1 - with pytest.raises( - ValueError, match=f"Dimension {dim}: chunk edge length must be an integer >= 1" - ): - RectilinearChunkGridMetadata.from_dict( - { - "name": "rectilinear", - "configuration": {"kind": "inline", "chunk_shapes": chunk_shapes}, - } - ) - - def test_serialization_error_zero_extent_rectilinear() -> None: """RectilinearChunkGridMetadata rejects empty edge tuples.""" with pytest.raises(ValueError, match="has no chunk edges"): @@ -3028,15 +3012,6 @@ def test_rectilinear_from_dict( assert grid.chunk_shapes == expected_chunk_shapes -@pytest.mark.parametrize("dim_spec", [4.5, None, "10"], ids=["float", "none", "string"]) -def test_rectilinear_from_dict_rejects_invalid_dim_spec(dim_spec: Any) -> None: - """A dimension that is neither an integer nor a list of edges is rejected.""" - with pytest.raises(TypeError, match="expected int or list"): - RectilinearChunkGridMetadata.from_dict( - {"name": "rectilinear", "configuration": {"kind": "inline", "chunk_shapes": [dim_spec]}} - ) - - @pytest.mark.parametrize( ("chunk_shapes", "expected_json_shapes"), [ From 9c9ee76c2a543c6315e59779afab7b908d8bb63d Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Sat, 26 Sep 2026 02:24:25 +0200 Subject: [PATCH 34/76] fix(array): check that metadata can be stored before touching the store `check_storable` is the one store-write gate for the rectilinear chunks flag. `ArrayV3Metadata.to_buffer_dict` calls it, and so does `GroupMetadata.to_buffer_dict` for each array in the group's consolidated metadata, so `consolidate_metadata` can no longer store a rectilinear grid with the flag off and leave the group unreadable without it. Every metadata write serializes before it writes. `resize` used to delete the chunks outside the new shape before storing the new metadata, so metadata that could not be stored lost data; it now stores the metadata first, for all metadata, and deletes afterwards. Assisted-by: ClaudeCode:claude-opus-5-5 Co-Authored-By: Claude Opus 5.5 --- changes/4375.bugfix.md | 4 +-- src/zarr/core/array.py | 32 ++++++++++------------- src/zarr/core/group.py | 5 ++++ src/zarr/core/metadata/v3.py | 15 +++++++++-- tests/test_metadata/test_upgrades.py | 38 ++++++++++++++++++++++++++++ 5 files changed, 72 insertions(+), 22 deletions(-) diff --git a/changes/4375.bugfix.md b/changes/4375.bugfix.md index 73c4caa4c1..d30dd159d0 100644 --- a/changes/4375.bugfix.md +++ b/changes/4375.bugfix.md @@ -1,3 +1,3 @@ -Arrays whose stored `regular` chunk grid lists chunk edge lengths for some dimensions, such as `"chunk_shape": [2, [5, 10, 5]]` (written by zarr 3.2.0 and 3.2.1 for `chunks=(2, (5, 10, 5))`), can be read again, without enabling `array.rectilinear_chunks`. The grid is read as the rectilinear chunk grid it describes, with a `ZarrUserWarning`; re-saving the metadata stores that rectilinear grid, which requires the flag. A regular chunk grid given edge lists is now rejected, so this metadata is no longer written. +Arrays whose stored `regular` chunk grid mixes chunk sizes with lists of chunk edge lengths, such as `"chunk_shape": [2, [5, 10, 5]]` (written by zarr 3.2.0 and 3.2.1 for `chunks=(2, (5, 10, 5))`), can be read again, without enabling `array.rectilinear_chunks`. The grid is read as the rectilinear chunk grid it describes, with a `ZarrUserWarning`; re-saving the metadata stores that rectilinear grid, which requires the flag. A regular chunk grid given edge lists is now rejected, so this metadata is no longer written. -The `array.rectilinear_chunks` flag now gates reading and storing array metadata that declares a rectilinear chunk grid, instead of constructing `RectilinearChunkGridMetadata`. +The `array.rectilinear_chunks` flag now gates reading and storing array metadata that declares a rectilinear chunk grid, including in a group's consolidated metadata, instead of constructing `RectilinearChunkGridMetadata`. `Array.resize` now stores the new metadata before deleting chunks outside the new shape, so a resize whose metadata cannot be stored leaves the array unchanged. diff --git a/src/zarr/core/array.py b/src/zarr/core/array.py index e39080b668..1b4bf82cea 100644 --- a/src/zarr/core/array.py +++ b/src/zarr/core/array.py @@ -5909,31 +5909,27 @@ async def _resize( # ensure deletion is only run if array is shrinking as the delete_outside_chunks path is unbounded in memory only_growing = all(new >= old for new, old in zip(new_shape, array.metadata.shape, strict=True)) - + outside_keys: list[tuple[str]] = [] if delete_outside_chunks and not only_growing: - # Remove all chunks outside of the new shape old_chunk_coords = set(array._chunk_grid.all_chunk_coords()) new_chunk_coords = set(new_chunk_grid.all_chunk_coords()) - - async def _delete_key(key: str) -> None: - await (array.store_path / key).delete() - - await concurrent_map( - [ - (array.metadata.encode_chunk_key(chunk_coords),) - for chunk_coords in old_chunk_coords.difference(new_chunk_coords) - ], - _delete_key, - zarr_config.get("async.concurrency"), - ) - - # Write new metadata + outside_keys = [ + (array.metadata.encode_chunk_key(chunk_coords),) + for chunk_coords in old_chunk_coords.difference(new_chunk_coords) + ] + + # Store the new metadata before deleting any chunk: metadata that cannot be stored + # then fails with the store untouched, and a failed deletion leaves only chunks + # outside the new shape, as `delete_outside_chunks=False` does. await save_metadata(array.store_path, new_metadata) - - # Update metadata and chunk_grid (in place) object.__setattr__(array, "metadata", new_metadata) object.__setattr__(array, "_chunk_grid", new_chunk_grid) + async def _delete_key(key: str) -> None: + await (array.store_path / key).delete() + + await concurrent_map(outside_keys, _delete_key, zarr_config.get("async.concurrency")) + async def _append( array: AsyncArray[ArrayV2Metadata] | AsyncArray[ArrayV3Metadata], diff --git a/src/zarr/core/group.py b/src/zarr/core/group.py index 78071300f0..f7dc79e97d 100644 --- a/src/zarr/core/group.py +++ b/src/zarr/core/group.py @@ -49,6 +49,7 @@ from zarr.core.json_parse import parse_field from zarr.core.metadata import ArrayV2Metadata, ArrayV3Metadata from zarr.core.metadata.io import save_metadata +from zarr.core.metadata.v3 import check_storable from zarr.core.sync import SyncMixin, sync from zarr.errors import ( ArrayNotFoundError, @@ -357,6 +358,10 @@ class GroupMetadata(Metadata): node_type: Literal["group"] = field(default="group", init=False) def to_buffer_dict(self, prototype: BufferPrototype) -> dict[str, Buffer]: + if self.consolidated_metadata is not None: + for member in self.consolidated_metadata.flattened_metadata.values(): + if isinstance(member, ArrayV3Metadata): + check_storable(member) indent = config.get("json_indent") if self.zarr_format == 3: return {ZARR_JSON: json_to_buffer(self.to_dict(), prototype=prototype, indent=indent)} diff --git a/src/zarr/core/metadata/v3.py b/src/zarr/core/metadata/v3.py index fc13010cde..28c2ce8684 100644 --- a/src/zarr/core/metadata/v3.py +++ b/src/zarr/core/metadata/v3.py @@ -370,6 +370,18 @@ def _check_rectilinear_chunks_enabled() -> None: ) +def check_storable(metadata: ArrayV3Metadata) -> None: + """Raise if `metadata` may not be stored: a rectilinear chunk grid requires the + rectilinear chunks flag. + + Every serialization of array metadata for a store calls this, before the store is + touched: `ArrayV3Metadata.to_buffer_dict` and, for the arrays in a group's + consolidated metadata, `GroupMetadata.to_buffer_dict`. + """ + if isinstance(metadata.chunk_grid, RectilinearChunkGridMetadata): + _check_rectilinear_chunks_enabled() + + def create_chunk_grid_metadata( chunks: ChunkGrid, ) -> ChunkGridMetadata: @@ -612,8 +624,7 @@ def encode_chunk_key(self, chunk_coords: tuple[int, ...]) -> str: return self.chunk_key_encoding.encode_chunk_key(chunk_coords) def to_buffer_dict(self, prototype: BufferPrototype) -> dict[str, Buffer]: - if isinstance(self.chunk_grid, RectilinearChunkGridMetadata): - _check_rectilinear_chunks_enabled() + check_storable(self) indent = config.get("json_indent") return {ZARR_JSON: json_to_buffer(self.to_dict(), prototype=prototype, indent=indent)} diff --git a/tests/test_metadata/test_upgrades.py b/tests/test_metadata/test_upgrades.py index 7cbdca9cb7..6850eb378f 100644 --- a/tests/test_metadata/test_upgrades.py +++ b/tests/test_metadata/test_upgrades.py @@ -559,3 +559,41 @@ def test_edge_lists_in_regular_grid_round_trip(tmp_path: Path) -> None: warnings.simplefilter("error", ZarrUserWarning) reopened = zarr.open_array(path) np.testing.assert_array_equal(reopened[...], data) + + +def test_resize_without_flag_leaves_store_intact(tmp_path: Path) -> None: + """Resizing an array read from the verbatim document without the flag cannot store + its metadata, and fails before deleting any chunk.""" + path = tmp_path / "mixed.zarr" + data = _store_mixed_array(path) + stored = {p: p.read_bytes() for p in path.rglob("*") if p.is_file()} + with zarr.config.set({"array.rectilinear_chunks": False}): + with pytest.warns(ZarrUserWarning, match="read as that rectilinear chunk grid"): + arr = zarr.open_array(path, mode="a") + with pytest.raises(ValueError, match="experimental and disabled by default"): + arr.resize((6, 5)) + assert {p: p.read_bytes() for p in path.rglob("*") if p.is_file()} == stored + assert arr.shape == data.shape + np.testing.assert_array_equal(arr[...], data) + + +@pytest.mark.filterwarnings("ignore:Consolidated metadata is currently not part:UserWarning") +def test_consolidate_without_flag_leaves_group_readable(tmp_path: Path) -> None: + """Consolidating a group holding an array read from the verbatim document would + store its rectilinear chunk grid, so without the flag it fails with the store + untouched, and the group still opens without the flag.""" + path = tmp_path / "group.zarr" + zarr.open_group(path, mode="w") + data = _store_mixed_array(path / "mixed") + group_doc = (path / "zarr.json").read_bytes() + with zarr.config.set({"array.rectilinear_chunks": False}): + with ( + pytest.warns(ZarrUserWarning, match="read as that rectilinear chunk grid"), + pytest.raises(ValueError, match="experimental and disabled by default"), + ): + zarr.consolidate_metadata(path) + assert (path / "zarr.json").read_bytes() == group_doc + with pytest.warns(ZarrUserWarning, match="read as that rectilinear chunk grid"): + mixed = zarr.open_group(path, mode="r")["mixed"] + assert isinstance(mixed, zarr.Array) + np.testing.assert_array_equal(mixed[...], data) From c0d4b3cc76c93cbd7364367a21140e4fc3017365 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Sat, 26 Sep 2026 05:58:34 +0200 Subject: [PATCH 35/76] fix(array): store upgraded metadata before writing the first chunk An array read from a stored document that the upgrades had to correct (a chunk size of 0, `false` or `true`) wrote chunks under the corrected layout while the store kept the old document, so zarr 3.2.1-3.4.0 then read only the fill value, and tensorstore and zarrs could not open it. `from_dict` now marks the metadata it read from an upgraded document (`_stored_document_upgraded`, a field outside the document and equality), and `AsyncArray._set_selection`, which every chunk write goes through (`setitem` now included), stores that metadata first and keeps the stored copy. Concurrent first writes store the same document. The array lifecycle state machine now ends the legacy state on a write and checks that no chunk is stored under a document that is still upgraded on read; `resave_metadata` runs only while the stored document is invalid, and re-saving valid metadata is checked once, at creation. The changelog also says which releases stored a chunk size of 0 on an axis of positive length. Assisted-by: ClaudeCode:claude-opus-5-5 Co-Authored-By: Claude Opus 5.5 --- changes/4334.bugfix.md | 2 +- src/zarr/abc/metadata.py | 5 ++- src/zarr/core/array.py | 20 +++++----- src/zarr/core/metadata/v2.py | 6 ++- src/zarr/core/metadata/v3.py | 4 ++ tests/test_array_stateful.py | 34 +++++++++++++--- tests/test_metadata/test_upgrades.py | 60 ++++++++++++++++++++++++++++ 7 files changed, 113 insertions(+), 18 deletions(-) diff --git a/changes/4334.bugfix.md b/changes/4334.bugfix.md index 77c211643e..4c55701757 100644 --- a/changes/4334.bugfix.md +++ b/changes/4334.bugfix.md @@ -2,4 +2,4 @@ A chunk edge length is now always at least 1, while an array extent may be 0. Ev Metadata built in code is strict about chunk edge lengths: `ArrayV2Metadata`, `RegularChunkGridMetadata`, `RectilinearChunkGridMetadata` and `ShardingCodec` take them as Python `int`s of at least 1, so a size of 0 now raises a `ValueError`, and a `bool`, a float or a NumPy integer (including a scalar `chunks=np.int64(5)` for `ArrayV2Metadata`) raises a `TypeError`; `FixedDimension(size=0, ...)` raises a `ValueError`. Stored rectilinear chunk grids with integral floats (`10.0`) are rejected too. The array creation functions still accept NumPy integers in `chunks=` and `shards=`. -Stored metadata with a regular chunk size of 0 or JSON `false`, as zarr-python wrote for arrays created with a zero-length axis until 3.4, now opens with a `ZarrUserWarning` and is read as one chunk spanning the axis (a multiple of the inner chunk size for sharded arrays, which previously failed to open). JSON `true`, as zarr-python 3.0 and 3.2 wrote for `chunks=(True,)`, is read as 1, in the chunk grid and in the inner chunk shape of a sharding codec. The warning names the array and says how to store valid metadata: `array.update_attributes({})`, then `zarr.consolidate_metadata` if the metadata is consolidated. +Stored metadata with a regular chunk size of 0 or JSON `false` now opens with a `ZarrUserWarning` and is read as one chunk spanning the axis (a multiple of the inner chunk size for sharded arrays, which previously failed to open). zarr-python wrote such sizes for arrays created with a zero-length axis until 3.4, and, from 2.18.7 to 3.2.1, for an explicit chunk size of 0 or `False` on an axis of any length, as in `chunks=(0,)` (those arrays could store no data). JSON `true`, as zarr-python 3.0 and 3.2 wrote for `chunks=(True,)`, is read as 1, in the chunk grid and in the inner chunk shape of a sharding codec. The warning names the array and says how to store valid metadata: `array.update_attributes({})`, then `zarr.consolidate_metadata` if the metadata is consolidated. Writing data to such an array first stores the upgraded metadata, so that other readers find the chunks written. diff --git a/src/zarr/abc/metadata.py b/src/zarr/abc/metadata.py index a56f986645..1104111b63 100644 --- a/src/zarr/abc/metadata.py +++ b/src/zarr/abc/metadata.py @@ -20,10 +20,13 @@ def to_dict(self) -> dict[str, JSON]: Recursively serialize this model to a dictionary. This method inspects the fields of self and calls `x.to_dict()` for any fields that are instances of `Metadata`. Sequences of `Metadata` are similarly recursed into, and - the output of that recursion is collected in a list. + the output of that recursion is collected in a list. Fields declared with + `compare=False` are not part of the document. """ out_dict = {} for field in fields(self): + if not field.compare: + continue key = field.name value = getattr(self, key) if isinstance(value, Metadata): diff --git a/src/zarr/core/array.py b/src/zarr/core/array.py index e39080b668..5f4cddb9b4 100644 --- a/src/zarr/core/array.py +++ b/src/zarr/core/array.py @@ -1623,6 +1623,12 @@ async def _set_selection( prototype: BufferPrototype, fields: Fields | None = None, ) -> None: + if self.metadata._stored_document_upgraded: + # Chunks are about to be stored under the upgraded metadata, so store it + # first: every reader of the store then agrees with them. + metadata = replace(self.metadata) + await self._save_metadata(metadata) + object.__setattr__(self, "metadata", metadata) return await _set_selection( self.store_path, self.metadata, @@ -1674,16 +1680,10 @@ async def setitem( - This method is asynchronous and should be awaited. - Supports basic indexing, where the selection is contiguous and does not involve advanced indexing. """ - return await _setitem( - self.store_path, - self.metadata, - self.codec_pipeline, - self.config, - self._chunk_grid, - selection, - value, - prototype=prototype, - ) + if prototype is None: + prototype = default_buffer_prototype() + indexer = BasicIndexer(selection, shape=self.metadata.shape, chunk_grid=self._chunk_grid) + return await self._set_selection(indexer, value, prototype=prototype) @property def oindex(self) -> AsyncOIndex[T_ArrayMetadata]: diff --git a/src/zarr/core/metadata/v2.py b/src/zarr/core/metadata/v2.py index b99ee442b2..b5701b3e78 100644 --- a/src/zarr/core/metadata/v2.py +++ b/src/zarr/core/metadata/v2.py @@ -72,6 +72,9 @@ class ArrayV2Metadata(Metadata): compressor: Numcodec | None attributes: dict[str, JSON] = field(default_factory=dict) zarr_format: Literal[2] = field(init=False, default=2) + _stored_document_upgraded: bool = field(default=False, init=False, compare=False, repr=False) + """Whether `from_dict` read this metadata from a stored document it had to upgrade, + so the store holds an invalid document until this metadata is stored.""" def __init__( self, @@ -184,7 +187,7 @@ def from_dict(cls, data: dict[str, Any], *, path: str | None = None) -> ArrayV2M # zarr v2 allowed arbitrary keys here. # We don't want the ArrayV2Metadata constructor to fail just because someone put an # extra key in the metadata. - expected = {x.name for x in fields(cls)} + expected = {x.name for x in fields(cls) if x.init} expected |= {"dtype", "chunks"} # check if `filters` is an empty sequence; if so use None instead and raise a warning @@ -206,6 +209,7 @@ def from_dict(cls, data: dict[str, Any], *, path: str | None = None) -> ArrayV2M metadata = cls(**_data) warn_readings(readings, path) + object.__setattr__(metadata, "_stored_document_upgraded", bool(readings)) return metadata def to_dict(self) -> dict[str, JSON]: diff --git a/src/zarr/core/metadata/v3.py b/src/zarr/core/metadata/v3.py index f1cbfcbe63..ab5ddf3d2f 100644 --- a/src/zarr/core/metadata/v3.py +++ b/src/zarr/core/metadata/v3.py @@ -478,6 +478,9 @@ class ArrayV3Metadata(Metadata): node_type: Literal["array"] = field(default="array", init=False) storage_transformers: tuple[dict[str, JSON], ...] extra_fields: dict[str, AllowedExtraField] + _stored_document_upgraded: bool = field(default=False, init=False, compare=False, repr=False) + """Whether `from_dict` read this metadata from a stored document it had to upgrade, + so the store holds an invalid document until this metadata is stored.""" def __init__( self, @@ -678,6 +681,7 @@ def from_dict(cls, data: dict[str, JSON], *, path: str | None = None) -> Self: storage_transformers=_data_typed.get("storage_transformers", ()), # type: ignore[arg-type] ) warn_readings(readings, path) + object.__setattr__(metadata, "_stored_document_upgraded", bool(readings)) return metadata def to_dict(self) -> dict[str, JSON]: diff --git a/tests/test_array_stateful.py b/tests/test_array_stateful.py index 0ea4da0167..7ce4780e5d 100644 --- a/tests/test_array_stateful.py +++ b/tests/test_array_stateful.py @@ -30,6 +30,7 @@ RuleBasedStateMachine, initialize, invariant, + precondition, rule, ) @@ -46,6 +47,7 @@ ] DTYPE = np.dtype("int16") +METADATA_KEYS = (".zarray", ".zattrs", "zarr.json") MAX_SIDE = 6 @@ -67,6 +69,10 @@ def _rectilinear_dim(extent: int) -> st.SearchStrategy[int | list[int]]: return steps | edges +async def _list(store: MemoryStore, prefix: str) -> list[str]: + return [key async for key in store.list_prefix(prefix)] + + class ArrayLifecycle(RuleBasedStateMachine): def __init__(self) -> None: super().__init__() @@ -136,6 +142,11 @@ def create(self, data: st.DataObject) -> None: stored_zero = data.draw(st.sampled_from([0, False]), label="stored zero") self._rewrite_stored_chunks(zarr_format, self.legacy_axes, stored_zero) event("legacy zero chunk size") + else: + # Re-saving valid metadata, as the warning tells users to, changes nothing. + arr = self._open() + arr.update_attributes({}) + assert self._open().metadata == arr.metadata def _rewrite_stored_chunks( self, zarr_format: Literal[2, 3], axes: list[int], value: Any @@ -233,9 +244,11 @@ def append(self, data: st.DataObject) -> None: @rule(data=st.data()) def resize(self, data: st.DataObject) -> None: arr = self._open() - new_shape = data.draw( - st.tuples(*(st.integers(0, MAX_SIDE) for _ in self.shape)), label="new shape" - ) + extents = [st.integers(0, MAX_SIDE) for _ in self.shape] + for axis in self.legacy_axes: + # What a user of an older release did next: grow an axis stored with chunk size 0. + extents[axis] = st.integers(self.shape[axis] + 1, MAX_SIDE + 1) | extents[axis] + new_shape = data.draw(st.tuples(*extents), label="new shape") note(f"resize {self.shape} -> {new_shape}") if any(new_shape[axis] > self.shape[axis] for axis in self.legacy_axes): event("grow an axis stored with chunk size 0") @@ -256,11 +269,14 @@ def write(self, data: st.DataObject) -> None: note(f"write {region}") arr[region] = values self._model_write(arr, region, values) + # Writing chunks first stores the metadata they are written under. + self.legacy_axes = [] + @precondition(lambda self: self.legacy_axes) @rule() def resave_metadata(self) -> None: - """What the warning for an invalid stored chunk size tells users to do. It stores - the metadata as read, which is a no-op for valid metadata.""" + """What the warning for an invalid stored chunk size tells users to do: store + the metadata as read.""" arr = self._open() read = arr.metadata arr.update_attributes({}) @@ -271,6 +287,14 @@ def teardown(self) -> None: self._rectilinear.__exit__(None, None, None) # ------------------------------------------------------------ invariants + @invariant() + def no_chunk_under_an_invalid_document(self) -> None: + """While the stored document is still one that is upgraded on read, which other + readers may reject or read differently, no chunk is stored under it.""" + if self.legacy_axes: + keys = sync(_list(self.store, f"{self.path}/")) + assert set(keys) <= {f"{self.path}/{name}" for name in METADATA_KEYS}, keys + @invariant() def matches_model(self) -> None: arr = self._open() diff --git a/tests/test_metadata/test_upgrades.py b/tests/test_metadata/test_upgrades.py index fabf22b1b8..b8e5ea3cc7 100644 --- a/tests/test_metadata/test_upgrades.py +++ b/tests/test_metadata/test_upgrades.py @@ -2,6 +2,7 @@ from __future__ import annotations +import asyncio import json import re import warnings @@ -20,6 +21,7 @@ upgrade_array_document, ) from zarr.core.metadata.v3 import RectilinearChunkGridMetadata, RegularChunkGridMetadata +from zarr.core.sync import sync from zarr.dtype import Int16 from zarr.errors import ZarrUserWarning @@ -345,6 +347,64 @@ def store_legacy(doc: dict[str, Any]) -> None: np.testing.assert_array_equal(reopened[...], np.concatenate([data, block])) +@pytest.mark.parametrize( + ("zarr_format", "shape", "inner", "expected"), + [(2, (3,), None, (3,)), (3, (3,), None, (3,)), (3, (10,), (4,), (12,))], + ids=["v2", "v3", "v3-sharded"], +) +@pytest.mark.parametrize("api", ["sync", "async", "async-concurrent"]) +def test_write_stores_upgraded_metadata_first( + tmp_path: Path, + zarr_format: Literal[2, 3], + shape: tuple[int, ...], + inner: tuple[int, ...] | None, + expected: tuple[int, ...], + api: str, +) -> None: + """Writing chunks to an array read from an upgraded document first stores the + upgraded metadata, so readers that do not upgrade (or read it differently) see + the chunks the write stored.""" + path = tmp_path / "legacy.zarr" + zarr.create_array( + store=path, + shape=shape, + chunks=inner or expected, + shards=expected if inner else None, + dtype="int16", + fill_value=0, + zarr_format=zarr_format, + ) + _rewrite_doc( + path, + zarr_format, + lambda doc: ( + doc.update(chunks=[0]) + if zarr_format == 2 + else doc["chunk_grid"]["configuration"].update(chunk_shape=[0]) + ), + ) + data = np.arange(1, shape[0] + 1, dtype="int16") + with pytest.warns(ZarrUserWarning, match="is read as"): + arr = zarr.open_array(store=path, mode="r+") + if api == "sync": + arr[:] = data + elif api == "async": + sync(arr.async_array.setitem(slice(None), data)) + else: + + async def write_twice() -> None: + # Both writes find the metadata not yet stored, and both store it. + await asyncio.gather(*(arr.async_array.setitem(slice(None), data) for _ in "ab")) + + sync(write_twice()) + + with warnings.catch_warnings(): + warnings.simplefilter("error", ZarrUserWarning) + reopened = zarr.open_array(store=path, mode="r") + assert (reopened.shards or reopened.chunks) == expected + np.testing.assert_array_equal(reopened[...], data) + + @pytest.mark.filterwarnings("ignore:Consolidated metadata is currently not part:UserWarning") @pytest.mark.parametrize("zarr_format", [2, 3]) def test_legacy_chunk_size_consolidated(tmp_path: Path, zarr_format: Literal[2, 3]) -> None: From 121b6b2f84b559590cc25f21e3d2988c0b3bd247 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Sat, 26 Sep 2026 05:59:03 +0200 Subject: [PATCH 36/76] refactor(metadata): name an array one way in upgrade warnings Warnings about an upgraded document name the array by its store path, `str(StorePath(store, path))`, on every path that reads one: opening an array, `AsyncArray.from_dict`, a group read with or without consolidated metadata, and consolidated members, which were named relative to the group. `GroupMetadata.from_dict` and `ConsolidatedMetadata.from_dict` take the group's path for that. The consolidated test now also opens the group with `use_consolidated=False` and checks each array's name, so dropping the path on either read path fails a test. Assisted-by: ClaudeCode:claude-opus-5-5 Co-Authored-By: Claude Opus 5.5 --- src/zarr/core/array.py | 2 +- src/zarr/core/group.py | 31 +++++++++++++++++----------- tests/test_metadata/test_upgrades.py | 21 ++++++++++++------- 3 files changed, 33 insertions(+), 21 deletions(-) diff --git a/src/zarr/core/array.py b/src/zarr/core/array.py index 5f4cddb9b4..2f33b2f7ed 100644 --- a/src/zarr/core/array.py +++ b/src/zarr/core/array.py @@ -765,7 +765,7 @@ def from_dict( ValueError If the dictionary data is invalid or incompatible with either Zarr format 2 or 3 array creation. """ - metadata = parse_array_metadata(data) + metadata = parse_array_metadata(data, str(store_path)) return cls(metadata=metadata, store_path=store_path) @classmethod diff --git a/src/zarr/core/group.py b/src/zarr/core/group.py index 78071300f0..d6cc572465 100644 --- a/src/zarr/core/group.py +++ b/src/zarr/core/group.py @@ -163,7 +163,9 @@ def to_dict(self) -> dict[str, JSON]: } @classmethod - def from_dict(cls, data: dict[str, JSON]) -> ConsolidatedMetadata: + def from_dict(cls, data: dict[str, JSON], *, path: str | None = None) -> ConsolidatedMetadata: + """Read consolidated metadata, naming each member by its path under the group at + `path` (or relative to that group, when `path` is not given) in warnings.""" data = dict(data) kind = data.get("kind") @@ -177,6 +179,7 @@ def from_dict(cls, data: dict[str, JSON]) -> ConsolidatedMetadata: metadata: dict[str, ArrayV2Metadata | ArrayV3Metadata | GroupMetadata] = {} if raw_metadata: for k, v in raw_metadata.items(): + member = k if path is None else _join_paths([path, k]) if not isinstance(v, dict): raise TypeError( f"Invalid value for metadata items. key='{k}', type='{type(v).__name__}'" @@ -188,16 +191,16 @@ def from_dict(cls, data: dict[str, JSON]) -> ConsolidatedMetadata: if zarr_format == 3: node_type = parse_node_type(v.get("node_type", None)) if node_type == "group": - metadata[k] = GroupMetadata.from_dict(v) + metadata[k] = GroupMetadata.from_dict(v, path=member) elif node_type == "array": - metadata[k] = ArrayV3Metadata.from_dict(v, path=k) + metadata[k] = ArrayV3Metadata.from_dict(v, path=member) else: assert_never(node_type) elif zarr_format == 2: if "shape" in v: - metadata[k] = ArrayV2Metadata.from_dict(v, path=k) + metadata[k] = ArrayV2Metadata.from_dict(v, path=member) else: - metadata[k] = GroupMetadata.from_dict(v) + metadata[k] = GroupMetadata.from_dict(v, path=member) else: assert_never(zarr_format) @@ -415,7 +418,9 @@ def __init__( object.__setattr__(self, "consolidated_metadata", consolidated_metadata) @classmethod - def from_dict(cls, data: dict[str, Any]) -> GroupMetadata: + def from_dict(cls, data: dict[str, Any], *, path: str | None = None) -> GroupMetadata: + """Read a stored group document; `path` names the group in warnings about the + consolidated metadata it holds.""" data = dict(data) node_type = data.pop("node_type", None) if node_type not in ("group", None): @@ -424,7 +429,9 @@ def from_dict(cls, data: dict[str, Any]) -> GroupMetadata: ) consolidated_metadata = data.pop("consolidated_metadata", None) if consolidated_metadata: - data["consolidated_metadata"] = ConsolidatedMetadata.from_dict(consolidated_metadata) + data["consolidated_metadata"] = ConsolidatedMetadata.from_dict( + consolidated_metadata, path=path + ) zarr_format = data.get("zarr_format") if zarr_format == 2 or zarr_format is None: @@ -695,7 +702,7 @@ def from_dict( msg = f"Node type in metadata ({node_type}) is not 'group'" raise GroupNotFoundError(msg) return cls( - metadata=GroupMetadata.from_dict(data), + metadata=GroupMetadata.from_dict(data, path=str(store_path)), store_path=store_path, ) @@ -3505,7 +3512,7 @@ async def _read_metadata_v3(store: Store, path: str) -> ArrayV3Metadata | GroupM if zarr_json_bytes is None: raise FileNotFoundError(path) return _build_metadata_v3( - buffer_to_json_object(zarr_json_bytes), path=_join_paths([str(store), path]) + buffer_to_json_object(zarr_json_bytes), path=str(StorePath(store, path)) ) @@ -3541,7 +3548,7 @@ async def _read_metadata_v2(store: Store, path: str) -> ArrayV2Metadata | GroupM else: zmeta = buffer_to_json_object(zgroup_bytes) - return _build_metadata_v2(zmeta, zattrs, path=_join_paths([str(store), path])) + return _build_metadata_v2(zmeta, zattrs, path=str(StorePath(store, path))) async def _read_group_metadata_v2(store: Store, path: str) -> GroupMetadata: @@ -3585,7 +3592,7 @@ def _build_metadata_v3( case {"node_type": "array"}: return ArrayV3Metadata.from_dict(zarr_json, path=path) case {"node_type": "group"}: - return GroupMetadata.from_dict(zarr_json) + return GroupMetadata.from_dict(zarr_json, path=path) case _: # pragma: no cover raise ValueError( "invalid value for `node_type` key in metadata document" @@ -3602,7 +3609,7 @@ def _build_metadata_v2( case {"shape": _}: return ArrayV2Metadata.from_dict(zarr_json | {"attributes": attrs_json}, path=path) case _: # pragma: no cover - return GroupMetadata.from_dict(zarr_json | {"attributes": attrs_json}) + return GroupMetadata.from_dict(zarr_json | {"attributes": attrs_json}, path=path) @overload diff --git a/tests/test_metadata/test_upgrades.py b/tests/test_metadata/test_upgrades.py index b8e5ea3cc7..9256463f61 100644 --- a/tests/test_metadata/test_upgrades.py +++ b/tests/test_metadata/test_upgrades.py @@ -408,9 +408,9 @@ async def write_twice() -> None: @pytest.mark.filterwarnings("ignore:Consolidated metadata is currently not part:UserWarning") @pytest.mark.parametrize("zarr_format", [2, 3]) def test_legacy_chunk_size_consolidated(tmp_path: Path, zarr_format: Literal[2, 3]) -> None: - """Consolidated metadata goes through the same upgrade, with one warning naming each - array; re-saving the arrays and consolidating again leaves a group that opens - without a warning.""" + """Consolidated metadata goes through the same upgrade as the arrays' own documents, + with one warning naming each array by its path; re-saving the arrays and + consolidating again leaves a group that opens without a warning.""" path = tmp_path / "group.zarr" group = zarr.open_group(path, mode="w", zarr_format=zarr_format) names = ("a", "b") @@ -439,11 +439,16 @@ def test_legacy_chunk_size_consolidated(tmp_path: Path, zarr_format: Literal[2, ].update(chunk_shape=[0]), ) - with pytest.warns(ZarrUserWarning, match="zarr.consolidate_metadata") as record: - group = zarr.open_group(path, mode="r+") - assert sorted(str(w.message).split(":")[0] for w in record) == ["Array 'a'", "Array 'b'"] - for name in names: - array = group[name] + for use_consolidated in (True, False): + with warnings.catch_warnings(record=True) as record: + warnings.simplefilter("always", ZarrUserWarning) + group = zarr.open_group(path, mode="r+", use_consolidated=use_consolidated) + arrays = [group[name] for name in names] + assert all("zarr.consolidate_metadata" in str(w.message) for w in record) + assert sorted(str(w.message).split(": ")[0] for w in record) == [ + f"Array '{group.store_path / name}'" for name in names + ] + for array in arrays: assert isinstance(array, zarr.Array) assert array.chunks == (1,) array.update_attributes({}) From c1353bf2a446eed53034446592b0804c993d90c1 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Sat, 26 Sep 2026 05:59:15 +0200 Subject: [PATCH 37/76] refactor(metadata): one rule for bare sizes and edge lists; one noun in messages - `_validate_chunk_shapes` decides "edge list or bare size" with one `match`: a list or tuple is an edge list, anything else is a bare size checked by `parse_chunk_edge`, so a string dimension such as `"10"` is rejected as not an int instead of being iterated per character. Stored rectilinear `from_dict` makes the same JSON split and passes the dimension to `expand_rle`, so an invalid edge or RLE count inside a list names its dimension. - Messages say "dimension" throughout, lowercase after the colon ("Dimension 0: chunk edge length must be >= 1, got 0"); the upgrade warnings read "0 in dimension 0 as one chunk spanning the dimension". - The upgrades test for JSON arrays with `list` only (a stored document holds no tuples), and `_read_chunk_size` says what `span=None` means. - `ArrayV2Metadata.from_dict` copies the document once. - The rejected stored chunk shapes get one test per error case. Assisted-by: ClaudeCode:claude-opus-5-5 Co-Authored-By: Claude Opus 5.5 --- src/zarr/core/common.py | 20 +++---- src/zarr/core/metadata/upgrades.py | 15 +++--- src/zarr/core/metadata/v2.py | 8 +-- src/zarr/core/metadata/v3.py | 19 +++---- tests/test_metadata/test_upgrades.py | 78 ++++++++++++++++++---------- 5 files changed, 85 insertions(+), 55 deletions(-) diff --git a/src/zarr/core/common.py b/src/zarr/core/common.py index cddbf91912..feb8eb1cac 100644 --- a/src/zarr/core/common.py +++ b/src/zarr/core/common.py @@ -276,11 +276,12 @@ def _default_zarr_format() -> ZarrFormat: return cast("ZarrFormat", int(zarr_config.get("default_zarr_format", 3))) -def _parse_positive_int(value: object, name: str) -> int: +def _parse_positive_int(value: object, name: str, axis: int | None) -> int: + subject = name[0].upper() + name[1:] if axis is None else f"Dimension {axis}: {name}" if isinstance(value, bool) or not isinstance(value, int): - raise TypeError(f"{name} must be an int, got {value!r}") + raise TypeError(f"{subject} must be an int, got {value!r}") if value < 1: - raise ValueError(f"{name} must be >= 1, got {value!r}") + raise ValueError(f"{subject} must be >= 1, got {value!r}") return value @@ -290,8 +291,7 @@ def parse_chunk_edge(size: object, axis: int | None = None) -> int: This is the one rule for chunk edge lengths in metadata: bare chunk sizes, explicit edges and run-length encoded sizes. `axis`, when given, is named in the error. """ - where = "" if axis is None else f"Dimension {axis}: " - return _parse_positive_int(size, f"{where}Chunk edge length") + return _parse_positive_int(size, "chunk edge length", axis) def parse_chunk_shape(data: object) -> tuple[int, ...]: @@ -301,8 +301,9 @@ def parse_chunk_shape(data: object) -> tuple[int, ...]: return tuple(parse_chunk_edge(size, axis) for axis, size in enumerate(data)) -def expand_rle(data: Sequence[object]) -> list[int]: - """Expand a mixed array of bare integers and RLE pairs. +def expand_rle(data: Sequence[object], axis: int | None = None) -> list[int]: + """Expand a mixed array of bare integers and RLE pairs, the edges of dimension + `axis` (named in errors, when given). Per the rectilinear chunk grid spec, each element can be: - a bare integer (an explicit edge length) @@ -314,9 +315,10 @@ def expand_rle(data: Sequence[object]) -> list[int]: if len(item) != 2: raise ValueError(f"RLE entries must be an integer or [size, count], got {item}") size, count = item - result.extend([parse_chunk_edge(size)] * _parse_positive_int(count, "RLE repeat count")) + repeat = _parse_positive_int(count, "RLE repeat count", axis) + result.extend([parse_chunk_edge(size, axis)] * repeat) else: - result.append(parse_chunk_edge(item)) + result.append(parse_chunk_edge(item, axis)) return result diff --git a/src/zarr/core/metadata/upgrades.py b/src/zarr/core/metadata/upgrades.py index 0cdac3151d..1cc6916c47 100644 --- a/src/zarr/core/metadata/upgrades.py +++ b/src/zarr/core/metadata/upgrades.py @@ -49,9 +49,9 @@ def warn_readings(readings: Sequence[str], path: str | None) -> None: warnings.warn(f"{subject}{' '.join(readings)} {RESAVE_HINT}", ZarrUserWarning, stacklevel=2) -def _is_int_list(value: object) -> TypeGuard[list[int] | tuple[int, ...]]: +def _is_int_list(value: object) -> TypeGuard[list[int]]: """Whether `value` is a JSON array of integers (JSON `false` and `true` count).""" - return isinstance(value, list | tuple) and all(isinstance(v, int) for v in value) + return isinstance(value, list) and all(isinstance(v, int) for v in value) def _read_chunk_size(size: JSON, span: int | None, unit: int) -> tuple[int, str | None] | None: @@ -61,7 +61,8 @@ def _read_chunk_size(size: JSON, span: int | None, unit: int) -> tuple[int, str entry cannot be read, which leaves it for the metadata constructors to reject. A JSON int >= 1 is kept, JSON `true` is read as 1, and 0 or JSON `false` is read as one chunk spanning the axis of length `span`, a multiple of `unit` (the inner chunk - size of a shard), when the span is known. + size of a shard). `span` is `None` where no stored 0 is known, as in the inner + chunk shape of a sharding codec: 0 is then left for the constructors to reject. """ match size: case True: @@ -70,7 +71,7 @@ def _read_chunk_size(size: JSON, span: int | None, unit: int) -> tuple[int, str return size, None case int() if size == 0 and span is not None: edge = full_span_chunk_size(span, unit) - how = f"one chunk spanning the axis ({edge})" + how = f"one chunk spanning the dimension ({edge})" if span > 0: how += ( ", and as no chunk can be stored under a chunk size of 0, the array " @@ -89,7 +90,7 @@ def _read_chunk_shape( Returns the chunk shape and, if an entry is invalid, a sentence saying how the `name` was read; `None` if it cannot be read. """ - if not (isinstance(stored, list | tuple) and len(stored) == len(spans)): + if not (isinstance(stored, list) and len(stored) == len(spans)): return None edges: list[int] = [] readings: list[str] = [] @@ -101,7 +102,7 @@ def _read_chunk_shape( edge, how = read edges.append(edge) if how is not None: - readings.append(f"{json.dumps(size)} on axis {axis} as {how}") + readings.append(f"{json.dumps(size)} in dimension {axis} as {how}") if not readings: return edges, None return edges, ( @@ -124,7 +125,7 @@ def _sharding_codec(doc: ArrayDocument) -> tuple[Sequence[JSON], int, Mapping[st """The codec list of a Zarr format 3 array document, with the position and the configuration of its sharding codec, if it has one.""" codecs = doc.get("codecs") - if isinstance(codecs, list | tuple): + if isinstance(codecs, list): for index, codec in enumerate(codecs): if isinstance(codec, Mapping) and codec.get("name") == "sharding_indexed": configuration = codec.get("configuration") diff --git a/src/zarr/core/metadata/v2.py b/src/zarr/core/metadata/v2.py index b5701b3e78..2a1bbeee73 100644 --- a/src/zarr/core/metadata/v2.py +++ b/src/zarr/core/metadata/v2.py @@ -155,8 +155,8 @@ def from_dict(cls, data: dict[str, Any], *, path: str | None = None) -> ArrayV2M """Read a stored `.zarray` document (with its attributes). An invalid document that `zarr.core.metadata.upgrades` can read warns, naming the array at `path`.""" upgraded, readings = upgrade_array_document(data, V2_ARRAY_UPGRADES) - data = dict(upgraded) - _data = data.copy() + # a new dict, because we are modifying it + _data: dict[str, Any] = dict(upgraded) # Check that the zarr_format attribute is correct. _ = parse_zarr_format(_data.pop("zarr_format")) @@ -164,7 +164,7 @@ def from_dict(cls, data: dict[str, Any], *, path: str | None = None) -> ArrayV2M # which could be in filters or as a compressor. # we will reference a hard-coded collection of object codec ids for this search. - _filters, _compressor = (data.get("filters"), data.get("compressor")) + _filters, _compressor = (_data.get("filters"), _data.get("compressor")) if _filters is not None: _filters = cast("tuple[dict[str, JSON], ...]", _filters) object_codec_id = get_object_codec_id(tuple(_filters) + (_compressor,)) @@ -173,7 +173,7 @@ def from_dict(cls, data: dict[str, Any], *, path: str | None = None) -> ArrayV2M # we add a layer of indirection here around the dtype attribute of the array metadata # because we also need to know the object codec id, if any, to resolve the data type dtype_spec: DTypeSpec_V2 = { - "name": data["dtype"], + "name": _data["dtype"], "object_codec_id": object_codec_id, } dtype = get_data_type_from_json(dtype_spec, zarr_format=2) diff --git a/src/zarr/core/metadata/v3.py b/src/zarr/core/metadata/v3.py index ab5ddf3d2f..4eb8dd352c 100644 --- a/src/zarr/core/metadata/v3.py +++ b/src/zarr/core/metadata/v3.py @@ -239,13 +239,14 @@ def _validate_chunk_shapes( """ result: list[int | tuple[int, ...]] = [] for dim_idx, dim_spec in enumerate(chunk_shapes): - if isinstance(dim_spec, Iterable): - edges = tuple(parse_chunk_edge(edge, dim_idx) for edge in dim_spec) - if not edges: - raise ValueError(f"Dimension {dim_idx} has no chunk edges.") - result.append(edges) - else: - result.append(parse_chunk_edge(dim_spec, dim_idx)) + match dim_spec: + case list() | tuple(): + edges = tuple(parse_chunk_edge(edge, dim_idx) for edge in dim_spec) + if not edges: + raise ValueError(f"Dimension {dim_idx} has no chunk edges.") + result.append(edges) + case _: + result.append(parse_chunk_edge(dim_spec, dim_idx)) return tuple(result) @@ -363,8 +364,8 @@ def from_dict(cls, data: RectilinearChunkGridMetadataJSON) -> Self: # type: ign validate_rectilinear_kind(configuration.get("kind")) raw_shapes = configuration["chunk_shapes"] parsed = [ - tuple(expand_rle(dim_spec)) if isinstance(dim_spec, list) else dim_spec - for dim_spec in raw_shapes + tuple(expand_rle(dim_spec, axis)) if isinstance(dim_spec, list) else dim_spec + for axis, dim_spec in enumerate(raw_shapes) ] return cls(chunk_shapes=tuple(parsed)) diff --git a/tests/test_metadata/test_upgrades.py b/tests/test_metadata/test_upgrades.py index 9256463f61..b976660eb2 100644 --- a/tests/test_metadata/test_upgrades.py +++ b/tests/test_metadata/test_upgrades.py @@ -100,32 +100,40 @@ def _chunk_shapes(metadata: ArrayV2Metadata | ArrayV3Metadata) -> tuple[Any, Any ( _v2_doc([0, 4], [0, 4]), ((1, 4), None), - r"0 on axis 0 as one chunk spanning the axis \(1\)\.", + r"0 in dimension 0 as one chunk spanning the dimension \(1\)\.", ), ( _v3_doc([0], [False]), ((1,), None), - r"false on axis 0 as one chunk spanning the axis \(1\)\.", + r"false in dimension 0 as one chunk spanning the dimension \(1\)\.", ), - (_v2_doc([5], [True]), ((1,), None), "true on axis 0 as 1"), + (_v2_doc([5], [True]), ((1,), None), "true in dimension 0 as 1"), ( _v3_doc([5, 4], [True, 4]), ((1, 4), None), - r"\[true, 4\] is invalid.*true on axis 0 as 1", + r"\[true, 4\] is invalid.*true in dimension 0 as 1", ), ( _v2_doc([3], [0]), ((3,), None), - r"spanning the axis \(3\), and .* holds only its fill value", + r"spanning the dimension \(3\), and .* holds only its fill value", ), - (_v3_doc([4, 3], [4, 0]), ((4, 3), None), "0 on axis 1 as .* holds only its fill value"), - (_v3_doc([0], [0], inner=[4]), ((4,), (4,)), r"spanning the axis \(4\)\."), + ( + _v3_doc([4, 3], [4, 0]), + ((4, 3), None), + "0 in dimension 1 as .* holds only its fill value", + ), + (_v3_doc([0], [0], inner=[4]), ((4,), (4,)), r"spanning the dimension \(4\)\."), ( _v3_doc([10], [0], inner=[4]), ((12,), (4,)), - r"spanning the axis \(12\), and .* holds only its fill value", + r"spanning the dimension \(12\), and .* holds only its fill value", + ), + ( + _v3_doc([0, 3], [0, 3], inner=[2, 3]), + ((2, 3), (2, 3)), + r"spanning the dimension \(2\)\.", ), - (_v3_doc([0, 3], [0, 3], inner=[2, 3]), ((2, 3), (2, 3)), r"spanning the axis \(2\)\."), ( _v3_doc([5], [True], inner=[True]), ((1,), (1,)), @@ -203,24 +211,40 @@ def test_invalid_upgraded_document_raises_without_warning(doc: dict[str, JSON], metadata_cls.from_dict(doc) -@pytest.mark.parametrize( - ("doc", "error"), - [ - (_v2_doc([4], [-1]), "Dimension 0: Chunk edge length must be >= 1, got -1"), - (_v3_doc([4, 4], [0]), "Dimension 0: Chunk edge length must be >= 1, got 0"), - (_v3_doc([4], [4.0]), "Dimension 0: Chunk edge length must be an int, got 4.0"), - (_v3_doc([4], [0], inner=[0]), "Dimension 0: Chunk edge length must be >= 1, got 0"), - ], - ids=["negative", "ndim-mismatch", "float", "sharded-inner-zero"], -) -def test_stored_chunk_shape_not_upgraded(doc: dict[str, JSON], error: str) -> None: - """Invalid chunk sizes no known writer stored, and chunk shapes with the wrong - number of axes, are not upgraded, only rejected.""" +def _read_strictly(doc: dict[str, JSON]) -> ArrayV2Metadata | ArrayV3Metadata: + """Read `doc`, failing on any warning that it was upgraded.""" metadata_cls = ArrayV2Metadata if doc["zarr_format"] == 2 else ArrayV3Metadata with warnings.catch_warnings(): warnings.simplefilter("error", ZarrUserWarning) - with pytest.raises((TypeError, ValueError), match=re.escape(error)): - metadata_cls.from_dict(doc) + return metadata_cls.from_dict(doc) + + +def test_stored_negative_chunk_size_rejected() -> None: + """No known writer stored a negative chunk size: it is rejected, not upgraded.""" + with pytest.raises(ValueError, match="^Dimension 0: chunk edge length must be >= 1, got -1$"): + _read_strictly(_v2_doc([4], [-1])) + + +def test_stored_chunk_shape_ndim_mismatch_rejected() -> None: + """A chunk shape with the wrong number of dimensions is not upgraded, so its 0 is + rejected.""" + with pytest.raises(ValueError, match="^Dimension 0: chunk edge length must be >= 1, got 0$"): + _read_strictly(_v3_doc([4, 4], [0])) + + +def test_stored_float_chunk_size_rejected() -> None: + """No known writer stored a float chunk size: it is rejected, not upgraded.""" + with pytest.raises( + TypeError, match=r"^Dimension 0: chunk edge length must be an int, got 4\.0$" + ): + _read_strictly(_v3_doc([4], [4.0])) + + +def test_stored_zero_inner_chunk_size_rejected() -> None: + """No known writer stored an inner chunk size of 0, and no span defines one: it is + rejected, not upgraded.""" + with pytest.raises(ValueError, match="^Dimension 0: chunk edge length must be >= 1, got 0$"): + _read_strictly(_v3_doc([4], [0], inner=[0])) def test_v2_constructor_rejects_chunks_of_wrong_length() -> None: @@ -264,7 +288,7 @@ def _rectilinear_from_dict(chunk_shapes: list[Any]) -> RectilinearChunkGridMetad def test_metadata_rejects_non_int_chunk_edge(site: str, size: object) -> None: """Metadata built in code takes chunk edge lengths as `int`s only, everywhere.""" with pytest.raises( - TypeError, match=re.escape(f"Chunk edge length must be an int, got {size!r}") + TypeError, match=re.escape(f"Dimension 0: chunk edge length must be an int, got {size!r}") ): CHUNK_EDGE_SITES[site](size) @@ -276,7 +300,9 @@ def test_metadata_rejects_chunk_edge_below_one(site: str, size: int) -> None: without a warning.""" with warnings.catch_warnings(): warnings.simplefilter("error", ZarrUserWarning) - with pytest.raises(ValueError, match=f"Chunk edge length must be >= 1, got {size}"): + with pytest.raises( + ValueError, match=f"Dimension 0: chunk edge length must be >= 1, got {size}" + ): CHUNK_EDGE_SITES[site](size) From d18261834533fa9f79a6a5b59b759ca1c43aad50 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Sat, 26 Sep 2026 06:16:28 +0200 Subject: [PATCH 38/76] fix: encode metadata before deleting store content `Array.resize`, `AsyncGroup.delitem` and `create_hierarchy(overwrite=True)` now encode every document they will store (so the rectilinear gate runs) before deleting anything, then store the encoded documents. A group member deletion or hierarchy overwrite that would store a rectilinear grid with the flag off no longer deletes data first. `_resize` goes back to deleting before storing, which restores the behaviour of stores that cannot delete (a ZipStore resize fails before writing duplicate metadata). The gate error raised from `GroupMetadata.to_buffer_dict` names the consolidated member. Tests: one table of operations that must leave the store untouched without the flag (resize, chunk write, member deletion, hierarchy overwrite), the consolidation test for a member in a subgroup, a write with the flag that stores the rectilinear document, an abbreviated long chunk shape, and a ZipStore resize. Assisted-by: ClaudeCode:claude-opus-5-5 Co-Authored-By: Claude Opus 5.5 --- changes/4375.bugfix.md | 2 +- src/zarr/core/array.py | 40 ++++++---- src/zarr/core/group.py | 85 +++++++++++--------- src/zarr/core/metadata/io.py | 14 ++++ src/zarr/core/metadata/v3.py | 12 +-- tests/test_metadata/test_upgrades.py | 112 ++++++++++++++++++++------- tests/test_store/test_zip.py | 22 +++++- 7 files changed, 199 insertions(+), 88 deletions(-) diff --git a/changes/4375.bugfix.md b/changes/4375.bugfix.md index d30dd159d0..875356a2b3 100644 --- a/changes/4375.bugfix.md +++ b/changes/4375.bugfix.md @@ -1,3 +1,3 @@ Arrays whose stored `regular` chunk grid mixes chunk sizes with lists of chunk edge lengths, such as `"chunk_shape": [2, [5, 10, 5]]` (written by zarr 3.2.0 and 3.2.1 for `chunks=(2, (5, 10, 5))`), can be read again, without enabling `array.rectilinear_chunks`. The grid is read as the rectilinear chunk grid it describes, with a `ZarrUserWarning`; re-saving the metadata stores that rectilinear grid, which requires the flag. A regular chunk grid given edge lists is now rejected, so this metadata is no longer written. -The `array.rectilinear_chunks` flag now gates reading and storing array metadata that declares a rectilinear chunk grid, including in a group's consolidated metadata, instead of constructing `RectilinearChunkGridMetadata`. `Array.resize` now stores the new metadata before deleting chunks outside the new shape, so a resize whose metadata cannot be stored leaves the array unchanged. +The `array.rectilinear_chunks` flag now gates reading and storing array metadata that declares a rectilinear chunk grid, including in a group's consolidated metadata, instead of constructing `RectilinearChunkGridMetadata`. `Array.resize`, deleting a group member, and `create_hierarchy(..., overwrite=True)` now encode the metadata they will store before deleting anything, so metadata that cannot be stored (for example, a rectilinear chunk grid with the flag off) fails with the store untouched. That error names the array when it is a member of a group's consolidated metadata. diff --git a/src/zarr/core/array.py b/src/zarr/core/array.py index 5924180fe1..df611fdc42 100644 --- a/src/zarr/core/array.py +++ b/src/zarr/core/array.py @@ -118,7 +118,7 @@ ArrayV2MetadataDict, ArrayV3Metadata, ) -from zarr.core.metadata.io import save_metadata +from zarr.core.metadata.io import save_metadata, store_documents from zarr.core.metadata.v2 import ( CompressorLikev2, get_object_codec_id, @@ -5909,26 +5909,34 @@ async def _resize( # ensure deletion is only run if array is shrinking as the delete_outside_chunks path is unbounded in memory only_growing = all(new >= old for new, old in zip(new_shape, array.metadata.shape, strict=True)) - outside_keys: list[tuple[str]] = [] + + # Encode the new metadata before deleting any chunk: metadata that cannot be stored + # then fails with the store untouched. + documents = new_metadata.to_buffer_dict(default_buffer_prototype()) + if delete_outside_chunks and not only_growing: + # Remove all chunks outside of the new shape old_chunk_coords = set(array._chunk_grid.all_chunk_coords()) new_chunk_coords = set(new_chunk_grid.all_chunk_coords()) - outside_keys = [ - (array.metadata.encode_chunk_key(chunk_coords),) - for chunk_coords in old_chunk_coords.difference(new_chunk_coords) - ] - - # Store the new metadata before deleting any chunk: metadata that cannot be stored - # then fails with the store untouched, and a failed deletion leaves only chunks - # outside the new shape, as `delete_outside_chunks=False` does. - await save_metadata(array.store_path, new_metadata) - object.__setattr__(array, "metadata", new_metadata) - object.__setattr__(array, "_chunk_grid", new_chunk_grid) - async def _delete_key(key: str) -> None: - await (array.store_path / key).delete() + async def _delete_key(key: str) -> None: + await (array.store_path / key).delete() + + await concurrent_map( + [ + (array.metadata.encode_chunk_key(chunk_coords),) + for chunk_coords in old_chunk_coords.difference(new_chunk_coords) + ], + _delete_key, + zarr_config.get("async.concurrency"), + ) + + # Write new metadata + await store_documents(array.store_path, documents) - await concurrent_map(outside_keys, _delete_key, zarr_config.get("async.concurrency")) + # Update metadata and chunk_grid (in place) + object.__setattr__(array, "metadata", new_metadata) + object.__setattr__(array, "_chunk_grid", new_chunk_grid) async def _append( diff --git a/src/zarr/core/group.py b/src/zarr/core/group.py index 06c74c3e52..08352063ed 100644 --- a/src/zarr/core/group.py +++ b/src/zarr/core/group.py @@ -48,7 +48,7 @@ from zarr.core.dtype import parse_data_type from zarr.core.json_parse import parse_field from zarr.core.metadata import ArrayV2Metadata, ArrayV3Metadata -from zarr.core.metadata.io import save_metadata +from zarr.core.metadata.io import save_metadata, store_documents from zarr.core.metadata.v3 import check_storable from zarr.core.sync import SyncMixin, sync from zarr.errors import ( @@ -362,9 +362,9 @@ class GroupMetadata(Metadata): def to_buffer_dict(self, prototype: BufferPrototype) -> dict[str, Buffer]: if self.consolidated_metadata is not None: - for member in self.consolidated_metadata.flattened_metadata.values(): + for path, member in self.consolidated_metadata.flattened_metadata.items(): if isinstance(member, ArrayV3Metadata): - check_storable(member) + check_storable(member, f"Array {path!r} in the consolidated metadata: ") indent = config.get("json_indent") if self.zarr_format == 3: return {ZARR_JSON: json_to_buffer(self.to_dict(), prototype=prototype, indent=indent)} @@ -811,11 +811,20 @@ async def delitem(self, key: str) -> None: Array or group name """ store_path = self.store_path / key - + consolidated = self.metadata.consolidated_metadata + if consolidated is None: + await store_path.delete_dir() + return + members = {name: node for name, node in consolidated.metadata.items() if name != key} + metadata = replace( + self.metadata, consolidated_metadata=replace(consolidated, metadata=members) + ) + # Encode the group metadata before deleting the member: metadata that cannot be + # stored then fails with the store untouched. + documents = metadata.to_buffer_dict(default_buffer_prototype()) await store_path.delete_dir() - if self.metadata.consolidated_metadata: - self.metadata.consolidated_metadata.metadata.pop(key, None) - await self._save_metadata() + await store_documents(self.store_path, documents) + object.__setattr__(self, "metadata", metadata) async def get[DefaultT]( self, key: str, default: DefaultT | None = None @@ -3090,6 +3099,7 @@ async def create_hierarchy( # ensure that all nodes have the same zarr_format, and add implicit groups as needed nodes_parsed = _parse_hierarchy_dict(data=nodes_normed_keys) redundant_implicit_groups = [] + to_delete_keys: list[str] = [] # empty hierarchies should be a no-op if len(nodes_parsed) > 0: @@ -3122,13 +3132,7 @@ async def create_hierarchy( if overwrite: # we will remove any nodes that collide with arrays and non-implicit groups defined in # nodes - - # track the keys of nodes we need to delete - to_delete_keys = [] - to_delete_keys.extend( - [k for k, v in nodes_parsed.items() if k not in implicit_group_keys] - ) - await asyncio.gather(*(store.delete_dir(key) for key in to_delete_keys)) + to_delete_keys = [k for k in nodes_parsed if k not in implicit_group_keys] else: # This type is long. coros: ( @@ -3188,7 +3192,11 @@ async def create_hierarchy( else: nodes_explicit[k] = v - async for key, node in create_nodes(store=store, nodes=nodes_explicit): + # Encode every node before deleting anything: a node whose metadata cannot be stored + # then fails with the store untouched. + documents = _encode_nodes(nodes_explicit) + await asyncio.gather(*(store.delete_dir(key) for key in to_delete_keys)) + async for key, node in _store_nodes(store, nodes_explicit, documents): yield key, node @@ -3218,15 +3226,35 @@ async def create_nodes( AsyncGroup | AsyncArray The created nodes in the order they are created. """ + async for key, node in _store_nodes(store, nodes, _encode_nodes(nodes)): + yield key, node + + +def _encode_nodes( + nodes: Mapping[str, GroupMetadata | ArrayV2Metadata | ArrayV3Metadata], +) -> dict[str, Buffer]: + """The metadata documents of `nodes`, by their keys in the store.""" + prototype = default_buffer_prototype() + return { + _join_paths([path, key]): value + for path, metadata in nodes.items() + for key, value in metadata.to_buffer_dict(prototype).items() + } + +async def _store_nodes( + store: Store, + nodes: Mapping[str, GroupMetadata | ArrayV2Metadata | ArrayV3Metadata], + documents: Mapping[str, Buffer], +) -> AsyncIterator[tuple[str, AsyncGroup | AnyAsyncArray]]: + """Store the `documents` encoded from `nodes` (see `create_nodes`).""" # Note: the only way to alter this value is via the config. If that's undesirable for some reason, # then we should consider adding a keyword argument to this function semaphore = asyncio.Semaphore(config.get("async.concurrency")) - create_tasks: list[Coroutine[None, None, str]] = [] - - for key, value in nodes.items(): - # make the key absolute - create_tasks.extend(_persist_metadata(store, key, value, semaphore=semaphore)) + create_tasks = [ + _set_return_key(store=store, key=key, value=value, semaphore=semaphore) + for key, value in documents.items() + ] created_object_keys = [] @@ -3739,23 +3767,6 @@ async def _set_return_key( return key -def _persist_metadata( - store: Store, - path: str, - metadata: ArrayV2Metadata | ArrayV3Metadata | GroupMetadata, - semaphore: asyncio.Semaphore | None = None, -) -> tuple[Coroutine[None, None, str], ...]: - """ - Prepare to save a metadata document to storage, returning a tuple of coroutines that must be awaited. - """ - - to_save = metadata.to_buffer_dict(default_buffer_prototype()) - return tuple( - _set_return_key(store=store, key=_join_paths([path, key]), value=value, semaphore=semaphore) - for key, value in to_save.items() - ) - - async def create_rooted_hierarchy( *, store: Store, diff --git a/src/zarr/core/metadata/io.py b/src/zarr/core/metadata/io.py index 7b63f5493b..926a637fd4 100644 --- a/src/zarr/core/metadata/io.py +++ b/src/zarr/core/metadata/io.py @@ -9,6 +9,9 @@ from zarr.storage._common import StorePath, ensure_no_existing_node if TYPE_CHECKING: + from collections.abc import Mapping + + from zarr.core.buffer import Buffer from zarr.core.common import ZarrFormat from zarr.core.group import GroupMetadata from zarr.core.metadata import ArrayMetadata @@ -33,6 +36,17 @@ def _build_parents(store_path: StorePath, zarr_format: ZarrFormat) -> dict[str, return parents +async def store_documents(store_path: StorePath, documents: Mapping[str, Buffer]) -> None: + """Store metadata documents encoded by `to_buffer_dict` under `store_path`. + + An operation that deletes or overwrites store content encodes the documents it will + store first, so that metadata that cannot be stored fails with the store untouched. + """ + await asyncio.gather( + *(set_or_delete(store_path / key, value) for key, value in documents.items()) + ) + + async def save_metadata( store_path: StorePath, metadata: ArrayMetadata | GroupMetadata, ensure_parents: bool = False ) -> None: diff --git a/src/zarr/core/metadata/v3.py b/src/zarr/core/metadata/v3.py index 3b3aa16a80..29948c4d56 100644 --- a/src/zarr/core/metadata/v3.py +++ b/src/zarr/core/metadata/v3.py @@ -357,30 +357,30 @@ def from_dict(cls, data: RectilinearChunkGridMetadataJSON) -> Self: # type: ign ChunkGridMetadata = RegularChunkGridMetadata | RectilinearChunkGridMetadata -def _check_rectilinear_chunks_enabled() -> None: - """Raise unless rectilinear chunks are enabled. +def _check_rectilinear_chunks_enabled(subject: str = "") -> None: + """Raise unless rectilinear chunks are enabled, with `subject` leading the error. The flag gates storing and reading array metadata documents that declare a rectilinear chunk grid; the chunk grid metadata classes themselves are not gated. """ if not config.get("array.rectilinear_chunks"): raise ValueError( - "Rectilinear chunk grids are experimental and disabled by default. " + f"{subject}Rectilinear chunk grids are experimental and disabled by default. " "Enable them with: zarr.config.set({'array.rectilinear_chunks': True}) " "or set the environment variable ZARR_ARRAY__RECTILINEAR_CHUNKS=True" ) -def check_storable(metadata: ArrayV3Metadata) -> None: +def check_storable(metadata: ArrayV3Metadata, subject: str = "") -> None: """Raise if `metadata` may not be stored: a rectilinear chunk grid requires the - rectilinear chunks flag. + rectilinear chunks flag. `subject`, when given, names the array in the error. Every serialization of array metadata for a store calls this, before the store is touched: `ArrayV3Metadata.to_buffer_dict` and, for the arrays in a group's consolidated metadata, `GroupMetadata.to_buffer_dict`. """ if isinstance(metadata.chunk_grid, RectilinearChunkGridMetadata): - _check_rectilinear_chunks_enabled() + _check_rectilinear_chunks_enabled(subject) def create_chunk_grid_metadata( diff --git a/tests/test_metadata/test_upgrades.py b/tests/test_metadata/test_upgrades.py index 08615b3430..564f6a33a4 100644 --- a/tests/test_metadata/test_upgrades.py +++ b/tests/test_metadata/test_upgrades.py @@ -24,6 +24,7 @@ from zarr.core.sync import sync from zarr.dtype import Int16 from zarr.errors import ZarrUserWarning +from zarr.storage import LocalStore if TYPE_CHECKING: from collections.abc import Callable @@ -538,9 +539,9 @@ def _mixed_doc(shape: list[int], chunk_shape: list[Any]) -> dict[str, JSON]: ), ), ( - _mixed_doc([4, 10_000], [2, [10] * 1000]), - (2, (10,) * 1000), - r"^The stored chunk grid .* \[1\]", + _mixed_doc([4, 10_000], [True, [10] * 1000]), + (1, (10,) * 1000), + r"^The stored chunk shape \[true, \[10, 10, .*\.\.\. is invalid: .* read as \[1, \[10, .*\.\.\.,", ), ], ids=[ @@ -627,9 +628,25 @@ def _store_mixed_array(path: Path) -> np.ndarray[Any, np.dtype[np.float32]]: return data -def test_edge_lists_in_regular_grid_round_trip(tmp_path: Path) -> None: - """A store holding the verbatim document opens without the rectilinear chunks flag, - reads its data, re-saves as rectilinear with the flag, and then opens cleanly.""" +def _update_attributes(arr: zarr.Array[Any], data: np.ndarray[Any, Any]) -> np.ndarray[Any, Any]: + arr.update_attributes({}) + return data + + +def _write(arr: zarr.Array[Any], data: np.ndarray[Any, Any]) -> np.ndarray[Any, Any]: + arr[:2] = -data[:2] + return np.concatenate([-data[:2], data[2:]]) + + +@pytest.mark.parametrize("store_metadata", [_update_attributes, _write], ids=["re-save", "write"]) +def test_edge_lists_in_regular_grid_round_trip( + tmp_path: Path, + store_metadata: Callable[[zarr.Array[Any], np.ndarray[Any, Any]], np.ndarray[Any, Any]], +) -> None: + """A store holding the verbatim document opens without the rectilinear chunks flag + and reads its data. With the flag, re-saving the metadata, or writing chunks (which + stores the metadata first), stores the rectilinear chunk grid, which then opens + cleanly.""" path = tmp_path / "mixed.zarr" data = _store_mixed_array(path) @@ -641,7 +658,7 @@ def test_edge_lists_in_regular_grid_round_trip(tmp_path: Path) -> None: np.testing.assert_array_equal(arr[...], data) with zarr.config.set({"array.rectilinear_chunks": True}): - arr.update_attributes({}) + expected = store_metadata(arr, data) assert json.loads((path / "zarr.json").read_text())["chunk_grid"] == { "name": "rectilinear", "configuration": {"kind": "inline", "chunk_shapes": [2, [5, 10, 5]]}, @@ -649,42 +666,83 @@ def test_edge_lists_in_regular_grid_round_trip(tmp_path: Path) -> None: with warnings.catch_warnings(): warnings.simplefilter("error", ZarrUserWarning) reopened = zarr.open_array(path) - np.testing.assert_array_equal(reopened[...], data) + np.testing.assert_array_equal(reopened[...], expected) -def test_resize_without_flag_leaves_store_intact(tmp_path: Path) -> None: - """Resizing an array read from the verbatim document without the flag cannot store - its metadata, and fails before deleting any chunk.""" - path = tmp_path / "mixed.zarr" - data = _store_mixed_array(path) +def _store_mixed_group(path: Path) -> None: + """A group at `path` holding the verbatim document at `mixed` and a regular array + `n`, with consolidated metadata that quotes the verbatim document.""" + group = zarr.open_group(path, mode="w") + _store_mixed_array(path / "mixed") + group.create_array("n", data=np.arange(4), chunks=(2,)) + with zarr.config.set({"array.rectilinear_chunks": True}): + zarr.consolidate_metadata(path) + group_doc = json.loads((path / "zarr.json").read_text()) + group_doc["consolidated_metadata"]["metadata"]["mixed"] = json.loads(MIXED_REGULAR_GRID_DOC) + (path / "zarr.json").write_text(json.dumps(group_doc)) + + +def _resize(path: Path) -> None: + zarr.open_array(path / "mixed", mode="a").resize((6, 5)) + + +def _write_chunks(path: Path) -> None: + zarr.open_array(path / "mixed", mode="a")[...] = 1 + + +def _delete_member(path: Path) -> None: + del zarr.open_group(path, mode="a")["n"] + + +def _overwrite_hierarchy(path: Path) -> None: + mixed = zarr.open_array(path / "mixed") + list(zarr.create_hierarchy(store=LocalStore(path), nodes={"n": mixed.metadata}, overwrite=True)) + + +@pytest.mark.filterwarnings( + "ignore:.*read as that rectilinear chunk grid:zarr.errors.ZarrUserWarning" +) +@pytest.mark.filterwarnings("ignore:Consolidated metadata is currently not part:UserWarning") +@pytest.mark.parametrize( + "action", + [_resize, _write_chunks, _delete_member, _overwrite_hierarchy], + ids=["resize", "write", "delete-member", "overwrite-hierarchy"], +) +def test_store_untouched_without_flag(tmp_path: Path, action: Callable[[Path], None]) -> None: + """An operation that would store the rectilinear chunk grid read from the verbatim + document fails without the flag before it deletes or writes anything.""" + path = tmp_path / "group.zarr" + _store_mixed_group(path) stored = {p: p.read_bytes() for p in path.rglob("*") if p.is_file()} - with zarr.config.set({"array.rectilinear_chunks": False}): - with pytest.warns(ZarrUserWarning, match="read as that rectilinear chunk grid"): - arr = zarr.open_array(path, mode="a") - with pytest.raises(ValueError, match="experimental and disabled by default"): - arr.resize((6, 5)) + with ( + zarr.config.set({"array.rectilinear_chunks": False}), + pytest.raises(ValueError, match="experimental and disabled by default"), + ): + action(path) assert {p: p.read_bytes() for p in path.rglob("*") if p.is_file()} == stored - assert arr.shape == data.shape - np.testing.assert_array_equal(arr[...], data) @pytest.mark.filterwarnings("ignore:Consolidated metadata is currently not part:UserWarning") -def test_consolidate_without_flag_leaves_group_readable(tmp_path: Path) -> None: +@pytest.mark.parametrize("member", ["mixed", "sub/mixed"]) +def test_consolidate_without_flag_leaves_group_readable(tmp_path: Path, member: str) -> None: """Consolidating a group holding an array read from the verbatim document would - store its rectilinear chunk grid, so without the flag it fails with the store - untouched, and the group still opens without the flag.""" + store its rectilinear chunk grid, so without the flag it fails, naming the array, + with the store untouched, and the group still opens without the flag.""" path = tmp_path / "group.zarr" - zarr.open_group(path, mode="w") - data = _store_mixed_array(path / "mixed") + zarr.open_group(path, mode="w").create_group("sub") + data = _store_mixed_array(path / member) group_doc = (path / "zarr.json").read_bytes() with zarr.config.set({"array.rectilinear_chunks": False}): with ( pytest.warns(ZarrUserWarning, match="read as that rectilinear chunk grid"), - pytest.raises(ValueError, match="experimental and disabled by default"), + pytest.raises( + ValueError, + match=f"^Array '{member}' in the consolidated metadata: .* experimental and", + ), ): zarr.consolidate_metadata(path) assert (path / "zarr.json").read_bytes() == group_doc with pytest.warns(ZarrUserWarning, match="read as that rectilinear chunk grid"): - mixed = zarr.open_group(path, mode="r")["mixed"] + mixed = zarr.open_group(path, mode="r")[member] assert isinstance(mixed, zarr.Array) np.testing.assert_array_equal(mixed[...], data) diff --git a/tests/test_store/test_zip.py b/tests/test_store/test_zip.py index 32b18c5273..8931875125 100644 --- a/tests/test_store/test_zip.py +++ b/tests/test_store/test_zip.py @@ -6,7 +6,7 @@ import shutil import tempfile import zipfile -from typing import TYPE_CHECKING +from typing import TYPE_CHECKING, Literal import numpy as np import pytest @@ -407,3 +407,23 @@ def test_zipstore_close_lifecycle(tmp_path: Path) -> None: lambda: ZipStoreLifecycleMachine(tmp_path), settings=settings(max_examples=50, deadline=None), ) + + +@pytest.mark.parametrize("zarr_format", [2, 3]) +def test_resize_needing_deletes_leaves_array_unchanged( + tmp_path: Path, zarr_format: Literal[2, 3] +) -> None: + """A zip store cannot delete, so a resize that must delete chunks fails before it + stores the new metadata.""" + path = tmp_path / "a.zip" + store = ZipStore(path, mode="w") + arr = create_array(store, shape=(20,), chunks=(5,), dtype="i4", zarr_format=zarr_format) + arr[:] = np.arange(20) + with pytest.raises(NotImplementedError): + arr.resize((5,)) + assert arr.shape == (20,) + store.close() + names = zipfile.ZipFile(path).namelist() + assert len(names) == len(set(names)) + reopened = zarr.open_array(ZipStore(path, mode="r"), mode="r") + np.testing.assert_array_equal(reopened[:], np.arange(20)) From d3a6eae3aa146087528268c4d7e48fa142b94498 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Sat, 26 Sep 2026 06:16:28 +0200 Subject: [PATCH 39/76] test: reject a run-length entry that is not a pair Assisted-by: ClaudeCode:claude-opus-5-5 Co-Authored-By: Claude Opus 5.5 --- tests/test_unified_chunk_grid.py | 2 ++ 1 file changed, 2 insertions(+) diff --git a/tests/test_unified_chunk_grid.py b/tests/test_unified_chunk_grid.py index 513e78a8ab..44610537ea 100644 --- a/tests/test_unified_chunk_grid.py +++ b/tests/test_unified_chunk_grid.py @@ -549,6 +549,7 @@ def test_rle_roundtrip() -> None: ([[-10, 2]], "Chunk edge length must be >= 1"), ([[5, 0]], "RLE repeat count must be >= 1"), ([[5, -1]], "RLE repeat count must be >= 1"), + ([[5, 2, 1]], r"RLE entries must be an integer or \[size, count\], got \[5, 2, 1\]"), ], ids=[ "zero-edge", @@ -557,6 +558,7 @@ def test_rle_roundtrip() -> None: "negative-rle-size", "zero-rle-count", "negative-rle-count", + "rle-entry-of-three", ], ) def test_rle_expand_rejects_invalid(rle_input: list[Any], match: str) -> None: From f081f1b6a2174ec542bdbe835b1d43c528499257 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Sat, 26 Sep 2026 10:12:54 +0200 Subject: [PATCH 40/76] fix(array): store the upgrade of the current stored document before writing A handle read from an upgraded document upserts the upgrade of what the store holds on its first non-empty chunk write: a stale handle no longer rolls back newer metadata, and an empty write stores nothing. consolidate_metadata does the same for each upgraded member before writing the consolidated document. Both go through two internal primitives in zarr.core.metadata.io: diff_documents compares a node's stored documents with those its metadata would store, value by value, and upsert_metadata encodes first, then stores only the documents that differ and returns the changes. The upgraded-document flag is a ClassVar set on the instance by one helper, mark_upgraded, which also warns; the upgrades are keyed by Zarr format; the Metadata.to_dict and V2 from_dict changes are reverted. _append goes through AsyncArray.setitem and _setitem is removed. A chunk shape that is not a list or tuple is rejected as a whole, and every expand_rle error names its dimension. Assisted-by: ClaudeCode:claude-opus-5-5 Co-Authored-By: Claude Opus 5.5 --- changes/4334.bugfix.md | 4 +- src/zarr/abc/metadata.py | 5 +- src/zarr/api/asynchronous.py | 11 +- src/zarr/codecs/sharding.py | 2 +- src/zarr/core/array.py | 89 ++++-------- src/zarr/core/common.py | 20 ++- src/zarr/core/metadata/io.py | 95 ++++++++++++- src/zarr/core/metadata/upgrades.py | 33 +++-- src/zarr/core/metadata/v2.py | 22 ++- src/zarr/core/metadata/v3.py | 34 ++--- tests/test_metadata/test_io.py | 199 +++++++++++++++++++++++++++ tests/test_metadata/test_upgrades.py | 131 ++++++++++++++++-- tests/test_unified_chunk_grid.py | 16 +++ 13 files changed, 515 insertions(+), 146 deletions(-) create mode 100644 tests/test_metadata/test_io.py diff --git a/changes/4334.bugfix.md b/changes/4334.bugfix.md index 4c55701757..460ba6ece8 100644 --- a/changes/4334.bugfix.md +++ b/changes/4334.bugfix.md @@ -1,5 +1,5 @@ A chunk edge length is now always at least 1, while an array extent may be 0. Every spelling of one chunk spanning an axis (`chunks=-1`, `chunks=False`, `chunks="auto"`, `shards=-1`, `shards=False`) gives chunk size 1 on a zero-length axis, and a shard spanning an axis is a multiple of the inner chunk size, so `shards=-1` no longer fails when the axis length is not a multiple of it. Rectilinear chunk grids can be created on a zero-length dimension with a non-empty list of positive chunk sizes, which are kept for later growth. -Metadata built in code is strict about chunk edge lengths: `ArrayV2Metadata`, `RegularChunkGridMetadata`, `RectilinearChunkGridMetadata` and `ShardingCodec` take them as Python `int`s of at least 1, so a size of 0 now raises a `ValueError`, and a `bool`, a float or a NumPy integer (including a scalar `chunks=np.int64(5)` for `ArrayV2Metadata`) raises a `TypeError`; `FixedDimension(size=0, ...)` raises a `ValueError`. Stored rectilinear chunk grids with integral floats (`10.0`) are rejected too. The array creation functions still accept NumPy integers in `chunks=` and `shards=`. +Metadata built in code is strict about chunk edge lengths: `ArrayV2Metadata`, `RegularChunkGridMetadata`, `RectilinearChunkGridMetadata` and `ShardingCodec` take them as Python `int`s of at least 1, so a size of 0 now raises a `ValueError`, and a `bool`, a float or a NumPy integer (including a scalar `chunks=np.int64(5)` for `ArrayV2Metadata`) raises a `TypeError`, as does a chunk shape that is not a list or tuple; `FixedDimension(size=0, ...)` raises a `ValueError`. Stored rectilinear chunk grids with integral floats (`10.0`) are rejected too. The array creation functions still accept NumPy integers in `chunks=` and `shards=`. -Stored metadata with a regular chunk size of 0 or JSON `false` now opens with a `ZarrUserWarning` and is read as one chunk spanning the axis (a multiple of the inner chunk size for sharded arrays, which previously failed to open). zarr-python wrote such sizes for arrays created with a zero-length axis until 3.4, and, from 2.18.7 to 3.2.1, for an explicit chunk size of 0 or `False` on an axis of any length, as in `chunks=(0,)` (those arrays could store no data). JSON `true`, as zarr-python 3.0 and 3.2 wrote for `chunks=(True,)`, is read as 1, in the chunk grid and in the inner chunk shape of a sharding codec. The warning names the array and says how to store valid metadata: `array.update_attributes({})`, then `zarr.consolidate_metadata` if the metadata is consolidated. Writing data to such an array first stores the upgraded metadata, so that other readers find the chunks written. +Stored metadata with a regular chunk size of 0 or JSON `false` now opens with a `ZarrUserWarning` and is read as one chunk spanning the axis (a multiple of the inner chunk size for sharded arrays, which previously failed to open). zarr-python wrote such sizes for arrays created with a zero-length axis until 3.4, and, from 2.18.7 to 3.2.1, for an explicit chunk size of 0 or `False` on an axis of any length, as in `chunks=(0,)` (those arrays could store no data). JSON `true`, as zarr-python 3.0 and 3.2 wrote for `chunks=(True,)`, is read as 1, in the chunk grid and in the inner chunk shape of a sharding codec. The warning names the array and says how to store valid metadata: `array.update_attributes({})`, then `zarr.consolidate_metadata` if the metadata is consolidated. Writing data to such an array (other than an empty selection) first stores the upgrade of the metadata the store then holds, unless another writer has stored valid metadata since, so that other readers find the chunks written; `zarr.consolidate_metadata` likewise stores the upgraded metadata of each such array before the consolidated metadata. diff --git a/src/zarr/abc/metadata.py b/src/zarr/abc/metadata.py index 1104111b63..a56f986645 100644 --- a/src/zarr/abc/metadata.py +++ b/src/zarr/abc/metadata.py @@ -20,13 +20,10 @@ def to_dict(self) -> dict[str, JSON]: Recursively serialize this model to a dictionary. This method inspects the fields of self and calls `x.to_dict()` for any fields that are instances of `Metadata`. Sequences of `Metadata` are similarly recursed into, and - the output of that recursion is collected in a list. Fields declared with - `compare=False` are not part of the document. + the output of that recursion is collected in a list. """ out_dict = {} for field in fields(self): - if not field.compare: - continue key = field.name value = getattr(self, key) if isinstance(value, Metadata): diff --git a/src/zarr/api/asynchronous.py b/src/zarr/api/asynchronous.py index 1fc10cdd1e..4832bafaba 100644 --- a/src/zarr/api/asynchronous.py +++ b/src/zarr/api/asynchronous.py @@ -231,10 +231,15 @@ async def consolidate_metadata( group = await AsyncGroup.open(store_path, zarr_format=zarr_format, use_consolidated=False) group.store_path.store._check_writable() - members_metadata = { - k: v.metadata - async for k, v in group.members(max_depth=None, use_consolidated_for_children=False) + members = { + k: v async for k, v in group.members(max_depth=None, use_consolidated_for_children=False) } + # Store the upgrade of every member document that had to be upgraded before the + # consolidated document that describes the upgraded metadata. + for member in members.values(): + if isinstance(member, AsyncArray): + await member._store_upgraded_document() + members_metadata = {k: v.metadata for k, v in members.items()} # While consolidating, we want to be explicit about when child groups # are empty by inserting an empty dict for consolidated_metadata.metadata for k, v in members_metadata.items(): diff --git a/src/zarr/codecs/sharding.py b/src/zarr/codecs/sharding.py index 37fe52c1a6..52d100bb91 100644 --- a/src/zarr/codecs/sharding.py +++ b/src/zarr/codecs/sharding.py @@ -458,7 +458,7 @@ class ShardingCodec( def __init__( self, *, - chunk_shape: Iterable[int], + chunk_shape: tuple[int, ...] | list[int], codecs: Iterable[Codec | dict[str, JSON]] = (BytesCodec(),), index_codecs: Iterable[Codec | dict[str, JSON]] = (BytesCodec(), Crc32cCodec()), index_location: ShardingCodecIndexLocation | IndexLocation = "end", diff --git a/src/zarr/core/array.py b/src/zarr/core/array.py index 2f33b2f7ed..64d6080838 100644 --- a/src/zarr/core/array.py +++ b/src/zarr/core/array.py @@ -118,7 +118,8 @@ ArrayV2MetadataDict, ArrayV3Metadata, ) -from zarr.core.metadata.io import save_metadata +from zarr.core.metadata.io import save_metadata, upsert_metadata +from zarr.core.metadata.upgrades import upgrade_array_document from zarr.core.metadata.v2 import ( CompressorLikev2, get_object_codec_id, @@ -1615,6 +1616,24 @@ async def _save_metadata(self, metadata: ArrayMetadata, ensure_parents: bool = F """ await save_metadata(self.store_path, metadata, ensure_parents=ensure_parents) + async def _store_upgraded_document(self) -> None: + """Store the upgrade of this array's current stored document, if it differs from + what the store holds. + + Only for metadata read from a document that had to be upgraded (see + `zarr.core.metadata.upgrades`). The document is read again, because the store may + hold a newer one than this handle's metadata: if that one needs no upgrade (the + array was re-saved or resized since), nothing is stored. Storing the same upgrade + twice is harmless, so concurrent callers need no coordination. + """ + if not self.metadata._stored_document_upgraded: + return + zarr_format = self.metadata.zarr_format + stored = await get_array_metadata(self.store_path, zarr_format=zarr_format) + upgraded, _ = upgrade_array_document(stored, zarr_format) + await upsert_metadata(self.store_path, parse_array_metadata(upgraded)) + object.__setattr__(self.metadata, "_stored_document_upgraded", False) + async def _set_selection( self, indexer: Indexer, @@ -1623,12 +1642,10 @@ async def _set_selection( prototype: BufferPrototype, fields: Fields | None = None, ) -> None: - if self.metadata._stored_document_upgraded: + if product(indexer.shape) > 0: # Chunks are about to be stored under the upgraded metadata, so store it # first: every reader of the store then agrees with them. - metadata = replace(self.metadata) - await self._save_metadata(metadata) - object.__setattr__(self, "metadata", metadata) + await self._store_upgraded_document() return await _set_selection( self.store_path, self.metadata, @@ -5827,58 +5844,6 @@ async def _set_selection( ) -async def _setitem( - store_path: StorePath, - metadata: ArrayMetadata, - codec_pipeline: CodecPipeline, - config: ArrayConfig, - chunk_grid: ChunkGrid, - selection: BasicSelection, - value: npt.ArrayLike, - prototype: BufferPrototype | None = None, -) -> None: - """ - Set values in the array using basic indexing. - - Parameters - ---------- - store_path : StorePath - The store path of the array. - metadata : ArrayMetadata - The array metadata. - codec_pipeline : CodecPipeline - The codec pipeline for encoding/decoding. - config : ArrayConfig - The array configuration. - chunk_grid : ChunkGrid - The chunk grid. - selection : BasicSelection - The selection defining the region of the array to set. - value : npt.ArrayLike - The values to be written into the selected region of the array. - prototype : BufferPrototype or None, optional - A prototype buffer that defines the structure and properties of the array chunks being modified. - If None, the default buffer prototype is used. - """ - if prototype is None: - prototype = default_buffer_prototype() - indexer = BasicIndexer( - selection, - shape=metadata.shape, - chunk_grid=chunk_grid, - ) - return await _set_selection( - store_path, - metadata, - codec_pipeline, - config, - chunk_grid, - indexer, - value, - prototype=prototype, - ) - - async def _resize( array: AsyncArray[ArrayV2Metadata] | AsyncArray[ArrayV3Metadata], new_shape: ShapeLike, @@ -5993,15 +5958,7 @@ async def _append( slice(None) if i != axis else slice(old_shape[i], new_shape[i]) for i in range(len(array.shape)) ) - await _setitem( - array.store_path, - array.metadata, - array.codec_pipeline, - array.config, - array._chunk_grid, - append_selection, - data, - ) + await array.setitem(append_selection, data) return new_shape diff --git a/src/zarr/core/common.py b/src/zarr/core/common.py index feb8eb1cac..3967c8ca34 100644 --- a/src/zarr/core/common.py +++ b/src/zarr/core/common.py @@ -276,8 +276,13 @@ def _default_zarr_format() -> ZarrFormat: return cast("ZarrFormat", int(zarr_config.get("default_zarr_format", 3))) +def _subject(name: str, axis: int | None) -> str: + """`name` as the subject of an error message, prefixed by the dimension `axis`.""" + return name[0].upper() + name[1:] if axis is None else f"Dimension {axis}: {name}" + + def _parse_positive_int(value: object, name: str, axis: int | None) -> int: - subject = name[0].upper() + name[1:] if axis is None else f"Dimension {axis}: {name}" + subject = _subject(name, axis) if isinstance(value, bool) or not isinstance(value, int): raise TypeError(f"{subject} must be an int, got {value!r}") if value < 1: @@ -295,10 +300,12 @@ def parse_chunk_edge(size: object, axis: int | None = None) -> int: def parse_chunk_shape(data: object) -> tuple[int, ...]: - """Check a regular chunk shape: one chunk edge length per axis (see `parse_chunk_edge`).""" - if not isinstance(data, Iterable): - raise TypeError(f"A chunk shape must be a sequence of chunk edge lengths, got {data!r}") - return tuple(parse_chunk_edge(size, axis) for axis, size in enumerate(data)) + """Check a regular chunk shape: a list or tuple of one chunk edge length per axis + (see `parse_chunk_edge`).""" + match data: + case list() | tuple(): + return tuple(parse_chunk_edge(size, axis) for axis, size in enumerate(data)) + raise TypeError(f"A chunk shape must be a list or tuple of chunk edge lengths, got {data!r}") def expand_rle(data: Sequence[object], axis: int | None = None) -> list[int]: @@ -313,7 +320,8 @@ def expand_rle(data: Sequence[object], axis: int | None = None) -> list[int]: for item in data: if isinstance(item, list): if len(item) != 2: - raise ValueError(f"RLE entries must be an integer or [size, count], got {item}") + subject = _subject("RLE entries", axis) + raise ValueError(f"{subject} must be an integer or [size, count], got {item}") size, count = item repeat = _parse_positive_int(count, "RLE repeat count", axis) result.extend([parse_chunk_edge(size, axis)] * repeat) diff --git a/src/zarr/core/metadata/io.py b/src/zarr/core/metadata/io.py index 7b63f5493b..7b5e0ced0d 100644 --- a/src/zarr/core/metadata/io.py +++ b/src/zarr/core/metadata/io.py @@ -1,19 +1,108 @@ from __future__ import annotations import asyncio -from typing import TYPE_CHECKING +import json +from enum import Enum +from itertools import zip_longest +from typing import TYPE_CHECKING, Final, NamedTuple from zarr.abc.store import set_or_delete +from zarr.core._json import buffer_to_json_object from zarr.core.buffer.core import default_buffer_prototype +from zarr.core.buffer.cpu import buffer_prototype as cpu_buffer_prototype from zarr.errors import ContainsArrayError from zarr.storage._common import StorePath, ensure_no_existing_node if TYPE_CHECKING: - from zarr.core.common import ZarrFormat + from collections.abc import Iterator, Mapping + + from zarr.core.buffer import Buffer + from zarr.core.common import JSON, ZarrFormat from zarr.core.group import GroupMetadata from zarr.core.metadata import ArrayMetadata +class _Absent(Enum): + ABSENT = "absent" + + +ABSENT: Final = _Absent.ABSENT +"""Where a `DocumentChange` has no value: the document, member or element is not there.""" + + +class DocumentChange(NamedTuple): + """A JSON value that differs between a node's stored metadata documents and the + documents its metadata would store.""" + + path: tuple[str | int, ...] + """Where the value is: the document's key (e.g. `.zarray`), then the object members + and array indices within it.""" + stored: JSON | _Absent + new: JSON | _Absent + + +def diff_documents( + stored: Mapping[str, JSON], new: Mapping[str, JSON] +) -> tuple[DocumentChange, ...]: + """What differs between a node's stored metadata documents and the documents its + metadata would store, each keyed by its store key (a document missing from `stored` + is not stored). Empty if they are identical. + + Objects and arrays are compared member by member, so a change is the smallest value + that differs; any other two values are identical only if their JSON encodings are, so + `true` differs from `1` and `1.0` from `1`. + """ + return tuple(_diff((), stored, new)) + + +def _diff( + path: tuple[str | int, ...], stored: JSON | _Absent, new: JSON | _Absent +) -> Iterator[DocumentChange]: + match stored, new: + case dict(), dict(): + for key in dict.fromkeys([*stored, *new]): + yield from _diff((*path, key), stored.get(key, ABSENT), new.get(key, ABSENT)) + case list(), list(): + for index, pair in enumerate(zip_longest(stored, new, fillvalue=ABSENT)): + yield from _diff((*path, index), *pair) + case _ if ABSENT in (stored, new) or json.dumps(stored) != json.dumps(new): + yield DocumentChange(path, stored, new) + + +async def store_documents(store_path: StorePath, documents: Mapping[str, Buffer]) -> None: + """Store metadata documents encoded by `to_buffer_dict` under `store_path`.""" + await asyncio.gather( + *(set_or_delete(store_path / key, value) for key, value in documents.items()) + ) + + +async def upsert_metadata( + store_path: StorePath, metadata: ArrayMetadata | GroupMetadata +) -> tuple[DocumentChange, ...]: + """Store the documents of `metadata` under `store_path` that differ from the stored + ones, and return how they differed (see `diff_documents`): empty if nothing was + stored. + + The documents are encoded before the store is read, so metadata that cannot be + stored fails with the store untouched. + """ + documents = metadata.to_buffer_dict(default_buffer_prototype()) + stored = await asyncio.gather( + *((store_path / key).get(prototype=cpu_buffer_prototype) for key in documents) + ) + changes = diff_documents( + { + key: buffer_to_json_object(buf) + for key, buf in zip(documents, stored, strict=True) + if buf is not None + }, + {key: buffer_to_json_object(buf) for key, buf in documents.items()}, + ) + changed = {change.path[0] for change in changes} + await store_documents(store_path, {k: v for k, v in documents.items() if k in changed}) + return changes + + def _build_parents(store_path: StorePath, zarr_format: ZarrFormat) -> dict[str, GroupMetadata]: from zarr.core.group import GroupMetadata @@ -52,7 +141,7 @@ async def save_metadata( ValueError """ to_save = metadata.to_buffer_dict(default_buffer_prototype()) - set_awaitables = [set_or_delete(store_path / key, value) for key, value in to_save.items()] + set_awaitables = [store_documents(store_path, to_save)] if ensure_parents: # To enable zarr.create(store, path="a/b/c"), we need to create all the intermediate groups. diff --git a/src/zarr/core/metadata/upgrades.py b/src/zarr/core/metadata/upgrades.py index 1cc6916c47..3d7d554f12 100644 --- a/src/zarr/core/metadata/upgrades.py +++ b/src/zarr/core/metadata/upgrades.py @@ -9,8 +9,7 @@ constructor. An invalid document therefore raises its own error, not a warning about how it was read. -To read another kind of invalid document, add an upgrade to `V2_ARRAY_UPGRADES` or -`V3_ARRAY_UPGRADES`. +To read another kind of invalid document, add an upgrade to `ARRAY_UPGRADES`. """ from __future__ import annotations @@ -25,7 +24,7 @@ from zarr.errors import ZarrUserWarning if TYPE_CHECKING: - from zarr.core.common import JSON + from zarr.core.common import JSON, ZarrFormat type ArrayDocument = Mapping[str, JSON] type Upgrade = Callable[[ArrayDocument], tuple[ArrayDocument, str] | None] @@ -39,14 +38,18 @@ ) -def warn_readings(readings: Sequence[str], path: str | None) -> None: - """Warn once with the readings returned by `upgrade_array_document`, naming the - array at `path` when the caller knows it.""" +def mark_upgraded[M](metadata: M, readings: Sequence[str], path: str | None) -> M: + """Record that `metadata` was read from a stored document that needed the upgrades + whose `readings` `upgrade_array_document` returned, if any: warn once, naming the + array at `path` when the caller knows it, and set `_stored_document_upgraded` on + `metadata`, so the array stores the upgrade before it writes chunks under it.""" if readings: subject = "" if path is None else f"Array {path!r}: " # The synchronous API parses metadata on zarr's IO thread, whose stack holds no # user code, so the warning points at the `from_dict` that read the document. warnings.warn(f"{subject}{' '.join(readings)} {RESAVE_HINT}", ZarrUserWarning, stacklevel=2) + object.__setattr__(metadata, "_stored_document_upgraded", True) + return metadata def _is_int_list(value: object) -> TypeGuard[list[int]]: @@ -177,23 +180,23 @@ def _invalid_chunk_sizes_v3(doc: ArrayDocument) -> tuple[ArrayDocument, str] | N return None -V2_ARRAY_UPGRADES: Final[tuple[Upgrade, ...]] = (_invalid_chunk_sizes_v2,) -V3_ARRAY_UPGRADES: Final[tuple[Upgrade, ...]] = ( - _invalid_inner_chunk_sizes_v3, - _invalid_chunk_sizes_v3, -) +ARRAY_UPGRADES: Final[Mapping[ZarrFormat, tuple[Upgrade, ...]]] = { + 2: (_invalid_chunk_sizes_v2,), + 3: (_invalid_inner_chunk_sizes_v3, _invalid_chunk_sizes_v3), +} +"""The upgrades of an array document of each Zarr format, applied in order.""" def upgrade_array_document( - doc: ArrayDocument, upgrades: Sequence[Upgrade] + doc: ArrayDocument, zarr_format: ZarrFormat ) -> tuple[ArrayDocument, list[str]]: - """Apply `upgrades` to a stored array metadata document, in order. + """Apply the upgrades for `zarr_format` to a stored array metadata document. Returns the upgraded document and the readings of the upgrades that changed it, - for `warn_readings` once the document has been validated. + for `mark_upgraded` once the document has been validated. """ readings: list[str] = [] - for upgrade in upgrades: + for upgrade in ARRAY_UPGRADES[zarr_format]: upgraded = upgrade(doc) if upgraded is not None: doc, reading = upgraded diff --git a/src/zarr/core/metadata/v2.py b/src/zarr/core/metadata/v2.py index 2a1bbeee73..58b6837b98 100644 --- a/src/zarr/core/metadata/v2.py +++ b/src/zarr/core/metadata/v2.py @@ -4,7 +4,7 @@ import warnings from collections.abc import Iterable, Sequence from functools import cached_property -from typing import TYPE_CHECKING, Any, Literal, TypedDict, cast +from typing import TYPE_CHECKING, Any, ClassVar, Literal, TypedDict, cast from zarr.abc.metadata import Metadata from zarr.abc.numcodec import Numcodec, _is_numcodec @@ -44,7 +44,7 @@ from zarr.core.config import config, parse_indexing_order from zarr.core.json_parse import parse_field from zarr.core.metadata.common import parse_attributes -from zarr.core.metadata.upgrades import V2_ARRAY_UPGRADES, upgrade_array_document, warn_readings +from zarr.core.metadata.upgrades import mark_upgraded, upgrade_array_document class ArrayV2MetadataDict(TypedDict): @@ -72,9 +72,10 @@ class ArrayV2Metadata(Metadata): compressor: Numcodec | None attributes: dict[str, JSON] = field(default_factory=dict) zarr_format: Literal[2] = field(init=False, default=2) - _stored_document_upgraded: bool = field(default=False, init=False, compare=False, repr=False) - """Whether `from_dict` read this metadata from a stored document it had to upgrade, - so the store holds an invalid document until this metadata is stored.""" + _stored_document_upgraded: ClassVar[bool] = False + """Whether `from_dict` read this metadata from a stored document it had to upgrade + (set on the instance by `mark_upgraded`), so the store may still hold that invalid + document.""" def __init__( self, @@ -154,7 +155,7 @@ def to_buffer_dict(self, prototype: BufferPrototype) -> dict[str, Buffer]: def from_dict(cls, data: dict[str, Any], *, path: str | None = None) -> ArrayV2Metadata: """Read a stored `.zarray` document (with its attributes). An invalid document that `zarr.core.metadata.upgrades` can read warns, naming the array at `path`.""" - upgraded, readings = upgrade_array_document(data, V2_ARRAY_UPGRADES) + upgraded, readings = upgrade_array_document(data, 2) # a new dict, because we are modifying it _data: dict[str, Any] = dict(upgraded) # Check that the zarr_format attribute is correct. @@ -187,7 +188,7 @@ def from_dict(cls, data: dict[str, Any], *, path: str | None = None) -> ArrayV2M # zarr v2 allowed arbitrary keys here. # We don't want the ArrayV2Metadata constructor to fail just because someone put an # extra key in the metadata. - expected = {x.name for x in fields(cls) if x.init} + expected = {x.name for x in fields(cls)} expected |= {"dtype", "chunks"} # check if `filters` is an empty sequence; if so use None instead and raise a warning @@ -207,10 +208,7 @@ def from_dict(cls, data: dict[str, Any], *, path: str | None = None) -> ArrayV2M _data = {k: v for k, v in _data.items() if k in expected} - metadata = cls(**_data) - warn_readings(readings, path) - object.__setattr__(metadata, "_stored_document_upgraded", bool(readings)) - return metadata + return mark_upgraded(cls(**_data), readings, path) def to_dict(self) -> dict[str, JSON]: zarray_dict = super().to_dict() @@ -331,7 +329,7 @@ def parse_compressor(data: object) -> Numcodec | None: raise ValueError(msg) -def parse_chunks(chunks: Iterable[int], shape: tuple[int, ...]) -> tuple[int, ...]: +def parse_chunks(chunks: object, shape: tuple[int, ...]) -> tuple[int, ...]: """Check a chunk shape: one chunk edge length (an `int` >= 1) per array axis.""" chunks_parsed = parse_chunk_shape(chunks) if len(chunks_parsed) != len(shape): diff --git a/src/zarr/core/metadata/v3.py b/src/zarr/core/metadata/v3.py index 4eb8dd352c..7231d9c51c 100644 --- a/src/zarr/core/metadata/v3.py +++ b/src/zarr/core/metadata/v3.py @@ -3,7 +3,7 @@ import json from collections.abc import Iterable, Mapping, Sequence from dataclasses import dataclass, field, replace -from typing import TYPE_CHECKING, Any, Final, Literal, NotRequired, TypeGuard, cast +from typing import TYPE_CHECKING, Any, ClassVar, Final, Literal, NotRequired, TypeGuard, cast from typing_extensions import TypedDict @@ -27,6 +27,7 @@ compress_rle, expand_rle, parse_chunk_edge, + parse_chunk_shape, parse_named_configuration, parse_shapelike, validate_rectilinear_edges, @@ -38,9 +39,8 @@ from zarr.core.json_parse import parse_field from zarr.core.metadata.common import parse_attributes from zarr.core.metadata.upgrades import ( - V3_ARRAY_UPGRADES, + mark_upgraded, upgrade_array_document, - warn_readings, ) from zarr.errors import MetadataValidationError, NodeTypeValidationError from zarr.registry import get_codec_class @@ -218,17 +218,6 @@ class RectilinearChunkGridMetadataConfig(TypedDict): ] -def _parse_chunk_shape(chunk_shape: Iterable[int]) -> tuple[int, ...]: - """Validate and normalize a regular chunk shape. - - Delegates to ``_validate_chunk_shapes`` — a regular chunk shape is just - a sequence of bare ints (one per dimension), each of which must be >= 1. - """ - result = _validate_chunk_shapes(tuple(chunk_shape)) - # Regular grids only have bare ints — cast is safe after validation - return cast(tuple[int, ...], result) - - def _validate_chunk_shapes( chunk_shapes: Sequence[int | Sequence[int]], ) -> tuple[int | tuple[int, ...], ...]: @@ -261,7 +250,7 @@ class RegularChunkGridMetadata(Metadata): chunk_shape: tuple[int, ...] def __post_init__(self) -> None: - chunk_shape_parsed = _parse_chunk_shape(self.chunk_shape) + chunk_shape_parsed = parse_chunk_shape(self.chunk_shape) object.__setattr__(self, "chunk_shape", chunk_shape_parsed) @property @@ -278,7 +267,7 @@ def to_dict(self) -> RegularChunkGridMetadataJSON: # type: ignore[override] def from_dict(cls, data: RegularChunkGridMetadataJSON) -> Self: # type: ignore[override] parse_named_configuration(data, "regular") # validate name configuration = data["configuration"] - return cls(chunk_shape=_parse_chunk_shape(configuration["chunk_shape"])) + return cls(chunk_shape=parse_chunk_shape(configuration["chunk_shape"])) @dataclass(frozen=True, kw_only=True) @@ -479,9 +468,10 @@ class ArrayV3Metadata(Metadata): node_type: Literal["array"] = field(default="array", init=False) storage_transformers: tuple[dict[str, JSON], ...] extra_fields: dict[str, AllowedExtraField] - _stored_document_upgraded: bool = field(default=False, init=False, compare=False, repr=False) - """Whether `from_dict` read this metadata from a stored document it had to upgrade, - so the store holds an invalid document until this metadata is stored.""" + _stored_document_upgraded: ClassVar[bool] = False + """Whether `from_dict` read this metadata from a stored document it had to upgrade + (set on the instance by `mark_upgraded`), so the store may still hold that invalid + document.""" def __init__( self, @@ -625,7 +615,7 @@ def to_buffer_dict(self, prototype: BufferPrototype) -> dict[str, Buffer]: def from_dict(cls, data: dict[str, JSON], *, path: str | None = None) -> Self: """Read a stored `zarr.json` array document. An invalid document that `zarr.core.metadata.upgrades` can read warns, naming the array at `path`.""" - upgraded, readings = upgrade_array_document(data, V3_ARRAY_UPGRADES) + upgraded, readings = upgrade_array_document(data, 3) # a new dict, because we are modifying it _data = dict(upgraded) @@ -681,9 +671,7 @@ def from_dict(cls, data: dict[str, JSON], *, path: str | None = None) -> Self: extra_fields=allowed_extra_fields, storage_transformers=_data_typed.get("storage_transformers", ()), # type: ignore[arg-type] ) - warn_readings(readings, path) - object.__setattr__(metadata, "_stored_document_upgraded", bool(readings)) - return metadata + return mark_upgraded(metadata, readings, path) def to_dict(self) -> dict[str, JSON]: out_dict = super().to_dict() diff --git a/tests/test_metadata/test_io.py b/tests/test_metadata/test_io.py new file mode 100644 index 0000000000..9accc0f79f --- /dev/null +++ b/tests/test_metadata/test_io.py @@ -0,0 +1,199 @@ +"""Tests for comparing stored metadata documents with the documents metadata would store, +and for storing only the documents that differ.""" + +from __future__ import annotations + +import json +from typing import TYPE_CHECKING, Any, Literal + +import pytest + +import zarr +from zarr.core.buffer import cpu +from zarr.core.metadata import ArrayV2Metadata, ArrayV3Metadata +from zarr.core.metadata.io import ABSENT, DocumentChange, diff_documents, upsert_metadata +from zarr.core.sync import sync +from zarr.storage import MemoryStore, StorePath + +if TYPE_CHECKING: + from zarr.abc.store import Store + from zarr.core.buffer import Buffer + from zarr.core.common import JSON + +V3_DOC: dict[str, JSON] = { + "zarr_format": 3, + "node_type": "array", + "shape": [3], + "chunk_grid": {"name": "regular", "configuration": {"chunk_shape": [3]}}, + "fill_value": 0, +} + + +def _with(doc: dict[str, JSON], **changes: JSON) -> dict[str, JSON]: + return {**doc, **changes} + + +@pytest.mark.parametrize( + ("stored", "new", "expected"), + [ + ({"zarr.json": V3_DOC}, {"zarr.json": V3_DOC}, ()), + ( + {"zarr.json": V3_DOC}, + {"zarr.json": _with(V3_DOC, fill_value=1)}, + ((("zarr.json", "fill_value"), 0, 1),), + ), + ( + { + "zarr.json": _with( + V3_DOC, chunk_grid={"name": "regular", "configuration": {"chunk_shape": [0]}} + ) + }, + {"zarr.json": V3_DOC}, + ((("zarr.json", "chunk_grid", "configuration", "chunk_shape", 0), 0, 3),), + ), + ( + {"zarr.json": V3_DOC}, + {"zarr.json": _with(V3_DOC, shape=[3, 4])}, + ((("zarr.json", "shape", 1), ABSENT, 4),), + ), + ( + {"zarr.json": _with(V3_DOC, attributes={"a": 1})}, + {"zarr.json": _with(V3_DOC, dimension_names=["x"])}, + ( + (("zarr.json", "attributes"), {"a": 1}, ABSENT), + (("zarr.json", "dimension_names"), ABSENT, ["x"]), + ), + ), + ( + {"zarr.json": _with(V3_DOC, fill_value=True)}, + {"zarr.json": _with(V3_DOC, fill_value=1)}, + ((("zarr.json", "fill_value"), True, 1),), + ), + ( + {"zarr.json": _with(V3_DOC, fill_value=1.0)}, + {"zarr.json": _with(V3_DOC, fill_value=1)}, + ((("zarr.json", "fill_value"), 1.0, 1),), + ), + ( + {"zarr.json": _with(V3_DOC, fill_value=float("nan"))}, + {"zarr.json": _with(V3_DOC, fill_value=float("nan"))}, + (), + ), + ( + {"zarr.json": _with(V3_DOC, shape={"0": 3})}, + {"zarr.json": V3_DOC}, + ((("zarr.json", "shape"), {"0": 3}, [3]),), + ), + ( + {".zarray": {"shape": [3], "chunks": [0]}}, + {".zarray": {"shape": [3], "chunks": [3]}, ".zattrs": {}}, + (((".zarray", "chunks", 0), 0, 3), ((".zattrs",), ABSENT, {})), + ), + ], + ids=[ + "identical", + "changed-scalar", + "changed-nested-list", + "longer-list", + "removed-and-added-key", + "bool-is-not-int", + "float-is-not-int", + "nan-is-nan", + "object-is-not-array", + "v2-two-documents", + ], +) +def test_diff_documents( + stored: dict[str, JSON], new: dict[str, JSON], expected: tuple[Any, ...] +) -> None: + """Documents are compared value by value, each change named by its JSON path under + the document's key; values other than objects and arrays are identical only if + their JSON encodings are. `diff_documents` is total over JSON: it raises no error.""" + assert diff_documents(stored, new) == tuple(DocumentChange(*change) for change in expected) + + +class _CountingStore(MemoryStore): + """A memory store that counts the values set in it.""" + + sets = 0 + + async def set(self, key: str, value: Buffer, byte_range: tuple[int, int] | None = None) -> None: + self.sets += 1 + await super().set(key, value, byte_range) + + +def _documents(store: Store) -> dict[str, Any]: + assert isinstance(store, MemoryStore) + return {key: json.loads(value.to_bytes()) for key, value in store._store_dict.items()} + + +def _legacy(zarr_format: Literal[2, 3]) -> tuple[StorePath, ArrayV2Metadata | ArrayV3Metadata]: + """An array stored with chunk shape `[0]`, and the metadata its upgrade reads.""" + store = _CountingStore() + array = zarr.create_array( + store, shape=(3,), chunks=(3,), dtype="int16", zarr_format=zarr_format + ) + key = ".zarray" if zarr_format == 2 else "zarr.json" + doc = _documents(store)[key] + if zarr_format == 2: + doc["chunks"] = [0] + else: + doc["chunk_grid"]["configuration"]["chunk_shape"] = [0] + sync(store.set(key, cpu.Buffer.from_bytes(json.dumps(doc).encode()))) + return StorePath(store), array.metadata + + +@pytest.mark.parametrize("zarr_format", [2, 3]) +def test_upsert_metadata_stores_documents_that_differ(zarr_format: Literal[2, 3]) -> None: + """The documents that differ from the stored ones are stored, and the changes are + returned.""" + store_path, metadata = _legacy(zarr_format) + assert isinstance(store_path.store, _CountingStore) + store_path.store.sets = 0 + key = ".zarray" if zarr_format == 2 else "zarr.json" + chunk_path = ( + ("chunks", 0) if zarr_format == 2 else ("chunk_grid", "configuration", "chunk_shape", 0) + ) + + changes = sync(upsert_metadata(store_path, metadata)) + + assert changes == (DocumentChange((key, *chunk_path), 0, 3),) + assert store_path.store.sets == 1 + stored = _documents(store_path.store) + assert stored[key] == json.loads(metadata.to_buffer_dict(cpu.buffer_prototype)[key].to_bytes()) + + +@pytest.mark.parametrize("zarr_format", [2, 3]) +def test_upsert_metadata_identical_stores_nothing(zarr_format: Literal[2, 3]) -> None: + """Metadata identical to what is stored stores nothing.""" + store = _CountingStore() + array = zarr.create_array( + store, shape=(3,), chunks=(3,), dtype="int16", zarr_format=zarr_format + ) + store.sets = 0 + + assert sync(upsert_metadata(StorePath(store), array.metadata)) == () + assert store.sets == 0 + + +def test_upsert_metadata_unstorable_leaves_store_untouched(monkeypatch: pytest.MonkeyPatch) -> None: + """Metadata that cannot be encoded fails before the store is read or written.""" + store_path, metadata = _legacy(3) + before = _documents(store_path.store) + + def refuse(*args: object) -> None: + raise ValueError("cannot be stored") + + monkeypatch.setattr(ArrayV3Metadata, "to_buffer_dict", refuse) + with pytest.raises(ValueError, match="cannot be stored"): + sync(upsert_metadata(store_path, metadata)) + assert _documents(store_path.store) == before + + +def test_upsert_metadata_stored_document_not_an_object() -> None: + """A stored document that is not a JSON object is not overwritten.""" + store_path, metadata = _legacy(3) + sync(store_path.store.set("zarr.json", cpu.Buffer.from_bytes(b"[]"))) + with pytest.raises(TypeError, match="Expected a JSON object, got list"): + sync(upsert_metadata(store_path, metadata)) + assert _documents(store_path.store) == {"zarr.json": []} diff --git a/tests/test_metadata/test_upgrades.py b/tests/test_metadata/test_upgrades.py index b976660eb2..6e1126fadc 100644 --- a/tests/test_metadata/test_upgrades.py +++ b/tests/test_metadata/test_upgrades.py @@ -6,30 +6,31 @@ import json import re import warnings -from typing import TYPE_CHECKING, Any, Literal +from typing import TYPE_CHECKING, Any, Literal, cast import numpy as np import pytest import zarr from zarr.codecs import ShardingCodec +from zarr.core.array import AsyncArray from zarr.core.metadata import ArrayV2Metadata, ArrayV3Metadata from zarr.core.metadata.upgrades import ( RESAVE_HINT, - V2_ARRAY_UPGRADES, - V3_ARRAY_UPGRADES, upgrade_array_document, ) from zarr.core.metadata.v3 import RectilinearChunkGridMetadata, RegularChunkGridMetadata from zarr.core.sync import sync from zarr.dtype import Int16 from zarr.errors import ZarrUserWarning +from zarr.storage._common import make_store_path if TYPE_CHECKING: from collections.abc import Callable from pathlib import Path - from zarr.core.common import JSON + from zarr.core.common import JSON, ZarrFormat + from zarr.types import AnyArray def _v2_doc(shape: list[int], chunks: list[Any]) -> dict[str, JSON]: @@ -168,8 +169,7 @@ def test_upgrade_array_document( `from_dict` warns once for the document, naming the array, saying how each part was read (and that the array holds only its fill value where a chunk size of 0 was stored for a non-empty axis) and how to re-save.""" - upgrades = V2_ARRAY_UPGRADES if doc["zarr_format"] == 2 else V3_ARRAY_UPGRADES - upgraded, readings = upgrade_array_document(doc, upgrades) + upgraded, readings = upgrade_array_document(doc, cast("ZarrFormat", doc["zarr_format"])) assert {k: v for k, v in upgraded.items() if k not in ("chunks", "chunk_grid", "codecs")} == { k: v for k, v in doc.items() if k not in ("chunks", "chunk_grid", "codecs") } @@ -204,7 +204,7 @@ def test_invalid_upgraded_document_raises_without_warning(doc: dict[str, JSON], """A document the upgrades read that the metadata constructor then rejects raises that error, without first warning how it was read.""" metadata_cls = ArrayV2Metadata if doc["zarr_format"] == 2 else ArrayV3Metadata - assert upgrade_array_document(doc, V2_ARRAY_UPGRADES + V3_ARRAY_UPGRADES)[1] + assert upgrade_array_document(doc, cast("ZarrFormat", doc["zarr_format"]))[1] with warnings.catch_warnings(): warnings.simplefilter("error", ZarrUserWarning) with pytest.raises(ValueError, match=error): @@ -306,10 +306,25 @@ def test_metadata_rejects_chunk_edge_below_one(site: str, size: int) -> None: CHUNK_EDGE_SITES[site](size) -@pytest.mark.parametrize("chunks", [4, np.int64(4)]) -def test_v2_constructor_rejects_scalar_chunks(chunks: object) -> None: - with pytest.raises(TypeError, match="A chunk shape must be a sequence of chunk edge lengths"): - _v2_metadata(chunks) +CHUNK_SHAPE_SITES: dict[str, Callable[[Any], object]] = { + "regular": lambda chunk_shape: RegularChunkGridMetadata(chunk_shape=chunk_shape), + "v2": _v2_metadata, + "sharding-inner": lambda chunk_shape: ShardingCodec(chunk_shape=chunk_shape), +} + + +@pytest.mark.parametrize("site", CHUNK_SHAPE_SITES) +@pytest.mark.parametrize("chunk_shape", [4, np.int64(4), "10", {"a": 1}, range(1, 2)]) +def test_metadata_rejects_chunk_shape_not_list_or_tuple(site: str, chunk_shape: object) -> None: + """A regular chunk shape is a list or tuple; anything else is rejected as a whole, + not iterated as if its elements were chunk edge lengths.""" + with pytest.raises( + TypeError, + match=re.escape( + f"A chunk shape must be a list or tuple of chunk edge lengths, got {chunk_shape!r}" + ), + ): + CHUNK_SHAPE_SITES[site](chunk_shape) def _rewrite_doc(path: Path, zarr_format: Literal[2, 3], edit: Any) -> None: @@ -424,6 +439,7 @@ async def write_twice() -> None: sync(write_twice()) + assert not arr.metadata._stored_document_upgraded with warnings.catch_warnings(): warnings.simplefilter("error", ZarrUserWarning) reopened = zarr.open_array(store=path, mode="r") @@ -431,6 +447,99 @@ async def write_twice() -> None: np.testing.assert_array_equal(reopened[...], data) +def _legacy_array(path: Path, zarr_format: Literal[2, 3]) -> None: + """Store an array of shape (3,) whose stored chunk shape is `[0]`.""" + zarr.create_array(store=path, shape=(3,), chunks=(3,), dtype="int16", zarr_format=zarr_format) + _rewrite_doc( + path, + zarr_format, + lambda doc: ( + doc.update(chunks=[0]) + if zarr_format == 2 + else doc["chunk_grid"]["configuration"].update(chunk_shape=[0]) + ), + ) + + +def _open_strictly(path: Path) -> AnyArray: + """Open the array at `path`, failing on any warning that its document was upgraded.""" + with warnings.catch_warnings(): + warnings.simplefilter("error", ZarrUserWarning) + array = zarr.open_array(store=path, mode="r") + assert isinstance(array, zarr.Array) + return array + + +@pytest.mark.parametrize("zarr_format", [2, 3]) +def test_stale_handle_write_keeps_newer_metadata( + tmp_path: Path, zarr_format: Literal[2, 3] +) -> None: + """A handle read from an upgraded document stores the upgrade of what the store + holds when it first writes chunks; if another handle stored valid metadata since, + it stores no metadata and writes only its chunks.""" + path = tmp_path / "legacy.zarr" + _legacy_array(path, zarr_format) + with pytest.warns(ZarrUserWarning, match="is read as"): + stale = zarr.open_array(store=path, mode="r+") + with pytest.warns(ZarrUserWarning, match="is read as"): + other = zarr.open_array(store=path, mode="r+") + other.append(np.arange(1, 7, dtype="int16")) + other.attrs["x"] = 1 + + stale[0] = 9 + + assert not stale.metadata._stored_document_upgraded + reopened = _open_strictly(path) + assert reopened.shape == (9,) + assert reopened.attrs.asdict() == {"x": 1} + np.testing.assert_array_equal(reopened[...], [9, 0, 0, 1, 2, 3, 4, 5, 6]) + + +@pytest.mark.parametrize("zarr_format", [2, 3]) +def test_empty_write_stores_no_metadata(tmp_path: Path, zarr_format: Literal[2, 3]) -> None: + """A write of an empty selection stores no chunks, so it stores no metadata either.""" + path = tmp_path / "legacy.zarr" + _legacy_array(path, zarr_format) + documents = {p.name: p.read_bytes() for p in path.iterdir()} + with pytest.warns(ZarrUserWarning, match="is read as"): + array = zarr.open_array(store=path, mode="r+") + + array[0:0] = np.empty(0, dtype="int16") + + assert array.metadata._stored_document_upgraded + assert {p.name: p.read_bytes() for p in path.iterdir()} == documents + + +@pytest.mark.parametrize("zarr_format", [2, 3]) +def test_async_array_from_dict_names_array(tmp_path: Path, zarr_format: Literal[2, 3]) -> None: + """`AsyncArray.from_dict` names the array at its store path in the upgrade warning.""" + store_path = sync(make_store_path(tmp_path / "legacy.zarr")) + doc = _v2_doc([4], [0]) if zarr_format == 2 else _v3_doc([4], [0]) + with pytest.warns(ZarrUserWarning, match=f"^Array {re.escape(repr(str(store_path)))}: "): + AsyncArray.from_dict(store_path, doc) + + +@pytest.mark.filterwarnings("ignore:Consolidated metadata is currently not part:UserWarning") +@pytest.mark.parametrize("zarr_format", [2, 3]) +def test_consolidate_stores_upgraded_members(tmp_path: Path, zarr_format: Literal[2, 3]) -> None: + """Consolidating a group stores the upgrade of every member document that needed + one before the consolidated document, so chunks written through the consolidated + metadata are stored under a member document that agrees with it.""" + path = tmp_path / "group.zarr" + zarr.open_group(path, mode="w", zarr_format=zarr_format) + _legacy_array(path / "a", zarr_format) + with pytest.warns(ZarrUserWarning, match="is read as"): + zarr.consolidate_metadata(path) + + member = _open_strictly(path / "a") + assert member.chunks == (3,) + group = zarr.open_group(path, mode="r+", use_consolidated=True) + array = group["a"] + assert isinstance(array, zarr.Array) + array[:] = [7, 8, 9] + np.testing.assert_array_equal(_open_strictly(path / "a")[...], [7, 8, 9]) + + @pytest.mark.filterwarnings("ignore:Consolidated metadata is currently not part:UserWarning") @pytest.mark.parametrize("zarr_format", [2, 3]) def test_legacy_chunk_size_consolidated(tmp_path: Path, zarr_format: Literal[2, 3]) -> None: diff --git a/tests/test_unified_chunk_grid.py b/tests/test_unified_chunk_grid.py index 4a8d6f791e..8f06863a05 100644 --- a/tests/test_unified_chunk_grid.py +++ b/tests/test_unified_chunk_grid.py @@ -562,6 +562,22 @@ def test_rle_expand_rejects_non_int(rle_input: list[Any], match: str) -> None: expand_rle(rle_input) +@pytest.mark.parametrize( + ("rle_input", "match"), + [ + ([0], "chunk edge length must be >= 1"), + ([10.0], "chunk edge length must be an int"), + ([[5, 0]], "RLE repeat count must be >= 1"), + ([[5, 2, 1]], r"RLE entries must be an integer or \[size, count\]"), + ], + ids=["zero-edge", "float-edge", "zero-rle-count", "rle-entry-of-three"], +) +def test_rle_expand_names_dimension(rle_input: list[Any], match: str) -> None: + """Given the dimension `axis` the edges belong to, every error of expand_rle names it.""" + with pytest.raises((TypeError, ValueError), match=f"^Dimension 2: {match}"): + expand_rle(rle_input, axis=2) + + # --------------------------------------------------------------------------- # _is_rectilinear_chunks tests # --------------------------------------------------------------------------- From 9e2933352d50c22220d9105df3d0f8e74f8d9657 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Sat, 26 Sep 2026 10:12:58 +0200 Subject: [PATCH 41/76] test: prefer zero-length axes for stored chunk size 0 in the lifecycle machine An empty write stores no metadata, so it no longer ends the legacy state. Assisted-by: ClaudeCode:claude-opus-5-5 Co-Authored-By: Claude Opus 5.5 --- tests/test_array_stateful.py | 17 ++++++++++------- 1 file changed, 10 insertions(+), 7 deletions(-) diff --git a/tests/test_array_stateful.py b/tests/test_array_stateful.py index 7ce4780e5d..6dc03768b3 100644 --- a/tests/test_array_stateful.py +++ b/tests/test_array_stateful.py @@ -134,11 +134,12 @@ def create(self, data: st.DataObject) -> None: if spelling != "rectilinear" and data.draw(st.booleans(), label="legacy zero"): # A stored chunk size of 0, as zarr-python wrote for arrays created with a # zero-length axis; older releases could then grow the axis without storing - # a chunk, so any extent is possible. Sharded arrays store it in the outer grid. - self.legacy_axes = data.draw( - st.lists(st.integers(0, len(shape) - 1), min_size=1, unique=True), - label="axes stored with chunk size 0", - ) + # a chunk, so any extent is possible, but zero-length axes come first. + # Sharded arrays store it in the outer grid. + axes = st.lists(st.integers(0, len(shape) - 1), min_size=1, unique=True) + if zero_axes := [axis for axis, extent in enumerate(shape) if extent == 0]: + axes = st.lists(st.sampled_from(zero_axes), min_size=1, unique=True) | axes + self.legacy_axes = data.draw(axes, label="axes stored with chunk size 0") stored_zero = data.draw(st.sampled_from([0, False]), label="stored zero") self._rewrite_stored_chunks(zarr_format, self.legacy_axes, stored_zero) event("legacy zero chunk size") @@ -269,8 +270,10 @@ def write(self, data: st.DataObject) -> None: note(f"write {region}") arr[region] = values self._model_write(arr, region, values) - # Writing chunks first stores the metadata they are written under. - self.legacy_axes = [] + if all(shape): + # Writing chunks first stores the metadata they are written under; a write + # of nothing stores nothing. + self.legacy_axes = [] @precondition(lambda self: self.legacy_axes) @rule() From 140a706a173be2196c9ac28e19fdb41501a1ebc5 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Sat, 26 Sep 2026 10:45:19 +0200 Subject: [PATCH 42/76] refactor: the caller that knows the node names it in gate errors `check_storable` and `_check_rectilinear_chunks_enabled` take no subject. `encode_documents(store_path, metadata)` in `metadata/io.py` encodes the documents an operation stores and, on failure, adds a note naming the node at `store_path` in the full-path form the upgrade warnings use; every encode-before-write site goes through it (`save_metadata`, `upsert_metadata`, `_resize`, `Group.delitem`, group attribute updates, `_encode_nodes` for `create_hierarchy`/`create_nodes`). `ArrayV3Metadata.from_dict` names the array at `path` when the flag refuses a rectilinear document, which is where `consolidate_metadata` and a first write to an upgraded 3.2.x mixed grid are refused without the flag. The upsert gate test now uses the real gate instead of a monkeypatched `to_buffer_dict`, and consolidating a 3.2.x mixed grid is checked both ways: refused with the store byte-identical without the flag, stored as rectilinear (member and consolidated document) with it. Assisted-by: ClaudeCode:claude-opus-5-5 Co-Authored-By: Claude Opus 5.5 --- src/zarr/core/array.py | 12 ++++++-- src/zarr/core/group.py | 28 +++++++++--------- src/zarr/core/metadata/io.py | 27 ++++++++++++++--- src/zarr/core/metadata/v3.py | 23 +++++++++------ tests/test_metadata/test_io.py | 24 +++++++++------- tests/test_metadata/test_upgrades.py | 43 ++++++++++++++++++++-------- 6 files changed, 106 insertions(+), 51 deletions(-) diff --git a/src/zarr/core/array.py b/src/zarr/core/array.py index e9e50bd136..820e782cac 100644 --- a/src/zarr/core/array.py +++ b/src/zarr/core/array.py @@ -118,7 +118,12 @@ ArrayV2MetadataDict, ArrayV3Metadata, ) -from zarr.core.metadata.io import save_metadata, store_documents, upsert_metadata +from zarr.core.metadata.io import ( + encode_documents, + save_metadata, + store_documents, + upsert_metadata, +) from zarr.core.metadata.upgrades import upgrade_array_document from zarr.core.metadata.v2 import ( CompressorLikev2, @@ -1631,7 +1636,8 @@ async def _store_upgraded_document(self) -> None: zarr_format = self.metadata.zarr_format stored = await get_array_metadata(self.store_path, zarr_format=zarr_format) upgraded, _ = upgrade_array_document(stored, zarr_format) - await upsert_metadata(self.store_path, parse_array_metadata(upgraded)) + metadata = parse_array_metadata(upgraded, str(self.store_path)) + await upsert_metadata(self.store_path, metadata) object.__setattr__(self.metadata, "_stored_document_upgraded", False) async def _set_selection( @@ -5877,7 +5883,7 @@ async def _resize( # Encode the new metadata before deleting any chunk: metadata that cannot be stored # then fails with the store untouched. - documents = new_metadata.to_buffer_dict(default_buffer_prototype()) + documents = encode_documents(array.store_path, new_metadata) if delete_outside_chunks and not only_growing: # Remove all chunks outside of the new shape diff --git a/src/zarr/core/group.py b/src/zarr/core/group.py index 08352063ed..9967d5e467 100644 --- a/src/zarr/core/group.py +++ b/src/zarr/core/group.py @@ -13,7 +13,6 @@ import zarr.api.asynchronous as async_api from zarr.abc.metadata import Metadata -from zarr.abc.store import Store, set_or_delete from zarr.core._info import GroupInfo from zarr.core._json import buffer_to_json_object, json_to_buffer from zarr.core.array import ( @@ -48,7 +47,7 @@ from zarr.core.dtype import parse_data_type from zarr.core.json_parse import parse_field from zarr.core.metadata import ArrayV2Metadata, ArrayV3Metadata -from zarr.core.metadata.io import save_metadata, store_documents +from zarr.core.metadata.io import encode_documents, save_metadata, store_documents from zarr.core.metadata.v3 import check_storable from zarr.core.sync import SyncMixin, sync from zarr.errors import ( @@ -76,6 +75,7 @@ ) from typing import Any + from zarr.abc.store import Store from zarr.core.array_spec import ArrayConfigLike from zarr.core.buffer import Buffer, BufferPrototype from zarr.core.chunk_key_encodings import ChunkKeyEncodingLike @@ -364,7 +364,11 @@ def to_buffer_dict(self, prototype: BufferPrototype) -> dict[str, Buffer]: if self.consolidated_metadata is not None: for path, member in self.consolidated_metadata.flattened_metadata.items(): if isinstance(member, ArrayV3Metadata): - check_storable(member, f"Array {path!r} in the consolidated metadata: ") + try: + check_storable(member) + except ValueError as e: + e.add_note(f"Array {path!r} in the consolidated metadata.") + raise indent = config.get("json_indent") if self.zarr_format == 3: return {ZARR_JSON: json_to_buffer(self.to_dict(), prototype=prototype, indent=indent)} @@ -821,7 +825,7 @@ async def delitem(self, key: str) -> None: ) # Encode the group metadata before deleting the member: metadata that cannot be # stored then fails with the store untouched. - documents = metadata.to_buffer_dict(default_buffer_prototype()) + documents = encode_documents(self.store_path, metadata) await store_path.delete_dir() await store_documents(self.store_path, documents) object.__setattr__(self, "metadata", metadata) @@ -2128,10 +2132,7 @@ async def update_attributes_async(self, new_attributes: dict[str, Any]) -> Group """ new_metadata = replace(self.metadata, attributes=new_attributes) - # Write new metadata - to_save = new_metadata.to_buffer_dict(default_buffer_prototype()) - awaitables = [set_or_delete(self.store_path / key, value) for key, value in to_save.items()] - await asyncio.gather(*awaitables) + await store_documents(self.store_path, encode_documents(self.store_path, new_metadata)) async_group = replace(self._async_group, metadata=new_metadata) return replace(self, _async_group=async_group) @@ -3194,7 +3195,7 @@ async def create_hierarchy( # Encode every node before deleting anything: a node whose metadata cannot be stored # then fails with the store untouched. - documents = _encode_nodes(nodes_explicit) + documents = _encode_nodes(store, nodes_explicit) await asyncio.gather(*(store.delete_dir(key) for key in to_delete_keys)) async for key, node in _store_nodes(store, nodes_explicit, documents): yield key, node @@ -3226,19 +3227,18 @@ async def create_nodes( AsyncGroup | AsyncArray The created nodes in the order they are created. """ - async for key, node in _store_nodes(store, nodes, _encode_nodes(nodes)): + async for key, node in _store_nodes(store, nodes, _encode_nodes(store, nodes)): yield key, node def _encode_nodes( - nodes: Mapping[str, GroupMetadata | ArrayV2Metadata | ArrayV3Metadata], + store: Store, nodes: Mapping[str, GroupMetadata | ArrayV2Metadata | ArrayV3Metadata] ) -> dict[str, Buffer]: - """The metadata documents of `nodes`, by their keys in the store.""" - prototype = default_buffer_prototype() + """The metadata documents of `nodes` in `store`, by their keys in the store.""" return { _join_paths([path, key]): value for path, metadata in nodes.items() - for key, value in metadata.to_buffer_dict(prototype).items() + for key, value in encode_documents(StorePath(store, path), metadata).items() } diff --git a/src/zarr/core/metadata/io.py b/src/zarr/core/metadata/io.py index 7b5e0ced0d..3290ea8d13 100644 --- a/src/zarr/core/metadata/io.py +++ b/src/zarr/core/metadata/io.py @@ -69,8 +69,28 @@ def _diff( yield DocumentChange(path, stored, new) +def encode_documents( + store_path: StorePath, metadata: ArrayMetadata | GroupMetadata +) -> dict[str, Buffer]: + """The metadata documents `metadata` stores under `store_path`, by key (see + `to_buffer_dict`). + + An operation that deletes or writes store content encodes its documents first, so + metadata that cannot be stored fails with the store untouched; the error then names + the node at `store_path`, as the warnings about stored documents do. + """ + from zarr.core.group import GroupMetadata + + try: + return metadata.to_buffer_dict(default_buffer_prototype()) + except ValueError as e: + node = "Group" if isinstance(metadata, GroupMetadata) else "Array" + e.add_note(f"{node} {str(store_path)!r}: nothing was stored.") + raise + + async def store_documents(store_path: StorePath, documents: Mapping[str, Buffer]) -> None: - """Store metadata documents encoded by `to_buffer_dict` under `store_path`.""" + """Store metadata documents encoded by `encode_documents` under `store_path`.""" await asyncio.gather( *(set_or_delete(store_path / key, value) for key, value in documents.items()) ) @@ -86,7 +106,7 @@ async def upsert_metadata( The documents are encoded before the store is read, so metadata that cannot be stored fails with the store untouched. """ - documents = metadata.to_buffer_dict(default_buffer_prototype()) + documents = encode_documents(store_path, metadata) stored = await asyncio.gather( *((store_path / key).get(prototype=cpu_buffer_prototype) for key in documents) ) @@ -140,8 +160,7 @@ async def save_metadata( ------ ValueError """ - to_save = metadata.to_buffer_dict(default_buffer_prototype()) - set_awaitables = [store_documents(store_path, to_save)] + set_awaitables = [store_documents(store_path, encode_documents(store_path, metadata))] if ensure_parents: # To enable zarr.create(store, path="a/b/c"), we need to create all the intermediate groups. diff --git a/src/zarr/core/metadata/v3.py b/src/zarr/core/metadata/v3.py index fdb671b91c..504517662b 100644 --- a/src/zarr/core/metadata/v3.py +++ b/src/zarr/core/metadata/v3.py @@ -356,30 +356,31 @@ def from_dict(cls, data: RectilinearChunkGridMetadataJSON) -> Self: # type: ign ChunkGridMetadata = RegularChunkGridMetadata | RectilinearChunkGridMetadata -def _check_rectilinear_chunks_enabled(subject: str = "") -> None: - """Raise unless rectilinear chunks are enabled, with `subject` leading the error. +def _check_rectilinear_chunks_enabled() -> None: + """Raise unless rectilinear chunks are enabled. The flag gates storing and reading array metadata documents that declare a rectilinear chunk grid; the chunk grid metadata classes themselves are not gated. """ if not config.get("array.rectilinear_chunks"): raise ValueError( - f"{subject}Rectilinear chunk grids are experimental and disabled by default. " + "Rectilinear chunk grids are experimental and disabled by default. " "Enable them with: zarr.config.set({'array.rectilinear_chunks': True}) " "or set the environment variable ZARR_ARRAY__RECTILINEAR_CHUNKS=True" ) -def check_storable(metadata: ArrayV3Metadata, subject: str = "") -> None: +def check_storable(metadata: ArrayV3Metadata) -> None: """Raise if `metadata` may not be stored: a rectilinear chunk grid requires the - rectilinear chunks flag. `subject`, when given, names the array in the error. + rectilinear chunks flag. `zarr.core.metadata.io.encode_documents` names the node in + the error. Every serialization of array metadata for a store calls this, before the store is touched: `ArrayV3Metadata.to_buffer_dict` and, for the arrays in a group's consolidated metadata, `GroupMetadata.to_buffer_dict`. """ if isinstance(metadata.chunk_grid, RectilinearChunkGridMetadata): - _check_rectilinear_chunks_enabled(subject) + _check_rectilinear_chunks_enabled() def create_chunk_grid_metadata( @@ -635,11 +636,17 @@ def to_buffer_dict(self, prototype: BufferPrototype) -> dict[str, Buffer]: @classmethod def from_dict(cls, data: dict[str, JSON], *, path: str | None = None) -> Self: """Read a stored `zarr.json` array document. An invalid document that - `zarr.core.metadata.upgrades` can read warns, naming the array at `path`.""" + `zarr.core.metadata.upgrades` can read warns, and a document the rectilinear + chunks flag refuses raises, naming the array at `path`.""" # The flag gates what the document declares, so it is checked before upgrades. chunk_grid = data.get("chunk_grid") if isinstance(chunk_grid, Mapping) and chunk_grid.get("name") == "rectilinear": - _check_rectilinear_chunks_enabled() + try: + _check_rectilinear_chunks_enabled() + except ValueError as e: + if path is not None: + e.add_note(f"Array {path!r}.") + raise upgraded, readings = upgrade_array_document(data, 3) # a new dict, because we are modifying it _data = dict(upgraded) diff --git a/tests/test_metadata/test_io.py b/tests/test_metadata/test_io.py index 9accc0f79f..51d67111b3 100644 --- a/tests/test_metadata/test_io.py +++ b/tests/test_metadata/test_io.py @@ -10,7 +10,6 @@ import zarr from zarr.core.buffer import cpu -from zarr.core.metadata import ArrayV2Metadata, ArrayV3Metadata from zarr.core.metadata.io import ABSENT, DocumentChange, diff_documents, upsert_metadata from zarr.core.sync import sync from zarr.storage import MemoryStore, StorePath @@ -19,6 +18,7 @@ from zarr.abc.store import Store from zarr.core.buffer import Buffer from zarr.core.common import JSON + from zarr.core.metadata import ArrayV2Metadata, ArrayV3Metadata V3_DOC: dict[str, JSON] = { "zarr_format": 3, @@ -176,17 +176,21 @@ def test_upsert_metadata_identical_stores_nothing(zarr_format: Literal[2, 3]) -> assert store.sets == 0 -def test_upsert_metadata_unstorable_leaves_store_untouched(monkeypatch: pytest.MonkeyPatch) -> None: - """Metadata that cannot be encoded fails before the store is read or written.""" - store_path, metadata = _legacy(3) +def test_upsert_metadata_unstorable_leaves_store_untouched() -> None: + """Metadata that may not be stored (a rectilinear chunk grid without the rectilinear + chunks flag) fails before the store is read or written, naming the array.""" + store_path, _ = _legacy(3) before = _documents(store_path.store) - - def refuse(*args: object) -> None: - raise ValueError("cannot be stored") - - monkeypatch.setattr(ArrayV3Metadata, "to_buffer_dict", refuse) - with pytest.raises(ValueError, match="cannot be stored"): + with zarr.config.set({"array.rectilinear_chunks": True}): + metadata = zarr.create_array( + MemoryStore(), shape=(3,), chunks=[[1, 2]], dtype="int16" + ).metadata + with ( + zarr.config.set({"array.rectilinear_chunks": False}), + pytest.raises(ValueError, match="experimental and disabled") as info, + ): sync(upsert_metadata(store_path, metadata)) + assert info.value.__notes__ == [f"Array {str(store_path)!r}: nothing was stored."] assert _documents(store_path.store) == before diff --git a/tests/test_metadata/test_upgrades.py b/tests/test_metadata/test_upgrades.py index 7a1fb12314..e3494082a0 100644 --- a/tests/test_metadata/test_upgrades.py +++ b/tests/test_metadata/test_upgrades.py @@ -808,14 +808,18 @@ def _overwrite_hierarchy(path: Path) -> None: list(zarr.create_hierarchy(store=LocalStore(path), nodes={"n": mixed.metadata}, overwrite=True)) +def _consolidate(path: Path) -> None: + zarr.consolidate_metadata(path) + + @pytest.mark.filterwarnings( "ignore:.*read as that rectilinear chunk grid:zarr.errors.ZarrUserWarning" ) @pytest.mark.filterwarnings("ignore:Consolidated metadata is currently not part:UserWarning") @pytest.mark.parametrize( "action", - [_resize, _write_chunks, _delete_member, _overwrite_hierarchy], - ids=["resize", "write", "delete-member", "overwrite-hierarchy"], + [_resize, _write_chunks, _delete_member, _overwrite_hierarchy, _consolidate], + ids=["resize", "write", "delete-member", "overwrite-hierarchy", "consolidate"], ) def test_store_untouched_without_flag(tmp_path: Path, action: Callable[[Path], None]) -> None: """An operation that would store the rectilinear chunk grid read from the verbatim @@ -833,25 +837,40 @@ def test_store_untouched_without_flag(tmp_path: Path, action: Callable[[Path], N @pytest.mark.filterwarnings("ignore:Consolidated metadata is currently not part:UserWarning") @pytest.mark.parametrize("member", ["mixed", "sub/mixed"]) -def test_consolidate_without_flag_leaves_group_readable(tmp_path: Path, member: str) -> None: - """Consolidating a group holding an array read from the verbatim document would - store its rectilinear chunk grid, so without the flag it fails, naming the array, - with the store untouched, and the group still opens without the flag.""" +def test_consolidate_edge_lists_in_regular_grid(tmp_path: Path, member: str) -> None: + """Consolidating a group holding the verbatim document stores the member's + rectilinear chunk grid first, so without the flag it fails, naming the array, and + the group still opens without the flag; with the flag, the member and the + consolidated metadata store the rectilinear chunk grid.""" path = tmp_path / "group.zarr" zarr.open_group(path, mode="w").create_group("sub") data = _store_mixed_array(path / member) - group_doc = (path / "zarr.json").read_bytes() + array_path = str(sync(make_store_path(path / member))) with zarr.config.set({"array.rectilinear_chunks": False}): with ( pytest.warns(ZarrUserWarning, match="read as that rectilinear chunk grid"), - pytest.raises( - ValueError, - match=f"^Array '{member}' in the consolidated metadata: .* experimental and", - ), + pytest.raises(ValueError, match="experimental and disabled") as info, ): zarr.consolidate_metadata(path) - assert (path / "zarr.json").read_bytes() == group_doc + assert info.value.__notes__ == [f"Array {array_path!r}."] with pytest.warns(ZarrUserWarning, match="read as that rectilinear chunk grid"): mixed = zarr.open_group(path, mode="r")[member] assert isinstance(mixed, zarr.Array) np.testing.assert_array_equal(mixed[...], data) + + rectilinear = { + "name": "rectilinear", + "configuration": {"kind": "inline", "chunk_shapes": [2, [5, 10, 5]]}, + } + with zarr.config.set({"array.rectilinear_chunks": True}): + with pytest.warns(ZarrUserWarning, match="read as that rectilinear chunk grid"): + zarr.consolidate_metadata(path) + group_doc = json.loads((path / "zarr.json").read_text()) + consolidated = group_doc["consolidated_metadata"]["metadata"][member] + assert consolidated["chunk_grid"] == rectilinear + assert json.loads((path / member / "zarr.json").read_text())["chunk_grid"] == rectilinear + with warnings.catch_warnings(): + warnings.simplefilter("error", ZarrUserWarning) + reopened = zarr.open_group(path, mode="r")[member] + assert isinstance(reopened, zarr.Array) + np.testing.assert_array_equal(reopened[...], data) From 299ecb34b4590e6e0416501552c3d71d32e4df17 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Sat, 26 Sep 2026 10:45:55 +0200 Subject: [PATCH 43/76] fix(group): delete a consolidated member in place, after encoding without it `Group.delitem` encodes the group documents from a copy without the member (so the gate runs with the store and the group untouched on failure), deletes the member, then pops it from the shared consolidated metadata in place, as main does, so a parent or subgroup handle sharing that consolidated metadata sees the deletion, then stores the documents. Tests: the consolidated delitem test runs for both Zarr formats and reopens from the store (V2 `.zmetadata` is rewritten); a subgroup handle read from its parent's consolidated metadata sees a deletion made through it; a deletion refused without the flag leaves the group listing the member. Assisted-by: ClaudeCode:claude-opus-5-5 Co-Authored-By: Claude Opus 5.5 --- src/zarr/core/group.py | 14 ++++--- tests/test_group.py | 58 ++++++++++++++++++++-------- tests/test_metadata/test_upgrades.py | 16 ++++++++ 3 files changed, 66 insertions(+), 22 deletions(-) diff --git a/src/zarr/core/group.py b/src/zarr/core/group.py index 9967d5e467..4b6455b3e2 100644 --- a/src/zarr/core/group.py +++ b/src/zarr/core/group.py @@ -819,16 +819,18 @@ async def delitem(self, key: str) -> None: if consolidated is None: await store_path.delete_dir() return + # Encode the group metadata without the member before deleting it: metadata that + # cannot be stored then fails with the store and this group untouched. members = {name: node for name, node in consolidated.metadata.items() if name != key} - metadata = replace( - self.metadata, consolidated_metadata=replace(consolidated, metadata=members) + documents = encode_documents( + self.store_path, + replace(self.metadata, consolidated_metadata=replace(consolidated, metadata=members)), ) - # Encode the group metadata before deleting the member: metadata that cannot be - # stored then fails with the store untouched. - documents = encode_documents(self.store_path, metadata) await store_path.delete_dir() + # In place, so every handle sharing this consolidated metadata (a parent's or a + # subgroup's) sees the deletion. + consolidated.metadata.pop(key, None) await store_documents(self.store_path, documents) - object.__setattr__(self, "metadata", metadata) async def get[DefaultT]( self, key: str, default: DefaultT | None = None diff --git a/tests/test_group.py b/tests/test_group.py index 31fbd138cd..a853890407 100644 --- a/tests/test_group.py +++ b/tests/test_group.py @@ -1620,11 +1620,13 @@ async def test_group_getitem_consolidated(self, store: Store) -> None: rg2 = await rg1.get_group("g2") assert rg2.metadata.consolidated_metadata == ConsolidatedMetadata(metadata={}) - async def test_group_delitem_consolidated(self, store: Store) -> None: + async def test_group_delitem_consolidated(self, store: Store, zarr_format: ZarrFormat) -> None: + """Deleting a member removes it from the consolidated metadata in memory and in + every document that stores it, so the group reopens without it.""" if isinstance(store, ZipStore): raise pytest.skip("Not implemented") - root = await AsyncGroup.from_store(store=store) + root = await AsyncGroup.from_store(store=store, zarr_format=zarr_format) # Set up the test structure with # / # g0/ # group /g0 @@ -1646,23 +1648,47 @@ async def test_group_delitem_consolidated(self, store: Store) -> None: x2 = await x1.create_group("x2") await x2.create_array("data", shape=(1,), dtype="uint8") - with pytest.warns( # noqa: PT031 - ZarrUserWarning, - match="Consolidated metadata is currently not part in the Zarr format 3 specification.", - ): - if isinstance(store, ZipStore): - with pytest.warns(UserWarning, match="Duplicate name"): - await zarr.api.asynchronous.consolidate_metadata(store) - else: - await zarr.api.asynchronous.consolidate_metadata(store) + with warnings.catch_warnings(): + warnings.filterwarnings( + "ignore", "Consolidated metadata is currently not part", ZarrUserWarning + ) + await zarr.api.asynchronous.consolidate_metadata(store) - group = await zarr.api.asynchronous.open_consolidated(store=store) - assert len(group.metadata.consolidated_metadata.metadata) == 2 - assert "g0" in group.metadata.consolidated_metadata.metadata + group = await zarr.api.asynchronous.open_consolidated(store=store, zarr_format=zarr_format) + assert group.metadata.consolidated_metadata is not None + assert sorted(group.metadata.consolidated_metadata.metadata) == ["g0", "x0"] await group.delitem("g0") - assert len(group.metadata.consolidated_metadata.metadata) == 1 - assert "g0" not in group.metadata.consolidated_metadata.metadata + assert sorted(group.metadata.consolidated_metadata.metadata) == ["x0"] + + reopened = await zarr.api.asynchronous.open_consolidated( + store=store, zarr_format=zarr_format + ) + assert reopened.metadata.consolidated_metadata is not None + assert sorted(reopened.metadata.consolidated_metadata.metadata) == ["x0"] + + def test_group_delitem_consolidated_aliased(self, store: Store) -> None: + """A subgroup read from its parent's consolidated metadata shares it, so a member + deleted through the subgroup is gone through the parent too.""" + if isinstance(store, ZipStore): + raise pytest.skip("Not implemented") + + root = zarr.create_group(store) + root.create_group("sub").create_array("b", shape=(4,), chunks=(2,), dtype="i4") + with warnings.catch_warnings(): + warnings.filterwarnings( + "ignore", "Consolidated metadata is currently not part", ZarrUserWarning + ) + zarr.consolidate_metadata(store) + + group = zarr.open_group(store, mode="r+", use_consolidated=True) + sub = group["sub"] + assert isinstance(sub, Group) + del sub["b"] + assert "b" not in sub + assert "b" not in group["sub"] + with pytest.raises(KeyError): + group["sub/b"] def test_open_consolidated_raises(self, store: Store) -> None: if isinstance(store, ZipStore): diff --git a/tests/test_metadata/test_upgrades.py b/tests/test_metadata/test_upgrades.py index e3494082a0..c0c06a7887 100644 --- a/tests/test_metadata/test_upgrades.py +++ b/tests/test_metadata/test_upgrades.py @@ -874,3 +874,19 @@ def test_consolidate_edge_lists_in_regular_grid(tmp_path: Path, member: str) -> reopened = zarr.open_group(path, mode="r")[member] assert isinstance(reopened, zarr.Array) np.testing.assert_array_equal(reopened[...], data) + + +@pytest.mark.filterwarnings( + "ignore:.*read as that rectilinear chunk grid:zarr.errors.ZarrUserWarning" +) +@pytest.mark.filterwarnings("ignore:Consolidated metadata is currently not part:UserWarning") +def test_delete_member_without_flag_keeps_group(tmp_path: Path) -> None: + """A deletion that fails without the flag leaves the group listing the member.""" + path = tmp_path / "group.zarr" + _store_mixed_group(path) + with zarr.config.set({"array.rectilinear_chunks": False}): + group = zarr.open_group(path, mode="a") + with pytest.raises(ValueError, match="experimental and disabled"): + del group["n"] + assert group.metadata.consolidated_metadata is not None + assert sorted(group.metadata.consolidated_metadata.metadata) == ["mixed", "n"] From 632ac2b539ee52bde947d7d79cdf5c790926607d Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Sat, 26 Sep 2026 11:13:45 +0200 Subject: [PATCH 44/76] fix(metadata): keep accepting the chunk sizes zarr 3.4.0 accepted in metadata constructors (patch release) Nothing in a patch release may reject a chunk size that zarr 3.4.0 accepted. The one chunk edge rule (`parse_chunk_edge`, which also reads RLE repeat counts) now reads any integral number as the `int` it equals: an `int`, a `bool`, a NumPy integer or an integral float, so rectilinear grids written with float edges (`[[4.0, 2]]`) read again and the grid constructors accept NumPy integers; a fractional float is still rejected. A regular chunk shape may again be any iterable. `ArrayV2Metadata(chunks=...)` and `ShardingCodec(chunk_shape=...)` read their chunk shape with `parse_shapelike`, as in 3.4.0: a scalar, NumPy integers, bools and 0 are accepted, and a 0 is written back as given (the kerchunk writer in VirtualiZarr relies on `chunks=(0,)` for empty inlined arrays). Stored documents with a chunk size of 0 are still read by the upgrades. The strict rule returns in the next minor release. Assisted-by: ClaudeCode:claude-opus-5-5 Co-Authored-By: Claude Opus 5.5 --- src/zarr/codecs/sharding.py | 9 +- src/zarr/core/common.py | 25 +++-- src/zarr/core/metadata/upgrades.py | 2 +- src/zarr/core/metadata/v2.py | 9 +- tests/test_metadata/test_upgrades.py | 140 +++++++++++++++++++-------- tests/test_unified_chunk_grid.py | 16 +-- 6 files changed, 136 insertions(+), 65 deletions(-) diff --git a/src/zarr/codecs/sharding.py b/src/zarr/codecs/sharding.py index 52d100bb91..7a053867d6 100644 --- a/src/zarr/codecs/sharding.py +++ b/src/zarr/codecs/sharding.py @@ -45,8 +45,9 @@ merge_and_encode_chunk, ) from zarr.core.common import ( - parse_chunk_shape, + ShapeLike, parse_named_configuration, + parse_shapelike, product, ) from zarr.core.config import config as zarr_config @@ -458,13 +459,13 @@ class ShardingCodec( def __init__( self, *, - chunk_shape: tuple[int, ...] | list[int], + chunk_shape: ShapeLike, codecs: Iterable[Codec | dict[str, JSON]] = (BytesCodec(),), index_codecs: Iterable[Codec | dict[str, JSON]] = (BytesCodec(), Crc32cCodec()), index_location: ShardingCodecIndexLocation | IndexLocation = "end", subchunk_write_order: SubchunkWriteOrder = "morton", ) -> None: - chunk_shape_parsed = parse_chunk_shape(chunk_shape) + chunk_shape_parsed = parse_shapelike(chunk_shape) codecs_parsed = parse_codecs(codecs) index_codecs_parsed = parse_codecs(index_codecs) _check_index_codecs_fixed_size(index_codecs_parsed) @@ -503,7 +504,7 @@ def __getstate__(self) -> dict[str, Any]: def __setstate__(self, state: dict[str, Any]) -> None: config = state["configuration"] - object.__setattr__(self, "chunk_shape", parse_chunk_shape(config["chunk_shape"])) + object.__setattr__(self, "chunk_shape", parse_shapelike(config["chunk_shape"])) object.__setattr__(self, "codecs", parse_codecs(config["codecs"])) object.__setattr__(self, "index_codecs", parse_codecs(config["index_codecs"])) object.__setattr__(self, "index_location", _parse_index_location(config["index_location"])) diff --git a/src/zarr/core/common.py b/src/zarr/core/common.py index 3967c8ca34..cfeb6d3371 100644 --- a/src/zarr/core/common.py +++ b/src/zarr/core/common.py @@ -2,6 +2,7 @@ import asyncio import math +import numbers import warnings from collections.abc import Iterable, Mapping, Sequence from enum import Enum @@ -282,16 +283,24 @@ def _subject(name: str, axis: int | None) -> str: def _parse_positive_int(value: object, name: str, axis: int | None) -> int: + """`value` as an `int` of at least 1. Any integral number is read as the `int` it + equals: an `int`, a `bool`, a NumPy integer or an integral float.""" subject = _subject(name, axis) - if isinstance(value, bool) or not isinstance(value, int): - raise TypeError(f"{subject} must be an int, got {value!r}") - if value < 1: + match value: + case numbers.Integral(): + parsed = int(value) + case float() if value.is_integer(): + parsed = int(value) + case _: + raise TypeError(f"{subject} must be an integer, got {value!r}") + if parsed < 1: raise ValueError(f"{subject} must be >= 1, got {value!r}") - return value + return parsed def parse_chunk_edge(size: object, axis: int | None = None) -> int: - """Check that `size` is a chunk edge length: an `int` (not a `bool`) of at least 1. + """Check that `size` is a chunk edge length: an integral number of at least 1, read + as an `int`. This is the one rule for chunk edge lengths in metadata: bare chunk sizes, explicit edges and run-length encoded sizes. `axis`, when given, is named in the error. @@ -300,12 +309,12 @@ def parse_chunk_edge(size: object, axis: int | None = None) -> int: def parse_chunk_shape(data: object) -> tuple[int, ...]: - """Check a regular chunk shape: a list or tuple of one chunk edge length per axis + """Check a regular chunk shape: an iterable of one chunk edge length per axis (see `parse_chunk_edge`).""" match data: - case list() | tuple(): + case Iterable(): return tuple(parse_chunk_edge(size, axis) for axis, size in enumerate(data)) - raise TypeError(f"A chunk shape must be a list or tuple of chunk edge lengths, got {data!r}") + raise TypeError(f"A chunk shape must be an iterable of chunk edge lengths, got {data!r}") def expand_rle(data: Sequence[object], axis: int | None = None) -> list[int]: diff --git a/src/zarr/core/metadata/upgrades.py b/src/zarr/core/metadata/upgrades.py index 3d7d554f12..8589a0f0c4 100644 --- a/src/zarr/core/metadata/upgrades.py +++ b/src/zarr/core/metadata/upgrades.py @@ -1,7 +1,7 @@ """Upgrades that read invalid stored array metadata documents written by older software. This is the only place invalid metadata is read leniently; the metadata constructors -are strict. An upgrade maps a stored array metadata document (parsed JSON) to a valid +never reinterpret a chunk size. An upgrade maps a stored array metadata document (parsed JSON) to a valid one and says how it read the document. `ArrayV2Metadata.from_dict` and `ArrayV3Metadata.from_dict` apply the upgrades for their Zarr format, so every path that parses a stored document, including consolidated metadata, goes through them, and diff --git a/src/zarr/core/metadata/v2.py b/src/zarr/core/metadata/v2.py index 58b6837b98..4adae2834e 100644 --- a/src/zarr/core/metadata/v2.py +++ b/src/zarr/core/metadata/v2.py @@ -38,7 +38,7 @@ ZARRAY_JSON, ZATTRS_JSON, MemoryOrder, - parse_chunk_shape, + ShapeLike, parse_shapelike, ) from zarr.core.config import config, parse_indexing_order @@ -329,9 +329,10 @@ def parse_compressor(data: object) -> Numcodec | None: raise ValueError(msg) -def parse_chunks(chunks: object, shape: tuple[int, ...]) -> tuple[int, ...]: - """Check a chunk shape: one chunk edge length (an `int` >= 1) per array axis.""" - chunks_parsed = parse_chunk_shape(chunks) +def parse_chunks(chunks: ShapeLike, shape: tuple[int, ...]) -> tuple[int, ...]: + """Check a chunk shape: one non-negative integer per array axis (see + `parse_shapelike`). Stored chunk sizes of 0 are read by `zarr.core.metadata.upgrades`.""" + chunks_parsed = parse_shapelike(chunks) if len(chunks_parsed) != len(shape): raise ValueError( f"The `shape` and `chunks` attributes must have the same length. " diff --git a/tests/test_metadata/test_upgrades.py b/tests/test_metadata/test_upgrades.py index 6e1126fadc..62ae417584 100644 --- a/tests/test_metadata/test_upgrades.py +++ b/tests/test_metadata/test_upgrades.py @@ -221,7 +221,7 @@ def _read_strictly(doc: dict[str, JSON]) -> ArrayV2Metadata | ArrayV3Metadata: def test_stored_negative_chunk_size_rejected() -> None: """No known writer stored a negative chunk size: it is rejected, not upgraded.""" - with pytest.raises(ValueError, match="^Dimension 0: chunk edge length must be >= 1, got -1$"): + with pytest.raises(ValueError, match="^Expected all values to be non-negative"): _read_strictly(_v2_doc([4], [-1])) @@ -232,19 +232,12 @@ def test_stored_chunk_shape_ndim_mismatch_rejected() -> None: _read_strictly(_v3_doc([4, 4], [0])) -def test_stored_float_chunk_size_rejected() -> None: - """No known writer stored a float chunk size: it is rejected, not upgraded.""" +def test_stored_fractional_chunk_size_rejected() -> None: + """A stored chunk size that is not an integral number is rejected, not upgraded.""" with pytest.raises( - TypeError, match=r"^Dimension 0: chunk edge length must be an int, got 4\.0$" + TypeError, match=r"^Dimension 0: chunk edge length must be an integer, got 4\.5$" ): - _read_strictly(_v3_doc([4], [4.0])) - - -def test_stored_zero_inner_chunk_size_rejected() -> None: - """No known writer stored an inner chunk size of 0, and no span defines one: it is - rejected, not upgraded.""" - with pytest.raises(ValueError, match="^Dimension 0: chunk edge length must be >= 1, got 0$"): - _read_strictly(_v3_doc([4], [0], inner=[0])) + _read_strictly(_v3_doc([4], [4.5])) def test_v2_constructor_rejects_chunks_of_wrong_length() -> None: @@ -271,33 +264,52 @@ def _rectilinear_from_dict(chunk_shapes: list[Any]) -> RectilinearChunkGridMetad ) +def _second_edge(grid: RectilinearChunkGridMetadata) -> int: + edges = grid.chunk_shapes[0] + assert isinstance(edges, tuple) + return edges[1] + + CHUNK_EDGE_SITES: dict[str, Callable[[Any], object]] = { - "regular": lambda size: RegularChunkGridMetadata(chunk_shape=(size,)), - "v2": lambda size: _v2_metadata((size,)), - "rectilinear-bare": lambda size: _rectilinear((size,)), - "rectilinear-edge": lambda size: _rectilinear(((4, size),)), - "rectilinear-bare-json": lambda size: _rectilinear_from_dict([size]), - "rectilinear-edge-json": lambda size: _rectilinear_from_dict([[4, size]]), - "rectilinear-rle-json": lambda size: _rectilinear_from_dict([[[size, 2]]]), - "sharding-inner": lambda size: ShardingCodec(chunk_shape=(size,)), + "regular": lambda size: RegularChunkGridMetadata(chunk_shape=(size,)).chunk_shape[0], + "rectilinear-bare": lambda size: _rectilinear((size,)).chunk_shapes[0], + "rectilinear-edge": lambda size: _second_edge(_rectilinear(((4, size),))), + "rectilinear-bare-json": lambda size: _rectilinear_from_dict([size]).chunk_shapes[0], + "rectilinear-edge-json": lambda size: _second_edge(_rectilinear_from_dict([[4, size]])), + "rectilinear-rle-json": lambda size: _second_edge(_rectilinear_from_dict([[[size, 2]]])), } +"""Each place chunk grid metadata reads a chunk edge length, returning the edge it read.""" @pytest.mark.parametrize("site", CHUNK_EDGE_SITES) -@pytest.mark.parametrize("size", [True, False, 4.0, np.int64(4), "4"]) -def test_metadata_rejects_non_int_chunk_edge(site: str, size: object) -> None: - """Metadata built in code takes chunk edge lengths as `int`s only, everywhere.""" +@pytest.mark.parametrize( + ("size", "expected"), + [(4, 4), (True, 1), (np.int64(4), 4), (4.0, 4), (np.float64(4.0), 4)], + ids=["int", "bool", "numpy-int", "float", "numpy-float"], +) +def test_metadata_reads_integral_chunk_edge(site: str, size: object, expected: int) -> None: + """Chunk grid metadata reads any integral number as the `int` chunk edge length it + equals.""" + edge = CHUNK_EDGE_SITES[site](size) + assert type(edge) is int + assert edge == expected + + +@pytest.mark.parametrize("site", CHUNK_EDGE_SITES) +@pytest.mark.parametrize("size", [4.5, float("inf"), "4", None]) +def test_metadata_rejects_non_integral_chunk_edge(site: str, size: object) -> None: with pytest.raises( - TypeError, match=re.escape(f"Dimension 0: chunk edge length must be an int, got {size!r}") + TypeError, + match=re.escape(f"Dimension 0: chunk edge length must be an integer, got {size!r}"), ): CHUNK_EDGE_SITES[site](size) @pytest.mark.parametrize("site", CHUNK_EDGE_SITES) -@pytest.mark.parametrize("size", [0, -1]) +@pytest.mark.parametrize("size", [0, False, -1]) def test_metadata_rejects_chunk_edge_below_one(site: str, size: int) -> None: - """Metadata built in code is strict: a chunk edge length below 1 is rejected, - without a warning.""" + """Chunk grid metadata built in code is strict: a chunk edge length below 1 is + rejected, without a warning.""" with warnings.catch_warnings(): warnings.simplefilter("error", ZarrUserWarning) with pytest.raises( @@ -306,25 +318,71 @@ def test_metadata_rejects_chunk_edge_below_one(site: str, size: int) -> None: CHUNK_EDGE_SITES[site](size) -CHUNK_SHAPE_SITES: dict[str, Callable[[Any], object]] = { - "regular": lambda chunk_shape: RegularChunkGridMetadata(chunk_shape=chunk_shape), - "v2": _v2_metadata, - "sharding-inner": lambda chunk_shape: ShardingCodec(chunk_shape=chunk_shape), -} - - -@pytest.mark.parametrize("site", CHUNK_SHAPE_SITES) -@pytest.mark.parametrize("chunk_shape", [4, np.int64(4), "10", {"a": 1}, range(1, 2)]) -def test_metadata_rejects_chunk_shape_not_list_or_tuple(site: str, chunk_shape: object) -> None: - """A regular chunk shape is a list or tuple; anything else is rejected as a whole, - not iterated as if its elements were chunk edge lengths.""" +@pytest.mark.parametrize("chunk_shape", [4, np.int64(4), None]) +def test_regular_chunk_grid_rejects_chunk_shape_not_iterable(chunk_shape: Any) -> None: with pytest.raises( TypeError, match=re.escape( - f"A chunk shape must be a list or tuple of chunk edge lengths, got {chunk_shape!r}" + f"A chunk shape must be an iterable of chunk edge lengths, got {chunk_shape!r}" ), ): - CHUNK_SHAPE_SITES[site](chunk_shape) + RegularChunkGridMetadata(chunk_shape=chunk_shape) + + +def _sharding_chunk_shape(chunks: Any) -> tuple[tuple[int, ...], object]: + codec = ShardingCodec(chunk_shape=chunks) + configuration = cast("dict[str, JSON]", codec.to_dict()["configuration"]) + return codec.chunk_shape, configuration["chunk_shape"] + + +CHUNK_SHAPE_SITES: dict[str, Callable[[Any], tuple[tuple[int, ...], object]]] = { + "v2": lambda chunks: ((md := _v2_metadata(chunks)).chunks, md.to_dict()["chunks"]), + "sharding-inner": _sharding_chunk_shape, +} +"""`ArrayV2Metadata` and `ShardingCodec` read a chunk shape as an array shape, returning +the chunk shape and the value `to_dict` writes for it.""" + + +@pytest.mark.parametrize("site", CHUNK_SHAPE_SITES) +@pytest.mark.parametrize( + ("chunks", "expected"), + [ + ((4,), (4,)), + ([4], (4,)), + (4, (4,)), + (np.int64(4), (4,)), + ((np.int64(4),), (4,)), + (np.array([4]), (4,)), + ((True,), (1,)), + ((0,), (0,)), + ((False,), (0,)), + (range(4, 5), (4,)), + ], +) +def test_chunk_shape_read_as_array_shape( + site: str, chunks: object, expected: tuple[int, ...] +) -> None: + """`ArrayV2Metadata` and `ShardingCodec` read their chunk shape as `parse_shapelike` + reads an array shape: an integer or an iterable of non-negative integers, including + NumPy integers and bools. A chunk size of 0 is written back as given; reading a + stored 0 is `zarr.core.metadata.upgrades`' business.""" + parsed, written = CHUNK_SHAPE_SITES[site](chunks) + assert parsed == expected + assert all(type(size) is int for size in parsed) + assert written == expected + + +@pytest.mark.parametrize("site", CHUNK_SHAPE_SITES) +def test_chunk_shape_read_as_array_shape_rejects_negative(site: str) -> None: + with pytest.raises(ValueError, match="Expected all values to be non-negative"): + CHUNK_SHAPE_SITES[site]((-1,)) + + +@pytest.mark.parametrize("site", CHUNK_SHAPE_SITES) +@pytest.mark.parametrize("chunks", [(4.0,), "4", None]) +def test_chunk_shape_read_as_array_shape_rejects_non_integer(site: str, chunks: object) -> None: + with pytest.raises(TypeError, match="Expected an"): + CHUNK_SHAPE_SITES[site](chunks) def _rewrite_doc(path: Path, zarr_format: Literal[2, 3], edit: Any) -> None: diff --git a/tests/test_unified_chunk_grid.py b/tests/test_unified_chunk_grid.py index 8f06863a05..ae3feacd3d 100644 --- a/tests/test_unified_chunk_grid.py +++ b/tests/test_unified_chunk_grid.py @@ -494,6 +494,8 @@ def test_chunk_grid_iter() -> None: [ ([[10, 3]], [10, 10, 10]), ([[10, 2], [20, 1]], [10, 10, 20]), + ([4.0, [4.0, 2], [5, 2.0]], [4, 4, 4, 5, 5]), + ([[True, 2], [np.int64(3), np.int64(1)]], [1, 1, 3]), ], ) def test_rle_expand(compressed: list[Any], expected: list[int]) -> None: @@ -550,14 +552,14 @@ def test_rle_expand_rejects_invalid(rle_input: list[Any], match: str) -> None: @pytest.mark.parametrize( ("rle_input", "match"), [ - ([10.0], "Chunk edge length must be an int, got 10.0"), - ([True], "Chunk edge length must be an int, got True"), - ([[10, 3.0]], "RLE repeat count must be an int, got 3.0"), + ([10.5], "Chunk edge length must be an integer, got 10.5"), + (["10"], "Chunk edge length must be an integer, got '10'"), + ([[10, 3.5]], "RLE repeat count must be an integer, got 3.5"), ], - ids=["float-edge", "bool-edge", "float-count"], + ids=["fractional-edge", "string-edge", "fractional-count"], ) def test_rle_expand_rejects_non_int(rle_input: list[Any], match: str) -> None: - """expand_rle takes JSON integers only: no stored document holds integral floats.""" + """expand_rle reads integral numbers only.""" with pytest.raises(TypeError, match=match): expand_rle(rle_input) @@ -566,11 +568,11 @@ def test_rle_expand_rejects_non_int(rle_input: list[Any], match: str) -> None: ("rle_input", "match"), [ ([0], "chunk edge length must be >= 1"), - ([10.0], "chunk edge length must be an int"), + ([10.5], "chunk edge length must be an integer"), ([[5, 0]], "RLE repeat count must be >= 1"), ([[5, 2, 1]], r"RLE entries must be an integer or \[size, count\]"), ], - ids=["zero-edge", "float-edge", "zero-rle-count", "rle-entry-of-three"], + ids=["zero-edge", "fractional-edge", "zero-rle-count", "rle-entry-of-three"], ) def test_rle_expand_names_dimension(rle_input: list[Any], match: str) -> None: """Given the dimension `axis` the edges belong to, every error of expand_rle names it.""" From 245acaed57544ec9801c1678181a4898cf331f34 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Sat, 26 Sep 2026 11:14:18 +0200 Subject: [PATCH 45/76] fix(array): leave a valid stored document as written on a stale handle's first write A handle read from an upgraded document re-reads the stored document before its first chunk write. If that document no longer needs an upgrade, store nothing: re-encoding it could differ from how another implementation wrote it, and nothing about it needs fixing. Assisted-by: ClaudeCode:claude-opus-5-5 Co-Authored-By: Claude Opus 5.5 --- src/zarr/core/array.py | 13 +++++++------ tests/test_metadata/test_upgrades.py | 27 +++++++++++++++++++++++++++ 2 files changed, 34 insertions(+), 6 deletions(-) diff --git a/src/zarr/core/array.py b/src/zarr/core/array.py index 64d6080838..da5b0595a8 100644 --- a/src/zarr/core/array.py +++ b/src/zarr/core/array.py @@ -1617,21 +1617,22 @@ async def _save_metadata(self, metadata: ArrayMetadata, ensure_parents: bool = F await save_metadata(self.store_path, metadata, ensure_parents=ensure_parents) async def _store_upgraded_document(self) -> None: - """Store the upgrade of this array's current stored document, if it differs from - what the store holds. + """Store the upgrade of this array's current stored document, if it needs one. Only for metadata read from a document that had to be upgraded (see `zarr.core.metadata.upgrades`). The document is read again, because the store may hold a newer one than this handle's metadata: if that one needs no upgrade (the - array was re-saved or resized since), nothing is stored. Storing the same upgrade - twice is harmless, so concurrent callers need no coordination. + array was re-saved or resized since, possibly by another implementation), it is + left as written. Storing the same upgrade twice is harmless, so concurrent + callers need no coordination. """ if not self.metadata._stored_document_upgraded: return zarr_format = self.metadata.zarr_format stored = await get_array_metadata(self.store_path, zarr_format=zarr_format) - upgraded, _ = upgrade_array_document(stored, zarr_format) - await upsert_metadata(self.store_path, parse_array_metadata(upgraded)) + upgraded, readings = upgrade_array_document(stored, zarr_format) + if readings: + await upsert_metadata(self.store_path, parse_array_metadata(upgraded)) object.__setattr__(self.metadata, "_stored_document_upgraded", False) async def _set_selection( diff --git a/tests/test_metadata/test_upgrades.py b/tests/test_metadata/test_upgrades.py index 62ae417584..5f86ed233c 100644 --- a/tests/test_metadata/test_upgrades.py +++ b/tests/test_metadata/test_upgrades.py @@ -553,6 +553,33 @@ def test_stale_handle_write_keeps_newer_metadata( np.testing.assert_array_equal(reopened[...], [9, 0, 0, 1, 2, 3, 4, 5, 6]) +@pytest.mark.parametrize("zarr_format", [2, 3]) +def test_stale_handle_write_keeps_valid_document_as_written( + tmp_path: Path, zarr_format: Literal[2, 3] +) -> None: + """If the document the store holds when a handle read from an upgraded document + first writes chunks needs no upgrade, it is left as written, even where zarr would + encode the same metadata differently (as another implementation may have written it).""" + path = tmp_path / "legacy.zarr" + _legacy_array(path, zarr_format) + with pytest.warns(ZarrUserWarning, match="is read as"): + stale = zarr.open_array(store=path, mode="r+") + # Valid metadata for the same array, as another writer might store it: chunk size + # 3, without the optional members zarr writes, as compact JSON. + doc_path = path / (".zarray" if zarr_format == 2 else "zarr.json") + doc = json.loads(doc_path.read_text()) + for optional in ("dimension_separator", "attributes", "storage_transformers"): + doc.pop(optional, None) + _stored_chunks(doc)[0] = 3 + doc_path.write_text(json.dumps(doc, separators=(",", ":"))) + written = doc_path.read_bytes() + + stale[0] = 9 + + assert doc_path.read_bytes() == written + np.testing.assert_array_equal(_open_strictly(path)[...], [9, 0, 0]) + + @pytest.mark.parametrize("zarr_format", [2, 3]) def test_empty_write_stores_no_metadata(tmp_path: Path, zarr_format: Literal[2, 3]) -> None: """A write of an empty selection stores no chunks, so it stores no metadata either.""" From 2d0121db7ea51eb0b4c318c51809bba0824b94ad Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Sat, 26 Sep 2026 11:15:35 +0200 Subject: [PATCH 46/76] docs: describe the 4334 changes to metadata constructors for a patch release Assisted-by: ClaudeCode:claude-opus-5-5 Co-Authored-By: Claude Opus 5.5 --- changes/4334.bugfix.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/changes/4334.bugfix.md b/changes/4334.bugfix.md index 460ba6ece8..220b04e805 100644 --- a/changes/4334.bugfix.md +++ b/changes/4334.bugfix.md @@ -1,5 +1,5 @@ A chunk edge length is now always at least 1, while an array extent may be 0. Every spelling of one chunk spanning an axis (`chunks=-1`, `chunks=False`, `chunks="auto"`, `shards=-1`, `shards=False`) gives chunk size 1 on a zero-length axis, and a shard spanning an axis is a multiple of the inner chunk size, so `shards=-1` no longer fails when the axis length is not a multiple of it. Rectilinear chunk grids can be created on a zero-length dimension with a non-empty list of positive chunk sizes, which are kept for later growth. -Metadata built in code is strict about chunk edge lengths: `ArrayV2Metadata`, `RegularChunkGridMetadata`, `RectilinearChunkGridMetadata` and `ShardingCodec` take them as Python `int`s of at least 1, so a size of 0 now raises a `ValueError`, and a `bool`, a float or a NumPy integer (including a scalar `chunks=np.int64(5)` for `ArrayV2Metadata`) raises a `TypeError`, as does a chunk shape that is not a list or tuple; `FixedDimension(size=0, ...)` raises a `ValueError`. Stored rectilinear chunk grids with integral floats (`10.0`) are rejected too. The array creation functions still accept NumPy integers in `chunks=` and `shards=`. +`FixedDimension(size=0, ...)` now raises a `ValueError`, so an array can no longer be built directly from metadata with a chunk size of 0; `ArrayV2Metadata(chunks=(0,))` itself is still accepted and written as given. The chunk grid metadata classes read a chunk edge length given as a NumPy integer, a `bool` or an integral float as the `int` it equals; `RectilinearChunkGridMetadata` used to keep such a value as given, so that it could not be stored (a NumPy integer) or read back (a `bool`). Stored metadata with a regular chunk size of 0 or JSON `false` now opens with a `ZarrUserWarning` and is read as one chunk spanning the axis (a multiple of the inner chunk size for sharded arrays, which previously failed to open). zarr-python wrote such sizes for arrays created with a zero-length axis until 3.4, and, from 2.18.7 to 3.2.1, for an explicit chunk size of 0 or `False` on an axis of any length, as in `chunks=(0,)` (those arrays could store no data). JSON `true`, as zarr-python 3.0 and 3.2 wrote for `chunks=(True,)`, is read as 1, in the chunk grid and in the inner chunk shape of a sharding codec. The warning names the array and says how to store valid metadata: `array.update_attributes({})`, then `zarr.consolidate_metadata` if the metadata is consolidated. Writing data to such an array (other than an empty selection) first stores the upgrade of the metadata the store then holds, unless another writer has stored valid metadata since, so that other readers find the chunks written; `zarr.consolidate_metadata` likewise stores the upgraded metadata of each such array before the consolidated metadata. From 8c6af467d18c83694fc7ee7d6d1df60ad1d26dc0 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Sat, 26 Sep 2026 11:17:08 +0200 Subject: [PATCH 47/76] test: expect the patch release's wording for non-integral chunk edges (patch release) The chunk edge rule reads any integral number (loop/4334), so a float or bool edge in a stored mixed regular grid is read as the int it equals; a fractional or string edge is still rejected, with "must be an integer". Assisted-by: ClaudeCode:claude-opus-5-5 Co-Authored-By: Claude Opus 5.5 --- tests/test_metadata/test_upgrades.py | 12 +++++++----- tests/test_metadata/test_v3.py | 3 ++- 2 files changed, 9 insertions(+), 6 deletions(-) diff --git a/tests/test_metadata/test_upgrades.py b/tests/test_metadata/test_upgrades.py index 995b7884e6..d5d2903734 100644 --- a/tests/test_metadata/test_upgrades.py +++ b/tests/test_metadata/test_upgrades.py @@ -786,20 +786,22 @@ def test_regular_grid_of_only_edge_lists_rejected() -> None: """A regular chunk shape made only of edge lists was never stored (a rectilinear chunk grid was), so it is not read as rectilinear.""" info = _rejected_without_warning(_mixed_doc([6, 20], [[1, 5], [5, 10, 5]])) - assert info.match(re.escape("Dimension 0: chunk edge length must be an int, got [1, 5]")) + assert info.match(re.escape("Dimension 0: chunk edge length must be an integer, got [1, 5]")) def test_run_length_encoded_edges_in_regular_grid_rejected() -> None: """Run-length encoded edges were never stored in a regular chunk shape.""" info = _rejected_without_warning(_mixed_doc([6, 20], [2, [[5, 2], 10]])) - assert info.match(re.escape("Dimension 1: chunk edge length must be an int, got [[5, 2], 10]")) + assert info.match( + re.escape("Dimension 1: chunk edge length must be an integer, got [[5, 2], 10]") + ) -@pytest.mark.parametrize("edge", [5.0, True], ids=["float", "bool"]) +@pytest.mark.parametrize("edge", [5.5, "5"], ids=["fractional", "string"]) def test_non_int_edge_in_regular_grid_rejected(edge: object) -> None: - """An edge that is not an int is reported as such, not blamed on its list.""" + """An edge that is not an integer is reported as such, not blamed on its list.""" info = _rejected_without_warning(_mixed_doc([6, 20], [2, [edge, 15]])) - assert info.match(re.escape(f"Dimension 1: chunk edge length must be an int, got {edge!r}")) + assert info.match(re.escape(f"Dimension 1: chunk edge length must be an integer, got {edge!r}")) def test_edge_below_one_in_regular_grid_rejected() -> None: diff --git a/tests/test_metadata/test_v3.py b/tests/test_metadata/test_v3.py index 4b16dbc4af..4374bfbd6c 100644 --- a/tests/test_metadata/test_v3.py +++ b/tests/test_metadata/test_v3.py @@ -159,7 +159,8 @@ def test_create_chunk_grid_metadata_unknown_dimension_type() -> None: def test_regular_chunk_grid_rejects_edge_lists() -> None: """A regular chunk grid only accepts integer chunk edge lengths.""" with pytest.raises( - TypeError, match=re.escape("Dimension 1: chunk edge length must be an int, got (5, 10, 5)") + TypeError, + match=re.escape("Dimension 1: chunk edge length must be an integer, got (5, 10, 5)"), ): RegularChunkGridMetadata(chunk_shape=(2, (5, 10, 5))) # type: ignore[arg-type] From 4c4fe219e124f3e71182a36148a1e1144d77f244 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Sat, 26 Sep 2026 12:05:59 +0200 Subject: [PATCH 48/76] fix(metadata): read float edges only in stored rectilinear documents The patch release widened the one chunk edge rule to accept integral floats so that stored rectilinear grids written with float edges (`[[4.0, 2]]`, as zarr 3.2 wrote for float edges) keep opening. That also made a stored regular chunk shape `[10.0]` open, which zarr 3.4.0 rejected and the next minor release would reject again. The constructor rule now accepts integers of any integer type (`int`, `bool`, NumPy integers) and rejects floats. A new document upgrade reads an integral JSON float of at least 1 as an `int` where zarr 3.2 stored one: the explicit edges and run-length encoded sizes of a rectilinear chunk grid, with the standard warning. Bare sizes, repeat counts, regular chunk shapes and inner chunk shapes stay for the constructors to reject. Assisted-by: ClaudeCode:claude-opus-5-5 Co-Authored-By: Claude Opus 5.5 --- changes/4334.bugfix.md | 4 +- src/zarr/core/common.py | 20 ++-- src/zarr/core/metadata/upgrades.py | 60 +++++++++++- tests/test_metadata/test_upgrades.py | 132 +++++++++++++++++++++++++-- tests/test_unified_chunk_grid.py | 15 ++- 5 files changed, 206 insertions(+), 25 deletions(-) diff --git a/changes/4334.bugfix.md b/changes/4334.bugfix.md index 220b04e805..4e719b78c6 100644 --- a/changes/4334.bugfix.md +++ b/changes/4334.bugfix.md @@ -1,5 +1,5 @@ A chunk edge length is now always at least 1, while an array extent may be 0. Every spelling of one chunk spanning an axis (`chunks=-1`, `chunks=False`, `chunks="auto"`, `shards=-1`, `shards=False`) gives chunk size 1 on a zero-length axis, and a shard spanning an axis is a multiple of the inner chunk size, so `shards=-1` no longer fails when the axis length is not a multiple of it. Rectilinear chunk grids can be created on a zero-length dimension with a non-empty list of positive chunk sizes, which are kept for later growth. -`FixedDimension(size=0, ...)` now raises a `ValueError`, so an array can no longer be built directly from metadata with a chunk size of 0; `ArrayV2Metadata(chunks=(0,))` itself is still accepted and written as given. The chunk grid metadata classes read a chunk edge length given as a NumPy integer, a `bool` or an integral float as the `int` it equals; `RectilinearChunkGridMetadata` used to keep such a value as given, so that it could not be stored (a NumPy integer) or read back (a `bool`). +`FixedDimension(size=0, ...)` now raises a `ValueError`, so an array can no longer be built directly from metadata with a chunk size of 0; `ArrayV2Metadata(chunks=(0,))` itself is still accepted and written as given. The chunk grid metadata classes read a chunk edge length given as a NumPy integer or a `bool` as the `int` it equals; `RectilinearChunkGridMetadata` used to keep such a value as given, so that it could not be stored (a NumPy integer) or read back (a `bool`). `RectilinearChunkGridMetadata` now rejects a float edge length such as `4.0` with a `TypeError`, as the other metadata classes do; it used to keep it and store it as a JSON float, which the Zarr specification does not allow. -Stored metadata with a regular chunk size of 0 or JSON `false` now opens with a `ZarrUserWarning` and is read as one chunk spanning the axis (a multiple of the inner chunk size for sharded arrays, which previously failed to open). zarr-python wrote such sizes for arrays created with a zero-length axis until 3.4, and, from 2.18.7 to 3.2.1, for an explicit chunk size of 0 or `False` on an axis of any length, as in `chunks=(0,)` (those arrays could store no data). JSON `true`, as zarr-python 3.0 and 3.2 wrote for `chunks=(True,)`, is read as 1, in the chunk grid and in the inner chunk shape of a sharding codec. The warning names the array and says how to store valid metadata: `array.update_attributes({})`, then `zarr.consolidate_metadata` if the metadata is consolidated. Writing data to such an array (other than an empty selection) first stores the upgrade of the metadata the store then holds, unless another writer has stored valid metadata since, so that other readers find the chunks written; `zarr.consolidate_metadata` likewise stores the upgraded metadata of each such array before the consolidated metadata. +Stored metadata with a regular chunk size of 0 or JSON `false` now opens with a `ZarrUserWarning` and is read as one chunk spanning the axis (a multiple of the inner chunk size for sharded arrays, which previously failed to open). zarr-python wrote such sizes for arrays created with a zero-length axis until 3.4, and, from 2.18.7 to 3.2.1, for an explicit chunk size of 0 or `False` on an axis of any length, as in `chunks=(0,)` (those arrays could store no data). JSON `true`, as zarr-python 3.0 and 3.2 wrote for `chunks=(True,)`, is read as 1, in the chunk grid and in the inner chunk shape of a sharding codec. A stored rectilinear chunk grid whose edge lengths are integral JSON floats (`[[4.0, 2]]`), as zarr-python 3.2 wrote for float edges, is read with those edges as integers, now with a warning; a float anywhere no release wrote one (a regular chunk shape, the inner chunk shape of a sharding codec, a run-length repeat count) is rejected. The warning names the array and says how to store valid metadata: `array.update_attributes({})`, then `zarr.consolidate_metadata` if the metadata is consolidated. Writing data to such an array (other than an empty selection) first stores the upgrade of the metadata the store then holds, unless another writer has stored valid metadata since, so that other readers find the chunks written; `zarr.consolidate_metadata` likewise stores the upgraded metadata of each such array before the consolidated metadata. diff --git a/src/zarr/core/common.py b/src/zarr/core/common.py index cfeb6d3371..dc85737289 100644 --- a/src/zarr/core/common.py +++ b/src/zarr/core/common.py @@ -283,24 +283,22 @@ def _subject(name: str, axis: int | None) -> str: def _parse_positive_int(value: object, name: str, axis: int | None) -> int: - """`value` as an `int` of at least 1. Any integral number is read as the `int` it - equals: an `int`, a `bool`, a NumPy integer or an integral float.""" + """`value` as an `int` of at least 1. An integer of any integer type (an `int`, a + `bool` or a NumPy integer) is read as the `int` it equals; a float is rejected, even + an integral one (stored documents with integral floats are read by + `zarr.core.metadata.upgrades`).""" subject = _subject(name, axis) - match value: - case numbers.Integral(): - parsed = int(value) - case float() if value.is_integer(): - parsed = int(value) - case _: - raise TypeError(f"{subject} must be an integer, got {value!r}") + if not isinstance(value, numbers.Integral): + raise TypeError(f"{subject} must be an integer, got {value!r}") + parsed = int(value) if parsed < 1: raise ValueError(f"{subject} must be >= 1, got {value!r}") return parsed def parse_chunk_edge(size: object, axis: int | None = None) -> int: - """Check that `size` is a chunk edge length: an integral number of at least 1, read - as an `int`. + """Check that `size` is a chunk edge length: an integer of at least 1, read as an + `int`. This is the one rule for chunk edge lengths in metadata: bare chunk sizes, explicit edges and run-length encoded sizes. `axis`, when given, is named in the error. diff --git a/src/zarr/core/metadata/upgrades.py b/src/zarr/core/metadata/upgrades.py index 8589a0f0c4..4d978d36c9 100644 --- a/src/zarr/core/metadata/upgrades.py +++ b/src/zarr/core/metadata/upgrades.py @@ -57,6 +57,12 @@ def _is_int_list(value: object) -> TypeGuard[list[int]]: return isinstance(value, list) and all(isinstance(v, int) for v in value) +def _abbreviate(value: JSON, limit: int = 60) -> str: + """`value` as JSON, cut to at most `limit` characters.""" + text = json.dumps(value) + return text if len(text) <= limit else f"{text[: limit - 3]}..." + + def _read_chunk_size(size: JSON, span: int | None, unit: int) -> tuple[int, str | None] | None: """Read one entry of a stored regular chunk shape as a chunk edge length. @@ -180,9 +186,61 @@ def _invalid_chunk_sizes_v3(doc: ArrayDocument) -> tuple[ArrayDocument, str] | N return None +def _read_edge_length(edge: JSON) -> JSON: + """Read one stored chunk edge length of a rectilinear chunk grid: an integral JSON + float of at least 1 (`4.0`) is read as the `int` it equals. Anything else is kept, + for the metadata constructors to check.""" + match edge: + case float() if edge.is_integer() and edge >= 1: + return int(edge) + return edge + + +def _read_rectilinear_axis(spec: JSON) -> JSON: + """Read the stored chunk edge lengths of one axis of a rectilinear chunk grid (see + `_read_edge_length`): its explicit edges and the sizes of its run-length encoded + `[size, count]` pairs. A bare chunk size and a repeat count are kept as stored: no + writer stored them as floats.""" + match spec: + case list(): + return [_read_rectilinear_entry(entry) for entry in spec] + return spec + + +def _read_rectilinear_entry(entry: JSON) -> JSON: + match entry: + case [size, count]: + return [_read_edge_length(size), count] + return _read_edge_length(entry) + + +def _float_edge_lengths_v3(doc: ArrayDocument) -> tuple[ArrayDocument, str] | None: + grid = doc.get("chunk_grid") + if not (isinstance(grid, Mapping) and grid.get("name") == "rectilinear"): + return None + configuration = grid.get("configuration") + if not isinstance(configuration, Mapping): + return None + stored = configuration.get("chunk_shapes") + if not isinstance(stored, list): + return None + read = [_read_rectilinear_axis(axis) for axis in stored] + # An integral float and the `int` it equals compare equal, but not as JSON text. + axes = [axis for axis, spec in enumerate(stored) if json.dumps(spec) != json.dumps(read[axis])] + if not axes: + return None + reading = ( + f"The stored chunk edge lengths {_abbreviate(stored)} are invalid: chunk edge " + f"lengths must be integers. They are read as {_abbreviate(read)}, reading each " + f"edge length stored as a float, in dimensions {axes}, as the integer it equals." + ) + upgraded = {**configuration, "chunk_shapes": read} + return {**doc, "chunk_grid": {**grid, "configuration": upgraded}}, reading + + ARRAY_UPGRADES: Final[Mapping[ZarrFormat, tuple[Upgrade, ...]]] = { 2: (_invalid_chunk_sizes_v2,), - 3: (_invalid_inner_chunk_sizes_v3, _invalid_chunk_sizes_v3), + 3: (_invalid_inner_chunk_sizes_v3, _invalid_chunk_sizes_v3, _float_edge_lengths_v3), } """The upgrades of an array document of each Zarr format, applied in order.""" diff --git a/tests/test_metadata/test_upgrades.py b/tests/test_metadata/test_upgrades.py index 5f86ed233c..f2da059b16 100644 --- a/tests/test_metadata/test_upgrades.py +++ b/tests/test_metadata/test_upgrades.py @@ -47,7 +47,7 @@ def _v2_doc(shape: list[int], chunks: list[Any]) -> dict[str, JSON]: def _v3_doc( - shape: list[int], chunk_shape: list[Any], inner: list[int] | None = None + shape: list[int], chunk_shape: list[Any], inner: list[Any] | None = None ) -> dict[str, JSON]: bytes_codec: dict[str, JSON] = {"name": "bytes", "configuration": {"endian": "little"}} codecs: list[JSON] = [bytes_codec] @@ -240,6 +240,116 @@ def test_stored_fractional_chunk_size_rejected() -> None: _read_strictly(_v3_doc([4], [4.5])) +def _rectilinear_doc(shape: list[int], chunk_shapes: list[Any]) -> dict[str, JSON]: + return _v3_doc(shape, [1] * len(shape)) | { + "chunk_grid": { + "name": "rectilinear", + "configuration": {"kind": "inline", "chunk_shapes": chunk_shapes}, + } + } + + +@pytest.mark.parametrize( + ("chunk_shapes", "expected", "warning"), + [ + ([[[4, 2]], [[5, 2]]], ((4, 4), (5, 5)), None), + ( + [[[4.0, 2]], [[5, 2]]], + ((4, 4), (5, 5)), + ( + r"^The stored chunk edge lengths \[\[\[4\.0, 2\]\], \[\[5, 2\]\]\] are " + r"invalid: .* read as \[\[\[4, 2\]\], \[\[5, 2\]\]\], .* in dimensions \[0\]," + ), + ), + ([[3.0, 5.0], [[5, 2]]], ((3, 5), (5, 5)), r"read as \[\[3, 5\], "), + ([[[4.0, 2], 2.0], [[5, 2]]], ((4, 4, 2), (5, 5)), r"read as \[\[\[4, 2\], 2\], "), + ([[[4.0, 2]], [10]], ((4, 4), (10,)), r"read as \[\[\[4, 2\]\], \[10\]\]"), + ([[[4.0, 2]], [5.0, 5]], ((4, 4), (5, 5)), r"in dimensions \[0, 1\],"), + ], + ids=["valid", "rle-size", "edges", "rle-size-and-edge", "sharded-outer", "2d"], +) +def test_read_float_edges_in_rectilinear_grid( + chunk_shapes: list[Any], + expected: tuple[tuple[int, ...], ...], + warning: str | None, +) -> None: + """A stored rectilinear chunk grid whose explicit edges or run-length encoded sizes + are integral floats, as zarr-python wrote them when given float edges, is read with + those edges as `int`s; `from_dict` warns once, naming the array and the dimensions, + and how to re-save.""" + shape = [sum(edges) for edges in expected] + doc = _rectilinear_doc(shape, chunk_shapes) + with ( + zarr.config.set({"array.rectilinear_chunks": True}), + warnings.catch_warnings(record=True) as record, + ): + warnings.simplefilter("always") + metadata = ArrayV3Metadata.from_dict(doc, path="group/array") + assert metadata.chunk_grid == RectilinearChunkGridMetadata(chunk_shapes=expected) + messages = [str(w.message) for w in record] + if warning is None: + assert messages == [] + else: + [message] = messages + assert message.startswith("Array 'group/array': ") + assert re.search(warning, message.removeprefix("Array 'group/array': ")) + assert message.endswith(RESAVE_HINT) + + +@pytest.mark.parametrize( + ("doc", "error"), + [ + (_v3_doc([20], [10.0]), "Dimension 0: chunk edge length must be an integer, got 10.0"), + ( + _v3_doc([8], [4], inner=[2.0]), + "Expected an iterable of integers. Got [2.0] instead.", + ), + (_v2_doc([20], [10.0]), "Expected an iterable of integers. Got [10.0] instead."), + ( + _rectilinear_doc([8], [[[4, 2.0]]]), + "Dimension 0: RLE repeat count must be an integer, got 2.0", + ), + ( + _rectilinear_doc([8], [4.0]), + "Dimension 0: chunk edge length must be an integer, got 4.0", + ), + ], + ids=["regular", "sharding-inner", "v2", "rle-count", "rectilinear-bare"], +) +def test_stored_float_chunk_size_rejected(doc: dict[str, JSON], error: str) -> None: + """A float chunk size is read only where zarr-python stored one, as the edges of a + rectilinear chunk grid. Anywhere else it is rejected, as zarr 3.4.0 rejected a + stored regular chunk size of `10.0`.""" + with zarr.config.set({"array.rectilinear_chunks": True}), pytest.raises(TypeError) as info: + _read_strictly(doc) + assert info.match(re.escape(error)) + + +def test_float_edges_round_trip(tmp_path: Path) -> None: + """A store whose rectilinear chunk grid holds the float edges zarr-python wrote opens + with a warning, reads its data and re-saves the edges as `int`s.""" + path = tmp_path / "float.zarr" + data = np.arange(80, dtype="int16").reshape(8, 10) + with zarr.config.set({"array.rectilinear_chunks": True}): + zarr.create_array(path, shape=data.shape, chunks=[[4, 4], [5, 5]], dtype="int16")[...] = ( + data + ) + + def store_floats(doc: dict[str, Any]) -> None: + doc["chunk_grid"]["configuration"]["chunk_shapes"] = [[[4.0, 2]], [[5, 2]]] + + _rewrite_doc(path, 3, store_floats) + with pytest.warns(ZarrUserWarning, match=r"read as \[\[\[4, 2\]\], \[\[5, 2\]\]\]"): + arr = zarr.open_array(path, mode="a") + np.testing.assert_array_equal(arr[...], data) + arr.update_attributes({}) + stored = json.loads((path / "zarr.json").read_text())["chunk_grid"]["configuration"] + assert json.dumps(stored["chunk_shapes"]) == "[[[4, 2]], [[5, 2]]]" + with warnings.catch_warnings(): + warnings.simplefilter("error", ZarrUserWarning) + np.testing.assert_array_equal(zarr.open_array(path)[...], data) + + def test_v2_constructor_rejects_chunks_of_wrong_length() -> None: with pytest.raises(ValueError, match="`chunks` has length 1, but `shape` has length 2"): ArrayV2Metadata(shape=(4, 4), chunks=(2,), dtype=Int16(), fill_value=0, order="C") @@ -284,20 +394,26 @@ def _second_edge(grid: RectilinearChunkGridMetadata) -> int: @pytest.mark.parametrize("site", CHUNK_EDGE_SITES) @pytest.mark.parametrize( ("size", "expected"), - [(4, 4), (True, 1), (np.int64(4), 4), (4.0, 4), (np.float64(4.0), 4)], - ids=["int", "bool", "numpy-int", "float", "numpy-float"], + [(4, 4), (True, 1), (np.int64(4), 4)], + ids=["int", "bool", "numpy-int"], ) -def test_metadata_reads_integral_chunk_edge(site: str, size: object, expected: int) -> None: - """Chunk grid metadata reads any integral number as the `int` chunk edge length it - equals.""" +def test_metadata_reads_integer_chunk_edge(site: str, size: object, expected: int) -> None: + """Chunk grid metadata reads an integer of any integer type as the `int` chunk edge + length it equals.""" edge = CHUNK_EDGE_SITES[site](size) assert type(edge) is int assert edge == expected @pytest.mark.parametrize("site", CHUNK_EDGE_SITES) -@pytest.mark.parametrize("size", [4.5, float("inf"), "4", None]) -def test_metadata_rejects_non_integral_chunk_edge(site: str, size: object) -> None: +@pytest.mark.parametrize( + "size", + [4.0, np.float64(4.0), 4.5, float("inf"), "4", None], + ids=["float", "numpy-float", "fractional", "inf", "str", "none"], +) +def test_metadata_rejects_non_integer_chunk_edge(site: str, size: object) -> None: + """A float is not a chunk edge length in metadata built in code, even an integral + one; stored documents with integral floats are read by the upgrades.""" with pytest.raises( TypeError, match=re.escape(f"Dimension 0: chunk edge length must be an integer, got {size!r}"), diff --git a/tests/test_unified_chunk_grid.py b/tests/test_unified_chunk_grid.py index ae3feacd3d..8df53df08b 100644 --- a/tests/test_unified_chunk_grid.py +++ b/tests/test_unified_chunk_grid.py @@ -494,7 +494,6 @@ def test_chunk_grid_iter() -> None: [ ([[10, 3]], [10, 10, 10]), ([[10, 2], [20, 1]], [10, 10, 20]), - ([4.0, [4.0, 2], [5, 2.0]], [4, 4, 4, 5, 5]), ([[True, 2], [np.int64(3), np.int64(1)]], [1, 1, 3]), ], ) @@ -553,13 +552,23 @@ def test_rle_expand_rejects_invalid(rle_input: list[Any], match: str) -> None: ("rle_input", "match"), [ ([10.5], "Chunk edge length must be an integer, got 10.5"), + ([10.0], "Chunk edge length must be an integer, got 10.0"), + ([[10.0, 3]], "Chunk edge length must be an integer, got 10.0"), (["10"], "Chunk edge length must be an integer, got '10'"), ([[10, 3.5]], "RLE repeat count must be an integer, got 3.5"), + ([[10, 3.0]], "RLE repeat count must be an integer, got 3.0"), + ], + ids=[ + "fractional-edge", + "float-edge", + "float-rle-size", + "string-edge", + "fractional-count", + "float-count", ], - ids=["fractional-edge", "string-edge", "fractional-count"], ) def test_rle_expand_rejects_non_int(rle_input: list[Any], match: str) -> None: - """expand_rle reads integral numbers only.""" + """expand_rle reads integers only, not floats.""" with pytest.raises(TypeError, match=match): expand_rle(rle_input) From c678c9c0598e448ed3c53ea56c8dff499923bdd6 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Sat, 26 Sep 2026 12:08:35 +0200 Subject: [PATCH 49/76] test: read float edges in a regular grid that lists chunk edge lengths zarr 3.2 stored `[2, [5.0, 10.0, 5.0]]` for float edges given to a regular grid. The mixed-grid upgrade reads it as a rectilinear grid, and the float edge upgrade that follows reads its edges as integers, both in one warning. Assisted-by: ClaudeCode:claude-opus-5-5 Co-Authored-By: Claude Opus 5.5 --- changes/4375.bugfix.md | 2 +- src/zarr/core/metadata/upgrades.py | 4 ++-- tests/test_metadata/test_upgrades.py | 12 +++++++++++- 3 files changed, 14 insertions(+), 4 deletions(-) diff --git a/changes/4375.bugfix.md b/changes/4375.bugfix.md index 875356a2b3..e037ad2be0 100644 --- a/changes/4375.bugfix.md +++ b/changes/4375.bugfix.md @@ -1,3 +1,3 @@ -Arrays whose stored `regular` chunk grid mixes chunk sizes with lists of chunk edge lengths, such as `"chunk_shape": [2, [5, 10, 5]]` (written by zarr 3.2.0 and 3.2.1 for `chunks=(2, (5, 10, 5))`), can be read again, without enabling `array.rectilinear_chunks`. The grid is read as the rectilinear chunk grid it describes, with a `ZarrUserWarning`; re-saving the metadata stores that rectilinear grid, which requires the flag. A regular chunk grid given edge lists is now rejected, so this metadata is no longer written. +Arrays whose stored `regular` chunk grid mixes chunk sizes with lists of chunk edge lengths, such as `"chunk_shape": [2, [5, 10, 5]]` (written by zarr 3.2.0 and 3.2.1 for `chunks=(2, (5, 10, 5))`, and as `[2, [5.0, 10.0, 5.0]]` for float edges), can be read again, without enabling `array.rectilinear_chunks`. The grid is read as the rectilinear chunk grid it describes, with a `ZarrUserWarning`; re-saving the metadata stores that rectilinear grid, which requires the flag. A regular chunk grid given edge lists is now rejected, so this metadata is no longer written. The `array.rectilinear_chunks` flag now gates reading and storing array metadata that declares a rectilinear chunk grid, including in a group's consolidated metadata, instead of constructing `RectilinearChunkGridMetadata`. `Array.resize`, deleting a group member, and `create_hierarchy(..., overwrite=True)` now encode the metadata they will store before deleting anything, so metadata that cannot be stored (for example, a rectilinear chunk grid with the flag off) fails with the store untouched. That error names the array when it is a member of a group's consolidated metadata. diff --git a/src/zarr/core/metadata/upgrades.py b/src/zarr/core/metadata/upgrades.py index f985bf6fae..fd1a01878b 100644 --- a/src/zarr/core/metadata/upgrades.py +++ b/src/zarr/core/metadata/upgrades.py @@ -76,8 +76,8 @@ def _read_chunk_size( size of a shard). `span` is `None` where no stored 0 is known, as in the inner chunk shape of a sharding codec: 0 is then left for the constructors to reject. A flat JSON list is kept as the chunk edge lengths of its axis, which only a - rectilinear chunk grid can declare (see `_invalid_chunk_sizes_v3`); the rectilinear - chunk grid checks each edge. + rectilinear chunk grid can declare (see `_invalid_chunk_sizes_v3`); its edges are + read as those of a stored rectilinear chunk grid (see `_float_edge_lengths_v3`). """ match size: case True: diff --git a/tests/test_metadata/test_upgrades.py b/tests/test_metadata/test_upgrades.py index 9a31123d5e..be1ecb7076 100644 --- a/tests/test_metadata/test_upgrades.py +++ b/tests/test_metadata/test_upgrades.py @@ -848,6 +848,14 @@ def _mixed_doc(shape: list[int], chunk_shape: list[Any]) -> dict[str, JSON]: r"\[1, \[5, 10, 5\]\], reading true in dimension 0 as 1\. The stored chunk grid" ), ), + ( + _mixed_doc([6, 20], [2, [5.0, 10.0, 5.0]]), + (2, (5, 10, 5)), + ( + r"^The stored chunk grid .* \[1\], .* The stored chunk edge lengths " + r"\[2, \[5\.0, 10\.0, 5\.0\]\] are invalid: .* read as \[2, \[5, 10, 5\]\]" + ), + ), ( _mixed_doc([4, 10_000], [True, [10] * 1000]), (1, (10,) * 1000), @@ -862,6 +870,7 @@ def _mixed_doc(shape: list[int], chunk_shape: list[Any]) -> dict[str, JSON]: "empty-int-axis", "empty-edge-axis", "true", + "float-edges", "long", ], ) @@ -885,8 +894,9 @@ def test_read_edge_lists_in_regular_grid( assert message.startswith("Array 'group/array': ") assert ( "Re-saving the metadata stores that rectilinear chunk grid, so each step that " - "follows requires `zarr.config.set({'array.rectilinear_chunks': True})`. " + RESAVE_HINT + "follows requires `zarr.config.set({'array.rectilinear_chunks': True})`. " ) in message + assert message.endswith(RESAVE_HINT) assert len(message) < 1000 From 6fde402d18b48e4721d11bed6b4849f727786b89 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Sat, 26 Sep 2026 14:31:27 +0200 Subject: [PATCH 50/76] test(codecs): pickle a ShardingCodec with an inner chunk size of 0 Assisted-by: ClaudeCode:claude-opus-5-5 Co-Authored-By: Claude Opus 5.5 --- tests/test_codecs/test_sharding.py | 25 ++++++++++++++++--------- 1 file changed, 16 insertions(+), 9 deletions(-) diff --git a/tests/test_codecs/test_sharding.py b/tests/test_codecs/test_sharding.py index e2619a6ce2..831285ae9f 100644 --- a/tests/test_codecs/test_sharding.py +++ b/tests/test_codecs/test_sharding.py @@ -760,17 +760,24 @@ def test_structured_dtype_fill_value() -> None: assert np.array_equal(arr[:], expected) -def test_pickle() -> None: +@pytest.mark.parametrize( + "codec", + [ + ShardingCodec(chunk_shape=(8, 8)), + ShardingCodec(chunk_shape=(8, 8), subchunk_write_order="lexicographic"), + ShardingCodec(chunk_shape=(0,)), + ], + ids=["default", "lexicographic", "zero-chunk-size"], +) +def test_pickle(codec: ShardingCodec) -> None: """ShardingCodec round-trips through pickle, including the non-serialized ``subchunk_write_order`` (which ``to_dict`` omits and which must not silently - revert to the ``morton`` default).""" - codec = ShardingCodec(chunk_shape=(8, 8)) - assert pickle.loads(pickle.dumps(codec)) == codec - - ordered = ShardingCodec(chunk_shape=(8, 8), subchunk_write_order="lexicographic") - restored = pickle.loads(pickle.dumps(ordered)) - assert restored == ordered - assert restored.subchunk_write_order == "lexicographic" + revert to the ``morton`` default), and an inner chunk size of 0, which the + constructor accepts.""" + restored = pickle.loads(pickle.dumps(codec)) + assert restored == codec + assert restored.chunk_shape == codec.chunk_shape + assert restored.subchunk_write_order == codec.subchunk_write_order @pytest.mark.parametrize("store", ["local", "memory"], indirect=["store"]) From b0ea125cb22e7feaeaa48c22dfaad5b589498d33 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Sat, 26 Sep 2026 14:31:31 +0200 Subject: [PATCH 51/76] fix(group): build every node before create_hierarchy deletes or stores anything A node whose metadata no array or group can be built from now fails with the store untouched, even with overwrite=True. Assisted-by: ClaudeCode:claude-opus-5-5 Co-Authored-By: Claude Opus 5.5 --- src/zarr/core/group.py | 38 ++++++++++++++++++++++++++++---------- tests/test_group.py | 24 ++++++++++++++++++++++++ 2 files changed, 52 insertions(+), 10 deletions(-) diff --git a/src/zarr/core/group.py b/src/zarr/core/group.py index d6cc572465..38eede55ec 100644 --- a/src/zarr/core/group.py +++ b/src/zarr/core/group.py @@ -3085,6 +3085,7 @@ async def create_hierarchy( # ensure that all nodes have the same zarr_format, and add implicit groups as needed nodes_parsed = _parse_hierarchy_dict(data=nodes_normed_keys) redundant_implicit_groups = [] + to_delete_keys: list[str] = [] # empty hierarchies should be a no-op if len(nodes_parsed) > 0: @@ -3117,13 +3118,7 @@ async def create_hierarchy( if overwrite: # we will remove any nodes that collide with arrays and non-implicit groups defined in # nodes - - # track the keys of nodes we need to delete - to_delete_keys = [] - to_delete_keys.extend( - [k for k, v in nodes_parsed.items() if k not in implicit_group_keys] - ) - await asyncio.gather(*(store.delete_dir(key) for key in to_delete_keys)) + to_delete_keys = [k for k in nodes_parsed if k not in implicit_group_keys] else: # This type is long. coros: ( @@ -3183,7 +3178,11 @@ async def create_hierarchy( else: nodes_explicit[k] = v - async for key, node in create_nodes(store=store, nodes=nodes_explicit): + # Build every node before deleting or storing anything: metadata that no array or + # group can be built from then fails with the store untouched. + built = _build_nodes(store, nodes_explicit) + await asyncio.gather(*(store.delete_dir(key) for key in to_delete_keys)) + async for key, node in _store_nodes(store, nodes_explicit, built): yield key, node @@ -3213,7 +3212,26 @@ async def create_nodes( AsyncGroup | AsyncArray The created nodes in the order they are created. """ + async for key, node in _store_nodes(store, nodes, _build_nodes(store, nodes)): + yield key, node + +def _build_nodes( + store: Store, nodes: Mapping[str, GroupMetadata | ArrayV2Metadata | ArrayV3Metadata] +) -> dict[str, AsyncGroup | AnyAsyncArray]: + """The array or group each of `nodes` describes, at its path in `store`.""" + return { + path: _build_node(store=store, path=path, metadata=meta) for path, meta in nodes.items() + } + + +async def _store_nodes( + store: Store, + nodes: Mapping[str, GroupMetadata | ArrayV2Metadata | ArrayV3Metadata], + built: Mapping[str, AsyncGroup | AnyAsyncArray], +) -> AsyncIterator[tuple[str, AsyncGroup | AnyAsyncArray]]: + """Store the metadata of `nodes` and yield the nodes `_build_nodes` built from them + (see `create_nodes`).""" # Note: the only way to alter this value is via the config. If that's undesirable for some reason, # then we should consider adding a keyword argument to this function semaphore = asyncio.Semaphore(config.get("async.concurrency")) @@ -3241,7 +3259,7 @@ async def create_nodes( node_name = created_key[: created_key.rfind("/")] meta_out = nodes[node_name] if meta_out.zarr_format == 3: - yield node_name, _build_node(store=store, path=node_name, metadata=meta_out) + yield node_name, built[node_name] else: # For zarr v2 # we only want to yield when both the metadata and attributes are created @@ -3256,7 +3274,7 @@ async def create_nodes( meta_done = _join_paths([node_name, ZARRAY_JSON]) in created_object_keys if meta_done and attrs_done: - yield node_name, _build_node(store=store, path=node_name, metadata=meta_out) + yield node_name, built[node_name] continue diff --git a/tests/test_group.py b/tests/test_group.py index 31fbd138cd..b6da55fa2a 100644 --- a/tests/test_group.py +++ b/tests/test_group.py @@ -1919,6 +1919,30 @@ async def test_create_hierarchy( assert expected_meta == {k: v.metadata for k, v in created.items()} +@pytest.mark.parametrize("zarr_format", [2, 3]) +def test_create_hierarchy_unbuildable_node_leaves_store_untouched( + monkeypatch: pytest.MonkeyPatch, zarr_format: ZarrFormat +) -> None: + """`create_hierarchy` builds every node before it deletes or stores anything, so a + node that cannot be built fails with the store untouched, even when overwriting.""" + store = MemoryStore() + group = zarr.create_group(store, zarr_format=zarr_format) + group.create_array("a", shape=(2,), chunks=(1,), dtype="int8")[:] = [1, 2] + before = dict(store._store_dict) + + def unbuildable(**kwargs: object) -> None: + raise RuntimeError("cannot build this node") + + monkeypatch.setattr(zarr.core.group, "_build_node", unbuildable) + with pytest.raises(RuntimeError, match="cannot build this node"): + dict( + zarr.create_hierarchy( + store=store, nodes={"a": GroupMetadata(zarr_format=zarr_format)}, overwrite=True + ) + ) + assert store._store_dict == before + + @pytest.mark.parametrize("store", ["memory"], indirect=True) @pytest.mark.parametrize("extant_node", ["array", "group"]) @pytest.mark.parametrize("impl", ["async", "sync"]) From a903d66a96c5bc15c46ca3ae8653b8916b1ea25e Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Sat, 26 Sep 2026 14:31:33 +0200 Subject: [PATCH 52/76] fix(metadata): warn only where the user must act; guard and refresh upgraded copies - Readings that give what zarr 3.4.0 read (0/false on an empty axis, JSON true, float rectilinear edges) are silent; the metadata is still marked upgraded, so the first write re-saves it. - JSON true is read as 1 in stored rectilinear edges and RLE sizes and in nested sharding codecs' inner chunk shapes. - A write through a handle whose stored document now lays out chunks differently raises and stores nothing; with no stored document it writes as before. - Storing group metadata refreshes each upgraded consolidated member from its own stored document (the one place: save_metadata); consolidate_metadata no longer needs its own pass. - Metadata built in code with chunk size 0 is read through the upgrades, so an array can be built from it, as in 3.4.0. - Chunk edges in metadata follow the 3.4.0 integer rule (int; bool read as its value); a string or mapping is rejected as a chunk shape as a whole. Assisted-by: ClaudeCode:claude-opus-5-5 Co-Authored-By: Claude Opus 5.5 --- changes/4334.bugfix.md | 6 +- src/zarr/api/asynchronous.py | 11 +- src/zarr/core/_json.py | 6 + src/zarr/core/array.py | 65 +++- src/zarr/core/common.py | 27 +- src/zarr/core/group.py | 6 +- src/zarr/core/metadata/io.py | 57 +++- src/zarr/core/metadata/upgrades.py | 211 ++++++------ src/zarr/core/metadata/v2.py | 3 +- src/zarr/core/metadata/v3.py | 3 +- tests/test_metadata/test_upgrades.py | 473 ++++++++++++++++++--------- tests/test_unified_chunk_grid.py | 25 +- 12 files changed, 572 insertions(+), 321 deletions(-) diff --git a/changes/4334.bugfix.md b/changes/4334.bugfix.md index 4e719b78c6..3828d097eb 100644 --- a/changes/4334.bugfix.md +++ b/changes/4334.bugfix.md @@ -1,5 +1,7 @@ A chunk edge length is now always at least 1, while an array extent may be 0. Every spelling of one chunk spanning an axis (`chunks=-1`, `chunks=False`, `chunks="auto"`, `shards=-1`, `shards=False`) gives chunk size 1 on a zero-length axis, and a shard spanning an axis is a multiple of the inner chunk size, so `shards=-1` no longer fails when the axis length is not a multiple of it. Rectilinear chunk grids can be created on a zero-length dimension with a non-empty list of positive chunk sizes, which are kept for later growth. -`FixedDimension(size=0, ...)` now raises a `ValueError`, so an array can no longer be built directly from metadata with a chunk size of 0; `ArrayV2Metadata(chunks=(0,))` itself is still accepted and written as given. The chunk grid metadata classes read a chunk edge length given as a NumPy integer or a `bool` as the `int` it equals; `RectilinearChunkGridMetadata` used to keep such a value as given, so that it could not be stored (a NumPy integer) or read back (a `bool`). `RectilinearChunkGridMetadata` now rejects a float edge length such as `4.0` with a `TypeError`, as the other metadata classes do; it used to keep it and store it as a JSON float, which the Zarr specification does not allow. +`FixedDimension(size=0, ...)` now raises a `ValueError`. `ArrayV2Metadata(chunks=(0,))` is still accepted and written as given; an array built from such metadata (with `create_hierarchy`, for example) reads the chunk size as a stored chunk size of 0 is read (below). `create_hierarchy` now builds every array and group before it deletes or stores anything, so a node that cannot be built fails with the store untouched. The regular chunk grid metadata class reads a `bool` chunk edge length as the `int` it equals, and rejects a string or a mapping as a chunk shape as a whole. `RectilinearChunkGridMetadata`, which is experimental, now reads a `bool` edge length as the `int` it equals and rejects a NumPy integer or a float edge length such as `4.0` with a `TypeError`; it used to keep them as given, so that it could not store a NumPy integer, could not read back a `bool`, and stored a float as a JSON float, which the Zarr specification does not allow. -Stored metadata with a regular chunk size of 0 or JSON `false` now opens with a `ZarrUserWarning` and is read as one chunk spanning the axis (a multiple of the inner chunk size for sharded arrays, which previously failed to open). zarr-python wrote such sizes for arrays created with a zero-length axis until 3.4, and, from 2.18.7 to 3.2.1, for an explicit chunk size of 0 or `False` on an axis of any length, as in `chunks=(0,)` (those arrays could store no data). JSON `true`, as zarr-python 3.0 and 3.2 wrote for `chunks=(True,)`, is read as 1, in the chunk grid and in the inner chunk shape of a sharding codec. A stored rectilinear chunk grid whose edge lengths are integral JSON floats (`[[4.0, 2]]`), as zarr-python 3.2 wrote for float edges, is read with those edges as integers, now with a warning; a float anywhere no release wrote one (a regular chunk shape, the inner chunk shape of a sharding codec, a run-length repeat count) is rejected. The warning names the array and says how to store valid metadata: `array.update_attributes({})`, then `zarr.consolidate_metadata` if the metadata is consolidated. Writing data to such an array (other than an empty selection) first stores the upgrade of the metadata the store then holds, unless another writer has stored valid metadata since, so that other readers find the chunks written; `zarr.consolidate_metadata` likewise stores the upgraded metadata of each such array before the consolidated metadata. +Stored metadata with a regular chunk size of 0 or JSON `false` (Zarr format 2 `chunks`, Zarr format 3 `regular` `chunk_shape`) is now read as one chunk spanning the axis (a multiple of the inner chunk size for sharded arrays, which previously failed to open). zarr-python wrote such sizes for arrays created with a zero-length axis until 3.4, and, from 2.18.7 to 3.2.1, for an explicit chunk size of 0 or `False` on an axis of any length, as in `chunks=(0,)` (those arrays could store no data). On a zero-length axis this opens silently, as before; appending to such an axis then stores chunks of size 1. On an axis of positive length it opens with a `ZarrUserWarning`, which says that the axis holds only the fill value and how to store valid metadata: `array.update_attributes({})`, then `zarr.consolidate_metadata` if the metadata is consolidated. JSON `true`, as zarr-python 3.0 and 3.2 wrote for a chunk size of `True`, is read as 1, silently, in the chunk grid (regular, and the explicit edges and run-length encoded sizes of a rectilinear one) and in the inner chunk shape of every sharding codec, nested or not. A stored rectilinear chunk grid whose edge lengths are integral JSON floats (`[[4.0, 2]]`), as zarr-python 3.2 wrote for float edges, is read with those edges as integers, silently; a float anywhere no release wrote one (a regular chunk shape, the inner chunk shape of a sharding codec, a run-length repeat count, an edge below 1) is rejected. + +Writing data to an array read from such metadata (other than an empty selection) first stores the upgrade of the metadata the store then holds, unless another writer has stored valid metadata since, so that other readers find the chunks written; in a `ZipStore` this adds a second entry for the metadata document, as every metadata update does. If the metadata the store then holds lays out chunks differently (another writer resized the array keeping the chunk size of 0, say), the write raises a `ValueError` asking to reopen the array, and stores nothing. Storing a group's metadata, which `zarr.consolidate_metadata` and every change to a group with consolidated metadata do, likewise first stores the upgrade of each such member's metadata and consolidates the member's metadata as the store then holds it. diff --git a/src/zarr/api/asynchronous.py b/src/zarr/api/asynchronous.py index 4832bafaba..1fc10cdd1e 100644 --- a/src/zarr/api/asynchronous.py +++ b/src/zarr/api/asynchronous.py @@ -231,15 +231,10 @@ async def consolidate_metadata( group = await AsyncGroup.open(store_path, zarr_format=zarr_format, use_consolidated=False) group.store_path.store._check_writable() - members = { - k: v async for k, v in group.members(max_depth=None, use_consolidated_for_children=False) + members_metadata = { + k: v.metadata + async for k, v in group.members(max_depth=None, use_consolidated_for_children=False) } - # Store the upgrade of every member document that had to be upgraded before the - # consolidated document that describes the upgraded metadata. - for member in members.values(): - if isinstance(member, AsyncArray): - await member._store_upgraded_document() - members_metadata = {k: v.metadata for k, v in members.items()} # While consolidating, we want to be explicit about when child groups # are empty by inserting an empty dict for consolidated_metadata.metadata for k, v in members_metadata.items(): diff --git a/src/zarr/core/_json.py b/src/zarr/core/_json.py index efe8152a4f..4fd5ffab95 100644 --- a/src/zarr/core/_json.py +++ b/src/zarr/core/_json.py @@ -41,6 +41,12 @@ def buffer_to_json(buffer: Buffer) -> JSON: return cast("JSON", json.loads(buffer.to_bytes())) +def json_equal(a: object, b: object) -> bool: + """Whether two JSON values have the same JSON encoding. Python compares `True` and + `1`, or `1.0` and `1`, as equal; JSON does not.""" + return json.dumps(a) == json.dumps(b) + + def buffer_to_json_object(buffer: Buffer) -> dict[str, JSON]: """Parse the contents of a `Buffer` as a JSON object (a `dict`). diff --git a/src/zarr/core/array.py b/src/zarr/core/array.py index da5b0595a8..a1d6a73033 100644 --- a/src/zarr/core/array.py +++ b/src/zarr/core/array.py @@ -118,8 +118,8 @@ ArrayV2MetadataDict, ArrayV3Metadata, ) -from zarr.core.metadata.io import save_metadata, upsert_metadata -from zarr.core.metadata.upgrades import upgrade_array_document +from zarr.core.metadata.io import read_stored_array, save_metadata, upsert_metadata +from zarr.core.metadata.upgrades import mark_upgraded, upgrade_array_document from zarr.core.metadata.v2 import ( CompressorLikev2, get_object_codec_id, @@ -202,10 +202,30 @@ def _chunk_sizes_from_shape( return tuple(result) +def _as_json(value: Any) -> Any: + """`value`, as `to_dict` returns it, with its tuples as JSON arrays.""" + match value: + case tuple() | list(): + return [_as_json(item) for item in value] + case dict(): + return {key: _as_json(item) for key, item in value.items()} + return value + + def parse_array_metadata(data: Any, path: str | None = None) -> ArrayMetadata: + """Array metadata from a metadata object or a metadata document, naming the array at + `path` in warnings about how an invalid document was read. + + The metadata constructors accept chunk sizes that only an invalid document holds + (such as 0), as they always have; such metadata is read as its document is (see + `zarr.core.metadata.upgrades`), so an array can be built from it. No data was read + or written under those chunk sizes, so none of its readings is a warning.""" if isinstance(data, ArrayMetadata): - return data - elif isinstance(data, dict): + document, readings = upgrade_array_document(_as_json(data.to_dict()), data.zarr_format) + if not readings: + return data + return mark_upgraded(parse_array_metadata(dict(document)), [None for _ in readings], path) + if isinstance(data, dict): zarr_format = data.get("zarr_format") if zarr_format == 3: meta_out = ArrayV3Metadata.from_dict(data, path=path) @@ -1617,22 +1637,31 @@ async def _save_metadata(self, metadata: ArrayMetadata, ensure_parents: bool = F await save_metadata(self.store_path, metadata, ensure_parents=ensure_parents) async def _store_upgraded_document(self) -> None: - """Store the upgrade of this array's current stored document, if it needs one. + """Store the upgrade of this array's current stored document, if it needs one, + before chunks are written under this handle's metadata. Only for metadata read from a document that had to be upgraded (see `zarr.core.metadata.upgrades`). The document is read again, because the store may - hold a newer one than this handle's metadata: if that one needs no upgrade (the - array was re-saved or resized since, possibly by another implementation), it is - left as written. Storing the same upgrade twice is harmless, so concurrent - callers need no coordination. + hold a newer one than this handle's metadata. If that one lays out chunks + differently (the array was resized since by software that kept the invalid chunk + size), this handle would write chunks no reader finds, so it raises and stores + nothing. If it needs no upgrade (the array was re-saved since, possibly by another + implementation), it is left as written; if there is none, there is nothing to + upgrade. Storing the same upgrade twice is harmless, so concurrent callers need + no coordination. """ if not self.metadata._stored_document_upgraded: return - zarr_format = self.metadata.zarr_format - stored = await get_array_metadata(self.store_path, zarr_format=zarr_format) - upgraded, readings = upgrade_array_document(stored, zarr_format) - if readings: - await upsert_metadata(self.store_path, parse_array_metadata(upgraded)) + read = await read_stored_array(self.store_path, self.metadata.zarr_format) + if read is not None: + current, upgraded = read + if _chunk_layout(current) != _chunk_layout(self.metadata): + raise ValueError( + f"The metadata stored for the array at {str(self.store_path)!r} has " + "changed since this array was opened: reopen the array to write to it." + ) + if upgraded: + await upsert_metadata(self.store_path, current) object.__setattr__(self.metadata, "_stored_document_upgraded", False) async def _set_selection( @@ -4864,6 +4893,14 @@ async def create_array( ) +def _chunk_layout(metadata: ArrayMetadata) -> tuple[object, tuple[int, ...] | None]: + """How an array's chunks are laid out: its chunk grid and, if it is sharded, the + inner chunk shape.""" + grid = metadata.chunks if isinstance(metadata, ArrayV2Metadata) else metadata.chunk_grid + sharding = _sharding_codec(metadata) + return grid, None if sharding is None else sharding.chunk_shape + + def _sharding_codec(metadata: ArrayMetadata) -> ShardingCodec | None: """The array's sharding codec, or None if the array is not sharded. diff --git a/src/zarr/core/common.py b/src/zarr/core/common.py index dc85737289..7ec4e2d04b 100644 --- a/src/zarr/core/common.py +++ b/src/zarr/core/common.py @@ -2,7 +2,6 @@ import asyncio import math -import numbers import warnings from collections.abc import Iterable, Mapping, Sequence from enum import Enum @@ -283,22 +282,20 @@ def _subject(name: str, axis: int | None) -> str: def _parse_positive_int(value: object, name: str, axis: int | None) -> int: - """`value` as an `int` of at least 1. An integer of any integer type (an `int`, a - `bool` or a NumPy integer) is read as the `int` it equals; a float is rejected, even - an integral one (stored documents with integral floats are read by - `zarr.core.metadata.upgrades`).""" + """`value` as an `int` of at least 1. A `bool` is read as the `int` it equals; any + other type, a NumPy integer or a float (even an integral one: stored documents with + integral floats are read by `zarr.core.metadata.upgrades`), is rejected.""" subject = _subject(name, axis) - if not isinstance(value, numbers.Integral): - raise TypeError(f"{subject} must be an integer, got {value!r}") - parsed = int(value) - if parsed < 1: + if not isinstance(value, int): + raise TypeError(f"{subject} must be an int, got {value!r}") + if value < 1: raise ValueError(f"{subject} must be >= 1, got {value!r}") - return parsed + return int(value) def parse_chunk_edge(size: object, axis: int | None = None) -> int: - """Check that `size` is a chunk edge length: an integer of at least 1, read as an - `int`. + """Check that `size` is a chunk edge length: an `int` of at least 1 (a `bool` is + read as the `int` it equals). This is the one rule for chunk edge lengths in metadata: bare chunk sizes, explicit edges and run-length encoded sizes. `axis`, when given, is named in the error. @@ -307,9 +304,11 @@ def parse_chunk_edge(size: object, axis: int | None = None) -> int: def parse_chunk_shape(data: object) -> tuple[int, ...]: - """Check a regular chunk shape: an iterable of one chunk edge length per axis - (see `parse_chunk_edge`).""" + """Check a regular chunk shape: an iterable, other than a string or a mapping, of one + chunk edge length per axis (see `parse_chunk_edge`).""" match data: + case str() | Mapping(): + pass case Iterable(): return tuple(parse_chunk_edge(size, axis) for axis, size in enumerate(data)) raise TypeError(f"A chunk shape must be an iterable of chunk edge lengths, got {data!r}") diff --git a/src/zarr/core/group.py b/src/zarr/core/group.py index 38eede55ec..5a52e3bbf3 100644 --- a/src/zarr/core/group.py +++ b/src/zarr/core/group.py @@ -13,7 +13,6 @@ import zarr.api.asynchronous as async_api from zarr.abc.metadata import Metadata -from zarr.abc.store import Store, set_or_delete from zarr.core._info import GroupInfo from zarr.core._json import buffer_to_json_object, json_to_buffer from zarr.core.array import ( @@ -75,6 +74,7 @@ ) from typing import Any + from zarr.abc.store import Store from zarr.core.array_spec import ArrayConfigLike from zarr.core.buffer import Buffer, BufferPrototype from zarr.core.chunk_key_encodings import ChunkKeyEncodingLike @@ -2115,9 +2115,7 @@ async def update_attributes_async(self, new_attributes: dict[str, Any]) -> Group new_metadata = replace(self.metadata, attributes=new_attributes) # Write new metadata - to_save = new_metadata.to_buffer_dict(default_buffer_prototype()) - awaitables = [set_or_delete(self.store_path / key, value) for key, value in to_save.items()] - await asyncio.gather(*awaitables) + await save_metadata(self.store_path, new_metadata) async_group = replace(self._async_group, metadata=new_metadata) return replace(self, _async_group=async_group) diff --git a/src/zarr/core/metadata/io.py b/src/zarr/core/metadata/io.py index 7b5e0ced0d..1503736e04 100644 --- a/src/zarr/core/metadata/io.py +++ b/src/zarr/core/metadata/io.py @@ -1,16 +1,17 @@ from __future__ import annotations import asyncio -import json +from dataclasses import replace from enum import Enum from itertools import zip_longest from typing import TYPE_CHECKING, Final, NamedTuple from zarr.abc.store import set_or_delete -from zarr.core._json import buffer_to_json_object +from zarr.core._json import buffer_to_json_object, json_equal from zarr.core.buffer.core import default_buffer_prototype from zarr.core.buffer.cpu import buffer_prototype as cpu_buffer_prototype -from zarr.errors import ContainsArrayError +from zarr.core.metadata.upgrades import upgrade_array_document +from zarr.errors import ArrayNotFoundError, ContainsArrayError from zarr.storage._common import StorePath, ensure_no_existing_node if TYPE_CHECKING: @@ -65,7 +66,7 @@ def _diff( case list(), list(): for index, pair in enumerate(zip_longest(stored, new, fillvalue=ABSENT)): yield from _diff((*path, index), *pair) - case _ if ABSENT in (stored, new) or json.dumps(stored) != json.dumps(new): + case _ if ABSENT in (stored, new) or not json_equal(stored, new): yield DocumentChange(path, stored, new) @@ -103,6 +104,49 @@ async def upsert_metadata( return changes +async def read_stored_array( + store_path: StorePath, zarr_format: ZarrFormat +) -> tuple[ArrayMetadata, bool] | None: + """The metadata of the array document stored at `store_path` as it now is, read with + the upgrades but without their warnings (the handle that asks has warned), and + whether the document had to be upgraded; `None` if no array document is stored + there.""" + from zarr.core.array import get_array_metadata, parse_array_metadata + + try: + stored = await get_array_metadata(store_path, zarr_format=zarr_format) + except ArrayNotFoundError: + return None + upgraded, readings = upgrade_array_document(stored, zarr_format) + return parse_array_metadata(upgraded, str(store_path)), bool(readings) + + +async def _refresh_consolidated(store_path: StorePath, metadata: GroupMetadata) -> GroupMetadata: + """`metadata` with each member of its consolidated metadata that was read from a + document that had to be upgraded replaced by the metadata of the member's own + document as it now is, after storing that document's upgrade if it needs one: the + member document may have changed since, so the consolidated copy is never stored + as if it were valid.""" + from zarr.core.group import GroupMetadata + + consolidated = metadata.consolidated_metadata + if consolidated is None: + return metadata + members = dict(consolidated.metadata) + for name, member in consolidated.metadata.items(): + if isinstance(member, GroupMetadata): + members[name] = await _refresh_consolidated(store_path / name, member) + elif member._stored_document_upgraded: + read = await read_stored_array(store_path / name, member.zarr_format) + if read is not None: + members[name], upgraded = read + if upgraded: + await upsert_metadata(store_path / name, members[name]) + if all(members[name] is member for name, member in consolidated.metadata.items()): + return metadata + return replace(metadata, consolidated_metadata=replace(consolidated, metadata=members)) + + def _build_parents(store_path: StorePath, zarr_format: ZarrFormat) -> dict[str, GroupMetadata]: from zarr.core.group import GroupMetadata @@ -140,6 +184,11 @@ async def save_metadata( ------ ValueError """ + from zarr.core.group import GroupMetadata + + if isinstance(metadata, GroupMetadata): + # The one place group metadata is stored, and with it consolidated metadata. + metadata = await _refresh_consolidated(store_path, metadata) to_save = metadata.to_buffer_dict(default_buffer_prototype()) set_awaitables = [store_documents(store_path, to_save)] diff --git a/src/zarr/core/metadata/upgrades.py b/src/zarr/core/metadata/upgrades.py index 4d978d36c9..52812e334c 100644 --- a/src/zarr/core/metadata/upgrades.py +++ b/src/zarr/core/metadata/upgrades.py @@ -1,13 +1,17 @@ """Upgrades that read invalid stored array metadata documents written by older software. -This is the only place invalid metadata is read leniently; the metadata constructors -never reinterpret a chunk size. An upgrade maps a stored array metadata document (parsed JSON) to a valid -one and says how it read the document. `ArrayV2Metadata.from_dict` and +This is the only place invalid metadata is read leniently. An upgrade maps a stored +array metadata document (parsed JSON) to a valid one. `ArrayV2Metadata.from_dict` and `ArrayV3Metadata.from_dict` apply the upgrades for their Zarr format, so every path -that parses a stored document, including consolidated metadata, goes through them, and -warn once, with every reading, after the upgraded document has passed the metadata -constructor. An invalid document therefore raises its own error, not a warning about -how it was read. +that parses a stored document, including consolidated metadata, goes through them. + +A reading warns only where the user must act on it; a reading that gives what zarr +read from the same document before these upgrades existed is silent, so a document +that opened without a warning still does. The warnings are given once per document, +after the upgraded document has passed the metadata constructor, so an invalid +document raises its own error, not a warning about how it was read. Silent or not, +metadata read from an upgraded document is marked (see `mark_upgraded`), so the array +stores the upgrade before it writes chunks under it. To read another kind of invalid document, add an upgrade to `ARRAY_UPGRADES`. """ @@ -18,8 +22,9 @@ import warnings from collections.abc import Callable, Iterable, Mapping, Sequence from itertools import chain, repeat -from typing import TYPE_CHECKING, Final, TypeGuard, cast +from typing import TYPE_CHECKING, Final, TypeGuard +from zarr.core._json import json_equal from zarr.core.chunk_grids import full_span_chunk_size from zarr.errors import ZarrUserWarning @@ -27,9 +32,9 @@ from zarr.core.common import JSON, ZarrFormat type ArrayDocument = Mapping[str, JSON] -type Upgrade = Callable[[ArrayDocument], tuple[ArrayDocument, str] | None] -"""Returns `None` if the document needs no upgrade, else the upgraded document and a -sentence saying how it was read.""" +type Upgrade = Callable[[ArrayDocument], tuple[ArrayDocument, str | None] | None] +"""Returns `None` if the document needs no upgrade, else the upgraded document and, +if the user must act on how it was read, a warning saying so (else `None`).""" RESAVE_HINT: Final = ( "To store valid metadata, open the array writable and call `array.update_attributes({})`; " @@ -38,17 +43,19 @@ ) -def mark_upgraded[M](metadata: M, readings: Sequence[str], path: str | None) -> M: +def mark_upgraded[M](metadata: M, readings: Sequence[str | None], path: str | None) -> M: """Record that `metadata` was read from a stored document that needed the upgrades - whose `readings` `upgrade_array_document` returned, if any: warn once, naming the - array at `path` when the caller knows it, and set `_stored_document_upgraded` on - `metadata`, so the array stores the upgrade before it writes chunks under it.""" + whose `readings` `upgrade_array_document` returned, if any: set + `_stored_document_upgraded` on `metadata`, so the array stores the upgrade before + it writes chunks under it, and warn once with the readings that are warnings, + naming the array at `path` when the caller knows it.""" if readings: + object.__setattr__(metadata, "_stored_document_upgraded", True) + if messages := [reading for reading in readings if reading is not None]: subject = "" if path is None else f"Array {path!r}: " # The synchronous API parses metadata on zarr's IO thread, whose stack holds no # user code, so the warning points at the `from_dict` that read the document. - warnings.warn(f"{subject}{' '.join(readings)} {RESAVE_HINT}", ZarrUserWarning, stacklevel=2) - object.__setattr__(metadata, "_stored_document_upgraded", True) + warnings.warn(f"{subject}{' '.join(messages)} {RESAVE_HINT}", ZarrUserWarning, stacklevel=2) return metadata @@ -57,47 +64,42 @@ def _is_int_list(value: object) -> TypeGuard[list[int]]: return isinstance(value, list) and all(isinstance(v, int) for v in value) -def _abbreviate(value: JSON, limit: int = 60) -> str: - """`value` as JSON, cut to at most `limit` characters.""" - text = json.dumps(value) - return text if len(text) <= limit else f"{text[: limit - 3]}..." - - def _read_chunk_size(size: JSON, span: int | None, unit: int) -> tuple[int, str | None] | None: """Read one entry of a stored regular chunk shape as a chunk edge length. - Returns the edge length and, for an invalid entry, how it was read; `None` if the - entry cannot be read, which leaves it for the metadata constructors to reject. A - JSON int >= 1 is kept, JSON `true` is read as 1, and 0 or JSON `false` is read as - one chunk spanning the axis of length `span`, a multiple of `unit` (the inner chunk - size of a shard). `span` is `None` where no stored 0 is known, as in the inner - chunk shape of a sharding codec: 0 is then left for the constructors to reject. + Returns the edge length and, where the user must act on how it was read, how it was + read; `None` if the entry cannot be read, which leaves it for the metadata + constructors to check. A JSON int >= 1 is kept, JSON `true` is read as 1, and 0 or + JSON `false` is read as one chunk spanning the axis of length `span`, a multiple of + `unit` (the inner chunk size of a shard): on an axis of positive length no chunk + can have been stored under it, so the array holds only its fill value. `span` is + `None` where no stored 0 is known, as in the inner chunk shape of a sharding codec: + 0 is then left as stored. """ match size: case True: - return 1, "1" + return 1, None case int() if size >= 1: return size, None case int() if size == 0 and span is not None: edge = full_span_chunk_size(span, unit) - how = f"one chunk spanning the dimension ({edge})" - if span > 0: - how += ( - ", and as no chunk can be stored under a chunk size of 0, the array " - "holds only its fill value" - ) - return edge, how + if span == 0: + return edge, None + return edge, ( + f"one chunk spanning the dimension ({edge}), and as no chunk can be stored " + "under a chunk size of 0, the array holds only its fill value" + ) return None def _read_chunk_shape( - stored: JSON, spans: Sequence[int | None], units: Iterable[int], name: str + stored: JSON, spans: Sequence[int | None], units: Iterable[int] = () ) -> tuple[list[int], str | None] | None: """Read a stored regular chunk shape, entry by entry (see `_read_chunk_size`), for axes of lengths `spans` whose chunks are multiples of `units` (1 where not given). - Returns the chunk shape and, if an entry is invalid, a sentence saying how the - `name` was read; `None` if it cannot be read. + Returns the chunk shape and, where the user must act on how an entry was read, a + sentence saying how the chunk shape was read; `None` if it cannot be read. """ if not (isinstance(stored, list) and len(stored) == len(spans)): return None @@ -115,61 +117,68 @@ def _read_chunk_shape( if not readings: return edges, None return edges, ( - f"The stored {name} {json.dumps(list(stored))} is invalid: chunk sizes must be " + f"The stored chunk shape {json.dumps(stored)} is invalid: chunk sizes must be " f"integers of at least 1. It is read as {edges}, reading {'; '.join(readings)}." ) -def _invalid_chunk_sizes_v2(doc: ArrayDocument) -> tuple[ArrayDocument, str] | None: +def _invalid_chunk_sizes_v2(doc: ArrayDocument) -> tuple[ArrayDocument, str | None] | None: shape = doc.get("shape") if not _is_int_list(shape): return None - match _read_chunk_shape(doc.get("chunks"), shape, (), "chunk shape"): - case chunks, str(reading): + stored = doc.get("chunks") + match _read_chunk_shape(stored, shape): + case chunks, reading if not json_equal(chunks, stored): return {**doc, "chunks": chunks}, reading return None -def _sharding_codec(doc: ArrayDocument) -> tuple[Sequence[JSON], int, Mapping[str, JSON]] | None: - """The codec list of a Zarr format 3 array document, with the position and the - configuration of its sharding codec, if it has one.""" - codecs = doc.get("codecs") - if isinstance(codecs, list): - for index, codec in enumerate(codecs): - if isinstance(codec, Mapping) and codec.get("name") == "sharding_indexed": - configuration = codec.get("configuration") - if isinstance(configuration, Mapping): - return codecs, index, configuration - return None - - -def _read_inner_chunk_shape(doc: ArrayDocument) -> tuple[list[int], str | None] | None: - """Read the inner chunk shape of a sharded array. No stored inner chunk size of 0 - or `false` is known, so the spans of its axes are not given.""" - shape = doc.get("shape") - sharding = _sharding_codec(doc) - if sharding is None or not _is_int_list(shape): +def _read_codec(codec: JSON) -> JSON: + """Read a stored codec: the inner chunk shape of a sharding codec is read as a chunk + shape with no known axis lengths (see `_read_chunk_shape`), and so are those of the + sharding codecs nested in its codecs.""" + if not (isinstance(codec, Mapping) and codec.get("name") == "sharding_indexed"): + return codec + configuration = codec.get("configuration") + if not isinstance(configuration, Mapping): + return codec + upgraded = dict(configuration) + stored = configuration.get("chunk_shape") + if isinstance(stored, list): + match _read_chunk_shape(stored, [None] * len(stored)): + case chunk_shape, _: + upgraded["chunk_shape"] = list(chunk_shape) + if isinstance(codecs := configuration.get("codecs"), list): + upgraded["codecs"] = [_read_codec(inner) for inner in codecs] + return {**codec, "configuration": upgraded} + + +def _invalid_inner_chunk_sizes_v3(doc: ArrayDocument) -> tuple[ArrayDocument, str | None] | None: + stored = doc.get("codecs") + if not isinstance(stored, list): return None - _, _, configuration = sharding - return _read_chunk_shape( - configuration.get("chunk_shape"), - [None] * len(shape), - (), - "inner chunk shape of the sharding codec", - ) + codecs = [_read_codec(codec) for codec in stored] + if json_equal(codecs, stored): + return None + return {**doc, "codecs": codecs}, None -def _invalid_inner_chunk_sizes_v3(doc: ArrayDocument) -> tuple[ArrayDocument, str] | None: - match _read_inner_chunk_shape(doc), _sharding_codec(doc): - case (inner, str(reading)), (codecs, index, configuration): - # `_sharding_codec` found a mapping at `index`. - codec = cast("Mapping[str, JSON]", codecs[index]) - upgraded = {**codec, "configuration": {**configuration, "chunk_shape": inner}} - return {**doc, "codecs": [*codecs[:index], upgraded, *codecs[index + 1 :]]}, reading - return None +def _inner_chunk_shape(doc: ArrayDocument) -> list[int]: + """The inner chunk shape of the sharding codec of a Zarr format 3 array document, if + it has one and that is a list of integers; else `[]`.""" + codecs = doc.get("codecs") + if isinstance(codecs, list): + for codec in codecs: + match codec: + case { + "name": "sharding_indexed", + "configuration": {"chunk_shape": list() as inner}, + }: + return inner if _is_int_list(inner) else [] + return [] -def _invalid_chunk_sizes_v3(doc: ArrayDocument) -> tuple[ArrayDocument, str] | None: +def _invalid_chunk_sizes_v3(doc: ArrayDocument) -> tuple[ArrayDocument, str | None] | None: grid = doc.get("chunk_grid") shape = doc.get("shape") if not (isinstance(grid, Mapping) and grid.get("name") == "regular" and _is_int_list(shape)): @@ -177,20 +186,21 @@ def _invalid_chunk_sizes_v3(doc: ArrayDocument) -> tuple[ArrayDocument, str] | N configuration = grid.get("configuration") if not isinstance(configuration, Mapping): return None - inner = _read_inner_chunk_shape(doc) - units = () if inner is None else inner[0] - match _read_chunk_shape(configuration.get("chunk_shape"), shape, units, "chunk shape"): - case chunk_shape, str(reading): + stored = configuration.get("chunk_shape") + match _read_chunk_shape(stored, shape, _inner_chunk_shape(doc)): + case chunk_shape, reading if not json_equal(chunk_shape, stored): upgraded = {**configuration, "chunk_shape": chunk_shape} return {**doc, "chunk_grid": {**grid, "configuration": upgraded}}, reading return None def _read_edge_length(edge: JSON) -> JSON: - """Read one stored chunk edge length of a rectilinear chunk grid: an integral JSON - float of at least 1 (`4.0`) is read as the `int` it equals. Anything else is kept, - for the metadata constructors to check.""" + """Read one stored chunk edge length of a rectilinear chunk grid: JSON `true` is + read as 1 and an integral JSON float of at least 1 (`4.0`) as the `int` it equals. + Anything else is kept, for the metadata constructors to check.""" match edge: + case True: + return 1 case float() if edge.is_integer() and edge >= 1: return int(edge) return edge @@ -200,7 +210,7 @@ def _read_rectilinear_axis(spec: JSON) -> JSON: """Read the stored chunk edge lengths of one axis of a rectilinear chunk grid (see `_read_edge_length`): its explicit edges and the sizes of its run-length encoded `[size, count]` pairs. A bare chunk size and a repeat count are kept as stored: no - writer stored them as floats.""" + writer stored them as `true` or as floats.""" match spec: case list(): return [_read_rectilinear_entry(entry) for entry in spec] @@ -214,7 +224,7 @@ def _read_rectilinear_entry(entry: JSON) -> JSON: return _read_edge_length(entry) -def _float_edge_lengths_v3(doc: ArrayDocument) -> tuple[ArrayDocument, str] | None: +def _invalid_edge_lengths_v3(doc: ArrayDocument) -> tuple[ArrayDocument, str | None] | None: grid = doc.get("chunk_grid") if not (isinstance(grid, Mapping) and grid.get("name") == "rectilinear"): return None @@ -225,35 +235,32 @@ def _float_edge_lengths_v3(doc: ArrayDocument) -> tuple[ArrayDocument, str] | No if not isinstance(stored, list): return None read = [_read_rectilinear_axis(axis) for axis in stored] - # An integral float and the `int` it equals compare equal, but not as JSON text. - axes = [axis for axis, spec in enumerate(stored) if json.dumps(spec) != json.dumps(read[axis])] - if not axes: + if json_equal(read, stored): return None - reading = ( - f"The stored chunk edge lengths {_abbreviate(stored)} are invalid: chunk edge " - f"lengths must be integers. They are read as {_abbreviate(read)}, reading each " - f"edge length stored as a float, in dimensions {axes}, as the integer it equals." - ) upgraded = {**configuration, "chunk_shapes": read} - return {**doc, "chunk_grid": {**grid, "configuration": upgraded}}, reading + return {**doc, "chunk_grid": {**grid, "configuration": upgraded}}, None ARRAY_UPGRADES: Final[Mapping[ZarrFormat, tuple[Upgrade, ...]]] = { 2: (_invalid_chunk_sizes_v2,), - 3: (_invalid_inner_chunk_sizes_v3, _invalid_chunk_sizes_v3, _float_edge_lengths_v3), + # The inner chunk shape is read first: it gives the unit of the outer chunk shape. + # The rectilinear edge lengths are read last, after any upgrade that yields a + # rectilinear chunk grid. + 3: (_invalid_inner_chunk_sizes_v3, _invalid_chunk_sizes_v3, _invalid_edge_lengths_v3), } """The upgrades of an array document of each Zarr format, applied in order.""" def upgrade_array_document( doc: ArrayDocument, zarr_format: ZarrFormat -) -> tuple[ArrayDocument, list[str]]: +) -> tuple[ArrayDocument, list[str | None]]: """Apply the upgrades for `zarr_format` to a stored array metadata document. - Returns the upgraded document and the readings of the upgrades that changed it, - for `mark_upgraded` once the document has been validated. + Returns the upgraded document and the reading of each upgrade that changed it (a + warning, or `None` for a silent one), for `mark_upgraded` once the document has + been validated. """ - readings: list[str] = [] + readings: list[str | None] = [] for upgrade in ARRAY_UPGRADES[zarr_format]: upgraded = upgrade(doc) if upgraded is not None: diff --git a/src/zarr/core/metadata/v2.py b/src/zarr/core/metadata/v2.py index 4adae2834e..7df14a513f 100644 --- a/src/zarr/core/metadata/v2.py +++ b/src/zarr/core/metadata/v2.py @@ -154,7 +154,8 @@ def to_buffer_dict(self, prototype: BufferPrototype) -> dict[str, Buffer]: @classmethod def from_dict(cls, data: dict[str, Any], *, path: str | None = None) -> ArrayV2Metadata: """Read a stored `.zarray` document (with its attributes). An invalid document - that `zarr.core.metadata.upgrades` can read warns, naming the array at `path`.""" + that `zarr.core.metadata.upgrades` can read is read as upgraded; a reading the user + must act on warns, naming the array at `path`.""" upgraded, readings = upgrade_array_document(data, 2) # a new dict, because we are modifying it _data: dict[str, Any] = dict(upgraded) diff --git a/src/zarr/core/metadata/v3.py b/src/zarr/core/metadata/v3.py index 7231d9c51c..ad16afc3e1 100644 --- a/src/zarr/core/metadata/v3.py +++ b/src/zarr/core/metadata/v3.py @@ -614,7 +614,8 @@ def to_buffer_dict(self, prototype: BufferPrototype) -> dict[str, Buffer]: @classmethod def from_dict(cls, data: dict[str, JSON], *, path: str | None = None) -> Self: """Read a stored `zarr.json` array document. An invalid document that - `zarr.core.metadata.upgrades` can read warns, naming the array at `path`.""" + `zarr.core.metadata.upgrades` can read is read as upgraded; a reading the user + must act on warns, naming the array at `path`.""" upgraded, readings = upgrade_array_document(data, 3) # a new dict, because we are modifying it _data = dict(upgraded) diff --git a/tests/test_metadata/test_upgrades.py b/tests/test_metadata/test_upgrades.py index f2da059b16..89dd1d236c 100644 --- a/tests/test_metadata/test_upgrades.py +++ b/tests/test_metadata/test_upgrades.py @@ -23,6 +23,7 @@ from zarr.core.sync import sync from zarr.dtype import Int16 from zarr.errors import ZarrUserWarning +from zarr.storage import MemoryStore, StorePath from zarr.storage._common import make_store_path if TYPE_CHECKING: @@ -83,66 +84,73 @@ def _stored_chunks(doc: dict[str, Any]) -> Any: ) -def _chunk_shapes(metadata: ArrayV2Metadata | ArrayV3Metadata) -> tuple[Any, Any]: - """The chunk shape of `metadata` and, if it is sharded, its inner chunk shape.""" +def _chunk_shapes(metadata: ArrayV2Metadata | ArrayV3Metadata) -> tuple[Any, ...]: + """The chunk shape of `metadata`, then the inner chunk shape of each sharding codec, + from the outermost in.""" if isinstance(metadata, ArrayV2Metadata): - return metadata.chunks, None + return (metadata.chunks,) assert isinstance(metadata.chunk_grid, RegularChunkGridMetadata) - inner = next((c.chunk_shape for c in metadata.codecs if isinstance(c, ShardingCodec)), None) - return metadata.chunk_grid.chunk_shape, inner + shapes: list[Any] = [metadata.chunk_grid.chunk_shape] + codecs: tuple[Any, ...] = metadata.codecs + while sharding := next((c for c in codecs if isinstance(c, ShardingCodec)), None): + shapes.append(sharding.chunk_shape) + codecs = sharding.codecs + return tuple(shapes) + + +def _nested_sharded_doc(inner: list[Any], nested: list[Any]) -> dict[str, JSON]: + """A Zarr format 3 document of shape `[8]` in one shard of chunk shape `inner`, whose + codecs shard each chunk again, in chunks of shape `nested`.""" + doc = _v3_doc([8], [8], inner=inner) + outer = cast("dict[str, Any]", cast("list[JSON]", doc["codecs"])[0]) + configuration = outer["configuration"] + configuration["codecs"] = [{**outer, "configuration": {**configuration, "chunk_shape": nested}}] + return doc @pytest.mark.parametrize( - ("doc", "expected", "warning"), + ("doc", "expected", "upgraded", "warning"), [ - (_v2_doc([10, 10], [4, 5]), ((4, 5), None), None), - (_v3_doc([0, 0], [1, 1]), ((1, 1), None), None), - (_v3_doc([10], [4], inner=[2]), ((4,), (2,)), None), - ( - _v2_doc([0, 4], [0, 4]), - ((1, 4), None), - r"0 in dimension 0 as one chunk spanning the dimension \(1\)\.", - ), - ( - _v3_doc([0], [False]), - ((1,), None), - r"false in dimension 0 as one chunk spanning the dimension \(1\)\.", - ), - (_v2_doc([5], [True]), ((1,), None), "true in dimension 0 as 1"), - ( - _v3_doc([5, 4], [True, 4]), - ((1, 4), None), - r"\[true, 4\] is invalid.*true in dimension 0 as 1", - ), + (_v2_doc([10, 10], [4, 5]), ((4, 5),), False, None), + (_v3_doc([0, 0], [1, 1]), ((1, 1),), False, None), + (_v3_doc([10], [4], inner=[2]), ((4,), (2,)), False, None), + (_v2_doc([0, 4], [0, 4]), ((1, 4),), True, None), + (_v3_doc([0], [False]), ((1,),), True, None), + (_v2_doc([5], [True]), ((1,),), True, None), + (_v3_doc([5, 4], [True, 4]), ((1, 4),), True, None), ( _v2_doc([3], [0]), - ((3,), None), - r"spanning the dimension \(3\), and .* holds only its fill value", + ((3,),), + True, + ( + r"^The stored chunk shape \[0\] is invalid: .* read as \[3\], reading 0 in " + r"dimension 0 as one chunk spanning the dimension \(3\), and .* holds only " + r"its fill value\.$" + ), ), ( _v3_doc([4, 3], [4, 0]), - ((4, 3), None), - "0 in dimension 1 as .* holds only its fill value", + ((4, 3),), + True, + r"reading 0 in dimension 1 as .* holds only its fill value\.$", ), - (_v3_doc([0], [0], inner=[4]), ((4,), (4,)), r"spanning the dimension \(4\)\."), + ( + _v2_doc([0, 3], [0, 0]), + ((1, 3),), + True, + r"read as \[1, 3\], reading 0 in dimension 1 as .* holds only its fill value\.$", + ), + (_v3_doc([0], [0], inner=[4]), ((4,), (4,)), True, None), ( _v3_doc([10], [0], inner=[4]), ((12,), (4,)), + True, r"spanning the dimension \(12\), and .* holds only its fill value", ), - ( - _v3_doc([0, 3], [0, 3], inner=[2, 3]), - ((2, 3), (2, 3)), - r"spanning the dimension \(2\)\.", - ), - ( - _v3_doc([5], [True], inner=[True]), - ((1,), (1,)), - ( - r"^The stored inner chunk shape of the sharding codec \[true\] is invalid: .* " - r"The stored chunk shape \[true\] is invalid: " - ), - ), + (_v3_doc([0, 3], [0, 3], inner=[2, 3]), ((2, 3), (2, 3)), True, None), + (_v3_doc([5], [True], inner=[True]), ((1,), (1,)), True, None), + (_nested_sharded_doc([4], [2]), ((8,), (4,), (2,)), False, None), + (_nested_sharded_doc([4], [True]), ((8,), (4,), (1,)), True, None), ], ids=[ "v2-valid", @@ -154,42 +162,48 @@ def _chunk_shapes(metadata: ArrayV2Metadata | ArrayV3Metadata) -> tuple[Any, Any "v3-true", "v2-zero-grown-axis", "v3-zero-grown-axis", + "v2-zero-empty-and-grown-axes", "v3-sharded-zero-empty-axis", "v3-sharded-zero-grown-axis", "v3-sharded-zero-2d", "v3-sharded-true-inner-and-outer", + "v3-nested-sharded-valid", + "v3-nested-sharded-true", ], ) def test_upgrade_array_document( - doc: dict[str, JSON], expected: tuple[Any, Any], warning: str | None + doc: dict[str, JSON], expected: tuple[Any, ...], upgraded: bool, warning: str | None ) -> None: - """Valid documents pass unchanged and silently. A stored chunk size of 0 or `false` - is read as one chunk spanning the axis (a multiple of the inner chunk when sharded) - and `true` as 1, in the chunk shape and in a sharding codec's inner chunk shape; - `from_dict` warns once for the document, naming the array, saying how each part was - read (and that the array holds only its fill value where a chunk size of 0 was - stored for a non-empty axis) and how to re-save.""" - upgraded, readings = upgrade_array_document(doc, cast("ZarrFormat", doc["zarr_format"])) - assert {k: v for k, v in upgraded.items() if k not in ("chunks", "chunk_grid", "codecs")} == { - k: v for k, v in doc.items() if k not in ("chunks", "chunk_grid", "codecs") - } + """Valid documents pass unchanged. A stored chunk size of 0 or `false` is read as one + chunk spanning the axis (a multiple of the inner chunk when sharded) and `true` as 1, + in the chunk shape and in the inner chunk shape of every sharding codec, nested or + not. `from_dict` marks the metadata of an upgraded document; it warns once, naming + the array, only where a chunk size of 0 was stored for a non-empty axis (which then + holds only its fill value), saying how that part was read and how to re-save. The + other readings give what zarr read before, so they are silent.""" + upgraded_doc, readings = upgrade_array_document(doc, cast("ZarrFormat", doc["zarr_format"])) + assert { + k: v for k, v in upgraded_doc.items() if k not in ("chunks", "chunk_grid", "codecs") + } == {k: v for k, v in doc.items() if k not in ("chunks", "chunk_grid", "codecs")} + assert bool(readings) is upgraded + if not upgraded: + assert upgraded_doc is doc metadata_cls = ArrayV2Metadata if doc["zarr_format"] == 2 else ArrayV3Metadata with warnings.catch_warnings(record=True) as record: warnings.simplefilter("always") metadata = metadata_cls.from_dict(dict(doc), path="group/array") assert _chunk_shapes(metadata) == expected + assert metadata._stored_document_upgraded is upgraded messages = [str(w.message) for w in record] if warning is None: - assert upgraded is doc - assert readings == [] assert messages == [] else: - assert len(messages) == 1 - message = messages[0] + [message] = messages assert message.startswith("Array 'group/array': ") - assert re.search(warning, message.removeprefix("Array 'group/array': ")) - assert ("holds only its fill value" in message) == ("fill value" in warning) assert message.endswith(RESAVE_HINT) + assert re.search( + warning, message.removeprefix("Array 'group/array': ").removesuffix(f" {RESAVE_HINT}") + ) @pytest.mark.parametrize( @@ -219,6 +233,15 @@ def _read_strictly(doc: dict[str, JSON]) -> ArrayV2Metadata | ArrayV3Metadata: return metadata_cls.from_dict(doc) +def _open_strictly(path: Path, mode: Literal["r", "a", "r+"] = "r") -> AnyArray: + """Open the array at `path`, failing on any warning that its document was upgraded.""" + with warnings.catch_warnings(): + warnings.simplefilter("error", ZarrUserWarning) + array = zarr.open_array(store=path, mode=mode) + assert isinstance(array, zarr.Array) + return array + + def test_stored_negative_chunk_size_rejected() -> None: """No known writer stored a negative chunk size: it is rejected, not upgraded.""" with pytest.raises(ValueError, match="^Expected all values to be non-negative"): @@ -232,14 +255,6 @@ def test_stored_chunk_shape_ndim_mismatch_rejected() -> None: _read_strictly(_v3_doc([4, 4], [0])) -def test_stored_fractional_chunk_size_rejected() -> None: - """A stored chunk size that is not an integral number is rejected, not upgraded.""" - with pytest.raises( - TypeError, match=r"^Dimension 0: chunk edge length must be an integer, got 4\.5$" - ): - _read_strictly(_v3_doc([4], [4.5])) - - def _rectilinear_doc(shape: list[int], chunk_shapes: list[Any]) -> dict[str, JSON]: return _v3_doc(shape, [1] * len(shape)) | { "chunk_grid": { @@ -250,56 +265,48 @@ def _rectilinear_doc(shape: list[int], chunk_shapes: list[Any]) -> dict[str, JSO @pytest.mark.parametrize( - ("chunk_shapes", "expected", "warning"), + ("chunk_shapes", "expected", "upgraded"), [ - ([[[4, 2]], [[5, 2]]], ((4, 4), (5, 5)), None), - ( - [[[4.0, 2]], [[5, 2]]], - ((4, 4), (5, 5)), - ( - r"^The stored chunk edge lengths \[\[\[4\.0, 2\]\], \[\[5, 2\]\]\] are " - r"invalid: .* read as \[\[\[4, 2\]\], \[\[5, 2\]\]\], .* in dimensions \[0\]," - ), - ), - ([[3.0, 5.0], [[5, 2]]], ((3, 5), (5, 5)), r"read as \[\[3, 5\], "), - ([[[4.0, 2], 2.0], [[5, 2]]], ((4, 4, 2), (5, 5)), r"read as \[\[\[4, 2\], 2\], "), - ([[[4.0, 2]], [10]], ((4, 4), (10,)), r"read as \[\[\[4, 2\]\], \[10\]\]"), - ([[[4.0, 2]], [5.0, 5]], ((4, 4), (5, 5)), r"in dimensions \[0, 1\],"), + ([[[4, 2]], [[5, 2]]], ((4, 4), (5, 5)), False), + ([[[4.0, 2]], [[5, 2]]], ((4, 4), (5, 5)), True), + ([[3.0, 5.0], [[5, 2]]], ((3, 5), (5, 5)), True), + ([[[4.0, 2], 2.0], [[5, 2]]], ((4, 4, 2), (5, 5)), True), + ([[[4.0, 2]], [10]], ((4, 4), (10,)), True), + ([[[4.0, 2]], [5.0, 5]], ((4, 4), (5, 5)), True), + ([[True, 4], [[5, 2]]], ((1, 4), (5, 5)), True), + ([[[True, 2], 3], [[5, 2]]], ((1, 1, 3), (5, 5)), True), + ], + ids=[ + "valid", + "float-rle-size", + "float-edges", + "float-rle-size-and-edge", + "float-sharded-outer", + "float-2d", + "true-edge", + "true-rle-size", ], - ids=["valid", "rle-size", "edges", "rle-size-and-edge", "sharded-outer", "2d"], ) -def test_read_float_edges_in_rectilinear_grid( - chunk_shapes: list[Any], - expected: tuple[tuple[int, ...], ...], - warning: str | None, +def test_read_invalid_edges_in_rectilinear_grid( + chunk_shapes: list[Any], expected: tuple[tuple[int, ...], ...], upgraded: bool ) -> None: """A stored rectilinear chunk grid whose explicit edges or run-length encoded sizes - are integral floats, as zarr-python wrote them when given float edges, is read with - those edges as `int`s; `from_dict` warns once, naming the array and the dimensions, - and how to re-save.""" + are integral floats or JSON `true`, as zarr-python wrote them when given float or + `True` edges, is read with those edges as the `int`s they equal. `from_dict` marks + the metadata as upgraded, silently: zarr read these edges so before.""" shape = [sum(edges) for edges in expected] doc = _rectilinear_doc(shape, chunk_shapes) - with ( - zarr.config.set({"array.rectilinear_chunks": True}), - warnings.catch_warnings(record=True) as record, - ): - warnings.simplefilter("always") - metadata = ArrayV3Metadata.from_dict(doc, path="group/array") + with zarr.config.set({"array.rectilinear_chunks": True}): + metadata = _read_strictly(doc) assert metadata.chunk_grid == RectilinearChunkGridMetadata(chunk_shapes=expected) - messages = [str(w.message) for w in record] - if warning is None: - assert messages == [] - else: - [message] = messages - assert message.startswith("Array 'group/array': ") - assert re.search(warning, message.removeprefix("Array 'group/array': ")) - assert message.endswith(RESAVE_HINT) + assert metadata._stored_document_upgraded is upgraded @pytest.mark.parametrize( ("doc", "error"), [ - (_v3_doc([20], [10.0]), "Dimension 0: chunk edge length must be an integer, got 10.0"), + (_v3_doc([20], [10.0]), "Dimension 0: chunk edge length must be an int, got 10.0"), + (_v3_doc([4], [4.5]), "Dimension 0: chunk edge length must be an int, got 4.5"), ( _v3_doc([8], [4], inner=[2.0]), "Expected an iterable of integers. Got [2.0] instead.", @@ -307,47 +314,64 @@ def test_read_float_edges_in_rectilinear_grid( (_v2_doc([20], [10.0]), "Expected an iterable of integers. Got [10.0] instead."), ( _rectilinear_doc([8], [[[4, 2.0]]]), - "Dimension 0: RLE repeat count must be an integer, got 2.0", + "Dimension 0: RLE repeat count must be an int, got 2.0", ), ( _rectilinear_doc([8], [4.0]), - "Dimension 0: chunk edge length must be an integer, got 4.0", + "Dimension 0: chunk edge length must be an int, got 4.0", + ), + ( + _rectilinear_doc([8], [[0.0, 8]]), + "Dimension 0: chunk edge length must be an int, got 0.0", ), ], - ids=["regular", "sharding-inner", "v2", "rle-count", "rectilinear-bare"], + ids=[ + "regular", + "regular-fractional", + "sharding-inner", + "v2", + "rle-count", + "rectilinear-bare", + "rectilinear-zero", + ], ) def test_stored_float_chunk_size_rejected(doc: dict[str, JSON], error: str) -> None: - """A float chunk size is read only where zarr-python stored one, as the edges of a - rectilinear chunk grid. Anywhere else it is rejected, as zarr 3.4.0 rejected a - stored regular chunk size of `10.0`.""" + """A float chunk size is read only where zarr-python stored one, as an edge of at + least 1 of a rectilinear chunk grid. Anywhere else it is rejected, as zarr 3.4.0 + rejected a stored regular chunk size of `10.0`.""" with zarr.config.set({"array.rectilinear_chunks": True}), pytest.raises(TypeError) as info: _read_strictly(doc) assert info.match(re.escape(error)) -def test_float_edges_round_trip(tmp_path: Path) -> None: - """A store whose rectilinear chunk grid holds the float edges zarr-python wrote opens - with a warning, reads its data and re-saves the edges as `int`s.""" - path = tmp_path / "float.zarr" +@pytest.mark.parametrize( + ("chunks", "stored", "resaved"), + [ + ([[4, 4], [5, 5]], [[[4.0, 2]], [[5, 2]]], "[[[4, 2]], [[5, 2]]]"), + ([[1, 3, 4], [5, 5]], [[True, 3, 4], [[5, 2]]], "[[1, 3, 4], [[5, 2]]]"), + ], + ids=["float", "true"], +) +def test_invalid_edges_round_trip( + tmp_path: Path, chunks: list[list[int]], stored: list[Any], resaved: str +) -> None: + """A store whose rectilinear chunk grid holds the float or `true` edges zarr-python + wrote opens silently, reads its data, and stores its edges as `int`s before the + first write.""" + path = tmp_path / "rectilinear.zarr" data = np.arange(80, dtype="int16").reshape(8, 10) with zarr.config.set({"array.rectilinear_chunks": True}): - zarr.create_array(path, shape=data.shape, chunks=[[4, 4], [5, 5]], dtype="int16")[...] = ( - data + zarr.create_array(path, shape=data.shape, chunks=chunks, dtype="int16")[...] = data + _rewrite_doc( + path, 3, lambda doc: doc["chunk_grid"]["configuration"].update(chunk_shapes=stored) ) - - def store_floats(doc: dict[str, Any]) -> None: - doc["chunk_grid"]["configuration"]["chunk_shapes"] = [[[4.0, 2]], [[5, 2]]] - - _rewrite_doc(path, 3, store_floats) - with pytest.warns(ZarrUserWarning, match=r"read as \[\[\[4, 2\]\], \[\[5, 2\]\]\]"): - arr = zarr.open_array(path, mode="a") + arr = _open_strictly(path, mode="a") np.testing.assert_array_equal(arr[...], data) - arr.update_attributes({}) - stored = json.loads((path / "zarr.json").read_text())["chunk_grid"]["configuration"] - assert json.dumps(stored["chunk_shapes"]) == "[[[4, 2]], [[5, 2]]]" - with warnings.catch_warnings(): - warnings.simplefilter("error", ZarrUserWarning) - np.testing.assert_array_equal(zarr.open_array(path)[...], data) + arr[0, 0] = -1 + written = json.loads((path / "zarr.json").read_text())["chunk_grid"]["configuration"] + assert json.dumps(written["chunk_shapes"]) == resaved + data[0, 0] = -1 + np.testing.assert_array_equal(_open_strictly(path)[...], data) def test_v2_constructor_rejects_chunks_of_wrong_length() -> None: @@ -394,12 +418,12 @@ def _second_edge(grid: RectilinearChunkGridMetadata) -> int: @pytest.mark.parametrize("site", CHUNK_EDGE_SITES) @pytest.mark.parametrize( ("size", "expected"), - [(4, 4), (True, 1), (np.int64(4), 4)], - ids=["int", "bool", "numpy-int"], + [(4, 4), (True, 1)], + ids=["int", "bool"], ) def test_metadata_reads_integer_chunk_edge(site: str, size: object, expected: int) -> None: - """Chunk grid metadata reads an integer of any integer type as the `int` chunk edge - length it equals.""" + """Chunk grid metadata reads an `int` or a `bool` as the `int` chunk edge length it + equals.""" edge = CHUNK_EDGE_SITES[site](size) assert type(edge) is int assert edge == expected @@ -408,15 +432,16 @@ def test_metadata_reads_integer_chunk_edge(site: str, size: object, expected: in @pytest.mark.parametrize("site", CHUNK_EDGE_SITES) @pytest.mark.parametrize( "size", - [4.0, np.float64(4.0), 4.5, float("inf"), "4", None], - ids=["float", "numpy-float", "fractional", "inf", "str", "none"], + [4.0, np.float64(4.0), 4.5, float("inf"), "4", None, np.int64(4)], + ids=["float", "numpy-float", "fractional", "inf", "str", "none", "numpy-int"], ) def test_metadata_rejects_non_integer_chunk_edge(site: str, size: object) -> None: - """A float is not a chunk edge length in metadata built in code, even an integral - one; stored documents with integral floats are read by the upgrades.""" + """A chunk edge length in metadata built in code is an `int`: a float is rejected, + even an integral one (stored documents with integral floats are read by the + upgrades), and so is a NumPy integer, as zarr 3.4.0 rejected one.""" with pytest.raises( TypeError, - match=re.escape(f"Dimension 0: chunk edge length must be an integer, got {size!r}"), + match=re.escape(f"Dimension 0: chunk edge length must be an int, got {size!r}"), ): CHUNK_EDGE_SITES[site](size) @@ -434,8 +459,10 @@ def test_metadata_rejects_chunk_edge_below_one(site: str, size: int) -> None: CHUNK_EDGE_SITES[site](size) -@pytest.mark.parametrize("chunk_shape", [4, np.int64(4), None]) -def test_regular_chunk_grid_rejects_chunk_shape_not_iterable(chunk_shape: Any) -> None: +@pytest.mark.parametrize("chunk_shape", [4, np.int64(4), None, "44", {"4": 4}]) +def test_regular_chunk_grid_rejects_chunk_shape_not_a_sequence(chunk_shape: Any) -> None: + """A regular chunk shape is an iterable of chunk edge lengths, but not a string or a + mapping, which is rejected as a whole, not entry by entry.""" with pytest.raises( TypeError, match=re.escape( @@ -509,11 +536,11 @@ def _rewrite_doc(path: Path, zarr_format: Literal[2, 3], edit: Any) -> None: @pytest.mark.parametrize( - ("zarr_format", "shape", "stored", "inner", "expected"), + ("zarr_format", "shape", "stored", "inner", "expected", "warns"), [ - (2, (0, 4), [0, 4], None, (1, 4)), - (3, (5,), [True], None, (1,)), - (3, (10,), [0], (4,), (12,)), + (2, (0, 4), [0, 4], None, (1, 4), False), + (3, (5,), [True], None, (1,), False), + (3, (10,), [0], (4,), (12,), True), ], ids=["v2-empty-2d", "v3-true", "v3-sharded-grown"], ) @@ -524,9 +551,11 @@ def test_legacy_chunk_size_round_trip( stored: list[Any], inner: tuple[int, ...] | None, expected: tuple[int, ...], + warns: bool, ) -> None: - """A store whose metadata holds a chunk size written by older software opens with a - warning, reads and appends under the upgraded grid, and re-saves valid metadata.""" + """A store whose metadata holds a chunk size written by older software opens (with a + warning where a non-empty axis was stored with chunk size 0), reads and appends under + the upgraded grid, and re-saves valid metadata.""" path = tmp_path / "legacy.zarr" arr = zarr.create_array( store=path, @@ -548,8 +577,13 @@ def store_legacy(doc: dict[str, Any]) -> None: _rewrite_doc(path, zarr_format, store_legacy) - with pytest.warns(ZarrUserWarning, match=r"^Array '.*legacy\.zarr': .* is read as"): + with warnings.catch_warnings(record=True) as record: + warnings.simplefilter("always", ZarrUserWarning) arr = zarr.open_array(store=path, mode="a") + assert [ + re.match(r"^Array '.*legacy\.zarr': .* is read as", str(w.message)) is not None + for w in record + ] == [True] * warns assert (arr.shards or arr.chunks) == expected np.testing.assert_array_equal(arr[...], data) @@ -635,15 +669,6 @@ def _legacy_array(path: Path, zarr_format: Literal[2, 3]) -> None: ) -def _open_strictly(path: Path) -> AnyArray: - """Open the array at `path`, failing on any warning that its document was upgraded.""" - with warnings.catch_warnings(): - warnings.simplefilter("error", ZarrUserWarning) - array = zarr.open_array(store=path, mode="r") - assert isinstance(array, zarr.Array) - return array - - @pytest.mark.parametrize("zarr_format", [2, 3]) def test_stale_handle_write_keeps_newer_metadata( tmp_path: Path, zarr_format: Literal[2, 3] @@ -696,6 +721,132 @@ def test_stale_handle_write_keeps_valid_document_as_written( np.testing.assert_array_equal(_open_strictly(path)[...], [9, 0, 0]) +@pytest.mark.parametrize("zarr_format", [2, 3]) +def test_stale_handle_write_after_chunk_grid_change_raises( + tmp_path: Path, zarr_format: Literal[2, 3] +) -> None: + """If the document the store holds when a handle read from an upgraded document + first writes chunks lays out chunks differently from the handle's metadata (here the + array was resized by software that kept the stored chunk size of 0, which now reads + as a larger chunk), the handle's chunks would not be found under it: the write + raises and stores nothing.""" + path = tmp_path / "legacy.zarr" + _legacy_array(path, zarr_format) + with pytest.warns(ZarrUserWarning, match="is read as"): + stale = zarr.open_array(store=path, mode="r+") + _rewrite_doc(path, zarr_format, lambda doc: doc.update(shape=[10])) + documents = {p.name: p.read_bytes() for p in path.iterdir()} + + with pytest.raises(ValueError, match="has changed since this array was opened: reopen"): + stale[0:3] = [7, 8, 9] + + assert stale.metadata._stored_document_upgraded + assert {p.name: p.read_bytes() for p in path.iterdir()} == documents + + +@pytest.mark.parametrize("zarr_format", [2, 3]) +def test_write_without_stored_document(zarr_format: Literal[2, 3]) -> None: + """An array read from an upgraded document that no store holds (as + `AsyncArray.from_dict` builds one) writes its chunks as any array does: there is no + stored document to upgrade.""" + store = MemoryStore() + doc = _v2_doc([3], [True]) if zarr_format == 2 else _v3_doc([3], [True]) + array = zarr.Array(AsyncArray.from_dict(StorePath(store), doc)) + upgraded = array.metadata._stored_document_upgraded + + array[:] = [1, 2, 3] + + assert (upgraded, array.metadata._stored_document_upgraded) == (True, False) + np.testing.assert_array_equal(array[:], [1, 2, 3]) + assert not [key for key in store._store_dict if key.endswith((".zarray", "zarr.json"))] + + +@pytest.mark.parametrize(("shape", "expected"), [((0,), (1,)), ((3,), (3,))]) +def test_array_from_metadata_with_chunk_size_zero(shape: tuple[int], expected: tuple[int]) -> None: + """`ArrayV2Metadata` accepts a chunk size of 0, as a stored document may hold it. An + array built from such metadata reads it as the upgrades read that document, silently + (no data was read or written under it): `create_hierarchy` stores the metadata as + given and yields such an array, which stores the upgrade before its first write.""" + metadata = ArrayV2Metadata(shape=shape, chunks=(0,), dtype=Int16(), fill_value=0, order="C") + store = MemoryStore() + with warnings.catch_warnings(): + warnings.simplefilter("error", ZarrUserWarning) + nodes = dict(zarr.create_hierarchy(store=store, nodes={"a": metadata})) + array = nodes["a"] + assert isinstance(array, zarr.Array) + assert array.chunks == expected + assert json.loads(store._store_dict["a/.zarray"].to_bytes())["chunks"] == [0] + + array[...] = 1 + + # Writing an empty selection stores no chunks, so it stores no metadata either. + resaved = list(expected) if array.size else [0] + assert json.loads(store._store_dict["a/.zarray"].to_bytes())["chunks"] == resaved + np.testing.assert_array_equal(zarr.open_array(store, path="a")[...], np.ones(shape)) + + +def _store_zero(doc: dict[str, Any]) -> None: + _stored_chunks(doc)[0] = 0 + + +def _rewrite_consolidated( + path: Path, zarr_format: Literal[2, 3], name: str, edit: Callable[[dict[str, Any]], None] +) -> None: + """Edit the document of the member `name` in the consolidated metadata at `path`.""" + if zarr_format == 2: + document = json.loads((path / ".zmetadata").read_text()) + edit(document["metadata"][f"{name}/.zarray"]) + (path / ".zmetadata").write_text(json.dumps(document)) + else: + _rewrite_doc(path, 3, lambda doc: edit(doc["consolidated_metadata"]["metadata"][name])) + + +def _consolidated_member(path: Path, zarr_format: Literal[2, 3], name: str) -> Any: + if zarr_format == 2: + return json.loads((path / ".zmetadata").read_text())["metadata"][f"{name}/.zarray"] + return json.loads((path / "zarr.json").read_text())["consolidated_metadata"]["metadata"][name] + + +@pytest.mark.filterwarnings("ignore:Consolidated metadata is currently not part:UserWarning") +@pytest.mark.parametrize("zarr_format", [2, 3]) +@pytest.mark.parametrize("operation", ["attrs", "update_attributes_async", "delete-member"]) +def test_group_write_refreshes_upgraded_consolidated_member( + tmp_path: Path, zarr_format: Literal[2, 3], operation: str +) -> None: + """Storing a group's metadata stores its consolidated metadata. Each member of it read + from a document that had to be upgraded is first read again from the member's own + document as it now is (here resized by software that kept the stored chunk size of + 0), whose upgrade is stored, so no group write stores a stale copy as if valid.""" + path = tmp_path / "group.zarr" + group = zarr.open_group(path, mode="w", zarr_format=zarr_format) + group.create_array("a", shape=(3,), chunks=(3,), dtype="int16") + group.create_array("b", shape=(1,), chunks=(1,), dtype="int16") + zarr.consolidate_metadata(path) + _rewrite_consolidated(path, zarr_format, "a", _store_zero) + + def resize_keeping_zero(doc: dict[str, Any]) -> None: + _store_zero(doc) + doc["shape"] = [10] + + _rewrite_doc(path / "a", zarr_format, resize_keeping_zero) + with pytest.warns(ZarrUserWarning, match="is read as"): + group = zarr.open_group(path, mode="r+", use_consolidated=True) + + if operation == "attrs": + group.attrs["x"] = 1 + elif operation == "update_attributes_async": + sync(group.update_attributes_async({"x": 1})) + else: + del group["b"] + + assert _stored_chunks(_consolidated_member(path, zarr_format, "a")) == [10] + reopened = zarr.open_group(path, mode="r", use_consolidated=True) + member = reopened["a"] + assert isinstance(member, zarr.Array) + assert (member.shape, member.chunks) == ((10,), (10,)) + assert _open_strictly(path / "a").chunks == (10,) + + @pytest.mark.parametrize("zarr_format", [2, 3]) def test_empty_write_stores_no_metadata(tmp_path: Path, zarr_format: Literal[2, 3]) -> None: """A write of an empty selection stores no chunks, so it stores no metadata either.""" @@ -751,10 +902,10 @@ def test_legacy_chunk_size_consolidated(tmp_path: Path, zarr_format: Literal[2, group = zarr.open_group(path, mode="w", zarr_format=zarr_format) names = ("a", "b") for name in names: - group.create_array(name, shape=(0,), chunks=(1,), dtype="int32") + group.create_array(name, shape=(2,), chunks=(2,), dtype="int32") zarr.consolidate_metadata(path) - # What zarr-python wrote for `chunks=(0,)` on an empty array, in both copies. + # What zarr-python wrote for `chunks=(0,)`, in both copies. for name in names: if zarr_format == 2: _rewrite_doc(path / name, 2, lambda doc: doc.update(chunks=[0])) @@ -786,7 +937,7 @@ def test_legacy_chunk_size_consolidated(tmp_path: Path, zarr_format: Literal[2, ] for array in arrays: assert isinstance(array, zarr.Array) - assert array.chunks == (1,) + assert array.chunks == (2,) array.update_attributes({}) zarr.consolidate_metadata(path) with warnings.catch_warnings(): @@ -797,4 +948,4 @@ def test_legacy_chunk_size_consolidated(tmp_path: Path, zarr_format: Literal[2, for name in names: array = reopened[name] assert isinstance(array, zarr.Array) - assert array.chunks == (1,) + assert array.chunks == (2,) diff --git a/tests/test_unified_chunk_grid.py b/tests/test_unified_chunk_grid.py index 8df53df08b..6a43cb4f3b 100644 --- a/tests/test_unified_chunk_grid.py +++ b/tests/test_unified_chunk_grid.py @@ -8,6 +8,7 @@ from __future__ import annotations +import re from typing import TYPE_CHECKING, Any import numpy as np @@ -494,7 +495,7 @@ def test_chunk_grid_iter() -> None: [ ([[10, 3]], [10, 10, 10]), ([[10, 2], [20, 1]], [10, 10, 20]), - ([[True, 2], [np.int64(3), np.int64(1)]], [1, 1, 3]), + ([[True, 2], [3, 1]], [1, 1, 3]), ], ) def test_rle_expand(compressed: list[Any], expected: list[int]) -> None: @@ -551,12 +552,14 @@ def test_rle_expand_rejects_invalid(rle_input: list[Any], match: str) -> None: @pytest.mark.parametrize( ("rle_input", "match"), [ - ([10.5], "Chunk edge length must be an integer, got 10.5"), - ([10.0], "Chunk edge length must be an integer, got 10.0"), - ([[10.0, 3]], "Chunk edge length must be an integer, got 10.0"), - (["10"], "Chunk edge length must be an integer, got '10'"), - ([[10, 3.5]], "RLE repeat count must be an integer, got 3.5"), - ([[10, 3.0]], "RLE repeat count must be an integer, got 3.0"), + ([10.5], "Chunk edge length must be an int, got 10.5"), + ([10.0], "Chunk edge length must be an int, got 10.0"), + ([[10.0, 3]], "Chunk edge length must be an int, got 10.0"), + (["10"], "Chunk edge length must be an int, got '10'"), + ([[10, 3.5]], "RLE repeat count must be an int, got 3.5"), + ([[10, 3.0]], "RLE repeat count must be an int, got 3.0"), + ([np.int64(10)], "Chunk edge length must be an int, got np.int64(10)"), + ([[10, np.int64(3)]], "RLE repeat count must be an int, got np.int64(3)"), ], ids=[ "fractional-edge", @@ -565,11 +568,13 @@ def test_rle_expand_rejects_invalid(rle_input: list[Any], match: str) -> None: "string-edge", "fractional-count", "float-count", + "numpy-int-edge", + "numpy-int-count", ], ) def test_rle_expand_rejects_non_int(rle_input: list[Any], match: str) -> None: - """expand_rle reads integers only, not floats.""" - with pytest.raises(TypeError, match=match): + """expand_rle reads `int`s (and `bool`s) only, not floats or NumPy integers.""" + with pytest.raises(TypeError, match=re.escape(match)): expand_rle(rle_input) @@ -577,7 +582,7 @@ def test_rle_expand_rejects_non_int(rle_input: list[Any], match: str) -> None: ("rle_input", "match"), [ ([0], "chunk edge length must be >= 1"), - ([10.5], "chunk edge length must be an integer"), + ([10.5], "chunk edge length must be an int,"), ([[5, 0]], "RLE repeat count must be >= 1"), ([[5, 2, 1]], r"RLE entries must be an integer or \[size, count\]"), ], From f78e1152614582c9e56fd52be45cd2f15c2396c3 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Sat, 26 Sep 2026 14:35:32 +0200 Subject: [PATCH 53/76] test: expect the lifecycle machine's upgrade warning only on non-empty axes Assisted-by: ClaudeCode:claude-opus-5-5 Co-Authored-By: Claude Opus 5.5 --- tests/test_array_stateful.py | 6 +++++- 1 file changed, 5 insertions(+), 1 deletion(-) diff --git a/tests/test_array_stateful.py b/tests/test_array_stateful.py index 6dc03768b3..109d65eec1 100644 --- a/tests/test_array_stateful.py +++ b/tests/test_array_stateful.py @@ -167,8 +167,12 @@ def _open(self) -> zarr.Array[Any]: with warnings.catch_warnings(record=True) as record: warnings.simplefilter("always", ZarrUserWarning) arr = zarr.open_array(self.store, path=self.path, mode="r+") + # A stored chunk size of 0 is read silently on an empty axis; on a non-empty one + # it warns that the axis holds only the fill value. Either way it is upgraded. warned = any(issubclass(w.category, ZarrUserWarning) for w in record) - assert warned is bool(self.legacy_axes), [str(w.message) for w in record] + must_warn = any(self.shape[axis] > 0 for axis in self.legacy_axes) + assert warned is must_warn, [str(w.message) for w in record] + assert arr.metadata._stored_document_upgraded is bool(self.legacy_axes) return arr # ----------------------------------------------------------------- model From 820f083670c1af9e1766fb77c6f03491f3b70f3b Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Sat, 26 Sep 2026 14:52:00 +0200 Subject: [PATCH 54/76] fix(group): encode a group's consolidated metadata before storing member upgrades After the merge, storing group metadata refreshed each upgraded consolidated member and stored its upgrade before encoding the group, and `Group.delitem` (which encodes before deleting) skipped the refresh. `encode_node` refreshes the consolidated members (reading only) and encodes the group, so the rectilinear gate runs for every member before anything is stored; `store_node` then stores the group's documents and the members' upgrades. `save_metadata` and `Group.delitem` both use the pair, so a group write refused without the flag stores no member upgrade either. Tests: group attribute writes join the store-untouched table; a refused group write stores no other member's upgrade; with the flag, deleting a member or setting a group attribute stores the rectilinear grid in the member and consolidated documents. Error-message expectations follow the patch release's integer rule wording, and the mixed-grid rows expect the JSON `true` and float edge readings to be silent. Assisted-by: ClaudeCode:claude-opus-5-5 Co-Authored-By: Claude Opus 5.5 --- changes/4375.bugfix.md | 2 +- src/zarr/core/group.py | 6 +- src/zarr/core/metadata/io.py | 71 +++++++++++++++----- tests/test_metadata/test_upgrades.py | 96 ++++++++++++++++++++++------ tests/test_metadata/test_v3.py | 2 +- 5 files changed, 137 insertions(+), 40 deletions(-) diff --git a/changes/4375.bugfix.md b/changes/4375.bugfix.md index e037ad2be0..dfb6304f55 100644 --- a/changes/4375.bugfix.md +++ b/changes/4375.bugfix.md @@ -1,3 +1,3 @@ Arrays whose stored `regular` chunk grid mixes chunk sizes with lists of chunk edge lengths, such as `"chunk_shape": [2, [5, 10, 5]]` (written by zarr 3.2.0 and 3.2.1 for `chunks=(2, (5, 10, 5))`, and as `[2, [5.0, 10.0, 5.0]]` for float edges), can be read again, without enabling `array.rectilinear_chunks`. The grid is read as the rectilinear chunk grid it describes, with a `ZarrUserWarning`; re-saving the metadata stores that rectilinear grid, which requires the flag. A regular chunk grid given edge lists is now rejected, so this metadata is no longer written. -The `array.rectilinear_chunks` flag now gates reading and storing array metadata that declares a rectilinear chunk grid, including in a group's consolidated metadata, instead of constructing `RectilinearChunkGridMetadata`. `Array.resize`, deleting a group member, and `create_hierarchy(..., overwrite=True)` now encode the metadata they will store before deleting anything, so metadata that cannot be stored (for example, a rectilinear chunk grid with the flag off) fails with the store untouched. That error names the array when it is a member of a group's consolidated metadata. +The `array.rectilinear_chunks` flag now gates reading and storing array metadata that declares a rectilinear chunk grid, including in a group's consolidated metadata, instead of constructing `RectilinearChunkGridMetadata`. `Array.resize`, deleting a group member, and `create_hierarchy(..., overwrite=True)` now encode the metadata they will store before deleting anything, and storing a group's metadata encodes its consolidated metadata before it stores the upgrade of any member's metadata, so metadata that cannot be stored (for example, a rectilinear chunk grid with the flag off) fails with the store untouched. That error names the array when it is a member of a group's consolidated metadata. diff --git a/src/zarr/core/group.py b/src/zarr/core/group.py index 04d9d5d434..9591e36991 100644 --- a/src/zarr/core/group.py +++ b/src/zarr/core/group.py @@ -47,7 +47,7 @@ from zarr.core.dtype import parse_data_type from zarr.core.json_parse import parse_field from zarr.core.metadata import ArrayV2Metadata, ArrayV3Metadata -from zarr.core.metadata.io import encode_documents, save_metadata, store_documents +from zarr.core.metadata.io import encode_documents, encode_node, save_metadata, store_node from zarr.core.metadata.v3 import check_storable from zarr.core.sync import SyncMixin, sync from zarr.errors import ( @@ -822,7 +822,7 @@ async def delitem(self, key: str) -> None: # Encode the group metadata without the member before deleting it: metadata that # cannot be stored then fails with the store and this group untouched. members = {name: node for name, node in consolidated.metadata.items() if name != key} - documents = encode_documents( + encoded = await encode_node( self.store_path, replace(self.metadata, consolidated_metadata=replace(consolidated, metadata=members)), ) @@ -830,7 +830,7 @@ async def delitem(self, key: str) -> None: # In place, so every handle sharing this consolidated metadata (a parent's or a # subgroup's) sees the deletion. consolidated.metadata.pop(key, None) - await store_documents(self.store_path, documents) + await store_node(self.store_path, encoded) async def get[DefaultT]( self, key: str, default: DefaultT | None = None diff --git a/src/zarr/core/metadata/io.py b/src/zarr/core/metadata/io.py index fc52838cdc..95f7fbce4e 100644 --- a/src/zarr/core/metadata/io.py +++ b/src/zarr/core/metadata/io.py @@ -130,7 +130,8 @@ async def read_stored_array( """The metadata of the array document stored at `store_path` as it now is, read with the upgrades but without their warnings (the handle that asks has warned), and whether the document had to be upgraded; `None` if no array document is stored - there.""" + there. Only operations that store metadata read it, so a document read as a + rectilinear chunk grid requires the rectilinear chunks flag, as storing it does.""" from zarr.core.array import get_array_metadata, parse_array_metadata try: @@ -141,30 +142,71 @@ async def read_stored_array( return parse_array_metadata(upgraded, str(store_path)), bool(readings) -async def _refresh_consolidated(store_path: StorePath, metadata: GroupMetadata) -> GroupMetadata: +async def _refresh_consolidated( + store_path: StorePath, metadata: GroupMetadata +) -> tuple[GroupMetadata, list[tuple[StorePath, ArrayMetadata]]]: """`metadata` with each member of its consolidated metadata that was read from a document that had to be upgraded replaced by the metadata of the member's own - document as it now is, after storing that document's upgrade if it needs one: the - member document may have changed since, so the consolidated copy is never stored - as if it were valid.""" + document as it now is (the member document may have changed since, so the + consolidated copy is never stored as if it were valid), and the members whose own + documents still need their upgrade stored. Reads the store; writes nothing.""" from zarr.core.group import GroupMetadata consolidated = metadata.consolidated_metadata if consolidated is None: - return metadata + return metadata, [] members = dict(consolidated.metadata) + to_upgrade: list[tuple[StorePath, ArrayMetadata]] = [] for name, member in consolidated.metadata.items(): if isinstance(member, GroupMetadata): - members[name] = await _refresh_consolidated(store_path / name, member) + members[name], nested = await _refresh_consolidated(store_path / name, member) + to_upgrade.extend(nested) elif member._stored_document_upgraded: read = await read_stored_array(store_path / name, member.zarr_format) if read is not None: - members[name], upgraded = read + current, upgraded = read + members[name] = current if upgraded: - await upsert_metadata(store_path / name, members[name]) + to_upgrade.append((store_path / name, current)) if all(members[name] is member for name, member in consolidated.metadata.items()): - return metadata - return replace(metadata, consolidated_metadata=replace(consolidated, metadata=members)) + return metadata, to_upgrade + refreshed = replace(metadata, consolidated_metadata=replace(consolidated, metadata=members)) + return refreshed, to_upgrade + + +class EncodedNode(NamedTuple): + """What storing a node's metadata writes, encoded before anything is written (see + `encode_node`).""" + + documents: dict[str, Buffer] + """The node's own documents, by key (see `encode_documents`).""" + members: list[tuple[StorePath, ArrayMetadata]] + """Members of a group's consolidated metadata whose own stored documents are + upgraded along with it.""" + + +async def encode_node( + store_path: StorePath, metadata: ArrayMetadata | GroupMetadata +) -> EncodedNode: + """Encode what storing `metadata` under `store_path` writes, so metadata that cannot + be stored fails with the store untouched. Group metadata is stored with its + consolidated metadata, whose upgraded members are first refreshed from their own + stored documents (see `_refresh_consolidated`); encoding the group then encodes + them too.""" + from zarr.core.group import GroupMetadata + + members: list[tuple[StorePath, ArrayMetadata]] = [] + if isinstance(metadata, GroupMetadata): + metadata, members = await _refresh_consolidated(store_path, metadata) + return EncodedNode(encode_documents(store_path, metadata), members) + + +async def store_node(store_path: StorePath, encoded: EncodedNode) -> None: + """Store what `encode_node` encoded under `store_path`.""" + await asyncio.gather( + store_documents(store_path, encoded.documents), + *(upsert_metadata(path, member) for path, member in encoded.members), + ) def _build_parents(store_path: StorePath, zarr_format: ZarrFormat) -> dict[str, GroupMetadata]: @@ -204,12 +246,7 @@ async def save_metadata( ------ ValueError """ - from zarr.core.group import GroupMetadata - - if isinstance(metadata, GroupMetadata): - # The one place group metadata is stored, and with it consolidated metadata. - metadata = await _refresh_consolidated(store_path, metadata) - set_awaitables = [store_documents(store_path, encode_documents(store_path, metadata))] + set_awaitables = [store_node(store_path, await encode_node(store_path, metadata))] if ensure_parents: # To enable zarr.create(store, path="a/b/c"), we need to create all the intermediate groups. diff --git a/tests/test_metadata/test_upgrades.py b/tests/test_metadata/test_upgrades.py index c0595df71e..231852e4fd 100644 --- a/tests/test_metadata/test_upgrades.py +++ b/tests/test_metadata/test_upgrades.py @@ -990,19 +990,13 @@ def _mixed_doc(shape: list[int], chunk_shape: list[Any]) -> dict[str, JSON]: (_mixed_doc([6, 12], [2, [5, 10, 5]]), (2, (5, 10, 5)), r"^The stored chunk grid"), (_mixed_doc([0, 20], [2, [5, 10, 5]]), (2, (5, 10, 5)), r"^The stored chunk grid"), (_mixed_doc([6, 0], [2, [5, 10, 5]]), (2, (5, 10, 5)), r"^The stored chunk grid"), - ( - _mixed_doc([6, 20], [True, [5, 10, 5]]), - (1, (5, 10, 5)), - ( - r"^The stored chunk shape \[true, \[5, 10, 5\]\] is invalid: .* read as " - r"\[1, \[5, 10, 5\]\], reading true in dimension 0 as 1\. The stored chunk grid" - ), - ), + # JSON true is read as 1 silently; only the grid reading warns. + (_mixed_doc([6, 20], [True, [5, 10, 5]]), (1, (5, 10, 5)), r"^The stored chunk grid"), (_mixed_doc([6, 20], [2, [5.0, 10.0, 5.0]]), (2, (5, 10, 5)), r"^The stored chunk grid"), ( - _mixed_doc([4, 10_000], [True, [10] * 1000]), - (1, (10,) * 1000), - r"^The stored chunk shape \[true, \[10, 10, .*\.\.\. is invalid: .* read as \[1, \[10, .*\.\.\.,", + _mixed_doc([4, 10_000], [0, [10] * 1000]), + (4, (10,) * 1000), + r"^The stored chunk shape \[0, \[10, 10, .*\.\.\. is invalid: .* read as \[4, \[10, .*\.\.\.,", ), ], ids=[ @@ -1055,22 +1049,20 @@ def test_regular_grid_of_only_edge_lists_rejected() -> None: """A regular chunk shape made only of edge lists was never stored (a rectilinear chunk grid was), so it is not read as rectilinear.""" info = _rejected_without_warning(_mixed_doc([6, 20], [[1, 5], [5, 10, 5]])) - assert info.match(re.escape("Dimension 0: chunk edge length must be an integer, got [1, 5]")) + assert info.match(re.escape("Dimension 0: chunk edge length must be an int, got [1, 5]")) def test_run_length_encoded_edges_in_regular_grid_rejected() -> None: """Run-length encoded edges were never stored in a regular chunk shape.""" info = _rejected_without_warning(_mixed_doc([6, 20], [2, [[5, 2], 10]])) - assert info.match( - re.escape("Dimension 1: chunk edge length must be an integer, got [[5, 2], 10]") - ) + assert info.match(re.escape("Dimension 1: chunk edge length must be an int, got [[5, 2], 10]")) @pytest.mark.parametrize("edge", [5.5, "5"], ids=["fractional", "string"]) def test_non_int_edge_in_regular_grid_rejected(edge: object) -> None: """An edge that is not an integer is reported as such, not blamed on its list.""" info = _rejected_without_warning(_mixed_doc([6, 20], [2, [edge, 15]])) - assert info.match(re.escape(f"Dimension 1: chunk edge length must be an integer, got {edge!r}")) + assert info.match(re.escape(f"Dimension 1: chunk edge length must be an int, got {edge!r}")) def test_edge_below_one_in_regular_grid_rejected() -> None: @@ -1168,14 +1160,25 @@ def _consolidate(path: Path) -> None: zarr.consolidate_metadata(path) +def _set_group_attribute(path: Path) -> None: + zarr.open_group(path, mode="a").attrs["x"] = 1 + + @pytest.mark.filterwarnings( "ignore:.*read as that rectilinear chunk grid:zarr.errors.ZarrUserWarning" ) @pytest.mark.filterwarnings("ignore:Consolidated metadata is currently not part:UserWarning") @pytest.mark.parametrize( "action", - [_resize, _write_chunks, _delete_member, _overwrite_hierarchy, _consolidate], - ids=["resize", "write", "delete-member", "overwrite-hierarchy", "consolidate"], + [ + _resize, + _write_chunks, + _delete_member, + _overwrite_hierarchy, + _consolidate, + _set_group_attribute, + ], + ids=["resize", "write", "delete-member", "overwrite-hierarchy", "consolidate", "group-attrs"], ) def test_store_untouched_without_flag(tmp_path: Path, action: Callable[[Path], None]) -> None: """An operation that would store the rectilinear chunk grid read from the verbatim @@ -1232,6 +1235,63 @@ def test_consolidate_edge_lists_in_regular_grid(tmp_path: Path, member: str) -> np.testing.assert_array_equal(reopened[...], data) +@pytest.mark.filterwarnings("ignore::zarr.errors.ZarrUserWarning") +@pytest.mark.filterwarnings("ignore:Consolidated metadata is currently not part:UserWarning") +def test_group_write_without_flag_stores_no_member_upgrade(tmp_path: Path) -> None: + """A group write refused without the flag stores no upgrade of another consolidated + member either (here `a`, stored with chunk size 0), although that one alone could be + stored: every member is refreshed and the group encoded before anything is stored.""" + path = tmp_path / "group.zarr" + _store_mixed_group(path) + zarr.open_group(path, mode="a").create_array("a", shape=(4,), chunks=(4,), dtype="int16") + with zarr.config.set({"array.rectilinear_chunks": True}): + zarr.consolidate_metadata(path) + _rewrite_doc(path / "a", 3, _store_zero) + _rewrite_consolidated(path, 3, "a", _store_zero) + # Consolidating with the flag stored the rectilinear grid: restore the verbatim document. + (path / "mixed" / "zarr.json").write_text(MIXED_REGULAR_GRID_DOC) + _rewrite_consolidated( + path, 3, "mixed", lambda doc: doc.update(json.loads(MIXED_REGULAR_GRID_DOC)) + ) + stored = {p: p.read_bytes() for p in path.rglob("*") if p.is_file()} + with ( + zarr.config.set({"array.rectilinear_chunks": False}), + pytest.raises(ValueError, match="experimental and disabled by default"), + ): + _set_group_attribute(path) + assert {p: p.read_bytes() for p in path.rglob("*") if p.is_file()} == stored + + +@pytest.mark.filterwarnings( + "ignore:.*read as that rectilinear chunk grid:zarr.errors.ZarrUserWarning" +) +@pytest.mark.filterwarnings("ignore:Consolidated metadata is currently not part:UserWarning") +@pytest.mark.parametrize("action", [_delete_member, _set_group_attribute], ids=["delete", "attrs"]) +def test_group_write_stores_mixed_member_with_flag( + tmp_path: Path, action: Callable[[Path], None] +) -> None: + """With the flag, a group write stores the rectilinear chunk grid read from the + verbatim document, in the member's own document and in the consolidated metadata, + which then open cleanly.""" + path = tmp_path / "group.zarr" + _store_mixed_group(path) + rectilinear = { + "name": "rectilinear", + "configuration": {"kind": "inline", "chunk_shapes": [2, [5, 10, 5]]}, + } + with zarr.config.set({"array.rectilinear_chunks": True}): + action(path) + group_doc = json.loads((path / "zarr.json").read_text()) + assert group_doc["consolidated_metadata"]["metadata"]["mixed"]["chunk_grid"] == rectilinear + assert json.loads((path / "mixed" / "zarr.json").read_text())["chunk_grid"] == rectilinear + with warnings.catch_warnings(): + warnings.simplefilter("error", ZarrUserWarning) + for use_consolidated in (True, False): + mixed = zarr.open_group(path, use_consolidated=use_consolidated)["mixed"] + assert isinstance(mixed, zarr.Array) + assert mixed.read_chunk_sizes == ((2, 2, 2), (5, 10, 5)) + + @pytest.mark.filterwarnings( "ignore:.*read as that rectilinear chunk grid:zarr.errors.ZarrUserWarning" ) diff --git a/tests/test_metadata/test_v3.py b/tests/test_metadata/test_v3.py index 4374bfbd6c..c2a4f82811 100644 --- a/tests/test_metadata/test_v3.py +++ b/tests/test_metadata/test_v3.py @@ -160,7 +160,7 @@ def test_regular_chunk_grid_rejects_edge_lists() -> None: """A regular chunk grid only accepts integer chunk edge lengths.""" with pytest.raises( TypeError, - match=re.escape("Dimension 1: chunk edge length must be an integer, got (5, 10, 5)"), + match=re.escape("Dimension 1: chunk edge length must be an int, got (5, 10, 5)"), ): RegularChunkGridMetadata(chunk_shape=(2, (5, 10, 5))) # type: ignore[arg-type] From bc306aa5f97890241e13fab483d33ac25fc04fc6 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Sat, 26 Sep 2026 16:42:48 +0200 Subject: [PATCH 55/76] fix(metadata): leave a sharded chunk size of 0 unread when the inner shape is invalid A stored outer chunk size of 0 of a sharded array is read in multiples of the inner chunk size. When that inner size is itself 0 or `false`, the unit is unknown: the upgrade now leaves the 0 for the constructor, which rejects it with the same `ValueError` zarr 3.4.0 raised, instead of a ZeroDivisionError. `_read_codec` now matches the sharding codec like `_inner_chunk_shape` does. Assisted-by: ClaudeCode:claude-opus-5-5 Co-Authored-By: Claude Opus 5.5 --- src/zarr/core/metadata/upgrades.py | 55 ++++++++++++++-------------- tests/test_metadata/test_upgrades.py | 11 ++++++ 2 files changed, 38 insertions(+), 28 deletions(-) diff --git a/src/zarr/core/metadata/upgrades.py b/src/zarr/core/metadata/upgrades.py index 52812e334c..3b09dab496 100644 --- a/src/zarr/core/metadata/upgrades.py +++ b/src/zarr/core/metadata/upgrades.py @@ -22,7 +22,7 @@ import warnings from collections.abc import Callable, Iterable, Mapping, Sequence from itertools import chain, repeat -from typing import TYPE_CHECKING, Final, TypeGuard +from typing import TYPE_CHECKING, Final, TypeGuard, cast from zarr.core._json import json_equal from zarr.core.chunk_grids import full_span_chunk_size @@ -137,20 +137,19 @@ def _read_codec(codec: JSON) -> JSON: """Read a stored codec: the inner chunk shape of a sharding codec is read as a chunk shape with no known axis lengths (see `_read_chunk_shape`), and so are those of the sharding codecs nested in its codecs.""" - if not (isinstance(codec, Mapping) and codec.get("name") == "sharding_indexed"): - return codec - configuration = codec.get("configuration") - if not isinstance(configuration, Mapping): - return codec - upgraded = dict(configuration) - stored = configuration.get("chunk_shape") - if isinstance(stored, list): - match _read_chunk_shape(stored, [None] * len(stored)): - case chunk_shape, _: - upgraded["chunk_shape"] = list(chunk_shape) - if isinstance(codecs := configuration.get("codecs"), list): - upgraded["codecs"] = [_read_codec(inner) for inner in codecs] - return {**codec, "configuration": upgraded} + match codec: + case {"name": "sharding_indexed", "configuration": Mapping() as configuration}: + upgraded = dict(configuration) + stored = configuration.get("chunk_shape") + if isinstance(stored, list) and ( + read := _read_chunk_shape(stored, [None] * len(stored)) + ): + upgraded["chunk_shape"] = read[0] + if isinstance(codecs := configuration.get("codecs"), list): + upgraded["codecs"] = [_read_codec(inner) for inner in codecs] + # The mapping pattern does not narrow `codec` for mypy. + return {**cast("Mapping[str, JSON]", codec), "configuration": upgraded} + return codec def _invalid_inner_chunk_sizes_v3(doc: ArrayDocument) -> tuple[ArrayDocument, str | None] | None: @@ -163,18 +162,16 @@ def _invalid_inner_chunk_sizes_v3(doc: ArrayDocument) -> tuple[ArrayDocument, st return {**doc, "codecs": codecs}, None -def _inner_chunk_shape(doc: ArrayDocument) -> list[int]: - """The inner chunk shape of the sharding codec of a Zarr format 3 array document, if - it has one and that is a list of integers; else `[]`.""" - codecs = doc.get("codecs") - if isinstance(codecs, list): - for codec in codecs: - match codec: - case { - "name": "sharding_indexed", - "configuration": {"chunk_shape": list() as inner}, - }: - return inner if _is_int_list(inner) else [] +def _inner_chunk_shape(doc: ArrayDocument) -> list[int] | None: + """The inner chunk shape of the sharding codec of a Zarr format 3 array document: + `[]` if it has none, `None` if its inner chunk sizes are not all integers of at + least 1 (the unit of its outer chunk shape is then unknown).""" + match doc.get("codecs"): + case list() as codecs: + for codec in codecs: + match codec: + case {"name": "sharding_indexed", "configuration": {"chunk_shape": inner}}: + return inner if _is_int_list(inner) and min(inner, default=1) >= 1 else None return [] @@ -183,11 +180,13 @@ def _invalid_chunk_sizes_v3(doc: ArrayDocument) -> tuple[ArrayDocument, str | No shape = doc.get("shape") if not (isinstance(grid, Mapping) and grid.get("name") == "regular" and _is_int_list(shape)): return None + if (units := _inner_chunk_shape(doc)) is None: + return None configuration = grid.get("configuration") if not isinstance(configuration, Mapping): return None stored = configuration.get("chunk_shape") - match _read_chunk_shape(stored, shape, _inner_chunk_shape(doc)): + match _read_chunk_shape(stored, shape, units): case chunk_shape, reading if not json_equal(chunk_shape, stored): upgraded = {**configuration, "chunk_shape": chunk_shape} return {**doc, "chunk_grid": {**grid, "configuration": upgraded}}, reading diff --git a/tests/test_metadata/test_upgrades.py b/tests/test_metadata/test_upgrades.py index 89dd1d236c..5e31bc4f96 100644 --- a/tests/test_metadata/test_upgrades.py +++ b/tests/test_metadata/test_upgrades.py @@ -255,6 +255,17 @@ def test_stored_chunk_shape_ndim_mismatch_rejected() -> None: _read_strictly(_v3_doc([4, 4], [0])) +@pytest.mark.parametrize("inner", [[0], [False]]) +def test_stored_zero_chunk_size_of_shard_with_invalid_inner_chunk_shape_rejected( + inner: list[Any], +) -> None: + """A stored chunk size of 0 of a sharded array is read in multiples of the inner + chunk size; if that is not an integer of at least 1, the 0 is not upgraded, so it + is rejected.""" + with pytest.raises(ValueError, match="^Dimension 0: chunk edge length must be >= 1, got 0$"): + _read_strictly(_v3_doc([4], [0], inner=inner)) + + def _rectilinear_doc(shape: list[int], chunk_shapes: list[Any]) -> dict[str, JSON]: return _v3_doc(shape, [1] * len(shape)) | { "chunk_grid": { From 3bc802fb0e0e88437f1f24ce24d5f18e9b8b5b2f Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Sat, 26 Sep 2026 16:43:13 +0200 Subject: [PATCH 56/76] perf(metadata): read each upgraded member once, concurrently, and adopt it A group handle kept the consolidated copies of upgraded members flagged, so every later group write re-read every such member, one after another, and read each twice on the first write (once to parse, once to diff). - `save_metadata` returns the metadata it stored; `AsyncGroup._save_metadata` (attrs, `update_attributes`, `delitem`, `consolidate_metadata`) and `Group.update_attributes_async` adopt it, so later writes read no member. - The refresh visits only nested groups and flagged members, and reads them with one `asyncio.gather`. The documents it reads are the ones the upsert diffs against (`read_documents` + `upsert_metadata(..., stored)`), and the array write path does the same. - `parse_stored_array` (documents -> silently marked metadata) replaces `read_stored_array`'s `(metadata, bool)` tuple, and is the one "read, mark, don't warn" path, also for code-built metadata. - A flagged member whose own document is gone or cannot be read (invalid, replaced by a group) keeps its consolidated copy instead of failing every group write. - `parse_array_metadata` of a metadata object reads it as a stored document only when `ArrayV2Metadata.chunks` holds a 0, the one chunk size that constructors still accept and the upgrades change. This removes the per-construction `to_dict` + upgrade (AsyncArray() back to 3.4.0 speed) and building arrays from codec configurations holding NumPy scalars works again, as in 3.4.0. - `create_hierarchy` documents that it stores a group's consolidated metadata as given. - Tests: second group write reads no member; nested member refresh; member without a readable document; NumPy-scalar codec configuration; stale handle whose inner chunk shape alone changed (kills `_chunk_layout` returning only the outer grid). Assisted-by: ClaudeCode:claude-opus-5-5 Co-Authored-By: Claude Opus 5.5 --- src/zarr/core/_json.py | 2 +- src/zarr/core/array.py | 52 +++++----- src/zarr/core/group.py | 8 +- src/zarr/core/metadata/io.py | 134 ++++++++++++++++--------- tests/test_metadata/test_io.py | 27 ++++-- tests/test_metadata/test_upgrades.py | 140 ++++++++++++++++++++++----- 6 files changed, 254 insertions(+), 109 deletions(-) diff --git a/src/zarr/core/_json.py b/src/zarr/core/_json.py index 4fd5ffab95..66f6236f44 100644 --- a/src/zarr/core/_json.py +++ b/src/zarr/core/_json.py @@ -41,7 +41,7 @@ def buffer_to_json(buffer: Buffer) -> JSON: return cast("JSON", json.loads(buffer.to_bytes())) -def json_equal(a: object, b: object) -> bool: +def json_equal(a: JSON, b: JSON) -> bool: """Whether two JSON values have the same JSON encoding. Python compares `True` and `1`, or `1.0` and `1`, as equal; JSON does not.""" return json.dumps(a) == json.dumps(b) diff --git a/src/zarr/core/array.py b/src/zarr/core/array.py index a1d6a73033..481aec46fb 100644 --- a/src/zarr/core/array.py +++ b/src/zarr/core/array.py @@ -118,8 +118,13 @@ ArrayV2MetadataDict, ArrayV3Metadata, ) -from zarr.core.metadata.io import read_stored_array, save_metadata, upsert_metadata -from zarr.core.metadata.upgrades import mark_upgraded, upgrade_array_document +from zarr.core.metadata.io import ( + ARRAY_DOCUMENTS, + parse_stored_array, + read_documents, + save_metadata, + upsert_metadata, +) from zarr.core.metadata.v2 import ( CompressorLikev2, get_object_codec_id, @@ -202,29 +207,18 @@ def _chunk_sizes_from_shape( return tuple(result) -def _as_json(value: Any) -> Any: - """`value`, as `to_dict` returns it, with its tuples as JSON arrays.""" - match value: - case tuple() | list(): - return [_as_json(item) for item in value] - case dict(): - return {key: _as_json(item) for key, item in value.items()} - return value - - def parse_array_metadata(data: Any, path: str | None = None) -> ArrayMetadata: """Array metadata from a metadata object or a metadata document, naming the array at `path` in warnings about how an invalid document was read. - The metadata constructors accept chunk sizes that only an invalid document holds - (such as 0), as they always have; such metadata is read as its document is (see - `zarr.core.metadata.upgrades`), so an array can be built from it. No data was read - or written under those chunk sizes, so none of its readings is a warning.""" + `ArrayV2Metadata` accepts a chunk size of 0, as it always has, though only an + invalid document holds one: such metadata is read as the documents it would store + are (see `zarr.core.metadata.upgrades`), so an array can be built from it. No data + was read or written under that chunk size, so the reading is silent.""" + if isinstance(data, ArrayV2Metadata) and 0 in data.chunks: + return parse_stored_array(data.to_buffer_dict(default_buffer_prototype()), 2) if isinstance(data, ArrayMetadata): - document, readings = upgrade_array_document(_as_json(data.to_dict()), data.zarr_format) - if not readings: - return data - return mark_upgraded(parse_array_metadata(dict(document)), [None for _ in readings], path) + return data if isinstance(data, dict): zarr_format = data.get("zarr_format") if zarr_format == 3: @@ -1652,16 +1646,20 @@ async def _store_upgraded_document(self) -> None: """ if not self.metadata._stored_document_upgraded: return - read = await read_stored_array(self.store_path, self.metadata.zarr_format) - if read is not None: - current, upgraded = read + zarr_format = self.metadata.zarr_format + documents = await read_documents(self.store_path, ARRAY_DOCUMENTS[zarr_format]) + try: + current = parse_stored_array(documents, zarr_format) + except ArrayNotFoundError: + pass + else: if _chunk_layout(current) != _chunk_layout(self.metadata): raise ValueError( f"The metadata stored for the array at {str(self.store_path)!r} has " "changed since this array was opened: reopen the array to write to it." ) - if upgraded: - await upsert_metadata(self.store_path, current) + if current._stored_document_upgraded: + await upsert_metadata(self.store_path, current, documents) object.__setattr__(self.metadata, "_stored_document_upgraded", False) async def _set_selection( @@ -4893,7 +4891,9 @@ async def create_array( ) -def _chunk_layout(metadata: ArrayMetadata) -> tuple[object, tuple[int, ...] | None]: +def _chunk_layout( + metadata: ArrayMetadata, +) -> tuple[tuple[int, ...] | ChunkGridMetadata, tuple[int, ...] | None]: """How an array's chunks are laid out: its chunk grid and, if it is sharded, the inner chunk shape.""" grid = metadata.chunks if isinstance(metadata, ArrayV2Metadata) else metadata.chunk_grid diff --git a/src/zarr/core/group.py b/src/zarr/core/group.py index 5a52e3bbf3..f0a9a98d7a 100644 --- a/src/zarr/core/group.py +++ b/src/zarr/core/group.py @@ -899,7 +899,9 @@ async def get_group(self, path: str) -> AsyncGroup: return node async def _save_metadata(self, ensure_parents: bool = False) -> None: - await save_metadata(self.store_path, self.metadata, ensure_parents=ensure_parents) + # Adopt the metadata stored: its consolidated members may be newer. + stored = await save_metadata(self.store_path, self.metadata, ensure_parents=ensure_parents) + object.__setattr__(self, "metadata", stored) @property def path(self) -> str: @@ -2115,7 +2117,7 @@ async def update_attributes_async(self, new_attributes: dict[str, Any]) -> Group new_metadata = replace(self.metadata, attributes=new_attributes) # Write new metadata - await save_metadata(self.store_path, new_metadata) + new_metadata = await save_metadata(self.store_path, new_metadata) async_group = replace(self._async_group, metadata=new_metadata) return replace(self, _async_group=async_group) @@ -3032,6 +3034,8 @@ async def create_hierarchy( ```{'': GroupMetadata, 'a': GroupMetadata, 'b': Groupmetadata}``` After input parsing, this function then creates all the nodes in the hierarchy concurrently. + The metadata of each node is stored as given: the consolidated metadata of a group is + stored as it is, without reading the documents of its members. Arrays and Groups are yielded in the order they are created. This order is not stable and should not be relied on. diff --git a/src/zarr/core/metadata/io.py b/src/zarr/core/metadata/io.py index 1503736e04..82b089ba0d 100644 --- a/src/zarr/core/metadata/io.py +++ b/src/zarr/core/metadata/io.py @@ -10,12 +10,13 @@ from zarr.core._json import buffer_to_json_object, json_equal from zarr.core.buffer.core import default_buffer_prototype from zarr.core.buffer.cpu import buffer_prototype as cpu_buffer_prototype -from zarr.core.metadata.upgrades import upgrade_array_document +from zarr.core.common import ZARR_JSON, ZARRAY_JSON, ZATTRS_JSON +from zarr.core.metadata.upgrades import mark_upgraded, upgrade_array_document from zarr.errors import ArrayNotFoundError, ContainsArrayError from zarr.storage._common import StorePath, ensure_no_existing_node if TYPE_CHECKING: - from collections.abc import Iterator, Mapping + from collections.abc import Iterable, Iterator, Mapping from zarr.core.buffer import Buffer from zarr.core.common import JSON, ZarrFormat @@ -66,7 +67,7 @@ def _diff( case list(), list(): for index, pair in enumerate(zip_longest(stored, new, fillvalue=ABSENT)): yield from _diff((*path, index), *pair) - case _ if ABSENT in (stored, new) or not json_equal(stored, new): + case _ if stored is ABSENT or new is ABSENT or not json_equal(stored, new): yield DocumentChange(path, stored, new) @@ -77,26 +78,29 @@ async def store_documents(store_path: StorePath, documents: Mapping[str, Buffer] ) +async def read_documents(store_path: StorePath, keys: Iterable[str]) -> dict[str, Buffer]: + """The documents stored under `store_path` at `keys`, read concurrently, by key; a + key with no document is left out.""" + keys = tuple(keys) + buffers = await asyncio.gather( + *((store_path / key).get(prototype=cpu_buffer_prototype) for key in keys) + ) + return {key: buf for key, buf in zip(keys, buffers, strict=True) if buf is not None} + + async def upsert_metadata( - store_path: StorePath, metadata: ArrayMetadata | GroupMetadata + store_path: StorePath, metadata: ArrayMetadata | GroupMetadata, stored: Mapping[str, Buffer] ) -> tuple[DocumentChange, ...]: - """Store the documents of `metadata` under `store_path` that differ from the stored - ones, and return how they differed (see `diff_documents`): empty if nothing was - stored. + """Store the documents of `metadata` under `store_path` that differ from `stored`, + the documents `read_documents` read there, and return how they differed (see + `diff_documents`): empty if nothing was stored. - The documents are encoded before the store is read, so metadata that cannot be - stored fails with the store untouched. + The documents are encoded before any is stored, so metadata that cannot be stored + fails with the store untouched. """ documents = metadata.to_buffer_dict(default_buffer_prototype()) - stored = await asyncio.gather( - *((store_path / key).get(prototype=cpu_buffer_prototype) for key in documents) - ) changes = diff_documents( - { - key: buffer_to_json_object(buf) - for key, buf in zip(documents, stored, strict=True) - if buf is not None - }, + {key: buffer_to_json_object(buf) for key, buf in stored.items() if key in documents}, {key: buffer_to_json_object(buf) for key, buf in documents.items()}, ) changed = {change.path[0] for change in changes} @@ -104,46 +108,77 @@ async def upsert_metadata( return changes -async def read_stored_array( - store_path: StorePath, zarr_format: ZarrFormat -) -> tuple[ArrayMetadata, bool] | None: - """The metadata of the array document stored at `store_path` as it now is, read with - the upgrades but without their warnings (the handle that asks has warned), and - whether the document had to be upgraded; `None` if no array document is stored - there.""" - from zarr.core.array import get_array_metadata, parse_array_metadata +ARRAY_DOCUMENTS: Final[Mapping[ZarrFormat, tuple[str, ...]]] = { + 2: (ZARRAY_JSON, ZATTRS_JSON), + 3: (ZARR_JSON,), +} +"""The store keys of the metadata documents of an array of each Zarr format.""" - try: - stored = await get_array_metadata(store_path, zarr_format=zarr_format) - except ArrayNotFoundError: - return None + +def parse_stored_array(documents: Mapping[str, Buffer], zarr_format: ZarrFormat) -> ArrayMetadata: + """The metadata of an array from its documents (by store key, see `ARRAY_DOCUMENTS`), + read with the upgrades but without their warnings (whoever asks has warned, or reads + metadata built in code), and marked (see `mark_upgraded`) if they had to be + upgraded. Raises `ArrayNotFoundError` if there is no array document among them.""" + from zarr.core.array import ( + _array_metadata_dict_v2, + _array_metadata_dict_v3, + parse_array_metadata, + ) + + if zarr_format == 2 and ZARRAY_JSON in documents: + stored = _array_metadata_dict_v2(documents[ZARRAY_JSON], documents.get(ZATTRS_JSON)) + elif zarr_format == 3 and ZARR_JSON in documents: + stored = _array_metadata_dict_v3(documents[ZARR_JSON]) + else: + raise ArrayNotFoundError(f"No Zarr format {zarr_format} array metadata document.") upgraded, readings = upgrade_array_document(stored, zarr_format) - return parse_array_metadata(upgraded, str(store_path)), bool(readings) + return mark_upgraded(parse_array_metadata(dict(upgraded)), [None] * len(readings), None) + + +async def _refresh_array(store_path: StorePath, member: ArrayMetadata) -> ArrayMetadata: + """The metadata of the array document stored at `store_path` as it now is, after + storing its upgrade if it needs one, for `member`, a member of consolidated metadata + that was read from a document that had to be upgraded; `member` itself if no array + document that can be read is stored there (it was deleted, or replaced by a group).""" + documents = await read_documents(store_path, ARRAY_DOCUMENTS[member.zarr_format]) + try: + current = parse_stored_array(documents, member.zarr_format) + except (KeyError, TypeError, ValueError): + return member + if current._stored_document_upgraded: + await upsert_metadata(store_path, current, documents) + object.__setattr__(current, "_stored_document_upgraded", False) + return current async def _refresh_consolidated(store_path: StorePath, metadata: GroupMetadata) -> GroupMetadata: """`metadata` with each member of its consolidated metadata that was read from a document that had to be upgraded replaced by the metadata of the member's own - document as it now is, after storing that document's upgrade if it needs one: the - member document may have changed since, so the consolidated copy is never stored - as if it were valid.""" + document as it now is (see `_refresh_array`), read concurrently: the member document + may have changed since, so the consolidated copy is never stored as if it were + valid.""" from zarr.core.group import GroupMetadata consolidated = metadata.consolidated_metadata if consolidated is None: return metadata - members = dict(consolidated.metadata) - for name, member in consolidated.metadata.items(): - if isinstance(member, GroupMetadata): - members[name] = await _refresh_consolidated(store_path / name, member) - elif member._stored_document_upgraded: - read = await read_stored_array(store_path / name, member.zarr_format) - if read is not None: - members[name], upgraded = read - if upgraded: - await upsert_metadata(store_path / name, members[name]) - if all(members[name] is member for name, member in consolidated.metadata.items()): + stale = { + name: member + for name, member in consolidated.metadata.items() + if isinstance(member, GroupMetadata) or member._stored_document_upgraded + } + refreshed = await asyncio.gather( + *( + _refresh_consolidated(store_path / name, member) + if isinstance(member, GroupMetadata) + else _refresh_array(store_path / name, member) + for name, member in stale.items() + ) + ) + if all(new is old for new, old in zip(refreshed, stale.values(), strict=True)): return metadata + members = {**consolidated.metadata, **dict(zip(stale, refreshed, strict=True))} return replace(metadata, consolidated_metadata=replace(consolidated, metadata=members)) @@ -166,10 +201,12 @@ def _build_parents(store_path: StorePath, zarr_format: ZarrFormat) -> dict[str, return parents -async def save_metadata( - store_path: StorePath, metadata: ArrayMetadata | GroupMetadata, ensure_parents: bool = False -) -> None: - """Asynchronously save the array or group metadata. +async def save_metadata[M: (ArrayMetadata, GroupMetadata)]( + store_path: StorePath, metadata: M, ensure_parents: bool = False +) -> M: + """Asynchronously save the array or group metadata, and return the metadata saved: + `metadata`, but for a group, with the consolidated members it stored (see + `_refresh_consolidated`), for the group to adopt. Parameters ---------- @@ -228,3 +265,4 @@ async def save_metadata( ) from e await asyncio.gather(*set_awaitables) + return metadata diff --git a/tests/test_metadata/test_io.py b/tests/test_metadata/test_io.py index 9accc0f79f..7fb4a06914 100644 --- a/tests/test_metadata/test_io.py +++ b/tests/test_metadata/test_io.py @@ -11,7 +11,14 @@ import zarr from zarr.core.buffer import cpu from zarr.core.metadata import ArrayV2Metadata, ArrayV3Metadata -from zarr.core.metadata.io import ABSENT, DocumentChange, diff_documents, upsert_metadata +from zarr.core.metadata.io import ( + ABSENT, + ARRAY_DOCUMENTS, + DocumentChange, + diff_documents, + read_documents, + upsert_metadata, +) from zarr.core.sync import sync from zarr.storage import MemoryStore, StorePath @@ -143,6 +150,14 @@ def _legacy(zarr_format: Literal[2, 3]) -> tuple[StorePath, ArrayV2Metadata | Ar return StorePath(store), array.metadata +def _upsert( + store_path: StorePath, metadata: ArrayV2Metadata | ArrayV3Metadata +) -> tuple[DocumentChange, ...]: + """Upsert `metadata` against the documents stored at `store_path`.""" + stored = sync(read_documents(store_path, ARRAY_DOCUMENTS[metadata.zarr_format])) + return sync(upsert_metadata(store_path, metadata, stored)) + + @pytest.mark.parametrize("zarr_format", [2, 3]) def test_upsert_metadata_stores_documents_that_differ(zarr_format: Literal[2, 3]) -> None: """The documents that differ from the stored ones are stored, and the changes are @@ -155,7 +170,7 @@ def test_upsert_metadata_stores_documents_that_differ(zarr_format: Literal[2, 3] ("chunks", 0) if zarr_format == 2 else ("chunk_grid", "configuration", "chunk_shape", 0) ) - changes = sync(upsert_metadata(store_path, metadata)) + changes = _upsert(store_path, metadata) assert changes == (DocumentChange((key, *chunk_path), 0, 3),) assert store_path.store.sets == 1 @@ -172,12 +187,12 @@ def test_upsert_metadata_identical_stores_nothing(zarr_format: Literal[2, 3]) -> ) store.sets = 0 - assert sync(upsert_metadata(StorePath(store), array.metadata)) == () + assert _upsert(StorePath(store), array.metadata) == () assert store.sets == 0 def test_upsert_metadata_unstorable_leaves_store_untouched(monkeypatch: pytest.MonkeyPatch) -> None: - """Metadata that cannot be encoded fails before the store is read or written.""" + """Metadata that cannot be encoded fails before the store is written.""" store_path, metadata = _legacy(3) before = _documents(store_path.store) @@ -186,7 +201,7 @@ def refuse(*args: object) -> None: monkeypatch.setattr(ArrayV3Metadata, "to_buffer_dict", refuse) with pytest.raises(ValueError, match="cannot be stored"): - sync(upsert_metadata(store_path, metadata)) + _upsert(store_path, metadata) assert _documents(store_path.store) == before @@ -195,5 +210,5 @@ def test_upsert_metadata_stored_document_not_an_object() -> None: store_path, metadata = _legacy(3) sync(store_path.store.set("zarr.json", cpu.Buffer.from_bytes(b"[]"))) with pytest.raises(TypeError, match="Expected a JSON object, got list"): - sync(upsert_metadata(store_path, metadata)) + _upsert(store_path, metadata) assert _documents(store_path.store) == {"zarr.json": []} diff --git a/tests/test_metadata/test_upgrades.py b/tests/test_metadata/test_upgrades.py index 5e31bc4f96..e299fe72ed 100644 --- a/tests/test_metadata/test_upgrades.py +++ b/tests/test_metadata/test_upgrades.py @@ -3,6 +3,7 @@ from __future__ import annotations import asyncio +import dataclasses import json import re import warnings @@ -13,6 +14,7 @@ import zarr from zarr.codecs import ShardingCodec +from zarr.codecs.numcodecs import Quantize from zarr.core.array import AsyncArray from zarr.core.metadata import ArrayV2Metadata, ArrayV3Metadata from zarr.core.metadata.upgrades import ( @@ -23,7 +25,7 @@ from zarr.core.sync import sync from zarr.dtype import Int16 from zarr.errors import ZarrUserWarning -from zarr.storage import MemoryStore, StorePath +from zarr.storage import LocalStore, MemoryStore, StorePath from zarr.storage._common import make_store_path if TYPE_CHECKING: @@ -732,20 +734,43 @@ def test_stale_handle_write_keeps_valid_document_as_written( np.testing.assert_array_equal(_open_strictly(path)[...], [9, 0, 0]) -@pytest.mark.parametrize("zarr_format", [2, 3]) +def _store_zero(doc: dict[str, Any]) -> None: + _stored_chunks(doc)[0] = 0 + + +def _resize_to_10(doc: dict[str, Any]) -> None: + doc["shape"] = [10] + + +def _halve_inner_chunk_shape(doc: dict[str, Any]) -> None: + doc["codecs"][0]["configuration"]["chunk_shape"] = [2] + + +@pytest.mark.parametrize( + ("zarr_format", "sharded", "change"), + [(2, False, _resize_to_10), (3, False, _resize_to_10), (3, True, _halve_inner_chunk_shape)], + ids=["v2-resized", "v3-resized", "v3-sharded-inner-chunk-shape"], +) def test_stale_handle_write_after_chunk_grid_change_raises( - tmp_path: Path, zarr_format: Literal[2, 3] + tmp_path: Path, + zarr_format: Literal[2, 3], + sharded: bool, + change: Callable[[dict[str, Any]], None], ) -> None: """If the document the store holds when a handle read from an upgraded document - first writes chunks lays out chunks differently from the handle's metadata (here the + first writes chunks lays out chunks differently from the handle's metadata (the array was resized by software that kept the stored chunk size of 0, which now reads - as a larger chunk), the handle's chunks would not be found under it: the write - raises and stores nothing.""" + as a larger chunk, or its inner chunk shape changed), the handle's chunks would not + be found under it: the write raises and stores nothing.""" path = tmp_path / "legacy.zarr" - _legacy_array(path, zarr_format) + if sharded: + zarr.create_array(store=path, shape=(3,), chunks=(4,), shards=(4,), dtype="int16") + _rewrite_doc(path, 3, _store_zero) + else: + _legacy_array(path, zarr_format) with pytest.warns(ZarrUserWarning, match="is read as"): stale = zarr.open_array(store=path, mode="r+") - _rewrite_doc(path, zarr_format, lambda doc: doc.update(shape=[10])) + _rewrite_doc(path, zarr_format, change) documents = {p.name: p.read_bytes() for p in path.iterdir()} with pytest.raises(ValueError, match="has changed since this array was opened: reopen"): @@ -796,8 +821,19 @@ def test_array_from_metadata_with_chunk_size_zero(shape: tuple[int], expected: t np.testing.assert_array_equal(zarr.open_array(store, path="a")[...], np.ones(shape)) -def _store_zero(doc: dict[str, Any]) -> None: - _stored_chunks(doc)[0] = 0 +def test_array_from_metadata_with_numpy_scalar_codec_configuration() -> None: + """An array is built from metadata whose codec configuration holds NumPy scalars + (which are not JSON values), as zarr always built one.""" + array = zarr.create_array( + MemoryStore(), shape=(4,), chunks=(2,), dtype="f8", filters=[Quantize(digits=3, dtype="f8")] + ) + assert isinstance(array.metadata, ArrayV3Metadata) + # The codec keeps its configuration as given, though it is typed as JSON. + digits = cast("JSON", np.int64(3)) + codecs = (Quantize(digits=digits, dtype="f8"), *array.metadata.codecs[1:]) + metadata = dataclasses.replace(array.metadata, codecs=codecs) + + assert AsyncArray(metadata, StorePath(MemoryStore())).metadata is metadata def _rewrite_consolidated( @@ -818,44 +854,96 @@ def _consolidated_member(path: Path, zarr_format: Literal[2, 3], name: str) -> A return json.loads((path / "zarr.json").read_text())["consolidated_metadata"]["metadata"][name] +def _flagged_consolidated_group(path: Path, zarr_format: Literal[2, 3], member: str) -> None: + """A group whose consolidated copy of the array `member` holds the stored chunk size + 0, and whose array `b` is valid.""" + group = zarr.open_group(path, mode="w", zarr_format=zarr_format) + parent, _, name = member.rpartition("/") + (group.require_group(parent) if parent else group).create_array( + name, shape=(3,), chunks=(3,), dtype="int16" + ) + group.create_array("b", shape=(1,), chunks=(1,), dtype="int16") + zarr.consolidate_metadata(path) + _rewrite_consolidated(path, zarr_format, member, _store_zero) + + @pytest.mark.filterwarnings("ignore:Consolidated metadata is currently not part:UserWarning") @pytest.mark.parametrize("zarr_format", [2, 3]) +@pytest.mark.parametrize("member", ["a", "g/a"]) @pytest.mark.parametrize("operation", ["attrs", "update_attributes_async", "delete-member"]) def test_group_write_refreshes_upgraded_consolidated_member( - tmp_path: Path, zarr_format: Literal[2, 3], operation: str + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, + zarr_format: Literal[2, 3], + member: str, + operation: str, ) -> None: """Storing a group's metadata stores its consolidated metadata. Each member of it read - from a document that had to be upgraded is first read again from the member's own - document as it now is (here resized by software that kept the stored chunk size of - 0), whose upgrade is stored, so no group write stores a stale copy as if valid.""" + from a document that had to be upgraded, at any depth, is first read again from the + member's own document as it now is (here resized by software that kept the stored + chunk size of 0), whose upgrade is stored, so no group write stores a stale copy as + if valid. The group adopts the members it stored, so later writes read no member.""" path = tmp_path / "group.zarr" - group = zarr.open_group(path, mode="w", zarr_format=zarr_format) - group.create_array("a", shape=(3,), chunks=(3,), dtype="int16") - group.create_array("b", shape=(1,), chunks=(1,), dtype="int16") - zarr.consolidate_metadata(path) - _rewrite_consolidated(path, zarr_format, "a", _store_zero) + _flagged_consolidated_group(path, zarr_format, member) def resize_keeping_zero(doc: dict[str, Any]) -> None: _store_zero(doc) doc["shape"] = [10] - _rewrite_doc(path / "a", zarr_format, resize_keeping_zero) + _rewrite_doc(path / member, zarr_format, resize_keeping_zero) with pytest.warns(ZarrUserWarning, match="is read as"): group = zarr.open_group(path, mode="r+", use_consolidated=True) if operation == "attrs": group.attrs["x"] = 1 elif operation == "update_attributes_async": - sync(group.update_attributes_async({"x": 1})) + group = sync(group.update_attributes_async({"x": 1})) else: del group["b"] - assert _stored_chunks(_consolidated_member(path, zarr_format, "a")) == [10] + assert _stored_chunks(_consolidated_member(path, zarr_format, member)) == [10] reopened = zarr.open_group(path, mode="r", use_consolidated=True) - member = reopened["a"] - assert isinstance(member, zarr.Array) - assert (member.shape, member.chunks) == ((10,), (10,)) - assert _open_strictly(path / "a").chunks == (10,) + array = reopened[member] + assert isinstance(array, zarr.Array) + assert (array.shape, array.chunks) == ((10,), (10,)) + assert _open_strictly(path / member).chunks == (10,) + + reads: list[str] = [] + get = LocalStore.get + + async def recording_get(self: LocalStore, key: str, *args: Any, **kwargs: Any) -> Any: + reads.append(key) + return await get(self, key, *args, **kwargs) + + monkeypatch.setattr(LocalStore, "get", recording_get) + group.attrs["y"] = 2 + assert reads == [] + + +@pytest.mark.filterwarnings("ignore:Consolidated metadata is currently not part:UserWarning") +@pytest.mark.parametrize("zarr_format", [2, 3]) +@pytest.mark.parametrize("replacement", ["group", "invalid", "none"]) +def test_group_write_keeps_upgraded_member_without_readable_document( + tmp_path: Path, zarr_format: Literal[2, 3], replacement: str +) -> None: + """A member of consolidated metadata read from a document that had to be upgraded, + whose own document is no longer an array document that can be read, has no document + to upgrade: a group write stores the consolidated copy as it is.""" + path = tmp_path / "group.zarr" + _flagged_consolidated_group(path, zarr_format, "a") + with pytest.warns(ZarrUserWarning, match="is read as"): + group = zarr.open_group(path, mode="r+", use_consolidated=True) + del zarr.open_group(path, mode="r+", use_consolidated=False)["a"] + if replacement == "group": + zarr.create_group(path / "a", zarr_format=zarr_format) + elif replacement == "invalid": + (path / "a").mkdir() + document = path / "a" / (".zarray" if zarr_format == 2 else "zarr.json") + document.write_text(json.dumps({"zarr_format": zarr_format, "node_type": "array"})) + + group.attrs["x"] = 1 + + assert _stored_chunks(_consolidated_member(path, zarr_format, "a")) == [3] @pytest.mark.parametrize("zarr_format", [2, 3]) From 76e334e99fe200cffbf87f16878a5690e60310e8 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Sat, 26 Sep 2026 16:57:50 +0200 Subject: [PATCH 57/76] fix(group): store the consolidated metadata as it is after a deletion `delitem` on a consolidated group stored the group metadata it had encoded before deleting, so concurrent deletions through one handle each stored a member list that still held the others' members. The up-front encode now only validates (metadata that cannot be stored still fails with the store untouched); after the deletion the handle's current metadata is stored and adopted, as zarr 3.4.0 stored it. Assisted-by: ClaudeCode:claude-opus-5-5 Co-Authored-By: Claude Opus 5.5 --- src/zarr/core/group.py | 18 ++++++++++-------- tests/test_group.py | 34 ++++++++++++++++++++++++++++++++++ 2 files changed, 44 insertions(+), 8 deletions(-) diff --git a/src/zarr/core/group.py b/src/zarr/core/group.py index 857a25d3c8..70a56c169d 100644 --- a/src/zarr/core/group.py +++ b/src/zarr/core/group.py @@ -47,7 +47,7 @@ from zarr.core.dtype import parse_data_type from zarr.core.json_parse import parse_field from zarr.core.metadata import ArrayV2Metadata, ArrayV3Metadata -from zarr.core.metadata.io import encode_documents, encode_node, save_metadata, store_node +from zarr.core.metadata.io import encode_documents, encode_node, save_metadata from zarr.core.metadata.v3 import check_storable from zarr.core.sync import SyncMixin, sync from zarr.errors import ( @@ -820,19 +820,21 @@ async def delitem(self, key: str) -> None: await store_path.delete_dir() return # Encode the group metadata without the member before deleting it: metadata that - # cannot be stored then fails with the store and this group untouched. + # cannot be stored then fails with the store and this group untouched. What is + # stored is encoded after the deletion, from the metadata as it then is, so + # concurrent deletions each store the deletions made before them. members = {name: node for name, node in consolidated.metadata.items() if name != key} - encoded = await encode_node( + await encode_node( self.store_path, replace(self.metadata, consolidated_metadata=replace(consolidated, metadata=members)), ) await store_path.delete_dir() # In place, so every handle sharing this consolidated metadata (a parent's or a - # subgroup's) sees the deletion. - consolidated.metadata.pop(key, None) - await store_node(self.store_path, encoded) - # Adopt the metadata stored: its consolidated members may be newer. - object.__setattr__(self, "metadata", encoded.metadata) + # subgroup's) sees the deletion; from the consolidated metadata this handle holds + # now, which a concurrent write may have replaced with the metadata it stored. + if (current := self.metadata.consolidated_metadata) is not None: + current.metadata.pop(key, None) + await self._save_metadata() async def get[DefaultT]( self, key: str, default: DefaultT | None = None diff --git a/tests/test_group.py b/tests/test_group.py index 166d39df71..5594e53f8d 100644 --- a/tests/test_group.py +++ b/tests/test_group.py @@ -1,5 +1,6 @@ from __future__ import annotations +import asyncio import contextlib import inspect import json @@ -1690,6 +1691,39 @@ def test_group_delitem_consolidated_aliased(self, store: Store) -> None: with pytest.raises(KeyError): group["sub/b"] + async def test_group_delitem_consolidated_concurrent(self, zarr_format: ZarrFormat) -> None: + """Concurrent deletions through one handle each store the deletions made before + them, so the stored consolidated metadata lists what the handle lists. Here the + deletions finish in the reverse of the order they started in.""" + + delays = [0.03, 0.02, 0.01] + + class SlowDeletes(LatencyStore): + async def delete_dir(self, prefix: str) -> None: + await asyncio.sleep(delays.pop(0)) + await super().delete_dir(prefix) + + store = SlowDeletes(MemoryStore()) + root = await AsyncGroup.from_store(store=store, zarr_format=zarr_format) + for name in "abcd": + await root.create_array(name, shape=(2,), dtype="i4") + with warnings.catch_warnings(): + warnings.filterwarnings( + "ignore", "Consolidated metadata is currently not part", ZarrUserWarning + ) + await zarr.api.asynchronous.consolidate_metadata(store) + group = await zarr.api.asynchronous.open_consolidated(store=store, zarr_format=zarr_format) + + await asyncio.gather(*(group.delitem(name) for name in "abc")) + + reopened = await zarr.api.asynchronous.open_consolidated( + store=store, zarr_format=zarr_format + ) + assert group.metadata.consolidated_metadata is not None + assert reopened.metadata.consolidated_metadata is not None + assert list(group.metadata.consolidated_metadata.metadata) == ["d"] + assert list(reopened.metadata.consolidated_metadata.metadata) == ["d"] + def test_open_consolidated_raises(self, store: Store) -> None: if isinstance(store, ZipStore): raise pytest.skip("Not implemented") From f78932d3149151eb30fc0c6b5b08585b858fcd86 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Sat, 26 Sep 2026 16:59:21 +0200 Subject: [PATCH 58/76] refactor: name a node in errors as "Array 'path': what happened" The rectilinear flag's read note, the stale-handle write error and the encode note now share the shape the stored-document warnings use: the node, its path, then what was (not) done. The consolidated-member note keeps its location qualifier ("in the consolidated metadata"), followed by the group's note. Assisted-by: ClaudeCode:claude-opus-5-5 Co-Authored-By: Claude Opus 5.5 --- src/zarr/core/array.py | 4 ++-- src/zarr/core/metadata/v3.py | 2 +- tests/test_metadata/test_upgrades.py | 2 +- tests/test_unified_chunk_grid.py | 13 +++++++++++++ 4 files changed, 17 insertions(+), 4 deletions(-) diff --git a/src/zarr/core/array.py b/src/zarr/core/array.py index 88adaf8e29..17dc55f605 100644 --- a/src/zarr/core/array.py +++ b/src/zarr/core/array.py @@ -1657,8 +1657,8 @@ async def _store_upgraded_document(self) -> None: else: if _chunk_layout(current) != _chunk_layout(self.metadata): raise ValueError( - f"The metadata stored for the array at {str(self.store_path)!r} has " - "changed since this array was opened: reopen the array to write to it." + f"Array {str(self.store_path)!r}: the metadata stored has changed since " + "this array was opened; reopen the array to write to it. Nothing was stored." ) if current._stored_document_upgraded: await upsert_metadata(self.store_path, current, documents) diff --git a/src/zarr/core/metadata/v3.py b/src/zarr/core/metadata/v3.py index bd8624f6c7..dcc74436be 100644 --- a/src/zarr/core/metadata/v3.py +++ b/src/zarr/core/metadata/v3.py @@ -646,7 +646,7 @@ def from_dict(cls, data: dict[str, JSON], *, path: str | None = None) -> Self: _check_rectilinear_chunks_enabled() except ValueError as e: if path is not None: - e.add_note(f"Array {path!r}.") + e.add_note(f"Array {path!r}: nothing was read.") raise upgraded, readings = upgrade_array_document(data, 3) # a new dict, because we are modifying it diff --git a/tests/test_metadata/test_upgrades.py b/tests/test_metadata/test_upgrades.py index 458b18b1d3..459c7635a0 100644 --- a/tests/test_metadata/test_upgrades.py +++ b/tests/test_metadata/test_upgrades.py @@ -773,7 +773,7 @@ def test_stale_handle_write_after_chunk_grid_change_raises( _rewrite_doc(path, zarr_format, change) documents = {p.name: p.read_bytes() for p in path.iterdir()} - with pytest.raises(ValueError, match="has changed since this array was opened: reopen"): + with pytest.raises(ValueError, match="has changed since this array was opened; reopen"): stale[0:3] = [7, 8, 9] assert stale.metadata._stored_document_upgraded diff --git a/tests/test_unified_chunk_grid.py b/tests/test_unified_chunk_grid.py index 435823a478..9bb2577bee 100644 --- a/tests/test_unified_chunk_grid.py +++ b/tests/test_unified_chunk_grid.py @@ -141,6 +141,19 @@ def test_rectilinear_feature_flag_blocked(action: Any) -> None: action() +def test_rectilinear_feature_flag_names_stored_array() -> None: + """Opening a stored array whose document the flag refuses names the array.""" + store = MemoryStore() + with zarr.config.set({"array.rectilinear_chunks": True}): + zarr.create_array(store, name="r", shape=(30,), chunks=[[10, 20]], dtype="int32") + with ( + zarr.config.set({"array.rectilinear_chunks": False}), + pytest.raises(ValueError, match="experimental and disabled by default") as info, + ): + zarr.open_array(store, path="r") + assert info.value.__notes__ == [f"Array {str(store) + '/r'!r}: nothing was read."] + + def test_rectilinear_metadata_classes_not_gated() -> None: """The flag gates stored documents, not the chunk grid metadata classes.""" with zarr.config.set({"array.rectilinear_chunks": False}): From dc50ec61d8ebd0facee9cf99b4433f338c2debf6 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Sat, 26 Sep 2026 18:02:28 +0200 Subject: [PATCH 59/76] fix(group): adopt refreshed consolidated members in place A group write replaced the handle's metadata with a copy holding the consolidated members it read again, so the handle no longer shared its consolidated dicts with subgroup handles taken earlier: an array deleted through such a subgroup was stored again by the parent's next write, and reappeared on reopen. `_refresh_consolidated` now returns, along with the copy it encodes, the members it replaced (with the consolidated dict that holds each), and `save_metadata` writes them back into those dicts once the store succeeds, unless the member was deleted or replaced meanwhile. `save_metadata` returns nothing again, and the group callers keep their metadata objects. A failed store leaves the members flagged, so a retry reads them again. Assisted-by: ClaudeCode:claude-opus-5-5 Co-Authored-By: Claude Opus 5.5 --- src/zarr/core/group.py | 6 +- src/zarr/core/metadata/io.py | 89 +++++++++++++++++++--------- tests/test_metadata/test_upgrades.py | 27 +++++++++ 3 files changed, 91 insertions(+), 31 deletions(-) diff --git a/src/zarr/core/group.py b/src/zarr/core/group.py index f0a9a98d7a..cd91582315 100644 --- a/src/zarr/core/group.py +++ b/src/zarr/core/group.py @@ -899,9 +899,7 @@ async def get_group(self, path: str) -> AsyncGroup: return node async def _save_metadata(self, ensure_parents: bool = False) -> None: - # Adopt the metadata stored: its consolidated members may be newer. - stored = await save_metadata(self.store_path, self.metadata, ensure_parents=ensure_parents) - object.__setattr__(self, "metadata", stored) + await save_metadata(self.store_path, self.metadata, ensure_parents=ensure_parents) @property def path(self) -> str: @@ -2117,7 +2115,7 @@ async def update_attributes_async(self, new_attributes: dict[str, Any]) -> Group new_metadata = replace(self.metadata, attributes=new_attributes) # Write new metadata - new_metadata = await save_metadata(self.store_path, new_metadata) + await save_metadata(self.store_path, new_metadata) async_group = replace(self._async_group, metadata=new_metadata) return replace(self, _async_group=async_group) diff --git a/src/zarr/core/metadata/io.py b/src/zarr/core/metadata/io.py index 82b089ba0d..281aee04ff 100644 --- a/src/zarr/core/metadata/io.py +++ b/src/zarr/core/metadata/io.py @@ -152,34 +152,66 @@ async def _refresh_array(store_path: StorePath, member: ArrayMetadata) -> ArrayM return current -async def _refresh_consolidated(store_path: StorePath, metadata: GroupMetadata) -> GroupMetadata: +class _Refreshed(NamedTuple): + """A member of consolidated metadata that `_refresh_consolidated` read again from the + member's own document.""" + + members: dict[str, ArrayMetadata | GroupMetadata] + """The consolidated metadata that holds the member, by member name.""" + name: str + stale: ArrayMetadata + current: ArrayMetadata + + def adopt(self) -> None: + """Replace the stale member with the current one in place, so every handle that + shares this consolidated metadata sees it; unless the member was deleted or + replaced since it was read.""" + if self.members.get(self.name) is self.stale: + self.members[self.name] = self.current + + +async def _refresh_consolidated( + store_path: StorePath, metadata: GroupMetadata +) -> tuple[GroupMetadata, list[_Refreshed]]: """`metadata` with each member of its consolidated metadata that was read from a - document that had to be upgraded replaced by the metadata of the member's own - document as it now is (see `_refresh_array`), read concurrently: the member document - may have changed since, so the consolidated copy is never stored as if it were - valid.""" + document that had to be upgraded, at any depth, replaced by the metadata of the + member's own document as it now is (see `_refresh_array`), read concurrently, and + the members it replaced: the member document may have changed since, so the + consolidated copy is never stored as if it were valid. `metadata` is left as it is, + for the group to adopt the members (see `_Refreshed.adopt`) once they are stored.""" from zarr.core.group import GroupMetadata consolidated = metadata.consolidated_metadata if consolidated is None: - return metadata - stale = { + return metadata, [] + groups = { name: member for name, member in consolidated.metadata.items() - if isinstance(member, GroupMetadata) or member._stored_document_upgraded + if isinstance(member, GroupMetadata) } - refreshed = await asyncio.gather( - *( - _refresh_consolidated(store_path / name, member) - if isinstance(member, GroupMetadata) - else _refresh_array(store_path / name, member) - for name, member in stale.items() - ) + arrays = { + name: member + for name, member in consolidated.metadata.items() + if not isinstance(member, GroupMetadata) and member._stored_document_upgraded + } + nested, current = await asyncio.gather( + asyncio.gather(*(_refresh_consolidated(store_path / n, m) for n, m in groups.items())), + asyncio.gather(*(_refresh_array(store_path / n, m) for n, m in arrays.items())), ) - if all(new is old for new, old in zip(refreshed, stale.values(), strict=True)): - return metadata - members = {**consolidated.metadata, **dict(zip(stale, refreshed, strict=True))} - return replace(metadata, consolidated_metadata=replace(consolidated, metadata=members)) + refreshed = [ + _Refreshed(consolidated.metadata, name, stale, new) + for (name, stale), new in zip(arrays.items(), current, strict=True) + if new is not stale + ] + members = dict(consolidated.metadata) + members.update({member.name: member.current for member in refreshed}) + for name, (group, inner) in zip(groups, nested, strict=True): + members[name] = group + refreshed.extend(inner) + if not refreshed: + return metadata, [] + consolidated = replace(consolidated, metadata=members) + return replace(metadata, consolidated_metadata=consolidated), refreshed def _build_parents(store_path: StorePath, zarr_format: ZarrFormat) -> dict[str, GroupMetadata]: @@ -201,12 +233,12 @@ def _build_parents(store_path: StorePath, zarr_format: ZarrFormat) -> dict[str, return parents -async def save_metadata[M: (ArrayMetadata, GroupMetadata)]( - store_path: StorePath, metadata: M, ensure_parents: bool = False -) -> M: - """Asynchronously save the array or group metadata, and return the metadata saved: - `metadata`, but for a group, with the consolidated members it stored (see - `_refresh_consolidated`), for the group to adopt. +async def save_metadata( + store_path: StorePath, metadata: ArrayMetadata | GroupMetadata, ensure_parents: bool = False +) -> None: + """Asynchronously save the array or group metadata. A group adopts, in place, the + members of its consolidated metadata that had to be read again to store it (see + `_refresh_consolidated`), once they are stored. Parameters ---------- @@ -223,9 +255,10 @@ async def save_metadata[M: (ArrayMetadata, GroupMetadata)]( """ from zarr.core.group import GroupMetadata + refreshed: list[_Refreshed] = [] if isinstance(metadata, GroupMetadata): # The one place group metadata is stored, and with it consolidated metadata. - metadata = await _refresh_consolidated(store_path, metadata) + metadata, refreshed = await _refresh_consolidated(store_path, metadata) to_save = metadata.to_buffer_dict(default_buffer_prototype()) set_awaitables = [store_documents(store_path, to_save)] @@ -265,4 +298,6 @@ async def save_metadata[M: (ArrayMetadata, GroupMetadata)]( ) from e await asyncio.gather(*set_awaitables) - return metadata + # Only once stored: a group whose store failed keeps the members flagged, to read again. + for member in refreshed: + member.adopt() diff --git a/tests/test_metadata/test_upgrades.py b/tests/test_metadata/test_upgrades.py index e299fe72ed..8da2b24a5e 100644 --- a/tests/test_metadata/test_upgrades.py +++ b/tests/test_metadata/test_upgrades.py @@ -920,6 +920,33 @@ async def recording_get(self: LocalStore, key: str, *args: Any, **kwargs: Any) - assert reads == [] +@pytest.mark.filterwarnings("ignore:Consolidated metadata is currently not part:UserWarning") +@pytest.mark.parametrize("zarr_format", [2, 3]) +def test_group_write_keeps_consolidated_metadata_shared_with_subgroups( + tmp_path: Path, zarr_format: Literal[2, 3] +) -> None: + """A subgroup handle taken from a group shares the group's consolidated metadata, also + once the group has adopted the upgraded members its first write stored: an array + deleted through the subgroup is gone from what the group stores next.""" + path = tmp_path / "group.zarr" + created = zarr.open_group(path, mode="w", zarr_format=zarr_format).create_group("g") + for name in ("a", "c"): + created.create_array(name, shape=(3,), chunks=(3,), dtype="int16") + zarr.consolidate_metadata(path) + _rewrite_consolidated(path, zarr_format, "g/a", _store_zero) + with pytest.warns(ZarrUserWarning, match="is read as"): + group = zarr.open_group(path, mode="r+", use_consolidated=True) + subgroup = group["g"] + assert isinstance(subgroup, zarr.Group) + + group.attrs["x"] = 1 + del subgroup["c"] + group.attrs["y"] = 2 + + reopened = zarr.open_group(path, mode="r", use_consolidated=True) + assert sorted(dict(reopened.members(max_depth=None))) == ["g", "g/a"] + + @pytest.mark.filterwarnings("ignore:Consolidated metadata is currently not part:UserWarning") @pytest.mark.parametrize("zarr_format", [2, 3]) @pytest.mark.parametrize("replacement", ["group", "invalid", "none"]) From cddb877261bc6c8b42961bc2c0382680e1386824 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Sat, 26 Sep 2026 18:02:38 +0200 Subject: [PATCH 60/76] fix(metadata): raise the rectilinear flag error when refreshing a consolidated member A group write reads each upgraded consolidated member again from its own document, and kept the consolidated copy when that document could not be read. That also swallowed the rectilinear chunks flag error, so a member whose document now declares a rectilinear grid had its stale copy stored as valid. The flag error is now `RectilinearChunksDisabledError`, a `ValueError`, and the refresh lets it propagate: the group write raises and stores nothing. Unreadable documents keep their consolidated copy as before. Assisted-by: ClaudeCode:claude-opus-5-5 Co-Authored-By: Claude Opus 5.5 --- src/zarr/core/metadata/io.py | 3 +++ src/zarr/core/metadata/v3.py | 6 +++++- tests/test_metadata/test_upgrades.py | 18 ++++++++++++++++++ 3 files changed, 26 insertions(+), 1 deletion(-) diff --git a/src/zarr/core/metadata/io.py b/src/zarr/core/metadata/io.py index 281aee04ff..c477517b54 100644 --- a/src/zarr/core/metadata/io.py +++ b/src/zarr/core/metadata/io.py @@ -12,6 +12,7 @@ from zarr.core.buffer.cpu import buffer_prototype as cpu_buffer_prototype from zarr.core.common import ZARR_JSON, ZARRAY_JSON, ZATTRS_JSON from zarr.core.metadata.upgrades import mark_upgraded, upgrade_array_document +from zarr.core.metadata.v3 import RectilinearChunksDisabledError from zarr.errors import ArrayNotFoundError, ContainsArrayError from zarr.storage._common import StorePath, ensure_no_existing_node @@ -144,6 +145,8 @@ async def _refresh_array(store_path: StorePath, member: ArrayMetadata) -> ArrayM documents = await read_documents(store_path, ARRAY_DOCUMENTS[member.zarr_format]) try: current = parse_stored_array(documents, member.zarr_format) + except RectilinearChunksDisabledError: + raise # A document that can be read, but only with the flag. except (KeyError, TypeError, ValueError): return member if current._stored_document_upgraded: diff --git a/src/zarr/core/metadata/v3.py b/src/zarr/core/metadata/v3.py index ad16afc3e1..17e88d17e4 100644 --- a/src/zarr/core/metadata/v3.py +++ b/src/zarr/core/metadata/v3.py @@ -270,6 +270,10 @@ def from_dict(cls, data: RegularChunkGridMetadataJSON) -> Self: # type: ignore[ return cls(chunk_shape=parse_chunk_shape(configuration["chunk_shape"])) +class RectilinearChunksDisabledError(ValueError): + """Rectilinear chunk grids are used while the `array.rectilinear_chunks` flag is off.""" + + @dataclass(frozen=True, kw_only=True) class RectilinearChunkGridMetadata(Metadata): """Metadata-only description of a rectilinear chunk grid. @@ -290,7 +294,7 @@ class RectilinearChunkGridMetadata(Metadata): def __post_init__(self) -> None: if not config.get("array.rectilinear_chunks"): - raise ValueError( + raise RectilinearChunksDisabledError( "Rectilinear chunk grids are experimental and disabled by default. " "Enable them with: zarr.config.set({'array.rectilinear_chunks': True}) " "or set the environment variable ZARR_ARRAY__RECTILINEAR_CHUNKS=True" diff --git a/tests/test_metadata/test_upgrades.py b/tests/test_metadata/test_upgrades.py index 8da2b24a5e..c2b76b12aa 100644 --- a/tests/test_metadata/test_upgrades.py +++ b/tests/test_metadata/test_upgrades.py @@ -947,6 +947,24 @@ def test_group_write_keeps_consolidated_metadata_shared_with_subgroups( assert sorted(dict(reopened.members(max_depth=None))) == ["g", "g/a"] +@pytest.mark.filterwarnings("ignore:Consolidated metadata is currently not part:UserWarning") +def test_group_write_refuses_upgraded_member_read_as_rectilinear(tmp_path: Path) -> None: + """A member of consolidated metadata read from a document that had to be upgraded, + whose own document now declares a rectilinear chunk grid, is read again only with the + rectilinear chunks flag: without it, a group write raises and stores nothing.""" + path = tmp_path / "group.zarr" + _flagged_consolidated_group(path, 3, "a") + with pytest.warns(ZarrUserWarning, match="is read as"): + group = zarr.open_group(path, mode="r+", use_consolidated=True) + (path / "a" / "zarr.json").write_text(json.dumps(_rectilinear_doc([3], [[1, 2]]))) + documents = {p: p.read_bytes() for p in path.rglob("*") if p.is_file()} + + with pytest.raises(ValueError, match="Rectilinear chunk grids are experimental"): + group.attrs["x"] = 1 + + assert {p: p.read_bytes() for p in path.rglob("*") if p.is_file()} == documents + + @pytest.mark.filterwarnings("ignore:Consolidated metadata is currently not part:UserWarning") @pytest.mark.parametrize("zarr_format", [2, 3]) @pytest.mark.parametrize("replacement", ["group", "invalid", "none"]) From b0628e6550f052b3e37aa94213bbbe948aed141f Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Sat, 26 Sep 2026 18:06:19 +0200 Subject: [PATCH 61/76] fix(group): say nothing was stored when a consolidated member needs the flag Storing a group's consolidated metadata reads each upgraded member again from its own document; one read as a rectilinear chunk grid now raises the flag error there (naming the array) instead of being kept stale. That error now also says the group stored nothing, as the encoding error does. `zarr.consolidate_metadata` without the flag reports the member this way. The refresh test gains a mixed regular grid row. Assisted-by: ClaudeCode:claude-opus-5-5 Co-Authored-By: Claude Opus 5.5 --- src/zarr/core/metadata/io.py | 6 +++++- tests/test_metadata/test_upgrades.py | 26 ++++++++++++++++++++------ 2 files changed, 25 insertions(+), 7 deletions(-) diff --git a/src/zarr/core/metadata/io.py b/src/zarr/core/metadata/io.py index e5060570bc..c0f365845f 100644 --- a/src/zarr/core/metadata/io.py +++ b/src/zarr/core/metadata/io.py @@ -276,7 +276,11 @@ async def encode_node( members: list[RefreshedMember] = [] if isinstance(metadata, GroupMetadata): - metadata, members = await _refresh_consolidated(store_path, metadata) + try: + metadata, members = await _refresh_consolidated(store_path, metadata) + except RectilinearChunksDisabledError as e: + e.add_note(f"Group {str(store_path)!r}: nothing was stored.") + raise return EncodedNode(encode_documents(store_path, metadata), members) diff --git a/tests/test_metadata/test_upgrades.py b/tests/test_metadata/test_upgrades.py index a923ae5d06..c1238f41f1 100644 --- a/tests/test_metadata/test_upgrades.py +++ b/tests/test_metadata/test_upgrades.py @@ -948,20 +948,33 @@ def test_group_write_keeps_consolidated_metadata_shared_with_subgroups( @pytest.mark.filterwarnings("ignore:Consolidated metadata is currently not part:UserWarning") -def test_group_write_refuses_upgraded_member_read_as_rectilinear(tmp_path: Path) -> None: +@pytest.mark.parametrize("document", ["rectilinear", "mixed"]) +def test_group_write_refuses_upgraded_member_read_as_rectilinear( + tmp_path: Path, document: str +) -> None: """A member of consolidated metadata read from a document that had to be upgraded, - whose own document now declares a rectilinear chunk grid, is read again only with the - rectilinear chunks flag: without it, a group write raises and stores nothing.""" + whose own document now declares a rectilinear chunk grid (or a regular grid listing + chunk edges, read as one), is read again only with the rectilinear chunks flag: + without it, a group write raises and stores nothing.""" path = tmp_path / "group.zarr" _flagged_consolidated_group(path, 3, "a") with pytest.warns(ZarrUserWarning, match="is read as"): group = zarr.open_group(path, mode="r+", use_consolidated=True) - (path / "a" / "zarr.json").write_text(json.dumps(_rectilinear_doc([3], [[1, 2]]))) + (path / "a" / "zarr.json").write_text( + json.dumps(_rectilinear_doc([3], [[1, 2]])) + if document == "rectilinear" + else MIXED_REGULAR_GRID_DOC + ) documents = {p: p.read_bytes() for p in path.rglob("*") if p.is_file()} + group_path = str(sync(make_store_path(path))) - with pytest.raises(ValueError, match="Rectilinear chunk grids are experimental"): + with pytest.raises(ValueError, match="Rectilinear chunk grids are experimental") as info: group.attrs["x"] = 1 + assert info.value.__notes__ == [ + f"Array {group_path + '/a'!r}: nothing was read.", + f"Group {group_path!r}: nothing was stored.", + ] assert {p: p.read_bytes() for p in path.rglob("*") if p.is_file()} == documents @@ -1355,8 +1368,9 @@ def test_consolidate_edge_lists_in_regular_grid(tmp_path: Path, member: str) -> pytest.raises(ValueError, match="experimental and disabled") as info, ): zarr.consolidate_metadata(path) + # Storing the consolidated metadata reads the member's document again. assert info.value.__notes__ == [ - f"Array {member!r} in the consolidated metadata.", + f"Array {group_path + '/' + member!r}: nothing was read.", f"Group {group_path!r}: nothing was stored.", ] with pytest.warns(ZarrUserWarning, match="read as that rectilinear chunk grid"): From 8753ef4bbddf5117fea1e4c3223f642910efd786 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Sat, 26 Sep 2026 18:07:49 +0200 Subject: [PATCH 62/76] perf(group): read each upgraded member once when deleting a member `delitem` encodes the group metadata without the member before deleting it, which reads each upgraded consolidated member again, then stored the metadata through `save_metadata`, which read them all a second time. It now adopts the members the first encoding read, after the deletion and in place, and stores the metadata as it then is with those members' upgrades (`store_node`), so the first deletion reads each member once. The handle's consolidated metadata is no longer replaced by writes, so the deletion pops from it directly. Assisted-by: ClaudeCode:claude-opus-5-5 Co-Authored-By: Claude Opus 5.5 --- src/zarr/core/group.py | 26 +++++++++++++++++++------- tests/test_metadata/test_upgrades.py | 21 +++++++++++++-------- 2 files changed, 32 insertions(+), 15 deletions(-) diff --git a/src/zarr/core/group.py b/src/zarr/core/group.py index 0626f6cfa9..3556a0d20c 100644 --- a/src/zarr/core/group.py +++ b/src/zarr/core/group.py @@ -47,7 +47,13 @@ from zarr.core.dtype import parse_data_type from zarr.core.json_parse import parse_field from zarr.core.metadata import ArrayV2Metadata, ArrayV3Metadata -from zarr.core.metadata.io import encode_documents, encode_node, save_metadata +from zarr.core.metadata.io import ( + EncodedNode, + encode_documents, + encode_node, + save_metadata, + store_node, +) from zarr.core.metadata.v3 import check_storable from zarr.core.sync import SyncMixin, sync from zarr.errors import ( @@ -824,17 +830,23 @@ async def delitem(self, key: str) -> None: # stored is encoded after the deletion, from the metadata as it then is, so # concurrent deletions each store the deletions made before them. members = {name: node for name, node in consolidated.metadata.items() if name != key} - await encode_node( + encoded = await encode_node( self.store_path, replace(self.metadata, consolidated_metadata=replace(consolidated, metadata=members)), ) await store_path.delete_dir() # In place, so every handle sharing this consolidated metadata (a parent's or a - # subgroup's) sees the deletion; from the consolidated metadata this handle holds - # now, which a concurrent write may have replaced with the metadata it stored. - if (current := self.metadata.consolidated_metadata) is not None: - current.metadata.pop(key, None) - await self._save_metadata() + # subgroup's) sees the deletion, and the members the encoding above read again, + # which are not read twice. + consolidated.metadata.pop(key, None) + refreshed = [ + member._replace(members=consolidated.metadata) if member.members is members else member + for member in encoded.members + ] + for member in refreshed: + member.adopt() + documents = encode_documents(self.store_path, self.metadata) + await store_node(self.store_path, EncodedNode(documents, refreshed)) async def get[DefaultT]( self, key: str, default: DefaultT | None = None diff --git a/tests/test_metadata/test_upgrades.py b/tests/test_metadata/test_upgrades.py index c1238f41f1..364917ff38 100644 --- a/tests/test_metadata/test_upgrades.py +++ b/tests/test_metadata/test_upgrades.py @@ -882,7 +882,8 @@ def test_group_write_refreshes_upgraded_consolidated_member( from a document that had to be upgraded, at any depth, is first read again from the member's own document as it now is (here resized by software that kept the stored chunk size of 0), whose upgrade is stored, so no group write stores a stale copy as - if valid. The group adopts the members it stored, so later writes read no member.""" + if valid. The first write reads each such member once; the group adopts the members + it stored, so later writes read no member.""" path = tmp_path / "group.zarr" _flagged_consolidated_group(path, zarr_format, member) @@ -893,6 +894,14 @@ def resize_keeping_zero(doc: dict[str, Any]) -> None: _rewrite_doc(path / member, zarr_format, resize_keeping_zero) with pytest.warns(ZarrUserWarning, match="is read as"): group = zarr.open_group(path, mode="r+", use_consolidated=True) + reads: list[str] = [] + get = LocalStore.get + + async def recording_get(self: LocalStore, key: str, *args: Any, **kwargs: Any) -> Any: + reads.append(key) + return await get(self, key, *args, **kwargs) + + monkeypatch.setattr(LocalStore, "get", recording_get) if operation == "attrs": group.attrs["x"] = 1 @@ -901,6 +910,8 @@ def resize_keeping_zero(doc: dict[str, Any]) -> None: else: del group["b"] + assert sorted(set(reads)) == sorted(reads) + monkeypatch.undo() assert _stored_chunks(_consolidated_member(path, zarr_format, member)) == [10] reopened = zarr.open_group(path, mode="r", use_consolidated=True) array = reopened[member] @@ -908,13 +919,7 @@ def resize_keeping_zero(doc: dict[str, Any]) -> None: assert (array.shape, array.chunks) == ((10,), (10,)) assert _open_strictly(path / member).chunks == (10,) - reads: list[str] = [] - get = LocalStore.get - - async def recording_get(self: LocalStore, key: str, *args: Any, **kwargs: Any) -> Any: - reads.append(key) - return await get(self, key, *args, **kwargs) - + reads.clear() monkeypatch.setattr(LocalStore, "get", recording_get) group.attrs["y"] = 2 assert reads == [] From 4b9f82c22dee55d88a623b1090145014dfda46e0 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Sat, 26 Sep 2026 19:20:54 +0200 Subject: [PATCH 63/76] fix(group): store a group's documents only; keep upgraded members as stored Group writes (attributes, `update_attributes`, deleting a member, `consolidate_metadata`, `create_hierarchy`) read each upgraded member of the consolidated metadata again from its own document and stored its upgrade. Those member writes raced concurrent deletions: two concurrent `delitem`s on a consolidated group recreated a deleted array's metadata document in most runs, which zarr 3.4.0 never did. A group write now stores only the group's own documents and reads none. Array metadata read from a document that had to be upgraded keeps that document (`_stored_document`, set by `mark_upgraded`, replacing the `_stored_document_upgraded` flag), and consolidated metadata stores such a member as it was stored. Every reader of the consolidated metadata then reads the member as upgraded again, and the array's own first chunk write stores the upgrade of its current document (or refuses a changed chunk grid) as before: the only write that upgrades a member document. Removed: the refresh and adoption of consolidated members in `save_metadata` (`_refresh_consolidated`, `_refresh_array`, `_Refreshed`), and `Group.update_attributes_async`'s detour through `save_metadata`. Tests pinning the refresh are replaced by tests that a group write reads nothing and writes only the group's documents, stores the member's consolidated copy byte for byte as stored, that concurrent deletions leave no member, and that the first write through consolidated metadata stores the member's upgrade or refuses a changed grid. Assisted-by: ClaudeCode:claude-opus-5-5 Co-Authored-By: Claude Opus 5.5 --- changes/4334.bugfix.md | 2 +- src/zarr/core/array.py | 6 +- src/zarr/core/group.py | 13 +- src/zarr/core/metadata/io.py | 99 +------------ src/zarr/core/metadata/upgrades.py | 18 ++- src/zarr/core/metadata/v2.py | 10 +- src/zarr/core/metadata/v3.py | 10 +- tests/test_array_stateful.py | 2 +- tests/test_metadata/test_upgrades.py | 203 +++++++++++++-------------- 9 files changed, 137 insertions(+), 226 deletions(-) diff --git a/changes/4334.bugfix.md b/changes/4334.bugfix.md index 3828d097eb..6279623347 100644 --- a/changes/4334.bugfix.md +++ b/changes/4334.bugfix.md @@ -4,4 +4,4 @@ A chunk edge length is now always at least 1, while an array extent may be 0. Ev Stored metadata with a regular chunk size of 0 or JSON `false` (Zarr format 2 `chunks`, Zarr format 3 `regular` `chunk_shape`) is now read as one chunk spanning the axis (a multiple of the inner chunk size for sharded arrays, which previously failed to open). zarr-python wrote such sizes for arrays created with a zero-length axis until 3.4, and, from 2.18.7 to 3.2.1, for an explicit chunk size of 0 or `False` on an axis of any length, as in `chunks=(0,)` (those arrays could store no data). On a zero-length axis this opens silently, as before; appending to such an axis then stores chunks of size 1. On an axis of positive length it opens with a `ZarrUserWarning`, which says that the axis holds only the fill value and how to store valid metadata: `array.update_attributes({})`, then `zarr.consolidate_metadata` if the metadata is consolidated. JSON `true`, as zarr-python 3.0 and 3.2 wrote for a chunk size of `True`, is read as 1, silently, in the chunk grid (regular, and the explicit edges and run-length encoded sizes of a rectilinear one) and in the inner chunk shape of every sharding codec, nested or not. A stored rectilinear chunk grid whose edge lengths are integral JSON floats (`[[4.0, 2]]`), as zarr-python 3.2 wrote for float edges, is read with those edges as integers, silently; a float anywhere no release wrote one (a regular chunk shape, the inner chunk shape of a sharding codec, a run-length repeat count, an edge below 1) is rejected. -Writing data to an array read from such metadata (other than an empty selection) first stores the upgrade of the metadata the store then holds, unless another writer has stored valid metadata since, so that other readers find the chunks written; in a `ZipStore` this adds a second entry for the metadata document, as every metadata update does. If the metadata the store then holds lays out chunks differently (another writer resized the array keeping the chunk size of 0, say), the write raises a `ValueError` asking to reopen the array, and stores nothing. Storing a group's metadata, which `zarr.consolidate_metadata` and every change to a group with consolidated metadata do, likewise first stores the upgrade of each such member's metadata and consolidates the member's metadata as the store then holds it. +Writing data to an array read from such metadata (other than an empty selection) first stores the upgrade of the metadata the store then holds, unless another writer has stored valid metadata since, so that other readers find the chunks written; in a `ZipStore` this adds a second entry for the metadata document, as every metadata update does. If the metadata the store then holds lays out chunks differently (another writer resized the array keeping the chunk size of 0, say), the write raises a `ValueError` asking to reopen the array, and stores nothing. This is the only operation that stores the upgrade: a group's consolidated metadata keeps such an array's metadata as it was stored (`zarr.consolidate_metadata` copies it as the array's own document holds it), and changing a group reads and writes no metadata of its members. diff --git a/src/zarr/core/array.py b/src/zarr/core/array.py index 481aec46fb..7b23f3b917 100644 --- a/src/zarr/core/array.py +++ b/src/zarr/core/array.py @@ -1644,7 +1644,7 @@ async def _store_upgraded_document(self) -> None: upgrade. Storing the same upgrade twice is harmless, so concurrent callers need no coordination. """ - if not self.metadata._stored_document_upgraded: + if self.metadata._stored_document is None: return zarr_format = self.metadata.zarr_format documents = await read_documents(self.store_path, ARRAY_DOCUMENTS[zarr_format]) @@ -1658,9 +1658,9 @@ async def _store_upgraded_document(self) -> None: f"The metadata stored for the array at {str(self.store_path)!r} has " "changed since this array was opened: reopen the array to write to it." ) - if current._stored_document_upgraded: + if current._stored_document is not None: await upsert_metadata(self.store_path, current, documents) - object.__setattr__(self.metadata, "_stored_document_upgraded", False) + object.__setattr__(self.metadata, "_stored_document", None) async def _set_selection( self, diff --git a/src/zarr/core/group.py b/src/zarr/core/group.py index cd91582315..4f60856425 100644 --- a/src/zarr/core/group.py +++ b/src/zarr/core/group.py @@ -13,6 +13,7 @@ import zarr.api.asynchronous as async_api from zarr.abc.metadata import Metadata +from zarr.abc.store import Store, set_or_delete from zarr.core._info import GroupInfo from zarr.core._json import buffer_to_json_object, json_to_buffer from zarr.core.array import ( @@ -74,7 +75,6 @@ ) from typing import Any - from zarr.abc.store import Store from zarr.core.array_spec import ArrayConfigLike from zarr.core.buffer import Buffer, BufferPrototype from zarr.core.chunk_key_encodings import ChunkKeyEncodingLike @@ -147,11 +147,16 @@ class ConsolidatedMetadata: must_understand: Literal[False] = False def to_dict(self) -> dict[str, JSON]: + """The consolidated metadata document. An array read from a stored document that + had to be upgraded is written as that document was stored: only the array's + own first chunk write stores its upgrade.""" return { "kind": self.kind, "must_understand": self.must_understand, "metadata": { k: v.to_dict() + if isinstance(v, GroupMetadata) or v._stored_document is None + else dict(v._stored_document) for k, v in sorted( self.flattened_metadata.items(), key=lambda item: ( @@ -2115,7 +2120,9 @@ async def update_attributes_async(self, new_attributes: dict[str, Any]) -> Group new_metadata = replace(self.metadata, attributes=new_attributes) # Write new metadata - await save_metadata(self.store_path, new_metadata) + to_save = new_metadata.to_buffer_dict(default_buffer_prototype()) + awaitables = [set_or_delete(self.store_path / key, value) for key, value in to_save.items()] + await asyncio.gather(*awaitables) async_group = replace(self._async_group, metadata=new_metadata) return replace(self, _async_group=async_group) @@ -3032,8 +3039,6 @@ async def create_hierarchy( ```{'': GroupMetadata, 'a': GroupMetadata, 'b': Groupmetadata}``` After input parsing, this function then creates all the nodes in the hierarchy concurrently. - The metadata of each node is stored as given: the consolidated metadata of a group is - stored as it is, without reading the documents of its members. Arrays and Groups are yielded in the order they are created. This order is not stable and should not be relied on. diff --git a/src/zarr/core/metadata/io.py b/src/zarr/core/metadata/io.py index c477517b54..c5c1c24da3 100644 --- a/src/zarr/core/metadata/io.py +++ b/src/zarr/core/metadata/io.py @@ -1,7 +1,6 @@ from __future__ import annotations import asyncio -from dataclasses import replace from enum import Enum from itertools import zip_longest from typing import TYPE_CHECKING, Final, NamedTuple @@ -12,7 +11,6 @@ from zarr.core.buffer.cpu import buffer_prototype as cpu_buffer_prototype from zarr.core.common import ZARR_JSON, ZARRAY_JSON, ZATTRS_JSON from zarr.core.metadata.upgrades import mark_upgraded, upgrade_array_document -from zarr.core.metadata.v3 import RectilinearChunksDisabledError from zarr.errors import ArrayNotFoundError, ContainsArrayError from zarr.storage._common import StorePath, ensure_no_existing_node @@ -134,87 +132,7 @@ def parse_stored_array(documents: Mapping[str, Buffer], zarr_format: ZarrFormat) else: raise ArrayNotFoundError(f"No Zarr format {zarr_format} array metadata document.") upgraded, readings = upgrade_array_document(stored, zarr_format) - return mark_upgraded(parse_array_metadata(dict(upgraded)), [None] * len(readings), None) - - -async def _refresh_array(store_path: StorePath, member: ArrayMetadata) -> ArrayMetadata: - """The metadata of the array document stored at `store_path` as it now is, after - storing its upgrade if it needs one, for `member`, a member of consolidated metadata - that was read from a document that had to be upgraded; `member` itself if no array - document that can be read is stored there (it was deleted, or replaced by a group).""" - documents = await read_documents(store_path, ARRAY_DOCUMENTS[member.zarr_format]) - try: - current = parse_stored_array(documents, member.zarr_format) - except RectilinearChunksDisabledError: - raise # A document that can be read, but only with the flag. - except (KeyError, TypeError, ValueError): - return member - if current._stored_document_upgraded: - await upsert_metadata(store_path, current, documents) - object.__setattr__(current, "_stored_document_upgraded", False) - return current - - -class _Refreshed(NamedTuple): - """A member of consolidated metadata that `_refresh_consolidated` read again from the - member's own document.""" - - members: dict[str, ArrayMetadata | GroupMetadata] - """The consolidated metadata that holds the member, by member name.""" - name: str - stale: ArrayMetadata - current: ArrayMetadata - - def adopt(self) -> None: - """Replace the stale member with the current one in place, so every handle that - shares this consolidated metadata sees it; unless the member was deleted or - replaced since it was read.""" - if self.members.get(self.name) is self.stale: - self.members[self.name] = self.current - - -async def _refresh_consolidated( - store_path: StorePath, metadata: GroupMetadata -) -> tuple[GroupMetadata, list[_Refreshed]]: - """`metadata` with each member of its consolidated metadata that was read from a - document that had to be upgraded, at any depth, replaced by the metadata of the - member's own document as it now is (see `_refresh_array`), read concurrently, and - the members it replaced: the member document may have changed since, so the - consolidated copy is never stored as if it were valid. `metadata` is left as it is, - for the group to adopt the members (see `_Refreshed.adopt`) once they are stored.""" - from zarr.core.group import GroupMetadata - - consolidated = metadata.consolidated_metadata - if consolidated is None: - return metadata, [] - groups = { - name: member - for name, member in consolidated.metadata.items() - if isinstance(member, GroupMetadata) - } - arrays = { - name: member - for name, member in consolidated.metadata.items() - if not isinstance(member, GroupMetadata) and member._stored_document_upgraded - } - nested, current = await asyncio.gather( - asyncio.gather(*(_refresh_consolidated(store_path / n, m) for n, m in groups.items())), - asyncio.gather(*(_refresh_array(store_path / n, m) for n, m in arrays.items())), - ) - refreshed = [ - _Refreshed(consolidated.metadata, name, stale, new) - for (name, stale), new in zip(arrays.items(), current, strict=True) - if new is not stale - ] - members = dict(consolidated.metadata) - members.update({member.name: member.current for member in refreshed}) - for name, (group, inner) in zip(groups, nested, strict=True): - members[name] = group - refreshed.extend(inner) - if not refreshed: - return metadata, [] - consolidated = replace(consolidated, metadata=members) - return replace(metadata, consolidated_metadata=consolidated), refreshed + return mark_upgraded(parse_array_metadata(dict(upgraded)), stored, [None] * len(readings), None) def _build_parents(store_path: StorePath, zarr_format: ZarrFormat) -> dict[str, GroupMetadata]: @@ -239,9 +157,7 @@ def _build_parents(store_path: StorePath, zarr_format: ZarrFormat) -> dict[str, async def save_metadata( store_path: StorePath, metadata: ArrayMetadata | GroupMetadata, ensure_parents: bool = False ) -> None: - """Asynchronously save the array or group metadata. A group adopts, in place, the - members of its consolidated metadata that had to be read again to store it (see - `_refresh_consolidated`), once they are stored. + """Asynchronously save the array or group metadata. Parameters ---------- @@ -256,14 +172,8 @@ async def save_metadata( ------ ValueError """ - from zarr.core.group import GroupMetadata - - refreshed: list[_Refreshed] = [] - if isinstance(metadata, GroupMetadata): - # The one place group metadata is stored, and with it consolidated metadata. - metadata, refreshed = await _refresh_consolidated(store_path, metadata) to_save = metadata.to_buffer_dict(default_buffer_prototype()) - set_awaitables = [store_documents(store_path, to_save)] + set_awaitables = [set_or_delete(store_path / key, value) for key, value in to_save.items()] if ensure_parents: # To enable zarr.create(store, path="a/b/c"), we need to create all the intermediate groups. @@ -301,6 +211,3 @@ async def save_metadata( ) from e await asyncio.gather(*set_awaitables) - # Only once stored: a group whose store failed keeps the members flagged, to read again. - for member in refreshed: - member.adopt() diff --git a/src/zarr/core/metadata/upgrades.py b/src/zarr/core/metadata/upgrades.py index 3b09dab496..728a0e53f5 100644 --- a/src/zarr/core/metadata/upgrades.py +++ b/src/zarr/core/metadata/upgrades.py @@ -18,6 +18,7 @@ from __future__ import annotations +import copy import json import warnings from collections.abc import Callable, Iterable, Mapping, Sequence @@ -43,14 +44,17 @@ ) -def mark_upgraded[M](metadata: M, readings: Sequence[str | None], path: str | None) -> M: - """Record that `metadata` was read from a stored document that needed the upgrades - whose `readings` `upgrade_array_document` returned, if any: set - `_stored_document_upgraded` on `metadata`, so the array stores the upgrade before - it writes chunks under it, and warn once with the readings that are warnings, - naming the array at `path` when the caller knows it.""" +def mark_upgraded[M]( + metadata: M, stored: ArrayDocument, readings: Sequence[str | None], path: str | None +) -> M: + """Record that `metadata` was read from the document `stored`, which needed the + upgrades whose `readings` `upgrade_array_document` returned, if any: keep a copy of + `stored` as `_stored_document` on `metadata`, so the array stores the upgrade before + it writes chunks under it and consolidated metadata stores it as it was stored, and + warn once with the readings that are warnings, naming the array at `path` when the + caller knows it.""" if readings: - object.__setattr__(metadata, "_stored_document_upgraded", True) + object.__setattr__(metadata, "_stored_document", copy.deepcopy(stored)) if messages := [reading for reading in readings if reading is not None]: subject = "" if path is None else f"Array {path!r}: " # The synchronous API parses metadata on zarr's IO thread, whose stack holds no diff --git a/src/zarr/core/metadata/v2.py b/src/zarr/core/metadata/v2.py index 7df14a513f..7f6a33e93b 100644 --- a/src/zarr/core/metadata/v2.py +++ b/src/zarr/core/metadata/v2.py @@ -25,6 +25,7 @@ TBaseScalar, ZDType, ) + from zarr.core.metadata.upgrades import ArrayDocument from dataclasses import dataclass, field, fields, replace @@ -72,10 +73,9 @@ class ArrayV2Metadata(Metadata): compressor: Numcodec | None attributes: dict[str, JSON] = field(default_factory=dict) zarr_format: Literal[2] = field(init=False, default=2) - _stored_document_upgraded: ClassVar[bool] = False - """Whether `from_dict` read this metadata from a stored document it had to upgrade - (set on the instance by `mark_upgraded`), so the store may still hold that invalid - document.""" + _stored_document: ClassVar[ArrayDocument | None] = None + """The stored document `from_dict` read this metadata from, if it had to upgrade it + (set on the instance by `mark_upgraded`): the store may still hold it.""" def __init__( self, @@ -209,7 +209,7 @@ def from_dict(cls, data: dict[str, Any], *, path: str | None = None) -> ArrayV2M _data = {k: v for k, v in _data.items() if k in expected} - return mark_upgraded(cls(**_data), readings, path) + return mark_upgraded(cls(**_data), data, readings, path) def to_dict(self) -> dict[str, JSON]: zarray_dict = super().to_dict() diff --git a/src/zarr/core/metadata/v3.py b/src/zarr/core/metadata/v3.py index 17e88d17e4..b654c36978 100644 --- a/src/zarr/core/metadata/v3.py +++ b/src/zarr/core/metadata/v3.py @@ -52,6 +52,7 @@ from zarr.core.buffer import Buffer, BufferPrototype from zarr.core.chunk_grids import ChunkGrid from zarr.core.dtype.wrapper import TBaseDType, TBaseScalar + from zarr.core.metadata.upgrades import ArrayDocument def parse_zarr_format(data: object) -> Literal[3]: @@ -472,10 +473,9 @@ class ArrayV3Metadata(Metadata): node_type: Literal["array"] = field(default="array", init=False) storage_transformers: tuple[dict[str, JSON], ...] extra_fields: dict[str, AllowedExtraField] - _stored_document_upgraded: ClassVar[bool] = False - """Whether `from_dict` read this metadata from a stored document it had to upgrade - (set on the instance by `mark_upgraded`), so the store may still hold that invalid - document.""" + _stored_document: ClassVar[ArrayDocument | None] = None + """The stored document `from_dict` read this metadata from, if it had to upgrade it + (set on the instance by `mark_upgraded`): the store may still hold it.""" def __init__( self, @@ -676,7 +676,7 @@ def from_dict(cls, data: dict[str, JSON], *, path: str | None = None) -> Self: extra_fields=allowed_extra_fields, storage_transformers=_data_typed.get("storage_transformers", ()), # type: ignore[arg-type] ) - return mark_upgraded(metadata, readings, path) + return mark_upgraded(metadata, data, readings, path) def to_dict(self) -> dict[str, JSON]: out_dict = super().to_dict() diff --git a/tests/test_array_stateful.py b/tests/test_array_stateful.py index 109d65eec1..5319dd6f89 100644 --- a/tests/test_array_stateful.py +++ b/tests/test_array_stateful.py @@ -172,7 +172,7 @@ def _open(self) -> zarr.Array[Any]: warned = any(issubclass(w.category, ZarrUserWarning) for w in record) must_warn = any(self.shape[axis] > 0 for axis in self.legacy_axes) assert warned is must_warn, [str(w.message) for w in record] - assert arr.metadata._stored_document_upgraded is bool(self.legacy_axes) + assert (arr.metadata._stored_document is not None) is bool(self.legacy_axes) return arr # ----------------------------------------------------------------- model diff --git a/tests/test_metadata/test_upgrades.py b/tests/test_metadata/test_upgrades.py index c2b76b12aa..3274351de5 100644 --- a/tests/test_metadata/test_upgrades.py +++ b/tests/test_metadata/test_upgrades.py @@ -195,7 +195,7 @@ def test_upgrade_array_document( warnings.simplefilter("always") metadata = metadata_cls.from_dict(dict(doc), path="group/array") assert _chunk_shapes(metadata) == expected - assert metadata._stored_document_upgraded is upgraded + assert metadata._stored_document == (doc if upgraded else None) messages = [str(w.message) for w in record] if warning is None: assert messages == [] @@ -312,7 +312,7 @@ def test_read_invalid_edges_in_rectilinear_grid( with zarr.config.set({"array.rectilinear_chunks": True}): metadata = _read_strictly(doc) assert metadata.chunk_grid == RectilinearChunkGridMetadata(chunk_shapes=expected) - assert metadata._stored_document_upgraded is upgraded + assert metadata._stored_document == (doc if upgraded else None) @pytest.mark.parametrize( @@ -660,7 +660,7 @@ async def write_twice() -> None: sync(write_twice()) - assert not arr.metadata._stored_document_upgraded + assert arr.metadata._stored_document is None with warnings.catch_warnings(): warnings.simplefilter("error", ZarrUserWarning) reopened = zarr.open_array(store=path, mode="r") @@ -700,7 +700,7 @@ def test_stale_handle_write_keeps_newer_metadata( stale[0] = 9 - assert not stale.metadata._stored_document_upgraded + assert stale.metadata._stored_document is None reopened = _open_strictly(path) assert reopened.shape == (9,) assert reopened.attrs.asdict() == {"x": 1} @@ -776,7 +776,7 @@ def test_stale_handle_write_after_chunk_grid_change_raises( with pytest.raises(ValueError, match="has changed since this array was opened: reopen"): stale[0:3] = [7, 8, 9] - assert stale.metadata._stored_document_upgraded + assert stale.metadata._stored_document is not None assert {p.name: p.read_bytes() for p in path.iterdir()} == documents @@ -788,11 +788,11 @@ def test_write_without_stored_document(zarr_format: Literal[2, 3]) -> None: store = MemoryStore() doc = _v2_doc([3], [True]) if zarr_format == 2 else _v3_doc([3], [True]) array = zarr.Array(AsyncArray.from_dict(StorePath(store), doc)) - upgraded = array.metadata._stored_document_upgraded + upgraded = array.metadata._stored_document is not None array[:] = [1, 2, 3] - assert (upgraded, array.metadata._stored_document_upgraded) == (True, False) + assert (upgraded, array.metadata._stored_document) == (True, None) np.testing.assert_array_equal(array[:], [1, 2, 3]) assert not [key for key in store._store_dict if key.endswith((".zarray", "zarr.json"))] @@ -867,128 +867,144 @@ def _flagged_consolidated_group(path: Path, zarr_format: Literal[2, 3], member: _rewrite_consolidated(path, zarr_format, member, _store_zero) +def _record_store_access(monkeypatch: pytest.MonkeyPatch) -> list[tuple[str, str]]: + """Record every `get` and `set` of a `LocalStore`, as `(method, key)`.""" + accesses: list[tuple[str, str]] = [] + for method in ("get", "set"): + original = getattr(LocalStore, method) + + async def recording( + self: LocalStore, + key: str, + *args: Any, + _method: str = method, + _original: Any = original, + **kwargs: Any, + ) -> Any: + accesses.append((_method, key)) + return await _original(self, key, *args, **kwargs) + + monkeypatch.setattr(LocalStore, method, recording) + return accesses + + @pytest.mark.filterwarnings("ignore:Consolidated metadata is currently not part:UserWarning") @pytest.mark.parametrize("zarr_format", [2, 3]) @pytest.mark.parametrize("member", ["a", "g/a"]) @pytest.mark.parametrize("operation", ["attrs", "update_attributes_async", "delete-member"]) -def test_group_write_refreshes_upgraded_consolidated_member( +def test_group_write_stores_upgraded_consolidated_member_as_stored( tmp_path: Path, monkeypatch: pytest.MonkeyPatch, zarr_format: Literal[2, 3], member: str, operation: str, ) -> None: - """Storing a group's metadata stores its consolidated metadata. Each member of it read - from a document that had to be upgraded, at any depth, is first read again from the - member's own document as it now is (here resized by software that kept the stored - chunk size of 0), whose upgrade is stored, so no group write stores a stale copy as - if valid. The group adopts the members it stored, so later writes read no member.""" + """A group write stores only the group's own documents, reading none: it reads and + writes no member document, and stores the consolidated copy of a member read from a + document that had to be upgraded exactly as it was stored, so every reader of the + consolidated metadata reads that member as upgraded again.""" path = tmp_path / "group.zarr" _flagged_consolidated_group(path, zarr_format, member) - - def resize_keeping_zero(doc: dict[str, Any]) -> None: - _store_zero(doc) - doc["shape"] = [10] - - _rewrite_doc(path / member, zarr_format, resize_keeping_zero) + stored = json.dumps(_consolidated_member(path, zarr_format, member)) with pytest.warns(ZarrUserWarning, match="is read as"): group = zarr.open_group(path, mode="r+", use_consolidated=True) + accesses = _record_store_access(monkeypatch) if operation == "attrs": group.attrs["x"] = 1 elif operation == "update_attributes_async": - group = sync(group.update_attributes_async({"x": 1})) + sync(group.update_attributes_async({"x": 1})) else: del group["b"] - assert _stored_chunks(_consolidated_member(path, zarr_format, member)) == [10] - reopened = zarr.open_group(path, mode="r", use_consolidated=True) - array = reopened[member] - assert isinstance(array, zarr.Array) - assert (array.shape, array.chunks) == ((10,), (10,)) - assert _open_strictly(path / member).chunks == (10,) - - reads: list[str] = [] - get = LocalStore.get - - async def recording_get(self: LocalStore, key: str, *args: Any, **kwargs: Any) -> Any: - reads.append(key) - return await get(self, key, *args, **kwargs) - - monkeypatch.setattr(LocalStore, "get", recording_get) - group.attrs["y"] = 2 - assert reads == [] + own = {"zarr.json"} if zarr_format == 3 else {".zgroup", ".zattrs", ".zmetadata"} + assert {method for method, _ in accesses} == {"set"} + assert {key for _, key in accesses} == own + assert json.dumps(_consolidated_member(path, zarr_format, member)) == stored + with pytest.warns(ZarrUserWarning, match="is read as"): + zarr.open_group(path, mode="r", use_consolidated=True) @pytest.mark.filterwarnings("ignore:Consolidated metadata is currently not part:UserWarning") @pytest.mark.parametrize("zarr_format", [2, 3]) -def test_group_write_keeps_consolidated_metadata_shared_with_subgroups( +def test_concurrent_deletions_leave_no_upgraded_member( tmp_path: Path, zarr_format: Literal[2, 3] ) -> None: - """A subgroup handle taken from a group shares the group's consolidated metadata, also - once the group has adopted the upgraded members its first write stored: an array - deleted through the subgroup is gone from what the group stores next.""" + """Members deleted concurrently through one consolidated group handle stay deleted, + also those read from documents that had to be upgraded: no group write stores a + member document.""" path = tmp_path / "group.zarr" - created = zarr.open_group(path, mode="w", zarr_format=zarr_format).create_group("g") - for name in ("a", "c"): - created.create_array(name, shape=(3,), chunks=(3,), dtype="int16") - zarr.consolidate_metadata(path) - _rewrite_consolidated(path, zarr_format, "g/a", _store_zero) + zarr.open_group(path, mode="w", zarr_format=zarr_format) + for name in ("a", "b"): + _legacy_array(path / name, zarr_format) + with pytest.warns(ZarrUserWarning, match="is read as"): + zarr.consolidate_metadata(path) with pytest.warns(ZarrUserWarning, match="is read as"): group = zarr.open_group(path, mode="r+", use_consolidated=True) - subgroup = group["g"] - assert isinstance(subgroup, zarr.Group) - group.attrs["x"] = 1 - del subgroup["c"] - group.attrs["y"] = 2 + async def delete_both() -> None: + await asyncio.gather(*(group._async_group.delitem(name) for name in ("a", "b"))) + + sync(delete_both()) + + assert not (path / "a").exists() + assert not (path / "b").exists() + - reopened = zarr.open_group(path, mode="r", use_consolidated=True) - assert sorted(dict(reopened.members(max_depth=None))) == ["g", "g/a"] +def _consolidated_legacy_member(path: Path, zarr_format: Literal[2, 3]) -> Path: + """A group whose array `a` is stored with chunk shape `[0]`, consolidated; the path + of the array's own document.""" + zarr.open_group(path, mode="w", zarr_format=zarr_format) + _legacy_array(path / "a", zarr_format) + with pytest.warns(ZarrUserWarning, match="is read as"): + zarr.consolidate_metadata(path) + return path / "a" / (".zarray" if zarr_format == 2 else "zarr.json") @pytest.mark.filterwarnings("ignore:Consolidated metadata is currently not part:UserWarning") -def test_group_write_refuses_upgraded_member_read_as_rectilinear(tmp_path: Path) -> None: - """A member of consolidated metadata read from a document that had to be upgraded, - whose own document now declares a rectilinear chunk grid, is read again only with the - rectilinear chunks flag: without it, a group write raises and stores nothing.""" +@pytest.mark.parametrize("zarr_format", [2, 3]) +def test_consolidated_upgraded_member_stored_by_its_first_write( + tmp_path: Path, zarr_format: Literal[2, 3] +) -> None: + """Consolidating a group copies the document of a member that had to be upgraded as + it is stored, and leaves that document as it is. The first chunk write through the + consolidated metadata stores the member's upgrade before its chunks, the one write + that stores it; consolidating again then copies the upgrade.""" path = tmp_path / "group.zarr" - _flagged_consolidated_group(path, 3, "a") - with pytest.warns(ZarrUserWarning, match="is read as"): - group = zarr.open_group(path, mode="r+", use_consolidated=True) - (path / "a" / "zarr.json").write_text(json.dumps(_rectilinear_doc([3], [[1, 2]]))) - documents = {p: p.read_bytes() for p in path.rglob("*") if p.is_file()} + document = _consolidated_legacy_member(path, zarr_format) + legacy = document.read_bytes() + assert json.dumps(_consolidated_member(path, zarr_format, "a")).encode() == legacy - with pytest.raises(ValueError, match="Rectilinear chunk grids are experimental"): - group.attrs["x"] = 1 + with pytest.warns(ZarrUserWarning, match="is read as"): + array = zarr.open_group(path, mode="r+", use_consolidated=True)["a"] + assert isinstance(array, zarr.Array) + array[:] = [7, 8, 9] - assert {p: p.read_bytes() for p in path.rglob("*") if p.is_file()} == documents + np.testing.assert_array_equal(_open_strictly(path / "a")[...], [7, 8, 9]) + zarr.consolidate_metadata(path) + assert _stored_chunks(_consolidated_member(path, zarr_format, "a")) == [3] @pytest.mark.filterwarnings("ignore:Consolidated metadata is currently not part:UserWarning") @pytest.mark.parametrize("zarr_format", [2, 3]) -@pytest.mark.parametrize("replacement", ["group", "invalid", "none"]) -def test_group_write_keeps_upgraded_member_without_readable_document( - tmp_path: Path, zarr_format: Literal[2, 3], replacement: str +def test_consolidated_upgraded_member_write_after_chunk_grid_change_raises( + tmp_path: Path, zarr_format: Literal[2, 3] ) -> None: - """A member of consolidated metadata read from a document that had to be upgraded, - whose own document is no longer an array document that can be read, has no document - to upgrade: a group write stores the consolidated copy as it is.""" + """A member read from its consolidated copy, whose own document has since been + resized by software that kept the stored chunk size of 0, would write chunks no + reader finds: its first write raises and stores nothing.""" path = tmp_path / "group.zarr" - _flagged_consolidated_group(path, zarr_format, "a") + _consolidated_legacy_member(path, zarr_format) with pytest.warns(ZarrUserWarning, match="is read as"): - group = zarr.open_group(path, mode="r+", use_consolidated=True) - del zarr.open_group(path, mode="r+", use_consolidated=False)["a"] - if replacement == "group": - zarr.create_group(path / "a", zarr_format=zarr_format) - elif replacement == "invalid": - (path / "a").mkdir() - document = path / "a" / (".zarray" if zarr_format == 2 else "zarr.json") - document.write_text(json.dumps({"zarr_format": zarr_format, "node_type": "array"})) + array = zarr.open_group(path, mode="r+", use_consolidated=True)["a"] + assert isinstance(array, zarr.Array) + _rewrite_doc(path / "a", zarr_format, _resize_to_10) + documents = {p: p.read_bytes() for p in path.rglob("*") if p.is_file()} - group.attrs["x"] = 1 + with pytest.raises(ValueError, match="has changed since this array was opened: reopen"): + array[0:3] = [7, 8, 9] - assert _stored_chunks(_consolidated_member(path, zarr_format, "a")) == [3] + assert {p: p.read_bytes() for p in path.rglob("*") if p.is_file()} == documents @pytest.mark.parametrize("zarr_format", [2, 3]) @@ -1002,7 +1018,7 @@ def test_empty_write_stores_no_metadata(tmp_path: Path, zarr_format: Literal[2, array[0:0] = np.empty(0, dtype="int16") - assert array.metadata._stored_document_upgraded + assert array.metadata._stored_document is not None assert {p.name: p.read_bytes() for p in path.iterdir()} == documents @@ -1015,27 +1031,6 @@ def test_async_array_from_dict_names_array(tmp_path: Path, zarr_format: Literal[ AsyncArray.from_dict(store_path, doc) -@pytest.mark.filterwarnings("ignore:Consolidated metadata is currently not part:UserWarning") -@pytest.mark.parametrize("zarr_format", [2, 3]) -def test_consolidate_stores_upgraded_members(tmp_path: Path, zarr_format: Literal[2, 3]) -> None: - """Consolidating a group stores the upgrade of every member document that needed - one before the consolidated document, so chunks written through the consolidated - metadata are stored under a member document that agrees with it.""" - path = tmp_path / "group.zarr" - zarr.open_group(path, mode="w", zarr_format=zarr_format) - _legacy_array(path / "a", zarr_format) - with pytest.warns(ZarrUserWarning, match="is read as"): - zarr.consolidate_metadata(path) - - member = _open_strictly(path / "a") - assert member.chunks == (3,) - group = zarr.open_group(path, mode="r+", use_consolidated=True) - array = group["a"] - assert isinstance(array, zarr.Array) - array[:] = [7, 8, 9] - np.testing.assert_array_equal(_open_strictly(path / "a")[...], [7, 8, 9]) - - @pytest.mark.filterwarnings("ignore:Consolidated metadata is currently not part:UserWarning") @pytest.mark.parametrize("zarr_format", [2, 3]) def test_legacy_chunk_size_consolidated(tmp_path: Path, zarr_format: Literal[2, 3]) -> None: From 65f3f8ee7abeaa104bc60b02eeac0d79992a651f Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Sat, 26 Sep 2026 19:34:20 +0200 Subject: [PATCH 64/76] docs(group): say what consolidated metadata stores for an upgraded array Assisted-by: ClaudeCode:claude-opus-5-5 Co-Authored-By: Claude Opus 5.5 --- src/zarr/core/group.py | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/src/zarr/core/group.py b/src/zarr/core/group.py index 4f60856425..d76106b1cd 100644 --- a/src/zarr/core/group.py +++ b/src/zarr/core/group.py @@ -148,8 +148,8 @@ class ConsolidatedMetadata: def to_dict(self) -> dict[str, JSON]: """The consolidated metadata document. An array read from a stored document that - had to be upgraded is written as that document was stored: only the array's - own first chunk write stores its upgrade.""" + had to be upgraded is written as that document was stored, so every reader of + the consolidated metadata reads it as upgraded again (see `mark_upgraded`).""" return { "kind": self.kind, "must_understand": self.must_understand, From faa6efe0af1165a57ca3764a7e9710ea52571360 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Sat, 26 Sep 2026 20:23:02 +0200 Subject: [PATCH 65/76] fix(array): clear the stored document whenever an array stores its metadata Setting attributes of an array read from a document that had to be upgraded stored the upgraded document with the new attributes, but left the metadata marked with the document it was read from. A consolidated group handle shares that metadata, so its next write stored the old document in the consolidated metadata and the new attributes were lost from it (zarr 3.4.0 kept them). Every write of an array's own documents (creating, resizing, setting attributes) now goes through `AsyncArray._save_metadata`, which clears the mark afterwards, as the first chunk write does after storing the upgrade. Both clear it through `_stored_document_replaced`, the one place that declares it: the first chunk write clears it also when it stores nothing (the store already holds a valid document, or none), so it cannot be folded into the save. The deep copy in `mark_upgraded` stays: the metadata's attributes share objects with the document it was read from, which the caller also holds. A test pins it. Assisted-by: ClaudeCode:claude-opus-5-5 Co-Authored-By: Claude Opus 5.5 --- changes/4334.bugfix.md | 2 +- src/zarr/core/array.py | 20 +++++++++---- tests/test_metadata/test_upgrades.py | 43 ++++++++++++++++++++++++++++ 3 files changed, 58 insertions(+), 7 deletions(-) diff --git a/changes/4334.bugfix.md b/changes/4334.bugfix.md index 6279623347..ea7ec480ce 100644 --- a/changes/4334.bugfix.md +++ b/changes/4334.bugfix.md @@ -4,4 +4,4 @@ A chunk edge length is now always at least 1, while an array extent may be 0. Ev Stored metadata with a regular chunk size of 0 or JSON `false` (Zarr format 2 `chunks`, Zarr format 3 `regular` `chunk_shape`) is now read as one chunk spanning the axis (a multiple of the inner chunk size for sharded arrays, which previously failed to open). zarr-python wrote such sizes for arrays created with a zero-length axis until 3.4, and, from 2.18.7 to 3.2.1, for an explicit chunk size of 0 or `False` on an axis of any length, as in `chunks=(0,)` (those arrays could store no data). On a zero-length axis this opens silently, as before; appending to such an axis then stores chunks of size 1. On an axis of positive length it opens with a `ZarrUserWarning`, which says that the axis holds only the fill value and how to store valid metadata: `array.update_attributes({})`, then `zarr.consolidate_metadata` if the metadata is consolidated. JSON `true`, as zarr-python 3.0 and 3.2 wrote for a chunk size of `True`, is read as 1, silently, in the chunk grid (regular, and the explicit edges and run-length encoded sizes of a rectilinear one) and in the inner chunk shape of every sharding codec, nested or not. A stored rectilinear chunk grid whose edge lengths are integral JSON floats (`[[4.0, 2]]`), as zarr-python 3.2 wrote for float edges, is read with those edges as integers, silently; a float anywhere no release wrote one (a regular chunk shape, the inner chunk shape of a sharding codec, a run-length repeat count, an edge below 1) is rejected. -Writing data to an array read from such metadata (other than an empty selection) first stores the upgrade of the metadata the store then holds, unless another writer has stored valid metadata since, so that other readers find the chunks written; in a `ZipStore` this adds a second entry for the metadata document, as every metadata update does. If the metadata the store then holds lays out chunks differently (another writer resized the array keeping the chunk size of 0, say), the write raises a `ValueError` asking to reopen the array, and stores nothing. This is the only operation that stores the upgrade: a group's consolidated metadata keeps such an array's metadata as it was stored (`zarr.consolidate_metadata` copies it as the array's own document holds it), and changing a group reads and writes no metadata of its members. +Writing data to an array read from such metadata (other than an empty selection) first stores the upgrade of the metadata the store then holds, unless another writer has stored valid metadata since, so that other readers find the chunks written; in a `ZipStore` this adds a second entry for the metadata document, as every metadata update does. If the metadata the store then holds lays out chunks differently (another writer resized the array keeping the chunk size of 0, say), the write raises a `ValueError` asking to reopen the array, and stores nothing. Apart from the array's own metadata writes (`update_attributes`, `resize`), which store the upgrade as before, this is the only operation that stores it: a group's consolidated metadata keeps such an array's metadata as it was stored (`zarr.consolidate_metadata` copies it as the array's own document holds it) until the array stores its upgrade through the group's handle, and changing a group reads and writes no metadata of its members. diff --git a/src/zarr/core/array.py b/src/zarr/core/array.py index 7b23f3b917..cf6420e5b7 100644 --- a/src/zarr/core/array.py +++ b/src/zarr/core/array.py @@ -1625,10 +1625,18 @@ async def get_coordinate_selection( return out_array async def _save_metadata(self, metadata: ArrayMetadata, ensure_parents: bool = False) -> None: - """ - Asynchronously save the array metadata. - """ + """Store `metadata` as this array's own documents: every write of them + (creating, resizing, setting attributes) goes through here.""" await save_metadata(self.store_path, metadata, ensure_parents=ensure_parents) + self._stored_document_replaced() + + def _stored_document_replaced(self) -> None: + """Record that the store no longer holds a document of this array that needs an + upgrade: it holds the upgrade, a valid document, or none. The metadata this handle + holds, which a consolidated group handle may share, then stops standing for the + document it was read from (see `mark_upgraded`), so no later write through either + handle stores that document again.""" + object.__setattr__(self.metadata, "_stored_document", None) async def _store_upgraded_document(self) -> None: """Store the upgrade of this array's current stored document, if it needs one, @@ -1660,7 +1668,7 @@ async def _store_upgraded_document(self) -> None: ) if current._stored_document is not None: await upsert_metadata(self.store_path, current, documents) - object.__setattr__(self.metadata, "_stored_document", None) + self._stored_document_replaced() async def _set_selection( self, @@ -5931,7 +5939,7 @@ async def _delete_key(key: str) -> None: ) # Write new metadata - await save_metadata(array.store_path, new_metadata) + await array._save_metadata(new_metadata) # Update metadata and chunk_grid (in place) object.__setattr__(array, "metadata", new_metadata) @@ -6023,7 +6031,7 @@ async def _update_attributes( array.metadata.attributes.update(new_attributes) # Write new metadata - await save_metadata(array.store_path, array.metadata) + await array._save_metadata(array.metadata) return array diff --git a/tests/test_metadata/test_upgrades.py b/tests/test_metadata/test_upgrades.py index 3274351de5..2698a5f198 100644 --- a/tests/test_metadata/test_upgrades.py +++ b/tests/test_metadata/test_upgrades.py @@ -16,6 +16,7 @@ from zarr.codecs import ShardingCodec from zarr.codecs.numcodecs import Quantize from zarr.core.array import AsyncArray +from zarr.core.group import ConsolidatedMetadata from zarr.core.metadata import ArrayV2Metadata, ArrayV3Metadata from zarr.core.metadata.upgrades import ( RESAVE_HINT, @@ -1007,6 +1008,48 @@ def test_consolidated_upgraded_member_write_after_chunk_grid_change_raises( assert {p: p.read_bytes() for p in path.rglob("*") if p.is_file()} == documents +@pytest.mark.filterwarnings("ignore:Consolidated metadata is currently not part:UserWarning") +@pytest.mark.parametrize("zarr_format", [2, 3]) +def test_consolidated_upgraded_member_attributes_kept_by_group_write( + tmp_path: Path, zarr_format: Literal[2, 3] +) -> None: + """Setting attributes of a member read from consolidated metadata stores the member's + upgraded document with them. A later write of the group, whose consolidated metadata + shares the member's metadata, then stores that upgrade: the new attributes, not the + legacy document as it was stored before.""" + path = tmp_path / "group.zarr" + _consolidated_legacy_member(path, zarr_format) + with pytest.warns(ZarrUserWarning, match="is read as"): + group = zarr.open_group(path, mode="r+", use_consolidated=True) + + group["a"].attrs["x"] = 1 + group.attrs["y"] = 2 + + with warnings.catch_warnings(record=True) as record: + warnings.simplefilter("always") + attributes = dict(zarr.open_group(path, mode="r", use_consolidated=True)["a"].attrs) + assert attributes == {"x": 1} + assert _stored_chunks(_consolidated_member(path, zarr_format, "a")) == [3] + assert not [w for w in record if "is read as" in str(w.message)] + + +@pytest.mark.parametrize("zarr_format", [2, 3]) +def test_upgraded_metadata_keeps_the_document_it_read(zarr_format: Literal[2, 3]) -> None: + """Metadata read from a document that had to be upgraded keeps that document as it + was read, whatever becomes of the caller's dict (or the objects in it, which the + metadata's attributes may hold): consolidated metadata stores it as read.""" + doc = _v2_doc([0], [0]) if zarr_format == 2 else _v3_doc([0], [0]) + doc["attributes"] = {"k": [1]} + read = json.loads(json.dumps(doc)) + metadata_cls = ArrayV2Metadata if zarr_format == 2 else ArrayV3Metadata + metadata = metadata_cls.from_dict(doc) + + cast("list[int]", metadata.attributes["k"]).append(2) + doc["shape"] = [4] + + assert ConsolidatedMetadata(metadata={"a": metadata}).to_dict()["metadata"] == {"a": read} + + @pytest.mark.parametrize("zarr_format", [2, 3]) def test_empty_write_stores_no_metadata(tmp_path: Path, zarr_format: Literal[2, 3]) -> None: """A write of an empty selection stores no chunks, so it stores no metadata either.""" From 30d7ca6282e1f2aede86ecc7d5553f85ad5edef2 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Sat, 26 Sep 2026 20:49:47 +0200 Subject: [PATCH 66/76] test(metadata): pin that a failed metadata save keeps the stored document mark Setting attributes on or resizing an array read from a document that needed an upgrade clears the `_stored_document` mark only after the store accepts the new metadata. A test now injects a store failure for the array's own document and checks the mark and the stored bytes survive, for Zarr formats 2 and 3. The `_save_metadata` docstring now says only what the method does. Assisted-by: ClaudeCode:claude-opus-5-5 Co-Authored-By: Claude Opus 5.5 --- src/zarr/core/array.py | 4 +-- tests/test_metadata/test_upgrades.py | 40 ++++++++++++++++++++++++++++ 2 files changed, 42 insertions(+), 2 deletions(-) diff --git a/src/zarr/core/array.py b/src/zarr/core/array.py index cf6420e5b7..e402382a33 100644 --- a/src/zarr/core/array.py +++ b/src/zarr/core/array.py @@ -1625,8 +1625,8 @@ async def get_coordinate_selection( return out_array async def _save_metadata(self, metadata: ArrayMetadata, ensure_parents: bool = False) -> None: - """Store `metadata` as this array's own documents: every write of them - (creating, resizing, setting attributes) goes through here.""" + """Store `metadata` as this array's own documents, then clear the + `_stored_document` mark (see `_stored_document_replaced`).""" await save_metadata(self.store_path, metadata, ensure_parents=ensure_parents) self._stored_document_replaced() diff --git a/tests/test_metadata/test_upgrades.py b/tests/test_metadata/test_upgrades.py index 2698a5f198..cf33aa2f15 100644 --- a/tests/test_metadata/test_upgrades.py +++ b/tests/test_metadata/test_upgrades.py @@ -781,6 +781,46 @@ def test_stale_handle_write_after_chunk_grid_change_raises( assert {p.name: p.read_bytes() for p in path.iterdir()} == documents +def _set_attribute(array: AnyArray) -> None: + array.attrs["x"] = 1 + + +def _grow(array: AnyArray) -> None: + array.resize((9,)) + + +@pytest.mark.parametrize("operation", [_set_attribute, _grow], ids=["attrs", "resize"]) +@pytest.mark.parametrize("zarr_format", [2, 3]) +def test_failed_metadata_save_keeps_stored_document( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, + zarr_format: Literal[2, 3], + operation: Callable[[AnyArray], None], +) -> None: + """If storing an array's metadata fails, the array still stands for the document + the store holds, which still needs its upgrade.""" + path = tmp_path / "legacy.zarr" + _legacy_array(path, zarr_format) + doc_name = ".zarray" if zarr_format == 2 else "zarr.json" + with pytest.warns(ZarrUserWarning, match="is read as"): + arr = zarr.open_array(store=path, mode="r+") + stored = (path / doc_name).read_bytes() + original_set = LocalStore.set + + async def failing_set(self: LocalStore, key: str, *args: Any, **kwargs: Any) -> None: + if key == doc_name: + raise OSError(f"cannot store {key}") + await original_set(self, key, *args, **kwargs) + + monkeypatch.setattr(LocalStore, "set", failing_set) + + with pytest.raises(OSError, match=f"cannot store {re.escape(doc_name)}"): + operation(arr) + + assert arr.metadata._stored_document is not None + assert (path / doc_name).read_bytes() == stored + + @pytest.mark.parametrize("zarr_format", [2, 3]) def test_write_without_stored_document(zarr_format: Literal[2, 3]) -> None: """An array read from an upgraded document that no store holds (as From 6707b45deade0889ff9f1defcf19acbfe08f082c Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Tue, 29 Sep 2026 14:37:24 +0200 Subject: [PATCH 67/76] fix(metadata): read a stored chunk size of 0 as 1, however long the axis A stored chunk size of 0 or `false` was read as one chunk spanning the axis at open time. On an axis that grew after the size was written, that made the chunk size depend on when the array was opened, and it could be very large: `chunks: [0, 100, 100]` on shape (10000, 100, 100) read as one 800 MB chunk. It is now read as the smallest valid chunk edge length (1, or the inner chunk size for a shard), which is what `chunks=-1` gives on a zero-length axis. A resize by software that kept the stored 0 no longer changes the layout a stale handle expects. Assisted-by: ClaudeCode:claude-opus-5-5 --- changes/4334.bugfix.md | 4 +- src/zarr/core/metadata/upgrades.py | 25 ++++---- tests/test_metadata/test_upgrades.py | 85 ++++++++++++++++++---------- 3 files changed, 72 insertions(+), 42 deletions(-) diff --git a/changes/4334.bugfix.md b/changes/4334.bugfix.md index ea7ec480ce..a58230fee3 100644 --- a/changes/4334.bugfix.md +++ b/changes/4334.bugfix.md @@ -2,6 +2,6 @@ A chunk edge length is now always at least 1, while an array extent may be 0. Ev `FixedDimension(size=0, ...)` now raises a `ValueError`. `ArrayV2Metadata(chunks=(0,))` is still accepted and written as given; an array built from such metadata (with `create_hierarchy`, for example) reads the chunk size as a stored chunk size of 0 is read (below). `create_hierarchy` now builds every array and group before it deletes or stores anything, so a node that cannot be built fails with the store untouched. The regular chunk grid metadata class reads a `bool` chunk edge length as the `int` it equals, and rejects a string or a mapping as a chunk shape as a whole. `RectilinearChunkGridMetadata`, which is experimental, now reads a `bool` edge length as the `int` it equals and rejects a NumPy integer or a float edge length such as `4.0` with a `TypeError`; it used to keep them as given, so that it could not store a NumPy integer, could not read back a `bool`, and stored a float as a JSON float, which the Zarr specification does not allow. -Stored metadata with a regular chunk size of 0 or JSON `false` (Zarr format 2 `chunks`, Zarr format 3 `regular` `chunk_shape`) is now read as one chunk spanning the axis (a multiple of the inner chunk size for sharded arrays, which previously failed to open). zarr-python wrote such sizes for arrays created with a zero-length axis until 3.4, and, from 2.18.7 to 3.2.1, for an explicit chunk size of 0 or `False` on an axis of any length, as in `chunks=(0,)` (those arrays could store no data). On a zero-length axis this opens silently, as before; appending to such an axis then stores chunks of size 1. On an axis of positive length it opens with a `ZarrUserWarning`, which says that the axis holds only the fill value and how to store valid metadata: `array.update_attributes({})`, then `zarr.consolidate_metadata` if the metadata is consolidated. JSON `true`, as zarr-python 3.0 and 3.2 wrote for a chunk size of `True`, is read as 1, silently, in the chunk grid (regular, and the explicit edges and run-length encoded sizes of a rectilinear one) and in the inner chunk shape of every sharding codec, nested or not. A stored rectilinear chunk grid whose edge lengths are integral JSON floats (`[[4.0, 2]]`), as zarr-python 3.2 wrote for float edges, is read with those edges as integers, silently; a float anywhere no release wrote one (a regular chunk shape, the inner chunk shape of a sharding codec, a run-length repeat count, an edge below 1) is rejected. +Stored metadata with a regular chunk size of 0 or JSON `false` (Zarr format 2 `chunks`, Zarr format 3 `regular` `chunk_shape`) is now read as chunk size 1 (the inner chunk size for sharded arrays, which previously failed to open), however long the axis is: that is the chunk size `chunks=-1` gives on a zero-length axis, and it does not depend on how far the axis has grown since. zarr-python wrote such sizes for arrays created with a zero-length axis until 3.4, and, from 2.18.7 to 3.2.1, for an explicit chunk size of 0 or `False` on an axis of any length, as in `chunks=(0,)` (those arrays could store no data). On a zero-length axis this opens silently, as before, and appending to it stores chunks of size 1. On an axis of positive length it opens with a `ZarrUserWarning`, which says that the axis holds only the fill value and how to store valid metadata: `array.update_attributes({})`, then `zarr.consolidate_metadata` if the metadata is consolidated. JSON `true`, as zarr-python 3.0 and 3.2 wrote for a chunk size of `True`, is read as 1, silently, in the chunk grid (regular, and the explicit edges and run-length encoded sizes of a rectilinear one) and in the inner chunk shape of every sharding codec, nested or not. A stored rectilinear chunk grid whose edge lengths are integral JSON floats (`[[4.0, 2]]`), as zarr-python 3.2 wrote for float edges, is read with those edges as integers, silently; a float anywhere no release wrote one (a regular chunk shape, the inner chunk shape of a sharding codec, a run-length repeat count, an edge below 1) is rejected. -Writing data to an array read from such metadata (other than an empty selection) first stores the upgrade of the metadata the store then holds, unless another writer has stored valid metadata since, so that other readers find the chunks written; in a `ZipStore` this adds a second entry for the metadata document, as every metadata update does. If the metadata the store then holds lays out chunks differently (another writer resized the array keeping the chunk size of 0, say), the write raises a `ValueError` asking to reopen the array, and stores nothing. Apart from the array's own metadata writes (`update_attributes`, `resize`), which store the upgrade as before, this is the only operation that stores it: a group's consolidated metadata keeps such an array's metadata as it was stored (`zarr.consolidate_metadata` copies it as the array's own document holds it) until the array stores its upgrade through the group's handle, and changing a group reads and writes no metadata of its members. +Writing data to an array read from such metadata (other than an empty selection) first stores the upgrade of the metadata the store then holds, unless another writer has stored valid metadata since, so that other readers find the chunks written; in a `ZipStore` this adds a second entry for the metadata document, as every metadata update does. If the metadata the store then holds lays out chunks differently (another writer stored a different chunk size since, say), the write raises a `ValueError` asking to reopen the array, and stores nothing. Apart from the array's own metadata writes (`update_attributes`, `resize`), which store the upgrade as before, this is the only operation that stores it: a group's consolidated metadata keeps such an array's metadata as it was stored (`zarr.consolidate_metadata` copies it as the array's own document holds it) until the array stores its upgrade through the group's handle, and changing a group reads and writes no metadata of its members. diff --git a/src/zarr/core/metadata/upgrades.py b/src/zarr/core/metadata/upgrades.py index 728a0e53f5..dbe56ae115 100644 --- a/src/zarr/core/metadata/upgrades.py +++ b/src/zarr/core/metadata/upgrades.py @@ -26,7 +26,6 @@ from typing import TYPE_CHECKING, Final, TypeGuard, cast from zarr.core._json import json_equal -from zarr.core.chunk_grids import full_span_chunk_size from zarr.errors import ZarrUserWarning if TYPE_CHECKING: @@ -74,11 +73,13 @@ def _read_chunk_size(size: JSON, span: int | None, unit: int) -> tuple[int, str Returns the edge length and, where the user must act on how it was read, how it was read; `None` if the entry cannot be read, which leaves it for the metadata constructors to check. A JSON int >= 1 is kept, JSON `true` is read as 1, and 0 or - JSON `false` is read as one chunk spanning the axis of length `span`, a multiple of - `unit` (the inner chunk size of a shard): on an axis of positive length no chunk - can have been stored under it, so the array holds only its fill value. `span` is - `None` where no stored 0 is known, as in the inner chunk shape of a sharding codec: - 0 is then left as stored. + JSON `false` is read as `unit`, the smallest chunk edge length the axis can have (1, + or the inner chunk size of a shard), whatever the length `span` of the axis. That is + what `chunks=-1` gives on an axis of length 0, where these sizes were written, and it + does not depend on how far the axis has grown since. On an axis of positive length no + chunk can have been stored under a chunk size of 0, so the array holds only its fill + value. `span` is `None` where no stored 0 is known, as in the inner chunk shape of a + sharding codec: 0 is then left as stored. """ match size: case True: @@ -86,12 +87,14 @@ def _read_chunk_size(size: JSON, span: int | None, unit: int) -> tuple[int, str case int() if size >= 1: return size, None case int() if size == 0 and span is not None: - edge = full_span_chunk_size(span, unit) if span == 0: - return edge, None - return edge, ( - f"one chunk spanning the dimension ({edge}), and as no chunk can be stored " - "under a chunk size of 0, the array holds only its fill value" + return unit, None + return unit, ( + "1, as no chunk can be stored under a chunk size of 0, so the array " + "holds only its fill value" + if unit == 1 + else f"{unit}, the inner chunk size, as no chunk can be stored under a " + "chunk size of 0, so the array holds only its fill value" ) return None diff --git a/tests/test_metadata/test_upgrades.py b/tests/test_metadata/test_upgrades.py index cf33aa2f15..5f17c05c15 100644 --- a/tests/test_metadata/test_upgrades.py +++ b/tests/test_metadata/test_upgrades.py @@ -123,32 +123,32 @@ def _nested_sharded_doc(inner: list[Any], nested: list[Any]) -> dict[str, JSON]: (_v3_doc([5, 4], [True, 4]), ((1, 4),), True, None), ( _v2_doc([3], [0]), - ((3,),), + ((1,),), True, ( - r"^The stored chunk shape \[0\] is invalid: .* read as \[3\], reading 0 in " - r"dimension 0 as one chunk spanning the dimension \(3\), and .* holds only " - r"its fill value\.$" + r"^The stored chunk shape \[0\] is invalid: .* read as \[1\], reading 0 in " + r"dimension 0 as 1, as no chunk can be stored under a chunk size of 0, so the " + r"array holds only its fill value\.$" ), ), ( _v3_doc([4, 3], [4, 0]), - ((4, 3),), + ((4, 1),), True, r"reading 0 in dimension 1 as .* holds only its fill value\.$", ), ( _v2_doc([0, 3], [0, 0]), - ((1, 3),), + ((1, 1),), True, - r"read as \[1, 3\], reading 0 in dimension 1 as .* holds only its fill value\.$", + r"read as \[1, 1\], reading 0 in dimension 1 as .* holds only its fill value\.$", ), (_v3_doc([0], [0], inner=[4]), ((4,), (4,)), True, None), ( _v3_doc([10], [0], inner=[4]), - ((12,), (4,)), + ((4,), (4,)), True, - r"spanning the dimension \(12\), and .* holds only its fill value", + r"as 4, the inner chunk size, as no chunk .* holds only its fill value", ), (_v3_doc([0, 3], [0, 3], inner=[2, 3]), ((2, 3), (2, 3)), True, None), (_v3_doc([5], [True], inner=[True]), ((1,), (1,)), True, None), @@ -177,8 +177,8 @@ def _nested_sharded_doc(inner: list[Any], nested: list[Any]) -> dict[str, JSON]: def test_upgrade_array_document( doc: dict[str, JSON], expected: tuple[Any, ...], upgraded: bool, warning: str | None ) -> None: - """Valid documents pass unchanged. A stored chunk size of 0 or `false` is read as one - chunk spanning the axis (a multiple of the inner chunk when sharded) and `true` as 1, + """Valid documents pass unchanged. A stored chunk size of 0 or `false` is read as 1 + (the inner chunk size when sharded), however long the axis, and `true` as 1, in the chunk shape and in the inner chunk shape of every sharding codec, nested or not. `from_dict` marks the metadata of an upgraded document; it warns once, naming the array, only where a chunk size of 0 was stored for a non-empty axis (which then @@ -554,7 +554,7 @@ def _rewrite_doc(path: Path, zarr_format: Literal[2, 3], edit: Any) -> None: [ (2, (0, 4), [0, 4], None, (1, 4), False), (3, (5,), [True], None, (1,), False), - (3, (10,), [0], (4,), (12,), True), + (3, (10,), [0], (4,), (4,), True), ], ids=["v2-empty-2d", "v3-true", "v3-sharded-grown"], ) @@ -612,7 +612,7 @@ def store_legacy(doc: dict[str, Any]) -> None: @pytest.mark.parametrize( ("zarr_format", "shape", "inner", "expected"), - [(2, (3,), None, (3,)), (3, (3,), None, (3,)), (3, (10,), (4,), (12,))], + [(2, (3,), None, (1,)), (3, (3,), None, (1,)), (3, (10,), (4,), (4,))], ids=["v2", "v3", "v3-sharded"], ) @pytest.mark.parametrize("api", ["sync", "async", "async-concurrent"]) @@ -720,12 +720,12 @@ def test_stale_handle_write_keeps_valid_document_as_written( with pytest.warns(ZarrUserWarning, match="is read as"): stale = zarr.open_array(store=path, mode="r+") # Valid metadata for the same array, as another writer might store it: chunk size - # 3, without the optional members zarr writes, as compact JSON. + # 1, without the optional members zarr writes, as compact JSON. doc_path = path / (".zarray" if zarr_format == 2 else "zarr.json") doc = json.loads(doc_path.read_text()) for optional in ("dimension_separator", "attributes", "storage_transformers"): doc.pop(optional, None) - _stored_chunks(doc)[0] = 3 + _stored_chunks(doc)[0] = 1 doc_path.write_text(json.dumps(doc, separators=(",", ":"))) written = doc_path.read_bytes() @@ -743,14 +743,22 @@ def _resize_to_10(doc: dict[str, Any]) -> None: doc["shape"] = [10] +def _store_chunk_size_3(doc: dict[str, Any]) -> None: + _stored_chunks(doc)[0] = 3 + + def _halve_inner_chunk_shape(doc: dict[str, Any]) -> None: doc["codecs"][0]["configuration"]["chunk_shape"] = [2] @pytest.mark.parametrize( ("zarr_format", "sharded", "change"), - [(2, False, _resize_to_10), (3, False, _resize_to_10), (3, True, _halve_inner_chunk_shape)], - ids=["v2-resized", "v3-resized", "v3-sharded-inner-chunk-shape"], + [ + (2, False, _store_chunk_size_3), + (3, False, _store_chunk_size_3), + (3, True, _halve_inner_chunk_shape), + ], + ids=["v2-chunk-size", "v3-chunk-size", "v3-sharded-inner-chunk-shape"], ) def test_stale_handle_write_after_chunk_grid_change_raises( tmp_path: Path, @@ -759,10 +767,9 @@ def test_stale_handle_write_after_chunk_grid_change_raises( change: Callable[[dict[str, Any]], None], ) -> None: """If the document the store holds when a handle read from an upgraded document - first writes chunks lays out chunks differently from the handle's metadata (the - array was resized by software that kept the stored chunk size of 0, which now reads - as a larger chunk, or its inner chunk shape changed), the handle's chunks would not - be found under it: the write raises and stores nothing.""" + first writes chunks lays out chunks differently from the handle's metadata (another + writer stored a different chunk size, or changed the inner chunk shape), the handle's + chunks would not be found under it: the write raises and stores nothing.""" path = tmp_path / "legacy.zarr" if sharded: zarr.create_array(store=path, shape=(3,), chunks=(4,), shards=(4,), dtype="int16") @@ -781,6 +788,26 @@ def test_stale_handle_write_after_chunk_grid_change_raises( assert {p.name: p.read_bytes() for p in path.iterdir()} == documents +@pytest.mark.parametrize("zarr_format", [2, 3]) +def test_stale_handle_write_after_resize_keeping_zero( + tmp_path: Path, zarr_format: Literal[2, 3] +) -> None: + """A stored chunk size of 0 is read as 1 however long the axis is, so a resize by + software that kept it lays out chunks as the handle does: the handle's first write + stores the upgrade of the resized document, then its chunks.""" + path = tmp_path / "legacy.zarr" + _legacy_array(path, zarr_format) + with pytest.warns(ZarrUserWarning, match="is read as"): + stale = zarr.open_array(store=path, mode="r+") + _rewrite_doc(path, zarr_format, _resize_to_10) + + stale[0:3] = [7, 8, 9] + + reopened = _open_strictly(path) + assert (reopened.shape, reopened.chunks) == ((10,), (1,)) + np.testing.assert_array_equal(reopened[...], [7, 8, 9, 0, 0, 0, 0, 0, 0, 0]) + + def _set_attribute(array: AnyArray) -> None: array.attrs["x"] = 1 @@ -838,7 +865,7 @@ def test_write_without_stored_document(zarr_format: Literal[2, 3]) -> None: assert not [key for key in store._store_dict if key.endswith((".zarray", "zarr.json"))] -@pytest.mark.parametrize(("shape", "expected"), [((0,), (1,)), ((3,), (3,))]) +@pytest.mark.parametrize(("shape", "expected"), [((0,), (1,)), ((3,), (1,))]) def test_array_from_metadata_with_chunk_size_zero(shape: tuple[int], expected: tuple[int]) -> None: """`ArrayV2Metadata` accepts a chunk size of 0, as a stored document may hold it. An array built from such metadata reads it as the upgrades read that document, silently @@ -1023,7 +1050,7 @@ def test_consolidated_upgraded_member_stored_by_its_first_write( np.testing.assert_array_equal(_open_strictly(path / "a")[...], [7, 8, 9]) zarr.consolidate_metadata(path) - assert _stored_chunks(_consolidated_member(path, zarr_format, "a")) == [3] + assert _stored_chunks(_consolidated_member(path, zarr_format, "a")) == [1] @pytest.mark.filterwarnings("ignore:Consolidated metadata is currently not part:UserWarning") @@ -1032,14 +1059,14 @@ def test_consolidated_upgraded_member_write_after_chunk_grid_change_raises( tmp_path: Path, zarr_format: Literal[2, 3] ) -> None: """A member read from its consolidated copy, whose own document has since been - resized by software that kept the stored chunk size of 0, would write chunks no - reader finds: its first write raises and stores nothing.""" + stored with a different chunk size, would write chunks no reader finds: its first + write raises and stores nothing.""" path = tmp_path / "group.zarr" _consolidated_legacy_member(path, zarr_format) with pytest.warns(ZarrUserWarning, match="is read as"): array = zarr.open_group(path, mode="r+", use_consolidated=True)["a"] assert isinstance(array, zarr.Array) - _rewrite_doc(path / "a", zarr_format, _resize_to_10) + _rewrite_doc(path / "a", zarr_format, _store_chunk_size_3) documents = {p: p.read_bytes() for p in path.rglob("*") if p.is_file()} with pytest.raises(ValueError, match="has changed since this array was opened: reopen"): @@ -1069,7 +1096,7 @@ def test_consolidated_upgraded_member_attributes_kept_by_group_write( warnings.simplefilter("always") attributes = dict(zarr.open_group(path, mode="r", use_consolidated=True)["a"].attrs) assert attributes == {"x": 1} - assert _stored_chunks(_consolidated_member(path, zarr_format, "a")) == [3] + assert _stored_chunks(_consolidated_member(path, zarr_format, "a")) == [1] assert not [w for w in record if "is read as" in str(w.message)] @@ -1159,7 +1186,7 @@ def test_legacy_chunk_size_consolidated(tmp_path: Path, zarr_format: Literal[2, ] for array in arrays: assert isinstance(array, zarr.Array) - assert array.chunks == (2,) + assert array.chunks == (1,) array.update_attributes({}) zarr.consolidate_metadata(path) with warnings.catch_warnings(): @@ -1170,4 +1197,4 @@ def test_legacy_chunk_size_consolidated(tmp_path: Path, zarr_format: Literal[2, for name in names: array = reopened[name] assert isinstance(array, zarr.Array) - assert array.chunks == (2,) + assert array.chunks == (1,) From 9e07aafdaac41a59b63d4ab063732f18e7076651 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Tue, 29 Sep 2026 16:20:28 +0200 Subject: [PATCH 68/76] fix(metadata): re-save only upgrades that move chunks, and read non-JSON input as is A stored `true` chunk size or an integral float edge is read as the value zarr already read it as, so chunks written under the upgraded metadata are where every reader looks. Marking such documents for re-saving turned a plain chunk write into a metadata write: in a ZipStore that adds a second `zarr.json` entry (a `UserWarning`, an error under `-W error`), and it opened a race with concurrent metadata writes that 3.4.0 did not have. Each upgrade now reports whether it moves chunks, and only those mark the metadata. The upgrades also detected changes by comparing JSON encodings, so `ArrayV3Metadata.from_dict` given metadata built in code (a codec instance, a NumPy integer in a sharding codec's `chunk_shape`) raised `TypeError`. Changes are now tracked where they are made. Assisted-by: ClaudeCode:claude-opus-5-5 --- changes/4334.bugfix.md | 2 +- src/zarr/core/array.py | 18 +-- src/zarr/core/metadata/io.py | 2 +- src/zarr/core/metadata/upgrades.py | 182 +++++++++++++++++---------- tests/test_metadata/test_upgrades.py | 152 +++++++++++++++------- 5 files changed, 241 insertions(+), 115 deletions(-) diff --git a/changes/4334.bugfix.md b/changes/4334.bugfix.md index a58230fee3..3f665ee60f 100644 --- a/changes/4334.bugfix.md +++ b/changes/4334.bugfix.md @@ -4,4 +4,4 @@ A chunk edge length is now always at least 1, while an array extent may be 0. Ev Stored metadata with a regular chunk size of 0 or JSON `false` (Zarr format 2 `chunks`, Zarr format 3 `regular` `chunk_shape`) is now read as chunk size 1 (the inner chunk size for sharded arrays, which previously failed to open), however long the axis is: that is the chunk size `chunks=-1` gives on a zero-length axis, and it does not depend on how far the axis has grown since. zarr-python wrote such sizes for arrays created with a zero-length axis until 3.4, and, from 2.18.7 to 3.2.1, for an explicit chunk size of 0 or `False` on an axis of any length, as in `chunks=(0,)` (those arrays could store no data). On a zero-length axis this opens silently, as before, and appending to it stores chunks of size 1. On an axis of positive length it opens with a `ZarrUserWarning`, which says that the axis holds only the fill value and how to store valid metadata: `array.update_attributes({})`, then `zarr.consolidate_metadata` if the metadata is consolidated. JSON `true`, as zarr-python 3.0 and 3.2 wrote for a chunk size of `True`, is read as 1, silently, in the chunk grid (regular, and the explicit edges and run-length encoded sizes of a rectilinear one) and in the inner chunk shape of every sharding codec, nested or not. A stored rectilinear chunk grid whose edge lengths are integral JSON floats (`[[4.0, 2]]`), as zarr-python 3.2 wrote for float edges, is read with those edges as integers, silently; a float anywhere no release wrote one (a regular chunk shape, the inner chunk shape of a sharding codec, a run-length repeat count, an edge below 1) is rejected. -Writing data to an array read from such metadata (other than an empty selection) first stores the upgrade of the metadata the store then holds, unless another writer has stored valid metadata since, so that other readers find the chunks written; in a `ZipStore` this adds a second entry for the metadata document, as every metadata update does. If the metadata the store then holds lays out chunks differently (another writer stored a different chunk size since, say), the write raises a `ValueError` asking to reopen the array, and stores nothing. Apart from the array's own metadata writes (`update_attributes`, `resize`), which store the upgrade as before, this is the only operation that stores it: a group's consolidated metadata keeps such an array's metadata as it was stored (`zarr.consolidate_metadata` copies it as the array's own document holds it) until the array stores its upgrade through the group's handle, and changing a group reads and writes no metadata of its members. +Writing data to an array whose stored metadata holds a regular chunk size of 0 or `false` (other than an empty selection) first stores the upgrade of the metadata the store then holds, unless another writer has stored valid metadata since, so that other readers find the chunks written (a `true` chunk size or an integral float edge is read as the value it equals, so a write stores no metadata for it); in a `ZipStore` this adds a second entry for the metadata document, as every metadata update does. If the metadata the store then holds lays out chunks differently (another writer stored a different chunk size since, say), the write raises a `ValueError` asking to reopen the array, and stores nothing. Apart from the array's own metadata writes (`update_attributes`, `resize`), which store the upgrade as before, this is the only operation that stores it: a group's consolidated metadata keeps such an array's metadata as it was stored (`zarr.consolidate_metadata` copies it as the array's own document holds it) until the array stores its upgrade through the group's handle, and changing a group reads and writes no metadata of its members. diff --git a/src/zarr/core/array.py b/src/zarr/core/array.py index bbe2871854..4026589f4e 100644 --- a/src/zarr/core/array.py +++ b/src/zarr/core/array.py @@ -1642,15 +1642,19 @@ async def _store_upgraded_document(self) -> None: """Store the upgrade of this array's current stored document, if it needs one, before chunks are written under this handle's metadata. - Only for metadata read from a document that had to be upgraded (see + Only for metadata read from a document whose upgrade moves chunks (see `zarr.core.metadata.upgrades`). The document is read again, because the store may hold a newer one than this handle's metadata. If that one lays out chunks - differently (the array was resized since by software that kept the invalid chunk - size), this handle would write chunks no reader finds, so it raises and stores - nothing. If it needs no upgrade (the array was re-saved since, possibly by another - implementation), it is left as written; if there is none, there is nothing to - upgrade. Storing the same upgrade twice is harmless, so concurrent callers need - no coordination. + differently (another writer stored a different chunk size since), this handle + would write chunks no reader finds, so it raises and stores nothing. If it needs + no upgrade (the array was re-saved since, possibly by another implementation), it + is left as written; if there is none, there is nothing to upgrade. + + Storing the same upgrade twice is harmless, so handles that write chunks + concurrently need no coordination. The read and the store are not one atomic + step, though: a metadata write by another handle between them (a resize, an + attribute update) is replaced by the upgrade of the document read before it, as + with any two metadata writes that race. """ if self.metadata._stored_document is None: return diff --git a/src/zarr/core/metadata/io.py b/src/zarr/core/metadata/io.py index c5c1c24da3..f3cd232d68 100644 --- a/src/zarr/core/metadata/io.py +++ b/src/zarr/core/metadata/io.py @@ -132,7 +132,7 @@ def parse_stored_array(documents: Mapping[str, Buffer], zarr_format: ZarrFormat) else: raise ArrayNotFoundError(f"No Zarr format {zarr_format} array metadata document.") upgraded, readings = upgrade_array_document(stored, zarr_format) - return mark_upgraded(parse_array_metadata(dict(upgraded)), stored, [None] * len(readings), None) + return mark_upgraded(parse_array_metadata(dict(upgraded)), stored, readings, None, warn=False) def _build_parents(store_path: StorePath, zarr_format: ZarrFormat) -> dict[str, GroupMetadata]: diff --git a/src/zarr/core/metadata/upgrades.py b/src/zarr/core/metadata/upgrades.py index dbe56ae115..0ef0ed41af 100644 --- a/src/zarr/core/metadata/upgrades.py +++ b/src/zarr/core/metadata/upgrades.py @@ -9,9 +9,11 @@ read from the same document before these upgrades existed is silent, so a document that opened without a warning still does. The warnings are given once per document, after the upgraded document has passed the metadata constructor, so an invalid -document raises its own error, not a warning about how it was read. Silent or not, -metadata read from an upgraded document is marked (see `mark_upgraded`), so the array -stores the upgrade before it writes chunks under it. +document raises its own error, not a warning about how it was read. Metadata read +from a document whose upgrade moves chunks (a chunk size read as another size) is +marked (see `mark_upgraded`), silent or not, so the array stores the upgrade before it +writes chunks under it. An upgrade that only respells a value zarr already read the +same way (`true` as 1, `4.0` as 4) moves no chunks, so it is not marked. To read another kind of invalid document, add an upgrade to `ARRAY_UPGRADES`. """ @@ -23,18 +25,30 @@ import warnings from collections.abc import Callable, Iterable, Mapping, Sequence from itertools import chain, repeat -from typing import TYPE_CHECKING, Final, TypeGuard, cast +from typing import TYPE_CHECKING, Final, NamedTuple, TypeGuard, cast -from zarr.core._json import json_equal from zarr.errors import ZarrUserWarning if TYPE_CHECKING: from zarr.core.common import JSON, ZarrFormat type ArrayDocument = Mapping[str, JSON] -type Upgrade = Callable[[ArrayDocument], tuple[ArrayDocument, str | None] | None] -"""Returns `None` if the document needs no upgrade, else the upgraded document and, -if the user must act on how it was read, a warning saying so (else `None`).""" + + +class Reading(NamedTuple): + """How an upgrade read a stored document.""" + + moves_chunks: bool + """Whether chunks written under the upgraded document are not where a reader of the + stored document looks for them, so the array must store the upgrade before it + writes chunks.""" + warning: str | None + """What the user must act on, if anything.""" + + +type Upgrade = Callable[[ArrayDocument], tuple[ArrayDocument, Reading] | None] +"""Returns `None` if the document needs no upgrade, else the upgraded document and how +it was read.""" RESAVE_HINT: Final = ( "To store valid metadata, open the array writable and call `array.update_attributes({})`; " @@ -44,17 +58,23 @@ def mark_upgraded[M]( - metadata: M, stored: ArrayDocument, readings: Sequence[str | None], path: str | None + metadata: M, + stored: ArrayDocument, + readings: Sequence[Reading], + path: str | None, + *, + warn: bool = True, ) -> M: """Record that `metadata` was read from the document `stored`, which needed the - upgrades whose `readings` `upgrade_array_document` returned, if any: keep a copy of - `stored` as `_stored_document` on `metadata`, so the array stores the upgrade before - it writes chunks under it and consolidated metadata stores it as it was stored, and - warn once with the readings that are warnings, naming the array at `path` when the - caller knows it.""" - if readings: + upgrades whose `readings` `upgrade_array_document` returned, if any. If a reading + moves chunks, keep a copy of `stored` as `_stored_document` on `metadata`, so the + array stores the upgrade before it writes chunks under it and consolidated metadata + stores it as it was stored. If `warn`, warn once with the readings' warnings, + naming the array at `path` when the caller knows it.""" + if any(reading.moves_chunks for reading in readings): object.__setattr__(metadata, "_stored_document", copy.deepcopy(stored)) - if messages := [reading for reading in readings if reading is not None]: + messages = [reading.warning for reading in readings if reading.warning is not None] + if warn and messages: subject = "" if path is None else f"Array {path!r}: " # The synchronous API parses metadata on zarr's IO thread, whose stack holds no # user code, so the warning points at the `from_dict` that read the document. @@ -67,10 +87,13 @@ def _is_int_list(value: object) -> TypeGuard[list[int]]: return isinstance(value, list) and all(isinstance(v, int) for v in value) -def _read_chunk_size(size: JSON, span: int | None, unit: int) -> tuple[int, str | None] | None: +def _read_chunk_size( + size: JSON, span: int | None, unit: int +) -> tuple[int, bool, str | None] | None: """Read one entry of a stored regular chunk shape as a chunk edge length. - Returns the edge length and, where the user must act on how it was read, how it was + Returns the edge length, whether it moves chunks (it is not the size a reader of + the stored entry uses), and, where the user must act on how it was read, how it was read; `None` if the entry cannot be read, which leaves it for the metadata constructors to check. A JSON int >= 1 is kept, JSON `true` is read as 1, and 0 or JSON `false` is read as `unit`, the smallest chunk edge length the axis can have (1, @@ -83,90 +106,115 @@ def _read_chunk_size(size: JSON, span: int | None, unit: int) -> tuple[int, str """ match size: case True: - return 1, None + return 1, False, None case int() if size >= 1: - return size, None + return size, False, None case int() if size == 0 and span is not None: if span == 0: - return unit, None - return unit, ( - "1, as no chunk can be stored under a chunk size of 0, so the array " - "holds only its fill value" - if unit == 1 - else f"{unit}, the inner chunk size, as no chunk can be stored under a " - "chunk size of 0, so the array holds only its fill value" + return unit, True, None + return ( + unit, + True, + ( + "1, as no chunk can be stored under a chunk size of 0, so the array " + "holds only its fill value" + if unit == 1 + else f"{unit}, the inner chunk size, as no chunk can be stored under a " + "chunk size of 0, so the array holds only its fill value" + ), ) return None def _read_chunk_shape( stored: JSON, spans: Sequence[int | None], units: Iterable[int] = () -) -> tuple[list[int], str | None] | None: +) -> tuple[list[int], Reading] | None: """Read a stored regular chunk shape, entry by entry (see `_read_chunk_size`), for axes of lengths `spans` whose chunks are multiples of `units` (1 where not given). - Returns the chunk shape and, where the user must act on how an entry was read, a - sentence saying how the chunk shape was read; `None` if it cannot be read. + Returns the chunk shape and how it was read, if any entry was read as another value + (the warning is a sentence saying how the chunk shape was read); `None` if it cannot + be read or needs no upgrade. """ if not (isinstance(stored, list) and len(stored) == len(spans)): return None edges: list[int] = [] + changed = moves_chunks = False readings: list[str] = [] axes = zip(stored, spans, chain(units, repeat(1)), strict=False) for axis, (size, span, unit) in enumerate(axes): read = _read_chunk_size(size, span, unit) if read is None: return None - edge, how = read + edge, moved, how = read edges.append(edge) + changed |= size is True or edge != size + moves_chunks |= moved if how is not None: readings.append(f"{json.dumps(size)} in dimension {axis} as {how}") - if not readings: - return edges, None - return edges, ( + if not changed: + return None + warning = ( f"The stored chunk shape {json.dumps(stored)} is invalid: chunk sizes must be " f"integers of at least 1. It is read as {edges}, reading {'; '.join(readings)}." + if readings + else None ) + return edges, Reading(moves_chunks, warning) -def _invalid_chunk_sizes_v2(doc: ArrayDocument) -> tuple[ArrayDocument, str | None] | None: +def _invalid_chunk_sizes_v2(doc: ArrayDocument) -> tuple[ArrayDocument, Reading] | None: shape = doc.get("shape") if not _is_int_list(shape): return None - stored = doc.get("chunks") - match _read_chunk_shape(stored, shape): - case chunks, reading if not json_equal(chunks, stored): + match _read_chunk_shape(doc.get("chunks"), shape): + case chunks, reading: return {**doc, "chunks": chunks}, reading return None -def _read_codec(codec: JSON) -> JSON: +def _read_codec(codec: JSON) -> JSON | None: """Read a stored codec: the inner chunk shape of a sharding codec is read as a chunk shape with no known axis lengths (see `_read_chunk_shape`), and so are those of the - sharding codecs nested in its codecs.""" + sharding codecs nested in its codecs. Returns `None` if nothing was read as another + value. No stored inner chunk size is read as a different size, so the reading moves + no chunks.""" match codec: case {"name": "sharding_indexed", "configuration": Mapping() as configuration}: - upgraded = dict(configuration) + upgraded: dict[str, JSON] = {} stored = configuration.get("chunk_shape") if isinstance(stored, list) and ( read := _read_chunk_shape(stored, [None] * len(stored)) ): upgraded["chunk_shape"] = read[0] - if isinstance(codecs := configuration.get("codecs"), list): - upgraded["codecs"] = [_read_codec(inner) for inner in codecs] + if isinstance(codecs := configuration.get("codecs"), list) and ( + inner := _read_codecs(codecs) + ): + upgraded["codecs"] = inner + if not upgraded: + return None # The mapping pattern does not narrow `codec` for mypy. - return {**cast("Mapping[str, JSON]", codec), "configuration": upgraded} - return codec + return { + **cast("Mapping[str, JSON]", codec), + "configuration": {**configuration, **upgraded}, + } + return None -def _invalid_inner_chunk_sizes_v3(doc: ArrayDocument) -> tuple[ArrayDocument, str | None] | None: - stored = doc.get("codecs") - if not isinstance(stored, list): +def _read_codecs(codecs: list[JSON]) -> list[JSON] | None: + """Read stored codecs (see `_read_codec`); `None` if nothing was read as another + value.""" + read = [_read_codec(codec) for codec in codecs] + if all(codec is None for codec in read): return None - codecs = [_read_codec(codec) for codec in stored] - if json_equal(codecs, stored): + return [stored if new is None else new for stored, new in zip(codecs, read, strict=True)] + + +def _invalid_inner_chunk_sizes_v3(doc: ArrayDocument) -> tuple[ArrayDocument, Reading] | None: + stored = doc.get("codecs") + if not isinstance(stored, list) or (codecs := _read_codecs(stored)) is None: return None - return {**doc, "codecs": codecs}, None + return {**doc, "codecs": codecs}, Reading(moves_chunks=False, warning=None) def _inner_chunk_shape(doc: ArrayDocument) -> list[int] | None: @@ -182,7 +230,7 @@ def _inner_chunk_shape(doc: ArrayDocument) -> list[int] | None: return [] -def _invalid_chunk_sizes_v3(doc: ArrayDocument) -> tuple[ArrayDocument, str | None] | None: +def _invalid_chunk_sizes_v3(doc: ArrayDocument) -> tuple[ArrayDocument, Reading] | None: grid = doc.get("chunk_grid") shape = doc.get("shape") if not (isinstance(grid, Mapping) and grid.get("name") == "regular" and _is_int_list(shape)): @@ -192,9 +240,8 @@ def _invalid_chunk_sizes_v3(doc: ArrayDocument) -> tuple[ArrayDocument, str | No configuration = grid.get("configuration") if not isinstance(configuration, Mapping): return None - stored = configuration.get("chunk_shape") - match _read_chunk_shape(stored, shape, units): - case chunk_shape, reading if not json_equal(chunk_shape, stored): + match _read_chunk_shape(configuration.get("chunk_shape"), shape, units): + case chunk_shape, reading: upgraded = {**configuration, "chunk_shape": chunk_shape} return {**doc, "chunk_grid": {**grid, "configuration": upgraded}}, reading return None @@ -230,7 +277,16 @@ def _read_rectilinear_entry(entry: JSON) -> JSON: return _read_edge_length(entry) -def _invalid_edge_lengths_v3(doc: ArrayDocument) -> tuple[ArrayDocument, str | None] | None: +def _respelled(read: JSON, stored: JSON) -> bool: + """Whether reading `stored` as `read` gave a value of another type anywhere in it + (`true` read as 1, `4.0` as 4). Compares types, not encodings, so a value that is + not JSON (a NumPy integer in metadata built in code) is simply kept.""" + if isinstance(stored, list) and isinstance(read, list): + return any(_respelled(r, s) for r, s in zip(read, stored, strict=True)) + return type(read) is not type(stored) + + +def _invalid_edge_lengths_v3(doc: ArrayDocument) -> tuple[ArrayDocument, Reading] | None: grid = doc.get("chunk_grid") if not (isinstance(grid, Mapping) and grid.get("name") == "rectilinear"): return None @@ -241,10 +297,11 @@ def _invalid_edge_lengths_v3(doc: ArrayDocument) -> tuple[ArrayDocument, str | N if not isinstance(stored, list): return None read = [_read_rectilinear_axis(axis) for axis in stored] - if json_equal(read, stored): + if not _respelled(read, stored): return None upgraded = {**configuration, "chunk_shapes": read} - return {**doc, "chunk_grid": {**grid, "configuration": upgraded}}, None + # A stored edge is read as the value it equals, so the reading moves no chunks. + return {**doc, "chunk_grid": {**grid, "configuration": upgraded}}, Reading(False, None) ARRAY_UPGRADES: Final[Mapping[ZarrFormat, tuple[Upgrade, ...]]] = { @@ -259,14 +316,13 @@ def _invalid_edge_lengths_v3(doc: ArrayDocument) -> tuple[ArrayDocument, str | N def upgrade_array_document( doc: ArrayDocument, zarr_format: ZarrFormat -) -> tuple[ArrayDocument, list[str | None]]: +) -> tuple[ArrayDocument, list[Reading]]: """Apply the upgrades for `zarr_format` to a stored array metadata document. - Returns the upgraded document and the reading of each upgrade that changed it (a - warning, or `None` for a silent one), for `mark_upgraded` once the document has - been validated. + Returns the upgraded document and the reading of each upgrade that changed it, for + `mark_upgraded` once the document has been validated. """ - readings: list[str | None] = [] + readings: list[Reading] = [] for upgrade in ARRAY_UPGRADES[zarr_format]: upgraded = upgrade(doc) if upgraded is not None: diff --git a/tests/test_metadata/test_upgrades.py b/tests/test_metadata/test_upgrades.py index 5f17c05c15..4588a90eaf 100644 --- a/tests/test_metadata/test_upgrades.py +++ b/tests/test_metadata/test_upgrades.py @@ -7,6 +7,7 @@ import json import re import warnings +import zipfile from typing import TYPE_CHECKING, Any, Literal, cast import numpy as np @@ -114,17 +115,17 @@ def _nested_sharded_doc(inner: list[Any], nested: list[Any]) -> dict[str, JSON]: @pytest.mark.parametrize( ("doc", "expected", "upgraded", "warning"), [ - (_v2_doc([10, 10], [4, 5]), ((4, 5),), False, None), - (_v3_doc([0, 0], [1, 1]), ((1, 1),), False, None), - (_v3_doc([10], [4], inner=[2]), ((4,), (2,)), False, None), - (_v2_doc([0, 4], [0, 4]), ((1, 4),), True, None), - (_v3_doc([0], [False]), ((1,),), True, None), - (_v2_doc([5], [True]), ((1,),), True, None), - (_v3_doc([5, 4], [True, 4]), ((1, 4),), True, None), + (_v2_doc([10, 10], [4, 5]), ((4, 5),), "no", None), + (_v3_doc([0, 0], [1, 1]), ((1, 1),), "no", None), + (_v3_doc([10], [4], inner=[2]), ((4,), (2,)), "no", None), + (_v2_doc([0, 4], [0, 4]), ((1, 4),), "moved", None), + (_v3_doc([0], [False]), ((1,),), "moved", None), + (_v2_doc([5], [True]), ((1,),), "respelled", None), + (_v3_doc([5, 4], [True, 4]), ((1, 4),), "respelled", None), ( _v2_doc([3], [0]), ((1,),), - True, + "moved", ( r"^The stored chunk shape \[0\] is invalid: .* read as \[1\], reading 0 in " r"dimension 0 as 1, as no chunk can be stored under a chunk size of 0, so the " @@ -134,26 +135,26 @@ def _nested_sharded_doc(inner: list[Any], nested: list[Any]) -> dict[str, JSON]: ( _v3_doc([4, 3], [4, 0]), ((4, 1),), - True, + "moved", r"reading 0 in dimension 1 as .* holds only its fill value\.$", ), ( _v2_doc([0, 3], [0, 0]), ((1, 1),), - True, + "moved", r"read as \[1, 1\], reading 0 in dimension 1 as .* holds only its fill value\.$", ), - (_v3_doc([0], [0], inner=[4]), ((4,), (4,)), True, None), + (_v3_doc([0], [0], inner=[4]), ((4,), (4,)), "moved", None), ( _v3_doc([10], [0], inner=[4]), ((4,), (4,)), - True, + "moved", r"as 4, the inner chunk size, as no chunk .* holds only its fill value", ), - (_v3_doc([0, 3], [0, 3], inner=[2, 3]), ((2, 3), (2, 3)), True, None), - (_v3_doc([5], [True], inner=[True]), ((1,), (1,)), True, None), - (_nested_sharded_doc([4], [2]), ((8,), (4,), (2,)), False, None), - (_nested_sharded_doc([4], [True]), ((8,), (4,), (1,)), True, None), + (_v3_doc([0, 3], [0, 3], inner=[2, 3]), ((2, 3), (2, 3)), "moved", None), + (_v3_doc([5], [True], inner=[True]), ((1,), (1,)), "respelled", None), + (_nested_sharded_doc([4], [2]), ((8,), (4,), (2,)), "no", None), + (_nested_sharded_doc([4], [True]), ((8,), (4,), (1,)), "respelled", None), ], ids=[ "v2-valid", @@ -175,12 +176,17 @@ def _nested_sharded_doc(inner: list[Any], nested: list[Any]) -> dict[str, JSON]: ], ) def test_upgrade_array_document( - doc: dict[str, JSON], expected: tuple[Any, ...], upgraded: bool, warning: str | None + doc: dict[str, JSON], + expected: tuple[Any, ...], + upgraded: Literal["no", "respelled", "moved"], + warning: str | None, ) -> None: """Valid documents pass unchanged. A stored chunk size of 0 or `false` is read as 1 (the inner chunk size when sharded), however long the axis, and `true` as 1, in the chunk shape and in the inner chunk shape of every sharding codec, nested or - not. `from_dict` marks the metadata of an upgraded document; it warns once, naming + not. `from_dict` marks the metadata of a document whose upgrade moves chunks (a + chunk size read as another size), so the array stores the upgrade before it writes + chunks, but not one that only respells a value (`true` as 1). It warns once, naming the array, only where a chunk size of 0 was stored for a non-empty axis (which then holds only its fill value), saying how that part was read and how to re-save. The other readings give what zarr read before, so they are silent.""" @@ -188,15 +194,15 @@ def test_upgrade_array_document( assert { k: v for k, v in upgraded_doc.items() if k not in ("chunks", "chunk_grid", "codecs") } == {k: v for k, v in doc.items() if k not in ("chunks", "chunk_grid", "codecs")} - assert bool(readings) is upgraded - if not upgraded: + assert bool(readings) is (upgraded != "no") + if upgraded == "no": assert upgraded_doc is doc metadata_cls = ArrayV2Metadata if doc["zarr_format"] == 2 else ArrayV3Metadata with warnings.catch_warnings(record=True) as record: warnings.simplefilter("always") metadata = metadata_cls.from_dict(dict(doc), path="group/array") assert _chunk_shapes(metadata) == expected - assert metadata._stored_document == (doc if upgraded else None) + assert metadata._stored_document == (doc if upgraded == "moved" else None) messages = [str(w.message) for w in record] if warning is None: assert messages == [] @@ -279,16 +285,16 @@ def _rectilinear_doc(shape: list[int], chunk_shapes: list[Any]) -> dict[str, JSO @pytest.mark.parametrize( - ("chunk_shapes", "expected", "upgraded"), + ("chunk_shapes", "expected"), [ - ([[[4, 2]], [[5, 2]]], ((4, 4), (5, 5)), False), - ([[[4.0, 2]], [[5, 2]]], ((4, 4), (5, 5)), True), - ([[3.0, 5.0], [[5, 2]]], ((3, 5), (5, 5)), True), - ([[[4.0, 2], 2.0], [[5, 2]]], ((4, 4, 2), (5, 5)), True), - ([[[4.0, 2]], [10]], ((4, 4), (10,)), True), - ([[[4.0, 2]], [5.0, 5]], ((4, 4), (5, 5)), True), - ([[True, 4], [[5, 2]]], ((1, 4), (5, 5)), True), - ([[[True, 2], 3], [[5, 2]]], ((1, 1, 3), (5, 5)), True), + ([[[4, 2]], [[5, 2]]], ((4, 4), (5, 5))), + ([[[4.0, 2]], [[5, 2]]], ((4, 4), (5, 5))), + ([[3.0, 5.0], [[5, 2]]], ((3, 5), (5, 5))), + ([[[4.0, 2], 2.0], [[5, 2]]], ((4, 4, 2), (5, 5))), + ([[[4.0, 2]], [10]], ((4, 4), (10,))), + ([[[4.0, 2]], [5.0, 5]], ((4, 4), (5, 5))), + ([[True, 4], [[5, 2]]], ((1, 4), (5, 5))), + ([[[True, 2], 3], [[5, 2]]], ((1, 1, 3), (5, 5))), ], ids=[ "valid", @@ -302,18 +308,19 @@ def _rectilinear_doc(shape: list[int], chunk_shapes: list[Any]) -> dict[str, JSO ], ) def test_read_invalid_edges_in_rectilinear_grid( - chunk_shapes: list[Any], expected: tuple[tuple[int, ...], ...], upgraded: bool + chunk_shapes: list[Any], expected: tuple[tuple[int, ...], ...] ) -> None: """A stored rectilinear chunk grid whose explicit edges or run-length encoded sizes are integral floats or JSON `true`, as zarr-python wrote them when given float or - `True` edges, is read with those edges as the `int`s they equal. `from_dict` marks - the metadata as upgraded, silently: zarr read these edges so before.""" + `True` edges, is read with those edges as the `int`s they equal, silently: zarr read + these edges so before. The reading moves no chunks, so the metadata is not marked + for re-saving.""" shape = [sum(edges) for edges in expected] doc = _rectilinear_doc(shape, chunk_shapes) with zarr.config.set({"array.rectilinear_chunks": True}): metadata = _read_strictly(doc) assert metadata.chunk_grid == RectilinearChunkGridMetadata(chunk_shapes=expected) - assert metadata._stored_document == (doc if upgraded else None) + assert metadata._stored_document is None @pytest.mark.parametrize( @@ -359,19 +366,19 @@ def test_stored_float_chunk_size_rejected(doc: dict[str, JSON], error: str) -> N @pytest.mark.parametrize( - ("chunks", "stored", "resaved"), + ("chunks", "stored"), [ - ([[4, 4], [5, 5]], [[[4.0, 2]], [[5, 2]]], "[[[4, 2]], [[5, 2]]]"), - ([[1, 3, 4], [5, 5]], [[True, 3, 4], [[5, 2]]], "[[1, 3, 4], [[5, 2]]]"), + ([[4, 4], [5, 5]], [[[4.0, 2]], [[5, 2]]]), + ([[1, 3, 4], [5, 5]], [[True, 3, 4], [[5, 2]]]), ], ids=["float", "true"], ) def test_invalid_edges_round_trip( - tmp_path: Path, chunks: list[list[int]], stored: list[Any], resaved: str + tmp_path: Path, chunks: list[list[int]], stored: list[Any] ) -> None: """A store whose rectilinear chunk grid holds the float or `true` edges zarr-python - wrote opens silently, reads its data, and stores its edges as `int`s before the - first write.""" + wrote opens silently and reads its data. Those edges are read as the values they + equal, so a write stores its chunks and leaves the document as it was stored.""" path = tmp_path / "rectilinear.zarr" data = np.arange(80, dtype="int16").reshape(8, 10) with zarr.config.set({"array.rectilinear_chunks": True}): @@ -383,7 +390,7 @@ def test_invalid_edges_round_trip( np.testing.assert_array_equal(arr[...], data) arr[0, 0] = -1 written = json.loads((path / "zarr.json").read_text())["chunk_grid"]["configuration"] - assert json.dumps(written["chunk_shapes"]) == resaved + assert json.dumps(written["chunk_shapes"]) == json.dumps(stored) data[0, 0] = -1 np.testing.assert_array_equal(_open_strictly(path)[...], data) @@ -854,8 +861,9 @@ def test_write_without_stored_document(zarr_format: Literal[2, 3]) -> None: `AsyncArray.from_dict` builds one) writes its chunks as any array does: there is no stored document to upgrade.""" store = MemoryStore() - doc = _v2_doc([3], [True]) if zarr_format == 2 else _v3_doc([3], [True]) - array = zarr.Array(AsyncArray.from_dict(StorePath(store), doc)) + doc = _v2_doc([3], [0]) if zarr_format == 2 else _v3_doc([3], [0]) + with pytest.warns(ZarrUserWarning, match="is read as"): + array = zarr.Array(AsyncArray.from_dict(StorePath(store), doc)) upgraded = array.metadata._stored_document is not None array[:] = [1, 2, 3] @@ -865,6 +873,64 @@ def test_write_without_stored_document(zarr_format: Literal[2, 3]) -> None: assert not [key for key in store._store_dict if key.endswith((".zarray", "zarr.json"))] +@pytest.mark.parametrize("zarr_format", [2, 3]) +def test_respelled_document_write_stores_only_chunks( + tmp_path: Path, zarr_format: Literal[2, 3] +) -> None: + """A document whose upgrade only respells a value (a stored chunk size `true`, read + as 1 as zarr read it before) moves no chunks, so a write stores its chunks and no + metadata: in a `ZipStore`, which cannot replace an entry, the write adds no second + metadata entry and gives no warning.""" + path = tmp_path / "legacy.zip" + doc = _v2_doc([4, 4], [True, 4]) if zarr_format == 2 else _v3_doc([4, 4], [True, 4]) + name = ".zarray" if zarr_format == 2 else "zarr.json" + with zipfile.ZipFile(path, "w") as archive: + archive.writestr(name, json.dumps(doc)) + store = zarr.storage.ZipStore(path, mode="a") + with warnings.catch_warnings(): + warnings.simplefilter("error") + array = zarr.open_array(store, mode="a") + array[0] = 5 + store.close() + + entries = [info.filename for info in zipfile.ZipFile(path).infolist()] + assert entries.count(name) == 1 + assert len(entries) == 2 + np.testing.assert_array_equal( + zarr.open_array(zarr.storage.ZipStore(path, mode="r"))[0], [5, 5, 5, 5] + ) + + +@pytest.mark.parametrize( + "codec", + [ + ShardingCodec(chunk_shape=(1,)), + { + "name": "sharding_indexed", + "configuration": { + "chunk_shape": [np.int64(2)], + "codecs": [{"name": "bytes", "configuration": {"endian": "little"}}], + "index_codecs": [ + {"name": "bytes", "configuration": {"endian": "little"}}, + {"name": "crc32c"}, + ], + "index_location": "end", + }, + }, + ], + ids=["codec-instance", "numpy-integer"], +) +def test_from_dict_keeps_values_that_are_not_json(codec: Any) -> None: + """`from_dict` takes metadata built in code as well as stored documents, so the + upgrades read values that are not JSON (codec instances, NumPy integers) as they are, + without encoding them, and the constructors read them as before.""" + doc: dict[str, Any] = _v3_doc([4], [4]) + doc["codecs"] = [codec] + metadata = ArrayV3Metadata.from_dict(doc) + assert metadata._stored_document is None + assert isinstance(metadata.codecs[0], ShardingCodec) + + @pytest.mark.parametrize(("shape", "expected"), [((0,), (1,)), ((3,), (1,))]) def test_array_from_metadata_with_chunk_size_zero(shape: tuple[int], expected: tuple[int]) -> None: """`ArrayV2Metadata` accepts a chunk size of 0, as a stored document may hold it. An From 23231e6b9b914e786c271ce6d07e5e0882ff5cf2 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Tue, 29 Sep 2026 16:26:20 +0200 Subject: [PATCH 69/76] fix(array): encode new array metadata before deleting an existing node With the rectilinear flag checked when metadata is stored rather than when it is built, `zarr.create(..., overwrite=True)` with rectilinear chunks and the flag off deleted the existing array and then raised; zarr 3.4.0 raised before deleting anything. `AsyncArray._create_v2`, `_create_v3` and `init_array` now build and encode the new metadata before `_prepare_overwrite`, so metadata that cannot be stored fails with the store untouched. This also covers `create_array(..., overwrite=True)`, which deleted the existing array before validating its arguments. Assisted-by: ClaudeCode:claude-opus-5-5 --- changes/4375.bugfix.md | 2 +- src/zarr/core/array.py | 18 ++++++++++++------ tests/test_metadata/test_upgrades.py | 18 ++++++++++++++++-- 3 files changed, 29 insertions(+), 9 deletions(-) diff --git a/changes/4375.bugfix.md b/changes/4375.bugfix.md index 4332ff7ae3..07f580785b 100644 --- a/changes/4375.bugfix.md +++ b/changes/4375.bugfix.md @@ -1,3 +1,3 @@ Arrays whose stored `regular` chunk grid mixes chunk sizes with lists of chunk edge lengths, such as `"chunk_shape": [2, [5, 10, 5]]` (written by zarr 3.2.0 and 3.2.1 for `chunks=(2, (5, 10, 5))`, and as `[2, [5.0, 10.0, 5.0]]` for float edges), can be read again, without enabling `array.rectilinear_chunks`. The grid is read as the rectilinear chunk grid it describes, with a `ZarrUserWarning`; re-saving the array's metadata (or writing chunks to it) stores that rectilinear grid, which requires the flag. A group's consolidated metadata keeps such an array's metadata as it was stored. A regular chunk grid given edge lists is now rejected, so this metadata is no longer written. -The `array.rectilinear_chunks` flag now gates reading and storing array metadata that declares a rectilinear chunk grid, including in a group's consolidated metadata, instead of constructing `RectilinearChunkGridMetadata`. `Array.resize`, deleting a group member, and `create_hierarchy(..., overwrite=True)` now encode the metadata they will store before deleting anything, so metadata that cannot be stored (for example, a rectilinear chunk grid with the flag off) fails with the store untouched. That error names the array when it is a member of a group's consolidated metadata. +The `array.rectilinear_chunks` flag now gates reading and storing array metadata that declares a rectilinear chunk grid, including in a group's consolidated metadata, instead of constructing `RectilinearChunkGridMetadata`. `Array.resize`, deleting a group member, `create_hierarchy(..., overwrite=True)`, and creating an array with `overwrite=True` (`zarr.create`, `zarr.create_array` and the functions built on them) now encode the metadata they will store before deleting anything, so metadata that cannot be stored (for example, a rectilinear chunk grid with the flag off) fails with the store untouched. That error names the array when it is a member of a group's consolidated metadata. diff --git a/src/zarr/core/array.py b/src/zarr/core/array.py index e3076664df..b2f2b9cf31 100644 --- a/src/zarr/core/array.py +++ b/src/zarr/core/array.py @@ -646,8 +646,6 @@ async def _create_v3( attributes: dict[str, JSON] | None = None, overwrite: bool = False, ) -> AsyncArrayV3: - await _prepare_overwrite(store_path, zarr_format=3, overwrite=overwrite) - if isinstance(chunk_key_encoding, tuple): chunk_key_encoding = ( V2ChunkKeyEncoding(separator=chunk_key_encoding[1]) @@ -665,6 +663,10 @@ async def _create_v3( dimension_names=dimension_names, attributes=attributes, ) + # Encode first, so metadata that cannot be stored fails before any existing + # node is deleted. + encode_documents(store_path, metadata) + await _prepare_overwrite(store_path, zarr_format=3, overwrite=overwrite) array = cls(metadata=metadata, store_path=store_path, config=config) await array._save_metadata(metadata, ensure_parents=True) @@ -721,8 +723,6 @@ async def _create_v2( attributes: dict[str, JSON] | None = None, overwrite: bool = False, ) -> AsyncArrayV2: - await _prepare_overwrite(store_path, zarr_format=2, overwrite=overwrite) - compressor_parsed: CompressorLikev2 if compressor == "auto": compressor_parsed = default_compressor_v2(dtype) @@ -748,6 +748,10 @@ async def _create_v2( compressor=compressor_parsed, attributes=attributes, ) + # Encode first, so metadata that cannot be stored fails before any existing + # node is deleted. + encode_documents(store_path, metadata) + await _prepare_overwrite(store_path, zarr_format=2, overwrite=overwrite) array = cls(metadata=metadata, store_path=store_path, config=config) await array._save_metadata(metadata, ensure_parents=True) @@ -4591,8 +4595,6 @@ async def init_array( chunk_key_encoding, zarr_format=zarr_format ) - await _prepare_overwrite(store_path, zarr_format=zarr_format, overwrite=overwrite) - # Validate rectilinear chunks constraints if _is_rectilinear_chunks(chunks): if zarr_format == 2: @@ -4705,6 +4707,10 @@ async def init_array( attributes=attributes, ) + # Encode first, so metadata that cannot be stored fails before any existing node is + # deleted. + encode_documents(store_path, meta) + await _prepare_overwrite(store_path, zarr_format=zarr_format, overwrite=overwrite) arr = AsyncArray(metadata=meta, store_path=store_path, config=config) await arr._save_metadata(meta, ensure_parents=True) return arr diff --git a/tests/test_metadata/test_upgrades.py b/tests/test_metadata/test_upgrades.py index 68262ac2d8..8e27ebee80 100644 --- a/tests/test_metadata/test_upgrades.py +++ b/tests/test_metadata/test_upgrades.py @@ -1471,6 +1471,14 @@ def _overwrite_hierarchy(path: Path) -> None: list(zarr.create_hierarchy(store=LocalStore(path), nodes={"n": mixed.metadata}, overwrite=True)) +def _overwrite_with_create(path: Path) -> None: + zarr.create(shape=(4,), chunks=[[2, 2]], dtype="int64", store=path / "n", overwrite=True) + + +def _overwrite_with_create_array(path: Path) -> None: + zarr.create_array(path / "n", shape=(4,), chunks=[[2, 2]], dtype="int64", overwrite=True) + + def _set_group_attribute(path: Path) -> None: zarr.open_group(path, mode="a").attrs["x"] = 1 @@ -1481,8 +1489,14 @@ def _set_group_attribute(path: Path) -> None: @pytest.mark.filterwarnings("ignore:Consolidated metadata is currently not part:UserWarning") @pytest.mark.parametrize( "action", - [_resize, _write_chunks, _overwrite_hierarchy], - ids=["resize", "write", "overwrite-hierarchy"], + [ + _resize, + _write_chunks, + _overwrite_hierarchy, + _overwrite_with_create, + _overwrite_with_create_array, + ], + ids=["resize", "write", "overwrite-hierarchy", "overwrite-create", "overwrite-create-array"], ) def test_store_untouched_without_flag(tmp_path: Path, action: Callable[[Path], None]) -> None: """An operation that would store the rectilinear chunk grid read from the verbatim From d7ff7230fd408a13607a13c6106cb3bd5efb7496 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Tue, 29 Sep 2026 16:58:19 +0200 Subject: [PATCH 70/76] test(metadata): type the rectilinear chunks passed to zarr.create in an overwrite test Assisted-by: ClaudeCode:claude-opus-5-5 --- tests/test_metadata/test_upgrades.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/tests/test_metadata/test_upgrades.py b/tests/test_metadata/test_upgrades.py index 8e27ebee80..95462c7d3d 100644 --- a/tests/test_metadata/test_upgrades.py +++ b/tests/test_metadata/test_upgrades.py @@ -1472,7 +1472,7 @@ def _overwrite_hierarchy(path: Path) -> None: def _overwrite_with_create(path: Path) -> None: - zarr.create(shape=(4,), chunks=[[2, 2]], dtype="int64", store=path / "n", overwrite=True) + zarr.create(shape=(4,), chunks=[[2, 2]], dtype="int64", store=path / "n", overwrite=True) # type: ignore[arg-type] def _overwrite_with_create_array(path: Path) -> None: From 5dccfb80a099646e08b39fe63180b7a2b24f10da Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Wed, 30 Sep 2026 14:41:19 +0200 Subject: [PATCH 71/76] fix(metadata): recommend recreating an array whose stored chunk size was 0 The warning for a stored chunk size of 0 on an axis of positive length now says that the array holds only its fill value, so recreating it with the wanted chunk shape loses nothing, and gives the re-save as the way to keep it instead. A re-save freezes the smallest chunk size into an array that holds no data yet. Each reading now carries its own advice, so mark_upgraded appends no shared hint. Assisted-by: ClaudeCode:claude-fable-5-1 Co-Authored-By: Claude Fable 5.1 --- changes/4334.bugfix.md | 2 +- src/zarr/core/metadata/upgrades.py | 32 +++++++++++++++++----------- tests/test_metadata/test_upgrades.py | 18 ++++++++-------- 3 files changed, 29 insertions(+), 23 deletions(-) diff --git a/changes/4334.bugfix.md b/changes/4334.bugfix.md index 3f665ee60f..dcde76ce73 100644 --- a/changes/4334.bugfix.md +++ b/changes/4334.bugfix.md @@ -2,6 +2,6 @@ A chunk edge length is now always at least 1, while an array extent may be 0. Ev `FixedDimension(size=0, ...)` now raises a `ValueError`. `ArrayV2Metadata(chunks=(0,))` is still accepted and written as given; an array built from such metadata (with `create_hierarchy`, for example) reads the chunk size as a stored chunk size of 0 is read (below). `create_hierarchy` now builds every array and group before it deletes or stores anything, so a node that cannot be built fails with the store untouched. The regular chunk grid metadata class reads a `bool` chunk edge length as the `int` it equals, and rejects a string or a mapping as a chunk shape as a whole. `RectilinearChunkGridMetadata`, which is experimental, now reads a `bool` edge length as the `int` it equals and rejects a NumPy integer or a float edge length such as `4.0` with a `TypeError`; it used to keep them as given, so that it could not store a NumPy integer, could not read back a `bool`, and stored a float as a JSON float, which the Zarr specification does not allow. -Stored metadata with a regular chunk size of 0 or JSON `false` (Zarr format 2 `chunks`, Zarr format 3 `regular` `chunk_shape`) is now read as chunk size 1 (the inner chunk size for sharded arrays, which previously failed to open), however long the axis is: that is the chunk size `chunks=-1` gives on a zero-length axis, and it does not depend on how far the axis has grown since. zarr-python wrote such sizes for arrays created with a zero-length axis until 3.4, and, from 2.18.7 to 3.2.1, for an explicit chunk size of 0 or `False` on an axis of any length, as in `chunks=(0,)` (those arrays could store no data). On a zero-length axis this opens silently, as before, and appending to it stores chunks of size 1. On an axis of positive length it opens with a `ZarrUserWarning`, which says that the axis holds only the fill value and how to store valid metadata: `array.update_attributes({})`, then `zarr.consolidate_metadata` if the metadata is consolidated. JSON `true`, as zarr-python 3.0 and 3.2 wrote for a chunk size of `True`, is read as 1, silently, in the chunk grid (regular, and the explicit edges and run-length encoded sizes of a rectilinear one) and in the inner chunk shape of every sharding codec, nested or not. A stored rectilinear chunk grid whose edge lengths are integral JSON floats (`[[4.0, 2]]`), as zarr-python 3.2 wrote for float edges, is read with those edges as integers, silently; a float anywhere no release wrote one (a regular chunk shape, the inner chunk shape of a sharding codec, a run-length repeat count, an edge below 1) is rejected. +Stored metadata with a regular chunk size of 0 or JSON `false` (Zarr format 2 `chunks`, Zarr format 3 `regular` `chunk_shape`) is now read as chunk size 1 (the inner chunk size for sharded arrays, which previously failed to open), however long the axis is: that is the chunk size `chunks=-1` gives on a zero-length axis, and it does not depend on how far the axis has grown since. zarr-python wrote such sizes for arrays created with a zero-length axis until 3.4, and, from 2.18.7 to 3.2.1, for an explicit chunk size of 0 or `False` on an axis of any length, as in `chunks=(0,)` (those arrays could store no data). On a zero-length axis this opens silently, as before, and appending to it stores chunks of size 1. On an axis of positive length it opens with a `ZarrUserWarning`, which says that the array holds only its fill value, so that recreating it with the wanted chunk shape loses nothing, and how to keep it by storing valid metadata instead: `array.update_attributes({})`, then `zarr.consolidate_metadata` if the metadata is consolidated. JSON `true`, as zarr-python 3.0 and 3.2 wrote for a chunk size of `True`, is read as 1, silently, in the chunk grid (regular, and the explicit edges and run-length encoded sizes of a rectilinear one) and in the inner chunk shape of every sharding codec, nested or not. A stored rectilinear chunk grid whose edge lengths are integral JSON floats (`[[4.0, 2]]`), as zarr-python 3.2 wrote for float edges, is read with those edges as integers, silently; a float anywhere no release wrote one (a regular chunk shape, the inner chunk shape of a sharding codec, a run-length repeat count, an edge below 1) is rejected. Writing data to an array whose stored metadata holds a regular chunk size of 0 or `false` (other than an empty selection) first stores the upgrade of the metadata the store then holds, unless another writer has stored valid metadata since, so that other readers find the chunks written (a `true` chunk size or an integral float edge is read as the value it equals, so a write stores no metadata for it); in a `ZipStore` this adds a second entry for the metadata document, as every metadata update does. If the metadata the store then holds lays out chunks differently (another writer stored a different chunk size since, say), the write raises a `ValueError` asking to reopen the array, and stores nothing. Apart from the array's own metadata writes (`update_attributes`, `resize`), which store the upgrade as before, this is the only operation that stores it: a group's consolidated metadata keeps such an array's metadata as it was stored (`zarr.consolidate_metadata` copies it as the array's own document holds it) until the array stores its upgrade through the group's handle, and changing a group reads and writes no metadata of its members. diff --git a/src/zarr/core/metadata/upgrades.py b/src/zarr/core/metadata/upgrades.py index 0ef0ed41af..7c22f94833 100644 --- a/src/zarr/core/metadata/upgrades.py +++ b/src/zarr/core/metadata/upgrades.py @@ -50,11 +50,15 @@ class Reading(NamedTuple): """Returns `None` if the document needs no upgrade, else the upgraded document and how it was read.""" -RESAVE_HINT: Final = ( - "To store valid metadata, open the array writable and call `array.update_attributes({})`; " - "if a group holds consolidated metadata for the array, then also call " - "`zarr.consolidate_metadata` on that group." +RECREATE_HINT: Final = ( + "The array holds only its fill value, so recreate it with the chunk shape you want; " + "nothing is lost. To keep it instead, store valid metadata: open the array writable and " + "call `array.update_attributes({})`; if a group holds consolidated metadata for the " + "array, then also call `zarr.consolidate_metadata` on that group." ) +"""How to act on a document whose array can hold no data: a stored chunk size of 0 is read +as the smallest chunk size, which is a poor chunk shape for the data the array grows into, +and a re-save would keep it.""" def mark_upgraded[M]( @@ -69,8 +73,9 @@ def mark_upgraded[M]( upgrades whose `readings` `upgrade_array_document` returned, if any. If a reading moves chunks, keep a copy of `stored` as `_stored_document` on `metadata`, so the array stores the upgrade before it writes chunks under it and consolidated metadata - stores it as it was stored. If `warn`, warn once with the readings' warnings, - naming the array at `path` when the caller knows it.""" + stores it as it was stored. If `warn`, warn once with the readings' warnings (each + says what the user must act on and how), naming the array at `path` when the caller + knows it.""" if any(reading.moves_chunks for reading in readings): object.__setattr__(metadata, "_stored_document", copy.deepcopy(stored)) messages = [reading.warning for reading in readings if reading.warning is not None] @@ -78,7 +83,7 @@ def mark_upgraded[M]( subject = "" if path is None else f"Array {path!r}: " # The synchronous API parses metadata on zarr's IO thread, whose stack holds no # user code, so the warning points at the `from_dict` that read the document. - warnings.warn(f"{subject}{' '.join(messages)} {RESAVE_HINT}", ZarrUserWarning, stacklevel=2) + warnings.warn(f"{subject}{' '.join(messages)}", ZarrUserWarning, stacklevel=2) return metadata @@ -116,11 +121,10 @@ def _read_chunk_size( unit, True, ( - "1, as no chunk can be stored under a chunk size of 0, so the array " - "holds only its fill value" + "1, as no chunk can be stored under a chunk size of 0" if unit == 1 else f"{unit}, the inner chunk size, as no chunk can be stored under a " - "chunk size of 0, so the array holds only its fill value" + "chunk size of 0" ), ) return None @@ -133,8 +137,9 @@ def _read_chunk_shape( axes of lengths `spans` whose chunks are multiples of `units` (1 where not given). Returns the chunk shape and how it was read, if any entry was read as another value - (the warning is a sentence saying how the chunk shape was read); `None` if it cannot - be read or needs no upgrade. + (the warning says how the chunk shape was read and, as the array then holds only its + fill value, recommends recreating it, see `RECREATE_HINT`); `None` if it cannot be + read or needs no upgrade. """ if not (isinstance(stored, list) and len(stored) == len(spans)): return None @@ -156,7 +161,8 @@ def _read_chunk_shape( return None warning = ( f"The stored chunk shape {json.dumps(stored)} is invalid: chunk sizes must be " - f"integers of at least 1. It is read as {edges}, reading {'; '.join(readings)}." + f"integers of at least 1. It is read as {edges}, reading {'; '.join(readings)}. " + f"{RECREATE_HINT}" if readings else None ) diff --git a/tests/test_metadata/test_upgrades.py b/tests/test_metadata/test_upgrades.py index 4588a90eaf..3b115fe73d 100644 --- a/tests/test_metadata/test_upgrades.py +++ b/tests/test_metadata/test_upgrades.py @@ -20,7 +20,7 @@ from zarr.core.group import ConsolidatedMetadata from zarr.core.metadata import ArrayV2Metadata, ArrayV3Metadata from zarr.core.metadata.upgrades import ( - RESAVE_HINT, + RECREATE_HINT, upgrade_array_document, ) from zarr.core.metadata.v3 import RectilinearChunkGridMetadata, RegularChunkGridMetadata @@ -128,28 +128,27 @@ def _nested_sharded_doc(inner: list[Any], nested: list[Any]) -> dict[str, JSON]: "moved", ( r"^The stored chunk shape \[0\] is invalid: .* read as \[1\], reading 0 in " - r"dimension 0 as 1, as no chunk can be stored under a chunk size of 0, so the " - r"array holds only its fill value\.$" + r"dimension 0 as 1, as no chunk can be stored under a chunk size of 0\.$" ), ), ( _v3_doc([4, 3], [4, 0]), ((4, 1),), "moved", - r"reading 0 in dimension 1 as .* holds only its fill value\.$", + r"reading 0 in dimension 1 as .* under a chunk size of 0\.$", ), ( _v2_doc([0, 3], [0, 0]), ((1, 1),), "moved", - r"read as \[1, 1\], reading 0 in dimension 1 as .* holds only its fill value\.$", + r"read as \[1, 1\], reading 0 in dimension 1 as .* under a chunk size of 0\.$", ), (_v3_doc([0], [0], inner=[4]), ((4,), (4,)), "moved", None), ( _v3_doc([10], [0], inner=[4]), ((4,), (4,)), "moved", - r"as 4, the inner chunk size, as no chunk .* holds only its fill value", + r"as 4, the inner chunk size, as no chunk can be stored under a chunk size of 0", ), (_v3_doc([0, 3], [0, 3], inner=[2, 3]), ((2, 3), (2, 3)), "moved", None), (_v3_doc([5], [True], inner=[True]), ((1,), (1,)), "respelled", None), @@ -188,7 +187,8 @@ def test_upgrade_array_document( chunk size read as another size), so the array stores the upgrade before it writes chunks, but not one that only respells a value (`true` as 1). It warns once, naming the array, only where a chunk size of 0 was stored for a non-empty axis (which then - holds only its fill value), saying how that part was read and how to re-save. The + holds only its fill value), saying how that part was read and that recreating the + array loses nothing (see `RECREATE_HINT`). The other readings give what zarr read before, so they are silent.""" upgraded_doc, readings = upgrade_array_document(doc, cast("ZarrFormat", doc["zarr_format"])) assert { @@ -209,9 +209,9 @@ def test_upgrade_array_document( else: [message] = messages assert message.startswith("Array 'group/array': ") - assert message.endswith(RESAVE_HINT) + assert message.endswith(RECREATE_HINT) assert re.search( - warning, message.removeprefix("Array 'group/array': ").removesuffix(f" {RESAVE_HINT}") + warning, message.removeprefix("Array 'group/array': ").removesuffix(f" {RECREATE_HINT}") ) From f766fb28e31b908520cfe4b06adcd5ac4e594b87 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Wed, 30 Sep 2026 14:45:35 +0200 Subject: [PATCH 72/76] test(metadata): widen the abbreviated-message bound for the two hints Assisted-by: ClaudeCode:claude-fable-5-1 Co-Authored-By: Claude Fable 5.1 --- tests/test_metadata/test_upgrades.py | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/tests/test_metadata/test_upgrades.py b/tests/test_metadata/test_upgrades.py index d0f2359495..ba86832daf 100644 --- a/tests/test_metadata/test_upgrades.py +++ b/tests/test_metadata/test_upgrades.py @@ -1350,7 +1350,9 @@ def test_read_edge_lists_in_regular_grid( "follows requires `zarr.config.set({'array.rectilinear_chunks': True})`. " ) in message assert message.endswith(RESAVE_HINT) - assert len(message) < 1000 + # The 1000-edge list is abbreviated: the message is two sentences and two hints, not + # a dump of the edges. + assert len(message) < 1300 def _rejected_without_warning(doc: dict[str, JSON]) -> pytest.ExceptionInfo[Exception]: From fcc24bbeee593e21b918261ca69b03d652237e3e Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Wed, 30 Sep 2026 14:58:17 +0200 Subject: [PATCH 73/76] fix(metadata): give the recipe for recreating an array read from a stored chunk size of 0 The warning names the call, zarr.from_array(array.store, name=array.path, data=array, chunks=..., overwrite=True, write_data=False), which keeps the data type, fill value, attributes and codecs, and a test pins the recipe on both formats. Assisted-by: ClaudeCode:claude-fable-5-1 Co-Authored-By: Claude Fable 5.1 --- changes/4334.bugfix.md | 2 +- src/zarr/core/metadata/upgrades.py | 9 ++++-- tests/test_metadata/test_upgrades.py | 45 ++++++++++++++++++++++++++++ 3 files changed, 52 insertions(+), 4 deletions(-) diff --git a/changes/4334.bugfix.md b/changes/4334.bugfix.md index dcde76ce73..fa5ca0c5a8 100644 --- a/changes/4334.bugfix.md +++ b/changes/4334.bugfix.md @@ -2,6 +2,6 @@ A chunk edge length is now always at least 1, while an array extent may be 0. Ev `FixedDimension(size=0, ...)` now raises a `ValueError`. `ArrayV2Metadata(chunks=(0,))` is still accepted and written as given; an array built from such metadata (with `create_hierarchy`, for example) reads the chunk size as a stored chunk size of 0 is read (below). `create_hierarchy` now builds every array and group before it deletes or stores anything, so a node that cannot be built fails with the store untouched. The regular chunk grid metadata class reads a `bool` chunk edge length as the `int` it equals, and rejects a string or a mapping as a chunk shape as a whole. `RectilinearChunkGridMetadata`, which is experimental, now reads a `bool` edge length as the `int` it equals and rejects a NumPy integer or a float edge length such as `4.0` with a `TypeError`; it used to keep them as given, so that it could not store a NumPy integer, could not read back a `bool`, and stored a float as a JSON float, which the Zarr specification does not allow. -Stored metadata with a regular chunk size of 0 or JSON `false` (Zarr format 2 `chunks`, Zarr format 3 `regular` `chunk_shape`) is now read as chunk size 1 (the inner chunk size for sharded arrays, which previously failed to open), however long the axis is: that is the chunk size `chunks=-1` gives on a zero-length axis, and it does not depend on how far the axis has grown since. zarr-python wrote such sizes for arrays created with a zero-length axis until 3.4, and, from 2.18.7 to 3.2.1, for an explicit chunk size of 0 or `False` on an axis of any length, as in `chunks=(0,)` (those arrays could store no data). On a zero-length axis this opens silently, as before, and appending to it stores chunks of size 1. On an axis of positive length it opens with a `ZarrUserWarning`, which says that the array holds only its fill value, so that recreating it with the wanted chunk shape loses nothing, and how to keep it by storing valid metadata instead: `array.update_attributes({})`, then `zarr.consolidate_metadata` if the metadata is consolidated. JSON `true`, as zarr-python 3.0 and 3.2 wrote for a chunk size of `True`, is read as 1, silently, in the chunk grid (regular, and the explicit edges and run-length encoded sizes of a rectilinear one) and in the inner chunk shape of every sharding codec, nested or not. A stored rectilinear chunk grid whose edge lengths are integral JSON floats (`[[4.0, 2]]`), as zarr-python 3.2 wrote for float edges, is read with those edges as integers, silently; a float anywhere no release wrote one (a regular chunk shape, the inner chunk shape of a sharding codec, a run-length repeat count, an edge below 1) is rejected. +Stored metadata with a regular chunk size of 0 or JSON `false` (Zarr format 2 `chunks`, Zarr format 3 `regular` `chunk_shape`) is now read as chunk size 1 (the inner chunk size for sharded arrays, which previously failed to open), however long the axis is: that is the chunk size `chunks=-1` gives on a zero-length axis, and it does not depend on how far the axis has grown since. zarr-python wrote such sizes for arrays created with a zero-length axis until 3.4, and, from 2.18.7 to 3.2.1, for an explicit chunk size of 0 or `False` on an axis of any length, as in `chunks=(0,)` (those arrays could store no data). On a zero-length axis this opens silently, as before, and appending to it stores chunks of size 1. On an axis of positive length it opens with a `ZarrUserWarning`, which says that the array holds only its fill value, so that recreating it with the wanted chunk shape loses nothing (`zarr.from_array(array.store, name=array.path, data=array, chunks=..., overwrite=True, write_data=False)` keeps its data type, fill value, attributes and codecs), and how to keep it by storing valid metadata instead: `array.update_attributes({})`, then `zarr.consolidate_metadata` if the metadata is consolidated. JSON `true`, as zarr-python 3.0 and 3.2 wrote for a chunk size of `True`, is read as 1, silently, in the chunk grid (regular, and the explicit edges and run-length encoded sizes of a rectilinear one) and in the inner chunk shape of every sharding codec, nested or not. A stored rectilinear chunk grid whose edge lengths are integral JSON floats (`[[4.0, 2]]`), as zarr-python 3.2 wrote for float edges, is read with those edges as integers, silently; a float anywhere no release wrote one (a regular chunk shape, the inner chunk shape of a sharding codec, a run-length repeat count, an edge below 1) is rejected. Writing data to an array whose stored metadata holds a regular chunk size of 0 or `false` (other than an empty selection) first stores the upgrade of the metadata the store then holds, unless another writer has stored valid metadata since, so that other readers find the chunks written (a `true` chunk size or an integral float edge is read as the value it equals, so a write stores no metadata for it); in a `ZipStore` this adds a second entry for the metadata document, as every metadata update does. If the metadata the store then holds lays out chunks differently (another writer stored a different chunk size since, say), the write raises a `ValueError` asking to reopen the array, and stores nothing. Apart from the array's own metadata writes (`update_attributes`, `resize`), which store the upgrade as before, this is the only operation that stores it: a group's consolidated metadata keeps such an array's metadata as it was stored (`zarr.consolidate_metadata` copies it as the array's own document holds it) until the array stores its upgrade through the group's handle, and changing a group reads and writes no metadata of its members. diff --git a/src/zarr/core/metadata/upgrades.py b/src/zarr/core/metadata/upgrades.py index 7c22f94833..31616cf2b1 100644 --- a/src/zarr/core/metadata/upgrades.py +++ b/src/zarr/core/metadata/upgrades.py @@ -52,9 +52,12 @@ class Reading(NamedTuple): RECREATE_HINT: Final = ( "The array holds only its fill value, so recreate it with the chunk shape you want; " - "nothing is lost. To keep it instead, store valid metadata: open the array writable and " - "call `array.update_attributes({})`; if a group holds consolidated metadata for the " - "array, then also call `zarr.consolidate_metadata` on that group." + "nothing is lost: `zarr.from_array(array.store, name=array.path, data=array, " + "chunks=, overwrite=True, write_data=False)` keeps its data type, fill " + "value, attributes and codecs. To keep the array as it is instead, store valid metadata " + "with `array.update_attributes({})`. Either way, open the array writable first, and if a " + "group holds consolidated metadata for the array, then also call " + "`zarr.consolidate_metadata` on that group." ) """How to act on a document whose array can hold no data: a stored chunk size of 0 is read as the smallest chunk size, which is a poor chunk shape for the data the array grows into, diff --git a/tests/test_metadata/test_upgrades.py b/tests/test_metadata/test_upgrades.py index 3b115fe73d..8e1708c04a 100644 --- a/tests/test_metadata/test_upgrades.py +++ b/tests/test_metadata/test_upgrades.py @@ -742,6 +742,51 @@ def test_stale_handle_write_keeps_valid_document_as_written( np.testing.assert_array_equal(_open_strictly(path)[...], [9, 0, 0]) +@pytest.mark.parametrize("zarr_format", [2, 3]) +def test_recreate_hint_recipe(tmp_path: Path, zarr_format: Literal[2, 3]) -> None: + """The recipe the warning gives for an array read from a stored chunk size of 0 + recreates it with the chunk shape asked for, keeping its data type, fill value, + attributes and codecs, and the array then reopens without a warning.""" + path = tmp_path / "legacy.zarr" + zarr.create_array( + store=path, + shape=(3,), + chunks=(3,), + dtype="int16", + fill_value=7, + attributes={"units": "m"}, + zarr_format=zarr_format, + ) + _rewrite_doc( + path, + zarr_format, + lambda doc: ( + doc.update(chunks=[0]) + if zarr_format == 2 + else doc["chunk_grid"]["configuration"].update(chunk_shape=[0]) + ), + ) + recipe = ( + "`zarr.from_array(array.store, name=array.path, data=array, chunks=, " + "overwrite=True, write_data=False)`" + ) + with pytest.warns(ZarrUserWarning, match=re.escape(recipe)): + array = zarr.open_array(store=path, mode="r+") + + zarr.from_array( + array.store, name=array.path, data=array, chunks=(2,), overwrite=True, write_data=False + ) + + reopened = _open_strictly(path) + assert reopened.chunks == (2,) + assert reopened.dtype == array.dtype + assert reopened.fill_value == 7 + assert reopened.attrs.asdict() == {"units": "m"} + assert reopened.compressors == array.compressors + assert reopened.filters == array.filters + np.testing.assert_array_equal(reopened[...], [7, 7, 7]) + + def _store_zero(doc: dict[str, Any]) -> None: _stored_chunks(doc)[0] = 0 From e88b6f9683fa3cb7b76e48a65adf356f8ea1adb5 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Wed, 30 Sep 2026 14:58:44 +0200 Subject: [PATCH 74/76] test(metadata): widen the abbreviated-message bound for the recipe Assisted-by: ClaudeCode:claude-fable-5-1 Co-Authored-By: Claude Fable 5.1 --- tests/test_metadata/test_upgrades.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/tests/test_metadata/test_upgrades.py b/tests/test_metadata/test_upgrades.py index 2db7c70dee..96005d91a0 100644 --- a/tests/test_metadata/test_upgrades.py +++ b/tests/test_metadata/test_upgrades.py @@ -1397,7 +1397,7 @@ def test_read_edge_lists_in_regular_grid( assert message.endswith(RESAVE_HINT) # The 1000-edge list is abbreviated: the message is two sentences and two hints, not # a dump of the edges. - assert len(message) < 1300 + assert len(message) < 1600 def _rejected_without_warning(doc: dict[str, JSON]) -> pytest.ExceptionInfo[Exception]: From 0992e948bb13e2c7594db894ecc7791d10b7f98b Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Thu, 1 Oct 2026 14:38:52 +0200 Subject: [PATCH 75/76] refactor(metadata): call the lenient readings of stored documents repairs, not upgrades The module is zarr.core.metadata.repair: repair_array_document, mark_repaired, ARRAY_REPAIRS and the Repair type, AsyncArray._store_repaired_document, and the test module test_repair.py. A reading that turns an invalid stored document into a valid one fixes it; it does not move it to a newer format. Assisted-by: ClaudeCode:claude-fable-5-1 Co-Authored-By: Claude Fable 5.1 --- changes/4334.bugfix.md | 2 +- src/zarr/core/array.py | 26 ++-- src/zarr/core/common.py | 2 +- src/zarr/core/group.py | 4 +- src/zarr/core/metadata/io.py | 12 +- .../core/metadata/{upgrades.py => repair.py} | 76 +++++------ src/zarr/core/metadata/v2.py | 18 +-- src/zarr/core/metadata/v3.py | 20 +-- tests/test_array_stateful.py | 4 +- tests/test_metadata/test_io.py | 2 +- .../{test_upgrades.py => test_repair.py} | 118 +++++++++--------- 11 files changed, 142 insertions(+), 142 deletions(-) rename src/zarr/core/metadata/{upgrades.py => repair.py} (84%) rename tests/test_metadata/{test_upgrades.py => test_repair.py} (93%) diff --git a/changes/4334.bugfix.md b/changes/4334.bugfix.md index fa5ca0c5a8..2d2435ddb8 100644 --- a/changes/4334.bugfix.md +++ b/changes/4334.bugfix.md @@ -4,4 +4,4 @@ A chunk edge length is now always at least 1, while an array extent may be 0. Ev Stored metadata with a regular chunk size of 0 or JSON `false` (Zarr format 2 `chunks`, Zarr format 3 `regular` `chunk_shape`) is now read as chunk size 1 (the inner chunk size for sharded arrays, which previously failed to open), however long the axis is: that is the chunk size `chunks=-1` gives on a zero-length axis, and it does not depend on how far the axis has grown since. zarr-python wrote such sizes for arrays created with a zero-length axis until 3.4, and, from 2.18.7 to 3.2.1, for an explicit chunk size of 0 or `False` on an axis of any length, as in `chunks=(0,)` (those arrays could store no data). On a zero-length axis this opens silently, as before, and appending to it stores chunks of size 1. On an axis of positive length it opens with a `ZarrUserWarning`, which says that the array holds only its fill value, so that recreating it with the wanted chunk shape loses nothing (`zarr.from_array(array.store, name=array.path, data=array, chunks=..., overwrite=True, write_data=False)` keeps its data type, fill value, attributes and codecs), and how to keep it by storing valid metadata instead: `array.update_attributes({})`, then `zarr.consolidate_metadata` if the metadata is consolidated. JSON `true`, as zarr-python 3.0 and 3.2 wrote for a chunk size of `True`, is read as 1, silently, in the chunk grid (regular, and the explicit edges and run-length encoded sizes of a rectilinear one) and in the inner chunk shape of every sharding codec, nested or not. A stored rectilinear chunk grid whose edge lengths are integral JSON floats (`[[4.0, 2]]`), as zarr-python 3.2 wrote for float edges, is read with those edges as integers, silently; a float anywhere no release wrote one (a regular chunk shape, the inner chunk shape of a sharding codec, a run-length repeat count, an edge below 1) is rejected. -Writing data to an array whose stored metadata holds a regular chunk size of 0 or `false` (other than an empty selection) first stores the upgrade of the metadata the store then holds, unless another writer has stored valid metadata since, so that other readers find the chunks written (a `true` chunk size or an integral float edge is read as the value it equals, so a write stores no metadata for it); in a `ZipStore` this adds a second entry for the metadata document, as every metadata update does. If the metadata the store then holds lays out chunks differently (another writer stored a different chunk size since, say), the write raises a `ValueError` asking to reopen the array, and stores nothing. Apart from the array's own metadata writes (`update_attributes`, `resize`), which store the upgrade as before, this is the only operation that stores it: a group's consolidated metadata keeps such an array's metadata as it was stored (`zarr.consolidate_metadata` copies it as the array's own document holds it) until the array stores its upgrade through the group's handle, and changing a group reads and writes no metadata of its members. +Writing data to an array whose stored metadata holds a regular chunk size of 0 or `false` (other than an empty selection) first stores the repair of the metadata the store then holds, unless another writer has stored valid metadata since, so that other readers find the chunks written (a `true` chunk size or an integral float edge is read as the value it equals, so a write stores no metadata for it); in a `ZipStore` this adds a second entry for the metadata document, as every metadata update does. If the metadata the store then holds lays out chunks differently (another writer stored a different chunk size since, say), the write raises a `ValueError` asking to reopen the array, and stores nothing. Apart from the array's own metadata writes (`update_attributes`, `resize`), which store the repair as before, this is the only operation that stores it: a group's consolidated metadata keeps such an array's metadata as it was stored (`zarr.consolidate_metadata` copies it as the array's own document holds it) until the array stores its repair through the group's handle, and changing a group reads and writes no metadata of its members. diff --git a/src/zarr/core/array.py b/src/zarr/core/array.py index cd44c2ee1f..9788f60743 100644 --- a/src/zarr/core/array.py +++ b/src/zarr/core/array.py @@ -214,7 +214,7 @@ def parse_array_metadata(data: Any, path: str | None = None) -> ArrayMetadata: `ArrayV2Metadata` accepts a chunk size of 0, as it always has, though only an invalid document holds one: such metadata is read as the documents it would store - are (see `zarr.core.metadata.upgrades`), so an array can be built from it. No data + are (see `zarr.core.metadata.repair`), so an array can be built from it. No data was read or written under that chunk size, so the reading is silent.""" if isinstance(data, ArrayV2Metadata) and 0 in data.chunks: return parse_stored_array(data.to_buffer_dict(default_buffer_prototype()), 2) @@ -1613,28 +1613,28 @@ async def _save_metadata(self, metadata: ArrayMetadata, ensure_parents: bool = F def _stored_document_replaced(self) -> None: """Record that the store no longer holds a document of this array that needs an - upgrade: it holds the upgrade, a valid document, or none. The metadata this handle + repair: it holds the repair, a valid document, or none. The metadata this handle holds, which a consolidated group handle may share, then stops standing for the - document it was read from (see `mark_upgraded`), so no later write through either + document it was read from (see `mark_repaired`), so no later write through either handle stores that document again.""" object.__setattr__(self.metadata, "_stored_document", None) - async def _store_upgraded_document(self) -> None: - """Store the upgrade of this array's current stored document, if it needs one, + async def _store_repaired_document(self) -> None: + """Store the repair of this array's current stored document, if it needs one, before chunks are written under this handle's metadata. - Only for metadata read from a document whose upgrade moves chunks (see - `zarr.core.metadata.upgrades`). The document is read again, because the store may + Only for metadata read from a document whose repair moves chunks (see + `zarr.core.metadata.repair`). The document is read again, because the store may hold a newer one than this handle's metadata. If that one lays out chunks differently (another writer stored a different chunk size since), this handle would write chunks no reader finds, so it raises and stores nothing. If it needs - no upgrade (the array was re-saved since, possibly by another implementation), it - is left as written; if there is none, there is nothing to upgrade. + no repair (the array was re-saved since, possibly by another implementation), it + is left as written; if there is none, there is nothing to repair. - Storing the same upgrade twice is harmless, so handles that write chunks + Storing the same repair twice is harmless, so handles that write chunks concurrently need no coordination. The read and the store are not one atomic step, though: a metadata write by another handle between them (a resize, an - attribute update) is replaced by the upgrade of the document read before it, as + attribute update) is replaced by the repair of the document read before it, as with any two metadata writes that race. """ if self.metadata._stored_document is None: @@ -1664,9 +1664,9 @@ async def _set_selection( fields: Fields | None = None, ) -> None: if product(indexer.shape) > 0: - # Chunks are about to be stored under the upgraded metadata, so store it + # Chunks are about to be stored under the repaired metadata, so store it # first: every reader of the store then agrees with them. - await self._store_upgraded_document() + await self._store_repaired_document() return await _set_selection( self.store_path, self.metadata, diff --git a/src/zarr/core/common.py b/src/zarr/core/common.py index fad94f9703..de009d0b50 100644 --- a/src/zarr/core/common.py +++ b/src/zarr/core/common.py @@ -290,7 +290,7 @@ def _subject(name: str, axis: int | None) -> str: def _parse_positive_int(value: object, name: str, axis: int | None) -> int: """`value` as an `int` of at least 1. A `bool` is read as the `int` it equals; any other type, a NumPy integer or a float (even an integral one: stored documents with - integral floats are read by `zarr.core.metadata.upgrades`), is rejected.""" + integral floats are read by `zarr.core.metadata.repair`), is rejected.""" subject = _subject(name, axis) if not isinstance(value, int): raise TypeError(f"{subject} must be an int, got {value!r}") diff --git a/src/zarr/core/group.py b/src/zarr/core/group.py index f94a05fd27..7d4e26a1dd 100644 --- a/src/zarr/core/group.py +++ b/src/zarr/core/group.py @@ -150,8 +150,8 @@ class ConsolidatedMetadata: def to_dict(self) -> dict[str, JSON]: """The consolidated metadata document. An array read from a stored document that - had to be upgraded is written as that document was stored, so every reader of - the consolidated metadata reads it as upgraded again (see `mark_upgraded`).""" + had to be repaired is written as that document was stored, so every reader of + the consolidated metadata reads it as repaired again (see `mark_repaired`).""" return { "kind": self.kind, "must_understand": self.must_understand, diff --git a/src/zarr/core/metadata/io.py b/src/zarr/core/metadata/io.py index 6094a162c8..5a9237327d 100644 --- a/src/zarr/core/metadata/io.py +++ b/src/zarr/core/metadata/io.py @@ -10,7 +10,7 @@ from zarr.core.buffer.core import default_buffer_prototype from zarr.core.buffer.cpu import buffer_prototype as cpu_buffer_prototype from zarr.core.common import ZARR_JSON, ZARRAY_JSON, ZATTRS_JSON -from zarr.core.metadata.upgrades import mark_upgraded, upgrade_array_document +from zarr.core.metadata.repair import mark_repaired, repair_array_document from zarr.errors import ArrayNotFoundError, ContainsArrayError from zarr.storage._common import StorePath, ensure_no_existing_node @@ -116,9 +116,9 @@ async def upsert_metadata( def parse_stored_array(documents: Mapping[str, Buffer], zarr_format: ZarrFormat) -> ArrayMetadata: """The metadata of an array from its documents (by store key, see `ARRAY_DOCUMENTS`), - read with the upgrades but without their warnings (whoever asks has warned, or reads - metadata built in code), and marked (see `mark_upgraded`) if they had to be - upgraded. Raises `ArrayNotFoundError` if there is no array document among them.""" + read with the repairs but without their warnings (whoever asks has warned, or reads + metadata built in code), and marked (see `mark_repaired`) if they had to be + repaired. Raises `ArrayNotFoundError` if there is no array document among them.""" from zarr.core.array import ( _array_metadata_dict_v2, _array_metadata_dict_v3, @@ -131,8 +131,8 @@ def parse_stored_array(documents: Mapping[str, Buffer], zarr_format: ZarrFormat) stored = _array_metadata_dict_v3(documents[ZARR_JSON]) else: raise ArrayNotFoundError(f"No Zarr format {zarr_format} array metadata document.") - upgraded, readings = upgrade_array_document(stored, zarr_format) - return mark_upgraded(parse_array_metadata(dict(upgraded)), stored, readings, None, warn=False) + repaired, readings = repair_array_document(stored, zarr_format) + return mark_repaired(parse_array_metadata(dict(repaired)), stored, readings, None, warn=False) def _build_parents(store_path: StorePath, zarr_format: ZarrFormat) -> dict[str, GroupMetadata]: diff --git a/src/zarr/core/metadata/upgrades.py b/src/zarr/core/metadata/repair.py similarity index 84% rename from src/zarr/core/metadata/upgrades.py rename to src/zarr/core/metadata/repair.py index 31616cf2b1..06ce97dce7 100644 --- a/src/zarr/core/metadata/upgrades.py +++ b/src/zarr/core/metadata/repair.py @@ -1,21 +1,21 @@ -"""Upgrades that read invalid stored array metadata documents written by older software. +"""Repairs that read invalid stored array metadata documents written by older software. -This is the only place invalid metadata is read leniently. An upgrade maps a stored +This is the only place invalid metadata is read leniently. A repair maps a stored array metadata document (parsed JSON) to a valid one. `ArrayV2Metadata.from_dict` and -`ArrayV3Metadata.from_dict` apply the upgrades for their Zarr format, so every path +`ArrayV3Metadata.from_dict` apply the repairs for their Zarr format, so every path that parses a stored document, including consolidated metadata, goes through them. A reading warns only where the user must act on it; a reading that gives what zarr -read from the same document before these upgrades existed is silent, so a document +read from the same document before these repairs existed is silent, so a document that opened without a warning still does. The warnings are given once per document, -after the upgraded document has passed the metadata constructor, so an invalid +after the repaired document has passed the metadata constructor, so an invalid document raises its own error, not a warning about how it was read. Metadata read -from a document whose upgrade moves chunks (a chunk size read as another size) is -marked (see `mark_upgraded`), silent or not, so the array stores the upgrade before it -writes chunks under it. An upgrade that only respells a value zarr already read the +from a document whose repair moves chunks (a chunk size read as another size) is +marked (see `mark_repaired`), silent or not, so the array stores the repair before it +writes chunks under it. A repair that only respells a value zarr already read the same way (`true` as 1, `4.0` as 4) moves no chunks, so it is not marked. -To read another kind of invalid document, add an upgrade to `ARRAY_UPGRADES`. +To read another kind of invalid document, add a repair to `ARRAY_REPAIRS`. """ from __future__ import annotations @@ -36,18 +36,18 @@ class Reading(NamedTuple): - """How an upgrade read a stored document.""" + """How a repair read a stored document.""" moves_chunks: bool - """Whether chunks written under the upgraded document are not where a reader of the - stored document looks for them, so the array must store the upgrade before it + """Whether chunks written under the repaired document are not where a reader of the + stored document looks for them, so the array must store the repair before it writes chunks.""" warning: str | None """What the user must act on, if anything.""" -type Upgrade = Callable[[ArrayDocument], tuple[ArrayDocument, Reading] | None] -"""Returns `None` if the document needs no upgrade, else the upgraded document and how +type Repair = Callable[[ArrayDocument], tuple[ArrayDocument, Reading] | None] +"""Returns `None` if the document needs no repair, else the repaired document and how it was read.""" RECREATE_HINT: Final = ( @@ -64,7 +64,7 @@ class Reading(NamedTuple): and a re-save would keep it.""" -def mark_upgraded[M]( +def mark_repaired[M]( metadata: M, stored: ArrayDocument, readings: Sequence[Reading], @@ -73,9 +73,9 @@ def mark_upgraded[M]( warn: bool = True, ) -> M: """Record that `metadata` was read from the document `stored`, which needed the - upgrades whose `readings` `upgrade_array_document` returned, if any. If a reading + repairs whose `readings` `repair_array_document` returned, if any. If a reading moves chunks, keep a copy of `stored` as `_stored_document` on `metadata`, so the - array stores the upgrade before it writes chunks under it and consolidated metadata + array stores the repair before it writes chunks under it and consolidated metadata stores it as it was stored. If `warn`, warn once with the readings' warnings (each says what the user must act on and how), naming the array at `path` when the caller knows it.""" @@ -142,7 +142,7 @@ def _read_chunk_shape( Returns the chunk shape and how it was read, if any entry was read as another value (the warning says how the chunk shape was read and, as the array then holds only its fill value, recommends recreating it, see `RECREATE_HINT`); `None` if it cannot be - read or needs no upgrade. + read or needs no repair. """ if not (isinstance(stored, list) and len(stored) == len(spans)): return None @@ -190,22 +190,22 @@ def _read_codec(codec: JSON) -> JSON | None: no chunks.""" match codec: case {"name": "sharding_indexed", "configuration": Mapping() as configuration}: - upgraded: dict[str, JSON] = {} + repaired: dict[str, JSON] = {} stored = configuration.get("chunk_shape") if isinstance(stored, list) and ( read := _read_chunk_shape(stored, [None] * len(stored)) ): - upgraded["chunk_shape"] = read[0] + repaired["chunk_shape"] = read[0] if isinstance(codecs := configuration.get("codecs"), list) and ( inner := _read_codecs(codecs) ): - upgraded["codecs"] = inner - if not upgraded: + repaired["codecs"] = inner + if not repaired: return None # The mapping pattern does not narrow `codec` for mypy. return { **cast("Mapping[str, JSON]", codec), - "configuration": {**configuration, **upgraded}, + "configuration": {**configuration, **repaired}, } return None @@ -251,8 +251,8 @@ def _invalid_chunk_sizes_v3(doc: ArrayDocument) -> tuple[ArrayDocument, Reading] return None match _read_chunk_shape(configuration.get("chunk_shape"), shape, units): case chunk_shape, reading: - upgraded = {**configuration, "chunk_shape": chunk_shape} - return {**doc, "chunk_grid": {**grid, "configuration": upgraded}}, reading + repaired = {**configuration, "chunk_shape": chunk_shape} + return {**doc, "chunk_grid": {**grid, "configuration": repaired}}, reading return None @@ -308,33 +308,33 @@ def _invalid_edge_lengths_v3(doc: ArrayDocument) -> tuple[ArrayDocument, Reading read = [_read_rectilinear_axis(axis) for axis in stored] if not _respelled(read, stored): return None - upgraded = {**configuration, "chunk_shapes": read} + repaired = {**configuration, "chunk_shapes": read} # A stored edge is read as the value it equals, so the reading moves no chunks. - return {**doc, "chunk_grid": {**grid, "configuration": upgraded}}, Reading(False, None) + return {**doc, "chunk_grid": {**grid, "configuration": repaired}}, Reading(False, None) -ARRAY_UPGRADES: Final[Mapping[ZarrFormat, tuple[Upgrade, ...]]] = { +ARRAY_REPAIRS: Final[Mapping[ZarrFormat, tuple[Repair, ...]]] = { 2: (_invalid_chunk_sizes_v2,), # The inner chunk shape is read first: it gives the unit of the outer chunk shape. - # The rectilinear edge lengths are read last, after any upgrade that yields a + # The rectilinear edge lengths are read last, after any repair that yields a # rectilinear chunk grid. 3: (_invalid_inner_chunk_sizes_v3, _invalid_chunk_sizes_v3, _invalid_edge_lengths_v3), } -"""The upgrades of an array document of each Zarr format, applied in order.""" +"""The repairs of an array document of each Zarr format, applied in order.""" -def upgrade_array_document( +def repair_array_document( doc: ArrayDocument, zarr_format: ZarrFormat ) -> tuple[ArrayDocument, list[Reading]]: - """Apply the upgrades for `zarr_format` to a stored array metadata document. + """Apply the repairs for `zarr_format` to a stored array metadata document. - Returns the upgraded document and the reading of each upgrade that changed it, for - `mark_upgraded` once the document has been validated. + Returns the repaired document and the reading of each repair that changed it, for + `mark_repaired` once the document has been validated. """ readings: list[Reading] = [] - for upgrade in ARRAY_UPGRADES[zarr_format]: - upgraded = upgrade(doc) - if upgraded is not None: - doc, reading = upgraded + for repair in ARRAY_REPAIRS[zarr_format]: + repaired = repair(doc) + if repaired is not None: + doc, reading = repaired readings.append(reading) return doc, readings diff --git a/src/zarr/core/metadata/v2.py b/src/zarr/core/metadata/v2.py index 7f6a33e93b..136267fc58 100644 --- a/src/zarr/core/metadata/v2.py +++ b/src/zarr/core/metadata/v2.py @@ -25,7 +25,7 @@ TBaseScalar, ZDType, ) - from zarr.core.metadata.upgrades import ArrayDocument + from zarr.core.metadata.repair import ArrayDocument from dataclasses import dataclass, field, fields, replace @@ -45,7 +45,7 @@ from zarr.core.config import config, parse_indexing_order from zarr.core.json_parse import parse_field from zarr.core.metadata.common import parse_attributes -from zarr.core.metadata.upgrades import mark_upgraded, upgrade_array_document +from zarr.core.metadata.repair import mark_repaired, repair_array_document class ArrayV2MetadataDict(TypedDict): @@ -74,8 +74,8 @@ class ArrayV2Metadata(Metadata): attributes: dict[str, JSON] = field(default_factory=dict) zarr_format: Literal[2] = field(init=False, default=2) _stored_document: ClassVar[ArrayDocument | None] = None - """The stored document `from_dict` read this metadata from, if it had to upgrade it - (set on the instance by `mark_upgraded`): the store may still hold it.""" + """The stored document `from_dict` read this metadata from, if it had to repair it + (set on the instance by `mark_repaired`): the store may still hold it.""" def __init__( self, @@ -154,11 +154,11 @@ def to_buffer_dict(self, prototype: BufferPrototype) -> dict[str, Buffer]: @classmethod def from_dict(cls, data: dict[str, Any], *, path: str | None = None) -> ArrayV2Metadata: """Read a stored `.zarray` document (with its attributes). An invalid document - that `zarr.core.metadata.upgrades` can read is read as upgraded; a reading the user + that `zarr.core.metadata.repair` can read is read as repaired; a reading the user must act on warns, naming the array at `path`.""" - upgraded, readings = upgrade_array_document(data, 2) + repaired, readings = repair_array_document(data, 2) # a new dict, because we are modifying it - _data: dict[str, Any] = dict(upgraded) + _data: dict[str, Any] = dict(repaired) # Check that the zarr_format attribute is correct. _ = parse_zarr_format(_data.pop("zarr_format")) @@ -209,7 +209,7 @@ def from_dict(cls, data: dict[str, Any], *, path: str | None = None) -> ArrayV2M _data = {k: v for k, v in _data.items() if k in expected} - return mark_upgraded(cls(**_data), data, readings, path) + return mark_repaired(cls(**_data), data, readings, path) def to_dict(self) -> dict[str, JSON]: zarray_dict = super().to_dict() @@ -332,7 +332,7 @@ def parse_compressor(data: object) -> Numcodec | None: def parse_chunks(chunks: ShapeLike, shape: tuple[int, ...]) -> tuple[int, ...]: """Check a chunk shape: one non-negative integer per array axis (see - `parse_shapelike`). Stored chunk sizes of 0 are read by `zarr.core.metadata.upgrades`.""" + `parse_shapelike`). Stored chunk sizes of 0 are read by `zarr.core.metadata.repair`.""" chunks_parsed = parse_shapelike(chunks) if len(chunks_parsed) != len(shape): raise ValueError( diff --git a/src/zarr/core/metadata/v3.py b/src/zarr/core/metadata/v3.py index 8df03217ac..e58082aeb9 100644 --- a/src/zarr/core/metadata/v3.py +++ b/src/zarr/core/metadata/v3.py @@ -40,9 +40,9 @@ from zarr.core.dtype.common import check_dtype_spec_v3 from zarr.core.json_parse import parse_field from zarr.core.metadata.common import parse_attributes -from zarr.core.metadata.upgrades import ( - mark_upgraded, - upgrade_array_document, +from zarr.core.metadata.repair import ( + mark_repaired, + repair_array_document, ) from zarr.errors import MetadataValidationError, NodeTypeValidationError from zarr.registry import get_codec_class @@ -54,7 +54,7 @@ from zarr.core.buffer import Buffer, BufferPrototype from zarr.core.chunk_grids import ChunkGrid from zarr.core.dtype.wrapper import TBaseDType, TBaseScalar - from zarr.core.metadata.upgrades import ArrayDocument + from zarr.core.metadata.repair import ArrayDocument def parse_zarr_format(data: object) -> Literal[3]: @@ -499,8 +499,8 @@ class ArrayV3Metadata(Metadata): storage_transformers: tuple[dict[str, JSON], ...] extra_fields: dict[str, AllowedExtraField] _stored_document: ClassVar[ArrayDocument | None] = None - """The stored document `from_dict` read this metadata from, if it had to upgrade it - (set on the instance by `mark_upgraded`): the store may still hold it.""" + """The stored document `from_dict` read this metadata from, if it had to repair it + (set on the instance by `mark_repaired`): the store may still hold it.""" def __init__( self, @@ -645,11 +645,11 @@ def to_buffer_dict(self, prototype: BufferPrototype) -> dict[str, Buffer]: @classmethod def from_dict(cls, data: dict[str, JSON], *, path: str | None = None) -> Self: """Read a stored `zarr.json` array document. An invalid document that - `zarr.core.metadata.upgrades` can read is read as upgraded; a reading the user + `zarr.core.metadata.repair` can read is read as repaired; a reading the user must act on warns, naming the array at `path`.""" - upgraded, readings = upgrade_array_document(data, 3) + repaired, readings = repair_array_document(data, 3) # a new dict, because we are modifying it - _data = dict(upgraded) + _data = dict(repaired) # check that the zarr_format attribute is correct _ = parse_zarr_format(_data.pop("zarr_format")) @@ -691,7 +691,7 @@ def from_dict(cls, data: dict[str, JSON], *, path: str | None = None) -> Self: extra_fields=allowed_extra_fields, storage_transformers=_data_typed.get("storage_transformers", ()), # type: ignore[arg-type] ) - return mark_upgraded(metadata, data, readings, path) + return mark_repaired(metadata, data, readings, path) def to_dict(self) -> dict[str, JSON]: out_dict = super().to_dict() diff --git a/tests/test_array_stateful.py b/tests/test_array_stateful.py index 5319dd6f89..a25b056c7d 100644 --- a/tests/test_array_stateful.py +++ b/tests/test_array_stateful.py @@ -168,7 +168,7 @@ def _open(self) -> zarr.Array[Any]: warnings.simplefilter("always", ZarrUserWarning) arr = zarr.open_array(self.store, path=self.path, mode="r+") # A stored chunk size of 0 is read silently on an empty axis; on a non-empty one - # it warns that the axis holds only the fill value. Either way it is upgraded. + # it warns that the axis holds only the fill value. Either way it is repaired. warned = any(issubclass(w.category, ZarrUserWarning) for w in record) must_warn = any(self.shape[axis] > 0 for axis in self.legacy_axes) assert warned is must_warn, [str(w.message) for w in record] @@ -296,7 +296,7 @@ def teardown(self) -> None: # ------------------------------------------------------------ invariants @invariant() def no_chunk_under_an_invalid_document(self) -> None: - """While the stored document is still one that is upgraded on read, which other + """While the stored document is still one that is repaired on read, which other readers may reject or read differently, no chunk is stored under it.""" if self.legacy_axes: keys = sync(_list(self.store, f"{self.path}/")) diff --git a/tests/test_metadata/test_io.py b/tests/test_metadata/test_io.py index 7fb4a06914..999e5a15a6 100644 --- a/tests/test_metadata/test_io.py +++ b/tests/test_metadata/test_io.py @@ -135,7 +135,7 @@ def _documents(store: Store) -> dict[str, Any]: def _legacy(zarr_format: Literal[2, 3]) -> tuple[StorePath, ArrayV2Metadata | ArrayV3Metadata]: - """An array stored with chunk shape `[0]`, and the metadata its upgrade reads.""" + """An array stored with chunk shape `[0]`, and the metadata its repair reads.""" store = _CountingStore() array = zarr.create_array( store, shape=(3,), chunks=(3,), dtype="int16", zarr_format=zarr_format diff --git a/tests/test_metadata/test_upgrades.py b/tests/test_metadata/test_repair.py similarity index 93% rename from tests/test_metadata/test_upgrades.py rename to tests/test_metadata/test_repair.py index 8e1708c04a..da35922da7 100644 --- a/tests/test_metadata/test_upgrades.py +++ b/tests/test_metadata/test_repair.py @@ -1,4 +1,4 @@ -"""Tests for the upgrades that read invalid stored array metadata documents.""" +"""Tests for the repairs that read invalid stored array metadata documents.""" from __future__ import annotations @@ -19,9 +19,9 @@ from zarr.core.array import AsyncArray from zarr.core.group import ConsolidatedMetadata from zarr.core.metadata import ArrayV2Metadata, ArrayV3Metadata -from zarr.core.metadata.upgrades import ( +from zarr.core.metadata.repair import ( RECREATE_HINT, - upgrade_array_document, + repair_array_document, ) from zarr.core.metadata.v3 import RectilinearChunkGridMetadata, RegularChunkGridMetadata from zarr.core.sync import sync @@ -113,7 +113,7 @@ def _nested_sharded_doc(inner: list[Any], nested: list[Any]) -> dict[str, JSON]: @pytest.mark.parametrize( - ("doc", "expected", "upgraded", "warning"), + ("doc", "expected", "repaired", "warning"), [ (_v2_doc([10, 10], [4, 5]), ((4, 5),), "no", None), (_v3_doc([0, 0], [1, 1]), ((1, 1),), "no", None), @@ -174,35 +174,35 @@ def _nested_sharded_doc(inner: list[Any], nested: list[Any]) -> dict[str, JSON]: "v3-nested-sharded-true", ], ) -def test_upgrade_array_document( +def test_repair_array_document( doc: dict[str, JSON], expected: tuple[Any, ...], - upgraded: Literal["no", "respelled", "moved"], + repaired: Literal["no", "respelled", "moved"], warning: str | None, ) -> None: """Valid documents pass unchanged. A stored chunk size of 0 or `false` is read as 1 (the inner chunk size when sharded), however long the axis, and `true` as 1, in the chunk shape and in the inner chunk shape of every sharding codec, nested or - not. `from_dict` marks the metadata of a document whose upgrade moves chunks (a - chunk size read as another size), so the array stores the upgrade before it writes + not. `from_dict` marks the metadata of a document whose repair moves chunks (a + chunk size read as another size), so the array stores the repair before it writes chunks, but not one that only respells a value (`true` as 1). It warns once, naming the array, only where a chunk size of 0 was stored for a non-empty axis (which then holds only its fill value), saying how that part was read and that recreating the array loses nothing (see `RECREATE_HINT`). The other readings give what zarr read before, so they are silent.""" - upgraded_doc, readings = upgrade_array_document(doc, cast("ZarrFormat", doc["zarr_format"])) + repaired_doc, readings = repair_array_document(doc, cast("ZarrFormat", doc["zarr_format"])) assert { - k: v for k, v in upgraded_doc.items() if k not in ("chunks", "chunk_grid", "codecs") + k: v for k, v in repaired_doc.items() if k not in ("chunks", "chunk_grid", "codecs") } == {k: v for k, v in doc.items() if k not in ("chunks", "chunk_grid", "codecs")} - assert bool(readings) is (upgraded != "no") - if upgraded == "no": - assert upgraded_doc is doc + assert bool(readings) is (repaired != "no") + if repaired == "no": + assert repaired_doc is doc metadata_cls = ArrayV2Metadata if doc["zarr_format"] == 2 else ArrayV3Metadata with warnings.catch_warnings(record=True) as record: warnings.simplefilter("always") metadata = metadata_cls.from_dict(dict(doc), path="group/array") assert _chunk_shapes(metadata) == expected - assert metadata._stored_document == (doc if upgraded == "moved" else None) + assert metadata._stored_document == (doc if repaired == "moved" else None) messages = [str(w.message) for w in record] if warning is None: assert messages == [] @@ -223,11 +223,11 @@ def test_upgrade_array_document( ], ids=["v2", "v3-sharded"], ) -def test_invalid_upgraded_document_raises_without_warning(doc: dict[str, JSON], error: str) -> None: - """A document the upgrades read that the metadata constructor then rejects raises +def test_invalid_repaired_document_raises_without_warning(doc: dict[str, JSON], error: str) -> None: + """A document the repairs read that the metadata constructor then rejects raises that error, without first warning how it was read.""" metadata_cls = ArrayV2Metadata if doc["zarr_format"] == 2 else ArrayV3Metadata - assert upgrade_array_document(doc, cast("ZarrFormat", doc["zarr_format"]))[1] + assert repair_array_document(doc, cast("ZarrFormat", doc["zarr_format"]))[1] with warnings.catch_warnings(): warnings.simplefilter("error", ZarrUserWarning) with pytest.raises(ValueError, match=error): @@ -235,7 +235,7 @@ def test_invalid_upgraded_document_raises_without_warning(doc: dict[str, JSON], def _read_strictly(doc: dict[str, JSON]) -> ArrayV2Metadata | ArrayV3Metadata: - """Read `doc`, failing on any warning that it was upgraded.""" + """Read `doc`, failing on any warning that it was repaired.""" metadata_cls = ArrayV2Metadata if doc["zarr_format"] == 2 else ArrayV3Metadata with warnings.catch_warnings(): warnings.simplefilter("error", ZarrUserWarning) @@ -243,7 +243,7 @@ def _read_strictly(doc: dict[str, JSON]) -> ArrayV2Metadata | ArrayV3Metadata: def _open_strictly(path: Path, mode: Literal["r", "a", "r+"] = "r") -> AnyArray: - """Open the array at `path`, failing on any warning that its document was upgraded.""" + """Open the array at `path`, failing on any warning that its document was repaired.""" with warnings.catch_warnings(): warnings.simplefilter("error", ZarrUserWarning) array = zarr.open_array(store=path, mode=mode) @@ -252,13 +252,13 @@ def _open_strictly(path: Path, mode: Literal["r", "a", "r+"] = "r") -> AnyArray: def test_stored_negative_chunk_size_rejected() -> None: - """No known writer stored a negative chunk size: it is rejected, not upgraded.""" + """No known writer stored a negative chunk size: it is rejected, not repaired.""" with pytest.raises(ValueError, match="^Expected all values to be non-negative"): _read_strictly(_v2_doc([4], [-1])) def test_stored_chunk_shape_ndim_mismatch_rejected() -> None: - """A chunk shape with the wrong number of dimensions is not upgraded, so its 0 is + """A chunk shape with the wrong number of dimensions is not repaired, so its 0 is rejected.""" with pytest.raises(ValueError, match="^Dimension 0: chunk edge length must be >= 1, got 0$"): _read_strictly(_v3_doc([4, 4], [0])) @@ -269,7 +269,7 @@ def test_stored_zero_chunk_size_of_shard_with_invalid_inner_chunk_shape_rejected inner: list[Any], ) -> None: """A stored chunk size of 0 of a sharded array is read in multiples of the inner - chunk size; if that is not an integer of at least 1, the 0 is not upgraded, so it + chunk size; if that is not an integer of at least 1, the 0 is not repaired, so it is rejected.""" with pytest.raises(ValueError, match="^Dimension 0: chunk edge length must be >= 1, got 0$"): _read_strictly(_v3_doc([4], [0], inner=inner)) @@ -459,7 +459,7 @@ def test_metadata_reads_integer_chunk_edge(site: str, size: object, expected: in def test_metadata_rejects_non_integer_chunk_edge(site: str, size: object) -> None: """A chunk edge length in metadata built in code is an `int`: a float is rejected, even an integral one (stored documents with integral floats are read by the - upgrades), and so is a NumPy integer, as zarr 3.4.0 rejected one.""" + repairs), and so is a NumPy integer, as zarr 3.4.0 rejected one.""" with pytest.raises( TypeError, match=re.escape(f"Dimension 0: chunk edge length must be an int, got {size!r}"), @@ -529,7 +529,7 @@ def test_chunk_shape_read_as_array_shape( """`ArrayV2Metadata` and `ShardingCodec` read their chunk shape as `parse_shapelike` reads an array shape: an integer or an iterable of non-negative integers, including NumPy integers and bools. A chunk size of 0 is written back as given; reading a - stored 0 is `zarr.core.metadata.upgrades`' business.""" + stored 0 is `zarr.core.metadata.repair`' business.""" parsed, written = CHUNK_SHAPE_SITES[site](chunks) assert parsed == expected assert all(type(size) is int for size in parsed) @@ -576,7 +576,7 @@ def test_legacy_chunk_size_round_trip( ) -> None: """A store whose metadata holds a chunk size written by older software opens (with a warning where a non-empty axis was stored with chunk size 0), reads and appends under - the upgraded grid, and re-saves valid metadata.""" + the repaired grid, and re-saves valid metadata.""" path = tmp_path / "legacy.zarr" arr = zarr.create_array( store=path, @@ -623,7 +623,7 @@ def store_legacy(doc: dict[str, Any]) -> None: ids=["v2", "v3", "v3-sharded"], ) @pytest.mark.parametrize("api", ["sync", "async", "async-concurrent"]) -def test_write_stores_upgraded_metadata_first( +def test_write_stores_repaired_metadata_first( tmp_path: Path, zarr_format: Literal[2, 3], shape: tuple[int, ...], @@ -631,8 +631,8 @@ def test_write_stores_upgraded_metadata_first( expected: tuple[int, ...], api: str, ) -> None: - """Writing chunks to an array read from an upgraded document first stores the - upgraded metadata, so readers that do not upgrade (or read it differently) see + """Writing chunks to an array read from a repaired document first stores the + repaired metadata, so readers that do not repair (or read it differently) see the chunks the write stored.""" path = tmp_path / "legacy.zarr" zarr.create_array( @@ -694,7 +694,7 @@ def _legacy_array(path: Path, zarr_format: Literal[2, 3]) -> None: def test_stale_handle_write_keeps_newer_metadata( tmp_path: Path, zarr_format: Literal[2, 3] ) -> None: - """A handle read from an upgraded document stores the upgrade of what the store + """A handle read from a repaired document stores the repair of what the store holds when it first writes chunks; if another handle stored valid metadata since, it stores no metadata and writes only its chunks.""" path = tmp_path / "legacy.zarr" @@ -719,8 +719,8 @@ def test_stale_handle_write_keeps_newer_metadata( def test_stale_handle_write_keeps_valid_document_as_written( tmp_path: Path, zarr_format: Literal[2, 3] ) -> None: - """If the document the store holds when a handle read from an upgraded document - first writes chunks needs no upgrade, it is left as written, even where zarr would + """If the document the store holds when a handle read from a repaired document + first writes chunks needs no repair, it is left as written, even where zarr would encode the same metadata differently (as another implementation may have written it).""" path = tmp_path / "legacy.zarr" _legacy_array(path, zarr_format) @@ -818,7 +818,7 @@ def test_stale_handle_write_after_chunk_grid_change_raises( sharded: bool, change: Callable[[dict[str, Any]], None], ) -> None: - """If the document the store holds when a handle read from an upgraded document + """If the document the store holds when a handle read from a repaired document first writes chunks lays out chunks differently from the handle's metadata (another writer stored a different chunk size, or changed the inner chunk shape), the handle's chunks would not be found under it: the write raises and stores nothing.""" @@ -846,7 +846,7 @@ def test_stale_handle_write_after_resize_keeping_zero( ) -> None: """A stored chunk size of 0 is read as 1 however long the axis is, so a resize by software that kept it lays out chunks as the handle does: the handle's first write - stores the upgrade of the resized document, then its chunks.""" + stores the repair of the resized document, then its chunks.""" path = tmp_path / "legacy.zarr" _legacy_array(path, zarr_format) with pytest.warns(ZarrUserWarning, match="is read as"): @@ -877,7 +877,7 @@ def test_failed_metadata_save_keeps_stored_document( operation: Callable[[AnyArray], None], ) -> None: """If storing an array's metadata fails, the array still stands for the document - the store holds, which still needs its upgrade.""" + the store holds, which still needs its repair.""" path = tmp_path / "legacy.zarr" _legacy_array(path, zarr_format) doc_name = ".zarray" if zarr_format == 2 else "zarr.json" @@ -902,18 +902,18 @@ async def failing_set(self: LocalStore, key: str, *args: Any, **kwargs: Any) -> @pytest.mark.parametrize("zarr_format", [2, 3]) def test_write_without_stored_document(zarr_format: Literal[2, 3]) -> None: - """An array read from an upgraded document that no store holds (as + """An array read from a repaired document that no store holds (as `AsyncArray.from_dict` builds one) writes its chunks as any array does: there is no - stored document to upgrade.""" + stored document to repair.""" store = MemoryStore() doc = _v2_doc([3], [0]) if zarr_format == 2 else _v3_doc([3], [0]) with pytest.warns(ZarrUserWarning, match="is read as"): array = zarr.Array(AsyncArray.from_dict(StorePath(store), doc)) - upgraded = array.metadata._stored_document is not None + repaired = array.metadata._stored_document is not None array[:] = [1, 2, 3] - assert (upgraded, array.metadata._stored_document) == (True, None) + assert (repaired, array.metadata._stored_document) == (True, None) np.testing.assert_array_equal(array[:], [1, 2, 3]) assert not [key for key in store._store_dict if key.endswith((".zarray", "zarr.json"))] @@ -922,7 +922,7 @@ def test_write_without_stored_document(zarr_format: Literal[2, 3]) -> None: def test_respelled_document_write_stores_only_chunks( tmp_path: Path, zarr_format: Literal[2, 3] ) -> None: - """A document whose upgrade only respells a value (a stored chunk size `true`, read + """A document whose repair only respells a value (a stored chunk size `true`, read as 1 as zarr read it before) moves no chunks, so a write stores its chunks and no metadata: in a `ZipStore`, which cannot replace an entry, the write adds no second metadata entry and gives no warning.""" @@ -967,7 +967,7 @@ def test_respelled_document_write_stores_only_chunks( ) def test_from_dict_keeps_values_that_are_not_json(codec: Any) -> None: """`from_dict` takes metadata built in code as well as stored documents, so the - upgrades read values that are not JSON (codec instances, NumPy integers) as they are, + repairs read values that are not JSON (codec instances, NumPy integers) as they are, without encoding them, and the constructors read them as before.""" doc: dict[str, Any] = _v3_doc([4], [4]) doc["codecs"] = [codec] @@ -979,9 +979,9 @@ def test_from_dict_keeps_values_that_are_not_json(codec: Any) -> None: @pytest.mark.parametrize(("shape", "expected"), [((0,), (1,)), ((3,), (1,))]) def test_array_from_metadata_with_chunk_size_zero(shape: tuple[int], expected: tuple[int]) -> None: """`ArrayV2Metadata` accepts a chunk size of 0, as a stored document may hold it. An - array built from such metadata reads it as the upgrades read that document, silently + array built from such metadata reads it as the repairs read that document, silently (no data was read or written under it): `create_hierarchy` stores the metadata as - given and yields such an array, which stores the upgrade before its first write.""" + given and yields such an array, which stores the repair before its first write.""" metadata = ArrayV2Metadata(shape=shape, chunks=(0,), dtype=Int16(), fill_value=0, order="C") store = MemoryStore() with warnings.catch_warnings(): @@ -1071,7 +1071,7 @@ async def recording( @pytest.mark.parametrize("zarr_format", [2, 3]) @pytest.mark.parametrize("member", ["a", "g/a"]) @pytest.mark.parametrize("operation", ["attrs", "update_attributes_async", "delete-member"]) -def test_group_write_stores_upgraded_consolidated_member_as_stored( +def test_group_write_stores_repaired_consolidated_member_as_stored( tmp_path: Path, monkeypatch: pytest.MonkeyPatch, zarr_format: Literal[2, 3], @@ -1080,8 +1080,8 @@ def test_group_write_stores_upgraded_consolidated_member_as_stored( ) -> None: """A group write stores only the group's own documents, reading none: it reads and writes no member document, and stores the consolidated copy of a member read from a - document that had to be upgraded exactly as it was stored, so every reader of the - consolidated metadata reads that member as upgraded again.""" + document that had to be repaired exactly as it was stored, so every reader of the + consolidated metadata reads that member as repaired again.""" path = tmp_path / "group.zarr" _flagged_consolidated_group(path, zarr_format, member) stored = json.dumps(_consolidated_member(path, zarr_format, member)) @@ -1106,11 +1106,11 @@ def test_group_write_stores_upgraded_consolidated_member_as_stored( @pytest.mark.filterwarnings("ignore:Consolidated metadata is currently not part:UserWarning") @pytest.mark.parametrize("zarr_format", [2, 3]) -def test_concurrent_deletions_leave_no_upgraded_member( +def test_concurrent_deletions_leave_no_repaired_member( tmp_path: Path, zarr_format: Literal[2, 3] ) -> None: """Members deleted concurrently through one consolidated group handle stay deleted, - also those read from documents that had to be upgraded: no group write stores a + also those read from documents that had to be repaired: no group write stores a member document.""" path = tmp_path / "group.zarr" zarr.open_group(path, mode="w", zarr_format=zarr_format) @@ -1142,13 +1142,13 @@ def _consolidated_legacy_member(path: Path, zarr_format: Literal[2, 3]) -> Path: @pytest.mark.filterwarnings("ignore:Consolidated metadata is currently not part:UserWarning") @pytest.mark.parametrize("zarr_format", [2, 3]) -def test_consolidated_upgraded_member_stored_by_its_first_write( +def test_consolidated_repaired_member_stored_by_its_first_write( tmp_path: Path, zarr_format: Literal[2, 3] ) -> None: - """Consolidating a group copies the document of a member that had to be upgraded as + """Consolidating a group copies the document of a member that had to be repaired as it is stored, and leaves that document as it is. The first chunk write through the - consolidated metadata stores the member's upgrade before its chunks, the one write - that stores it; consolidating again then copies the upgrade.""" + consolidated metadata stores the member's repair before its chunks, the one write + that stores it; consolidating again then copies the repair.""" path = tmp_path / "group.zarr" document = _consolidated_legacy_member(path, zarr_format) legacy = document.read_bytes() @@ -1166,7 +1166,7 @@ def test_consolidated_upgraded_member_stored_by_its_first_write( @pytest.mark.filterwarnings("ignore:Consolidated metadata is currently not part:UserWarning") @pytest.mark.parametrize("zarr_format", [2, 3]) -def test_consolidated_upgraded_member_write_after_chunk_grid_change_raises( +def test_consolidated_repaired_member_write_after_chunk_grid_change_raises( tmp_path: Path, zarr_format: Literal[2, 3] ) -> None: """A member read from its consolidated copy, whose own document has since been @@ -1188,12 +1188,12 @@ def test_consolidated_upgraded_member_write_after_chunk_grid_change_raises( @pytest.mark.filterwarnings("ignore:Consolidated metadata is currently not part:UserWarning") @pytest.mark.parametrize("zarr_format", [2, 3]) -def test_consolidated_upgraded_member_attributes_kept_by_group_write( +def test_consolidated_repaired_member_attributes_kept_by_group_write( tmp_path: Path, zarr_format: Literal[2, 3] ) -> None: """Setting attributes of a member read from consolidated metadata stores the member's - upgraded document with them. A later write of the group, whose consolidated metadata - shares the member's metadata, then stores that upgrade: the new attributes, not the + repaired document with them. A later write of the group, whose consolidated metadata + shares the member's metadata, then stores that repair: the new attributes, not the legacy document as it was stored before.""" path = tmp_path / "group.zarr" _consolidated_legacy_member(path, zarr_format) @@ -1212,8 +1212,8 @@ def test_consolidated_upgraded_member_attributes_kept_by_group_write( @pytest.mark.parametrize("zarr_format", [2, 3]) -def test_upgraded_metadata_keeps_the_document_it_read(zarr_format: Literal[2, 3]) -> None: - """Metadata read from a document that had to be upgraded keeps that document as it +def test_repaired_metadata_keeps_the_document_it_read(zarr_format: Literal[2, 3]) -> None: + """Metadata read from a document that had to be repaired keeps that document as it was read, whatever becomes of the caller's dict (or the objects in it, which the metadata's attributes may hold): consolidated metadata stores it as read.""" doc = _v2_doc([0], [0]) if zarr_format == 2 else _v3_doc([0], [0]) @@ -1245,7 +1245,7 @@ def test_empty_write_stores_no_metadata(tmp_path: Path, zarr_format: Literal[2, @pytest.mark.parametrize("zarr_format", [2, 3]) def test_async_array_from_dict_names_array(tmp_path: Path, zarr_format: Literal[2, 3]) -> None: - """`AsyncArray.from_dict` names the array at its store path in the upgrade warning.""" + """`AsyncArray.from_dict` names the array at its store path in the repair warning.""" store_path = sync(make_store_path(tmp_path / "legacy.zarr")) doc = _v2_doc([4], [0]) if zarr_format == 2 else _v3_doc([4], [0]) with pytest.warns(ZarrUserWarning, match=f"^Array {re.escape(repr(str(store_path)))}: "): @@ -1255,7 +1255,7 @@ def test_async_array_from_dict_names_array(tmp_path: Path, zarr_format: Literal[ @pytest.mark.filterwarnings("ignore:Consolidated metadata is currently not part:UserWarning") @pytest.mark.parametrize("zarr_format", [2, 3]) def test_legacy_chunk_size_consolidated(tmp_path: Path, zarr_format: Literal[2, 3]) -> None: - """Consolidated metadata goes through the same upgrade as the arrays' own documents, + """Consolidated metadata goes through the same repair as the arrays' own documents, with one warning naming each array by its path; re-saving the arrays and consolidating again leaves a group that opens without a warning.""" path = tmp_path / "group.zarr" From 01dd403d6c66561786c996606c5a364ff0e69e8a Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Thu, 1 Oct 2026 23:01:45 +0200 Subject: [PATCH 76/76] docs(changes): trim the #4375 fragment to what the PR adds over main Rejecting edge lists in a regular chunk grid landed with #4334, and encoding a new node's metadata before deleting the existing one landed with #4410. Assisted-by: ClaudeCode:claude-fable-5-1 Co-Authored-By: Claude Fable 5.1 --- changes/4375.bugfix.md | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/changes/4375.bugfix.md b/changes/4375.bugfix.md index 07f580785b..689d8e3be7 100644 --- a/changes/4375.bugfix.md +++ b/changes/4375.bugfix.md @@ -1,3 +1,3 @@ -Arrays whose stored `regular` chunk grid mixes chunk sizes with lists of chunk edge lengths, such as `"chunk_shape": [2, [5, 10, 5]]` (written by zarr 3.2.0 and 3.2.1 for `chunks=(2, (5, 10, 5))`, and as `[2, [5.0, 10.0, 5.0]]` for float edges), can be read again, without enabling `array.rectilinear_chunks`. The grid is read as the rectilinear chunk grid it describes, with a `ZarrUserWarning`; re-saving the array's metadata (or writing chunks to it) stores that rectilinear grid, which requires the flag. A group's consolidated metadata keeps such an array's metadata as it was stored. A regular chunk grid given edge lists is now rejected, so this metadata is no longer written. +Arrays whose stored `regular` chunk grid mixes chunk sizes with lists of chunk edge lengths, such as `"chunk_shape": [2, [5, 10, 5]]` (written by zarr 3.2.0 and 3.2.1 for `chunks=(2, (5, 10, 5))`, and as `[2, [5.0, 10.0, 5.0]]` for float edges), can be read again, without enabling `array.rectilinear_chunks`. The grid is read as the rectilinear chunk grid it describes, with a `ZarrUserWarning`; re-saving the array's metadata (or writing chunks to it) stores that rectilinear grid, which requires the flag. A group's consolidated metadata keeps such an array's metadata as it was stored. -The `array.rectilinear_chunks` flag now gates reading and storing array metadata that declares a rectilinear chunk grid, including in a group's consolidated metadata, instead of constructing `RectilinearChunkGridMetadata`. `Array.resize`, deleting a group member, `create_hierarchy(..., overwrite=True)`, and creating an array with `overwrite=True` (`zarr.create`, `zarr.create_array` and the functions built on them) now encode the metadata they will store before deleting anything, so metadata that cannot be stored (for example, a rectilinear chunk grid with the flag off) fails with the store untouched. That error names the array when it is a member of a group's consolidated metadata. +The `array.rectilinear_chunks` flag now gates reading and storing array metadata that declares a rectilinear chunk grid, including in a group's consolidated metadata, instead of constructing `RectilinearChunkGridMetadata`. An operation that would store such metadata with the flag off (creating an array, `Array.resize`, deleting a member of a group with consolidated metadata, `create_hierarchy`) raises before it deletes or writes anything, and the error names the array, also when it is a member of a group's consolidated metadata.