From cd7d2641935a8178d4591ff81ed49584e4e3c3b9 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Wed, 9 Sep 2026 20:16:37 +0200 Subject: [PATCH 01/41] fix(chunk-grids): enforce one zero-length-axis invariant across model, clamps and metadata Invariant: a chunk edge length is always >= 1; a dimension's extent may be 0, in which case the dimension has zero chunks (ceildiv(0, size) == 0). Zero-length-axis bugs have recurred since 2017 (#150, #241, #303, #972, #1977, #2434, #3711, #4305, #4307, #4328) because the layers disagreed on this invariant and every span-derived chunk spelling clamped on its own: - The metadata layer (common.py, metadata/v3.py) required chunk edges >= 1, but the in-memory FixedDimension allowed size == 0 with four special-case branches left over from #2434, so normalization could build a grid the metadata constructor then rejected. FixedDimension now rejects size < 1 and the four `if self.size == 0` branches are gone. VaryingDimension already required edges > 0 and is unchanged. - `chunks=-1`, `chunks=False`, `chunks="auto"` (_guess_regular_chunks, both the typesize == 0 early return and the np.maximum line) and `shards="auto"` each derived "one chunk covering the axis" independently. They now all go through one helper, `_full_span_chunk_size(span) = max(span, 1)`, which is the single definition of that phrase for a possibly zero-length axis. - Zarr format 2 metadata had no chunk >= 1 check, so a legacy `chunks: [0]` document opened fine and read uninitialised memory after a resize. It now raises a clear ValueError at parse time, matching the format 3 grid. - Rectilinear grids had no creation-time spelling for a zero-length axis: normalize_chunks_1d required sum(edges) == span, which no list of positive edges can satisfy for span 0, even though the same state is reachable via resize((0,)) and round-trips through reopen. For span == 0 any non-empty list of positive edges is now accepted verbatim, producing the same VaryingDimension(edges, extent=0) that resize produces; the strict sum check is kept for span > 0. Tests: the per-spelling regression test from #4328 is replaced by one matrix over {-1, False, "auto", 1, (1,...), [[2, 2]]} x {(0,), (0, 4), (4, 0), (0, 0), ()} x {v2, v3} x {no shards, shards="auto" with and without a byte budget, explicit shards}, with separate small tests for each error case. Tests that constructed FixedDimension(size=0) now assert it raises, and a zero-extent test covers the behaviour the old special cases were guarding. Assisted-by: ClaudeCode:claude-fable-5-1 --- changes/0000.bugfix.md | 1 + docs/user-guide/arrays.md | 6 + src/zarr/core/chunk_grids.py | 75 ++++++---- src/zarr/core/metadata/v2.py | 7 + tests/test_chunk_grids.py | 229 ++++++++++++++++++++++++------- tests/test_metadata/test_v2.py | 12 ++ tests/test_unified_chunk_grid.py | 77 +++++------ 7 files changed, 284 insertions(+), 123 deletions(-) create mode 100644 changes/0000.bugfix.md diff --git a/changes/0000.bugfix.md b/changes/0000.bugfix.md new file mode 100644 index 0000000000..8b224c3f76 --- /dev/null +++ b/changes/0000.bugfix.md @@ -0,0 +1 @@ +Zero-length dimensions are now handled by a single rule instead of a per-spelling patch: a chunk edge length is always at least 1, while a dimension's extent may be 0 (such a dimension simply has zero chunks). Every way of asking for "one chunk covering the axis" — `chunks=-1`, `chunks=False`, `chunks="auto"`, and `shards="auto"` — now derives the chunk size from the same helper, so they agree on chunk size 1 for a zero-length axis in both Zarr formats. Zarr format 2 metadata now rejects a chunk edge length of 0 with a clear error, matching Zarr format 3, instead of reading uninitialised data after a resize. Rectilinear chunk grids (`chunks=[[...], ...]`) can now be created on a zero-length dimension: since no list of positive edge lengths can sum to 0, the given edge lengths are stored as-is and describe the chunks the dimension will grow into on `append` or `resize`, exactly the state a rectilinear dimension is in after being resized down to 0. The private `FixedDimension(size=0, ...)` model, which previously carried its own zero-size special cases, now raises `ValueError`. diff --git a/docs/user-guide/arrays.md b/docs/user-guide/arrays.md index 707fc1a1a5..ebc4b9aee1 100644 --- a/docs/user-guide/arrays.md +++ b/docs/user-guide/arrays.md @@ -708,6 +708,12 @@ z.append(np.arange(10, dtype='float64')) print(f"After append: shape={z.shape}, chunk_sizes={z.write_chunk_sizes}") ``` +A rectilinear array can also be created with a zero-length dimension: because no +list of positive chunk sizes can sum to 0, the chunk sizes given for such a +dimension are stored as-is and describe the chunks the dimension will grow into +on `append` or `resize` — the same state as resizing an existing rectilinear +dimension down to 0. + ### Compressors and filters Rectilinear arrays work with all codecs — compressors, filters, and checksums. diff --git a/src/zarr/core/chunk_grids.py b/src/zarr/core/chunk_grids.py index 6242994fdf..f759e914b6 100644 --- a/src/zarr/core/chunk_grids.py +++ b/src/zarr/core/chunk_grids.py @@ -45,22 +45,24 @@ @dataclass(frozen=True) class FixedDimension: """Uniform chunk size. Boundary chunks contain less data but are - encoded at full size by the codec pipeline.""" + encoded at full size by the codec pipeline. - size: int # chunk edge length (>= 0) - extent: int # array dimension length + The chunk edge length is always at least 1, matching the invariant the + metadata layer enforces for every stored chunk grid. The extent may be 0: + a zero-length axis simply has zero chunks (``ceildiv(0, size) == 0``). + """ + + size: int # chunk edge length (>= 1) + extent: int # array dimension length (>= 0) nchunks: int = field(init=False, repr=False) ngridcells: int = field(init=False, repr=False) def __post_init__(self) -> None: - if self.size < 0: - raise ValueError(f"FixedDimension size must be >= 0, got {self.size}") + if self.size < 1: + raise ValueError(f"FixedDimension size must be >= 1, got {self.size}") if self.extent < 0: raise ValueError(f"FixedDimension extent must be >= 0, got {self.extent}") - if self.size == 0: - n = 0 - else: - n = ceildiv(self.extent, self.size) + n = ceildiv(self.extent, self.size) object.__setattr__(self, "nchunks", n) object.__setattr__(self, "ngridcells", n) @@ -69,8 +71,6 @@ def index_to_chunk(self, idx: int) -> int: raise IndexError(f"Negative index {idx} is not allowed") if idx >= self.extent: raise IndexError(f"Index {idx} is out of bounds for extent {self.extent}") - if self.size == 0: - return 0 return idx // self.size def chunk_offset(self, chunk_ix: int) -> int: @@ -95,8 +95,6 @@ def data_size(self, chunk_ix: int) -> int: Does not validate *chunk_ix* — callers must ensure it is in ``[0, nchunks)``. Use ``ChunkGrid.__getitem__`` for safe access. """ - if self.size == 0: - return 0 return max(0, min(self.size, self.extent - chunk_ix * self.size)) @property @@ -110,8 +108,6 @@ def _unique_edge_lengths(self) -> Iterable[int]: return (self.size,) def indices_to_chunks(self, indices: npt.NDArray[np.intp]) -> npt.NDArray[np.intp]: - if self.size == 0: - return np.zeros_like(indices) return indices // self.size def with_extent(self, new_extent: int) -> FixedDimension: @@ -640,6 +636,20 @@ class ChunkLayout(NamedTuple): inner: ChunkLayout | None = None +def _full_span_chunk_size(span: int) -> int: + """The edge length of one chunk covering an entire axis of length *span*. + + This is *the* definition of "one chunk spans the axis" for a possibly + zero-length axis. Chunk edge lengths must be at least 1 (the invariant + shared by `FixedDimension`, `VaryingDimension` and the stored chunk grid + metadata), so a zero-length axis gets chunk size 1 and zero chunks. Every + spelling that derives a chunk size from a span — ``chunks=-1``, + ``chunks=False``, ``chunks="auto"``, ``shards="auto"`` — must route + through this helper rather than clamping on its own. + """ + return max(span, 1) + + def _guess_regular_chunks( shape: tuple[int, ...] | int, typesize: int, @@ -677,11 +687,10 @@ def _guess_regular_chunks( shape = (shape,) if typesize == 0: - return shape + return tuple(_full_span_chunk_size(s) for s in shape) ndims = len(shape) - # require chunks to have non-zero length for all dimensions - chunks = np.maximum(np.array(shape, dtype="=f8"), 1) + chunks = np.array([_full_span_chunk_size(s) for s in shape], dtype="=f8") # Determine the optimal chunk size in bytes using a PyTables expression. # This is kept as a float. @@ -724,13 +733,21 @@ def normalize_chunks_1d(chunks: int | Iterable[object], span: int) -> DimensionG the span, and the uniform form is O(1) in the number of chunks — a dimension with `2**62` chunks must not materialize one entry per chunk. - `-1` means "one chunk covering the entire span." + `-1` means "one chunk covering the entire span" (see `_full_span_chunk_size` + for what that means on a zero-length span). Explicit chunk size lists must sum to the span exactly and always produce `VaryingDimension`, even when the sizes happen to be uniform: the input syntax declares the grid kind, so a per-chunk list is preserved as a rectilinear dimension rather than silently collapsed to a regular one, which would change how the dimension grows on resize. For scalar sizes the last chunk may overhang the span. + + The one exception to the sum rule is a zero-length span: no list of + positive edges can sum to 0, so any non-empty list is accepted verbatim + and the edges describe the chunks the axis will grow into on `append` / + `resize`. This is the same state a rectilinear axis reaches when it is + resized down to 0 — `VaryingDimension` allows trailing edges beyond the + extent — so creating at length 0 and shrinking to 0 are indistinguishable. """ # `numbers.Integral` rather than `int` so that numpy integer scalars (which are not # `int` subclasses) take the uniform-chunk path instead of being treated as a sequence. @@ -741,9 +758,7 @@ def normalize_chunks_1d(chunks: int | Iterable[object], span: int) -> DimensionG if chunk_size < -1 or chunk_size == 0: raise ValueError(f"Chunk size must be positive or -1, got {chunk_size}") if chunk_size == -1: - # A zero-length span still gets chunk size 1 (chunk sizes must be positive), - # matching the auto-chunking clamp in _guess_regular_chunks. - return FixedDimension(size=max(span, 1), extent=span) + return FixedDimension(size=_full_span_chunk_size(span), extent=span) return FixedDimension(size=chunk_size, extent=span) else: try: @@ -768,7 +783,9 @@ def normalize_chunks_1d(chunks: int | Iterable[object], span: int) -> DimensionG ints: list[int] = [int(c) for c in chunk_list] # type: ignore[call-overload] if any(c <= 0 for c in ints): raise ValueError(f"All chunk sizes must be positive, got {ints}") - if sum(ints) != span: + # A zero-length span cannot be covered by positive edges; the edges are the + # chunks the axis will grow into, exactly as after ``resize(0)``. + if span > 0 and sum(ints) != span: raise ValueError(f"Chunk sizes {ints} do not sum to span {span}") return VaryingDimension(ints, extent=span) @@ -809,7 +826,7 @@ def normalize_chunks_nd( ) # handle no chunking: one chunk covering every axis. Routed through the -1 sentinel so - # the zero-length-axis clamp lives in one place (normalize_chunks_1d). + # the zero-length-axis rule lives in one place (_full_span_chunk_size). if chunks is False: chunks = -1 @@ -864,8 +881,10 @@ def _guess_num_chunks_per_axis_shard( In other words the shard would be a (2,2,2) grid of (2,2,2) chunks i.e., prod(chunk_shape) * (returned_val ** len(chunk_shape)) * item_size = 256 bytes. - Degenerate chunk shapes — a 0-dimensional shape, or one containing a zero-length - axis — return 1, as the search loop's stopping conditions can never be met. + Degenerate inputs — a 0-dimensional chunk shape, or a zero-byte chunk (``item_size`` + of 0; chunk edge lengths themselves are always at least 1) — return 1, as the + search loop's stopping conditions can never be met. A zero-length *array* axis + needs no special case: the array-bound check fails immediately for it. Parameters ---------- @@ -886,8 +905,8 @@ def _guess_num_chunks_per_axis_shard( if max_bytes < bytes_per_chunk: return 1 num_axes = len(chunk_shape) - # For a 0-dimensional chunk shape or one with a zero-length axis, both loop - # conditions below are constant, so the loop would never terminate. + # For a 0-dimensional chunk shape or a zero-byte chunk, both loop conditions + # below are constant, so the loop would never terminate. if num_axes == 0 or bytes_per_chunk == 0: return 1 chunks_per_shard = 1 diff --git a/src/zarr/core/metadata/v2.py b/src/zarr/core/metadata/v2.py index 5822a228b9..b1fa2d866d 100644 --- a/src/zarr/core/metadata/v2.py +++ b/src/zarr/core/metadata/v2.py @@ -89,6 +89,13 @@ def __init__( """ shape_parsed = parse_shapelike(shape) chunks_parsed = parse_shapelike(chunks) + # Same invariant as the Zarr format 3 chunk grid metadata: every chunk edge + # length is at least 1, even on a zero-length axis. + for dim_idx, chunk in enumerate(chunks_parsed): + if chunk < 1: + raise ValueError( + f"Dimension {dim_idx}: chunk edge length must be >= 1, got {chunk}" + ) compressor_parsed = parse_compressor(compressor) order_parsed = parse_indexing_order(order) dimension_separator_parsed = parse_separator(dimension_separator) diff --git a/tests/test_chunk_grids.py b/tests/test_chunk_grids.py index 40133700a8..86383d31ed 100644 --- a/tests/test_chunk_grids.py +++ b/tests/test_chunk_grids.py @@ -367,66 +367,191 @@ def test_create_0d_array_auto_shards_with_target_shard_size() -> None: assert arr.shards == () -@pytest.mark.parametrize("chunks", [-1, False], ids=["minus-one", "false"]) -@pytest.mark.parametrize("shape", [(0,), (0, 4), (4, 0)], ids=["1d", "2d-lead", "2d-trail"]) +# -- Zero-length dimensions -- +# +# One invariant: a chunk edge length is always >= 1, an extent may be 0. Every spelling +# that derives a chunk size from a span (-1, False, "auto", shards="auto") must agree on +# chunk size 1 for a zero-length axis, in both Zarr formats, with or without sharding. +# Historically each spelling clamped (or failed to clamp) on its own; see #4304, #4305, +# #4307, #4328 and, further back, #150, #241, #303, #972, #1977, #2434, #3711. + +ZeroLengthChunkSpelling = Literal["minus-one", "false", "auto", "one", "ones", "rectilinear"] +ZeroLengthShards = Literal["auto", "auto-budget", "explicit"] | None + +# Spellings whose chunk size is derived from the axis span rather than given explicitly. +_SPAN_DERIVED_SPELLINGS: frozenset[ZeroLengthChunkSpelling] = frozenset( + {"minus-one", "false", "auto"} +) + + +def _zero_length_chunks_arg(spelling: ZeroLengthChunkSpelling, shape: tuple[int, ...]) -> Any: + """Translate a chunk-spelling id into the `chunks=` argument for `shape`.""" + match spelling: + case "minus-one": + return -1 + case "false": + return False + case "auto": + return "auto" + case "one": + return 1 + case "ones": + return (1,) * len(shape) + case "rectilinear": + return [[2, 2]] * len(shape) + + +@pytest.mark.parametrize("spelling", ["minus-one", "false", "auto", "one", "ones", "rectilinear"]) @pytest.mark.parametrize( - ("zarr_format", "shards", "target_shard_size_bytes"), - [ - (2, None, None), - (3, None, None), - (3, "auto", None), - (3, "auto", 128 * 1024 * 1024), - ], - ids=["v2", "v3", "v3-auto-shards", "v3-auto-shards-budget"], + "shape", + [(0,), (0, 4), (4, 0), (0, 0), ()], + ids=["1d", "2d-lead", "2d-trail", "2d-both", "0d"], ) -def test_create_zero_length_array_full_span_chunks( - chunks: int | bool, +@pytest.mark.parametrize( + ("zarr_format", "shards"), + [(2, None), (3, None), (3, "auto"), (3, "auto-budget"), (3, "explicit")], + ids=["v2", "v3", "v3-auto-shards", "v3-auto-shards-budget", "v3-explicit-shards"], +) +def test_create_zero_length_array( + spelling: ZeroLengthChunkSpelling, shape: tuple[int, ...], zarr_format: Literal[2, 3], - shards: Literal["auto"] | None, - target_shard_size_bytes: int | None, + shards: ZeroLengthShards, ) -> None: - """`chunks=-1` and `chunks=False` on a zero-length axis must resolve to chunk size 1. - - Both spellings mean "one chunk covering the whole axis". They used to resolve to chunk - size 0 on zero-length axes, which broke every downstream path differently: a ValueError - from the Zarr format 3 chunk grid metadata, a ZeroDivisionError with shards="auto", an - infinite loop with a shard size budget (https://github.com/zarr-developers/zarr-python/issues/4304), - and invalid `chunks: [0]` metadata for Zarr format 2 that silently corrupted reads after - a resize. + """Every chunk spelling produces a valid, usable grid on a zero-length axis. + + Span-derived spellings resolve to chunk size 1 on zero-length axes (and the full span + elsewhere, for these small shapes); explicit spellings are stored verbatim. In every case + the stored metadata matches `arr.chunks` / `arr.shards`, the array can grow along the + empty axis, round-trip data, and shrink back to empty. """ - expected_chunks = tuple(max(s, 1) for s in shape) + ndim = len(shape) + if spelling == "rectilinear": + if zarr_format == 2: + pytest.skip("Zarr format 2 does not support rectilinear chunk grids") + if shards is not None: + pytest.skip("rectilinear chunks with sharding is not supported") + if ndim == 0: + pytest.skip("a 0-d array has no dimension to chunk rectilinearly") + if shards == "explicit" and ndim == 0: + pytest.skip("a 0-d array has no axis to shard explicitly") + + chunks = _zero_length_chunks_arg(spelling, shape) + expected_chunks: tuple[int, ...] | None + if spelling in _SPAN_DERIVED_SPELLINGS: + expected_chunks = tuple(max(s, 1) for s in shape) + elif spelling == "rectilinear": + expected_chunks = None + else: + expected_chunks = (1,) * ndim + + shards_arg: Any + expected_shards: tuple[int, ...] | None + match shards: + case None: + shards_arg, expected_shards = None, None + case "auto" | "auto-budget": + # Axes this short never split, so the guessed shard equals the chunk. + shards_arg, expected_shards = "auto", expected_chunks + case "explicit": + # A shard larger than the (zero) extent is fine: the axis has zero shards. + shards_arg = tuple(2 if s == 0 else s for s in shape) + expected_shards = shards_arg + warns = ( pytest.warns(ZarrUserWarning, match="Automatic shard shape inference is experimental") - if shards == "auto" + if shards_arg == "auto" else contextlib.nullcontext() ) - with zarr.config.set({"array.target_shard_size_bytes": target_shard_size_bytes}), warns: - arr = zarr.create_array( - store={}, - shape=shape, - dtype="int64", - chunks=chunks, - shards=shards, - zarr_format=zarr_format, - ) - assert arr.chunks == expected_chunks - assert arr.shards == (expected_chunks if shards == "auto" else None) + budget = 128 * 1024 * 1024 if shards == "auto-budget" else None + # The rectilinear flag must stay set for the array's whole life, not just creation. + with zarr.config.set( + {"array.rectilinear_chunks": True, "array.target_shard_size_bytes": budget} + ): + with warns: + arr = zarr.create_array( + store={}, + shape=shape, + dtype="int64", + chunks=chunks, + shards=shards_arg, + zarr_format=zarr_format, + ) - # The stored chunk grid must be the clamped shape, whichever format wrote it. - meta = cast(dict[str, Any], arr.metadata.to_dict()) - if zarr_format == 2: - assert meta["chunks"] == expected_chunks - else: - assert meta["chunk_grid"]["configuration"]["chunk_shape"] == expected_chunks - - # The array must remain usable: grow the empty axis and round-trip data through it. - axis = shape.index(0) - grown = tuple(2 if s == 0 else s for s in shape) - arr.append(np.full(grown, 7, dtype="int64"), axis=axis) - assert arr.shape == grown - np.testing.assert_array_equal(arr[...], np.full(grown, 7, dtype="int64")) - resized = tuple(3 if s == 0 else s for s in shape) - arr.resize(resized) - assert arr.shape == resized - assert int(np.asarray(arr[...]).sum()) == 7 * np.prod(grown) + # In-memory view and stored metadata agree with the invariant. + assert arr.shards == expected_shards + meta = cast(dict[str, Any], arr.metadata.to_dict()) + if spelling == "rectilinear": + grid = meta["chunk_grid"] + assert grid["name"] == "rectilinear" + # Stored verbatim on zero-length axes too, run-length encoded as [size, count]. + assert list(grid["configuration"]["chunk_shapes"]) == [[[2, 2]]] * ndim + assert arr.write_chunk_sizes == tuple(() if s == 0 else (2, 2) for s in shape) + else: + assert arr.chunks == expected_chunks + if zarr_format == 2: + assert meta["chunks"] == expected_chunks + else: + stored = meta["chunk_grid"]["configuration"]["chunk_shape"] + assert stored == (expected_chunks if expected_shards is None else expected_shards) + assert all(c >= 1 for c in arr.chunks) + + # The array must remain usable. + if ndim == 0: + arr[...] = 7 + assert arr[...] == 7 + return + axis = shape.index(0) + grown = tuple(2 if i == axis else s for i, s in enumerate(shape)) + data = np.full(grown, 7, dtype="int64") + arr.append(data, axis=axis) + assert arr.shape == grown + np.testing.assert_array_equal(arr[...], data) + arr.resize(shape) + assert arr.shape == shape + assert np.asarray(arr[...]).shape == shape + + +@pytest.mark.parametrize("zarr_format", [2, 3]) +def test_create_zero_chunk_rejected(zarr_format: Literal[2, 3]) -> None: + """An explicit chunk size of 0 is rejected up front, even for a zero-length axis.""" + with pytest.raises(ValueError, match="Chunk size must be positive or -1, got 0"): + zarr.create_array(store={}, shape=(0,), chunks=(0,), dtype="int64", zarr_format=zarr_format) + + +def test_rectilinear_zero_extent_matches_resize() -> None: + """Creating a rectilinear axis at length 0 equals resizing one down to 0. + + Both leave a `VaryingDimension` whose edges lie entirely beyond the extent, so the + stored grids are identical and both grow into the same chunks on append. + """ + with zarr.config.set({"array.rectilinear_chunks": True}): + created = zarr.create_array(store={}, shape=(0,), chunks=[[2, 2]], dtype="int64") + resized = zarr.create_array(store={}, shape=(4,), chunks=[[2, 2]], dtype="int64") + resized.resize((0,)) + created_meta = cast(dict[str, Any], created.metadata.to_dict()) + resized_meta = cast(dict[str, Any], resized.metadata.to_dict()) + assert created_meta["chunk_grid"] == resized_meta["chunk_grid"] + assert created_meta["shape"] == resized_meta["shape"] == (0,) + + created.append(np.arange(3, dtype="int64")) + resized.append(np.arange(3, dtype="int64")) + np.testing.assert_array_equal(created[...], np.arange(3)) + np.testing.assert_array_equal(resized[...], np.arange(3)) + assert created.write_chunk_sizes == resized.write_chunk_sizes == ((2, 1),) + + +def test_normalize_chunks_1d_zero_span_accepts_any_edges() -> None: + """On a zero-length span the explicit edge list is stored verbatim.""" + dim = normalize_chunks_1d([3, 5], span=0) + assert isinstance(dim, VaryingDimension) + assert dim.edges == (3, 5) + assert dim.extent == 0 + assert dim.nchunks == 0 + assert dim.resize(4) == VaryingDimension([3, 5], extent=4) + + +def test_normalize_chunks_1d_nonzero_span_still_requires_exact_sum() -> None: + """Relaxing the sum rule for span 0 must not leak into positive spans.""" + with pytest.raises(ValueError, match="do not sum to span 1"): + normalize_chunks_1d([3, 5], span=1) diff --git a/tests/test_metadata/test_v2.py b/tests/test_metadata/test_v2.py index 1358f458d6..ac3ee4c029 100644 --- a/tests/test_metadata/test_v2.py +++ b/tests/test_metadata/test_v2.py @@ -309,6 +309,18 @@ def test_from_dict_extra_fields() -> None: assert result == expected +@pytest.mark.parametrize(("shape", "chunks"), [((0,), (0,)), ((4, 0), (4, 0)), ((5,), (0,))]) +def test_zero_chunk_edge_rejected(shape: tuple[int, ...], chunks: tuple[int, ...]) -> None: + """A chunk edge length of 0 is invalid metadata, whatever the array shape. + + Older releases could write `chunks: [0]` for a zero-length axis; such documents read + uninitialised memory once resized. The v2 layer now enforces the same `>= 1` rule as + the Zarr format 3 chunk grid. + """ + with pytest.raises(ValueError, match="chunk edge length must be >= 1, got 0"): + ArrayV2Metadata(shape=shape, dtype=Float64(), chunks=chunks, fill_value=0.0, order="C") + + def test_eq_nan_fill_value() -> None: """Two metadata objects with an identical NaN fill_value compare equal. diff --git a/tests/test_unified_chunk_grid.py b/tests/test_unified_chunk_grid.py index b8289d2135..2cf3f7cc10 100644 --- a/tests/test_unified_chunk_grid.py +++ b/tests/test_unified_chunk_grid.py @@ -149,9 +149,9 @@ def test_rectilinear_feature_flag_enabled() -> None: (10, 100, 1, 10, 10, 10, 10), (10, 100, 9, 10, 10, 10, 90), (10, 95, 9, 10, 10, 5, 90), # boundary chunk - (0, 0, None, 0, None, None, None), # zero-size + (10, 0, None, 0, None, None, None), # zero-extent: no chunks, size still >= 1 ], - ids=["start", "middle", "end", "boundary", "zero-size"], + ids=["start", "middle", "end", "boundary", "zero-extent"], ) def test_fixed_dimension( size: int, @@ -190,11 +190,21 @@ def test_fixed_dimension_indices_to_chunks() -> None: @pytest.mark.parametrize( ("size", "extent", "match"), - [(-1, 100, "must be >= 0"), (10, -1, "must be >= 0")], - ids=["negative-size", "negative-extent"], + [ + (-1, 100, "size must be >= 1"), + (0, 100, "size must be >= 1"), + (0, 0, "size must be >= 1"), + (10, -1, "extent must be >= 0"), + ], + ids=["negative-size", "zero-size", "zero-size-zero-extent", "negative-extent"], ) -def test_fixed_dimension_rejects_negative(size: int, extent: int, match: str) -> None: - """FixedDimension raises ValueError for negative size or extent""" +def test_fixed_dimension_rejects_invalid(size: int, extent: int, match: str) -> None: + """FixedDimension raises ValueError for a size below 1 or a negative extent. + + A chunk edge length of 0 is never valid, whatever the extent: the metadata layer + requires every chunk edge length to be >= 1, and the in-memory model enforces the + same invariant so the two can never disagree. + """ with pytest.raises(ValueError, match=match): FixedDimension(size=size, extent=extent) @@ -1421,46 +1431,27 @@ def test_edge_case_chunk_grid_boundary_shape() -> None: # -- Zero-size and zero-extent -- -@pytest.mark.parametrize( - ("size", "extent"), - [(0, 0), (0, 5), (10, 0)], - ids=["zero-size-zero-extent", "zero-size-nonzero-extent", "zero-extent-nonzero-size"], -) -def test_edge_case_zero_size_or_extent(size: int, extent: int) -> None: - """FixedDimension with zero size or extent has zero chunks and getitem returns None""" - d = FixedDimension(size=size, extent=extent) - assert d.nchunks == 0 - g = ChunkGrid(dimensions=(d,)) - assert g[0] is None - - -def test_edge_case_zero_size_data_and_indices() -> None: - """FixedDimension(size=0) handles data_size, index_to_chunk, and indices_to_chunks safely.""" - d = FixedDimension(size=0, extent=0) - # Zero-sized chunks have zero data - assert d.data_size(0) == 0 - # Vectorized lookup maps every index to chunk 0 (avoids division by zero) - indices = np.array([0, 0, 0], dtype=np.intp) - np.testing.assert_array_equal(d.indices_to_chunks(indices), np.zeros(3, dtype=np.intp)) +@pytest.mark.parametrize("size", [1, 10], ids=["size-1", "size-10"]) +def test_fixed_dimension_zero_extent(size: int) -> None: + """A zero-length axis has zero chunks and behaves like an empty grid. - -def test_edge_case_zero_size_nonzero_extent_index() -> None: - """FixedDimension(size=0, extent>0) maps valid indices to chunk 0 without dividing by zero.""" - d = FixedDimension(size=0, extent=5) + The extent may be 0 even though the chunk size may not: `ceildiv(0, size)` is 0, + so there is nothing to look up, and the vectorized index mapping of an empty index + array is empty. + """ + d = FixedDimension(size=size, extent=0) assert d.nchunks == 0 - # index_to_chunk avoids division by zero and returns 0 - assert d.index_to_chunk(0) == 0 - assert d.index_to_chunk(4) == 0 - - -def test_edge_case_zero_size_data_and_index() -> None: - """FixedDimension(size=0) returns zero for data_size and maps indices to chunk 0.""" - d = FixedDimension(size=0, extent=0) - # data_size returns 0 for a zero-sized chunk + assert d.ngridcells == 0 assert d.data_size(0) == 0 - # vectorized indices_to_chunks returns zeros - indices = np.array([0, 0, 0], dtype=np.intp) - np.testing.assert_array_equal(d.indices_to_chunks(indices), np.zeros(3, dtype=np.intp)) + assert d.with_extent(0) == d + assert d.with_extent(3) == FixedDimension(size=size, extent=3) + empty = np.array([], dtype=np.intp) + np.testing.assert_array_equal(d.indices_to_chunks(empty), empty) + with pytest.raises(IndexError): + d.index_to_chunk(0) + g = ChunkGrid(dimensions=(d,)) + assert g[0] is None + assert list(g) == [] # -- 0-d grid -- From a171507980ed67b19cbeaa9e33ed20604b4ba944 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Wed, 9 Sep 2026 20:20:39 +0200 Subject: [PATCH 02/41] fix(metadata): read a legacy v2 zero chunk edge on an empty axis as 1 MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit zarr-python 2.18.7 writes `chunks: [0]` for `zarr.zeros((0,), chunks=False)` and for `chunks=(0,)`, so stores with that document exist. Rejecting them at open would turn a previously-readable array into an error; leaving the 0 in place read uninitialised memory after a resize. Normalize the edge to 1 with a ZarrUserWarning instead — the same grid every other "one chunk spans the axis" spelling produces — and keep rejecting a zero edge on an axis that has data. Assisted-by: ClaudeCode:claude-fable-5-1 --- changes/0000.bugfix.md | 2 +- src/zarr/core/metadata/v2.py | 22 ++++++++++++++++++---- tests/test_metadata/test_v2.py | 32 +++++++++++++++++++++++++------- 3 files changed, 44 insertions(+), 12 deletions(-) diff --git a/changes/0000.bugfix.md b/changes/0000.bugfix.md index 8b224c3f76..dc0cd52e28 100644 --- a/changes/0000.bugfix.md +++ b/changes/0000.bugfix.md @@ -1 +1 @@ -Zero-length dimensions are now handled by a single rule instead of a per-spelling patch: a chunk edge length is always at least 1, while a dimension's extent may be 0 (such a dimension simply has zero chunks). Every way of asking for "one chunk covering the axis" — `chunks=-1`, `chunks=False`, `chunks="auto"`, and `shards="auto"` — now derives the chunk size from the same helper, so they agree on chunk size 1 for a zero-length axis in both Zarr formats. Zarr format 2 metadata now rejects a chunk edge length of 0 with a clear error, matching Zarr format 3, instead of reading uninitialised data after a resize. Rectilinear chunk grids (`chunks=[[...], ...]`) can now be created on a zero-length dimension: since no list of positive edge lengths can sum to 0, the given edge lengths are stored as-is and describe the chunks the dimension will grow into on `append` or `resize`, exactly the state a rectilinear dimension is in after being resized down to 0. The private `FixedDimension(size=0, ...)` model, which previously carried its own zero-size special cases, now raises `ValueError`. +Zero-length dimensions are now handled by a single rule instead of a per-spelling patch: a chunk edge length is always at least 1, while a dimension's extent may be 0 (such a dimension simply has zero chunks). Every way of asking for "one chunk covering the axis" — `chunks=-1`, `chunks=False`, `chunks="auto"`, and `shards="auto"` — now derives the chunk size from the same helper, so they agree on chunk size 1 for a zero-length axis in both Zarr formats. Zarr format 2 metadata now applies the same rule: a stored chunk edge length of 0 on a zero-length axis — which zarr-python 2.x wrote for `chunks=False` and `chunks=(0,)` — is read as 1 (with a `ZarrUserWarning`) so those stores stay readable and no longer read uninitialised data after a resize, while a chunk edge of 0 on an axis that has data is rejected with a clear error. Rectilinear chunk grids (`chunks=[[...], ...]`) can now be created on a zero-length dimension: since no list of positive edge lengths can sum to 0, the given edge lengths are stored as-is and describe the chunks the dimension will grow into on `append` or `resize`, exactly the state a rectilinear dimension is in after being resized down to 0. The private `FixedDimension(size=0, ...)` model, which previously carried its own zero-size special cases, now raises `ValueError`. diff --git a/src/zarr/core/metadata/v2.py b/src/zarr/core/metadata/v2.py index b1fa2d866d..7d6e43fc61 100644 --- a/src/zarr/core/metadata/v2.py +++ b/src/zarr/core/metadata/v2.py @@ -90,12 +90,26 @@ def __init__( shape_parsed = parse_shapelike(shape) chunks_parsed = parse_shapelike(chunks) # Same invariant as the Zarr format 3 chunk grid metadata: every chunk edge - # length is at least 1, even on a zero-length axis. - for dim_idx, chunk in enumerate(chunks_parsed): + # length is at least 1. zarr-python 2.x wrote `chunks: [0]` for a zero-length + # axis created with `chunks=False` or `chunks=(0,)`. Such an axis holds no + # chunks, so those documents are readable; the edge is normalized to 1 so a + # later resize does not divide by zero. On an axis that has data, 0 is invalid. + normalized_chunks: list[int] = [] + for dim_idx, (extent, chunk) in enumerate(zip(shape_parsed, chunks_parsed, strict=False)): if chunk < 1: - raise ValueError( - f"Dimension {dim_idx}: chunk edge length must be >= 1, got {chunk}" + if chunk < 0 or extent != 0: + raise ValueError( + f"Dimension {dim_idx}: chunk edge length must be >= 1, got {chunk}" + ) + warnings.warn( + f"Dimension {dim_idx}: chunk edge length 0 on a zero-length axis " + "(as written by zarr-python 2.x) is treated as 1.", + ZarrUserWarning, + stacklevel=2, ) + chunk = 1 + normalized_chunks.append(chunk) + chunks_parsed = tuple(normalized_chunks) + chunks_parsed[len(shape_parsed) :] compressor_parsed = parse_compressor(compressor) order_parsed = parse_indexing_order(order) dimension_separator_parsed = parse_separator(dimension_separator) diff --git a/tests/test_metadata/test_v2.py b/tests/test_metadata/test_v2.py index ac3ee4c029..39f5961259 100644 --- a/tests/test_metadata/test_v2.py +++ b/tests/test_metadata/test_v2.py @@ -309,15 +309,33 @@ def test_from_dict_extra_fields() -> None: assert result == expected -@pytest.mark.parametrize(("shape", "chunks"), [((0,), (0,)), ((4, 0), (4, 0)), ((5,), (0,))]) -def test_zero_chunk_edge_rejected(shape: tuple[int, ...], chunks: tuple[int, ...]) -> None: - """A chunk edge length of 0 is invalid metadata, whatever the array shape. +@pytest.mark.parametrize( + ("shape", "chunks", "expected"), + [((0,), (0,), (1,)), ((4, 0), (4, 0), (4, 1)), ((0, 0), (0, 0), (1, 1))], +) +def test_zero_chunk_edge_on_empty_axis_normalized( + shape: tuple[int, ...], chunks: tuple[int, ...], expected: tuple[int, ...] +) -> None: + """A stored chunk edge of 0 on a zero-length axis is read as 1, with a warning. - Older releases could write `chunks: [0]` for a zero-length axis; such documents read - uninitialised memory once resized. The v2 layer now enforces the same `>= 1` rule as - the Zarr format 3 chunk grid. + zarr-python 2.x wrote `chunks: [0]` for such an axis (`chunks=False` or + `chunks=(0,)`), and those documents must stay readable. Left at 0, a later resize + read uninitialised memory; normalizing to 1 gives the axis the same grid every other + "one chunk spans the axis" spelling produces. """ - with pytest.raises(ValueError, match="chunk edge length must be >= 1, got 0"): + with pytest.warns(ZarrUserWarning, match="chunk edge length 0 on a zero-length axis"): + meta = ArrayV2Metadata( + shape=shape, dtype=Float64(), chunks=chunks, fill_value=0.0, order="C" + ) + assert meta.chunks == expected + + +@pytest.mark.parametrize(("shape", "chunks"), [((5,), (0,)), ((4, 3), (4, 0))]) +def test_zero_chunk_edge_with_data_rejected( + shape: tuple[int, ...], chunks: tuple[int, ...] +) -> None: + """A chunk edge of 0 on an axis that has data is invalid metadata.""" + with pytest.raises(ValueError, match="chunk edge length must be >= 1"): ArrayV2Metadata(shape=shape, dtype=Float64(), chunks=chunks, fill_value=0.0, order="C") From 921be0d73916abefd50aa1843d4fa35d841ca17c Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Wed, 9 Sep 2026 20:20:47 +0200 Subject: [PATCH 03/41] chore: rename changelog fragment to the PR number Assisted-by: ClaudeCode:claude-fable-5-1 --- changes/{0000.bugfix.md => 4334.bugfix.md} | 0 1 file changed, 0 insertions(+), 0 deletions(-) rename changes/{0000.bugfix.md => 4334.bugfix.md} (100%) diff --git a/changes/0000.bugfix.md b/changes/4334.bugfix.md similarity index 100% rename from changes/0000.bugfix.md rename to changes/4334.bugfix.md From 047a92e3246e9ead0329be04afd5f22e12ba04bf Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Wed, 9 Sep 2026 20:24:28 +0200 Subject: [PATCH 04/41] docs: state what 2.x actually did with a zero chunk edge Measured against zarr 2.18.7: `zeros((0,), chunks=False)`, `chunks=-1` and `chunks=(0,)` all write `chunks: [0]`, after which nchunks, read, write, append, resize and reopen-then-read every raise ZeroDivisionError. There was never a working behaviour to preserve; normalizing the edge to 1 makes such arrays usable for the first time. Say so in the comment and fragment instead of claiming the stores were previously readable. Assisted-by: ClaudeCode:claude-fable-5-1 --- changes/4334.bugfix.md | 2 +- src/zarr/core/metadata/v2.py | 9 ++++++--- 2 files changed, 7 insertions(+), 4 deletions(-) diff --git a/changes/4334.bugfix.md b/changes/4334.bugfix.md index dc0cd52e28..8a04bec7e1 100644 --- a/changes/4334.bugfix.md +++ b/changes/4334.bugfix.md @@ -1 +1 @@ -Zero-length dimensions are now handled by a single rule instead of a per-spelling patch: a chunk edge length is always at least 1, while a dimension's extent may be 0 (such a dimension simply has zero chunks). Every way of asking for "one chunk covering the axis" — `chunks=-1`, `chunks=False`, `chunks="auto"`, and `shards="auto"` — now derives the chunk size from the same helper, so they agree on chunk size 1 for a zero-length axis in both Zarr formats. Zarr format 2 metadata now applies the same rule: a stored chunk edge length of 0 on a zero-length axis — which zarr-python 2.x wrote for `chunks=False` and `chunks=(0,)` — is read as 1 (with a `ZarrUserWarning`) so those stores stay readable and no longer read uninitialised data after a resize, while a chunk edge of 0 on an axis that has data is rejected with a clear error. Rectilinear chunk grids (`chunks=[[...], ...]`) can now be created on a zero-length dimension: since no list of positive edge lengths can sum to 0, the given edge lengths are stored as-is and describe the chunks the dimension will grow into on `append` or `resize`, exactly the state a rectilinear dimension is in after being resized down to 0. The private `FixedDimension(size=0, ...)` model, which previously carried its own zero-size special cases, now raises `ValueError`. +Zero-length dimensions are now handled by a single rule instead of a per-spelling patch: a chunk edge length is always at least 1, while a dimension's extent may be 0 (such a dimension simply has zero chunks). Every way of asking for "one chunk covering the axis" — `chunks=-1`, `chunks=False`, `chunks="auto"`, and `shards="auto"` — now derives the chunk size from the same helper, so they agree on chunk size 1 for a zero-length axis in both Zarr formats. Zarr format 2 metadata now applies the same rule: a stored chunk edge length of 0 on a zero-length axis — which zarr-python 2.x wrote for `chunks=False`, `chunks=-1` and `chunks=(0,)` and then could never read, write, append to or resize (every operation raised `ZeroDivisionError`), and which 3.0–3.3 opened but lost data on append — is read as 1 with a `ZarrUserWarning`, making such arrays usable, while a chunk edge of 0 on an axis that has data is rejected with a clear error. Rectilinear chunk grids (`chunks=[[...], ...]`) can now be created on a zero-length dimension: since no list of positive edge lengths can sum to 0, the given edge lengths are stored as-is and describe the chunks the dimension will grow into on `append` or `resize`, exactly the state a rectilinear dimension is in after being resized down to 0. The private `FixedDimension(size=0, ...)` model, which previously carried its own zero-size special cases, now raises `ValueError`. diff --git a/src/zarr/core/metadata/v2.py b/src/zarr/core/metadata/v2.py index 7d6e43fc61..a098132ab9 100644 --- a/src/zarr/core/metadata/v2.py +++ b/src/zarr/core/metadata/v2.py @@ -91,9 +91,12 @@ def __init__( chunks_parsed = parse_shapelike(chunks) # Same invariant as the Zarr format 3 chunk grid metadata: every chunk edge # length is at least 1. zarr-python 2.x wrote `chunks: [0]` for a zero-length - # axis created with `chunks=False` or `chunks=(0,)`. Such an axis holds no - # chunks, so those documents are readable; the edge is normalized to 1 so a - # later resize does not divide by zero. On an axis that has data, 0 is invalid. + # axis created with `chunks=False`, `-1` or `(0,)`, and then could not read, + # write, append to or resize the array (every operation divided by zero); + # 3.0-3.3 opened such documents but lost data on append. The axis holds no + # chunks, so the edge is normalized to 1 — the grid every other "one chunk + # spans the axis" spelling produces — which makes the array usable at last. + # On an axis that has data, 0 is invalid and any data was never stored. normalized_chunks: list[int] = [] for dim_idx, (extent, chunk) in enumerate(zip(shape_parsed, chunks_parsed, strict=False)): if chunk < 1: From 2297c62745f18ad698f3357a12796ff34ad95ec6 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Sun, 13 Sep 2026 17:41:59 +0200 Subject: [PATCH 05/41] docs: qualify zero-chunk compatibility history Assisted-by: Codex:GPT-6 --- changes/4334.bugfix.md | 2 +- docs/user-guide/arrays.md | 2 +- src/zarr/core/chunk_grids.py | 4 ++-- src/zarr/core/metadata/v2.py | 14 +++++++------- tests/test_metadata/test_v2.py | 11 +++++------ 5 files changed, 16 insertions(+), 17 deletions(-) diff --git a/changes/4334.bugfix.md b/changes/4334.bugfix.md index 8a04bec7e1..5fc5373ee5 100644 --- a/changes/4334.bugfix.md +++ b/changes/4334.bugfix.md @@ -1 +1 @@ -Zero-length dimensions are now handled by a single rule instead of a per-spelling patch: a chunk edge length is always at least 1, while a dimension's extent may be 0 (such a dimension simply has zero chunks). Every way of asking for "one chunk covering the axis" — `chunks=-1`, `chunks=False`, `chunks="auto"`, and `shards="auto"` — now derives the chunk size from the same helper, so they agree on chunk size 1 for a zero-length axis in both Zarr formats. Zarr format 2 metadata now applies the same rule: a stored chunk edge length of 0 on a zero-length axis — which zarr-python 2.x wrote for `chunks=False`, `chunks=-1` and `chunks=(0,)` and then could never read, write, append to or resize (every operation raised `ZeroDivisionError`), and which 3.0–3.3 opened but lost data on append — is read as 1 with a `ZarrUserWarning`, making such arrays usable, while a chunk edge of 0 on an axis that has data is rejected with a clear error. Rectilinear chunk grids (`chunks=[[...], ...]`) can now be created on a zero-length dimension: since no list of positive edge lengths can sum to 0, the given edge lengths are stored as-is and describe the chunks the dimension will grow into on `append` or `resize`, exactly the state a rectilinear dimension is in after being resized down to 0. The private `FixedDimension(size=0, ...)` model, which previously carried its own zero-size special cases, now raises `ValueError`. +Chunk-grid normalization now shares a helper that selects chunk size 1 for a zero-length axis when inferring a full-span chunk size. Chunk edge lengths remain positive while array extents may be zero. Zarr format 2 metadata with a stored chunk edge of 0 on a zero-length axis is interpreted as chunk size 1 with a `ZarrUserWarning`; a zero chunk edge on a positive-length axis is rejected. This supports legacy metadata such as the zero chunk sizes written by zarr-python 2.18.7 for empty arrays created with `chunks=False`, `chunks=-1`, or `chunks=(0,)`, without claiming that every historical reader behaved identically. Rectilinear chunk grids can be created on a zero-length dimension with a non-empty list of positive chunk sizes; these sizes are retained for subsequent growth. `FixedDimension(size=0, ...)` now raises `ValueError`. diff --git a/docs/user-guide/arrays.md b/docs/user-guide/arrays.md index 1fc0670c1b..e8df79a102 100644 --- a/docs/user-guide/arrays.md +++ b/docs/user-guide/arrays.md @@ -708,7 +708,7 @@ print(f"After append: shape={z.shape}, chunk_sizes={z.write_chunk_sizes}") ``` A rectilinear array can also be created with a zero-length dimension: because no -list of positive chunk sizes can sum to 0, the chunk sizes given for such a +non-empty list of positive chunk sizes can sum to 0, the chunk sizes given for such a dimension are stored as-is and describe the chunks the dimension will grow into on `append` or `resize` — the same state as resizing an existing rectilinear dimension down to 0. diff --git a/src/zarr/core/chunk_grids.py b/src/zarr/core/chunk_grids.py index 340d228e15..545aa3c581 100644 --- a/src/zarr/core/chunk_grids.py +++ b/src/zarr/core/chunk_grids.py @@ -762,8 +762,8 @@ def normalize_chunks_1d(chunks: int | Iterable[object], span: int) -> DimensionG which would change how the dimension grows on resize. For scalar sizes the last chunk may overhang the span. - The one exception to the sum rule is a zero-length span: no list of - positive edges can sum to 0, so any non-empty list is accepted verbatim + The one exception to the sum rule is a zero-length span: no non-empty list of + positive edges can sum to 0, so a non-empty list of positive integers is retained and the edges describe the chunks the axis will grow into on `append` / `resize`. This is the same state a rectilinear axis reaches when it is resized down to 0 — `VaryingDimension` allows trailing edges beyond the diff --git a/src/zarr/core/metadata/v2.py b/src/zarr/core/metadata/v2.py index a098132ab9..6389146e7e 100644 --- a/src/zarr/core/metadata/v2.py +++ b/src/zarr/core/metadata/v2.py @@ -90,13 +90,13 @@ def __init__( shape_parsed = parse_shapelike(shape) chunks_parsed = parse_shapelike(chunks) # Same invariant as the Zarr format 3 chunk grid metadata: every chunk edge - # length is at least 1. zarr-python 2.x wrote `chunks: [0]` for a zero-length - # axis created with `chunks=False`, `-1` or `(0,)`, and then could not read, - # write, append to or resize the array (every operation divided by zero); - # 3.0-3.3 opened such documents but lost data on append. The axis holds no - # chunks, so the edge is normalized to 1 — the grid every other "one chunk - # spans the axis" spelling produces — which makes the array usable at last. - # On an axis that has data, 0 is invalid and any data was never stored. + # length is at least 1. zarr-python 2.18.7 can write `chunks: [0]` for + # a zero-length axis created with `chunks=False`, `-1` or `(0,)`. + # Normalize that empty axis to chunk size 1 so it can use the positive-size + # grid model. This is a compatibility policy for legacy metadata, not a + # statement about every historical reader. A zero chunk size on a + # positive-length axis is rejected; metadata alone cannot establish + # whether the store contains chunk payloads. normalized_chunks: list[int] = [] for dim_idx, (extent, chunk) in enumerate(zip(shape_parsed, chunks_parsed, strict=False)): if chunk < 1: diff --git a/tests/test_metadata/test_v2.py b/tests/test_metadata/test_v2.py index 39f5961259..1e578c033c 100644 --- a/tests/test_metadata/test_v2.py +++ b/tests/test_metadata/test_v2.py @@ -318,10 +318,9 @@ def test_zero_chunk_edge_on_empty_axis_normalized( ) -> None: """A stored chunk edge of 0 on a zero-length axis is read as 1, with a warning. - zarr-python 2.x wrote `chunks: [0]` for such an axis (`chunks=False` or - `chunks=(0,)`), and those documents must stay readable. Left at 0, a later resize - read uninitialised memory; normalizing to 1 gives the axis the same grid every other - "one chunk spans the axis" spelling produces. + This checks the compatibility policy for legacy metadata: the in-memory + chunk size becomes positive while the extent remains zero. It does not + test historical reader behavior or perform array I/O. """ with pytest.warns(ZarrUserWarning, match="chunk edge length 0 on a zero-length axis"): meta = ArrayV2Metadata( @@ -331,10 +330,10 @@ def test_zero_chunk_edge_on_empty_axis_normalized( @pytest.mark.parametrize(("shape", "chunks"), [((5,), (0,)), ((4, 3), (4, 0))]) -def test_zero_chunk_edge_with_data_rejected( +def test_zero_chunk_edge_with_positive_extent_rejected( shape: tuple[int, ...], chunks: tuple[int, ...] ) -> None: - """A chunk edge of 0 on an axis that has data is invalid metadata.""" + """A chunk edge of 0 on a positive-length axis is rejected.""" with pytest.raises(ValueError, match="chunk edge length must be >= 1"): ArrayV2Metadata(shape=shape, dtype=Float64(), chunks=chunks, fill_value=0.0, order="C") From 49564dc4b59c111d7faecbbcc437eecb8554483c Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Sat, 19 Sep 2026 17:50:20 +0200 Subject: [PATCH 06/41] fix(metadata): read a legacy zero chunk size on an empty axis in Zarr format 3 MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The compatibility policy for a stored chunk size of 0 on a zero-length axis covered Zarr format 2 only, so arrays written by zarr-python 3.0 and 3.1 with `chunk_shape: [0]` — or `[false]`, which 3.0 wrote for `chunks=False` — still could not be opened at all. `ArrayV3Metadata` now applies the same policy to a regular chunk grid: a stored chunk size of 0 on a zero-length axis is read as 1 with a `ZarrUserWarning`, and a zero chunk size on a positive-length axis is left for the chunk grid parser to reject. It runs in `__init__` rather than in the grid parser because the policy needs the array shape, which chunk grid metadata does not carry. Both warnings now say how to store a corrected chunk size — open the array writable and call `array.update_attributes({})`, which rewrites the whole document from the parsed metadata — from one shared constant. The Zarr format 2 warning also named only zarr-python 2.x; measured against real installs, every 3.x release before 3.4 wrote a zero chunk size for an empty array too (3.3.0 for `chunks=-1` and `chunks=False`). Tested against stores written by zarr 3.0.10, 3.1.6 and 3.3.0: they open, append without losing data, and re-save to a chunk size that reopens without a warning. Assisted-by: ClaudeCode:claude-opus-5 --- changes/4334.bugfix.md | 4 ++- src/zarr/core/metadata/common.py | 12 ++++++- src/zarr/core/metadata/v2.py | 9 ++--- src/zarr/core/metadata/v3.py | 55 +++++++++++++++++++++++++++-- tests/test_chunk_grids.py | 59 ++++++++++++++++++++++++++++++++ tests/test_metadata/test_v3.py | 32 +++++++++++++++++ 6 files changed, 162 insertions(+), 9 deletions(-) diff --git a/changes/4334.bugfix.md b/changes/4334.bugfix.md index 5fc5373ee5..610413b42e 100644 --- a/changes/4334.bugfix.md +++ b/changes/4334.bugfix.md @@ -1 +1,3 @@ -Chunk-grid normalization now shares a helper that selects chunk size 1 for a zero-length axis when inferring a full-span chunk size. Chunk edge lengths remain positive while array extents may be zero. Zarr format 2 metadata with a stored chunk edge of 0 on a zero-length axis is interpreted as chunk size 1 with a `ZarrUserWarning`; a zero chunk edge on a positive-length axis is rejected. This supports legacy metadata such as the zero chunk sizes written by zarr-python 2.18.7 for empty arrays created with `chunks=False`, `chunks=-1`, or `chunks=(0,)`, without claiming that every historical reader behaved identically. Rectilinear chunk grids can be created on a zero-length dimension with a non-empty list of positive chunk sizes; these sizes are retained for subsequent growth. `FixedDimension(size=0, ...)` now raises `ValueError`. +Chunk-grid normalization now shares a helper that selects chunk size 1 for a zero-length axis when inferring a full-span chunk size. Chunk edge lengths remain positive while array extents may be zero. `FixedDimension(size=0, ...)` now raises `ValueError`. Rectilinear chunk grids can be created on a zero-length dimension with a non-empty list of positive chunk sizes; these sizes are retained for subsequent growth. + +A stored chunk size of 0 on a zero-length axis is read as 1 with a `ZarrUserWarning`, in both Zarr format 2 (`chunks`) and Zarr format 3 (a regular grid's `chunk_shape`, including the JSON `false` written by zarr-python 3.0 for `chunks=False`); a zero chunk size on a positive-length axis is rejected. zarr-python wrote such metadata for empty arrays until 3.4 — Zarr format 2 through 3.3 and by zarr-python 2.x, Zarr format 3 in 3.0 and 3.1 — and appending to one of these Zarr format 2 arrays previously reported success while writing no chunk, so the appended data read back as the fill value. The warning explains how to store a corrected chunk size: open the array writable and call `array.update_attributes({})`. diff --git a/src/zarr/core/metadata/common.py b/src/zarr/core/metadata/common.py index 6367bdb28a..83267bb16f 100644 --- a/src/zarr/core/metadata/common.py +++ b/src/zarr/core/metadata/common.py @@ -1,10 +1,20 @@ from __future__ import annotations -from typing import TYPE_CHECKING +from typing import TYPE_CHECKING, Final if TYPE_CHECKING: from zarr.core.common import JSON +RESAVE_METADATA_HINT: Final = ( + "Re-save the array metadata to store the corrected value: open the array " + "writable and call `array.update_attributes({})`." +) +"""How to persist metadata that was read under a compatibility policy. + +`update_attributes` rewrites the whole metadata document from the parsed +(corrected) metadata, so an empty update is enough. +""" + def parse_attributes(data: dict[str, JSON] | None) -> dict[str, JSON]: if data is None: diff --git a/src/zarr/core/metadata/v2.py b/src/zarr/core/metadata/v2.py index 6389146e7e..b30109ce47 100644 --- a/src/zarr/core/metadata/v2.py +++ b/src/zarr/core/metadata/v2.py @@ -42,7 +42,7 @@ ) from zarr.core.config import config, parse_indexing_order from zarr.core.json_parse import parse_field -from zarr.core.metadata.common import parse_attributes +from zarr.core.metadata.common import RESAVE_METADATA_HINT, parse_attributes class ArrayV2MetadataDict(TypedDict): @@ -90,8 +90,8 @@ def __init__( shape_parsed = parse_shapelike(shape) chunks_parsed = parse_shapelike(chunks) # Same invariant as the Zarr format 3 chunk grid metadata: every chunk edge - # length is at least 1. zarr-python 2.18.7 can write `chunks: [0]` for - # a zero-length axis created with `chunks=False`, `-1` or `(0,)`. + # length is at least 1. zarr-python 2.18.7, and 3.x before 3.4, can write + # `chunks: [0]` for a zero-length axis (e.g. `chunks=False`, `-1` or `(0,)`). # Normalize that empty axis to chunk size 1 so it can use the positive-size # grid model. This is a compatibility policy for legacy metadata, not a # statement about every historical reader. A zero chunk size on a @@ -106,7 +106,8 @@ def __init__( ) warnings.warn( f"Dimension {dim_idx}: chunk edge length 0 on a zero-length axis " - "(as written by zarr-python 2.x) is treated as 1.", + "(as written by zarr-python 2.x, and by 3.x before 3.4) is treated " + f"as 1. {RESAVE_METADATA_HINT}", ZarrUserWarning, stacklevel=2, ) diff --git a/src/zarr/core/metadata/v3.py b/src/zarr/core/metadata/v3.py index 11f3eb593d..0faf042858 100644 --- a/src/zarr/core/metadata/v3.py +++ b/src/zarr/core/metadata/v3.py @@ -1,10 +1,12 @@ from __future__ import annotations import json +import warnings from collections.abc import Iterable, Mapping, Sequence from dataclasses import dataclass, field, replace from typing import TYPE_CHECKING, Any, Final, Literal, NotRequired, TypeGuard, cast +import numpy as np from typing_extensions import TypedDict from zarr.abc.codec import ArrayArrayCodec, ArrayBytesCodec, BytesBytesCodec, Codec @@ -35,8 +37,8 @@ from zarr.core.dtype import VariableLengthUTF8, ZDType, get_data_type_from_json from zarr.core.dtype.common import check_dtype_spec_v3 from zarr.core.json_parse import parse_field -from zarr.core.metadata.common import parse_attributes -from zarr.errors import MetadataValidationError, NodeTypeValidationError +from zarr.core.metadata.common import RESAVE_METADATA_HINT, parse_attributes +from zarr.errors import MetadataValidationError, NodeTypeValidationError, ZarrUserWarning from zarr.registry import get_codec_class if TYPE_CHECKING: @@ -476,6 +478,51 @@ class ArrayMetadataJSON_V3(TypedDict, extra_items=AllowedExtraField): # type: i } +def _read_legacy_zero_chunk_sizes( + chunk_grid: dict[str, JSON] | ChunkGridMetadata | NamedConfig[str, Any], + shape: tuple[int, ...], +) -> dict[str, JSON] | ChunkGridMetadata | NamedConfig[str, Any]: + """Read a stored regular chunk size of 0 on a zero-length axis as 1. + + zarr-python 3.0 and 3.1 stored `chunk_shape: [0]` for an array created with + a zero-length axis and `chunks=(0,)` or `chunks=False` (3.0 stored `false` + for the latter). Chunk sizes must be at least 1, and a zero-length axis has + no chunks for the size to describe until the array grows, so the size is + read as 1 with a warning. This is the policy `ArrayV2Metadata` applies to + Zarr format 2 metadata. It needs the array shape, which the chunk grid + metadata does not carry, so it runs here rather than in the grid parser. + A zero chunk size on a positive-length axis is left for that parser to + reject. + """ + if not isinstance(chunk_grid, Mapping) or chunk_grid.get("name") != "regular": + return chunk_grid + configuration = chunk_grid.get("configuration") + if not isinstance(configuration, Mapping): + return chunk_grid + chunk_shape = configuration.get("chunk_shape") + if not isinstance(chunk_shape, Sequence) or isinstance(chunk_shape, str): + return chunk_grid + if len(chunk_shape) != len(shape): + return chunk_grid # the dimensionality check reports this + normalized = list(chunk_shape) + changed = False + for dim_idx, (size, extent) in enumerate(zip(chunk_shape, shape, strict=True)): + if isinstance(size, int | np.integer) and size == 0 and extent == 0: + warnings.warn( + f"Dimension {dim_idx}: chunk edge length {size!r} on a zero-length axis " + "(as written by zarr-python 3.0 and 3.1) is treated as 1. " + f"{RESAVE_METADATA_HINT}", + ZarrUserWarning, + stacklevel=2, + ) + normalized[dim_idx] = 1 + changed = True + if not changed: + return chunk_grid + corrected: dict[str, JSON] = {**configuration, "chunk_shape": normalized} + return {"name": "regular", "configuration": corrected} + + @dataclass(frozen=True, kw_only=True) class ArrayV3Metadata(Metadata): shape: tuple[int, ...] @@ -510,7 +557,9 @@ def __init__( """ shape_parsed = parse_shapelike(shape) - chunk_grid_parsed = parse_chunk_grid(chunk_grid) + chunk_grid_parsed = parse_chunk_grid( + _read_legacy_zero_chunk_sizes(chunk_grid, shape_parsed) + ) chunk_key_encoding_parsed = parse_chunk_key_encoding(chunk_key_encoding) dimension_names_parsed = parse_dimension_names(dimension_names) # Note: relying on a type method is numpy-specific diff --git a/tests/test_chunk_grids.py b/tests/test_chunk_grids.py index 86383d31ed..4ba5d19f26 100644 --- a/tests/test_chunk_grids.py +++ b/tests/test_chunk_grids.py @@ -1,4 +1,7 @@ import contextlib +import json +import warnings +from pathlib import Path from typing import Any, Literal, cast import numpy as np @@ -541,6 +544,62 @@ def test_rectilinear_zero_extent_matches_resize() -> None: assert created.write_chunk_sizes == resized.write_chunk_sizes == ((2, 1),) +def _store_legacy_zero_chunk(path: Any, zarr_format: Literal[2, 3], stored: Any) -> None: + """Rewrite an array's stored chunk size to *stored*, as older zarr-python did.""" + doc_name = ".zarray" if zarr_format == 2 else "zarr.json" + doc_path = path / doc_name + doc = json.loads(doc_path.read_text()) + if zarr_format == 2: + doc["chunks"] = [stored] + else: + doc["chunk_grid"]["configuration"]["chunk_shape"] = [stored] + doc_path.write_text(json.dumps(doc)) + + +@pytest.mark.parametrize("zarr_format", [2, 3]) +@pytest.mark.parametrize("stored", [0, False], ids=["zero", "false"]) +def test_legacy_zero_chunk_on_empty_axis_round_trip( + tmp_path: Path, zarr_format: Literal[2, 3], stored: Any +) -> None: + """An empty array whose stored chunk size is 0 opens, appends without losing + data, and re-saves as a valid chunk size. + + zarr-python wrote such metadata for an empty array until 3.4: Zarr format 2 + through 3.3, and Zarr format 3 in 3.0 and 3.1, which also wrote JSON `false` + for `chunks=False`. Appending to one of these arrays used to report success + while writing no chunk, so the data read back as the fill value. + """ + path = tmp_path / "legacy.zarr" + zarr.create_array(store=path, shape=(0,), chunks=(4,), dtype="int64", zarr_format=zarr_format) + _store_legacy_zero_chunk(path, zarr_format, stored) + + with pytest.warns(ZarrUserWarning, match="zero-length axis"): + arr = zarr.open_array(store=path, mode="a") + assert arr.chunks == (1,) + + arr.append(np.arange(3, dtype="int64")) + np.testing.assert_array_equal(zarr.open_array(store=path)[...], np.arange(3)) + + # The warning says to do this; it must leave metadata that reopens cleanly. + arr.update_attributes({}) + with warnings.catch_warnings(): + warnings.simplefilter("error", ZarrUserWarning) + reopened = zarr.open_array(store=path) + assert reopened.chunks == (1,) + + +@pytest.mark.parametrize("zarr_format", [2, 3]) +def test_stored_zero_chunk_on_positive_axis_rejected( + tmp_path: Path, zarr_format: Literal[2, 3] +) -> None: + """A stored chunk size of 0 is only tolerated on a zero-length axis.""" + path = tmp_path / "bad.zarr" + zarr.create_array(store=path, shape=(5,), chunks=(4,), dtype="int64", zarr_format=zarr_format) + _store_legacy_zero_chunk(path, zarr_format, 0) + with pytest.raises(ValueError, match="must be >= 1"): + zarr.open_array(store=path) + + def test_normalize_chunks_1d_zero_span_accepts_any_edges() -> None: """On a zero-length span the explicit edge list is stored verbatim.""" dim = normalize_chunks_1d([3, 5], span=0) diff --git a/tests/test_metadata/test_v3.py b/tests/test_metadata/test_v3.py index 9f78ed7b70..664fb6f092 100644 --- a/tests/test_metadata/test_v3.py +++ b/tests/test_metadata/test_v3.py @@ -463,3 +463,35 @@ def test_group_metadata_to_dict_consolidated(attributes: dict[str, Any] | None) }, }, } + + +@pytest.mark.parametrize( + ("shape", "chunk_shape", "expected"), + [ + ((0,), [0], (1,)), + ((4, 0), [4, 0], (4, 1)), + ((0, 0), [0, 0], (1, 1)), + ((0,), [False], (1,)), + ], + ids=["1d", "one-empty-axis", "all-empty-axes", "json-false"], +) +def test_zero_chunk_size_on_empty_axis_normalized( + shape: tuple[int, ...], chunk_shape: list[Any], expected: tuple[int, ...] +) -> None: + """A stored regular chunk size of 0 on a zero-length axis is read as 1, with a warning. + + zarr-python 3.0 and 3.1 wrote this for an array created with a zero-length + axis; 3.0 wrote JSON `false` for `chunks=False`. This is the Zarr format 3 + counterpart of the policy `ArrayV2Metadata` applies. + """ + from zarr.core.metadata.v3 import RegularChunkGridMetadata + from zarr.errors import ZarrUserWarning + + d = minimal_metadata_dict_v3( + shape=shape, + chunk_grid={"name": "regular", "configuration": {"chunk_shape": chunk_shape}}, + ) + with pytest.warns(ZarrUserWarning, match="zero-length axis"): + meta = ArrayV3Metadata.from_dict(d) # type: ignore[arg-type] + assert isinstance(meta.chunk_grid, RegularChunkGridMetadata) + assert meta.chunk_grid.chunk_shape == expected From 5bb1f97e1ab67093dd6539ca126ecc6b99e520d8 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Sat, 19 Sep 2026 18:05:39 +0200 Subject: [PATCH 07/41] refactor(metadata): check stored chunk shapes against the array shape in one routine The legacy zero-chunk policy was written out twice: inline in `ArrayV2Metadata.__init__`, and again in a Zarr format 3 helper. The V2 copy also zipped with `strict=False` and re-appended any trailing chunk entries only so that a separate length check, `parse_metadata`, could report a dimensionality mismatch after construction. `parse_stored_chunk_shape` in `zarr.core.metadata.common` is now the one place a stored chunk shape is checked against its array's shape, for both formats: one entry per axis, every integer chunk size at least 1, and a size of 0 (or JSON `false`) on a zero-length axis read as 1 with a warning that names the writer and how to re-save. Non-integer entries, such as edge lists, pass through for the caller's own parser. `ArrayV2Metadata.__init__` calls it directly and `parse_metadata` is gone. The Zarr format 3 adapter only locates a regular grid's `chunk_shape` in the stored document and hands it over; it still runs in `ArrayV3Metadata.__init__` because chunk grid metadata has no array shape. The Zarr format 3 import changes that existed only for the old helper are reverted. Tests for the policy now target the routine: one table of valid and legacy inputs, and one test per rejection (dimension mismatch, zero on a non-empty axis, negative). They replace metadata-level tests in test_v2.py and test_v3.py that only re-tested the same rules; the end-to-end tests still cover both formats' wiring against stored arrays. Assisted-by: ClaudeCode:claude-opus-5 --- src/zarr/core/metadata/common.py | 50 ++++++++++++++++++++- src/zarr/core/metadata/v2.py | 44 +++--------------- src/zarr/core/metadata/v3.py | 46 +++++-------------- tests/test_metadata/test_common.py | 72 ++++++++++++++++++++++++++++++ tests/test_metadata/test_v2.py | 29 ------------ tests/test_metadata/test_v3.py | 32 ------------- 6 files changed, 139 insertions(+), 134 deletions(-) create mode 100644 tests/test_metadata/test_common.py diff --git a/src/zarr/core/metadata/common.py b/src/zarr/core/metadata/common.py index 83267bb16f..92fb2b447e 100644 --- a/src/zarr/core/metadata/common.py +++ b/src/zarr/core/metadata/common.py @@ -1,8 +1,14 @@ from __future__ import annotations -from typing import TYPE_CHECKING, Final +import numbers +import warnings +from typing import TYPE_CHECKING, Any, Final + +from zarr.errors import ZarrUserWarning if TYPE_CHECKING: + from collections.abc import Sequence + from zarr.core.common import JSON RESAVE_METADATA_HINT: Final = ( @@ -21,3 +27,45 @@ def parse_attributes(data: dict[str, JSON] | None) -> dict[str, JSON]: return {} return dict(data) + + +def parse_stored_chunk_shape( + chunk_shape: Sequence[Any], shape: Sequence[int], *, legacy_writers: str +) -> tuple[Any, ...]: + """Validate a stored chunk shape against the array shape. + + This is the one place a stored per-axis chunk shape is checked against the + array it belongs to, for both Zarr formats. The chunk shape must have one + entry per array axis, and every integer chunk size must be at least 1. + + The exception is a chunk size of 0 on a zero-length axis, which + `legacy_writers` stored for arrays created empty. An empty axis has no + chunk for the size to describe, so it is read as 1 with a + `ZarrUserWarning` that says how to re-save valid metadata. A chunk size of + 0 on a positive-length axis is rejected, because the metadata cannot say + how the stored chunks were laid out. JSON `false` counts as 0. + + Entries that are not integers, such as explicit chunk edge lists, are + returned unchanged for the caller's own parser. + """ + if len(chunk_shape) != len(shape): + raise ValueError( + f"The chunk shape {tuple(chunk_shape)} and the array shape {tuple(shape)} " + "must have the same number of dimensions." + ) + parsed: list[Any] = [] + for dim_idx, (size, extent) in enumerate(zip(chunk_shape, shape, strict=True)): + if isinstance(size, numbers.Integral) and size < 1: + if size < 0 or extent != 0: + raise ValueError( + f"Dimension {dim_idx}: chunk edge length must be >= 1, got {size!r}" + ) + warnings.warn( + f"Dimension {dim_idx}: chunk edge length {size!r} on a zero-length axis " + f"(as written by {legacy_writers}) is treated as 1. {RESAVE_METADATA_HINT}", + ZarrUserWarning, + stacklevel=3, + ) + size = 1 + parsed.append(size) + return tuple(parsed) diff --git a/src/zarr/core/metadata/v2.py b/src/zarr/core/metadata/v2.py index b30109ce47..5513f1337b 100644 --- a/src/zarr/core/metadata/v2.py +++ b/src/zarr/core/metadata/v2.py @@ -42,7 +42,7 @@ ) from zarr.core.config import config, parse_indexing_order from zarr.core.json_parse import parse_field -from zarr.core.metadata.common import RESAVE_METADATA_HINT, parse_attributes +from zarr.core.metadata.common import parse_attributes, parse_stored_chunk_shape class ArrayV2MetadataDict(TypedDict): @@ -88,32 +88,11 @@ def __init__( Metadata for a Zarr format 2 array. """ shape_parsed = parse_shapelike(shape) - chunks_parsed = parse_shapelike(chunks) - # Same invariant as the Zarr format 3 chunk grid metadata: every chunk edge - # length is at least 1. zarr-python 2.18.7, and 3.x before 3.4, can write - # `chunks: [0]` for a zero-length axis (e.g. `chunks=False`, `-1` or `(0,)`). - # Normalize that empty axis to chunk size 1 so it can use the positive-size - # grid model. This is a compatibility policy for legacy metadata, not a - # statement about every historical reader. A zero chunk size on a - # positive-length axis is rejected; metadata alone cannot establish - # whether the store contains chunk payloads. - normalized_chunks: list[int] = [] - for dim_idx, (extent, chunk) in enumerate(zip(shape_parsed, chunks_parsed, strict=False)): - if chunk < 1: - if chunk < 0 or extent != 0: - raise ValueError( - f"Dimension {dim_idx}: chunk edge length must be >= 1, got {chunk}" - ) - warnings.warn( - f"Dimension {dim_idx}: chunk edge length 0 on a zero-length axis " - "(as written by zarr-python 2.x, and by 3.x before 3.4) is treated " - f"as 1. {RESAVE_METADATA_HINT}", - ZarrUserWarning, - stacklevel=2, - ) - chunk = 1 - normalized_chunks.append(chunk) - chunks_parsed = tuple(normalized_chunks) + chunks_parsed[len(shape_parsed) :] + chunks_parsed = parse_stored_chunk_shape( + parse_shapelike(chunks), + shape_parsed, + legacy_writers="zarr-python 2.x, and by 3.x before 3.4", + ) compressor_parsed = parse_compressor(compressor) order_parsed = parse_indexing_order(order) dimension_separator_parsed = parse_separator(dimension_separator) @@ -136,7 +115,6 @@ def __init__( object.__setattr__(self, "attributes", attributes_parsed) # ensure that the metadata document is consistent - _ = parse_metadata(self) @property def ndim(self) -> int: @@ -348,16 +326,6 @@ def parse_compressor(data: object) -> Numcodec | None: raise ValueError(msg) -def parse_metadata(data: ArrayV2Metadata) -> ArrayV2Metadata: - if (l_chunks := len(data.chunks)) != (l_shape := len(data.shape)): - msg = ( - f"The `shape` and `chunks` attributes must have the same length. " - f"`chunks` has length {l_chunks}, but `shape` has length {l_shape}." - ) - raise ValueError(msg) - return data - - def get_object_codec_id(maybe_object_codecs: Sequence[JSON]) -> str | None: """ Inspect a sequence of codecs / filters for an "object codec", i.e. a codec diff --git a/src/zarr/core/metadata/v3.py b/src/zarr/core/metadata/v3.py index 0faf042858..962564243a 100644 --- a/src/zarr/core/metadata/v3.py +++ b/src/zarr/core/metadata/v3.py @@ -1,12 +1,10 @@ from __future__ import annotations import json -import warnings from collections.abc import Iterable, Mapping, Sequence from dataclasses import dataclass, field, replace from typing import TYPE_CHECKING, Any, Final, Literal, NotRequired, TypeGuard, cast -import numpy as np from typing_extensions import TypedDict from zarr.abc.codec import ArrayArrayCodec, ArrayBytesCodec, BytesBytesCodec, Codec @@ -37,8 +35,8 @@ from zarr.core.dtype import VariableLengthUTF8, ZDType, get_data_type_from_json from zarr.core.dtype.common import check_dtype_spec_v3 from zarr.core.json_parse import parse_field -from zarr.core.metadata.common import RESAVE_METADATA_HINT, parse_attributes -from zarr.errors import MetadataValidationError, NodeTypeValidationError, ZarrUserWarning +from zarr.core.metadata.common import parse_attributes, parse_stored_chunk_shape +from zarr.errors import MetadataValidationError, NodeTypeValidationError from zarr.registry import get_codec_class if TYPE_CHECKING: @@ -482,17 +480,12 @@ def _read_legacy_zero_chunk_sizes( chunk_grid: dict[str, JSON] | ChunkGridMetadata | NamedConfig[str, Any], shape: tuple[int, ...], ) -> dict[str, JSON] | ChunkGridMetadata | NamedConfig[str, Any]: - """Read a stored regular chunk size of 0 on a zero-length axis as 1. - - zarr-python 3.0 and 3.1 stored `chunk_shape: [0]` for an array created with - a zero-length axis and `chunks=(0,)` or `chunks=False` (3.0 stored `false` - for the latter). Chunk sizes must be at least 1, and a zero-length axis has - no chunks for the size to describe until the array grows, so the size is - read as 1 with a warning. This is the policy `ArrayV2Metadata` applies to - Zarr format 2 metadata. It needs the array shape, which the chunk grid - metadata does not carry, so it runs here rather than in the grid parser. - A zero chunk size on a positive-length axis is left for that parser to - reject. + """Check a stored regular grid's chunk shape against the array shape. + + zarr-python 3.0 and 3.1 stored `chunk_shape: [0]` (and 3.0 `[false]`) for + an array created with a zero-length axis; `parse_stored_chunk_shape` + decides what that means. It needs the array shape, which chunk grid + metadata does not carry, so this runs here rather than in the grid parser. """ if not isinstance(chunk_grid, Mapping) or chunk_grid.get("name") != "regular": return chunk_grid @@ -502,25 +495,10 @@ def _read_legacy_zero_chunk_sizes( chunk_shape = configuration.get("chunk_shape") if not isinstance(chunk_shape, Sequence) or isinstance(chunk_shape, str): return chunk_grid - if len(chunk_shape) != len(shape): - return chunk_grid # the dimensionality check reports this - normalized = list(chunk_shape) - changed = False - for dim_idx, (size, extent) in enumerate(zip(chunk_shape, shape, strict=True)): - if isinstance(size, int | np.integer) and size == 0 and extent == 0: - warnings.warn( - f"Dimension {dim_idx}: chunk edge length {size!r} on a zero-length axis " - "(as written by zarr-python 3.0 and 3.1) is treated as 1. " - f"{RESAVE_METADATA_HINT}", - ZarrUserWarning, - stacklevel=2, - ) - normalized[dim_idx] = 1 - changed = True - if not changed: - return chunk_grid - corrected: dict[str, JSON] = {**configuration, "chunk_shape": normalized} - return {"name": "regular", "configuration": corrected} + parsed = parse_stored_chunk_shape(chunk_shape, shape, legacy_writers="zarr-python 3.0 and 3.1") + corrected: dict[str, Any] = dict(chunk_grid) + corrected["configuration"] = {**configuration, "chunk_shape": list(parsed)} + return corrected @dataclass(frozen=True, kw_only=True) diff --git a/tests/test_metadata/test_common.py b/tests/test_metadata/test_common.py new file mode 100644 index 0000000000..18e6a47e1c --- /dev/null +++ b/tests/test_metadata/test_common.py @@ -0,0 +1,72 @@ +"""Tests for metadata helpers shared by both Zarr formats.""" + +from __future__ import annotations + +import warnings +from typing import Any + +import numpy as np +import pytest + +from zarr.core.metadata.common import parse_stored_chunk_shape +from zarr.errors import ZarrUserWarning + + +@pytest.mark.parametrize( + ("chunk_shape", "shape", "expected", "warns"), + [ + ((4, 5), (10, 10), (4, 5), False), + ((1, 1), (0, 0), (1, 1), False), + ((0,), (0,), (1,), True), + ((False,), (0,), (1,), True), + ((np.int64(0),), (0,), (1,), True), + ((4, 0), (4, 0), (4, 1), True), + ((0, 0), (0, 0), (1, 1), True), + ((2, [5, 10, 5]), (6, 20), (2, [5, 10, 5]), False), + ], + ids=[ + "valid", + "valid-on-empty-axes", + "legacy-zero", + "legacy-json-false", + "legacy-numpy-zero", + "only-empty-axis-corrected", + "every-empty-axis-corrected", + "edge-list-passed-through", + ], +) +def test_parse_stored_chunk_shape( + chunk_shape: tuple[Any, ...], shape: tuple[int, ...], expected: tuple[Any, ...], warns: bool +) -> None: + """A valid chunk shape is returned as is; a chunk size of 0 on a zero-length + axis is read as 1, with a warning naming the writer and how to re-save.""" + with warnings.catch_warnings(record=True) as record: + warnings.simplefilter("always") + parsed = parse_stored_chunk_shape(chunk_shape, shape, legacy_writers="an old writer") + assert parsed == expected + messages = [str(w.message) for w in record if issubclass(w.category, ZarrUserWarning)] + assert bool(messages) is warns + for message in messages: + assert "an old writer" in message + assert "update_attributes({})" in message + + +def test_parse_stored_chunk_shape_rejects_dimension_mismatch() -> None: + """The chunk shape needs one entry per array axis.""" + with pytest.raises(ValueError, match="same number of dimensions"): + parse_stored_chunk_shape((4,), (10, 10), legacy_writers="an old writer") + + +@pytest.mark.parametrize(("chunk_shape", "shape"), [((0,), (5,)), ((4, 0), (4, 3))]) +def test_parse_stored_chunk_shape_rejects_zero_on_nonempty_axis( + chunk_shape: tuple[int, ...], shape: tuple[int, ...] +) -> None: + """A chunk size of 0 is only tolerated on a zero-length axis.""" + with pytest.raises(ValueError, match="chunk edge length must be >= 1, got 0"): + parse_stored_chunk_shape(chunk_shape, shape, legacy_writers="an old writer") + + +def test_parse_stored_chunk_shape_rejects_negative() -> None: + """A negative chunk size is rejected even on a zero-length axis.""" + with pytest.raises(ValueError, match="chunk edge length must be >= 1, got -1"): + parse_stored_chunk_shape((-1,), (0,), legacy_writers="an old writer") diff --git a/tests/test_metadata/test_v2.py b/tests/test_metadata/test_v2.py index 1e578c033c..1358f458d6 100644 --- a/tests/test_metadata/test_v2.py +++ b/tests/test_metadata/test_v2.py @@ -309,35 +309,6 @@ def test_from_dict_extra_fields() -> None: assert result == expected -@pytest.mark.parametrize( - ("shape", "chunks", "expected"), - [((0,), (0,), (1,)), ((4, 0), (4, 0), (4, 1)), ((0, 0), (0, 0), (1, 1))], -) -def test_zero_chunk_edge_on_empty_axis_normalized( - shape: tuple[int, ...], chunks: tuple[int, ...], expected: tuple[int, ...] -) -> None: - """A stored chunk edge of 0 on a zero-length axis is read as 1, with a warning. - - This checks the compatibility policy for legacy metadata: the in-memory - chunk size becomes positive while the extent remains zero. It does not - test historical reader behavior or perform array I/O. - """ - with pytest.warns(ZarrUserWarning, match="chunk edge length 0 on a zero-length axis"): - meta = ArrayV2Metadata( - shape=shape, dtype=Float64(), chunks=chunks, fill_value=0.0, order="C" - ) - assert meta.chunks == expected - - -@pytest.mark.parametrize(("shape", "chunks"), [((5,), (0,)), ((4, 3), (4, 0))]) -def test_zero_chunk_edge_with_positive_extent_rejected( - shape: tuple[int, ...], chunks: tuple[int, ...] -) -> None: - """A chunk edge of 0 on a positive-length axis is rejected.""" - with pytest.raises(ValueError, match="chunk edge length must be >= 1"): - ArrayV2Metadata(shape=shape, dtype=Float64(), chunks=chunks, fill_value=0.0, order="C") - - def test_eq_nan_fill_value() -> None: """Two metadata objects with an identical NaN fill_value compare equal. diff --git a/tests/test_metadata/test_v3.py b/tests/test_metadata/test_v3.py index 664fb6f092..9f78ed7b70 100644 --- a/tests/test_metadata/test_v3.py +++ b/tests/test_metadata/test_v3.py @@ -463,35 +463,3 @@ def test_group_metadata_to_dict_consolidated(attributes: dict[str, Any] | None) }, }, } - - -@pytest.mark.parametrize( - ("shape", "chunk_shape", "expected"), - [ - ((0,), [0], (1,)), - ((4, 0), [4, 0], (4, 1)), - ((0, 0), [0, 0], (1, 1)), - ((0,), [False], (1,)), - ], - ids=["1d", "one-empty-axis", "all-empty-axes", "json-false"], -) -def test_zero_chunk_size_on_empty_axis_normalized( - shape: tuple[int, ...], chunk_shape: list[Any], expected: tuple[int, ...] -) -> None: - """A stored regular chunk size of 0 on a zero-length axis is read as 1, with a warning. - - zarr-python 3.0 and 3.1 wrote this for an array created with a zero-length - axis; 3.0 wrote JSON `false` for `chunks=False`. This is the Zarr format 3 - counterpart of the policy `ArrayV2Metadata` applies. - """ - from zarr.core.metadata.v3 import RegularChunkGridMetadata - from zarr.errors import ZarrUserWarning - - d = minimal_metadata_dict_v3( - shape=shape, - chunk_grid={"name": "regular", "configuration": {"chunk_shape": chunk_shape}}, - ) - with pytest.warns(ZarrUserWarning, match="zero-length axis"): - meta = ArrayV3Metadata.from_dict(d) # type: ignore[arg-type] - assert isinstance(meta.chunk_grid, RegularChunkGridMetadata) - assert meta.chunk_grid.chunk_shape == expected From f93193de1eead0419c4e3e53f4ae3d93a064e45c Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Sat, 19 Sep 2026 18:17:02 +0200 Subject: [PATCH 08/41] refactor(metadata): scope the stored chunk shape check to regular chunk grids `parse_stored_chunk_shape` passed non-integer entries through "for the caller's own parser", which made a regular-grid policy look like a general chunk shape routine and let it decide what a 0-length chunk means for grids it does not own. A rectilinear grid, or any other grid, is free to define its own semantics for 0-length chunks. It is now `parse_stored_regular_chunk_shape`, typed `Sequence[int]`, with no pass-through, and its docstring says it applies to Zarr format 2 `chunks` and Zarr format 3 `regular` grids only. The Zarr format 3 caller hands it a chunk shape only when the grid is named `regular` and every entry is an integer (`_is_regular_chunk_shape`); anything else is not a regular chunk shape and goes to the chunk grid parser untouched. Assisted-by: ClaudeCode:claude-opus-5 --- src/zarr/core/metadata/common.py | 32 +++++++++++------------- src/zarr/core/metadata/v2.py | 4 +-- src/zarr/core/metadata/v3.py | 40 ++++++++++++++++++++++-------- tests/test_metadata/test_common.py | 22 ++++++++-------- 4 files changed, 57 insertions(+), 41 deletions(-) diff --git a/src/zarr/core/metadata/common.py b/src/zarr/core/metadata/common.py index 92fb2b447e..6e6bab48a5 100644 --- a/src/zarr/core/metadata/common.py +++ b/src/zarr/core/metadata/common.py @@ -1,8 +1,7 @@ from __future__ import annotations -import numbers import warnings -from typing import TYPE_CHECKING, Any, Final +from typing import TYPE_CHECKING, Final from zarr.errors import ZarrUserWarning @@ -29,33 +28,32 @@ def parse_attributes(data: dict[str, JSON] | None) -> dict[str, JSON]: return dict(data) -def parse_stored_chunk_shape( - chunk_shape: Sequence[Any], shape: Sequence[int], *, legacy_writers: str -) -> tuple[Any, ...]: - """Validate a stored chunk shape against the array shape. +def parse_stored_regular_chunk_shape( + chunk_shape: Sequence[int], shape: Sequence[int], *, legacy_writers: str +) -> tuple[int, ...]: + """Validate a stored regular chunk grid's chunk shape against the array shape. - This is the one place a stored per-axis chunk shape is checked against the - array it belongs to, for both Zarr formats. The chunk shape must have one - entry per array axis, and every integer chunk size must be at least 1. + This is for regular chunk grids only: Zarr format 2 `chunks`, and the + `chunk_shape` of a Zarr format 3 `regular` grid. Another chunk grid, such + as the rectilinear grid, is free to define its own meaning for a chunk of + length 0, so its chunk sizes must not be passed here. - The exception is a chunk size of 0 on a zero-length axis, which - `legacy_writers` stored for arrays created empty. An empty axis has no - chunk for the size to describe, so it is read as 1 with a + The chunk shape must have one entry per array axis, and every chunk size + must be at least 1. The exception is a chunk size of 0 on a zero-length + axis, which `legacy_writers` stored for arrays created empty. An empty axis + has no chunk for the size to describe, so it is read as 1 with a `ZarrUserWarning` that says how to re-save valid metadata. A chunk size of 0 on a positive-length axis is rejected, because the metadata cannot say how the stored chunks were laid out. JSON `false` counts as 0. - - Entries that are not integers, such as explicit chunk edge lists, are - returned unchanged for the caller's own parser. """ if len(chunk_shape) != len(shape): raise ValueError( f"The chunk shape {tuple(chunk_shape)} and the array shape {tuple(shape)} " "must have the same number of dimensions." ) - parsed: list[Any] = [] + parsed: list[int] = [] for dim_idx, (size, extent) in enumerate(zip(chunk_shape, shape, strict=True)): - if isinstance(size, numbers.Integral) and size < 1: + if size < 1: if size < 0 or extent != 0: raise ValueError( f"Dimension {dim_idx}: chunk edge length must be >= 1, got {size!r}" diff --git a/src/zarr/core/metadata/v2.py b/src/zarr/core/metadata/v2.py index 5513f1337b..c213bc98b6 100644 --- a/src/zarr/core/metadata/v2.py +++ b/src/zarr/core/metadata/v2.py @@ -42,7 +42,7 @@ ) from zarr.core.config import config, parse_indexing_order from zarr.core.json_parse import parse_field -from zarr.core.metadata.common import parse_attributes, parse_stored_chunk_shape +from zarr.core.metadata.common import parse_attributes, parse_stored_regular_chunk_shape class ArrayV2MetadataDict(TypedDict): @@ -88,7 +88,7 @@ def __init__( Metadata for a Zarr format 2 array. """ shape_parsed = parse_shapelike(shape) - chunks_parsed = parse_stored_chunk_shape( + chunks_parsed = parse_stored_regular_chunk_shape( parse_shapelike(chunks), shape_parsed, legacy_writers="zarr-python 2.x, and by 3.x before 3.4", diff --git a/src/zarr/core/metadata/v3.py b/src/zarr/core/metadata/v3.py index 962564243a..c5a70e2be4 100644 --- a/src/zarr/core/metadata/v3.py +++ b/src/zarr/core/metadata/v3.py @@ -5,6 +5,7 @@ from dataclasses import dataclass, field, replace from typing import TYPE_CHECKING, Any, Final, Literal, NotRequired, TypeGuard, cast +import numpy as np from typing_extensions import TypedDict from zarr.abc.codec import ArrayArrayCodec, ArrayBytesCodec, BytesBytesCodec, Codec @@ -35,7 +36,7 @@ from zarr.core.dtype import VariableLengthUTF8, ZDType, get_data_type_from_json from zarr.core.dtype.common import check_dtype_spec_v3 from zarr.core.json_parse import parse_field -from zarr.core.metadata.common import parse_attributes, parse_stored_chunk_shape +from zarr.core.metadata.common import parse_attributes, parse_stored_regular_chunk_shape from zarr.errors import MetadataValidationError, NodeTypeValidationError from zarr.registry import get_codec_class @@ -476,16 +477,31 @@ class ArrayMetadataJSON_V3(TypedDict, extra_items=AllowedExtraField): # type: i } -def _read_legacy_zero_chunk_sizes( +def _is_regular_chunk_shape(value: object) -> TypeGuard[Sequence[int]]: + """Whether a stored `chunk_shape` is a regular chunk shape: a sequence of integers. + + JSON `false` counts, because `bool` is an integer type. + """ + return ( + isinstance(value, Sequence) + and not isinstance(value, str) + and all(isinstance(size, int | np.integer) for size in value) + ) + + +def _parse_stored_regular_chunk_grid( chunk_grid: dict[str, JSON] | ChunkGridMetadata | NamedConfig[str, Any], shape: tuple[int, ...], ) -> dict[str, JSON] | ChunkGridMetadata | NamedConfig[str, Any]: - """Check a stored regular grid's chunk shape against the array shape. - - zarr-python 3.0 and 3.1 stored `chunk_shape: [0]` (and 3.0 `[false]`) for - an array created with a zero-length axis; `parse_stored_chunk_shape` - decides what that means. It needs the array shape, which chunk grid - metadata does not carry, so this runs here rather than in the grid parser. + """Check a stored regular chunk grid's chunk shape against the array shape. + + Only a `regular` grid whose `chunk_shape` is all integers is a regular chunk + shape, and only that is handed to `parse_stored_regular_chunk_shape`. + Anything else is not a regular chunk shape and is left for the chunk grid + parser: other grids define their own chunk semantics. zarr-python 3.0 and + 3.1 stored `chunk_shape: [0]` (and 3.0 `[false]`) for an array created with + a zero-length axis. This runs here rather than in the grid parser because + it needs the array shape, which chunk grid metadata does not carry. """ if not isinstance(chunk_grid, Mapping) or chunk_grid.get("name") != "regular": return chunk_grid @@ -493,9 +509,11 @@ def _read_legacy_zero_chunk_sizes( if not isinstance(configuration, Mapping): return chunk_grid chunk_shape = configuration.get("chunk_shape") - if not isinstance(chunk_shape, Sequence) or isinstance(chunk_shape, str): + if not _is_regular_chunk_shape(chunk_shape): return chunk_grid - parsed = parse_stored_chunk_shape(chunk_shape, shape, legacy_writers="zarr-python 3.0 and 3.1") + parsed = parse_stored_regular_chunk_shape( + chunk_shape, shape, legacy_writers="zarr-python 3.0 and 3.1" + ) corrected: dict[str, Any] = dict(chunk_grid) corrected["configuration"] = {**configuration, "chunk_shape": list(parsed)} return corrected @@ -536,7 +554,7 @@ def __init__( shape_parsed = parse_shapelike(shape) chunk_grid_parsed = parse_chunk_grid( - _read_legacy_zero_chunk_sizes(chunk_grid, shape_parsed) + _parse_stored_regular_chunk_grid(chunk_grid, shape_parsed) ) chunk_key_encoding_parsed = parse_chunk_key_encoding(chunk_key_encoding) dimension_names_parsed = parse_dimension_names(dimension_names) diff --git a/tests/test_metadata/test_common.py b/tests/test_metadata/test_common.py index 18e6a47e1c..3ebaaac9ab 100644 --- a/tests/test_metadata/test_common.py +++ b/tests/test_metadata/test_common.py @@ -8,7 +8,7 @@ import numpy as np import pytest -from zarr.core.metadata.common import parse_stored_chunk_shape +from zarr.core.metadata.common import parse_stored_regular_chunk_shape from zarr.errors import ZarrUserWarning @@ -22,7 +22,6 @@ ((np.int64(0),), (0,), (1,), True), ((4, 0), (4, 0), (4, 1), True), ((0, 0), (0, 0), (1, 1), True), - ((2, [5, 10, 5]), (6, 20), (2, [5, 10, 5]), False), ], ids=[ "valid", @@ -32,17 +31,18 @@ "legacy-numpy-zero", "only-empty-axis-corrected", "every-empty-axis-corrected", - "edge-list-passed-through", ], ) -def test_parse_stored_chunk_shape( +def test_parse_stored_regular_chunk_shape( chunk_shape: tuple[Any, ...], shape: tuple[int, ...], expected: tuple[Any, ...], warns: bool ) -> None: """A valid chunk shape is returned as is; a chunk size of 0 on a zero-length axis is read as 1, with a warning naming the writer and how to re-save.""" with warnings.catch_warnings(record=True) as record: warnings.simplefilter("always") - parsed = parse_stored_chunk_shape(chunk_shape, shape, legacy_writers="an old writer") + parsed = parse_stored_regular_chunk_shape( + chunk_shape, shape, legacy_writers="an old writer" + ) assert parsed == expected messages = [str(w.message) for w in record if issubclass(w.category, ZarrUserWarning)] assert bool(messages) is warns @@ -51,22 +51,22 @@ def test_parse_stored_chunk_shape( assert "update_attributes({})" in message -def test_parse_stored_chunk_shape_rejects_dimension_mismatch() -> None: +def test_parse_stored_regular_chunk_shape_rejects_dimension_mismatch() -> None: """The chunk shape needs one entry per array axis.""" with pytest.raises(ValueError, match="same number of dimensions"): - parse_stored_chunk_shape((4,), (10, 10), legacy_writers="an old writer") + parse_stored_regular_chunk_shape((4,), (10, 10), legacy_writers="an old writer") @pytest.mark.parametrize(("chunk_shape", "shape"), [((0,), (5,)), ((4, 0), (4, 3))]) -def test_parse_stored_chunk_shape_rejects_zero_on_nonempty_axis( +def test_parse_stored_regular_chunk_shape_rejects_zero_on_nonempty_axis( chunk_shape: tuple[int, ...], shape: tuple[int, ...] ) -> None: """A chunk size of 0 is only tolerated on a zero-length axis.""" with pytest.raises(ValueError, match="chunk edge length must be >= 1, got 0"): - parse_stored_chunk_shape(chunk_shape, shape, legacy_writers="an old writer") + parse_stored_regular_chunk_shape(chunk_shape, shape, legacy_writers="an old writer") -def test_parse_stored_chunk_shape_rejects_negative() -> None: +def test_parse_stored_regular_chunk_shape_rejects_negative() -> None: """A negative chunk size is rejected even on a zero-length axis.""" with pytest.raises(ValueError, match="chunk edge length must be >= 1, got -1"): - parse_stored_chunk_shape((-1,), (0,), legacy_writers="an old writer") + parse_stored_regular_chunk_shape((-1,), (0,), legacy_writers="an old writer") From 8414ea3deda8d314ec85160a6203cb08f29192e9 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Fri, 25 Sep 2026 12:35:42 +0200 Subject: [PATCH 09/41] fix(metadata): read a stored zero chunk size on a grown axis as one spanning chunk A stored chunk size of 0 was tolerated only on a zero-length axis and rejected otherwise, because the metadata supposedly could not say how the stored chunks were laid out. But a chunk size of 0 gives a grid of zero chunks, so no release could store a chunk under it, and the writers of that metadata let the axis grow: 3.4.0 appends to a Zarr format 2 array created empty by 3.3.0 (shape grows, no chunk written), and 3.1.6 records a Zarr format 3 resize and the resize half of a failed append. Measured with real installs. Those arrays open in 3.4.0, attributes included; the rejection would have made them unopenable. `parse_stored_regular_chunk_shape` now reads a stored 0 (or JSON `false`) on any axis as one chunk spanning it, `max(extent, 1)`, which is what the `-1`/`False` spec that wrote it meant. On a grown axis the warning also says that data written to it was not saved. Negative sizes are still rejected. Assisted-by: ClaudeCode:claude-opus-5-5 Co-Authored-By: Claude Opus 5.5 --- changes/4334.bugfix.md | 2 +- src/zarr/core/metadata/common.py | 42 ++++++++++++++----------- src/zarr/core/metadata/v3.py | 2 +- tests/test_chunk_grids.py | 46 ++++++++++++---------------- tests/test_metadata/test_common.py | 49 ++++++++++++++++-------------- 5 files changed, 73 insertions(+), 68 deletions(-) diff --git a/changes/4334.bugfix.md b/changes/4334.bugfix.md index 610413b42e..68ae5ade9e 100644 --- a/changes/4334.bugfix.md +++ b/changes/4334.bugfix.md @@ -1,3 +1,3 @@ Chunk-grid normalization now shares a helper that selects chunk size 1 for a zero-length axis when inferring a full-span chunk size. Chunk edge lengths remain positive while array extents may be zero. `FixedDimension(size=0, ...)` now raises `ValueError`. Rectilinear chunk grids can be created on a zero-length dimension with a non-empty list of positive chunk sizes; these sizes are retained for subsequent growth. -A stored chunk size of 0 on a zero-length axis is read as 1 with a `ZarrUserWarning`, in both Zarr format 2 (`chunks`) and Zarr format 3 (a regular grid's `chunk_shape`, including the JSON `false` written by zarr-python 3.0 for `chunks=False`); a zero chunk size on a positive-length axis is rejected. zarr-python wrote such metadata for empty arrays until 3.4 — Zarr format 2 through 3.3 and by zarr-python 2.x, Zarr format 3 in 3.0 and 3.1 — and appending to one of these Zarr format 2 arrays previously reported success while writing no chunk, so the appended data read back as the fill value. The warning explains how to store a corrected chunk size: open the array writable and call `array.update_attributes({})`. +A stored chunk size of 0 is read as one chunk spanning the axis, with a `ZarrUserWarning`, in both Zarr format 2 (`chunks`) and Zarr format 3 (a regular grid's `chunk_shape`, including the JSON `false` written by zarr-python 3.0 for `chunks=False`). zarr-python wrote such metadata for arrays created with a zero-length axis until 3.4 — Zarr format 2 through 3.3 and by zarr-python 2.x, Zarr format 3 in 3.0 and 3.1. Those versions could then grow the axis without storing any chunk: appending to one of these Zarr format 2 arrays in 3.4.0 reported success while the appended data read back as the fill value, and 3.1 recorded a Zarr format 3 resize or failed append. Such grown arrays open too, and the warning says that data written to the grown axis was not saved. A negative chunk size is rejected. The warning explains how to store a corrected chunk size: open the array writable and call `array.update_attributes({})`. diff --git a/src/zarr/core/metadata/common.py b/src/zarr/core/metadata/common.py index 6e6bab48a5..14c6b8dd66 100644 --- a/src/zarr/core/metadata/common.py +++ b/src/zarr/core/metadata/common.py @@ -39,12 +39,15 @@ def parse_stored_regular_chunk_shape( length 0, so its chunk sizes must not be passed here. The chunk shape must have one entry per array axis, and every chunk size - must be at least 1. The exception is a chunk size of 0 on a zero-length - axis, which `legacy_writers` stored for arrays created empty. An empty axis - has no chunk for the size to describe, so it is read as 1 with a - `ZarrUserWarning` that says how to re-save valid metadata. A chunk size of - 0 on a positive-length axis is rejected, because the metadata cannot say - how the stored chunks were laid out. JSON `false` counts as 0. + must be at least 1, with one exception. `legacy_writers` stored a chunk + size of 0 (or JSON `false`) for an array created with a zero-length axis + and a chunk spec meaning "one chunk spanning the axis". That is how the + size is read, `max(extent, 1)`, with a `ZarrUserWarning` that says how to + re-save valid metadata. The axis may have grown since: those writers let + the array be resized or appended to, but a chunk size of 0 gives a grid of + zero chunks, so no chunk was ever stored for it and any chunk size reads + the store correctly. The warning then says that the appended data was not + saved. A negative chunk size is rejected. """ if len(chunk_shape) != len(shape): raise ValueError( @@ -53,17 +56,22 @@ def parse_stored_regular_chunk_shape( ) parsed: list[int] = [] for dim_idx, (size, extent) in enumerate(zip(chunk_shape, shape, strict=True)): - if size < 1: - if size < 0 or extent != 0: - raise ValueError( - f"Dimension {dim_idx}: chunk edge length must be >= 1, got {size!r}" - ) - warnings.warn( - f"Dimension {dim_idx}: chunk edge length {size!r} on a zero-length axis " - f"(as written by {legacy_writers}) is treated as 1. {RESAVE_METADATA_HINT}", - ZarrUserWarning, - stacklevel=3, + if size < 0: + raise ValueError(f"Dimension {dim_idx}: chunk edge length must be >= 1, got {size!r}") + if size == 0: + corrected = max(extent, 1) + msg = ( + f"Dimension {dim_idx}: chunk edge length {size!r} (as written by " + f"{legacy_writers} for an array created with a zero-length axis) is read " + f"as one chunk spanning the axis, of size {corrected}." ) - size = 1 + if extent > 0: + msg += ( + f" The axis has since grown to {extent}, but no chunk can be stored " + "under a chunk size of 0, so data written to it before now was not " + "saved and reads as the fill value." + ) + warnings.warn(f"{msg} {RESAVE_METADATA_HINT}", ZarrUserWarning, stacklevel=3) + size = corrected parsed.append(size) return tuple(parsed) diff --git a/src/zarr/core/metadata/v3.py b/src/zarr/core/metadata/v3.py index c5a70e2be4..ef7a5bf0e2 100644 --- a/src/zarr/core/metadata/v3.py +++ b/src/zarr/core/metadata/v3.py @@ -500,7 +500,7 @@ def _parse_stored_regular_chunk_grid( Anything else is not a regular chunk shape and is left for the chunk grid parser: other grids define their own chunk semantics. zarr-python 3.0 and 3.1 stored `chunk_shape: [0]` (and 3.0 `[false]`) for an array created with - a zero-length axis. This runs here rather than in the grid parser because + a zero-length axis, and 3.1 kept it when the axis grew. This runs here rather than in the grid parser because it needs the array shape, which chunk grid metadata does not carry. """ if not isinstance(chunk_grid, Mapping) or chunk_grid.get("name") != "regular": diff --git a/tests/test_chunk_grids.py b/tests/test_chunk_grids.py index 4ba5d19f26..cfd2e42258 100644 --- a/tests/test_chunk_grids.py +++ b/tests/test_chunk_grids.py @@ -558,46 +558,40 @@ def _store_legacy_zero_chunk(path: Any, zarr_format: Literal[2, 3], stored: Any) @pytest.mark.parametrize("zarr_format", [2, 3]) @pytest.mark.parametrize("stored", [0, False], ids=["zero", "false"]) -def test_legacy_zero_chunk_on_empty_axis_round_trip( - tmp_path: Path, zarr_format: Literal[2, 3], stored: Any +@pytest.mark.parametrize("extent", [0, 3], ids=["empty-axis", "grown-axis"]) +def test_legacy_zero_chunk_round_trip( + tmp_path: Path, zarr_format: Literal[2, 3], stored: Any, extent: int ) -> None: - """An empty array whose stored chunk size is 0 opens, appends without losing - data, and re-saves as a valid chunk size. - - zarr-python wrote such metadata for an empty array until 3.4: Zarr format 2 - through 3.3, and Zarr format 3 in 3.0 and 3.1, which also wrote JSON `false` - for `chunks=False`. Appending to one of these arrays used to report success - while writing no chunk, so the data read back as the fill value. + """An array whose stored chunk size is 0 opens with one chunk spanning the + axis, appends without losing data, and re-saves as a valid chunk size. + + zarr-python wrote such metadata for an array created with a zero-length axis + until 3.4: Zarr format 2 through 3.3, and Zarr format 3 in 3.0 and 3.1, which + also wrote JSON `false` for `chunks=False`. Those versions could then grow + the axis (a 3.4.0 append, a 3.1 resize) while storing no chunk, so the + grown axis holds only the fill value. """ path = tmp_path / "legacy.zarr" - zarr.create_array(store=path, shape=(0,), chunks=(4,), dtype="int64", zarr_format=zarr_format) + zarr.create_array( + store=path, shape=(extent,), chunks=(4,), dtype="int64", zarr_format=zarr_format + ) _store_legacy_zero_chunk(path, zarr_format, stored) - with pytest.warns(ZarrUserWarning, match="zero-length axis"): + with pytest.warns(ZarrUserWarning, match="one chunk spanning the axis"): arr = zarr.open_array(store=path, mode="a") - assert arr.chunks == (1,) + assert arr.chunks == (max(extent, 1),) + np.testing.assert_array_equal(arr[...], np.zeros(extent, dtype="int64")) arr.append(np.arange(3, dtype="int64")) - np.testing.assert_array_equal(zarr.open_array(store=path)[...], np.arange(3)) + expected = np.concatenate([np.zeros(extent, dtype="int64"), np.arange(3)]) + np.testing.assert_array_equal(zarr.open_array(store=path)[...], expected) # The warning says to do this; it must leave metadata that reopens cleanly. arr.update_attributes({}) with warnings.catch_warnings(): warnings.simplefilter("error", ZarrUserWarning) reopened = zarr.open_array(store=path) - assert reopened.chunks == (1,) - - -@pytest.mark.parametrize("zarr_format", [2, 3]) -def test_stored_zero_chunk_on_positive_axis_rejected( - tmp_path: Path, zarr_format: Literal[2, 3] -) -> None: - """A stored chunk size of 0 is only tolerated on a zero-length axis.""" - path = tmp_path / "bad.zarr" - zarr.create_array(store=path, shape=(5,), chunks=(4,), dtype="int64", zarr_format=zarr_format) - _store_legacy_zero_chunk(path, zarr_format, 0) - with pytest.raises(ValueError, match="must be >= 1"): - zarr.open_array(store=path) + assert reopened.chunks == (max(extent, 1),) def test_normalize_chunks_1d_zero_span_accepts_any_edges() -> None: diff --git a/tests/test_metadata/test_common.py b/tests/test_metadata/test_common.py index 3ebaaac9ab..f35f2fa954 100644 --- a/tests/test_metadata/test_common.py +++ b/tests/test_metadata/test_common.py @@ -2,6 +2,7 @@ from __future__ import annotations +import re import warnings from typing import Any @@ -13,15 +14,17 @@ @pytest.mark.parametrize( - ("chunk_shape", "shape", "expected", "warns"), + ("chunk_shape", "shape", "expected", "warning"), [ - ((4, 5), (10, 10), (4, 5), False), - ((1, 1), (0, 0), (1, 1), False), - ((0,), (0,), (1,), True), - ((False,), (0,), (1,), True), - ((np.int64(0),), (0,), (1,), True), - ((4, 0), (4, 0), (4, 1), True), - ((0, 0), (0, 0), (1, 1), True), + ((4, 5), (10, 10), (4, 5), None), + ((1, 1), (0, 0), (1, 1), None), + ((0,), (0,), (1,), "of size 1"), + ((False,), (0,), (1,), "of size 1"), + ((np.int64(0),), (0,), (1,), "of size 1"), + ((4, 0), (4, 0), (4, 1), "Dimension 1"), + ((0, 0), (0, 0), (1, 1), "Dimension 0"), + ((0,), (5,), (5,), "grown to 5.*was not saved"), + ((4, 0), (4, 3), (4, 3), "grown to 3.*was not saved"), ], ids=[ "valid", @@ -29,15 +32,21 @@ "legacy-zero", "legacy-json-false", "legacy-numpy-zero", - "only-empty-axis-corrected", - "every-empty-axis-corrected", + "only-zero-size-corrected", + "every-zero-size-corrected", + "legacy-zero-on-grown-axis", + "legacy-zero-on-grown-axis-2d", ], ) def test_parse_stored_regular_chunk_shape( - chunk_shape: tuple[Any, ...], shape: tuple[int, ...], expected: tuple[Any, ...], warns: bool + chunk_shape: tuple[Any, ...], + shape: tuple[int, ...], + expected: tuple[Any, ...], + warning: str | None, ) -> None: - """A valid chunk shape is returned as is; a chunk size of 0 on a zero-length - axis is read as 1, with a warning naming the writer and how to re-save.""" + """A valid chunk shape is returned as is; a chunk size of 0 is read as one + chunk spanning the axis, with a warning naming the writer and how to re-save, + and, if the axis has grown, that data written to it was not saved.""" with warnings.catch_warnings(record=True) as record: warnings.simplefilter("always") parsed = parse_stored_regular_chunk_shape( @@ -45,7 +54,10 @@ def test_parse_stored_regular_chunk_shape( ) assert parsed == expected messages = [str(w.message) for w in record if issubclass(w.category, ZarrUserWarning)] - assert bool(messages) is warns + if warning is None: + assert messages == [] + else: + assert any(re.search(warning, message) for message in messages) for message in messages: assert "an old writer" in message assert "update_attributes({})" in message @@ -57,15 +69,6 @@ def test_parse_stored_regular_chunk_shape_rejects_dimension_mismatch() -> None: parse_stored_regular_chunk_shape((4,), (10, 10), legacy_writers="an old writer") -@pytest.mark.parametrize(("chunk_shape", "shape"), [((0,), (5,)), ((4, 0), (4, 3))]) -def test_parse_stored_regular_chunk_shape_rejects_zero_on_nonempty_axis( - chunk_shape: tuple[int, ...], shape: tuple[int, ...] -) -> None: - """A chunk size of 0 is only tolerated on a zero-length axis.""" - with pytest.raises(ValueError, match="chunk edge length must be >= 1, got 0"): - parse_stored_regular_chunk_shape(chunk_shape, shape, legacy_writers="an old writer") - - def test_parse_stored_regular_chunk_shape_rejects_negative() -> None: """A negative chunk size is rejected even on a zero-length axis.""" with pytest.raises(ValueError, match="chunk edge length must be >= 1, got -1"): From 2bf32df2294270de30ab6f8c85506c0ed8d589ce Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Fri, 25 Sep 2026 12:43:38 +0200 Subject: [PATCH 10/41] test: a stateful test of one array's create/append/resize/write life The only state machine that touched arrays compared zarr on one store with zarr on a MemoryStore, so a chunk grid bug showed up identically on both sides; it had no append rule, covered Zarr format 3 only, and kept every empty axis at 0 when resizing, which is where the zero-length bugs live. `ArrayLifecycle` checks one array against a NumPy model across both formats, every chunk spelling (-1, False, "auto", ints, sharded, rectilinear) and the stored chunk size of 0 that releases before 3.4 wrote, including on an axis those releases grew. Rules append, resize (growing and shrinking to and from 0), write and re-save the metadata; the invariant reopens the array and compares shape, values and whether the legacy warning is due. Deliberately breaking the grown-axis policy, the legacy warning, or append on an empty axis each fails it. `resize` keeps partly retained chunks whole, so cells cut off by a shrink can come back with their old values when the axis grows (as in 2.x); the model marks such cells unknown until written instead of encoding that. Assisted-by: ClaudeCode:claude-opus-5-5 Co-Authored-By: Claude Opus 5.5 --- tests/test_array_stateful.py | 252 +++++++++++++++++++++++++++++++++++ 1 file changed, 252 insertions(+) create mode 100644 tests/test_array_stateful.py diff --git a/tests/test_array_stateful.py b/tests/test_array_stateful.py new file mode 100644 index 0000000000..75a6b12d59 --- /dev/null +++ b/tests/test_array_stateful.py @@ -0,0 +1,252 @@ +"""A stateful test of one array's life: create, append, resize, write, reopen. + +The model is a NumPy array, not a second zarr array, so a bug in zarr's chunk +grid logic cannot hide by being made on both sides. Zero-length axes are drawn +on purpose, both at creation and by resizing and appending, and so are the +stored chunk sizes of 0 that zarr-python wrote for empty arrays before 3.4. + +`resize` deletes only the chunks that fall entirely outside the new shape, so +cells cut off by a shrink can come back with their old values when the axis +grows again (as in zarr-python 2.x). The model does not encode that chunk-level +behaviour: a cell cut off and brought back is unknown until it is written. +""" + +from __future__ import annotations + +import json +import warnings +from typing import Any, Literal + +import hypothesis.extra.numpy as npst +import hypothesis.strategies as st +import numpy as np +import pytest +from hypothesis import event, note, settings +from hypothesis.stateful import ( + RuleBasedStateMachine, + initialize, + invariant, + rule, +) + +import zarr +from zarr.core.buffer import cpu, default_buffer_prototype +from zarr.core.sync import sync +from zarr.errors import ZarrUserWarning +from zarr.storage import MemoryStore + +pytestmark = pytest.mark.filterwarnings( + "ignore::zarr.core.dtype.common.UnstableSpecificationWarning" +) + +DTYPE = np.dtype("int16") +MAX_SIDE = 6 + + +def _rectilinear_dim(extent: int) -> st.SearchStrategy[int | list[int]]: + """A bare step, or an edge list covering `extent` (any edges for extent 0).""" + steps = st.integers(min_value=1, max_value=MAX_SIDE) + if extent == 0: + return steps | st.lists(steps, min_size=1, max_size=3) + if extent == 1: + return steps | st.just([1]) + cuts = st.lists(st.integers(min_value=1, max_value=extent - 1), unique=True, max_size=3) + edges = cuts.map( + lambda c: [b - a for a, b in zip([0, *sorted(c)], [*sorted(c), extent], strict=True)] + ) + return steps | edges + + +class ArrayLifecycle(RuleBasedStateMachine): + def __init__(self) -> None: + super().__init__() + self._rectilinear = zarr.config.set({"array.rectilinear_chunks": True}) + self._rectilinear.__enter__() + self.store = MemoryStore() + self.path = "a" + self.model: np.ndarray[Any, np.dtype[np.int16]] = np.zeros((0,), dtype=DTYPE) + # Cells whose value the model knows; see the module docstring. + self.known: np.ndarray[Any, np.dtype[np.bool_]] = np.ones((0,), dtype=bool) + # Every shape the array has had, to find cells a resize brings back. + self.past_shapes: list[tuple[int, ...]] = [] + self.fill = 0 + # A legacy store warns until its metadata is re-saved. + self.expect_open_warning = False + + # -------------------------------------------------------------- creation + @initialize(data=st.data()) + def create(self, data: st.DataObject) -> None: + zarr_format: Literal[2, 3] = data.draw(st.sampled_from([2, 3]), label="zarr_format") + shape = data.draw( + npst.array_shapes(min_dims=1, max_dims=3, min_side=0, max_side=MAX_SIDE), + label="shape", + ) + self.fill = data.draw(st.integers(-3, 3), label="fill_value") + # sampled_from favours early entries; the less common spellings go first. + spellings = ["ints", "legacy-zero", "-1", "False", "auto"] + if zarr_format == 3: + spellings.insert(1, "rectilinear") + spelling = data.draw(st.sampled_from(spellings), label="chunk spelling") + event(f"chunks: {spelling}") + if any(s == 0 for s in shape): + event("created with a zero-length axis") + + chunks: Any + shards: Any = None + if spelling == "-1": + chunks = -1 + elif spelling == "False": + chunks = False + elif spelling == "auto": + chunks = "auto" + elif spelling == "rectilinear": + chunks = [data.draw(_rectilinear_dim(s)) for s in shape] + if not any(isinstance(c, list) for c in chunks): + chunks[0] = [chunks[0]] if shape[0] == 0 else [shape[0]] + else: + chunks = tuple(data.draw(st.integers(1, 4)) for _ in shape) + if ( + spelling == "ints" + and zarr_format == 3 + and data.draw(st.booleans(), label="sharded") + ): + shards = tuple(c * data.draw(st.integers(1, 2)) for c in chunks) + event("sharded") + note(f"create {shape=} {chunks=} {shards=} {zarr_format=} fill={self.fill}") + zarr.create_array( + self.store, + name=self.path, + shape=shape, + chunks=chunks, + shards=shards, + dtype=DTYPE, + fill_value=self.fill, + zarr_format=zarr_format, + ) + self.model = np.full(shape, self.fill, dtype=DTYPE) + self.known = np.ones(shape, dtype=bool) + self.past_shapes = [shape] + + if spelling == "legacy-zero": + # What zarr-python wrote before 3.4 for an array created with a + # zero-length axis and one chunk spanning it; older releases could + # grow that axis without storing a chunk, so any extent is possible. + zero_axes = data.draw( + st.lists(st.integers(0, len(shape) - 1), min_size=1, unique=True), + label="axes stored with chunk size 0", + ) + stored_zero = data.draw(st.sampled_from([0, False]), label="stored zero") + self._rewrite_stored_chunks(zarr_format, zero_axes, stored_zero) + self.expect_open_warning = True + if any(shape[i] > 0 for i in zero_axes): + event("legacy zero chunk on a grown axis") + + def _rewrite_stored_chunks( + self, zarr_format: Literal[2, 3], axes: list[int], value: Any + ) -> None: + key = f"{self.path}/{'.zarray' if zarr_format == 2 else 'zarr.json'}" + buf = sync(self.store.get(key, prototype=default_buffer_prototype())) + assert buf is not None + doc = json.loads(buf.to_bytes()) + sizes = ( + doc["chunks"] if zarr_format == 2 else doc["chunk_grid"]["configuration"]["chunk_shape"] + ) + for axis in axes: + sizes[axis] = value + sync(self.store.set(key, cpu.Buffer.from_bytes(json.dumps(doc).encode()))) + + def _open(self) -> zarr.Array[Any]: + with warnings.catch_warnings(record=True) as record: + warnings.simplefilter("always", ZarrUserWarning) + arr = zarr.open_array(self.store, path=self.path, mode="r+") + warned = any(issubclass(w.category, ZarrUserWarning) for w in record) + assert warned is self.expect_open_warning, [str(w.message) for w in record] + return arr + + # ----------------------------------------------------------------- rules + @rule(data=st.data()) + def append(self, data: st.DataObject) -> None: + arr = self._open() + axis = data.draw(st.integers(0, self.model.ndim - 1), label="axis") + block_shape = list(self.model.shape) + block_shape[axis] = data.draw(st.integers(0, 4), label="rows") + block = data.draw(npst.arrays(DTYPE, tuple(block_shape)), label="block") + note(f"append {block.shape} along {axis} to {self.model.shape}") + if self.model.shape[axis] == 0: + event("append to a zero-length axis") + arr.append(block, axis=axis) + self._reshape_model(arr.shape) + tail = tuple( + slice(-block.shape[axis], None) if i == axis and block.shape[axis] else slice(None) + for i in range(self.model.ndim) + ) + if block.shape[axis]: + self.model[tail] = block + self.known[tail] = True + # Growing the array rewrites its metadata, which stores any correction. + self.expect_open_warning = False + + @rule(data=st.data()) + def resize(self, data: st.DataObject) -> None: + arr = self._open() + new_shape = data.draw( + st.tuples(*(st.integers(0, MAX_SIDE) for _ in self.model.shape)), label="new shape" + ) + note(f"resize {self.model.shape} -> {new_shape}") + if any(o == 0 and n > 0 for o, n in zip(self.model.shape, new_shape, strict=True)): + event("resize grows a zero-length axis") + arr.resize(new_shape) + self._reshape_model(new_shape) + self.expect_open_warning = False + + def _reshape_model(self, new_shape: tuple[int, ...]) -> None: + """Resize the model: kept cells keep their values, new cells hold the fill + value, and cells that an earlier shape held but the current one cut off + become unknown.""" + overlap = tuple( + slice(0, min(o, n)) for o, n in zip(self.model.shape, new_shape, strict=True) + ) + model = np.full(new_shape, self.fill, dtype=DTYPE) + known = np.ones(new_shape, dtype=bool) + for past in self.past_shapes: + known[tuple(slice(0, min(p, n)) for p, n in zip(past, new_shape, strict=True))] = False + model[overlap] = self.model[overlap] + known[overlap] = self.known[overlap] + if not known.all(): + event("resize brings back cells cut off earlier") + self.model, self.known = model, known + self.past_shapes.append(tuple(new_shape)) + + @rule(data=st.data()) + def write(self, data: st.DataObject) -> None: + arr = self._open() + region = tuple( + slice(*sorted(data.draw(st.tuples(st.integers(0, s), st.integers(0, s))))) + for s in self.model.shape + ) + values = data.draw(npst.arrays(DTYPE, self.model[region].shape), label="values") + note(f"write {region}") + arr[region] = values + self.model[region] = values + self.known[region] = True + + @rule() + def resave_metadata(self) -> None: + """What the legacy warning tells users to do.""" + self._open().update_attributes({}) + self.expect_open_warning = False + + def teardown(self) -> None: + self._rectilinear.__exit__(None, None, None) + + # ------------------------------------------------------------ invariants + @invariant() + def matches_model(self) -> None: + arr = self._open() + assert arr.shape == self.model.shape + actual = np.asarray(arr[...]) + np.testing.assert_array_equal(actual[self.known], self.model[self.known]) + + +ArrayLifecycle.TestCase.settings = settings(max_examples=200, stateful_step_count=12, deadline=None) +TestArrayLifecycle = ArrayLifecycle.TestCase From 1723f0b43f0b14b44aa76414dd7ce3111564a444 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Fri, 25 Sep 2026 15:26:25 +0200 Subject: [PATCH 11/41] fix(metadata): stop the zero chunk size warning from naming writers The warning for a stored chunk size of 0 said which zarr-python releases wrote it. The check runs on every metadata construction, so metadata built in code, such as VirtualiZarr's kerchunk writer passing an empty array's shape as `chunks`, was told it came from zarr-python 2.x. The warning now says what the chunk size is read as and, on an axis of positive length, that the axis holds only the fill value. `legacy_writers` is gone, and the docstrings no longer narrate release history; the changelog keeps it. Assisted-by: ClaudeCode:claude-opus-5-5 Co-Authored-By: Claude Opus 5.5 --- src/zarr/core/metadata/common.py | 26 +++++++++++--------------- src/zarr/core/metadata/v2.py | 6 +----- src/zarr/core/metadata/v3.py | 11 ++++------- tests/test_array_stateful.py | 4 ++-- tests/test_metadata/test_common.py | 17 +++++++---------- 5 files changed, 25 insertions(+), 39 deletions(-) diff --git a/src/zarr/core/metadata/common.py b/src/zarr/core/metadata/common.py index 14c6b8dd66..00f1e96254 100644 --- a/src/zarr/core/metadata/common.py +++ b/src/zarr/core/metadata/common.py @@ -29,7 +29,7 @@ def parse_attributes(data: dict[str, JSON] | None) -> dict[str, JSON]: def parse_stored_regular_chunk_shape( - chunk_shape: Sequence[int], shape: Sequence[int], *, legacy_writers: str + chunk_shape: Sequence[int], shape: Sequence[int] ) -> tuple[int, ...]: """Validate a stored regular chunk grid's chunk shape against the array shape. @@ -39,15 +39,13 @@ def parse_stored_regular_chunk_shape( length 0, so its chunk sizes must not be passed here. The chunk shape must have one entry per array axis, and every chunk size - must be at least 1, with one exception. `legacy_writers` stored a chunk - size of 0 (or JSON `false`) for an array created with a zero-length axis - and a chunk spec meaning "one chunk spanning the axis". That is how the - size is read, `max(extent, 1)`, with a `ZarrUserWarning` that says how to - re-save valid metadata. The axis may have grown since: those writers let - the array be resized or appended to, but a chunk size of 0 gives a grid of - zero chunks, so no chunk was ever stored for it and any chunk size reads - the store correctly. The warning then says that the appended data was not - saved. A negative chunk size is rejected. + must be at least 1, with one exception: a chunk size of 0 (or JSON + `false`) is read as one chunk spanning its axis, `max(extent, 1)`, with a + `ZarrUserWarning` that says how to re-save valid metadata. A chunk size of + 0 gives a grid of zero chunks, so no chunk can be stored under it and any + positive chunk size reads the store correctly; if the axis has positive + length, the warning also says that it holds only the fill value. A + negative chunk size is rejected. """ if len(chunk_shape) != len(shape): raise ValueError( @@ -61,15 +59,13 @@ def parse_stored_regular_chunk_shape( if size == 0: corrected = max(extent, 1) msg = ( - f"Dimension {dim_idx}: chunk edge length {size!r} (as written by " - f"{legacy_writers} for an array created with a zero-length axis) is read " + f"Dimension {dim_idx}: chunk edge length {size!r} is invalid and is read " f"as one chunk spanning the axis, of size {corrected}." ) if extent > 0: msg += ( - f" The axis has since grown to {extent}, but no chunk can be stored " - "under a chunk size of 0, so data written to it before now was not " - "saved and reads as the fill value." + f" No chunk can be stored under a chunk size of 0, so the {extent} " + "elements along this axis hold only the fill value." ) warnings.warn(f"{msg} {RESAVE_METADATA_HINT}", ZarrUserWarning, stacklevel=3) size = corrected diff --git a/src/zarr/core/metadata/v2.py b/src/zarr/core/metadata/v2.py index c213bc98b6..08c290fab2 100644 --- a/src/zarr/core/metadata/v2.py +++ b/src/zarr/core/metadata/v2.py @@ -88,11 +88,7 @@ def __init__( Metadata for a Zarr format 2 array. """ shape_parsed = parse_shapelike(shape) - chunks_parsed = parse_stored_regular_chunk_shape( - parse_shapelike(chunks), - shape_parsed, - legacy_writers="zarr-python 2.x, and by 3.x before 3.4", - ) + chunks_parsed = parse_stored_regular_chunk_shape(parse_shapelike(chunks), shape_parsed) compressor_parsed = parse_compressor(compressor) order_parsed = parse_indexing_order(order) dimension_separator_parsed = parse_separator(dimension_separator) diff --git a/src/zarr/core/metadata/v3.py b/src/zarr/core/metadata/v3.py index ef7a5bf0e2..1db659beb8 100644 --- a/src/zarr/core/metadata/v3.py +++ b/src/zarr/core/metadata/v3.py @@ -498,10 +498,9 @@ def _parse_stored_regular_chunk_grid( Only a `regular` grid whose `chunk_shape` is all integers is a regular chunk shape, and only that is handed to `parse_stored_regular_chunk_shape`. Anything else is not a regular chunk shape and is left for the chunk grid - parser: other grids define their own chunk semantics. zarr-python 3.0 and - 3.1 stored `chunk_shape: [0]` (and 3.0 `[false]`) for an array created with - a zero-length axis, and 3.1 kept it when the axis grew. This runs here rather than in the grid parser because - it needs the array shape, which chunk grid metadata does not carry. + parser: other grids define their own chunk semantics. This runs here rather + than in the grid parser because it needs the array shape, which chunk grid + metadata does not carry. """ if not isinstance(chunk_grid, Mapping) or chunk_grid.get("name") != "regular": return chunk_grid @@ -511,9 +510,7 @@ def _parse_stored_regular_chunk_grid( chunk_shape = configuration.get("chunk_shape") if not _is_regular_chunk_shape(chunk_shape): return chunk_grid - parsed = parse_stored_regular_chunk_shape( - chunk_shape, shape, legacy_writers="zarr-python 3.0 and 3.1" - ) + parsed = parse_stored_regular_chunk_shape(chunk_shape, shape) corrected: dict[str, Any] = dict(chunk_grid) corrected["configuration"] = {**configuration, "chunk_shape": list(parsed)} return corrected diff --git a/tests/test_array_stateful.py b/tests/test_array_stateful.py index 75a6b12d59..a5a6627a5d 100644 --- a/tests/test_array_stateful.py +++ b/tests/test_array_stateful.py @@ -7,8 +7,8 @@ `resize` deletes only the chunks that fall entirely outside the new shape, so cells cut off by a shrink can come back with their old values when the axis -grows again (as in zarr-python 2.x). The model does not encode that chunk-level -behaviour: a cell cut off and brought back is unknown until it is written. +grows again. The model does not encode that chunk-level behaviour: a cell cut +off and brought back is unknown until it is written. """ from __future__ import annotations diff --git a/tests/test_metadata/test_common.py b/tests/test_metadata/test_common.py index f35f2fa954..c3ed9416eb 100644 --- a/tests/test_metadata/test_common.py +++ b/tests/test_metadata/test_common.py @@ -23,8 +23,8 @@ ((np.int64(0),), (0,), (1,), "of size 1"), ((4, 0), (4, 0), (4, 1), "Dimension 1"), ((0, 0), (0, 0), (1, 1), "Dimension 0"), - ((0,), (5,), (5,), "grown to 5.*was not saved"), - ((4, 0), (4, 3), (4, 3), "grown to 3.*was not saved"), + ((0,), (5,), (5,), "5 elements along this axis hold only the fill value"), + ((4, 0), (4, 3), (4, 3), "3 elements along this axis hold only the fill value"), ], ids=[ "valid", @@ -45,13 +45,11 @@ def test_parse_stored_regular_chunk_shape( warning: str | None, ) -> None: """A valid chunk shape is returned as is; a chunk size of 0 is read as one - chunk spanning the axis, with a warning naming the writer and how to re-save, - and, if the axis has grown, that data written to it was not saved.""" + chunk spanning the axis, with a warning saying how to re-save and, on an axis + of positive length, that it holds only the fill value.""" with warnings.catch_warnings(record=True) as record: warnings.simplefilter("always") - parsed = parse_stored_regular_chunk_shape( - chunk_shape, shape, legacy_writers="an old writer" - ) + parsed = parse_stored_regular_chunk_shape(chunk_shape, shape) assert parsed == expected messages = [str(w.message) for w in record if issubclass(w.category, ZarrUserWarning)] if warning is None: @@ -59,17 +57,16 @@ def test_parse_stored_regular_chunk_shape( else: assert any(re.search(warning, message) for message in messages) for message in messages: - assert "an old writer" in message assert "update_attributes({})" in message def test_parse_stored_regular_chunk_shape_rejects_dimension_mismatch() -> None: """The chunk shape needs one entry per array axis.""" with pytest.raises(ValueError, match="same number of dimensions"): - parse_stored_regular_chunk_shape((4,), (10, 10), legacy_writers="an old writer") + parse_stored_regular_chunk_shape((4,), (10, 10)) def test_parse_stored_regular_chunk_shape_rejects_negative() -> None: """A negative chunk size is rejected even on a zero-length axis.""" with pytest.raises(ValueError, match="chunk edge length must be >= 1, got -1"): - parse_stored_regular_chunk_shape((-1,), (0,), legacy_writers="an old writer") + parse_stored_regular_chunk_shape((-1,), (0,)) From 3d223f18f4035ae25eb2d88d2d352ec3dbb3fafb Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Fri, 25 Sep 2026 20:56:13 +0200 Subject: [PATCH 12/41] chore(metadata): drop an orphaned comment in ArrayV2Metadata.__init__ The comment introduced a consistency check (`parse_metadata`) that this branch removed; the check now lives in `parse_stored_regular_chunk_shape`. Assisted-by: ClaudeCode:claude-opus-5-5 Co-Authored-By: Claude Opus 5.5 --- src/zarr/core/metadata/v2.py | 2 -- 1 file changed, 2 deletions(-) diff --git a/src/zarr/core/metadata/v2.py b/src/zarr/core/metadata/v2.py index 08c290fab2..3ae2a8bd96 100644 --- a/src/zarr/core/metadata/v2.py +++ b/src/zarr/core/metadata/v2.py @@ -110,8 +110,6 @@ def __init__( object.__setattr__(self, "fill_value", fill_value_parsed) object.__setattr__(self, "attributes", attributes_parsed) - # ensure that the metadata document is consistent - @property def ndim(self) -> int: return len(self.shape) From 339c514cac483c6555619fd2741a772d95965982 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Fri, 25 Sep 2026 21:54:57 +0200 Subject: [PATCH 13/41] fix(chunk-grids): size a full-span shard as a multiple of the inner chunk One definition of "one chunk spanning the axis", full_span_chunk_size(span, unit), is the smallest positive multiple of unit covering span. shards=-1 and shards=False now use the inner chunk size as unit, so a zero-length axis, or one whose length is not a multiple of the inner chunk, gets a valid shard instead of a divisibility error. The zero-length array test drops the explicit-size spellings and keeps the spellings that derive a chunk from the span. Assisted-by: ClaudeCode:claude-opus-5-5 Co-Authored-By: Claude Opus 5.5 --- src/zarr/core/chunk_grids.py | 66 +++++----- tests/test_chunk_grids.py | 247 ++++++++++------------------------- 2 files changed, 96 insertions(+), 217 deletions(-) diff --git a/src/zarr/core/chunk_grids.py b/src/zarr/core/chunk_grids.py index 545aa3c581..d5d61ee010 100644 --- a/src/zarr/core/chunk_grids.py +++ b/src/zarr/core/chunk_grids.py @@ -47,12 +47,7 @@ @dataclass(frozen=True) class FixedDimension: """Uniform chunk size. Boundary chunks contain less data but are - encoded at full size by the codec pipeline. - - The chunk edge length is always at least 1, matching the invariant the - metadata layer enforces for every stored chunk grid. The extent may be 0: - a zero-length axis simply has zero chunks (``ceildiv(0, size) == 0``). - """ + encoded at full size by the codec pipeline.""" size: int # chunk edge length (>= 1) extent: int # array dimension length (>= 0) @@ -656,18 +651,15 @@ class ChunkLayout(NamedTuple): inner: ChunkLayout | None = None -def _full_span_chunk_size(span: int) -> int: - """The edge length of one chunk covering an entire axis of length *span*. +def full_span_chunk_size(span: int, unit: int = 1) -> int: + """The edge length of one chunk spanning an axis of length `span`. - This is *the* definition of "one chunk spans the axis" for a possibly - zero-length axis. Chunk edge lengths must be at least 1 (the invariant - shared by `FixedDimension`, `VaryingDimension` and the stored chunk grid - metadata), so a zero-length axis gets chunk size 1 and zero chunks. Every - spelling that derives a chunk size from a span — ``chunks=-1``, - ``chunks=False``, ``chunks="auto"``, ``shards="auto"`` — must route - through this helper rather than clamping on its own. + This is the smallest positive multiple of `unit` that covers `span`, so a + zero-length axis gets a chunk of size `unit` and zero chunks. `unit` is the + size the chunk must be a multiple of: the inner chunk size for a shard, 1 + otherwise. """ - return max(span, 1) + return unit * max(1, ceildiv(span, unit)) def _guess_regular_chunks( @@ -707,10 +699,11 @@ def _guess_regular_chunks( shape = (shape,) if typesize == 0: - return tuple(_full_span_chunk_size(s) for s in shape) + return tuple(full_span_chunk_size(s) for s in shape) ndims = len(shape) - chunks = np.array([_full_span_chunk_size(s) for s in shape], dtype="=f8") + # require chunks to have non-zero length for all dimensions + chunks = np.maximum(np.array(shape, dtype="=f8"), 1) # Determine the optimal chunk size in bytes using a PyTables expression. # This is kept as a float. @@ -745,7 +738,7 @@ def _guess_regular_chunks( return tuple(int(x) for x in chunks) -def normalize_chunks_1d(chunks: int | Iterable[object], span: int) -> DimensionGrid: +def normalize_chunks_1d(chunks: int | Iterable[object], span: int, unit: int = 1) -> DimensionGrid: """ Normalize a one-dimensional chunk specification into a dimension grid: `FixedDimension` for scalar chunk sizes, `VaryingDimension` for explicit @@ -753,21 +746,15 @@ def normalize_chunks_1d(chunks: int | Iterable[object], span: int) -> DimensionG the span, and the uniform form is O(1) in the number of chunks — a dimension with `2**62` chunks must not materialize one entry per chunk. - `-1` means "one chunk covering the entire span" (see `_full_span_chunk_size` - for what that means on a zero-length span). + `-1` means "one chunk covering the entire span", sized by + `full_span_chunk_size(span, unit)`. Explicit chunk size lists must sum to the span exactly and always produce `VaryingDimension`, even when the sizes happen to be uniform: the input syntax declares the grid kind, so a per-chunk list is preserved as a rectilinear dimension rather than silently collapsed to a regular one, which would change how the dimension grows on resize. For scalar sizes - the last chunk may overhang the span. - - The one exception to the sum rule is a zero-length span: no non-empty list of - positive edges can sum to 0, so a non-empty list of positive integers is retained - and the edges describe the chunks the axis will grow into on `append` / - `resize`. This is the same state a rectilinear axis reaches when it is - resized down to 0 — `VaryingDimension` allows trailing edges beyond the - extent — so creating at length 0 and shrinking to 0 are indistinguishable. + the last chunk may overhang the span. On a zero-length span any non-empty + list of positive sizes is kept: the chunks the axis grows into. """ # `numbers.Integral` rather than `int` so that numpy integer scalars (which are not # `int` subclasses) take the uniform-chunk path instead of being treated as a sequence. @@ -778,7 +765,7 @@ def normalize_chunks_1d(chunks: int | Iterable[object], span: int) -> DimensionG if chunk_size < -1 or chunk_size == 0: raise ValueError(f"Chunk size must be positive or -1, got {chunk_size}") if chunk_size == -1: - return FixedDimension(size=_full_span_chunk_size(span), extent=span) + return FixedDimension(size=full_span_chunk_size(span, unit), extent=span) return FixedDimension(size=chunk_size, extent=span) else: try: @@ -803,8 +790,6 @@ def normalize_chunks_1d(chunks: int | Iterable[object], span: int) -> DimensionG ints: list[int] = [int(c) for c in chunk_list] # type: ignore[call-overload] if any(c <= 0 for c in ints): raise ValueError(f"All chunk sizes must be positive, got {ints}") - # A zero-length span cannot be covered by positive edges; the edges are the - # chunks the axis will grow into, exactly as after ``resize(0)``. if span > 0 and sum(ints) != span: raise ValueError(f"Chunk sizes {ints} do not sum to span {span}") return VaryingDimension(ints, extent=span) @@ -813,6 +798,7 @@ def normalize_chunks_1d(chunks: int | Iterable[object], span: int) -> DimensionG def normalize_chunks_nd( chunks: Any, shape: tuple[int, ...], + unit: tuple[int, ...] | None = None, ) -> ChunkGrid: """ Normalize a chunk specification into a `ChunkGrid`. @@ -832,6 +818,10 @@ def normalize_chunks_nd( `ChunkGrid` directly. `chunks=None` and `chunks=True` are rejected here — the caller is responsible for choosing between explicit sizes and auto-chunking. + + `unit` gives, per axis, the size a chunk must be a multiple of (the inner + chunk shape, when normalizing a shard shape); it only affects the chunks + that `-1` and `False` derive from the span. """ from zarr.core.metadata.v3 import RectilinearChunkGridMetadata, RegularChunkGridMetadata @@ -845,8 +835,7 @@ def normalize_chunks_nd( f'{chunks!r} is not a valid chunk input. Use chunks=None or chunks="auto" from the top-level API for auto-chunking, or pass an int / tuple of ints.' ) - # handle no chunking: one chunk covering every axis. Routed through the -1 sentinel so - # the zero-length-axis rule lives in one place (_full_span_chunk_size). + # handle no chunking: one chunk covering every axis. if chunks is False: chunks = -1 @@ -860,8 +849,13 @@ def normalize_chunks_nd( f"chunks has {len(chunks)} dimensions but shape has {len(shape)} dimensions" ) + if unit is None: + unit = (1,) * len(shape) return ChunkGrid( - dimensions=tuple(normalize_chunks_1d(c, span=s) for c, s in zip(chunks, shape, strict=True)) + dimensions=tuple( + normalize_chunks_1d(c, span=s, unit=u) + for c, s, u in zip(chunks, shape, unit, strict=True) + ) ) @@ -1009,5 +1003,5 @@ def resolve_outer_and_inner_chunks( else: shard_flat = cast("tuple[int, ...]", shard_shape) - outer = normalize_chunks_nd(shard_flat, array_shape) + outer = normalize_chunks_nd(shard_flat, array_shape, unit=chunk_shape_flat) return ChunkLayout(outer_chunks=outer, inner=ChunkLayout(outer_chunks=chunks)) diff --git a/tests/test_chunk_grids.py b/tests/test_chunk_grids.py index cfd2e42258..d171937898 100644 --- a/tests/test_chunk_grids.py +++ b/tests/test_chunk_grids.py @@ -1,7 +1,4 @@ import contextlib -import json -import warnings -from pathlib import Path from typing import Any, Literal, cast import numpy as np @@ -16,6 +13,7 @@ VaryingDimension, _guess_num_chunks_per_axis_shard, _guess_regular_chunks, + full_span_chunk_size, normalize_chunks_1d, normalize_chunks_nd, resolve_outer_and_inner_chunks, @@ -371,148 +369,89 @@ def test_create_0d_array_auto_shards_with_target_shard_size() -> None: # -- Zero-length dimensions -- -# -# One invariant: a chunk edge length is always >= 1, an extent may be 0. Every spelling -# that derives a chunk size from a span (-1, False, "auto", shards="auto") must agree on -# chunk size 1 for a zero-length axis, in both Zarr formats, with or without sharding. -# Historically each spelling clamped (or failed to clamp) on its own; see #4304, #4305, -# #4307, #4328 and, further back, #150, #241, #303, #972, #1977, #2434, #3711. - -ZeroLengthChunkSpelling = Literal["minus-one", "false", "auto", "one", "ones", "rectilinear"] -ZeroLengthShards = Literal["auto", "auto-budget", "explicit"] | None - -# Spellings whose chunk size is derived from the axis span rather than given explicitly. -_SPAN_DERIVED_SPELLINGS: frozenset[ZeroLengthChunkSpelling] = frozenset( - {"minus-one", "false", "auto"} + + +@pytest.mark.parametrize( + ("span", "unit", "expected"), + [(0, 1, 1), (5, 1, 5), (0, 4, 4), (8, 4, 8), (10, 4, 12)], ) +def test_full_span_chunk_size(span: int, unit: int, expected: int) -> None: + """One chunk spanning an axis is the smallest positive multiple of `unit` covering it.""" + assert full_span_chunk_size(span, unit) == expected -def _zero_length_chunks_arg(spelling: ZeroLengthChunkSpelling, shape: tuple[int, ...]) -> Any: - """Translate a chunk-spelling id into the `chunks=` argument for `shape`.""" - match spelling: - case "minus-one": - return -1 - case "false": - return False - case "auto": - return "auto" - case "one": - return 1 - case "ones": - return (1,) * len(shape) - case "rectilinear": - return [[2, 2]] * len(shape) - - -@pytest.mark.parametrize("spelling", ["minus-one", "false", "auto", "one", "ones", "rectilinear"]) +@pytest.mark.parametrize("chunks", [-1, False, "auto"]) @pytest.mark.parametrize( "shape", [(0,), (0, 4), (4, 0), (0, 0), ()], ids=["1d", "2d-lead", "2d-trail", "2d-both", "0d"], ) @pytest.mark.parametrize( - ("zarr_format", "shards"), - [(2, None), (3, None), (3, "auto"), (3, "auto-budget"), (3, "explicit")], - ids=["v2", "v3", "v3-auto-shards", "v3-auto-shards-budget", "v3-explicit-shards"], + ("zarr_format", "shards", "target_shard_size_bytes"), + [(2, None, None), (3, None, None), (3, "auto", None), (3, "auto", 128 * 1024 * 1024)], + ids=["v2", "v3", "v3-auto-shards", "v3-auto-shards-budget"], ) def test_create_zero_length_array( - spelling: ZeroLengthChunkSpelling, + chunks: Any, shape: tuple[int, ...], zarr_format: Literal[2, 3], - shards: ZeroLengthShards, + shards: Literal["auto"] | None, + target_shard_size_bytes: int | None, ) -> None: - """Every chunk spelling produces a valid, usable grid on a zero-length axis. - - Span-derived spellings resolve to chunk size 1 on zero-length axes (and the full span - elsewhere, for these small shapes); explicit spellings are stored verbatim. In every case - the stored metadata matches `arr.chunks` / `arr.shards`, the array can grow along the - empty axis, round-trip data, and shrink back to empty. - """ - ndim = len(shape) - if spelling == "rectilinear": - if zarr_format == 2: - pytest.skip("Zarr format 2 does not support rectilinear chunk grids") - if shards is not None: - pytest.skip("rectilinear chunks with sharding is not supported") - if ndim == 0: - pytest.skip("a 0-d array has no dimension to chunk rectilinearly") - if shards == "explicit" and ndim == 0: - pytest.skip("a 0-d array has no axis to shard explicitly") - - chunks = _zero_length_chunks_arg(spelling, shape) - expected_chunks: tuple[int, ...] | None - if spelling in _SPAN_DERIVED_SPELLINGS: - expected_chunks = tuple(max(s, 1) for s in shape) - elif spelling == "rectilinear": - expected_chunks = None - else: - expected_chunks = (1,) * ndim - - shards_arg: Any - expected_shards: tuple[int, ...] | None - match shards: - case None: - shards_arg, expected_shards = None, None - case "auto" | "auto-budget": - # Axes this short never split, so the guessed shard equals the chunk. - shards_arg, expected_shards = "auto", expected_chunks - case "explicit": - # A shard larger than the (zero) extent is fine: the axis has zero shards. - shards_arg = tuple(2 if s == 0 else s for s in shape) - expected_shards = shards_arg - + """Every spelling of one chunk spanning the axis gives chunk size 1 on a + zero-length axis, and the array can grow along that axis and shrink back.""" + expected = tuple(max(s, 1) for s in shape) warns = ( pytest.warns(ZarrUserWarning, match="Automatic shard shape inference is experimental") - if shards_arg == "auto" + if shards == "auto" else contextlib.nullcontext() ) - budget = 128 * 1024 * 1024 if shards == "auto-budget" else None - # The rectilinear flag must stay set for the array's whole life, not just creation. - with zarr.config.set( - {"array.rectilinear_chunks": True, "array.target_shard_size_bytes": budget} - ): - with warns: - arr = zarr.create_array( - store={}, - shape=shape, - dtype="int64", - chunks=chunks, - shards=shards_arg, - zarr_format=zarr_format, - ) + with zarr.config.set({"array.target_shard_size_bytes": target_shard_size_bytes}), warns: + arr = zarr.create_array( + store={}, + shape=shape, + dtype="int64", + chunks=chunks, + shards=shards, + zarr_format=zarr_format, + ) + assert arr.chunks == expected + assert arr.shards == (None if shards is None else expected) + meta = cast(dict[str, Any], arr.metadata.to_dict()) + stored = ( + meta["chunks"] if zarr_format == 2 else meta["chunk_grid"]["configuration"]["chunk_shape"] + ) + assert tuple(stored) == expected - # In-memory view and stored metadata agree with the invariant. - assert arr.shards == expected_shards - meta = cast(dict[str, Any], arr.metadata.to_dict()) - if spelling == "rectilinear": - grid = meta["chunk_grid"] - assert grid["name"] == "rectilinear" - # Stored verbatim on zero-length axes too, run-length encoded as [size, count]. - assert list(grid["configuration"]["chunk_shapes"]) == [[[2, 2]]] * ndim - assert arr.write_chunk_sizes == tuple(() if s == 0 else (2, 2) for s in shape) - else: - assert arr.chunks == expected_chunks - if zarr_format == 2: - assert meta["chunks"] == expected_chunks - else: - stored = meta["chunk_grid"]["configuration"]["chunk_shape"] - assert stored == (expected_chunks if expected_shards is None else expected_shards) - assert all(c >= 1 for c in arr.chunks) - - # The array must remain usable. - if ndim == 0: - arr[...] = 7 - assert arr[...] == 7 - return - axis = shape.index(0) - grown = tuple(2 if i == axis else s for i, s in enumerate(shape)) - data = np.full(grown, 7, dtype="int64") - arr.append(data, axis=axis) - assert arr.shape == grown - np.testing.assert_array_equal(arr[...], data) - arr.resize(shape) - assert arr.shape == shape - assert np.asarray(arr[...]).shape == shape + if not shape: + arr[...] = 7 + assert arr[...] == 7 + return + axis = shape.index(0) + grown = tuple(2 if i == axis else s for i, s in enumerate(shape)) + data = np.full(grown, 7, dtype="int64") + arr.append(data, axis=axis) + np.testing.assert_array_equal(arr[...], data) + arr.resize(shape) + assert np.asarray(arr[...]).shape == shape + + +@pytest.mark.parametrize( + ("shape", "chunks", "shards", "expected"), + [ + ((0, 20), (5, 5), -1, (5, 20)), + ((0,), (4,), False, (4,)), + ((10,), (4,), -1, (12,)), + ((8, 0), (4, 3), (-1, 6), (8, 6)), + ], +) +def test_create_full_span_shards( + shape: tuple[int, ...], chunks: tuple[int, ...], shards: Any, expected: tuple[int, ...] +) -> None: + """A shard spanning the axis is a multiple of the inner chunk, even on a zero-length axis.""" + arr = zarr.create_array(store={}, shape=shape, chunks=chunks, shards=shards, dtype="int8") + assert arr.shards == expected + assert arr.chunks == chunks @pytest.mark.parametrize("zarr_format", [2, 3]) @@ -523,11 +462,7 @@ def test_create_zero_chunk_rejected(zarr_format: Literal[2, 3]) -> None: def test_rectilinear_zero_extent_matches_resize() -> None: - """Creating a rectilinear axis at length 0 equals resizing one down to 0. - - Both leave a `VaryingDimension` whose edges lie entirely beyond the extent, so the - stored grids are identical and both grow into the same chunks on append. - """ + """Creating a rectilinear axis at length 0 equals resizing one down to 0.""" with zarr.config.set({"array.rectilinear_chunks": True}): created = zarr.create_array(store={}, shape=(0,), chunks=[[2, 2]], dtype="int64") resized = zarr.create_array(store={}, shape=(4,), chunks=[[2, 2]], dtype="int64") @@ -544,56 +479,6 @@ def test_rectilinear_zero_extent_matches_resize() -> None: assert created.write_chunk_sizes == resized.write_chunk_sizes == ((2, 1),) -def _store_legacy_zero_chunk(path: Any, zarr_format: Literal[2, 3], stored: Any) -> None: - """Rewrite an array's stored chunk size to *stored*, as older zarr-python did.""" - doc_name = ".zarray" if zarr_format == 2 else "zarr.json" - doc_path = path / doc_name - doc = json.loads(doc_path.read_text()) - if zarr_format == 2: - doc["chunks"] = [stored] - else: - doc["chunk_grid"]["configuration"]["chunk_shape"] = [stored] - doc_path.write_text(json.dumps(doc)) - - -@pytest.mark.parametrize("zarr_format", [2, 3]) -@pytest.mark.parametrize("stored", [0, False], ids=["zero", "false"]) -@pytest.mark.parametrize("extent", [0, 3], ids=["empty-axis", "grown-axis"]) -def test_legacy_zero_chunk_round_trip( - tmp_path: Path, zarr_format: Literal[2, 3], stored: Any, extent: int -) -> None: - """An array whose stored chunk size is 0 opens with one chunk spanning the - axis, appends without losing data, and re-saves as a valid chunk size. - - zarr-python wrote such metadata for an array created with a zero-length axis - until 3.4: Zarr format 2 through 3.3, and Zarr format 3 in 3.0 and 3.1, which - also wrote JSON `false` for `chunks=False`. Those versions could then grow - the axis (a 3.4.0 append, a 3.1 resize) while storing no chunk, so the - grown axis holds only the fill value. - """ - path = tmp_path / "legacy.zarr" - zarr.create_array( - store=path, shape=(extent,), chunks=(4,), dtype="int64", zarr_format=zarr_format - ) - _store_legacy_zero_chunk(path, zarr_format, stored) - - with pytest.warns(ZarrUserWarning, match="one chunk spanning the axis"): - arr = zarr.open_array(store=path, mode="a") - assert arr.chunks == (max(extent, 1),) - np.testing.assert_array_equal(arr[...], np.zeros(extent, dtype="int64")) - - arr.append(np.arange(3, dtype="int64")) - expected = np.concatenate([np.zeros(extent, dtype="int64"), np.arange(3)]) - np.testing.assert_array_equal(zarr.open_array(store=path)[...], expected) - - # The warning says to do this; it must leave metadata that reopens cleanly. - arr.update_attributes({}) - with warnings.catch_warnings(): - warnings.simplefilter("error", ZarrUserWarning) - reopened = zarr.open_array(store=path) - assert reopened.chunks == (max(extent, 1),) - - def test_normalize_chunks_1d_zero_span_accepts_any_edges() -> None: """On a zero-length span the explicit edge list is stored verbatim.""" dim = normalize_chunks_1d([3, 5], span=0) From e49eb7e7b4c1f66dab6bf5a803f086199c572922 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Fri, 25 Sep 2026 21:55:20 +0200 Subject: [PATCH 14/41] refactor(metadata): read invalid stored chunk sizes in one module of document upgrades The metadata constructors are strict again: a chunk edge length is an int of at least 1, so metadata built in code with 0 or True raises, without a warning. Stored documents are read leniently only in zarr.core.metadata.upgrades, whose upgrades map a stored array document to a valid one before the constructors run. ArrayV2Metadata.from_dict and ArrayV3Metadata.from_dict apply them, so opening an array and parsing consolidated metadata take the same path. The first upgrade reads a regular chunk size of 0 or JSON false as one chunk spanning the axis, a multiple of the inner chunk size when the array is sharded (this opens sharded stores with an outer chunk size of 0), and JSON true, which real releases wrote for chunks=(True,), as 1. Its one warning says what is invalid, how it is read, and how to re-save, including zarr.consolidate_metadata. Assisted-by: ClaudeCode:claude-opus-5-5 Co-Authored-By: Claude Opus 5.5 --- src/zarr/core/metadata/common.py | 63 +----- src/zarr/core/metadata/upgrades.py | 135 ++++++++++++ src/zarr/core/metadata/v2.py | 18 +- src/zarr/core/metadata/v3.py | 59 +----- tests/test_metadata/test_common.py | 72 ------- tests/test_metadata/test_upgrades.py | 293 +++++++++++++++++++++++++++ 6 files changed, 456 insertions(+), 184 deletions(-) create mode 100644 src/zarr/core/metadata/upgrades.py delete mode 100644 tests/test_metadata/test_common.py create mode 100644 tests/test_metadata/test_upgrades.py diff --git a/src/zarr/core/metadata/common.py b/src/zarr/core/metadata/common.py index 00f1e96254..b9e86f537a 100644 --- a/src/zarr/core/metadata/common.py +++ b/src/zarr/core/metadata/common.py @@ -1,25 +1,10 @@ from __future__ import annotations -import warnings -from typing import TYPE_CHECKING, Final - -from zarr.errors import ZarrUserWarning +from typing import TYPE_CHECKING if TYPE_CHECKING: - from collections.abc import Sequence - from zarr.core.common import JSON -RESAVE_METADATA_HINT: Final = ( - "Re-save the array metadata to store the corrected value: open the array " - "writable and call `array.update_attributes({})`." -) -"""How to persist metadata that was read under a compatibility policy. - -`update_attributes` rewrites the whole metadata document from the parsed -(corrected) metadata, so an empty update is enough. -""" - def parse_attributes(data: dict[str, JSON] | None) -> dict[str, JSON]: if data is None: @@ -28,46 +13,10 @@ def parse_attributes(data: dict[str, JSON] | None) -> dict[str, JSON]: return dict(data) -def parse_stored_regular_chunk_shape( - chunk_shape: Sequence[int], shape: Sequence[int] -) -> tuple[int, ...]: - """Validate a stored regular chunk grid's chunk shape against the array shape. - - This is for regular chunk grids only: Zarr format 2 `chunks`, and the - `chunk_shape` of a Zarr format 3 `regular` grid. Another chunk grid, such - as the rectilinear grid, is free to define its own meaning for a chunk of - length 0, so its chunk sizes must not be passed here. - - The chunk shape must have one entry per array axis, and every chunk size - must be at least 1, with one exception: a chunk size of 0 (or JSON - `false`) is read as one chunk spanning its axis, `max(extent, 1)`, with a - `ZarrUserWarning` that says how to re-save valid metadata. A chunk size of - 0 gives a grid of zero chunks, so no chunk can be stored under it and any - positive chunk size reads the store correctly; if the axis has positive - length, the warning also says that it holds only the fill value. A - negative chunk size is rejected. - """ - if len(chunk_shape) != len(shape): +def parse_chunk_edge(size: object, axis: int) -> int: + """Check that `size` is a chunk edge length: an `int` (not a `bool`) of at least 1.""" + if isinstance(size, bool) or not isinstance(size, int) or size < 1: raise ValueError( - f"The chunk shape {tuple(chunk_shape)} and the array shape {tuple(shape)} " - "must have the same number of dimensions." + f"Dimension {axis}: chunk edge length must be an integer >= 1, got {size!r}" ) - parsed: list[int] = [] - for dim_idx, (size, extent) in enumerate(zip(chunk_shape, shape, strict=True)): - if size < 0: - raise ValueError(f"Dimension {dim_idx}: chunk edge length must be >= 1, got {size!r}") - if size == 0: - corrected = max(extent, 1) - msg = ( - f"Dimension {dim_idx}: chunk edge length {size!r} is invalid and is read " - f"as one chunk spanning the axis, of size {corrected}." - ) - if extent > 0: - msg += ( - f" No chunk can be stored under a chunk size of 0, so the {extent} " - "elements along this axis hold only the fill value." - ) - warnings.warn(f"{msg} {RESAVE_METADATA_HINT}", ZarrUserWarning, stacklevel=3) - size = corrected - parsed.append(size) - return tuple(parsed) + return size diff --git a/src/zarr/core/metadata/upgrades.py b/src/zarr/core/metadata/upgrades.py new file mode 100644 index 0000000000..933b4c53e2 --- /dev/null +++ b/src/zarr/core/metadata/upgrades.py @@ -0,0 +1,135 @@ +"""Upgrades that read invalid stored array metadata documents written by older software. + +This is the only place invalid metadata is read leniently; the metadata constructors +are strict. An upgrade maps a stored array metadata document (parsed JSON) to a valid +one and warns once when it changes anything. `ArrayV2Metadata.from_dict` and +`ArrayV3Metadata.from_dict` apply the upgrades for their Zarr format, so every path +that parses a stored document, including consolidated metadata, goes through them. + +To read another kind of invalid document, add an upgrade to `V2_ARRAY_UPGRADES` or +`V3_ARRAY_UPGRADES`. +""" + +from __future__ import annotations + +import json +import warnings +from collections.abc import Callable, Mapping, Sequence +from typing import TYPE_CHECKING, Final + +from typing_extensions import TypeIs + +from zarr.core.chunk_grids import full_span_chunk_size +from zarr.errors import ZarrUserWarning + +if TYPE_CHECKING: + from zarr.core.common import JSON + +type ArrayDocument = Mapping[str, JSON] +type Upgrade = Callable[[ArrayDocument], ArrayDocument] + +RESAVE_HINT: Final = ( + "To store valid metadata, open the array writable and call `array.update_attributes({})`; " + "if a group holds consolidated metadata for the array, then also call " + "`zarr.consolidate_metadata` on that group." +) + + +def _warn(message: str) -> None: + # The synchronous API parses metadata on zarr's IO thread, whose stack holds no + # user code, so the warning points at the upgrade on every path. + warnings.warn(f"{message} {RESAVE_HINT}", ZarrUserWarning, stacklevel=2) + + +def _is_int_list(value: object) -> TypeIs[list[int] | tuple[int, ...]]: + """Whether `value` is a JSON array of integers (JSON `false` and `true` count).""" + return isinstance(value, list | tuple) and all(isinstance(v, int) for v in value) + + +def _read_invalid_chunk_sizes(chunk_shape: JSON, shape: JSON, unit: JSON) -> list[int] | None: + """Read a regular chunk shape whose chunk sizes include 0, JSON `false` or JSON `true`. + + 0 and `false` are read as one chunk spanning the axis, a multiple of `unit` on that + axis (the inner chunk shape of a sharded array); `true` is read as 1. Returns `None` + when there is nothing to read this way: anything else that is invalid is left for + the metadata constructors to reject. + """ + if not (_is_int_list(chunk_shape) and _is_int_list(shape)) or len(chunk_shape) != len(shape): + return None + invalid_axes = [ + axis for axis, size in enumerate(chunk_shape) if size == 0 or isinstance(size, bool) + ] + if not invalid_axes: + return None + units = ( + list(unit) + if _is_int_list(unit) and len(unit) == len(shape) and all(u >= 1 for u in unit) + else [1] * len(shape) + ) + upgraded = [ + full_span_chunk_size(extent, int(u)) if size == 0 else int(size) + for size, extent, u in zip(chunk_shape, shape, units, strict=True) + ] + readings = "; ".join( + f"{json.dumps(chunk_shape[axis])} on axis {axis} as " + + ("1" if chunk_shape[axis] else f"one chunk spanning the axis ({upgraded[axis]})") + for axis in invalid_axes + ) + message = ( + f"The stored chunk shape {json.dumps(list(chunk_shape))} is invalid: chunk sizes " + f"must be integers of at least 1. It is read as {upgraded}, reading {readings}." + ) + if any(chunk_shape[axis] == 0 and shape[axis] > 0 for axis in invalid_axes): + message += ( + " No chunk can be stored under a chunk size of 0, so the array holds only its " + "fill value." + ) + _warn(message) + return upgraded + + +def _invalid_chunk_sizes_v2(doc: ArrayDocument) -> ArrayDocument: + chunks = _read_invalid_chunk_sizes(doc.get("chunks"), doc.get("shape"), None) + return doc if chunks is None else {**doc, "chunks": chunks} + + +def _sharding_chunk_shape(codecs: JSON) -> JSON: + """The inner chunk shape of a sharding codec in a Zarr format 3 codec list, if any.""" + if isinstance(codecs, Sequence) and not isinstance(codecs, str): + for codec in codecs: + if isinstance(codec, Mapping) and codec.get("name") == "sharding_indexed": + configuration = codec.get("configuration") + if isinstance(configuration, Mapping): + return configuration.get("chunk_shape") + return None + + +def _invalid_chunk_sizes_v3(doc: ArrayDocument) -> ArrayDocument: + grid = doc.get("chunk_grid") + if not isinstance(grid, Mapping) or grid.get("name") != "regular": + return doc + configuration = grid.get("configuration") + if not isinstance(configuration, Mapping): + return doc + chunk_shape = _read_invalid_chunk_sizes( + configuration.get("chunk_shape"), + doc.get("shape"), + _sharding_chunk_shape(doc.get("codecs")), + ) + if chunk_shape is None: + return doc + return { + **doc, + "chunk_grid": {**grid, "configuration": {**configuration, "chunk_shape": chunk_shape}}, + } + + +V2_ARRAY_UPGRADES: Final[tuple[Upgrade, ...]] = (_invalid_chunk_sizes_v2,) +V3_ARRAY_UPGRADES: Final[tuple[Upgrade, ...]] = (_invalid_chunk_sizes_v3,) + + +def upgrade_array_document(doc: ArrayDocument, upgrades: Sequence[Upgrade]) -> ArrayDocument: + """Apply `upgrades` to a stored array metadata document, in order.""" + for upgrade in upgrades: + doc = upgrade(doc) + return doc diff --git a/src/zarr/core/metadata/v2.py b/src/zarr/core/metadata/v2.py index 3ae2a8bd96..0eb28efb11 100644 --- a/src/zarr/core/metadata/v2.py +++ b/src/zarr/core/metadata/v2.py @@ -42,7 +42,8 @@ ) from zarr.core.config import config, parse_indexing_order from zarr.core.json_parse import parse_field -from zarr.core.metadata.common import parse_attributes, parse_stored_regular_chunk_shape +from zarr.core.metadata.common import parse_attributes, parse_chunk_edge +from zarr.core.metadata.upgrades import V2_ARRAY_UPGRADES, upgrade_array_document class ArrayV2MetadataDict(TypedDict): @@ -88,7 +89,7 @@ def __init__( Metadata for a Zarr format 2 array. """ shape_parsed = parse_shapelike(shape) - chunks_parsed = parse_stored_regular_chunk_shape(parse_shapelike(chunks), shape_parsed) + chunks_parsed = parse_chunks(chunks, shape_parsed) compressor_parsed = parse_compressor(compressor) order_parsed = parse_indexing_order(order) dimension_separator_parsed = parse_separator(dimension_separator) @@ -147,7 +148,7 @@ def to_buffer_dict(self, prototype: BufferPrototype) -> dict[str, Buffer]: @classmethod def from_dict(cls, data: dict[str, Any]) -> ArrayV2Metadata: - # Make a copy to protect the original from modification. + data = dict(upgrade_array_document(data, V2_ARRAY_UPGRADES)) _data = data.copy() # Check that the zarr_format attribute is correct. _ = parse_zarr_format(_data.pop("zarr_format")) @@ -320,6 +321,17 @@ def parse_compressor(data: object) -> Numcodec | None: raise ValueError(msg) +def parse_chunks(chunks: Iterable[int], shape: tuple[int, ...]) -> tuple[int, ...]: + """Check a chunk shape: one chunk edge length (an integer >= 1) per array axis.""" + chunks_parsed = tuple(parse_chunk_edge(size, axis) for axis, size in enumerate(chunks)) + if len(chunks_parsed) != len(shape): + raise ValueError( + f"The `shape` and `chunks` attributes must have the same length. " + f"`chunks` has length {len(chunks_parsed)}, but `shape` has length {len(shape)}." + ) + return chunks_parsed + + def get_object_codec_id(maybe_object_codecs: Sequence[JSON]) -> str | None: """ Inspect a sequence of codecs / filters for an "object codec", i.e. a codec diff --git a/src/zarr/core/metadata/v3.py b/src/zarr/core/metadata/v3.py index 1db659beb8..4a27d495b3 100644 --- a/src/zarr/core/metadata/v3.py +++ b/src/zarr/core/metadata/v3.py @@ -5,7 +5,6 @@ from dataclasses import dataclass, field, replace from typing import TYPE_CHECKING, Any, Final, Literal, NotRequired, TypeGuard, cast -import numpy as np from typing_extensions import TypedDict from zarr.abc.codec import ArrayArrayCodec, ArrayBytesCodec, BytesBytesCodec, Codec @@ -36,7 +35,8 @@ from zarr.core.dtype import VariableLengthUTF8, ZDType, get_data_type_from_json from zarr.core.dtype.common import check_dtype_spec_v3 from zarr.core.json_parse import parse_field -from zarr.core.metadata.common import parse_attributes, parse_stored_regular_chunk_shape +from zarr.core.metadata.common import parse_attributes, parse_chunk_edge +from zarr.core.metadata.upgrades import V3_ARRAY_UPGRADES, upgrade_array_document from zarr.errors import MetadataValidationError, NodeTypeValidationError from zarr.registry import get_codec_class @@ -235,16 +235,12 @@ def _validate_chunk_shapes( result: list[int | tuple[int, ...]] = [] for dim_idx, dim_spec in enumerate(chunk_shapes): if isinstance(dim_spec, int): - if dim_spec < 1: - raise ValueError( - f"Dimension {dim_idx}: integer chunk edge length must be >= 1, got {dim_spec}" - ) - result.append(dim_spec) + result.append(parse_chunk_edge(dim_spec, dim_idx)) else: edges = tuple(dim_spec) if not edges: raise ValueError(f"Dimension {dim_idx} has no chunk edges.") - bad = [i for i, e in enumerate(edges) if e < 1] + bad = [i for i, e in enumerate(edges) if isinstance(e, bool) or e < 1] if bad: raise ValueError( f"Dimension {dim_idx} has invalid edge lengths at indices {bad}: " @@ -477,45 +473,6 @@ class ArrayMetadataJSON_V3(TypedDict, extra_items=AllowedExtraField): # type: i } -def _is_regular_chunk_shape(value: object) -> TypeGuard[Sequence[int]]: - """Whether a stored `chunk_shape` is a regular chunk shape: a sequence of integers. - - JSON `false` counts, because `bool` is an integer type. - """ - return ( - isinstance(value, Sequence) - and not isinstance(value, str) - and all(isinstance(size, int | np.integer) for size in value) - ) - - -def _parse_stored_regular_chunk_grid( - chunk_grid: dict[str, JSON] | ChunkGridMetadata | NamedConfig[str, Any], - shape: tuple[int, ...], -) -> dict[str, JSON] | ChunkGridMetadata | NamedConfig[str, Any]: - """Check a stored regular chunk grid's chunk shape against the array shape. - - Only a `regular` grid whose `chunk_shape` is all integers is a regular chunk - shape, and only that is handed to `parse_stored_regular_chunk_shape`. - Anything else is not a regular chunk shape and is left for the chunk grid - parser: other grids define their own chunk semantics. This runs here rather - than in the grid parser because it needs the array shape, which chunk grid - metadata does not carry. - """ - if not isinstance(chunk_grid, Mapping) or chunk_grid.get("name") != "regular": - return chunk_grid - configuration = chunk_grid.get("configuration") - if not isinstance(configuration, Mapping): - return chunk_grid - chunk_shape = configuration.get("chunk_shape") - if not _is_regular_chunk_shape(chunk_shape): - return chunk_grid - parsed = parse_stored_regular_chunk_shape(chunk_shape, shape) - corrected: dict[str, Any] = dict(chunk_grid) - corrected["configuration"] = {**configuration, "chunk_shape": list(parsed)} - return corrected - - @dataclass(frozen=True, kw_only=True) class ArrayV3Metadata(Metadata): shape: tuple[int, ...] @@ -550,9 +507,7 @@ def __init__( """ shape_parsed = parse_shapelike(shape) - chunk_grid_parsed = parse_chunk_grid( - _parse_stored_regular_chunk_grid(chunk_grid, shape_parsed) - ) + chunk_grid_parsed = parse_chunk_grid(chunk_grid) chunk_key_encoding_parsed = parse_chunk_key_encoding(chunk_key_encoding) dimension_names_parsed = parse_dimension_names(dimension_names) # Note: relying on a type method is numpy-specific @@ -673,8 +628,8 @@ def to_buffer_dict(self, prototype: BufferPrototype) -> dict[str, Buffer]: @classmethod def from_dict(cls, data: dict[str, JSON]) -> Self: - # make a copy because we are modifying the dict - _data = data.copy() + # a new dict, because we are modifying it + _data = dict(upgrade_array_document(data, V3_ARRAY_UPGRADES)) # check that the zarr_format attribute is correct _ = parse_zarr_format(_data.pop("zarr_format")) diff --git a/tests/test_metadata/test_common.py b/tests/test_metadata/test_common.py deleted file mode 100644 index c3ed9416eb..0000000000 --- a/tests/test_metadata/test_common.py +++ /dev/null @@ -1,72 +0,0 @@ -"""Tests for metadata helpers shared by both Zarr formats.""" - -from __future__ import annotations - -import re -import warnings -from typing import Any - -import numpy as np -import pytest - -from zarr.core.metadata.common import parse_stored_regular_chunk_shape -from zarr.errors import ZarrUserWarning - - -@pytest.mark.parametrize( - ("chunk_shape", "shape", "expected", "warning"), - [ - ((4, 5), (10, 10), (4, 5), None), - ((1, 1), (0, 0), (1, 1), None), - ((0,), (0,), (1,), "of size 1"), - ((False,), (0,), (1,), "of size 1"), - ((np.int64(0),), (0,), (1,), "of size 1"), - ((4, 0), (4, 0), (4, 1), "Dimension 1"), - ((0, 0), (0, 0), (1, 1), "Dimension 0"), - ((0,), (5,), (5,), "5 elements along this axis hold only the fill value"), - ((4, 0), (4, 3), (4, 3), "3 elements along this axis hold only the fill value"), - ], - ids=[ - "valid", - "valid-on-empty-axes", - "legacy-zero", - "legacy-json-false", - "legacy-numpy-zero", - "only-zero-size-corrected", - "every-zero-size-corrected", - "legacy-zero-on-grown-axis", - "legacy-zero-on-grown-axis-2d", - ], -) -def test_parse_stored_regular_chunk_shape( - chunk_shape: tuple[Any, ...], - shape: tuple[int, ...], - expected: tuple[Any, ...], - warning: str | None, -) -> None: - """A valid chunk shape is returned as is; a chunk size of 0 is read as one - chunk spanning the axis, with a warning saying how to re-save and, on an axis - of positive length, that it holds only the fill value.""" - with warnings.catch_warnings(record=True) as record: - warnings.simplefilter("always") - parsed = parse_stored_regular_chunk_shape(chunk_shape, shape) - assert parsed == expected - messages = [str(w.message) for w in record if issubclass(w.category, ZarrUserWarning)] - if warning is None: - assert messages == [] - else: - assert any(re.search(warning, message) for message in messages) - for message in messages: - assert "update_attributes({})" in message - - -def test_parse_stored_regular_chunk_shape_rejects_dimension_mismatch() -> None: - """The chunk shape needs one entry per array axis.""" - with pytest.raises(ValueError, match="same number of dimensions"): - parse_stored_regular_chunk_shape((4,), (10, 10)) - - -def test_parse_stored_regular_chunk_shape_rejects_negative() -> None: - """A negative chunk size is rejected even on a zero-length axis.""" - with pytest.raises(ValueError, match="chunk edge length must be >= 1, got -1"): - parse_stored_regular_chunk_shape((-1,), (0,)) diff --git a/tests/test_metadata/test_upgrades.py b/tests/test_metadata/test_upgrades.py new file mode 100644 index 0000000000..7506bb6875 --- /dev/null +++ b/tests/test_metadata/test_upgrades.py @@ -0,0 +1,293 @@ +"""Tests for the upgrades that read invalid stored array metadata documents.""" + +from __future__ import annotations + +import json +import re +import warnings +from typing import TYPE_CHECKING, Any, Literal + +import numpy as np +import pytest + +import zarr +from zarr.core.metadata import ArrayV2Metadata, ArrayV3Metadata +from zarr.core.metadata.upgrades import ( + V2_ARRAY_UPGRADES, + V3_ARRAY_UPGRADES, + upgrade_array_document, +) +from zarr.core.metadata.v3 import RegularChunkGridMetadata +from zarr.dtype import Int16 +from zarr.errors import ZarrUserWarning + +if TYPE_CHECKING: + from pathlib import Path + + from zarr.core.common import JSON + + +def _v2_doc(shape: list[int], chunks: list[Any]) -> dict[str, JSON]: + return { + "zarr_format": 2, + "shape": shape, + "chunks": chunks, + "dtype": " dict[str, JSON]: + bytes_codec: dict[str, JSON] = {"name": "bytes", "configuration": {"endian": "little"}} + codecs: list[JSON] = [bytes_codec] + if inner is not None: + codecs = [ + { + "name": "sharding_indexed", + "configuration": { + "chunk_shape": inner, + "codecs": [bytes_codec], + "index_codecs": [bytes_codec, {"name": "crc32c"}], + "index_location": "end", + }, + } + ] + return { + "zarr_format": 3, + "node_type": "array", + "shape": shape, + "data_type": "int16", + "chunk_grid": {"name": "regular", "configuration": {"chunk_shape": chunk_shape}}, + "chunk_key_encoding": {"name": "default", "configuration": {"separator": "/"}}, + "fill_value": 0, + "codecs": codecs, + } + + +def _stored_chunks(doc: dict[str, Any]) -> Any: + return ( + doc["chunks"] + if doc["zarr_format"] == 2 + else doc["chunk_grid"]["configuration"]["chunk_shape"] + ) + + +@pytest.mark.parametrize( + ("doc", "expected", "warning"), + [ + (_v2_doc([10, 10], [4, 5]), [4, 5], None), + (_v3_doc([0, 0], [1, 1]), [1, 1], None), + (_v3_doc([10], [4], inner=[2]), [4], None), + (_v2_doc([0, 4], [0, 4]), [1, 4], "0 on axis 0 as one chunk spanning the axis"), + (_v3_doc([0], [False]), [1], "false on axis 0 as one chunk spanning the axis"), + (_v2_doc([5], [True]), [1], "true on axis 0 as 1"), + (_v3_doc([5, 4], [True, 4]), [1, 4], "true on axis 0 as 1"), + (_v2_doc([3], [0]), [3], "holds only its fill value"), + (_v3_doc([4, 3], [4, 0]), [4, 3], "holds only its fill value"), + (_v3_doc([0], [0], inner=[4]), [4], "spanning the axis \\(4\\)"), + (_v3_doc([10], [0], inner=[4]), [12], "spanning the axis \\(12\\)"), + (_v3_doc([0, 3], [0, 3], inner=[2, 3]), [2, 3], "spanning the axis \\(2\\)"), + ], + ids=[ + "v2-valid", + "v3-valid-empty-axes", + "v3-valid-sharded", + "v2-zero-empty-axis", + "v3-false-empty-axis", + "v2-true", + "v3-true", + "v2-zero-grown-axis", + "v3-zero-grown-axis", + "v3-sharded-zero-empty-axis", + "v3-sharded-zero-grown-axis", + "v3-sharded-zero-2d", + ], +) +def test_upgrade_array_document( + doc: dict[str, JSON], expected: list[int], warning: str | None +) -> None: + """Valid documents pass unchanged and silently. A stored chunk size of 0 or `false` + is read as one chunk spanning the axis (a multiple of the inner chunk when sharded) + and `true` as 1, with one warning that says how to re-save.""" + upgrades = V2_ARRAY_UPGRADES if doc["zarr_format"] == 2 else V3_ARRAY_UPGRADES + with warnings.catch_warnings(record=True) as record: + warnings.simplefilter("always") + upgraded = upgrade_array_document(doc, upgrades) + assert _stored_chunks(dict(upgraded)) == expected + assert {k: v for k, v in upgraded.items() if k not in ("chunks", "chunk_grid")} == { + k: v for k, v in doc.items() if k not in ("chunks", "chunk_grid") + } + messages = [str(w.message) for w in record] + if warning is None: + assert upgraded is doc + assert messages == [] + else: + assert len(messages) == 1 + assert re.search(warning, messages[0]) + assert "update_attributes({})" in messages[0] + assert "zarr.consolidate_metadata" in messages[0] + metadata_cls = ArrayV2Metadata if doc["zarr_format"] == 2 else ArrayV3Metadata + with warnings.catch_warnings(): + warnings.simplefilter("ignore", ZarrUserWarning) + metadata_cls.from_dict(dict(doc)) + + +@pytest.mark.parametrize("doc", [_v2_doc([4], [-1]), _v3_doc([4], [-1])], ids=["v2", "v3"]) +def test_stored_negative_chunk_size_rejected(doc: dict[str, JSON]) -> None: + metadata_cls = ArrayV2Metadata if doc["zarr_format"] == 2 else ArrayV3Metadata + with pytest.raises(ValueError, match="chunk edge length must be an integer >= 1, got -1"): + metadata_cls.from_dict(doc) + + +@pytest.mark.parametrize("doc", [_v2_doc([4, 4], [0]), _v3_doc([4, 4], [0])], ids=["v2", "v3"]) +def test_stored_chunk_shape_ndim_mismatch_rejected(doc: dict[str, JSON]) -> None: + """A chunk shape with the wrong number of axes is not upgraded, only rejected.""" + metadata_cls = ArrayV2Metadata if doc["zarr_format"] == 2 else ArrayV3Metadata + with warnings.catch_warnings(): + warnings.simplefilter("error", ZarrUserWarning) + with pytest.raises(ValueError, match="chunk edge length|same length|same number"): + metadata_cls.from_dict(doc) + + +@pytest.mark.parametrize("size", [0, True], ids=["zero", "true"]) +def test_constructor_rejects_invalid_chunk_size(size: int) -> None: + """Metadata built in code is strict: 0 and `True` are rejected, without a warning.""" + with warnings.catch_warnings(): + warnings.simplefilter("error", ZarrUserWarning) + with pytest.raises(ValueError, match=f"got {size!r}"): + RegularChunkGridMetadata(chunk_shape=(size,)) + with pytest.raises(ValueError, match=f"got {size!r}"): + ArrayV2Metadata( + shape=(0,), + chunks=(size,), + dtype=Int16(), + fill_value=0, + order="C", + ) + + +def _rewrite_doc(path: Path, zarr_format: Literal[2, 3], edit: Any) -> None: + doc_path = path / (".zarray" if zarr_format == 2 else "zarr.json") + doc = json.loads(doc_path.read_text()) + edit(doc) + doc_path.write_text(json.dumps(doc)) + + +@pytest.mark.parametrize( + ("zarr_format", "shape", "stored", "inner", "expected"), + [ + (2, (0, 4), [0, 4], None, (1, 4)), + (2, (3,), [0], None, (3,)), + (2, (5,), [True], None, (1,)), + (3, (0,), [False], None, (1,)), + (3, (3,), [0], None, (3,)), + (3, (5,), [True], None, (1,)), + (3, (0,), [0], (4,), (4,)), + (3, (10,), [0], (4,), (12,)), + (3, (0, 3), [0, 3], (2, 3), (2, 3)), + ], + ids=[ + "v2-empty-2d", + "v2-grown", + "v2-true", + "v3-false-empty", + "v3-grown", + "v3-true", + "v3-sharded-empty", + "v3-sharded-grown", + "v3-sharded-2d", + ], +) +def test_legacy_chunk_size_round_trip( + tmp_path: Path, + zarr_format: Literal[2, 3], + shape: tuple[int, ...], + stored: list[Any], + inner: tuple[int, ...] | None, + expected: tuple[int, ...], +) -> None: + """A store whose metadata holds a chunk size written by older software opens with a + warning, reads and appends under the upgraded grid, and re-saves valid metadata.""" + path = tmp_path / "legacy.zarr" + arr = zarr.create_array( + store=path, + shape=shape, + chunks=inner or expected, + shards=expected if inner else None, + dtype="int16", + fill_value=0, + zarr_format=zarr_format, + ) + data = np.arange(np.prod(shape), dtype="int16").reshape(shape) + arr[...] = data + + def store_legacy(doc: dict[str, Any]) -> None: + if zarr_format == 2: + doc["chunks"] = stored + else: + doc["chunk_grid"]["configuration"]["chunk_shape"] = stored + + _rewrite_doc(path, zarr_format, store_legacy) + + with pytest.warns(ZarrUserWarning, match="is read as"): + arr = zarr.open_array(store=path, mode="a") + assert (arr.shards or arr.chunks) == expected + np.testing.assert_array_equal(arr[...], data) + + block = np.full((2, *shape[1:]), 7, dtype="int16") + arr.append(block) + arr.update_attributes({}) + with warnings.catch_warnings(): + warnings.simplefilter("error", ZarrUserWarning) + reopened = zarr.open_array(store=path) + np.testing.assert_array_equal(reopened[...], np.concatenate([data, block])) + + +@pytest.mark.filterwarnings("ignore:Consolidated metadata is currently not part:UserWarning") +@pytest.mark.parametrize("zarr_format", [2, 3]) +def test_legacy_chunk_size_consolidated(tmp_path: Path, zarr_format: Literal[2, 3]) -> None: + """Consolidated metadata goes through the same upgrade; re-saving the array and + consolidating again leaves a group that opens without a warning.""" + path = tmp_path / "group.zarr" + group = zarr.open_group(path, mode="w", zarr_format=zarr_format) + group.create_array("a", shape=(0,), chunks=(1,), dtype="int32") + zarr.consolidate_metadata(path) + + # What zarr-python wrote for `chunks=(0,)` on an empty array, in both copies. + if zarr_format == 2: + _rewrite_doc(path / "a", 2, lambda doc: doc.update(chunks=[0])) + zmetadata = json.loads((path / ".zmetadata").read_text()) + zmetadata["metadata"]["a/.zarray"]["chunks"] = [0] + (path / ".zmetadata").write_text(json.dumps(zmetadata)) + else: + _rewrite_doc( + path / "a", 3, lambda doc: doc["chunk_grid"]["configuration"].update(chunk_shape=[0]) + ) + _rewrite_doc( + path, + 3, + lambda doc: doc["consolidated_metadata"]["metadata"]["a"]["chunk_grid"][ + "configuration" + ].update(chunk_shape=[0]), + ) + + with pytest.warns(ZarrUserWarning, match="zarr.consolidate_metadata"): + group = zarr.open_group(path, mode="r+") + array = group["a"] + assert isinstance(array, zarr.Array) + assert array.chunks == (1,) + + array.update_attributes({}) + zarr.consolidate_metadata(path) + with warnings.catch_warnings(): + warnings.simplefilter("error", ZarrUserWarning) + warnings.filterwarnings("ignore", "Consolidated metadata is currently not part") + for use_consolidated in (True, False): + reopened = zarr.open_group(path, mode="r", use_consolidated=use_consolidated)["a"] + assert isinstance(reopened, zarr.Array) + assert reopened.chunks == (1,) From f6f5c7b5cd97500987220f74bd20032b00e4e0f0 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Fri, 25 Sep 2026 22:22:48 +0200 Subject: [PATCH 15/41] docs: rewrite the 4334 changelog fragment and trim restating docstrings The fragment now covers the full-span shard rule, strict constructors, and the stored-document upgrade (including JSON true) in two short paragraphs. Assisted-by: ClaudeCode:claude-opus-5-5 Co-Authored-By: Claude Opus 5.5 --- changes/4334.bugfix.md | 4 ++-- src/zarr/core/chunk_grids.py | 6 ++---- tests/test_unified_chunk_grid.py | 14 ++------------ 3 files changed, 6 insertions(+), 18 deletions(-) diff --git a/changes/4334.bugfix.md b/changes/4334.bugfix.md index 68ae5ade9e..26228e668f 100644 --- a/changes/4334.bugfix.md +++ b/changes/4334.bugfix.md @@ -1,3 +1,3 @@ -Chunk-grid normalization now shares a helper that selects chunk size 1 for a zero-length axis when inferring a full-span chunk size. Chunk edge lengths remain positive while array extents may be zero. `FixedDimension(size=0, ...)` now raises `ValueError`. Rectilinear chunk grids can be created on a zero-length dimension with a non-empty list of positive chunk sizes; these sizes are retained for subsequent growth. +A chunk edge length is now always at least 1, while an array extent may be 0. Every spelling of one chunk spanning an axis (`chunks=-1`, `chunks=False`, `chunks="auto"`, `shards=-1`, `shards=False`) gives chunk size 1 on a zero-length axis, and a shard spanning an axis is a multiple of the inner chunk size, so `shards=-1` no longer fails when the axis length is not a multiple of it. Array metadata built in code with a chunk size of 0 or `True` now raises, as does `FixedDimension(size=0, ...)`. Rectilinear chunk grids can be created on a zero-length dimension with a non-empty list of positive chunk sizes, which are kept for later growth. -A stored chunk size of 0 is read as one chunk spanning the axis, with a `ZarrUserWarning`, in both Zarr format 2 (`chunks`) and Zarr format 3 (a regular grid's `chunk_shape`, including the JSON `false` written by zarr-python 3.0 for `chunks=False`). zarr-python wrote such metadata for arrays created with a zero-length axis until 3.4 — Zarr format 2 through 3.3 and by zarr-python 2.x, Zarr format 3 in 3.0 and 3.1. Those versions could then grow the axis without storing any chunk: appending to one of these Zarr format 2 arrays in 3.4.0 reported success while the appended data read back as the fill value, and 3.1 recorded a Zarr format 3 resize or failed append. Such grown arrays open too, and the warning says that data written to the grown axis was not saved. A negative chunk size is rejected. The warning explains how to store a corrected chunk size: open the array writable and call `array.update_attributes({})`. +Stored metadata with a regular chunk size of 0 or JSON `false`, as zarr-python wrote for arrays created with a zero-length axis until 3.4, now opens with a `ZarrUserWarning` and is read as one chunk spanning the axis (a multiple of the inner chunk size for sharded arrays, which previously failed to open), and JSON `true`, as zarr-python 3.0 and 3.2 wrote for `chunks=(True,)`, is read as 1. The warning says how to store valid metadata: `array.update_attributes({})`, then `zarr.consolidate_metadata` if the metadata is consolidated. diff --git a/src/zarr/core/chunk_grids.py b/src/zarr/core/chunk_grids.py index d5d61ee010..17b3225725 100644 --- a/src/zarr/core/chunk_grids.py +++ b/src/zarr/core/chunk_grids.py @@ -895,10 +895,8 @@ def _guess_num_chunks_per_axis_shard( In other words the shard would be a (2,2,2) grid of (2,2,2) chunks i.e., prod(chunk_shape) * (returned_val ** len(chunk_shape)) * item_size = 256 bytes. - Degenerate inputs — a 0-dimensional chunk shape, or a zero-byte chunk (``item_size`` - of 0; chunk edge lengths themselves are always at least 1) — return 1, as the - search loop's stopping conditions can never be met. A zero-length *array* axis - needs no special case: the array-bound check fails immediately for it. + Degenerate inputs — a 0-dimensional chunk shape, or a zero-byte chunk — return 1, + as the search loop's stopping conditions can never be met. Parameters ---------- diff --git a/tests/test_unified_chunk_grid.py b/tests/test_unified_chunk_grid.py index 2cf3f7cc10..6e2379e2fd 100644 --- a/tests/test_unified_chunk_grid.py +++ b/tests/test_unified_chunk_grid.py @@ -199,12 +199,7 @@ def test_fixed_dimension_indices_to_chunks() -> None: ids=["negative-size", "zero-size", "zero-size-zero-extent", "negative-extent"], ) def test_fixed_dimension_rejects_invalid(size: int, extent: int, match: str) -> None: - """FixedDimension raises ValueError for a size below 1 or a negative extent. - - A chunk edge length of 0 is never valid, whatever the extent: the metadata layer - requires every chunk edge length to be >= 1, and the in-memory model enforces the - same invariant so the two can never disagree. - """ + """FixedDimension raises ValueError for a size below 1 or a negative extent.""" with pytest.raises(ValueError, match=match): FixedDimension(size=size, extent=extent) @@ -1433,12 +1428,7 @@ def test_edge_case_chunk_grid_boundary_shape() -> None: @pytest.mark.parametrize("size", [1, 10], ids=["size-1", "size-10"]) def test_fixed_dimension_zero_extent(size: int) -> None: - """A zero-length axis has zero chunks and behaves like an empty grid. - - The extent may be 0 even though the chunk size may not: `ceildiv(0, size)` is 0, - so there is nothing to look up, and the vectorized index mapping of an empty index - array is empty. - """ + """A zero-length axis has zero chunks and behaves like an empty grid.""" d = FixedDimension(size=size, extent=0) assert d.nchunks == 0 assert d.ngridcells == 0 From f87335852b1677bf78bcbb70ad767a9e3112621a Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Fri, 25 Sep 2026 22:28:18 +0200 Subject: [PATCH 16/41] test: model exactly what the store holds in the array lifecycle state machine The model now tracks cells beyond the array's shape: a shrinking resize deletes exactly the chunks outside the new grid, kept chunks keep their out-of-bounds cells, and a write covering every in-bounds cell of an unsharded chunk resets its out-of-bounds cells to the fill value. A resize that never deletes chunks now fails the test. Sharding is its own chunk spelling, a stored chunk size of 0 applies to sharded arrays too (in the outer grid), re-saving metadata only runs when the store warns, and the test takes its settings from the repository's hypothesis profiles under the slow_hypothesis marker, like the other stateful tests. Assisted-by: ClaudeCode:claude-opus-5-5 Co-Authored-By: Claude Opus 5.5 --- tests/test_array_stateful.py | 201 +++++++++++++++++++---------------- 1 file changed, 111 insertions(+), 90 deletions(-) diff --git a/tests/test_array_stateful.py b/tests/test_array_stateful.py index a5a6627a5d..6e9bea8f8b 100644 --- a/tests/test_array_stateful.py +++ b/tests/test_array_stateful.py @@ -1,18 +1,22 @@ """A stateful test of one array's life: create, append, resize, write, reopen. -The model is a NumPy array, not a second zarr array, so a bug in zarr's chunk -grid logic cannot hide by being made on both sides. Zero-length axes are drawn -on purpose, both at creation and by resizing and appending, and so are the -stored chunk sizes of 0 that zarr-python wrote for empty arrays before 3.4. - -`resize` deletes only the chunks that fall entirely outside the new shape, so -cells cut off by a shrink can come back with their old values when the axis -grows again. The model does not encode that chunk-level behaviour: a cell cut -off and brought back is unknown until it is written. +The model is a NumPy array of what the store holds, not a second zarr array, so a bug +in zarr's chunk grid logic cannot hide by being made on both sides. Zero-length axes +are drawn on purpose, both at creation and by resizing and appending, and so are +stored chunk sizes of 0. + +The model tracks cells beyond the array's shape too, because chunks do: a shrinking +`resize` deletes exactly the chunks outside the new grid, a chunk it keeps keeps its +cells beyond the new shape (and they come back if the array grows again), and a write +that covers every in-bounds cell of an unsharded chunk rewrites the whole chunk, +resetting its cells beyond the shape to the fill value. A sharded array rewrites a +shard through its inner chunks, each judged against the shard rather than the array +shape, so a write never resets cells beyond the shape. """ from __future__ import annotations +import itertools import json import warnings from typing import Any, Literal @@ -21,30 +25,37 @@ import hypothesis.strategies as st import numpy as np import pytest -from hypothesis import event, note, settings +from hypothesis import event, note from hypothesis.stateful import ( RuleBasedStateMachine, initialize, invariant, + precondition, rule, ) import zarr from zarr.core.buffer import cpu, default_buffer_prototype +from zarr.core.chunk_grids import ChunkGrid from zarr.core.sync import sync from zarr.errors import ZarrUserWarning from zarr.storage import MemoryStore -pytestmark = pytest.mark.filterwarnings( - "ignore::zarr.core.dtype.common.UnstableSpecificationWarning" -) +pytestmark = [ + pytest.mark.slow_hypothesis, + pytest.mark.filterwarnings("ignore::zarr.core.dtype.common.UnstableSpecificationWarning"), +] DTYPE = np.dtype("int16") MAX_SIDE = 6 def _rectilinear_dim(extent: int) -> st.SearchStrategy[int | list[int]]: - """A bare step, or an edge list covering `extent` (any edges for extent 0).""" + """A bare step, or an edge list covering `extent` (any edges for extent 0). + + A small local copy of what `zarr.testing.strategies` draws for rectilinear + declarations, so this test does not depend on that module's experimental API. + """ steps = st.integers(min_value=1, max_value=MAX_SIDE) if extent == 0: return steps | st.lists(steps, min_size=1, max_size=3) @@ -64,39 +75,33 @@ def __init__(self) -> None: self._rectilinear.__enter__() self.store = MemoryStore() self.path = "a" - self.model: np.ndarray[Any, np.dtype[np.int16]] = np.zeros((0,), dtype=DTYPE) - # Cells whose value the model knows; see the module docstring. - self.known: np.ndarray[Any, np.dtype[np.bool_]] = np.ones((0,), dtype=bool) - # Every shape the array has had, to find cells a resize brings back. - self.past_shapes: list[tuple[int, ...]] = [] + self.shape: tuple[int, ...] = (0,) self.fill = 0 - # A legacy store warns until its metadata is re-saved. + # What the store holds, indexed like the array and extending past its shape. + self.stored: np.ndarray[Any, np.dtype[np.int16]] = np.zeros((0,), dtype=DTYPE) + # A store with an invalid stored chunk size warns until its metadata is re-saved. self.expect_open_warning = False # -------------------------------------------------------------- creation @initialize(data=st.data()) def create(self, data: st.DataObject) -> None: - zarr_format: Literal[2, 3] = data.draw(st.sampled_from([2, 3]), label="zarr_format") + zarr_format: Literal[2, 3] = data.draw(st.sampled_from([3, 2]), label="zarr_format") shape = data.draw( npst.array_shapes(min_dims=1, max_dims=3, min_side=0, max_side=MAX_SIDE), label="shape", ) self.fill = data.draw(st.integers(-3, 3), label="fill_value") # sampled_from favours early entries; the less common spellings go first. - spellings = ["ints", "legacy-zero", "-1", "False", "auto"] + spellings = ["ints", "-1", "False", "auto"] if zarr_format == 3: - spellings.insert(1, "rectilinear") + spellings[:0] = ["sharded", "rectilinear"] spelling = data.draw(st.sampled_from(spellings), label="chunk spelling") event(f"chunks: {spelling}") - if any(s == 0 for s in shape): - event("created with a zero-length axis") chunks: Any shards: Any = None - if spelling == "-1": - chunks = -1 - elif spelling == "False": - chunks = False + if spelling in ("-1", "False"): + chunks = {"-1": -1, "False": False}[spelling] elif spelling == "auto": chunks = "auto" elif spelling == "rectilinear": @@ -104,14 +109,9 @@ def create(self, data: st.DataObject) -> None: if not any(isinstance(c, list) for c in chunks): chunks[0] = [chunks[0]] if shape[0] == 0 else [shape[0]] else: - chunks = tuple(data.draw(st.integers(1, 4)) for _ in shape) - if ( - spelling == "ints" - and zarr_format == 3 - and data.draw(st.booleans(), label="sharded") - ): - shards = tuple(c * data.draw(st.integers(1, 2)) for c in chunks) - event("sharded") + chunks = tuple(data.draw(st.integers(1, 3)) for _ in shape) + if spelling == "sharded": + shards = tuple(c * data.draw(st.integers(1, 3)) for c in chunks) note(f"create {shape=} {chunks=} {shards=} {zarr_format=} fill={self.fill}") zarr.create_array( self.store, @@ -123,14 +123,13 @@ def create(self, data: st.DataObject) -> None: fill_value=self.fill, zarr_format=zarr_format, ) - self.model = np.full(shape, self.fill, dtype=DTYPE) - self.known = np.ones(shape, dtype=bool) - self.past_shapes = [shape] - - if spelling == "legacy-zero": - # What zarr-python wrote before 3.4 for an array created with a - # zero-length axis and one chunk spanning it; older releases could - # grow that axis without storing a chunk, so any extent is possible. + self.shape = shape + self.stored = np.full(shape, self.fill, dtype=DTYPE) + + if spelling != "rectilinear" and data.draw(st.booleans(), label="legacy zero"): + # A stored chunk size of 0, as zarr-python wrote for arrays created with a + # zero-length axis; older releases could then grow the axis without storing + # a chunk, so any extent is possible. Sharded arrays store it in the outer grid. zero_axes = data.draw( st.lists(st.integers(0, len(shape) - 1), min_size=1, unique=True), label="axes stored with chunk size 0", @@ -138,8 +137,7 @@ def create(self, data: st.DataObject) -> None: stored_zero = data.draw(st.sampled_from([0, False]), label="stored zero") self._rewrite_stored_chunks(zarr_format, zero_axes, stored_zero) self.expect_open_warning = True - if any(shape[i] > 0 for i in zero_axes): - event("legacy zero chunk on a grown axis") + event("legacy zero chunk size") def _rewrite_stored_chunks( self, zarr_format: Literal[2, 3], axes: list[int], value: Any @@ -163,26 +161,68 @@ def _open(self) -> zarr.Array[Any]: assert warned is self.expect_open_warning, [str(w.message) for w in record] return arr + # ----------------------------------------------------------------- model + def _cover(self, shape: tuple[int, ...]) -> None: + """Grow `stored` with the fill value so that it covers `shape`.""" + pad = [(0, max(0, n - s)) for n, s in zip(shape, self.stored.shape, strict=True)] + if any(after for _, after in pad): + self.stored = np.pad(self.stored, pad, constant_values=self.fill) + + def _model_resize(self, grid: ChunkGrid, new_shape: tuple[int, ...]) -> None: + """Delete exactly the chunks of `grid` outside the grid for `new_shape`.""" + kept = [] + for dim, old, new in zip(grid.dimensions, self.shape, new_shape, strict=True): + if new >= old: + kept.append(slice(None)) + elif new == 0: + kept.append(slice(0, 0)) + else: + last = dim.index_to_chunk(new - 1) + kept.append(slice(0, dim.chunk_offset(last) + dim.chunk_size(last))) + stored = np.full_like(self.stored, self.fill) + stored[tuple(kept)] = self.stored[tuple(kept)] + self.stored = stored + self.shape = new_shape + self._cover(new_shape) + + def _model_write(self, arr: zarr.Array[Any], region: tuple[slice, ...], values: Any) -> None: + """Write `values`; an unsharded chunk whose in-bounds cells are all written is + rewritten whole.""" + grid = ChunkGrid.from_metadata(arr.metadata) + if arr.shards is None and all(r.stop > r.start for r in region): + # Per axis, each chunk the region touches: (start, stop, all in-bounds cells written). + per_axis: list[list[tuple[int, int, bool]]] = [] + for dim, r, extent in zip(grid.dimensions, region, self.shape, strict=True): + spans = [] + for c in range(dim.index_to_chunk(r.start), dim.index_to_chunk(r.stop - 1) + 1): + lo = dim.chunk_offset(c) + hi = lo + dim.chunk_size(c) + spans.append((lo, hi, r.start <= lo and r.stop >= min(hi, extent))) + per_axis.append(spans) + self._cover(tuple(max(hi for _, hi, _ in axis) for axis in per_axis)) + for combo in itertools.product(*per_axis): + if all(complete for *_, complete in combo): + self.stored[tuple(slice(lo, hi) for lo, hi, _ in combo)] = self.fill + self.stored[region] = values + # ----------------------------------------------------------------- rules @rule(data=st.data()) def append(self, data: st.DataObject) -> None: arr = self._open() - axis = data.draw(st.integers(0, self.model.ndim - 1), label="axis") - block_shape = list(self.model.shape) + axis = data.draw(st.integers(0, len(self.shape) - 1), label="axis") + block_shape = list(self.shape) block_shape[axis] = data.draw(st.integers(0, 4), label="rows") block = data.draw(npst.arrays(DTYPE, tuple(block_shape)), label="block") - note(f"append {block.shape} along {axis} to {self.model.shape}") - if self.model.shape[axis] == 0: + note(f"append {block.shape} along {axis} to {self.shape}") + if self.shape[axis] == 0 and block.shape[axis]: event("append to a zero-length axis") + old_extent = self.shape[axis] arr.append(block, axis=axis) - self._reshape_model(arr.shape) - tail = tuple( - slice(-block.shape[axis], None) if i == axis and block.shape[axis] else slice(None) - for i in range(self.model.ndim) + self._model_resize(ChunkGrid.from_metadata(arr.metadata), arr.shape) + region = tuple( + slice(old_extent, s) if i == axis else slice(0, s) for i, s in enumerate(arr.shape) ) - if block.shape[axis]: - self.model[tail] = block - self.known[tail] = True + self._model_write(arr, region, block) # Growing the array rewrites its metadata, which stores any correction. self.expect_open_warning = False @@ -190,49 +230,31 @@ def append(self, data: st.DataObject) -> None: def resize(self, data: st.DataObject) -> None: arr = self._open() new_shape = data.draw( - st.tuples(*(st.integers(0, MAX_SIDE) for _ in self.model.shape)), label="new shape" + st.tuples(*(st.integers(0, MAX_SIDE) for _ in self.shape)), label="new shape" ) - note(f"resize {self.model.shape} -> {new_shape}") - if any(o == 0 and n > 0 for o, n in zip(self.model.shape, new_shape, strict=True)): - event("resize grows a zero-length axis") + note(f"resize {self.shape} -> {new_shape}") + grid = ChunkGrid.from_metadata(arr.metadata) arr.resize(new_shape) - self._reshape_model(new_shape) + self._model_resize(grid, new_shape) self.expect_open_warning = False - def _reshape_model(self, new_shape: tuple[int, ...]) -> None: - """Resize the model: kept cells keep their values, new cells hold the fill - value, and cells that an earlier shape held but the current one cut off - become unknown.""" - overlap = tuple( - slice(0, min(o, n)) for o, n in zip(self.model.shape, new_shape, strict=True) - ) - model = np.full(new_shape, self.fill, dtype=DTYPE) - known = np.ones(new_shape, dtype=bool) - for past in self.past_shapes: - known[tuple(slice(0, min(p, n)) for p, n in zip(past, new_shape, strict=True))] = False - model[overlap] = self.model[overlap] - known[overlap] = self.known[overlap] - if not known.all(): - event("resize brings back cells cut off earlier") - self.model, self.known = model, known - self.past_shapes.append(tuple(new_shape)) - @rule(data=st.data()) def write(self, data: st.DataObject) -> None: arr = self._open() region = tuple( slice(*sorted(data.draw(st.tuples(st.integers(0, s), st.integers(0, s))))) - for s in self.model.shape + for s in self.shape ) - values = data.draw(npst.arrays(DTYPE, self.model[region].shape), label="values") + shape = tuple(r.stop - r.start for r in region) + values = data.draw(npst.arrays(DTYPE, shape), label="values") note(f"write {region}") arr[region] = values - self.model[region] = values - self.known[region] = True + self._model_write(arr, region, values) + @precondition(lambda self: self.expect_open_warning) @rule() def resave_metadata(self) -> None: - """What the legacy warning tells users to do.""" + """What the warning for an invalid stored chunk size tells users to do.""" self._open().update_attributes({}) self.expect_open_warning = False @@ -243,10 +265,9 @@ def teardown(self) -> None: @invariant() def matches_model(self) -> None: arr = self._open() - assert arr.shape == self.model.shape - actual = np.asarray(arr[...]) - np.testing.assert_array_equal(actual[self.known], self.model[self.known]) + assert arr.shape == self.shape + expected = self.stored[tuple(slice(0, s) for s in self.shape)] + np.testing.assert_array_equal(np.asarray(arr[...]), expected) -ArrayLifecycle.TestCase.settings = settings(max_examples=200, stateful_step_count=12, deadline=None) TestArrayLifecycle = ArrayLifecycle.TestCase From d9549b8710d638e96bd2d6fda9345b54feee63e4 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Fri, 25 Sep 2026 22:46:45 +0200 Subject: [PATCH 17/41] refactor(metadata): warn about an upgraded document only once it validates An upgrade now returns the upgraded document and how it read it, and `from_dict` warns with those readings after the metadata constructor accepts the upgraded document. A stored document that is still invalid after an upgrade raises its own error instead of first warning how it was read. Assisted-by: ClaudeCode:claude-opus-5-5 Co-Authored-By: Claude Opus 5.5 --- src/zarr/core/metadata/upgrades.py | 76 ++++++++++++++++++---------- src/zarr/core/metadata/v2.py | 9 ++-- src/zarr/core/metadata/v3.py | 13 +++-- tests/test_metadata/test_upgrades.py | 30 ++++++++--- 4 files changed, 88 insertions(+), 40 deletions(-) diff --git a/src/zarr/core/metadata/upgrades.py b/src/zarr/core/metadata/upgrades.py index 933b4c53e2..e15ffd30fe 100644 --- a/src/zarr/core/metadata/upgrades.py +++ b/src/zarr/core/metadata/upgrades.py @@ -2,9 +2,11 @@ This is the only place invalid metadata is read leniently; the metadata constructors are strict. An upgrade maps a stored array metadata document (parsed JSON) to a valid -one and warns once when it changes anything. `ArrayV2Metadata.from_dict` and +one and says how it read the document. `ArrayV2Metadata.from_dict` and `ArrayV3Metadata.from_dict` apply the upgrades for their Zarr format, so every path -that parses a stored document, including consolidated metadata, goes through them. +that parses a stored document, including consolidated metadata, goes through them, and +warn with each reading once the upgraded document has passed the metadata constructor. +An invalid document therefore raises its own error, not a warning about how it was read. To read another kind of invalid document, add an upgrade to `V2_ARRAY_UPGRADES` or `V3_ARRAY_UPGRADES`. @@ -14,7 +16,7 @@ import json import warnings -from collections.abc import Callable, Mapping, Sequence +from collections.abc import Callable, Iterable, Mapping, Sequence from typing import TYPE_CHECKING, Final from typing_extensions import TypeIs @@ -26,7 +28,9 @@ from zarr.core.common import JSON type ArrayDocument = Mapping[str, JSON] -type Upgrade = Callable[[ArrayDocument], ArrayDocument] +type Upgrade = Callable[[ArrayDocument], tuple[ArrayDocument, str] | None] +"""Returns `None` if the document needs no upgrade, else the upgraded document and a +sentence saying how it was read.""" RESAVE_HINT: Final = ( "To store valid metadata, open the array writable and call `array.update_attributes({})`; " @@ -35,10 +39,12 @@ ) -def _warn(message: str) -> None: - # The synchronous API parses metadata on zarr's IO thread, whose stack holds no - # user code, so the warning points at the upgrade on every path. - warnings.warn(f"{message} {RESAVE_HINT}", ZarrUserWarning, stacklevel=2) +def warn_readings(readings: Iterable[str]) -> None: + """Warn once for each reading returned by `upgrade_array_document`.""" + for reading in readings: + # The synchronous API parses metadata on zarr's IO thread, whose stack holds no + # user code, so the warning points at the `from_dict` that read the document. + warnings.warn(f"{reading} {RESAVE_HINT}", ZarrUserWarning, stacklevel=2) def _is_int_list(value: object) -> TypeIs[list[int] | tuple[int, ...]]: @@ -46,12 +52,15 @@ def _is_int_list(value: object) -> TypeIs[list[int] | tuple[int, ...]]: return isinstance(value, list | tuple) and all(isinstance(v, int) for v in value) -def _read_invalid_chunk_sizes(chunk_shape: JSON, shape: JSON, unit: JSON) -> list[int] | None: +def _read_invalid_chunk_sizes( + chunk_shape: JSON, shape: JSON, unit: JSON +) -> tuple[list[int], str] | None: """Read a regular chunk shape whose chunk sizes include 0, JSON `false` or JSON `true`. 0 and `false` are read as one chunk spanning the axis, a multiple of `unit` on that - axis (the inner chunk shape of a sharded array); `true` is read as 1. Returns `None` - when there is nothing to read this way: anything else that is invalid is left for + axis (the inner chunk shape of a sharded array); `true` is read as 1. Returns the + upgraded chunk shape and how it was read, or `None` when there is nothing to read + this way: anything else that is invalid is left for the metadata constructors to reject. """ if not (_is_int_list(chunk_shape) and _is_int_list(shape)) or len(chunk_shape) != len(shape): @@ -84,13 +93,15 @@ def _read_invalid_chunk_sizes(chunk_shape: JSON, shape: JSON, unit: JSON) -> lis " No chunk can be stored under a chunk size of 0, so the array holds only its " "fill value." ) - _warn(message) - return upgraded + return upgraded, message -def _invalid_chunk_sizes_v2(doc: ArrayDocument) -> ArrayDocument: - chunks = _read_invalid_chunk_sizes(doc.get("chunks"), doc.get("shape"), None) - return doc if chunks is None else {**doc, "chunks": chunks} +def _invalid_chunk_sizes_v2(doc: ArrayDocument) -> tuple[ArrayDocument, str] | None: + read = _read_invalid_chunk_sizes(doc.get("chunks"), doc.get("shape"), None) + if read is None: + return None + chunks, reading = read + return {**doc, "chunks": chunks}, reading def _sharding_chunk_shape(codecs: JSON) -> JSON: @@ -104,32 +115,43 @@ def _sharding_chunk_shape(codecs: JSON) -> JSON: return None -def _invalid_chunk_sizes_v3(doc: ArrayDocument) -> ArrayDocument: +def _invalid_chunk_sizes_v3(doc: ArrayDocument) -> tuple[ArrayDocument, str] | None: grid = doc.get("chunk_grid") if not isinstance(grid, Mapping) or grid.get("name") != "regular": - return doc + return None configuration = grid.get("configuration") if not isinstance(configuration, Mapping): - return doc - chunk_shape = _read_invalid_chunk_sizes( + return None + read = _read_invalid_chunk_sizes( configuration.get("chunk_shape"), doc.get("shape"), _sharding_chunk_shape(doc.get("codecs")), ) - if chunk_shape is None: - return doc + if read is None: + return None + chunk_shape, reading = read return { **doc, "chunk_grid": {**grid, "configuration": {**configuration, "chunk_shape": chunk_shape}}, - } + }, reading V2_ARRAY_UPGRADES: Final[tuple[Upgrade, ...]] = (_invalid_chunk_sizes_v2,) V3_ARRAY_UPGRADES: Final[tuple[Upgrade, ...]] = (_invalid_chunk_sizes_v3,) -def upgrade_array_document(doc: ArrayDocument, upgrades: Sequence[Upgrade]) -> ArrayDocument: - """Apply `upgrades` to a stored array metadata document, in order.""" +def upgrade_array_document( + doc: ArrayDocument, upgrades: Sequence[Upgrade] +) -> tuple[ArrayDocument, list[str]]: + """Apply `upgrades` to a stored array metadata document, in order. + + Returns the upgraded document and the readings of the upgrades that changed it, + for `warn_readings` once the document has been validated. + """ + readings: list[str] = [] for upgrade in upgrades: - doc = upgrade(doc) - return doc + upgraded = upgrade(doc) + if upgraded is not None: + doc, reading = upgraded + readings.append(reading) + return doc, readings diff --git a/src/zarr/core/metadata/v2.py b/src/zarr/core/metadata/v2.py index 0eb28efb11..c648f271e6 100644 --- a/src/zarr/core/metadata/v2.py +++ b/src/zarr/core/metadata/v2.py @@ -43,7 +43,7 @@ from zarr.core.config import config, parse_indexing_order from zarr.core.json_parse import parse_field from zarr.core.metadata.common import parse_attributes, parse_chunk_edge -from zarr.core.metadata.upgrades import V2_ARRAY_UPGRADES, upgrade_array_document +from zarr.core.metadata.upgrades import V2_ARRAY_UPGRADES, upgrade_array_document, warn_readings class ArrayV2MetadataDict(TypedDict): @@ -148,7 +148,8 @@ def to_buffer_dict(self, prototype: BufferPrototype) -> dict[str, Buffer]: @classmethod def from_dict(cls, data: dict[str, Any]) -> ArrayV2Metadata: - data = dict(upgrade_array_document(data, V2_ARRAY_UPGRADES)) + upgraded, readings = upgrade_array_document(data, V2_ARRAY_UPGRADES) + data = dict(upgraded) _data = data.copy() # Check that the zarr_format attribute is correct. _ = parse_zarr_format(_data.pop("zarr_format")) @@ -200,7 +201,9 @@ def from_dict(cls, data: dict[str, Any]) -> ArrayV2Metadata: _data = {k: v for k, v in _data.items() if k in expected} - return cls(**_data) + metadata = cls(**_data) + warn_readings(readings) + return metadata def to_dict(self) -> dict[str, JSON]: zarray_dict = super().to_dict() diff --git a/src/zarr/core/metadata/v3.py b/src/zarr/core/metadata/v3.py index 4a27d495b3..b6929220ce 100644 --- a/src/zarr/core/metadata/v3.py +++ b/src/zarr/core/metadata/v3.py @@ -36,7 +36,11 @@ from zarr.core.dtype.common import check_dtype_spec_v3 from zarr.core.json_parse import parse_field from zarr.core.metadata.common import parse_attributes, parse_chunk_edge -from zarr.core.metadata.upgrades import V3_ARRAY_UPGRADES, upgrade_array_document +from zarr.core.metadata.upgrades import ( + V3_ARRAY_UPGRADES, + upgrade_array_document, + warn_readings, +) from zarr.errors import MetadataValidationError, NodeTypeValidationError from zarr.registry import get_codec_class @@ -628,8 +632,9 @@ def to_buffer_dict(self, prototype: BufferPrototype) -> dict[str, Buffer]: @classmethod def from_dict(cls, data: dict[str, JSON]) -> Self: + upgraded, readings = upgrade_array_document(data, V3_ARRAY_UPGRADES) # a new dict, because we are modifying it - _data = dict(upgrade_array_document(data, V3_ARRAY_UPGRADES)) + _data = dict(upgraded) # check that the zarr_format attribute is correct _ = parse_zarr_format(_data.pop("zarr_format")) @@ -669,7 +674,7 @@ def from_dict(cls, data: dict[str, JSON]) -> Self: # TODO: replace this with a real type check! _data_typed = cast(ArrayMetadataJSON_V3, _data) - return cls( + metadata = cls( shape=_data_typed["shape"], chunk_grid=_data_typed["chunk_grid"], # type: ignore[arg-type] chunk_key_encoding=_data_typed["chunk_key_encoding"], # type: ignore[arg-type] @@ -683,6 +688,8 @@ def from_dict(cls, data: dict[str, JSON]) -> Self: extra_fields=allowed_extra_fields, storage_transformers=_data_typed.get("storage_transformers", ()), # type: ignore[arg-type] ) + warn_readings(readings) + return metadata def to_dict(self) -> dict[str, JSON]: out_dict = super().to_dict() diff --git a/tests/test_metadata/test_upgrades.py b/tests/test_metadata/test_upgrades.py index 7506bb6875..a82a8a7717 100644 --- a/tests/test_metadata/test_upgrades.py +++ b/tests/test_metadata/test_upgrades.py @@ -113,28 +113,44 @@ def test_upgrade_array_document( ) -> None: """Valid documents pass unchanged and silently. A stored chunk size of 0 or `false` is read as one chunk spanning the axis (a multiple of the inner chunk when sharded) - and `true` as 1, with one warning that says how to re-save.""" + and `true` as 1; `from_dict` warns once with the reading and how to re-save.""" upgrades = V2_ARRAY_UPGRADES if doc["zarr_format"] == 2 else V3_ARRAY_UPGRADES - with warnings.catch_warnings(record=True) as record: - warnings.simplefilter("always") - upgraded = upgrade_array_document(doc, upgrades) + upgraded, readings = upgrade_array_document(doc, upgrades) assert _stored_chunks(dict(upgraded)) == expected assert {k: v for k, v in upgraded.items() if k not in ("chunks", "chunk_grid")} == { k: v for k, v in doc.items() if k not in ("chunks", "chunk_grid") } + metadata_cls = ArrayV2Metadata if doc["zarr_format"] == 2 else ArrayV3Metadata + with warnings.catch_warnings(record=True) as record: + warnings.simplefilter("always") + metadata_cls.from_dict(dict(doc)) messages = [str(w.message) for w in record] if warning is None: assert upgraded is doc + assert readings == [] assert messages == [] else: + assert len(readings) == 1 + assert re.search(warning, readings[0]) assert len(messages) == 1 - assert re.search(warning, messages[0]) + assert messages[0].startswith(readings[0]) assert "update_attributes({})" in messages[0] assert "zarr.consolidate_metadata" in messages[0] + + +@pytest.mark.parametrize( + "doc", + [_v2_doc([4], [0]) | {"dtype": " None: + """A document that is invalid after its upgrade raises its own error, without first + warning how it was read.""" metadata_cls = ArrayV2Metadata if doc["zarr_format"] == 2 else ArrayV3Metadata with warnings.catch_warnings(): - warnings.simplefilter("ignore", ZarrUserWarning) - metadata_cls.from_dict(dict(doc)) + warnings.simplefilter("error", ZarrUserWarning) + with pytest.raises(ValueError): + metadata_cls.from_dict(doc) @pytest.mark.parametrize("doc", [_v2_doc([4], [-1]), _v3_doc([4], [-1])], ids=["v2", "v3"]) From 5ef0f51f787fc523332d99a4c2d4c8cf4dc65d2c Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Sat, 26 Sep 2026 02:08:34 +0200 Subject: [PATCH 18/41] refactor(metadata): one per-axis rule for stored chunk sizes, one rule for edges Stored regular chunk shapes, in both formats and in a sharding codec's inner chunk shape, are read entry by entry by one rule in upgrades.py: an int >= 1 is kept, `true` is read as 1 (zarr 3.0.10 also wrote it as the inner chunk size of a shard, which the sharding codec used to read silently), and 0 or `false` as one chunk spanning the axis. Integral floats are not read: no release wrote a regular grid with them. A document warns once, naming the array where the caller knows its path. Metadata constructors check every chunk edge length with one function, `parse_chunk_edge`, which tells a wrong type (TypeError) from a wrong value (ValueError); this covers bare sizes, rectilinear edges, RLE sizes and the sharding codec's inner chunk shape. Assisted-by: ClaudeCode:claude-opus-5-5 Co-Authored-By: Claude Opus 5.5 --- changes/4334.bugfix.md | 6 +- src/zarr/codecs/sharding.py | 9 +- src/zarr/core/array.py | 8 +- src/zarr/core/common.py | 46 +++-- src/zarr/core/group.py | 20 +- src/zarr/core/metadata/common.py | 9 - src/zarr/core/metadata/upgrades.py | 185 ++++++++++------- src/zarr/core/metadata/v2.py | 13 +- src/zarr/core/metadata/v3.py | 39 ++-- tests/test_metadata/test_upgrades.py | 292 ++++++++++++++++++--------- tests/test_unified_chunk_grid.py | 26 +-- 11 files changed, 396 insertions(+), 257 deletions(-) diff --git a/changes/4334.bugfix.md b/changes/4334.bugfix.md index 26228e668f..77c211643e 100644 --- a/changes/4334.bugfix.md +++ b/changes/4334.bugfix.md @@ -1,3 +1,5 @@ -A chunk edge length is now always at least 1, while an array extent may be 0. Every spelling of one chunk spanning an axis (`chunks=-1`, `chunks=False`, `chunks="auto"`, `shards=-1`, `shards=False`) gives chunk size 1 on a zero-length axis, and a shard spanning an axis is a multiple of the inner chunk size, so `shards=-1` no longer fails when the axis length is not a multiple of it. Array metadata built in code with a chunk size of 0 or `True` now raises, as does `FixedDimension(size=0, ...)`. Rectilinear chunk grids can be created on a zero-length dimension with a non-empty list of positive chunk sizes, which are kept for later growth. +A chunk edge length is now always at least 1, while an array extent may be 0. Every spelling of one chunk spanning an axis (`chunks=-1`, `chunks=False`, `chunks="auto"`, `shards=-1`, `shards=False`) gives chunk size 1 on a zero-length axis, and a shard spanning an axis is a multiple of the inner chunk size, so `shards=-1` no longer fails when the axis length is not a multiple of it. Rectilinear chunk grids can be created on a zero-length dimension with a non-empty list of positive chunk sizes, which are kept for later growth. -Stored metadata with a regular chunk size of 0 or JSON `false`, as zarr-python wrote for arrays created with a zero-length axis until 3.4, now opens with a `ZarrUserWarning` and is read as one chunk spanning the axis (a multiple of the inner chunk size for sharded arrays, which previously failed to open), and JSON `true`, as zarr-python 3.0 and 3.2 wrote for `chunks=(True,)`, is read as 1. The warning says how to store valid metadata: `array.update_attributes({})`, then `zarr.consolidate_metadata` if the metadata is consolidated. +Metadata built in code is strict about chunk edge lengths: `ArrayV2Metadata`, `RegularChunkGridMetadata`, `RectilinearChunkGridMetadata` and `ShardingCodec` take them as Python `int`s of at least 1, so a size of 0 now raises a `ValueError`, and a `bool`, a float or a NumPy integer (including a scalar `chunks=np.int64(5)` for `ArrayV2Metadata`) raises a `TypeError`; `FixedDimension(size=0, ...)` raises a `ValueError`. Stored rectilinear chunk grids with integral floats (`10.0`) are rejected too. The array creation functions still accept NumPy integers in `chunks=` and `shards=`. + +Stored metadata with a regular chunk size of 0 or JSON `false`, as zarr-python wrote for arrays created with a zero-length axis until 3.4, now opens with a `ZarrUserWarning` and is read as one chunk spanning the axis (a multiple of the inner chunk size for sharded arrays, which previously failed to open). JSON `true`, as zarr-python 3.0 and 3.2 wrote for `chunks=(True,)`, is read as 1, in the chunk grid and in the inner chunk shape of a sharding codec. The warning names the array and says how to store valid metadata: `array.update_attributes({})`, then `zarr.consolidate_metadata` if the metadata is consolidated. diff --git a/src/zarr/codecs/sharding.py b/src/zarr/codecs/sharding.py index 7a053867d6..37fe52c1a6 100644 --- a/src/zarr/codecs/sharding.py +++ b/src/zarr/codecs/sharding.py @@ -45,9 +45,8 @@ merge_and_encode_chunk, ) from zarr.core.common import ( - ShapeLike, + parse_chunk_shape, parse_named_configuration, - parse_shapelike, product, ) from zarr.core.config import config as zarr_config @@ -459,13 +458,13 @@ class ShardingCodec( def __init__( self, *, - chunk_shape: ShapeLike, + chunk_shape: Iterable[int], codecs: Iterable[Codec | dict[str, JSON]] = (BytesCodec(),), index_codecs: Iterable[Codec | dict[str, JSON]] = (BytesCodec(), Crc32cCodec()), index_location: ShardingCodecIndexLocation | IndexLocation = "end", subchunk_write_order: SubchunkWriteOrder = "morton", ) -> None: - chunk_shape_parsed = parse_shapelike(chunk_shape) + chunk_shape_parsed = parse_chunk_shape(chunk_shape) codecs_parsed = parse_codecs(codecs) index_codecs_parsed = parse_codecs(index_codecs) _check_index_codecs_fixed_size(index_codecs_parsed) @@ -504,7 +503,7 @@ def __getstate__(self) -> dict[str, Any]: def __setstate__(self, state: dict[str, Any]) -> None: config = state["configuration"] - object.__setattr__(self, "chunk_shape", parse_shapelike(config["chunk_shape"])) + object.__setattr__(self, "chunk_shape", parse_chunk_shape(config["chunk_shape"])) object.__setattr__(self, "codecs", parse_codecs(config["codecs"])) object.__setattr__(self, "index_codecs", parse_codecs(config["index_codecs"])) object.__setattr__(self, "index_location", _parse_index_location(config["index_location"])) diff --git a/src/zarr/core/array.py b/src/zarr/core/array.py index 5a8d6bf57e..e39080b668 100644 --- a/src/zarr/core/array.py +++ b/src/zarr/core/array.py @@ -201,13 +201,13 @@ def _chunk_sizes_from_shape( return tuple(result) -def parse_array_metadata(data: Any) -> ArrayMetadata: +def parse_array_metadata(data: Any, path: str | None = None) -> ArrayMetadata: if isinstance(data, ArrayMetadata): return data elif isinstance(data, dict): zarr_format = data.get("zarr_format") if zarr_format == 3: - meta_out = ArrayV3Metadata.from_dict(data) + meta_out = ArrayV3Metadata.from_dict(data, path=path) if len(meta_out.storage_transformers) > 0: msg = ( f"Array metadata contains storage transformers: {meta_out.storage_transformers}." @@ -216,7 +216,7 @@ def parse_array_metadata(data: Any) -> ArrayMetadata: raise ValueError(msg) return meta_out elif zarr_format == 2: - return ArrayV2Metadata.from_dict(data) + return ArrayV2Metadata.from_dict(data, path=path) else: raise ValueError(f"Invalid zarr_format: {zarr_format}. Expected 2 or 3") raise TypeError # pragma: no cover @@ -404,7 +404,7 @@ def __init__( store_path: StorePath, config: ArrayConfigLike | None = None, ) -> None: - metadata_parsed = parse_array_metadata(metadata) + metadata_parsed = parse_array_metadata(metadata, str(store_path)) config_parsed = parse_array_config(config) object.__setattr__(self, "metadata", metadata_parsed) diff --git a/src/zarr/core/common.py b/src/zarr/core/common.py index ba01f6c19f..cddbf91912 100644 --- a/src/zarr/core/common.py +++ b/src/zarr/core/common.py @@ -276,7 +276,32 @@ def _default_zarr_format() -> ZarrFormat: return cast("ZarrFormat", int(zarr_config.get("default_zarr_format", 3))) -def expand_rle(data: Sequence[int | list[int]]) -> list[int]: +def _parse_positive_int(value: object, name: str) -> int: + if isinstance(value, bool) or not isinstance(value, int): + raise TypeError(f"{name} must be an int, got {value!r}") + if value < 1: + raise ValueError(f"{name} must be >= 1, got {value!r}") + return value + + +def parse_chunk_edge(size: object, axis: int | None = None) -> int: + """Check that `size` is a chunk edge length: an `int` (not a `bool`) of at least 1. + + This is the one rule for chunk edge lengths in metadata: bare chunk sizes, explicit + edges and run-length encoded sizes. `axis`, when given, is named in the error. + """ + where = "" if axis is None else f"Dimension {axis}: " + return _parse_positive_int(size, f"{where}Chunk edge length") + + +def parse_chunk_shape(data: object) -> tuple[int, ...]: + """Check a regular chunk shape: one chunk edge length per axis (see `parse_chunk_edge`).""" + if not isinstance(data, Iterable): + raise TypeError(f"A chunk shape must be a sequence of chunk edge lengths, got {data!r}") + return tuple(parse_chunk_edge(size, axis) for axis, size in enumerate(data)) + + +def expand_rle(data: Sequence[object]) -> list[int]: """Expand a mixed array of bare integers and RLE pairs. Per the rectilinear chunk grid spec, each element can be: @@ -285,20 +310,13 @@ def expand_rle(data: Sequence[int | list[int]]) -> list[int]: """ result: list[int] = [] for item in data: - if isinstance(item, (int, float)) and not isinstance(item, bool): - val = int(item) - if val < 1: - raise ValueError(f"Chunk edge length must be >= 1, got {val}") - result.append(val) - elif isinstance(item, list) and len(item) == 2: - size, count = int(item[0]), int(item[1]) - if size < 1: - raise ValueError(f"Chunk edge length must be >= 1, got {size}") - if count < 1: - raise ValueError(f"RLE repeat count must be >= 1, got {count}") - result.extend([size] * count) + if isinstance(item, list): + if len(item) != 2: + raise ValueError(f"RLE entries must be an integer or [size, count], got {item}") + size, count = item + result.extend([parse_chunk_edge(size)] * _parse_positive_int(count, "RLE repeat count")) else: - raise ValueError(f"RLE entries must be an integer or [size, count], got {item}") + result.append(parse_chunk_edge(item)) return result diff --git a/src/zarr/core/group.py b/src/zarr/core/group.py index d734e6b7cd..78071300f0 100644 --- a/src/zarr/core/group.py +++ b/src/zarr/core/group.py @@ -190,12 +190,12 @@ def from_dict(cls, data: dict[str, JSON]) -> ConsolidatedMetadata: if node_type == "group": metadata[k] = GroupMetadata.from_dict(v) elif node_type == "array": - metadata[k] = ArrayV3Metadata.from_dict(v) + metadata[k] = ArrayV3Metadata.from_dict(v, path=k) else: assert_never(node_type) elif zarr_format == 2: if "shape" in v: - metadata[k] = ArrayV2Metadata.from_dict(v) + metadata[k] = ArrayV2Metadata.from_dict(v, path=k) else: metadata[k] = GroupMetadata.from_dict(v) else: @@ -3504,7 +3504,9 @@ async def _read_metadata_v3(store: Store, path: str) -> ArrayV3Metadata | GroupM ) if zarr_json_bytes is None: raise FileNotFoundError(path) - return _build_metadata_v3(buffer_to_json_object(zarr_json_bytes)) + return _build_metadata_v3( + buffer_to_json_object(zarr_json_bytes), path=_join_paths([str(store), path]) + ) async def _read_metadata_v2(store: Store, path: str) -> ArrayV2Metadata | GroupMetadata: @@ -3539,7 +3541,7 @@ async def _read_metadata_v2(store: Store, path: str) -> ArrayV2Metadata | GroupM else: zmeta = buffer_to_json_object(zgroup_bytes) - return _build_metadata_v2(zmeta, zattrs) + return _build_metadata_v2(zmeta, zattrs, path=_join_paths([str(store), path])) async def _read_group_metadata_v2(store: Store, path: str) -> GroupMetadata: @@ -3570,7 +3572,9 @@ async def _read_group_metadata( return await _read_group_metadata_v3(store=store, path=path) -def _build_metadata_v3(zarr_json: dict[str, JSON]) -> ArrayV3Metadata | GroupMetadata: +def _build_metadata_v3( + zarr_json: dict[str, JSON], *, path: str | None = None +) -> ArrayV3Metadata | GroupMetadata: """ Convert a dict representation of Zarr V3 metadata into the corresponding metadata class. """ @@ -3579,7 +3583,7 @@ def _build_metadata_v3(zarr_json: dict[str, JSON]) -> ArrayV3Metadata | GroupMet raise MetadataValidationError(msg) match zarr_json: case {"node_type": "array"}: - return ArrayV3Metadata.from_dict(zarr_json) + return ArrayV3Metadata.from_dict(zarr_json, path=path) case {"node_type": "group"}: return GroupMetadata.from_dict(zarr_json) case _: # pragma: no cover @@ -3589,14 +3593,14 @@ def _build_metadata_v3(zarr_json: dict[str, JSON]) -> ArrayV3Metadata | GroupMet def _build_metadata_v2( - zarr_json: dict[str, JSON], attrs_json: dict[str, JSON] + zarr_json: dict[str, JSON], attrs_json: dict[str, JSON], *, path: str | None = None ) -> ArrayV2Metadata | GroupMetadata: """ Convert a dict representation of Zarr V2 metadata into the corresponding metadata class. """ match zarr_json: case {"shape": _}: - return ArrayV2Metadata.from_dict(zarr_json | {"attributes": attrs_json}) + return ArrayV2Metadata.from_dict(zarr_json | {"attributes": attrs_json}, path=path) case _: # pragma: no cover return GroupMetadata.from_dict(zarr_json | {"attributes": attrs_json}) diff --git a/src/zarr/core/metadata/common.py b/src/zarr/core/metadata/common.py index b9e86f537a..6367bdb28a 100644 --- a/src/zarr/core/metadata/common.py +++ b/src/zarr/core/metadata/common.py @@ -11,12 +11,3 @@ def parse_attributes(data: dict[str, JSON] | None) -> dict[str, JSON]: return {} return dict(data) - - -def parse_chunk_edge(size: object, axis: int) -> int: - """Check that `size` is a chunk edge length: an `int` (not a `bool`) of at least 1.""" - if isinstance(size, bool) or not isinstance(size, int) or size < 1: - raise ValueError( - f"Dimension {axis}: chunk edge length must be an integer >= 1, got {size!r}" - ) - return size diff --git a/src/zarr/core/metadata/upgrades.py b/src/zarr/core/metadata/upgrades.py index e15ffd30fe..0cdac3151d 100644 --- a/src/zarr/core/metadata/upgrades.py +++ b/src/zarr/core/metadata/upgrades.py @@ -5,8 +5,9 @@ one and says how it read the document. `ArrayV2Metadata.from_dict` and `ArrayV3Metadata.from_dict` apply the upgrades for their Zarr format, so every path that parses a stored document, including consolidated metadata, goes through them, and -warn with each reading once the upgraded document has passed the metadata constructor. -An invalid document therefore raises its own error, not a warning about how it was read. +warn once, with every reading, after the upgraded document has passed the metadata +constructor. An invalid document therefore raises its own error, not a warning about +how it was read. To read another kind of invalid document, add an upgrade to `V2_ARRAY_UPGRADES` or `V3_ARRAY_UPGRADES`. @@ -17,9 +18,8 @@ import json import warnings from collections.abc import Callable, Iterable, Mapping, Sequence -from typing import TYPE_CHECKING, Final - -from typing_extensions import TypeIs +from itertools import chain, repeat +from typing import TYPE_CHECKING, Final, TypeGuard, cast from zarr.core.chunk_grids import full_span_chunk_size from zarr.errors import ZarrUserWarning @@ -39,105 +39,148 @@ ) -def warn_readings(readings: Iterable[str]) -> None: - """Warn once for each reading returned by `upgrade_array_document`.""" - for reading in readings: +def warn_readings(readings: Sequence[str], path: str | None) -> None: + """Warn once with the readings returned by `upgrade_array_document`, naming the + array at `path` when the caller knows it.""" + if readings: + subject = "" if path is None else f"Array {path!r}: " # The synchronous API parses metadata on zarr's IO thread, whose stack holds no # user code, so the warning points at the `from_dict` that read the document. - warnings.warn(f"{reading} {RESAVE_HINT}", ZarrUserWarning, stacklevel=2) + warnings.warn(f"{subject}{' '.join(readings)} {RESAVE_HINT}", ZarrUserWarning, stacklevel=2) -def _is_int_list(value: object) -> TypeIs[list[int] | tuple[int, ...]]: +def _is_int_list(value: object) -> TypeGuard[list[int] | tuple[int, ...]]: """Whether `value` is a JSON array of integers (JSON `false` and `true` count).""" return isinstance(value, list | tuple) and all(isinstance(v, int) for v in value) -def _read_invalid_chunk_sizes( - chunk_shape: JSON, shape: JSON, unit: JSON -) -> tuple[list[int], str] | None: - """Read a regular chunk shape whose chunk sizes include 0, JSON `false` or JSON `true`. +def _read_chunk_size(size: JSON, span: int | None, unit: int) -> tuple[int, str | None] | None: + """Read one entry of a stored regular chunk shape as a chunk edge length. - 0 and `false` are read as one chunk spanning the axis, a multiple of `unit` on that - axis (the inner chunk shape of a sharded array); `true` is read as 1. Returns the - upgraded chunk shape and how it was read, or `None` when there is nothing to read - this way: anything else that is invalid is left for - the metadata constructors to reject. + Returns the edge length and, for an invalid entry, how it was read; `None` if the + entry cannot be read, which leaves it for the metadata constructors to reject. A + JSON int >= 1 is kept, JSON `true` is read as 1, and 0 or JSON `false` is read as + one chunk spanning the axis of length `span`, a multiple of `unit` (the inner chunk + size of a shard), when the span is known. """ - if not (_is_int_list(chunk_shape) and _is_int_list(shape)) or len(chunk_shape) != len(shape): - return None - invalid_axes = [ - axis for axis, size in enumerate(chunk_shape) if size == 0 or isinstance(size, bool) - ] - if not invalid_axes: + match size: + case True: + return 1, "1" + case int() if size >= 1: + return size, None + case int() if size == 0 and span is not None: + edge = full_span_chunk_size(span, unit) + how = f"one chunk spanning the axis ({edge})" + if span > 0: + how += ( + ", and as no chunk can be stored under a chunk size of 0, the array " + "holds only its fill value" + ) + return edge, how + return None + + +def _read_chunk_shape( + stored: JSON, spans: Sequence[int | None], units: Iterable[int], name: str +) -> tuple[list[int], str | None] | None: + """Read a stored regular chunk shape, entry by entry (see `_read_chunk_size`), for + axes of lengths `spans` whose chunks are multiples of `units` (1 where not given). + + Returns the chunk shape and, if an entry is invalid, a sentence saying how the + `name` was read; `None` if it cannot be read. + """ + if not (isinstance(stored, list | tuple) and len(stored) == len(spans)): return None - units = ( - list(unit) - if _is_int_list(unit) and len(unit) == len(shape) and all(u >= 1 for u in unit) - else [1] * len(shape) - ) - upgraded = [ - full_span_chunk_size(extent, int(u)) if size == 0 else int(size) - for size, extent, u in zip(chunk_shape, shape, units, strict=True) - ] - readings = "; ".join( - f"{json.dumps(chunk_shape[axis])} on axis {axis} as " - + ("1" if chunk_shape[axis] else f"one chunk spanning the axis ({upgraded[axis]})") - for axis in invalid_axes - ) - message = ( - f"The stored chunk shape {json.dumps(list(chunk_shape))} is invalid: chunk sizes " - f"must be integers of at least 1. It is read as {upgraded}, reading {readings}." + edges: list[int] = [] + readings: list[str] = [] + axes = zip(stored, spans, chain(units, repeat(1)), strict=False) + for axis, (size, span, unit) in enumerate(axes): + read = _read_chunk_size(size, span, unit) + if read is None: + return None + edge, how = read + edges.append(edge) + if how is not None: + readings.append(f"{json.dumps(size)} on axis {axis} as {how}") + if not readings: + return edges, None + return edges, ( + f"The stored {name} {json.dumps(list(stored))} is invalid: chunk sizes must be " + f"integers of at least 1. It is read as {edges}, reading {'; '.join(readings)}." ) - if any(chunk_shape[axis] == 0 and shape[axis] > 0 for axis in invalid_axes): - message += ( - " No chunk can be stored under a chunk size of 0, so the array holds only its " - "fill value." - ) - return upgraded, message def _invalid_chunk_sizes_v2(doc: ArrayDocument) -> tuple[ArrayDocument, str] | None: - read = _read_invalid_chunk_sizes(doc.get("chunks"), doc.get("shape"), None) - if read is None: + shape = doc.get("shape") + if not _is_int_list(shape): return None - chunks, reading = read - return {**doc, "chunks": chunks}, reading + match _read_chunk_shape(doc.get("chunks"), shape, (), "chunk shape"): + case chunks, str(reading): + return {**doc, "chunks": chunks}, reading + return None -def _sharding_chunk_shape(codecs: JSON) -> JSON: - """The inner chunk shape of a sharding codec in a Zarr format 3 codec list, if any.""" - if isinstance(codecs, Sequence) and not isinstance(codecs, str): - for codec in codecs: +def _sharding_codec(doc: ArrayDocument) -> tuple[Sequence[JSON], int, Mapping[str, JSON]] | None: + """The codec list of a Zarr format 3 array document, with the position and the + configuration of its sharding codec, if it has one.""" + codecs = doc.get("codecs") + if isinstance(codecs, list | tuple): + for index, codec in enumerate(codecs): if isinstance(codec, Mapping) and codec.get("name") == "sharding_indexed": configuration = codec.get("configuration") if isinstance(configuration, Mapping): - return configuration.get("chunk_shape") + return codecs, index, configuration + return None + + +def _read_inner_chunk_shape(doc: ArrayDocument) -> tuple[list[int], str | None] | None: + """Read the inner chunk shape of a sharded array. No stored inner chunk size of 0 + or `false` is known, so the spans of its axes are not given.""" + shape = doc.get("shape") + sharding = _sharding_codec(doc) + if sharding is None or not _is_int_list(shape): + return None + _, _, configuration = sharding + return _read_chunk_shape( + configuration.get("chunk_shape"), + [None] * len(shape), + (), + "inner chunk shape of the sharding codec", + ) + + +def _invalid_inner_chunk_sizes_v3(doc: ArrayDocument) -> tuple[ArrayDocument, str] | None: + match _read_inner_chunk_shape(doc), _sharding_codec(doc): + case (inner, str(reading)), (codecs, index, configuration): + # `_sharding_codec` found a mapping at `index`. + codec = cast("Mapping[str, JSON]", codecs[index]) + upgraded = {**codec, "configuration": {**configuration, "chunk_shape": inner}} + return {**doc, "codecs": [*codecs[:index], upgraded, *codecs[index + 1 :]]}, reading return None def _invalid_chunk_sizes_v3(doc: ArrayDocument) -> tuple[ArrayDocument, str] | None: grid = doc.get("chunk_grid") - if not isinstance(grid, Mapping) or grid.get("name") != "regular": + shape = doc.get("shape") + if not (isinstance(grid, Mapping) and grid.get("name") == "regular" and _is_int_list(shape)): return None configuration = grid.get("configuration") if not isinstance(configuration, Mapping): return None - read = _read_invalid_chunk_sizes( - configuration.get("chunk_shape"), - doc.get("shape"), - _sharding_chunk_shape(doc.get("codecs")), - ) - if read is None: - return None - chunk_shape, reading = read - return { - **doc, - "chunk_grid": {**grid, "configuration": {**configuration, "chunk_shape": chunk_shape}}, - }, reading + inner = _read_inner_chunk_shape(doc) + units = () if inner is None else inner[0] + match _read_chunk_shape(configuration.get("chunk_shape"), shape, units, "chunk shape"): + case chunk_shape, str(reading): + upgraded = {**configuration, "chunk_shape": chunk_shape} + return {**doc, "chunk_grid": {**grid, "configuration": upgraded}}, reading + return None V2_ARRAY_UPGRADES: Final[tuple[Upgrade, ...]] = (_invalid_chunk_sizes_v2,) -V3_ARRAY_UPGRADES: Final[tuple[Upgrade, ...]] = (_invalid_chunk_sizes_v3,) +V3_ARRAY_UPGRADES: Final[tuple[Upgrade, ...]] = ( + _invalid_inner_chunk_sizes_v3, + _invalid_chunk_sizes_v3, +) def upgrade_array_document( diff --git a/src/zarr/core/metadata/v2.py b/src/zarr/core/metadata/v2.py index c648f271e6..b99ee442b2 100644 --- a/src/zarr/core/metadata/v2.py +++ b/src/zarr/core/metadata/v2.py @@ -38,11 +38,12 @@ ZARRAY_JSON, ZATTRS_JSON, MemoryOrder, + parse_chunk_shape, parse_shapelike, ) from zarr.core.config import config, parse_indexing_order from zarr.core.json_parse import parse_field -from zarr.core.metadata.common import parse_attributes, parse_chunk_edge +from zarr.core.metadata.common import parse_attributes from zarr.core.metadata.upgrades import V2_ARRAY_UPGRADES, upgrade_array_document, warn_readings @@ -147,7 +148,9 @@ def to_buffer_dict(self, prototype: BufferPrototype) -> dict[str, Buffer]: } @classmethod - def from_dict(cls, data: dict[str, Any]) -> ArrayV2Metadata: + def from_dict(cls, data: dict[str, Any], *, path: str | None = None) -> ArrayV2Metadata: + """Read a stored `.zarray` document (with its attributes). An invalid document + that `zarr.core.metadata.upgrades` can read warns, naming the array at `path`.""" upgraded, readings = upgrade_array_document(data, V2_ARRAY_UPGRADES) data = dict(upgraded) _data = data.copy() @@ -202,7 +205,7 @@ def from_dict(cls, data: dict[str, Any]) -> ArrayV2Metadata: _data = {k: v for k, v in _data.items() if k in expected} metadata = cls(**_data) - warn_readings(readings) + warn_readings(readings, path) return metadata def to_dict(self) -> dict[str, JSON]: @@ -325,8 +328,8 @@ def parse_compressor(data: object) -> Numcodec | None: def parse_chunks(chunks: Iterable[int], shape: tuple[int, ...]) -> tuple[int, ...]: - """Check a chunk shape: one chunk edge length (an integer >= 1) per array axis.""" - chunks_parsed = tuple(parse_chunk_edge(size, axis) for axis, size in enumerate(chunks)) + """Check a chunk shape: one chunk edge length (an `int` >= 1) per array axis.""" + chunks_parsed = parse_chunk_shape(chunks) if len(chunks_parsed) != len(shape): raise ValueError( f"The `shape` and `chunks` attributes must have the same length. " diff --git a/src/zarr/core/metadata/v3.py b/src/zarr/core/metadata/v3.py index b6929220ce..f1cbfcbe63 100644 --- a/src/zarr/core/metadata/v3.py +++ b/src/zarr/core/metadata/v3.py @@ -26,6 +26,7 @@ NamedRequiredConfig, compress_rle, expand_rle, + parse_chunk_edge, parse_named_configuration, parse_shapelike, validate_rectilinear_edges, @@ -35,7 +36,7 @@ from zarr.core.dtype import VariableLengthUTF8, ZDType, get_data_type_from_json from zarr.core.dtype.common import check_dtype_spec_v3 from zarr.core.json_parse import parse_field -from zarr.core.metadata.common import parse_attributes, parse_chunk_edge +from zarr.core.metadata.common import parse_attributes from zarr.core.metadata.upgrades import ( V3_ARRAY_UPGRADES, upgrade_array_document, @@ -238,19 +239,13 @@ def _validate_chunk_shapes( """ result: list[int | tuple[int, ...]] = [] for dim_idx, dim_spec in enumerate(chunk_shapes): - if isinstance(dim_spec, int): - result.append(parse_chunk_edge(dim_spec, dim_idx)) - else: - edges = tuple(dim_spec) + if isinstance(dim_spec, Iterable): + edges = tuple(parse_chunk_edge(edge, dim_idx) for edge in dim_spec) if not edges: raise ValueError(f"Dimension {dim_idx} has no chunk edges.") - bad = [i for i, e in enumerate(edges) if isinstance(e, bool) or e < 1] - if bad: - raise ValueError( - f"Dimension {dim_idx} has invalid edge lengths at indices {bad}: " - f"{[edges[i] for i in bad]}" - ) result.append(edges) + else: + result.append(parse_chunk_edge(dim_spec, dim_idx)) return tuple(result) @@ -367,18 +362,10 @@ def from_dict(cls, data: RectilinearChunkGridMetadataJSON) -> Self: # type: ign configuration = data["configuration"] validate_rectilinear_kind(configuration.get("kind")) raw_shapes = configuration["chunk_shapes"] - parsed: list[int | tuple[int, ...]] = [] - for dim_spec in raw_shapes: - if isinstance(dim_spec, int): - if dim_spec < 1: - raise ValueError(f"Integer chunk edge length must be >= 1, got {dim_spec}") - parsed.append(dim_spec) - elif isinstance(dim_spec, list): - parsed.append(tuple(expand_rle(dim_spec))) - else: - raise TypeError( - f"Invalid chunk_shapes entry: expected int or list, got {type(dim_spec)}" - ) + parsed = [ + tuple(expand_rle(dim_spec)) if isinstance(dim_spec, list) else dim_spec + for dim_spec in raw_shapes + ] return cls(chunk_shapes=tuple(parsed)) @@ -631,7 +618,9 @@ def to_buffer_dict(self, prototype: BufferPrototype) -> dict[str, Buffer]: return {ZARR_JSON: json_to_buffer(self.to_dict(), prototype=prototype, indent=indent)} @classmethod - def from_dict(cls, data: dict[str, JSON]) -> Self: + def from_dict(cls, data: dict[str, JSON], *, path: str | None = None) -> Self: + """Read a stored `zarr.json` array document. An invalid document that + `zarr.core.metadata.upgrades` can read warns, naming the array at `path`.""" upgraded, readings = upgrade_array_document(data, V3_ARRAY_UPGRADES) # a new dict, because we are modifying it _data = dict(upgraded) @@ -688,7 +677,7 @@ def from_dict(cls, data: dict[str, JSON]) -> Self: extra_fields=allowed_extra_fields, storage_transformers=_data_typed.get("storage_transformers", ()), # type: ignore[arg-type] ) - warn_readings(readings) + warn_readings(readings, path) return metadata def to_dict(self) -> dict[str, JSON]: diff --git a/tests/test_metadata/test_upgrades.py b/tests/test_metadata/test_upgrades.py index a82a8a7717..fabf22b1b8 100644 --- a/tests/test_metadata/test_upgrades.py +++ b/tests/test_metadata/test_upgrades.py @@ -11,17 +11,20 @@ import pytest import zarr +from zarr.codecs import ShardingCodec from zarr.core.metadata import ArrayV2Metadata, ArrayV3Metadata from zarr.core.metadata.upgrades import ( + RESAVE_HINT, V2_ARRAY_UPGRADES, V3_ARRAY_UPGRADES, upgrade_array_document, ) -from zarr.core.metadata.v3 import RegularChunkGridMetadata +from zarr.core.metadata.v3 import RectilinearChunkGridMetadata, RegularChunkGridMetadata from zarr.dtype import Int16 from zarr.errors import ZarrUserWarning if TYPE_CHECKING: + from collections.abc import Callable from pathlib import Path from zarr.core.common import JSON @@ -77,21 +80,58 @@ def _stored_chunks(doc: dict[str, Any]) -> Any: ) +def _chunk_shapes(metadata: ArrayV2Metadata | ArrayV3Metadata) -> tuple[Any, Any]: + """The chunk shape of `metadata` and, if it is sharded, its inner chunk shape.""" + if isinstance(metadata, ArrayV2Metadata): + return metadata.chunks, None + assert isinstance(metadata.chunk_grid, RegularChunkGridMetadata) + inner = next((c.chunk_shape for c in metadata.codecs if isinstance(c, ShardingCodec)), None) + return metadata.chunk_grid.chunk_shape, inner + + @pytest.mark.parametrize( ("doc", "expected", "warning"), [ - (_v2_doc([10, 10], [4, 5]), [4, 5], None), - (_v3_doc([0, 0], [1, 1]), [1, 1], None), - (_v3_doc([10], [4], inner=[2]), [4], None), - (_v2_doc([0, 4], [0, 4]), [1, 4], "0 on axis 0 as one chunk spanning the axis"), - (_v3_doc([0], [False]), [1], "false on axis 0 as one chunk spanning the axis"), - (_v2_doc([5], [True]), [1], "true on axis 0 as 1"), - (_v3_doc([5, 4], [True, 4]), [1, 4], "true on axis 0 as 1"), - (_v2_doc([3], [0]), [3], "holds only its fill value"), - (_v3_doc([4, 3], [4, 0]), [4, 3], "holds only its fill value"), - (_v3_doc([0], [0], inner=[4]), [4], "spanning the axis \\(4\\)"), - (_v3_doc([10], [0], inner=[4]), [12], "spanning the axis \\(12\\)"), - (_v3_doc([0, 3], [0, 3], inner=[2, 3]), [2, 3], "spanning the axis \\(2\\)"), + (_v2_doc([10, 10], [4, 5]), ((4, 5), None), None), + (_v3_doc([0, 0], [1, 1]), ((1, 1), None), None), + (_v3_doc([10], [4], inner=[2]), ((4,), (2,)), None), + ( + _v2_doc([0, 4], [0, 4]), + ((1, 4), None), + r"0 on axis 0 as one chunk spanning the axis \(1\)\.", + ), + ( + _v3_doc([0], [False]), + ((1,), None), + r"false on axis 0 as one chunk spanning the axis \(1\)\.", + ), + (_v2_doc([5], [True]), ((1,), None), "true on axis 0 as 1"), + ( + _v3_doc([5, 4], [True, 4]), + ((1, 4), None), + r"\[true, 4\] is invalid.*true on axis 0 as 1", + ), + ( + _v2_doc([3], [0]), + ((3,), None), + r"spanning the axis \(3\), and .* holds only its fill value", + ), + (_v3_doc([4, 3], [4, 0]), ((4, 3), None), "0 on axis 1 as .* holds only its fill value"), + (_v3_doc([0], [0], inner=[4]), ((4,), (4,)), r"spanning the axis \(4\)\."), + ( + _v3_doc([10], [0], inner=[4]), + ((12,), (4,)), + r"spanning the axis \(12\), and .* holds only its fill value", + ), + (_v3_doc([0, 3], [0, 3], inner=[2, 3]), ((2, 3), (2, 3)), r"spanning the axis \(2\)\."), + ( + _v3_doc([5], [True], inner=[True]), + ((1,), (1,)), + ( + r"^The stored inner chunk shape of the sharding codec \[true\] is invalid: .* " + r"The stored chunk shape \[true\] is invalid: " + ), + ), ], ids=[ "v2-valid", @@ -106,85 +146,142 @@ def _stored_chunks(doc: dict[str, Any]) -> Any: "v3-sharded-zero-empty-axis", "v3-sharded-zero-grown-axis", "v3-sharded-zero-2d", + "v3-sharded-true-inner-and-outer", ], ) def test_upgrade_array_document( - doc: dict[str, JSON], expected: list[int], warning: str | None + doc: dict[str, JSON], expected: tuple[Any, Any], warning: str | None ) -> None: """Valid documents pass unchanged and silently. A stored chunk size of 0 or `false` is read as one chunk spanning the axis (a multiple of the inner chunk when sharded) - and `true` as 1; `from_dict` warns once with the reading and how to re-save.""" + and `true` as 1, in the chunk shape and in a sharding codec's inner chunk shape; + `from_dict` warns once for the document, naming the array, saying how each part was + read (and that the array holds only its fill value where a chunk size of 0 was + stored for a non-empty axis) and how to re-save.""" upgrades = V2_ARRAY_UPGRADES if doc["zarr_format"] == 2 else V3_ARRAY_UPGRADES upgraded, readings = upgrade_array_document(doc, upgrades) - assert _stored_chunks(dict(upgraded)) == expected - assert {k: v for k, v in upgraded.items() if k not in ("chunks", "chunk_grid")} == { - k: v for k, v in doc.items() if k not in ("chunks", "chunk_grid") + assert {k: v for k, v in upgraded.items() if k not in ("chunks", "chunk_grid", "codecs")} == { + k: v for k, v in doc.items() if k not in ("chunks", "chunk_grid", "codecs") } metadata_cls = ArrayV2Metadata if doc["zarr_format"] == 2 else ArrayV3Metadata with warnings.catch_warnings(record=True) as record: warnings.simplefilter("always") - metadata_cls.from_dict(dict(doc)) + metadata = metadata_cls.from_dict(dict(doc), path="group/array") + assert _chunk_shapes(metadata) == expected messages = [str(w.message) for w in record] if warning is None: assert upgraded is doc assert readings == [] assert messages == [] else: - assert len(readings) == 1 - assert re.search(warning, readings[0]) assert len(messages) == 1 - assert messages[0].startswith(readings[0]) - assert "update_attributes({})" in messages[0] - assert "zarr.consolidate_metadata" in messages[0] + message = messages[0] + assert message.startswith("Array 'group/array': ") + assert re.search(warning, message.removeprefix("Array 'group/array': ")) + assert ("holds only its fill value" in message) == ("fill value" in warning) + assert message.endswith(RESAVE_HINT) @pytest.mark.parametrize( - "doc", - [_v2_doc([4], [0]) | {"dtype": " None: - """A document that is invalid after its upgrade raises its own error, without first - warning how it was read.""" +def test_invalid_upgraded_document_raises_without_warning(doc: dict[str, JSON], error: str) -> None: + """A document the upgrades read that the metadata constructor then rejects raises + that error, without first warning how it was read.""" metadata_cls = ArrayV2Metadata if doc["zarr_format"] == 2 else ArrayV3Metadata + assert upgrade_array_document(doc, V2_ARRAY_UPGRADES + V3_ARRAY_UPGRADES)[1] with warnings.catch_warnings(): warnings.simplefilter("error", ZarrUserWarning) - with pytest.raises(ValueError): + with pytest.raises(ValueError, match=error): metadata_cls.from_dict(doc) -@pytest.mark.parametrize("doc", [_v2_doc([4], [-1]), _v3_doc([4], [-1])], ids=["v2", "v3"]) -def test_stored_negative_chunk_size_rejected(doc: dict[str, JSON]) -> None: - metadata_cls = ArrayV2Metadata if doc["zarr_format"] == 2 else ArrayV3Metadata - with pytest.raises(ValueError, match="chunk edge length must be an integer >= 1, got -1"): - metadata_cls.from_dict(doc) - - -@pytest.mark.parametrize("doc", [_v2_doc([4, 4], [0]), _v3_doc([4, 4], [0])], ids=["v2", "v3"]) -def test_stored_chunk_shape_ndim_mismatch_rejected(doc: dict[str, JSON]) -> None: - """A chunk shape with the wrong number of axes is not upgraded, only rejected.""" +@pytest.mark.parametrize( + ("doc", "error"), + [ + (_v2_doc([4], [-1]), "Dimension 0: Chunk edge length must be >= 1, got -1"), + (_v3_doc([4, 4], [0]), "Dimension 0: Chunk edge length must be >= 1, got 0"), + (_v3_doc([4], [4.0]), "Dimension 0: Chunk edge length must be an int, got 4.0"), + (_v3_doc([4], [0], inner=[0]), "Dimension 0: Chunk edge length must be >= 1, got 0"), + ], + ids=["negative", "ndim-mismatch", "float", "sharded-inner-zero"], +) +def test_stored_chunk_shape_not_upgraded(doc: dict[str, JSON], error: str) -> None: + """Invalid chunk sizes no known writer stored, and chunk shapes with the wrong + number of axes, are not upgraded, only rejected.""" metadata_cls = ArrayV2Metadata if doc["zarr_format"] == 2 else ArrayV3Metadata with warnings.catch_warnings(): warnings.simplefilter("error", ZarrUserWarning) - with pytest.raises(ValueError, match="chunk edge length|same length|same number"): + with pytest.raises((TypeError, ValueError), match=re.escape(error)): metadata_cls.from_dict(doc) -@pytest.mark.parametrize("size", [0, True], ids=["zero", "true"]) -def test_constructor_rejects_invalid_chunk_size(size: int) -> None: - """Metadata built in code is strict: 0 and `True` are rejected, without a warning.""" +def test_v2_constructor_rejects_chunks_of_wrong_length() -> None: + with pytest.raises(ValueError, match="`chunks` has length 1, but `shape` has length 2"): + ArrayV2Metadata(shape=(4, 4), chunks=(2,), dtype=Int16(), fill_value=0, order="C") + + +def _v2_metadata(chunks: Any) -> ArrayV2Metadata: + return ArrayV2Metadata(shape=(4,), chunks=chunks, dtype=Int16(), fill_value=0, order="C") + + +def _rectilinear(chunk_shapes: tuple[Any, ...]) -> RectilinearChunkGridMetadata: + with zarr.config.set({"array.rectilinear_chunks": True}): + return RectilinearChunkGridMetadata(chunk_shapes=chunk_shapes) + + +def _rectilinear_from_dict(chunk_shapes: list[Any]) -> RectilinearChunkGridMetadata: + with zarr.config.set({"array.rectilinear_chunks": True}): + return RectilinearChunkGridMetadata.from_dict( + { + "name": "rectilinear", + "configuration": {"kind": "inline", "chunk_shapes": chunk_shapes}, + } + ) + + +CHUNK_EDGE_SITES: dict[str, Callable[[Any], object]] = { + "regular": lambda size: RegularChunkGridMetadata(chunk_shape=(size,)), + "v2": lambda size: _v2_metadata((size,)), + "rectilinear-bare": lambda size: _rectilinear((size,)), + "rectilinear-edge": lambda size: _rectilinear(((4, size),)), + "rectilinear-bare-json": lambda size: _rectilinear_from_dict([size]), + "rectilinear-edge-json": lambda size: _rectilinear_from_dict([[4, size]]), + "rectilinear-rle-json": lambda size: _rectilinear_from_dict([[[size, 2]]]), + "sharding-inner": lambda size: ShardingCodec(chunk_shape=(size,)), +} + + +@pytest.mark.parametrize("site", CHUNK_EDGE_SITES) +@pytest.mark.parametrize("size", [True, False, 4.0, np.int64(4), "4"]) +def test_metadata_rejects_non_int_chunk_edge(site: str, size: object) -> None: + """Metadata built in code takes chunk edge lengths as `int`s only, everywhere.""" + with pytest.raises( + TypeError, match=re.escape(f"Chunk edge length must be an int, got {size!r}") + ): + CHUNK_EDGE_SITES[site](size) + + +@pytest.mark.parametrize("site", CHUNK_EDGE_SITES) +@pytest.mark.parametrize("size", [0, -1]) +def test_metadata_rejects_chunk_edge_below_one(site: str, size: int) -> None: + """Metadata built in code is strict: a chunk edge length below 1 is rejected, + without a warning.""" with warnings.catch_warnings(): warnings.simplefilter("error", ZarrUserWarning) - with pytest.raises(ValueError, match=f"got {size!r}"): - RegularChunkGridMetadata(chunk_shape=(size,)) - with pytest.raises(ValueError, match=f"got {size!r}"): - ArrayV2Metadata( - shape=(0,), - chunks=(size,), - dtype=Int16(), - fill_value=0, - order="C", - ) + with pytest.raises(ValueError, match=f"Chunk edge length must be >= 1, got {size}"): + CHUNK_EDGE_SITES[site](size) + + +@pytest.mark.parametrize("chunks", [4, np.int64(4)]) +def test_v2_constructor_rejects_scalar_chunks(chunks: object) -> None: + with pytest.raises(TypeError, match="A chunk shape must be a sequence of chunk edge lengths"): + _v2_metadata(chunks) def _rewrite_doc(path: Path, zarr_format: Literal[2, 3], edit: Any) -> None: @@ -198,26 +295,10 @@ def _rewrite_doc(path: Path, zarr_format: Literal[2, 3], edit: Any) -> None: ("zarr_format", "shape", "stored", "inner", "expected"), [ (2, (0, 4), [0, 4], None, (1, 4)), - (2, (3,), [0], None, (3,)), - (2, (5,), [True], None, (1,)), - (3, (0,), [False], None, (1,)), - (3, (3,), [0], None, (3,)), (3, (5,), [True], None, (1,)), - (3, (0,), [0], (4,), (4,)), (3, (10,), [0], (4,), (12,)), - (3, (0, 3), [0, 3], (2, 3), (2, 3)), - ], - ids=[ - "v2-empty-2d", - "v2-grown", - "v2-true", - "v3-false-empty", - "v3-grown", - "v3-true", - "v3-sharded-empty", - "v3-sharded-grown", - "v3-sharded-2d", ], + ids=["v2-empty-2d", "v3-true", "v3-sharded-grown"], ) def test_legacy_chunk_size_round_trip( tmp_path: Path, @@ -250,7 +331,7 @@ def store_legacy(doc: dict[str, Any]) -> None: _rewrite_doc(path, zarr_format, store_legacy) - with pytest.warns(ZarrUserWarning, match="is read as"): + with pytest.warns(ZarrUserWarning, match=r"^Array '.*legacy\.zarr': .* is read as"): arr = zarr.open_array(store=path, mode="a") assert (arr.shards or arr.chunks) == expected np.testing.assert_array_equal(arr[...], data) @@ -267,43 +348,52 @@ def store_legacy(doc: dict[str, Any]) -> None: @pytest.mark.filterwarnings("ignore:Consolidated metadata is currently not part:UserWarning") @pytest.mark.parametrize("zarr_format", [2, 3]) def test_legacy_chunk_size_consolidated(tmp_path: Path, zarr_format: Literal[2, 3]) -> None: - """Consolidated metadata goes through the same upgrade; re-saving the array and - consolidating again leaves a group that opens without a warning.""" + """Consolidated metadata goes through the same upgrade, with one warning naming each + array; re-saving the arrays and consolidating again leaves a group that opens + without a warning.""" path = tmp_path / "group.zarr" group = zarr.open_group(path, mode="w", zarr_format=zarr_format) - group.create_array("a", shape=(0,), chunks=(1,), dtype="int32") + names = ("a", "b") + for name in names: + group.create_array(name, shape=(0,), chunks=(1,), dtype="int32") zarr.consolidate_metadata(path) # What zarr-python wrote for `chunks=(0,)` on an empty array, in both copies. - if zarr_format == 2: - _rewrite_doc(path / "a", 2, lambda doc: doc.update(chunks=[0])) - zmetadata = json.loads((path / ".zmetadata").read_text()) - zmetadata["metadata"]["a/.zarray"]["chunks"] = [0] - (path / ".zmetadata").write_text(json.dumps(zmetadata)) - else: - _rewrite_doc( - path / "a", 3, lambda doc: doc["chunk_grid"]["configuration"].update(chunk_shape=[0]) - ) - _rewrite_doc( - path, - 3, - lambda doc: doc["consolidated_metadata"]["metadata"]["a"]["chunk_grid"][ - "configuration" - ].update(chunk_shape=[0]), - ) + for name in names: + if zarr_format == 2: + _rewrite_doc(path / name, 2, lambda doc: doc.update(chunks=[0])) + zmetadata = json.loads((path / ".zmetadata").read_text()) + zmetadata["metadata"][f"{name}/.zarray"]["chunks"] = [0] + (path / ".zmetadata").write_text(json.dumps(zmetadata)) + else: + _rewrite_doc( + path / name, + 3, + lambda doc: doc["chunk_grid"]["configuration"].update(chunk_shape=[0]), + ) + _rewrite_doc( + path, + 3, + lambda doc, name=name: doc["consolidated_metadata"]["metadata"][name]["chunk_grid"][ + "configuration" + ].update(chunk_shape=[0]), + ) - with pytest.warns(ZarrUserWarning, match="zarr.consolidate_metadata"): + with pytest.warns(ZarrUserWarning, match="zarr.consolidate_metadata") as record: group = zarr.open_group(path, mode="r+") - array = group["a"] - assert isinstance(array, zarr.Array) - assert array.chunks == (1,) - - array.update_attributes({}) + assert sorted(str(w.message).split(":")[0] for w in record) == ["Array 'a'", "Array 'b'"] + for name in names: + array = group[name] + assert isinstance(array, zarr.Array) + assert array.chunks == (1,) + array.update_attributes({}) zarr.consolidate_metadata(path) with warnings.catch_warnings(): warnings.simplefilter("error", ZarrUserWarning) warnings.filterwarnings("ignore", "Consolidated metadata is currently not part") for use_consolidated in (True, False): - reopened = zarr.open_group(path, mode="r", use_consolidated=use_consolidated)["a"] - assert isinstance(reopened, zarr.Array) - assert reopened.chunks == (1,) + reopened = zarr.open_group(path, mode="r", use_consolidated=use_consolidated) + for name in names: + array = reopened[name] + assert isinstance(array, zarr.Array) + assert array.chunks == (1,) diff --git a/tests/test_unified_chunk_grid.py b/tests/test_unified_chunk_grid.py index 6e2379e2fd..4a8d6f791e 100644 --- a/tests/test_unified_chunk_grid.py +++ b/tests/test_unified_chunk_grid.py @@ -547,19 +547,19 @@ def test_rle_expand_rejects_invalid(rle_input: list[Any], match: str) -> None: expand_rle(rle_input) -# -- expand_rle handles JSON floats -- - - -def test_expand_rle_bare_integer_floats_accepted() -> None: - """JSON parsers may emit 10.0 for the integer 10; expand_rle should handle it.""" - result = expand_rle([10.0, 20.0]) # type: ignore[list-item] - assert result == [10, 20] - - -def test_expand_rle_pair_with_float_count() -> None: - """expand_rle accepts float repeat counts that are integer-valued""" - result = expand_rle([[10, 3.0]]) # type: ignore[list-item] - assert result == [10, 10, 10] +@pytest.mark.parametrize( + ("rle_input", "match"), + [ + ([10.0], "Chunk edge length must be an int, got 10.0"), + ([True], "Chunk edge length must be an int, got True"), + ([[10, 3.0]], "RLE repeat count must be an int, got 3.0"), + ], + ids=["float-edge", "bool-edge", "float-count"], +) +def test_rle_expand_rejects_non_int(rle_input: list[Any], match: str) -> None: + """expand_rle takes JSON integers only: no stored document holds integral floats.""" + with pytest.raises(TypeError, match=match): + expand_rle(rle_input) # --------------------------------------------------------------------------- From 192f86c446257ab0255976cae068c3218284f432 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Sat, 26 Sep 2026 02:08:38 +0200 Subject: [PATCH 19/41] refactor(chunk-grids): start chunk guessing from the one full-span rule `_guess_regular_chunks` clamped zero-length axes with `np.maximum`, restating `full_span_chunk_size` with unit 1. Fold the zero-span normalizer tests into the `normalize_chunks_1d` tables. Assisted-by: ClaudeCode:claude-opus-5-5 Co-Authored-By: Claude Opus 5.5 --- src/zarr/core/chunk_grids.py | 6 +++--- tests/test_chunk_grids.py | 25 +++++++++---------------- 2 files changed, 12 insertions(+), 19 deletions(-) diff --git a/src/zarr/core/chunk_grids.py b/src/zarr/core/chunk_grids.py index 17b3225725..399943f72d 100644 --- a/src/zarr/core/chunk_grids.py +++ b/src/zarr/core/chunk_grids.py @@ -698,12 +698,12 @@ def _guess_regular_chunks( if isinstance(shape, int): shape = (shape,) + # Start from one chunk spanning each axis, then halve axes until the chunk is small enough. + chunks = np.array([full_span_chunk_size(s) for s in shape], dtype="=f8") if typesize == 0: - return tuple(full_span_chunk_size(s) for s in shape) + return tuple(int(x) for x in chunks) ndims = len(shape) - # require chunks to have non-zero length for all dimensions - chunks = np.maximum(np.array(shape, dtype="=f8"), 1) # Determine the optimal chunk size in bytes using a PyTables expression. # This is kept as a float. diff --git a/tests/test_chunk_grids.py b/tests/test_chunk_grids.py index d171937898..207f121d11 100644 --- a/tests/test_chunk_grids.py +++ b/tests/test_chunk_grids.py @@ -210,6 +210,10 @@ def test_chunk_layout_nested() -> None: ExpectFail( input=([10, 20], 100), exception=ValueError, id="wrong-sum", msg="do not sum to span" ), + # Only a zero-length span keeps edges that do not sum to it. + ExpectFail( + input=([3, 5], 1), exception=ValueError, id="over-sum", msg="do not sum to span 1" + ), # Nested/RLE form for a single dim is rejected with offending indices. ExpectFail( input=([[3, 3], 1], 7), @@ -319,6 +323,11 @@ def test_normalize_chunks_nd_errors(case: ExpectFail[tuple[Any, tuple[int, ...]] output=VaryingDimension([10, 20, 70], extent=100), id="explicit-irregular", ), + # on a zero-length span any non-empty list of positive edges is kept: the + # chunks the axis grows into. + Expect( + input=([3, 5], 0), output=VaryingDimension([3, 5], extent=0), id="explicit-zero-span" + ), ], ids=lambda c: c.id, ) @@ -477,19 +486,3 @@ def test_rectilinear_zero_extent_matches_resize() -> None: np.testing.assert_array_equal(created[...], np.arange(3)) np.testing.assert_array_equal(resized[...], np.arange(3)) assert created.write_chunk_sizes == resized.write_chunk_sizes == ((2, 1),) - - -def test_normalize_chunks_1d_zero_span_accepts_any_edges() -> None: - """On a zero-length span the explicit edge list is stored verbatim.""" - dim = normalize_chunks_1d([3, 5], span=0) - assert isinstance(dim, VaryingDimension) - assert dim.edges == (3, 5) - assert dim.extent == 0 - assert dim.nchunks == 0 - assert dim.resize(4) == VaryingDimension([3, 5], extent=4) - - -def test_normalize_chunks_1d_nonzero_span_still_requires_exact_sum() -> None: - """Relaxing the sum rule for span 0 must not leak into positive spans.""" - with pytest.raises(ValueError, match="do not sum to span 1"): - normalize_chunks_1d([3, 5], span=1) From 2e878cf41e346857a40ec03458340e94ca249d80 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Sat, 26 Sep 2026 02:08:39 +0200 Subject: [PATCH 20/41] test: run the array lifecycle state machine in the slow Hypothesis job Add it to `just hypothesis`. Re-saving metadata is always enabled and must leave valid metadata unchanged; appends favour an axis stored with chunk size 0, which raises how often a legacy axis grows before it is re-saved. A fixed example pins that a write to a shard kept by a shrink leaves its cells beyond the shape alone. Assisted-by: ClaudeCode:claude-opus-5-5 Co-Authored-By: Claude Opus 5.5 --- Justfile | 2 +- tests/test_array.py | 12 ++++++++++++ tests/test_array_stateful.py | 37 ++++++++++++++++++++++-------------- 3 files changed, 36 insertions(+), 15 deletions(-) diff --git a/Justfile b/Justfile index 6e56644207..b3f187e231 100644 --- a/Justfile +++ b/Justfile @@ -57,7 +57,7 @@ coverage-serve *args: # Run slow Hypothesis tests and write coverage.xml hypothesis *args: - hatch run {{ quote(hatch_env) }}:coverage run --source=src -m pytest -nauto --run-slow-hypothesis tests/test_properties.py tests/test_store/test_stateful* "$@" + hatch run {{ quote(hatch_env) }}:coverage run --source=src -m pytest -nauto --run-slow-hypothesis tests/test_properties.py tests/test_store/test_stateful* tests/test_array_stateful.py "$@" hatch run {{ quote(hatch_env) }}:coverage xml # Validate executable documentation code blocks diff --git a/tests/test_array.py b/tests/test_array.py index 3ae502a4e8..b153f24c0f 100644 --- a/tests/test_array.py +++ b/tests/test_array.py @@ -732,6 +732,18 @@ def test_resize_1d(store: MemoryStore, zarr_format: ZarrFormat) -> None: assert new_shape == result.shape +@pytest.mark.parametrize("chunks", [(1,), (2,), (4,)]) +def test_resize_sharded_keeps_cells_beyond_shape(chunks: tuple[int, ...]) -> None: + """A shard kept by a shrinking resize keeps its cells beyond the new shape, and a + later write to the shard leaves them alone, so they come back when the array grows.""" + arr = zarr.create_array({}, shape=(4,), chunks=chunks, shards=(4,), dtype="int16", fill_value=0) + arr[:] = [1, 2, 3, 4] + arr.resize((2,)) + arr[:] = [9, 9] + arr.resize((4,)) + np.testing.assert_array_equal(arr[:], [9, 9, 3, 4]) + + @pytest.mark.parametrize("store", ["memory"], indirect=True) def test_resize_2d(store: MemoryStore, zarr_format: ZarrFormat) -> None: z = zarr.create( diff --git a/tests/test_array_stateful.py b/tests/test_array_stateful.py index 6e9bea8f8b..0ea4da0167 100644 --- a/tests/test_array_stateful.py +++ b/tests/test_array_stateful.py @@ -30,7 +30,6 @@ RuleBasedStateMachine, initialize, invariant, - precondition, rule, ) @@ -79,8 +78,8 @@ def __init__(self) -> None: self.fill = 0 # What the store holds, indexed like the array and extending past its shape. self.stored: np.ndarray[Any, np.dtype[np.int16]] = np.zeros((0,), dtype=DTYPE) - # A store with an invalid stored chunk size warns until its metadata is re-saved. - self.expect_open_warning = False + # Axes stored with chunk size 0; the store warns until its metadata is re-saved. + self.legacy_axes: list[int] = [] # -------------------------------------------------------------- creation @initialize(data=st.data()) @@ -130,13 +129,12 @@ def create(self, data: st.DataObject) -> None: # A stored chunk size of 0, as zarr-python wrote for arrays created with a # zero-length axis; older releases could then grow the axis without storing # a chunk, so any extent is possible. Sharded arrays store it in the outer grid. - zero_axes = data.draw( + self.legacy_axes = data.draw( st.lists(st.integers(0, len(shape) - 1), min_size=1, unique=True), label="axes stored with chunk size 0", ) stored_zero = data.draw(st.sampled_from([0, False]), label="stored zero") - self._rewrite_stored_chunks(zarr_format, zero_axes, stored_zero) - self.expect_open_warning = True + self._rewrite_stored_chunks(zarr_format, self.legacy_axes, stored_zero) event("legacy zero chunk size") def _rewrite_stored_chunks( @@ -158,7 +156,7 @@ def _open(self) -> zarr.Array[Any]: warnings.simplefilter("always", ZarrUserWarning) arr = zarr.open_array(self.store, path=self.path, mode="r+") warned = any(issubclass(w.category, ZarrUserWarning) for w in record) - assert warned is self.expect_open_warning, [str(w.message) for w in record] + assert warned is bool(self.legacy_axes), [str(w.message) for w in record] return arr # ----------------------------------------------------------------- model @@ -209,13 +207,19 @@ def _model_write(self, arr: zarr.Array[Any], region: tuple[slice, ...], values: @rule(data=st.data()) def append(self, data: st.DataObject) -> None: arr = self._open() - axis = data.draw(st.integers(0, len(self.shape) - 1), label="axis") + axes = st.integers(0, len(self.shape) - 1) + if self.legacy_axes: + # What a user of an older release did next: grow an axis stored with chunk size 0. + axes = st.sampled_from(self.legacy_axes) | axes + axis = data.draw(axes, label="axis") block_shape = list(self.shape) block_shape[axis] = data.draw(st.integers(0, 4), label="rows") block = data.draw(npst.arrays(DTYPE, tuple(block_shape)), label="block") note(f"append {block.shape} along {axis} to {self.shape}") if self.shape[axis] == 0 and block.shape[axis]: event("append to a zero-length axis") + if axis in self.legacy_axes and block.shape[axis]: + event("grow an axis stored with chunk size 0") old_extent = self.shape[axis] arr.append(block, axis=axis) self._model_resize(ChunkGrid.from_metadata(arr.metadata), arr.shape) @@ -224,7 +228,7 @@ def append(self, data: st.DataObject) -> None: ) self._model_write(arr, region, block) # Growing the array rewrites its metadata, which stores any correction. - self.expect_open_warning = False + self.legacy_axes = [] @rule(data=st.data()) def resize(self, data: st.DataObject) -> None: @@ -233,10 +237,12 @@ def resize(self, data: st.DataObject) -> None: st.tuples(*(st.integers(0, MAX_SIDE) for _ in self.shape)), label="new shape" ) note(f"resize {self.shape} -> {new_shape}") + if any(new_shape[axis] > self.shape[axis] for axis in self.legacy_axes): + event("grow an axis stored with chunk size 0") grid = ChunkGrid.from_metadata(arr.metadata) arr.resize(new_shape) self._model_resize(grid, new_shape) - self.expect_open_warning = False + self.legacy_axes = [] @rule(data=st.data()) def write(self, data: st.DataObject) -> None: @@ -251,12 +257,15 @@ def write(self, data: st.DataObject) -> None: arr[region] = values self._model_write(arr, region, values) - @precondition(lambda self: self.expect_open_warning) @rule() def resave_metadata(self) -> None: - """What the warning for an invalid stored chunk size tells users to do.""" - self._open().update_attributes({}) - self.expect_open_warning = False + """What the warning for an invalid stored chunk size tells users to do. It stores + the metadata as read, which is a no-op for valid metadata.""" + arr = self._open() + read = arr.metadata + arr.update_attributes({}) + self.legacy_axes = [] + assert self._open().metadata == read def teardown(self) -> None: self._rectilinear.__exit__(None, None, None) From c0d4b3cc76c93cbd7364367a21140e4fc3017365 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Sat, 26 Sep 2026 05:58:34 +0200 Subject: [PATCH 21/41] fix(array): store upgraded metadata before writing the first chunk An array read from a stored document that the upgrades had to correct (a chunk size of 0, `false` or `true`) wrote chunks under the corrected layout while the store kept the old document, so zarr 3.2.1-3.4.0 then read only the fill value, and tensorstore and zarrs could not open it. `from_dict` now marks the metadata it read from an upgraded document (`_stored_document_upgraded`, a field outside the document and equality), and `AsyncArray._set_selection`, which every chunk write goes through (`setitem` now included), stores that metadata first and keeps the stored copy. Concurrent first writes store the same document. The array lifecycle state machine now ends the legacy state on a write and checks that no chunk is stored under a document that is still upgraded on read; `resave_metadata` runs only while the stored document is invalid, and re-saving valid metadata is checked once, at creation. The changelog also says which releases stored a chunk size of 0 on an axis of positive length. Assisted-by: ClaudeCode:claude-opus-5-5 Co-Authored-By: Claude Opus 5.5 --- changes/4334.bugfix.md | 2 +- src/zarr/abc/metadata.py | 5 ++- src/zarr/core/array.py | 20 +++++----- src/zarr/core/metadata/v2.py | 6 ++- src/zarr/core/metadata/v3.py | 4 ++ tests/test_array_stateful.py | 34 +++++++++++++--- tests/test_metadata/test_upgrades.py | 60 ++++++++++++++++++++++++++++ 7 files changed, 113 insertions(+), 18 deletions(-) diff --git a/changes/4334.bugfix.md b/changes/4334.bugfix.md index 77c211643e..4c55701757 100644 --- a/changes/4334.bugfix.md +++ b/changes/4334.bugfix.md @@ -2,4 +2,4 @@ A chunk edge length is now always at least 1, while an array extent may be 0. Ev Metadata built in code is strict about chunk edge lengths: `ArrayV2Metadata`, `RegularChunkGridMetadata`, `RectilinearChunkGridMetadata` and `ShardingCodec` take them as Python `int`s of at least 1, so a size of 0 now raises a `ValueError`, and a `bool`, a float or a NumPy integer (including a scalar `chunks=np.int64(5)` for `ArrayV2Metadata`) raises a `TypeError`; `FixedDimension(size=0, ...)` raises a `ValueError`. Stored rectilinear chunk grids with integral floats (`10.0`) are rejected too. The array creation functions still accept NumPy integers in `chunks=` and `shards=`. -Stored metadata with a regular chunk size of 0 or JSON `false`, as zarr-python wrote for arrays created with a zero-length axis until 3.4, now opens with a `ZarrUserWarning` and is read as one chunk spanning the axis (a multiple of the inner chunk size for sharded arrays, which previously failed to open). JSON `true`, as zarr-python 3.0 and 3.2 wrote for `chunks=(True,)`, is read as 1, in the chunk grid and in the inner chunk shape of a sharding codec. The warning names the array and says how to store valid metadata: `array.update_attributes({})`, then `zarr.consolidate_metadata` if the metadata is consolidated. +Stored metadata with a regular chunk size of 0 or JSON `false` now opens with a `ZarrUserWarning` and is read as one chunk spanning the axis (a multiple of the inner chunk size for sharded arrays, which previously failed to open). zarr-python wrote such sizes for arrays created with a zero-length axis until 3.4, and, from 2.18.7 to 3.2.1, for an explicit chunk size of 0 or `False` on an axis of any length, as in `chunks=(0,)` (those arrays could store no data). JSON `true`, as zarr-python 3.0 and 3.2 wrote for `chunks=(True,)`, is read as 1, in the chunk grid and in the inner chunk shape of a sharding codec. The warning names the array and says how to store valid metadata: `array.update_attributes({})`, then `zarr.consolidate_metadata` if the metadata is consolidated. Writing data to such an array first stores the upgraded metadata, so that other readers find the chunks written. diff --git a/src/zarr/abc/metadata.py b/src/zarr/abc/metadata.py index a56f986645..1104111b63 100644 --- a/src/zarr/abc/metadata.py +++ b/src/zarr/abc/metadata.py @@ -20,10 +20,13 @@ def to_dict(self) -> dict[str, JSON]: Recursively serialize this model to a dictionary. This method inspects the fields of self and calls `x.to_dict()` for any fields that are instances of `Metadata`. Sequences of `Metadata` are similarly recursed into, and - the output of that recursion is collected in a list. + the output of that recursion is collected in a list. Fields declared with + `compare=False` are not part of the document. """ out_dict = {} for field in fields(self): + if not field.compare: + continue key = field.name value = getattr(self, key) if isinstance(value, Metadata): diff --git a/src/zarr/core/array.py b/src/zarr/core/array.py index e39080b668..5f4cddb9b4 100644 --- a/src/zarr/core/array.py +++ b/src/zarr/core/array.py @@ -1623,6 +1623,12 @@ async def _set_selection( prototype: BufferPrototype, fields: Fields | None = None, ) -> None: + if self.metadata._stored_document_upgraded: + # Chunks are about to be stored under the upgraded metadata, so store it + # first: every reader of the store then agrees with them. + metadata = replace(self.metadata) + await self._save_metadata(metadata) + object.__setattr__(self, "metadata", metadata) return await _set_selection( self.store_path, self.metadata, @@ -1674,16 +1680,10 @@ async def setitem( - This method is asynchronous and should be awaited. - Supports basic indexing, where the selection is contiguous and does not involve advanced indexing. """ - return await _setitem( - self.store_path, - self.metadata, - self.codec_pipeline, - self.config, - self._chunk_grid, - selection, - value, - prototype=prototype, - ) + if prototype is None: + prototype = default_buffer_prototype() + indexer = BasicIndexer(selection, shape=self.metadata.shape, chunk_grid=self._chunk_grid) + return await self._set_selection(indexer, value, prototype=prototype) @property def oindex(self) -> AsyncOIndex[T_ArrayMetadata]: diff --git a/src/zarr/core/metadata/v2.py b/src/zarr/core/metadata/v2.py index b99ee442b2..b5701b3e78 100644 --- a/src/zarr/core/metadata/v2.py +++ b/src/zarr/core/metadata/v2.py @@ -72,6 +72,9 @@ class ArrayV2Metadata(Metadata): compressor: Numcodec | None attributes: dict[str, JSON] = field(default_factory=dict) zarr_format: Literal[2] = field(init=False, default=2) + _stored_document_upgraded: bool = field(default=False, init=False, compare=False, repr=False) + """Whether `from_dict` read this metadata from a stored document it had to upgrade, + so the store holds an invalid document until this metadata is stored.""" def __init__( self, @@ -184,7 +187,7 @@ def from_dict(cls, data: dict[str, Any], *, path: str | None = None) -> ArrayV2M # zarr v2 allowed arbitrary keys here. # We don't want the ArrayV2Metadata constructor to fail just because someone put an # extra key in the metadata. - expected = {x.name for x in fields(cls)} + expected = {x.name for x in fields(cls) if x.init} expected |= {"dtype", "chunks"} # check if `filters` is an empty sequence; if so use None instead and raise a warning @@ -206,6 +209,7 @@ def from_dict(cls, data: dict[str, Any], *, path: str | None = None) -> ArrayV2M metadata = cls(**_data) warn_readings(readings, path) + object.__setattr__(metadata, "_stored_document_upgraded", bool(readings)) return metadata def to_dict(self) -> dict[str, JSON]: diff --git a/src/zarr/core/metadata/v3.py b/src/zarr/core/metadata/v3.py index f1cbfcbe63..ab5ddf3d2f 100644 --- a/src/zarr/core/metadata/v3.py +++ b/src/zarr/core/metadata/v3.py @@ -478,6 +478,9 @@ class ArrayV3Metadata(Metadata): node_type: Literal["array"] = field(default="array", init=False) storage_transformers: tuple[dict[str, JSON], ...] extra_fields: dict[str, AllowedExtraField] + _stored_document_upgraded: bool = field(default=False, init=False, compare=False, repr=False) + """Whether `from_dict` read this metadata from a stored document it had to upgrade, + so the store holds an invalid document until this metadata is stored.""" def __init__( self, @@ -678,6 +681,7 @@ def from_dict(cls, data: dict[str, JSON], *, path: str | None = None) -> Self: storage_transformers=_data_typed.get("storage_transformers", ()), # type: ignore[arg-type] ) warn_readings(readings, path) + object.__setattr__(metadata, "_stored_document_upgraded", bool(readings)) return metadata def to_dict(self) -> dict[str, JSON]: diff --git a/tests/test_array_stateful.py b/tests/test_array_stateful.py index 0ea4da0167..7ce4780e5d 100644 --- a/tests/test_array_stateful.py +++ b/tests/test_array_stateful.py @@ -30,6 +30,7 @@ RuleBasedStateMachine, initialize, invariant, + precondition, rule, ) @@ -46,6 +47,7 @@ ] DTYPE = np.dtype("int16") +METADATA_KEYS = (".zarray", ".zattrs", "zarr.json") MAX_SIDE = 6 @@ -67,6 +69,10 @@ def _rectilinear_dim(extent: int) -> st.SearchStrategy[int | list[int]]: return steps | edges +async def _list(store: MemoryStore, prefix: str) -> list[str]: + return [key async for key in store.list_prefix(prefix)] + + class ArrayLifecycle(RuleBasedStateMachine): def __init__(self) -> None: super().__init__() @@ -136,6 +142,11 @@ def create(self, data: st.DataObject) -> None: stored_zero = data.draw(st.sampled_from([0, False]), label="stored zero") self._rewrite_stored_chunks(zarr_format, self.legacy_axes, stored_zero) event("legacy zero chunk size") + else: + # Re-saving valid metadata, as the warning tells users to, changes nothing. + arr = self._open() + arr.update_attributes({}) + assert self._open().metadata == arr.metadata def _rewrite_stored_chunks( self, zarr_format: Literal[2, 3], axes: list[int], value: Any @@ -233,9 +244,11 @@ def append(self, data: st.DataObject) -> None: @rule(data=st.data()) def resize(self, data: st.DataObject) -> None: arr = self._open() - new_shape = data.draw( - st.tuples(*(st.integers(0, MAX_SIDE) for _ in self.shape)), label="new shape" - ) + extents = [st.integers(0, MAX_SIDE) for _ in self.shape] + for axis in self.legacy_axes: + # What a user of an older release did next: grow an axis stored with chunk size 0. + extents[axis] = st.integers(self.shape[axis] + 1, MAX_SIDE + 1) | extents[axis] + new_shape = data.draw(st.tuples(*extents), label="new shape") note(f"resize {self.shape} -> {new_shape}") if any(new_shape[axis] > self.shape[axis] for axis in self.legacy_axes): event("grow an axis stored with chunk size 0") @@ -256,11 +269,14 @@ def write(self, data: st.DataObject) -> None: note(f"write {region}") arr[region] = values self._model_write(arr, region, values) + # Writing chunks first stores the metadata they are written under. + self.legacy_axes = [] + @precondition(lambda self: self.legacy_axes) @rule() def resave_metadata(self) -> None: - """What the warning for an invalid stored chunk size tells users to do. It stores - the metadata as read, which is a no-op for valid metadata.""" + """What the warning for an invalid stored chunk size tells users to do: store + the metadata as read.""" arr = self._open() read = arr.metadata arr.update_attributes({}) @@ -271,6 +287,14 @@ def teardown(self) -> None: self._rectilinear.__exit__(None, None, None) # ------------------------------------------------------------ invariants + @invariant() + def no_chunk_under_an_invalid_document(self) -> None: + """While the stored document is still one that is upgraded on read, which other + readers may reject or read differently, no chunk is stored under it.""" + if self.legacy_axes: + keys = sync(_list(self.store, f"{self.path}/")) + assert set(keys) <= {f"{self.path}/{name}" for name in METADATA_KEYS}, keys + @invariant() def matches_model(self) -> None: arr = self._open() diff --git a/tests/test_metadata/test_upgrades.py b/tests/test_metadata/test_upgrades.py index fabf22b1b8..b8e5ea3cc7 100644 --- a/tests/test_metadata/test_upgrades.py +++ b/tests/test_metadata/test_upgrades.py @@ -2,6 +2,7 @@ from __future__ import annotations +import asyncio import json import re import warnings @@ -20,6 +21,7 @@ upgrade_array_document, ) from zarr.core.metadata.v3 import RectilinearChunkGridMetadata, RegularChunkGridMetadata +from zarr.core.sync import sync from zarr.dtype import Int16 from zarr.errors import ZarrUserWarning @@ -345,6 +347,64 @@ def store_legacy(doc: dict[str, Any]) -> None: np.testing.assert_array_equal(reopened[...], np.concatenate([data, block])) +@pytest.mark.parametrize( + ("zarr_format", "shape", "inner", "expected"), + [(2, (3,), None, (3,)), (3, (3,), None, (3,)), (3, (10,), (4,), (12,))], + ids=["v2", "v3", "v3-sharded"], +) +@pytest.mark.parametrize("api", ["sync", "async", "async-concurrent"]) +def test_write_stores_upgraded_metadata_first( + tmp_path: Path, + zarr_format: Literal[2, 3], + shape: tuple[int, ...], + inner: tuple[int, ...] | None, + expected: tuple[int, ...], + api: str, +) -> None: + """Writing chunks to an array read from an upgraded document first stores the + upgraded metadata, so readers that do not upgrade (or read it differently) see + the chunks the write stored.""" + path = tmp_path / "legacy.zarr" + zarr.create_array( + store=path, + shape=shape, + chunks=inner or expected, + shards=expected if inner else None, + dtype="int16", + fill_value=0, + zarr_format=zarr_format, + ) + _rewrite_doc( + path, + zarr_format, + lambda doc: ( + doc.update(chunks=[0]) + if zarr_format == 2 + else doc["chunk_grid"]["configuration"].update(chunk_shape=[0]) + ), + ) + data = np.arange(1, shape[0] + 1, dtype="int16") + with pytest.warns(ZarrUserWarning, match="is read as"): + arr = zarr.open_array(store=path, mode="r+") + if api == "sync": + arr[:] = data + elif api == "async": + sync(arr.async_array.setitem(slice(None), data)) + else: + + async def write_twice() -> None: + # Both writes find the metadata not yet stored, and both store it. + await asyncio.gather(*(arr.async_array.setitem(slice(None), data) for _ in "ab")) + + sync(write_twice()) + + with warnings.catch_warnings(): + warnings.simplefilter("error", ZarrUserWarning) + reopened = zarr.open_array(store=path, mode="r") + assert (reopened.shards or reopened.chunks) == expected + np.testing.assert_array_equal(reopened[...], data) + + @pytest.mark.filterwarnings("ignore:Consolidated metadata is currently not part:UserWarning") @pytest.mark.parametrize("zarr_format", [2, 3]) def test_legacy_chunk_size_consolidated(tmp_path: Path, zarr_format: Literal[2, 3]) -> None: From 121b6b2f84b559590cc25f21e3d2988c0b3bd247 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Sat, 26 Sep 2026 05:59:03 +0200 Subject: [PATCH 22/41] refactor(metadata): name an array one way in upgrade warnings Warnings about an upgraded document name the array by its store path, `str(StorePath(store, path))`, on every path that reads one: opening an array, `AsyncArray.from_dict`, a group read with or without consolidated metadata, and consolidated members, which were named relative to the group. `GroupMetadata.from_dict` and `ConsolidatedMetadata.from_dict` take the group's path for that. The consolidated test now also opens the group with `use_consolidated=False` and checks each array's name, so dropping the path on either read path fails a test. Assisted-by: ClaudeCode:claude-opus-5-5 Co-Authored-By: Claude Opus 5.5 --- src/zarr/core/array.py | 2 +- src/zarr/core/group.py | 31 +++++++++++++++++----------- tests/test_metadata/test_upgrades.py | 21 ++++++++++++------- 3 files changed, 33 insertions(+), 21 deletions(-) diff --git a/src/zarr/core/array.py b/src/zarr/core/array.py index 5f4cddb9b4..2f33b2f7ed 100644 --- a/src/zarr/core/array.py +++ b/src/zarr/core/array.py @@ -765,7 +765,7 @@ def from_dict( ValueError If the dictionary data is invalid or incompatible with either Zarr format 2 or 3 array creation. """ - metadata = parse_array_metadata(data) + metadata = parse_array_metadata(data, str(store_path)) return cls(metadata=metadata, store_path=store_path) @classmethod diff --git a/src/zarr/core/group.py b/src/zarr/core/group.py index 78071300f0..d6cc572465 100644 --- a/src/zarr/core/group.py +++ b/src/zarr/core/group.py @@ -163,7 +163,9 @@ def to_dict(self) -> dict[str, JSON]: } @classmethod - def from_dict(cls, data: dict[str, JSON]) -> ConsolidatedMetadata: + def from_dict(cls, data: dict[str, JSON], *, path: str | None = None) -> ConsolidatedMetadata: + """Read consolidated metadata, naming each member by its path under the group at + `path` (or relative to that group, when `path` is not given) in warnings.""" data = dict(data) kind = data.get("kind") @@ -177,6 +179,7 @@ def from_dict(cls, data: dict[str, JSON]) -> ConsolidatedMetadata: metadata: dict[str, ArrayV2Metadata | ArrayV3Metadata | GroupMetadata] = {} if raw_metadata: for k, v in raw_metadata.items(): + member = k if path is None else _join_paths([path, k]) if not isinstance(v, dict): raise TypeError( f"Invalid value for metadata items. key='{k}', type='{type(v).__name__}'" @@ -188,16 +191,16 @@ def from_dict(cls, data: dict[str, JSON]) -> ConsolidatedMetadata: if zarr_format == 3: node_type = parse_node_type(v.get("node_type", None)) if node_type == "group": - metadata[k] = GroupMetadata.from_dict(v) + metadata[k] = GroupMetadata.from_dict(v, path=member) elif node_type == "array": - metadata[k] = ArrayV3Metadata.from_dict(v, path=k) + metadata[k] = ArrayV3Metadata.from_dict(v, path=member) else: assert_never(node_type) elif zarr_format == 2: if "shape" in v: - metadata[k] = ArrayV2Metadata.from_dict(v, path=k) + metadata[k] = ArrayV2Metadata.from_dict(v, path=member) else: - metadata[k] = GroupMetadata.from_dict(v) + metadata[k] = GroupMetadata.from_dict(v, path=member) else: assert_never(zarr_format) @@ -415,7 +418,9 @@ def __init__( object.__setattr__(self, "consolidated_metadata", consolidated_metadata) @classmethod - def from_dict(cls, data: dict[str, Any]) -> GroupMetadata: + def from_dict(cls, data: dict[str, Any], *, path: str | None = None) -> GroupMetadata: + """Read a stored group document; `path` names the group in warnings about the + consolidated metadata it holds.""" data = dict(data) node_type = data.pop("node_type", None) if node_type not in ("group", None): @@ -424,7 +429,9 @@ def from_dict(cls, data: dict[str, Any]) -> GroupMetadata: ) consolidated_metadata = data.pop("consolidated_metadata", None) if consolidated_metadata: - data["consolidated_metadata"] = ConsolidatedMetadata.from_dict(consolidated_metadata) + data["consolidated_metadata"] = ConsolidatedMetadata.from_dict( + consolidated_metadata, path=path + ) zarr_format = data.get("zarr_format") if zarr_format == 2 or zarr_format is None: @@ -695,7 +702,7 @@ def from_dict( msg = f"Node type in metadata ({node_type}) is not 'group'" raise GroupNotFoundError(msg) return cls( - metadata=GroupMetadata.from_dict(data), + metadata=GroupMetadata.from_dict(data, path=str(store_path)), store_path=store_path, ) @@ -3505,7 +3512,7 @@ async def _read_metadata_v3(store: Store, path: str) -> ArrayV3Metadata | GroupM if zarr_json_bytes is None: raise FileNotFoundError(path) return _build_metadata_v3( - buffer_to_json_object(zarr_json_bytes), path=_join_paths([str(store), path]) + buffer_to_json_object(zarr_json_bytes), path=str(StorePath(store, path)) ) @@ -3541,7 +3548,7 @@ async def _read_metadata_v2(store: Store, path: str) -> ArrayV2Metadata | GroupM else: zmeta = buffer_to_json_object(zgroup_bytes) - return _build_metadata_v2(zmeta, zattrs, path=_join_paths([str(store), path])) + return _build_metadata_v2(zmeta, zattrs, path=str(StorePath(store, path))) async def _read_group_metadata_v2(store: Store, path: str) -> GroupMetadata: @@ -3585,7 +3592,7 @@ def _build_metadata_v3( case {"node_type": "array"}: return ArrayV3Metadata.from_dict(zarr_json, path=path) case {"node_type": "group"}: - return GroupMetadata.from_dict(zarr_json) + return GroupMetadata.from_dict(zarr_json, path=path) case _: # pragma: no cover raise ValueError( "invalid value for `node_type` key in metadata document" @@ -3602,7 +3609,7 @@ def _build_metadata_v2( case {"shape": _}: return ArrayV2Metadata.from_dict(zarr_json | {"attributes": attrs_json}, path=path) case _: # pragma: no cover - return GroupMetadata.from_dict(zarr_json | {"attributes": attrs_json}) + return GroupMetadata.from_dict(zarr_json | {"attributes": attrs_json}, path=path) @overload diff --git a/tests/test_metadata/test_upgrades.py b/tests/test_metadata/test_upgrades.py index b8e5ea3cc7..9256463f61 100644 --- a/tests/test_metadata/test_upgrades.py +++ b/tests/test_metadata/test_upgrades.py @@ -408,9 +408,9 @@ async def write_twice() -> None: @pytest.mark.filterwarnings("ignore:Consolidated metadata is currently not part:UserWarning") @pytest.mark.parametrize("zarr_format", [2, 3]) def test_legacy_chunk_size_consolidated(tmp_path: Path, zarr_format: Literal[2, 3]) -> None: - """Consolidated metadata goes through the same upgrade, with one warning naming each - array; re-saving the arrays and consolidating again leaves a group that opens - without a warning.""" + """Consolidated metadata goes through the same upgrade as the arrays' own documents, + with one warning naming each array by its path; re-saving the arrays and + consolidating again leaves a group that opens without a warning.""" path = tmp_path / "group.zarr" group = zarr.open_group(path, mode="w", zarr_format=zarr_format) names = ("a", "b") @@ -439,11 +439,16 @@ def test_legacy_chunk_size_consolidated(tmp_path: Path, zarr_format: Literal[2, ].update(chunk_shape=[0]), ) - with pytest.warns(ZarrUserWarning, match="zarr.consolidate_metadata") as record: - group = zarr.open_group(path, mode="r+") - assert sorted(str(w.message).split(":")[0] for w in record) == ["Array 'a'", "Array 'b'"] - for name in names: - array = group[name] + for use_consolidated in (True, False): + with warnings.catch_warnings(record=True) as record: + warnings.simplefilter("always", ZarrUserWarning) + group = zarr.open_group(path, mode="r+", use_consolidated=use_consolidated) + arrays = [group[name] for name in names] + assert all("zarr.consolidate_metadata" in str(w.message) for w in record) + assert sorted(str(w.message).split(": ")[0] for w in record) == [ + f"Array '{group.store_path / name}'" for name in names + ] + for array in arrays: assert isinstance(array, zarr.Array) assert array.chunks == (1,) array.update_attributes({}) From c1353bf2a446eed53034446592b0804c993d90c1 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Sat, 26 Sep 2026 05:59:15 +0200 Subject: [PATCH 23/41] refactor(metadata): one rule for bare sizes and edge lists; one noun in messages - `_validate_chunk_shapes` decides "edge list or bare size" with one `match`: a list or tuple is an edge list, anything else is a bare size checked by `parse_chunk_edge`, so a string dimension such as `"10"` is rejected as not an int instead of being iterated per character. Stored rectilinear `from_dict` makes the same JSON split and passes the dimension to `expand_rle`, so an invalid edge or RLE count inside a list names its dimension. - Messages say "dimension" throughout, lowercase after the colon ("Dimension 0: chunk edge length must be >= 1, got 0"); the upgrade warnings read "0 in dimension 0 as one chunk spanning the dimension". - The upgrades test for JSON arrays with `list` only (a stored document holds no tuples), and `_read_chunk_size` says what `span=None` means. - `ArrayV2Metadata.from_dict` copies the document once. - The rejected stored chunk shapes get one test per error case. Assisted-by: ClaudeCode:claude-opus-5-5 Co-Authored-By: Claude Opus 5.5 --- src/zarr/core/common.py | 20 +++---- src/zarr/core/metadata/upgrades.py | 15 +++--- src/zarr/core/metadata/v2.py | 8 +-- src/zarr/core/metadata/v3.py | 19 +++---- tests/test_metadata/test_upgrades.py | 78 ++++++++++++++++++---------- 5 files changed, 85 insertions(+), 55 deletions(-) diff --git a/src/zarr/core/common.py b/src/zarr/core/common.py index cddbf91912..feb8eb1cac 100644 --- a/src/zarr/core/common.py +++ b/src/zarr/core/common.py @@ -276,11 +276,12 @@ def _default_zarr_format() -> ZarrFormat: return cast("ZarrFormat", int(zarr_config.get("default_zarr_format", 3))) -def _parse_positive_int(value: object, name: str) -> int: +def _parse_positive_int(value: object, name: str, axis: int | None) -> int: + subject = name[0].upper() + name[1:] if axis is None else f"Dimension {axis}: {name}" if isinstance(value, bool) or not isinstance(value, int): - raise TypeError(f"{name} must be an int, got {value!r}") + raise TypeError(f"{subject} must be an int, got {value!r}") if value < 1: - raise ValueError(f"{name} must be >= 1, got {value!r}") + raise ValueError(f"{subject} must be >= 1, got {value!r}") return value @@ -290,8 +291,7 @@ def parse_chunk_edge(size: object, axis: int | None = None) -> int: This is the one rule for chunk edge lengths in metadata: bare chunk sizes, explicit edges and run-length encoded sizes. `axis`, when given, is named in the error. """ - where = "" if axis is None else f"Dimension {axis}: " - return _parse_positive_int(size, f"{where}Chunk edge length") + return _parse_positive_int(size, "chunk edge length", axis) def parse_chunk_shape(data: object) -> tuple[int, ...]: @@ -301,8 +301,9 @@ def parse_chunk_shape(data: object) -> tuple[int, ...]: return tuple(parse_chunk_edge(size, axis) for axis, size in enumerate(data)) -def expand_rle(data: Sequence[object]) -> list[int]: - """Expand a mixed array of bare integers and RLE pairs. +def expand_rle(data: Sequence[object], axis: int | None = None) -> list[int]: + """Expand a mixed array of bare integers and RLE pairs, the edges of dimension + `axis` (named in errors, when given). Per the rectilinear chunk grid spec, each element can be: - a bare integer (an explicit edge length) @@ -314,9 +315,10 @@ def expand_rle(data: Sequence[object]) -> list[int]: if len(item) != 2: raise ValueError(f"RLE entries must be an integer or [size, count], got {item}") size, count = item - result.extend([parse_chunk_edge(size)] * _parse_positive_int(count, "RLE repeat count")) + repeat = _parse_positive_int(count, "RLE repeat count", axis) + result.extend([parse_chunk_edge(size, axis)] * repeat) else: - result.append(parse_chunk_edge(item)) + result.append(parse_chunk_edge(item, axis)) return result diff --git a/src/zarr/core/metadata/upgrades.py b/src/zarr/core/metadata/upgrades.py index 0cdac3151d..1cc6916c47 100644 --- a/src/zarr/core/metadata/upgrades.py +++ b/src/zarr/core/metadata/upgrades.py @@ -49,9 +49,9 @@ def warn_readings(readings: Sequence[str], path: str | None) -> None: warnings.warn(f"{subject}{' '.join(readings)} {RESAVE_HINT}", ZarrUserWarning, stacklevel=2) -def _is_int_list(value: object) -> TypeGuard[list[int] | tuple[int, ...]]: +def _is_int_list(value: object) -> TypeGuard[list[int]]: """Whether `value` is a JSON array of integers (JSON `false` and `true` count).""" - return isinstance(value, list | tuple) and all(isinstance(v, int) for v in value) + return isinstance(value, list) and all(isinstance(v, int) for v in value) def _read_chunk_size(size: JSON, span: int | None, unit: int) -> tuple[int, str | None] | None: @@ -61,7 +61,8 @@ def _read_chunk_size(size: JSON, span: int | None, unit: int) -> tuple[int, str entry cannot be read, which leaves it for the metadata constructors to reject. A JSON int >= 1 is kept, JSON `true` is read as 1, and 0 or JSON `false` is read as one chunk spanning the axis of length `span`, a multiple of `unit` (the inner chunk - size of a shard), when the span is known. + size of a shard). `span` is `None` where no stored 0 is known, as in the inner + chunk shape of a sharding codec: 0 is then left for the constructors to reject. """ match size: case True: @@ -70,7 +71,7 @@ def _read_chunk_size(size: JSON, span: int | None, unit: int) -> tuple[int, str return size, None case int() if size == 0 and span is not None: edge = full_span_chunk_size(span, unit) - how = f"one chunk spanning the axis ({edge})" + how = f"one chunk spanning the dimension ({edge})" if span > 0: how += ( ", and as no chunk can be stored under a chunk size of 0, the array " @@ -89,7 +90,7 @@ def _read_chunk_shape( Returns the chunk shape and, if an entry is invalid, a sentence saying how the `name` was read; `None` if it cannot be read. """ - if not (isinstance(stored, list | tuple) and len(stored) == len(spans)): + if not (isinstance(stored, list) and len(stored) == len(spans)): return None edges: list[int] = [] readings: list[str] = [] @@ -101,7 +102,7 @@ def _read_chunk_shape( edge, how = read edges.append(edge) if how is not None: - readings.append(f"{json.dumps(size)} on axis {axis} as {how}") + readings.append(f"{json.dumps(size)} in dimension {axis} as {how}") if not readings: return edges, None return edges, ( @@ -124,7 +125,7 @@ def _sharding_codec(doc: ArrayDocument) -> tuple[Sequence[JSON], int, Mapping[st """The codec list of a Zarr format 3 array document, with the position and the configuration of its sharding codec, if it has one.""" codecs = doc.get("codecs") - if isinstance(codecs, list | tuple): + if isinstance(codecs, list): for index, codec in enumerate(codecs): if isinstance(codec, Mapping) and codec.get("name") == "sharding_indexed": configuration = codec.get("configuration") diff --git a/src/zarr/core/metadata/v2.py b/src/zarr/core/metadata/v2.py index b5701b3e78..2a1bbeee73 100644 --- a/src/zarr/core/metadata/v2.py +++ b/src/zarr/core/metadata/v2.py @@ -155,8 +155,8 @@ def from_dict(cls, data: dict[str, Any], *, path: str | None = None) -> ArrayV2M """Read a stored `.zarray` document (with its attributes). An invalid document that `zarr.core.metadata.upgrades` can read warns, naming the array at `path`.""" upgraded, readings = upgrade_array_document(data, V2_ARRAY_UPGRADES) - data = dict(upgraded) - _data = data.copy() + # a new dict, because we are modifying it + _data: dict[str, Any] = dict(upgraded) # Check that the zarr_format attribute is correct. _ = parse_zarr_format(_data.pop("zarr_format")) @@ -164,7 +164,7 @@ def from_dict(cls, data: dict[str, Any], *, path: str | None = None) -> ArrayV2M # which could be in filters or as a compressor. # we will reference a hard-coded collection of object codec ids for this search. - _filters, _compressor = (data.get("filters"), data.get("compressor")) + _filters, _compressor = (_data.get("filters"), _data.get("compressor")) if _filters is not None: _filters = cast("tuple[dict[str, JSON], ...]", _filters) object_codec_id = get_object_codec_id(tuple(_filters) + (_compressor,)) @@ -173,7 +173,7 @@ def from_dict(cls, data: dict[str, Any], *, path: str | None = None) -> ArrayV2M # we add a layer of indirection here around the dtype attribute of the array metadata # because we also need to know the object codec id, if any, to resolve the data type dtype_spec: DTypeSpec_V2 = { - "name": data["dtype"], + "name": _data["dtype"], "object_codec_id": object_codec_id, } dtype = get_data_type_from_json(dtype_spec, zarr_format=2) diff --git a/src/zarr/core/metadata/v3.py b/src/zarr/core/metadata/v3.py index ab5ddf3d2f..4eb8dd352c 100644 --- a/src/zarr/core/metadata/v3.py +++ b/src/zarr/core/metadata/v3.py @@ -239,13 +239,14 @@ def _validate_chunk_shapes( """ result: list[int | tuple[int, ...]] = [] for dim_idx, dim_spec in enumerate(chunk_shapes): - if isinstance(dim_spec, Iterable): - edges = tuple(parse_chunk_edge(edge, dim_idx) for edge in dim_spec) - if not edges: - raise ValueError(f"Dimension {dim_idx} has no chunk edges.") - result.append(edges) - else: - result.append(parse_chunk_edge(dim_spec, dim_idx)) + match dim_spec: + case list() | tuple(): + edges = tuple(parse_chunk_edge(edge, dim_idx) for edge in dim_spec) + if not edges: + raise ValueError(f"Dimension {dim_idx} has no chunk edges.") + result.append(edges) + case _: + result.append(parse_chunk_edge(dim_spec, dim_idx)) return tuple(result) @@ -363,8 +364,8 @@ def from_dict(cls, data: RectilinearChunkGridMetadataJSON) -> Self: # type: ign validate_rectilinear_kind(configuration.get("kind")) raw_shapes = configuration["chunk_shapes"] parsed = [ - tuple(expand_rle(dim_spec)) if isinstance(dim_spec, list) else dim_spec - for dim_spec in raw_shapes + tuple(expand_rle(dim_spec, axis)) if isinstance(dim_spec, list) else dim_spec + for axis, dim_spec in enumerate(raw_shapes) ] return cls(chunk_shapes=tuple(parsed)) diff --git a/tests/test_metadata/test_upgrades.py b/tests/test_metadata/test_upgrades.py index 9256463f61..b976660eb2 100644 --- a/tests/test_metadata/test_upgrades.py +++ b/tests/test_metadata/test_upgrades.py @@ -100,32 +100,40 @@ def _chunk_shapes(metadata: ArrayV2Metadata | ArrayV3Metadata) -> tuple[Any, Any ( _v2_doc([0, 4], [0, 4]), ((1, 4), None), - r"0 on axis 0 as one chunk spanning the axis \(1\)\.", + r"0 in dimension 0 as one chunk spanning the dimension \(1\)\.", ), ( _v3_doc([0], [False]), ((1,), None), - r"false on axis 0 as one chunk spanning the axis \(1\)\.", + r"false in dimension 0 as one chunk spanning the dimension \(1\)\.", ), - (_v2_doc([5], [True]), ((1,), None), "true on axis 0 as 1"), + (_v2_doc([5], [True]), ((1,), None), "true in dimension 0 as 1"), ( _v3_doc([5, 4], [True, 4]), ((1, 4), None), - r"\[true, 4\] is invalid.*true on axis 0 as 1", + r"\[true, 4\] is invalid.*true in dimension 0 as 1", ), ( _v2_doc([3], [0]), ((3,), None), - r"spanning the axis \(3\), and .* holds only its fill value", + r"spanning the dimension \(3\), and .* holds only its fill value", ), - (_v3_doc([4, 3], [4, 0]), ((4, 3), None), "0 on axis 1 as .* holds only its fill value"), - (_v3_doc([0], [0], inner=[4]), ((4,), (4,)), r"spanning the axis \(4\)\."), + ( + _v3_doc([4, 3], [4, 0]), + ((4, 3), None), + "0 in dimension 1 as .* holds only its fill value", + ), + (_v3_doc([0], [0], inner=[4]), ((4,), (4,)), r"spanning the dimension \(4\)\."), ( _v3_doc([10], [0], inner=[4]), ((12,), (4,)), - r"spanning the axis \(12\), and .* holds only its fill value", + r"spanning the dimension \(12\), and .* holds only its fill value", + ), + ( + _v3_doc([0, 3], [0, 3], inner=[2, 3]), + ((2, 3), (2, 3)), + r"spanning the dimension \(2\)\.", ), - (_v3_doc([0, 3], [0, 3], inner=[2, 3]), ((2, 3), (2, 3)), r"spanning the axis \(2\)\."), ( _v3_doc([5], [True], inner=[True]), ((1,), (1,)), @@ -203,24 +211,40 @@ def test_invalid_upgraded_document_raises_without_warning(doc: dict[str, JSON], metadata_cls.from_dict(doc) -@pytest.mark.parametrize( - ("doc", "error"), - [ - (_v2_doc([4], [-1]), "Dimension 0: Chunk edge length must be >= 1, got -1"), - (_v3_doc([4, 4], [0]), "Dimension 0: Chunk edge length must be >= 1, got 0"), - (_v3_doc([4], [4.0]), "Dimension 0: Chunk edge length must be an int, got 4.0"), - (_v3_doc([4], [0], inner=[0]), "Dimension 0: Chunk edge length must be >= 1, got 0"), - ], - ids=["negative", "ndim-mismatch", "float", "sharded-inner-zero"], -) -def test_stored_chunk_shape_not_upgraded(doc: dict[str, JSON], error: str) -> None: - """Invalid chunk sizes no known writer stored, and chunk shapes with the wrong - number of axes, are not upgraded, only rejected.""" +def _read_strictly(doc: dict[str, JSON]) -> ArrayV2Metadata | ArrayV3Metadata: + """Read `doc`, failing on any warning that it was upgraded.""" metadata_cls = ArrayV2Metadata if doc["zarr_format"] == 2 else ArrayV3Metadata with warnings.catch_warnings(): warnings.simplefilter("error", ZarrUserWarning) - with pytest.raises((TypeError, ValueError), match=re.escape(error)): - metadata_cls.from_dict(doc) + return metadata_cls.from_dict(doc) + + +def test_stored_negative_chunk_size_rejected() -> None: + """No known writer stored a negative chunk size: it is rejected, not upgraded.""" + with pytest.raises(ValueError, match="^Dimension 0: chunk edge length must be >= 1, got -1$"): + _read_strictly(_v2_doc([4], [-1])) + + +def test_stored_chunk_shape_ndim_mismatch_rejected() -> None: + """A chunk shape with the wrong number of dimensions is not upgraded, so its 0 is + rejected.""" + with pytest.raises(ValueError, match="^Dimension 0: chunk edge length must be >= 1, got 0$"): + _read_strictly(_v3_doc([4, 4], [0])) + + +def test_stored_float_chunk_size_rejected() -> None: + """No known writer stored a float chunk size: it is rejected, not upgraded.""" + with pytest.raises( + TypeError, match=r"^Dimension 0: chunk edge length must be an int, got 4\.0$" + ): + _read_strictly(_v3_doc([4], [4.0])) + + +def test_stored_zero_inner_chunk_size_rejected() -> None: + """No known writer stored an inner chunk size of 0, and no span defines one: it is + rejected, not upgraded.""" + with pytest.raises(ValueError, match="^Dimension 0: chunk edge length must be >= 1, got 0$"): + _read_strictly(_v3_doc([4], [0], inner=[0])) def test_v2_constructor_rejects_chunks_of_wrong_length() -> None: @@ -264,7 +288,7 @@ def _rectilinear_from_dict(chunk_shapes: list[Any]) -> RectilinearChunkGridMetad def test_metadata_rejects_non_int_chunk_edge(site: str, size: object) -> None: """Metadata built in code takes chunk edge lengths as `int`s only, everywhere.""" with pytest.raises( - TypeError, match=re.escape(f"Chunk edge length must be an int, got {size!r}") + TypeError, match=re.escape(f"Dimension 0: chunk edge length must be an int, got {size!r}") ): CHUNK_EDGE_SITES[site](size) @@ -276,7 +300,9 @@ def test_metadata_rejects_chunk_edge_below_one(site: str, size: int) -> None: without a warning.""" with warnings.catch_warnings(): warnings.simplefilter("error", ZarrUserWarning) - with pytest.raises(ValueError, match=f"Chunk edge length must be >= 1, got {size}"): + with pytest.raises( + ValueError, match=f"Dimension 0: chunk edge length must be >= 1, got {size}" + ): CHUNK_EDGE_SITES[site](size) From f081f1b6a2174ec542bdbe835b1d43c528499257 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Sat, 26 Sep 2026 10:12:54 +0200 Subject: [PATCH 24/41] fix(array): store the upgrade of the current stored document before writing A handle read from an upgraded document upserts the upgrade of what the store holds on its first non-empty chunk write: a stale handle no longer rolls back newer metadata, and an empty write stores nothing. consolidate_metadata does the same for each upgraded member before writing the consolidated document. Both go through two internal primitives in zarr.core.metadata.io: diff_documents compares a node's stored documents with those its metadata would store, value by value, and upsert_metadata encodes first, then stores only the documents that differ and returns the changes. The upgraded-document flag is a ClassVar set on the instance by one helper, mark_upgraded, which also warns; the upgrades are keyed by Zarr format; the Metadata.to_dict and V2 from_dict changes are reverted. _append goes through AsyncArray.setitem and _setitem is removed. A chunk shape that is not a list or tuple is rejected as a whole, and every expand_rle error names its dimension. Assisted-by: ClaudeCode:claude-opus-5-5 Co-Authored-By: Claude Opus 5.5 --- changes/4334.bugfix.md | 4 +- src/zarr/abc/metadata.py | 5 +- src/zarr/api/asynchronous.py | 11 +- src/zarr/codecs/sharding.py | 2 +- src/zarr/core/array.py | 89 ++++-------- src/zarr/core/common.py | 20 ++- src/zarr/core/metadata/io.py | 95 ++++++++++++- src/zarr/core/metadata/upgrades.py | 33 +++-- src/zarr/core/metadata/v2.py | 22 ++- src/zarr/core/metadata/v3.py | 34 ++--- tests/test_metadata/test_io.py | 199 +++++++++++++++++++++++++++ tests/test_metadata/test_upgrades.py | 131 ++++++++++++++++-- tests/test_unified_chunk_grid.py | 16 +++ 13 files changed, 515 insertions(+), 146 deletions(-) create mode 100644 tests/test_metadata/test_io.py diff --git a/changes/4334.bugfix.md b/changes/4334.bugfix.md index 4c55701757..460ba6ece8 100644 --- a/changes/4334.bugfix.md +++ b/changes/4334.bugfix.md @@ -1,5 +1,5 @@ A chunk edge length is now always at least 1, while an array extent may be 0. Every spelling of one chunk spanning an axis (`chunks=-1`, `chunks=False`, `chunks="auto"`, `shards=-1`, `shards=False`) gives chunk size 1 on a zero-length axis, and a shard spanning an axis is a multiple of the inner chunk size, so `shards=-1` no longer fails when the axis length is not a multiple of it. Rectilinear chunk grids can be created on a zero-length dimension with a non-empty list of positive chunk sizes, which are kept for later growth. -Metadata built in code is strict about chunk edge lengths: `ArrayV2Metadata`, `RegularChunkGridMetadata`, `RectilinearChunkGridMetadata` and `ShardingCodec` take them as Python `int`s of at least 1, so a size of 0 now raises a `ValueError`, and a `bool`, a float or a NumPy integer (including a scalar `chunks=np.int64(5)` for `ArrayV2Metadata`) raises a `TypeError`; `FixedDimension(size=0, ...)` raises a `ValueError`. Stored rectilinear chunk grids with integral floats (`10.0`) are rejected too. The array creation functions still accept NumPy integers in `chunks=` and `shards=`. +Metadata built in code is strict about chunk edge lengths: `ArrayV2Metadata`, `RegularChunkGridMetadata`, `RectilinearChunkGridMetadata` and `ShardingCodec` take them as Python `int`s of at least 1, so a size of 0 now raises a `ValueError`, and a `bool`, a float or a NumPy integer (including a scalar `chunks=np.int64(5)` for `ArrayV2Metadata`) raises a `TypeError`, as does a chunk shape that is not a list or tuple; `FixedDimension(size=0, ...)` raises a `ValueError`. Stored rectilinear chunk grids with integral floats (`10.0`) are rejected too. The array creation functions still accept NumPy integers in `chunks=` and `shards=`. -Stored metadata with a regular chunk size of 0 or JSON `false` now opens with a `ZarrUserWarning` and is read as one chunk spanning the axis (a multiple of the inner chunk size for sharded arrays, which previously failed to open). zarr-python wrote such sizes for arrays created with a zero-length axis until 3.4, and, from 2.18.7 to 3.2.1, for an explicit chunk size of 0 or `False` on an axis of any length, as in `chunks=(0,)` (those arrays could store no data). JSON `true`, as zarr-python 3.0 and 3.2 wrote for `chunks=(True,)`, is read as 1, in the chunk grid and in the inner chunk shape of a sharding codec. The warning names the array and says how to store valid metadata: `array.update_attributes({})`, then `zarr.consolidate_metadata` if the metadata is consolidated. Writing data to such an array first stores the upgraded metadata, so that other readers find the chunks written. +Stored metadata with a regular chunk size of 0 or JSON `false` now opens with a `ZarrUserWarning` and is read as one chunk spanning the axis (a multiple of the inner chunk size for sharded arrays, which previously failed to open). zarr-python wrote such sizes for arrays created with a zero-length axis until 3.4, and, from 2.18.7 to 3.2.1, for an explicit chunk size of 0 or `False` on an axis of any length, as in `chunks=(0,)` (those arrays could store no data). JSON `true`, as zarr-python 3.0 and 3.2 wrote for `chunks=(True,)`, is read as 1, in the chunk grid and in the inner chunk shape of a sharding codec. The warning names the array and says how to store valid metadata: `array.update_attributes({})`, then `zarr.consolidate_metadata` if the metadata is consolidated. Writing data to such an array (other than an empty selection) first stores the upgrade of the metadata the store then holds, unless another writer has stored valid metadata since, so that other readers find the chunks written; `zarr.consolidate_metadata` likewise stores the upgraded metadata of each such array before the consolidated metadata. diff --git a/src/zarr/abc/metadata.py b/src/zarr/abc/metadata.py index 1104111b63..a56f986645 100644 --- a/src/zarr/abc/metadata.py +++ b/src/zarr/abc/metadata.py @@ -20,13 +20,10 @@ def to_dict(self) -> dict[str, JSON]: Recursively serialize this model to a dictionary. This method inspects the fields of self and calls `x.to_dict()` for any fields that are instances of `Metadata`. Sequences of `Metadata` are similarly recursed into, and - the output of that recursion is collected in a list. Fields declared with - `compare=False` are not part of the document. + the output of that recursion is collected in a list. """ out_dict = {} for field in fields(self): - if not field.compare: - continue key = field.name value = getattr(self, key) if isinstance(value, Metadata): diff --git a/src/zarr/api/asynchronous.py b/src/zarr/api/asynchronous.py index 1fc10cdd1e..4832bafaba 100644 --- a/src/zarr/api/asynchronous.py +++ b/src/zarr/api/asynchronous.py @@ -231,10 +231,15 @@ async def consolidate_metadata( group = await AsyncGroup.open(store_path, zarr_format=zarr_format, use_consolidated=False) group.store_path.store._check_writable() - members_metadata = { - k: v.metadata - async for k, v in group.members(max_depth=None, use_consolidated_for_children=False) + members = { + k: v async for k, v in group.members(max_depth=None, use_consolidated_for_children=False) } + # Store the upgrade of every member document that had to be upgraded before the + # consolidated document that describes the upgraded metadata. + for member in members.values(): + if isinstance(member, AsyncArray): + await member._store_upgraded_document() + members_metadata = {k: v.metadata for k, v in members.items()} # While consolidating, we want to be explicit about when child groups # are empty by inserting an empty dict for consolidated_metadata.metadata for k, v in members_metadata.items(): diff --git a/src/zarr/codecs/sharding.py b/src/zarr/codecs/sharding.py index 37fe52c1a6..52d100bb91 100644 --- a/src/zarr/codecs/sharding.py +++ b/src/zarr/codecs/sharding.py @@ -458,7 +458,7 @@ class ShardingCodec( def __init__( self, *, - chunk_shape: Iterable[int], + chunk_shape: tuple[int, ...] | list[int], codecs: Iterable[Codec | dict[str, JSON]] = (BytesCodec(),), index_codecs: Iterable[Codec | dict[str, JSON]] = (BytesCodec(), Crc32cCodec()), index_location: ShardingCodecIndexLocation | IndexLocation = "end", diff --git a/src/zarr/core/array.py b/src/zarr/core/array.py index 2f33b2f7ed..64d6080838 100644 --- a/src/zarr/core/array.py +++ b/src/zarr/core/array.py @@ -118,7 +118,8 @@ ArrayV2MetadataDict, ArrayV3Metadata, ) -from zarr.core.metadata.io import save_metadata +from zarr.core.metadata.io import save_metadata, upsert_metadata +from zarr.core.metadata.upgrades import upgrade_array_document from zarr.core.metadata.v2 import ( CompressorLikev2, get_object_codec_id, @@ -1615,6 +1616,24 @@ async def _save_metadata(self, metadata: ArrayMetadata, ensure_parents: bool = F """ await save_metadata(self.store_path, metadata, ensure_parents=ensure_parents) + async def _store_upgraded_document(self) -> None: + """Store the upgrade of this array's current stored document, if it differs from + what the store holds. + + Only for metadata read from a document that had to be upgraded (see + `zarr.core.metadata.upgrades`). The document is read again, because the store may + hold a newer one than this handle's metadata: if that one needs no upgrade (the + array was re-saved or resized since), nothing is stored. Storing the same upgrade + twice is harmless, so concurrent callers need no coordination. + """ + if not self.metadata._stored_document_upgraded: + return + zarr_format = self.metadata.zarr_format + stored = await get_array_metadata(self.store_path, zarr_format=zarr_format) + upgraded, _ = upgrade_array_document(stored, zarr_format) + await upsert_metadata(self.store_path, parse_array_metadata(upgraded)) + object.__setattr__(self.metadata, "_stored_document_upgraded", False) + async def _set_selection( self, indexer: Indexer, @@ -1623,12 +1642,10 @@ async def _set_selection( prototype: BufferPrototype, fields: Fields | None = None, ) -> None: - if self.metadata._stored_document_upgraded: + if product(indexer.shape) > 0: # Chunks are about to be stored under the upgraded metadata, so store it # first: every reader of the store then agrees with them. - metadata = replace(self.metadata) - await self._save_metadata(metadata) - object.__setattr__(self, "metadata", metadata) + await self._store_upgraded_document() return await _set_selection( self.store_path, self.metadata, @@ -5827,58 +5844,6 @@ async def _set_selection( ) -async def _setitem( - store_path: StorePath, - metadata: ArrayMetadata, - codec_pipeline: CodecPipeline, - config: ArrayConfig, - chunk_grid: ChunkGrid, - selection: BasicSelection, - value: npt.ArrayLike, - prototype: BufferPrototype | None = None, -) -> None: - """ - Set values in the array using basic indexing. - - Parameters - ---------- - store_path : StorePath - The store path of the array. - metadata : ArrayMetadata - The array metadata. - codec_pipeline : CodecPipeline - The codec pipeline for encoding/decoding. - config : ArrayConfig - The array configuration. - chunk_grid : ChunkGrid - The chunk grid. - selection : BasicSelection - The selection defining the region of the array to set. - value : npt.ArrayLike - The values to be written into the selected region of the array. - prototype : BufferPrototype or None, optional - A prototype buffer that defines the structure and properties of the array chunks being modified. - If None, the default buffer prototype is used. - """ - if prototype is None: - prototype = default_buffer_prototype() - indexer = BasicIndexer( - selection, - shape=metadata.shape, - chunk_grid=chunk_grid, - ) - return await _set_selection( - store_path, - metadata, - codec_pipeline, - config, - chunk_grid, - indexer, - value, - prototype=prototype, - ) - - async def _resize( array: AsyncArray[ArrayV2Metadata] | AsyncArray[ArrayV3Metadata], new_shape: ShapeLike, @@ -5993,15 +5958,7 @@ async def _append( slice(None) if i != axis else slice(old_shape[i], new_shape[i]) for i in range(len(array.shape)) ) - await _setitem( - array.store_path, - array.metadata, - array.codec_pipeline, - array.config, - array._chunk_grid, - append_selection, - data, - ) + await array.setitem(append_selection, data) return new_shape diff --git a/src/zarr/core/common.py b/src/zarr/core/common.py index feb8eb1cac..3967c8ca34 100644 --- a/src/zarr/core/common.py +++ b/src/zarr/core/common.py @@ -276,8 +276,13 @@ def _default_zarr_format() -> ZarrFormat: return cast("ZarrFormat", int(zarr_config.get("default_zarr_format", 3))) +def _subject(name: str, axis: int | None) -> str: + """`name` as the subject of an error message, prefixed by the dimension `axis`.""" + return name[0].upper() + name[1:] if axis is None else f"Dimension {axis}: {name}" + + def _parse_positive_int(value: object, name: str, axis: int | None) -> int: - subject = name[0].upper() + name[1:] if axis is None else f"Dimension {axis}: {name}" + subject = _subject(name, axis) if isinstance(value, bool) or not isinstance(value, int): raise TypeError(f"{subject} must be an int, got {value!r}") if value < 1: @@ -295,10 +300,12 @@ def parse_chunk_edge(size: object, axis: int | None = None) -> int: def parse_chunk_shape(data: object) -> tuple[int, ...]: - """Check a regular chunk shape: one chunk edge length per axis (see `parse_chunk_edge`).""" - if not isinstance(data, Iterable): - raise TypeError(f"A chunk shape must be a sequence of chunk edge lengths, got {data!r}") - return tuple(parse_chunk_edge(size, axis) for axis, size in enumerate(data)) + """Check a regular chunk shape: a list or tuple of one chunk edge length per axis + (see `parse_chunk_edge`).""" + match data: + case list() | tuple(): + return tuple(parse_chunk_edge(size, axis) for axis, size in enumerate(data)) + raise TypeError(f"A chunk shape must be a list or tuple of chunk edge lengths, got {data!r}") def expand_rle(data: Sequence[object], axis: int | None = None) -> list[int]: @@ -313,7 +320,8 @@ def expand_rle(data: Sequence[object], axis: int | None = None) -> list[int]: for item in data: if isinstance(item, list): if len(item) != 2: - raise ValueError(f"RLE entries must be an integer or [size, count], got {item}") + subject = _subject("RLE entries", axis) + raise ValueError(f"{subject} must be an integer or [size, count], got {item}") size, count = item repeat = _parse_positive_int(count, "RLE repeat count", axis) result.extend([parse_chunk_edge(size, axis)] * repeat) diff --git a/src/zarr/core/metadata/io.py b/src/zarr/core/metadata/io.py index 7b63f5493b..7b5e0ced0d 100644 --- a/src/zarr/core/metadata/io.py +++ b/src/zarr/core/metadata/io.py @@ -1,19 +1,108 @@ from __future__ import annotations import asyncio -from typing import TYPE_CHECKING +import json +from enum import Enum +from itertools import zip_longest +from typing import TYPE_CHECKING, Final, NamedTuple from zarr.abc.store import set_or_delete +from zarr.core._json import buffer_to_json_object from zarr.core.buffer.core import default_buffer_prototype +from zarr.core.buffer.cpu import buffer_prototype as cpu_buffer_prototype from zarr.errors import ContainsArrayError from zarr.storage._common import StorePath, ensure_no_existing_node if TYPE_CHECKING: - from zarr.core.common import ZarrFormat + from collections.abc import Iterator, Mapping + + from zarr.core.buffer import Buffer + from zarr.core.common import JSON, ZarrFormat from zarr.core.group import GroupMetadata from zarr.core.metadata import ArrayMetadata +class _Absent(Enum): + ABSENT = "absent" + + +ABSENT: Final = _Absent.ABSENT +"""Where a `DocumentChange` has no value: the document, member or element is not there.""" + + +class DocumentChange(NamedTuple): + """A JSON value that differs between a node's stored metadata documents and the + documents its metadata would store.""" + + path: tuple[str | int, ...] + """Where the value is: the document's key (e.g. `.zarray`), then the object members + and array indices within it.""" + stored: JSON | _Absent + new: JSON | _Absent + + +def diff_documents( + stored: Mapping[str, JSON], new: Mapping[str, JSON] +) -> tuple[DocumentChange, ...]: + """What differs between a node's stored metadata documents and the documents its + metadata would store, each keyed by its store key (a document missing from `stored` + is not stored). Empty if they are identical. + + Objects and arrays are compared member by member, so a change is the smallest value + that differs; any other two values are identical only if their JSON encodings are, so + `true` differs from `1` and `1.0` from `1`. + """ + return tuple(_diff((), stored, new)) + + +def _diff( + path: tuple[str | int, ...], stored: JSON | _Absent, new: JSON | _Absent +) -> Iterator[DocumentChange]: + match stored, new: + case dict(), dict(): + for key in dict.fromkeys([*stored, *new]): + yield from _diff((*path, key), stored.get(key, ABSENT), new.get(key, ABSENT)) + case list(), list(): + for index, pair in enumerate(zip_longest(stored, new, fillvalue=ABSENT)): + yield from _diff((*path, index), *pair) + case _ if ABSENT in (stored, new) or json.dumps(stored) != json.dumps(new): + yield DocumentChange(path, stored, new) + + +async def store_documents(store_path: StorePath, documents: Mapping[str, Buffer]) -> None: + """Store metadata documents encoded by `to_buffer_dict` under `store_path`.""" + await asyncio.gather( + *(set_or_delete(store_path / key, value) for key, value in documents.items()) + ) + + +async def upsert_metadata( + store_path: StorePath, metadata: ArrayMetadata | GroupMetadata +) -> tuple[DocumentChange, ...]: + """Store the documents of `metadata` under `store_path` that differ from the stored + ones, and return how they differed (see `diff_documents`): empty if nothing was + stored. + + The documents are encoded before the store is read, so metadata that cannot be + stored fails with the store untouched. + """ + documents = metadata.to_buffer_dict(default_buffer_prototype()) + stored = await asyncio.gather( + *((store_path / key).get(prototype=cpu_buffer_prototype) for key in documents) + ) + changes = diff_documents( + { + key: buffer_to_json_object(buf) + for key, buf in zip(documents, stored, strict=True) + if buf is not None + }, + {key: buffer_to_json_object(buf) for key, buf in documents.items()}, + ) + changed = {change.path[0] for change in changes} + await store_documents(store_path, {k: v for k, v in documents.items() if k in changed}) + return changes + + def _build_parents(store_path: StorePath, zarr_format: ZarrFormat) -> dict[str, GroupMetadata]: from zarr.core.group import GroupMetadata @@ -52,7 +141,7 @@ async def save_metadata( ValueError """ to_save = metadata.to_buffer_dict(default_buffer_prototype()) - set_awaitables = [set_or_delete(store_path / key, value) for key, value in to_save.items()] + set_awaitables = [store_documents(store_path, to_save)] if ensure_parents: # To enable zarr.create(store, path="a/b/c"), we need to create all the intermediate groups. diff --git a/src/zarr/core/metadata/upgrades.py b/src/zarr/core/metadata/upgrades.py index 1cc6916c47..3d7d554f12 100644 --- a/src/zarr/core/metadata/upgrades.py +++ b/src/zarr/core/metadata/upgrades.py @@ -9,8 +9,7 @@ constructor. An invalid document therefore raises its own error, not a warning about how it was read. -To read another kind of invalid document, add an upgrade to `V2_ARRAY_UPGRADES` or -`V3_ARRAY_UPGRADES`. +To read another kind of invalid document, add an upgrade to `ARRAY_UPGRADES`. """ from __future__ import annotations @@ -25,7 +24,7 @@ from zarr.errors import ZarrUserWarning if TYPE_CHECKING: - from zarr.core.common import JSON + from zarr.core.common import JSON, ZarrFormat type ArrayDocument = Mapping[str, JSON] type Upgrade = Callable[[ArrayDocument], tuple[ArrayDocument, str] | None] @@ -39,14 +38,18 @@ ) -def warn_readings(readings: Sequence[str], path: str | None) -> None: - """Warn once with the readings returned by `upgrade_array_document`, naming the - array at `path` when the caller knows it.""" +def mark_upgraded[M](metadata: M, readings: Sequence[str], path: str | None) -> M: + """Record that `metadata` was read from a stored document that needed the upgrades + whose `readings` `upgrade_array_document` returned, if any: warn once, naming the + array at `path` when the caller knows it, and set `_stored_document_upgraded` on + `metadata`, so the array stores the upgrade before it writes chunks under it.""" if readings: subject = "" if path is None else f"Array {path!r}: " # The synchronous API parses metadata on zarr's IO thread, whose stack holds no # user code, so the warning points at the `from_dict` that read the document. warnings.warn(f"{subject}{' '.join(readings)} {RESAVE_HINT}", ZarrUserWarning, stacklevel=2) + object.__setattr__(metadata, "_stored_document_upgraded", True) + return metadata def _is_int_list(value: object) -> TypeGuard[list[int]]: @@ -177,23 +180,23 @@ def _invalid_chunk_sizes_v3(doc: ArrayDocument) -> tuple[ArrayDocument, str] | N return None -V2_ARRAY_UPGRADES: Final[tuple[Upgrade, ...]] = (_invalid_chunk_sizes_v2,) -V3_ARRAY_UPGRADES: Final[tuple[Upgrade, ...]] = ( - _invalid_inner_chunk_sizes_v3, - _invalid_chunk_sizes_v3, -) +ARRAY_UPGRADES: Final[Mapping[ZarrFormat, tuple[Upgrade, ...]]] = { + 2: (_invalid_chunk_sizes_v2,), + 3: (_invalid_inner_chunk_sizes_v3, _invalid_chunk_sizes_v3), +} +"""The upgrades of an array document of each Zarr format, applied in order.""" def upgrade_array_document( - doc: ArrayDocument, upgrades: Sequence[Upgrade] + doc: ArrayDocument, zarr_format: ZarrFormat ) -> tuple[ArrayDocument, list[str]]: - """Apply `upgrades` to a stored array metadata document, in order. + """Apply the upgrades for `zarr_format` to a stored array metadata document. Returns the upgraded document and the readings of the upgrades that changed it, - for `warn_readings` once the document has been validated. + for `mark_upgraded` once the document has been validated. """ readings: list[str] = [] - for upgrade in upgrades: + for upgrade in ARRAY_UPGRADES[zarr_format]: upgraded = upgrade(doc) if upgraded is not None: doc, reading = upgraded diff --git a/src/zarr/core/metadata/v2.py b/src/zarr/core/metadata/v2.py index 2a1bbeee73..58b6837b98 100644 --- a/src/zarr/core/metadata/v2.py +++ b/src/zarr/core/metadata/v2.py @@ -4,7 +4,7 @@ import warnings from collections.abc import Iterable, Sequence from functools import cached_property -from typing import TYPE_CHECKING, Any, Literal, TypedDict, cast +from typing import TYPE_CHECKING, Any, ClassVar, Literal, TypedDict, cast from zarr.abc.metadata import Metadata from zarr.abc.numcodec import Numcodec, _is_numcodec @@ -44,7 +44,7 @@ from zarr.core.config import config, parse_indexing_order from zarr.core.json_parse import parse_field from zarr.core.metadata.common import parse_attributes -from zarr.core.metadata.upgrades import V2_ARRAY_UPGRADES, upgrade_array_document, warn_readings +from zarr.core.metadata.upgrades import mark_upgraded, upgrade_array_document class ArrayV2MetadataDict(TypedDict): @@ -72,9 +72,10 @@ class ArrayV2Metadata(Metadata): compressor: Numcodec | None attributes: dict[str, JSON] = field(default_factory=dict) zarr_format: Literal[2] = field(init=False, default=2) - _stored_document_upgraded: bool = field(default=False, init=False, compare=False, repr=False) - """Whether `from_dict` read this metadata from a stored document it had to upgrade, - so the store holds an invalid document until this metadata is stored.""" + _stored_document_upgraded: ClassVar[bool] = False + """Whether `from_dict` read this metadata from a stored document it had to upgrade + (set on the instance by `mark_upgraded`), so the store may still hold that invalid + document.""" def __init__( self, @@ -154,7 +155,7 @@ def to_buffer_dict(self, prototype: BufferPrototype) -> dict[str, Buffer]: def from_dict(cls, data: dict[str, Any], *, path: str | None = None) -> ArrayV2Metadata: """Read a stored `.zarray` document (with its attributes). An invalid document that `zarr.core.metadata.upgrades` can read warns, naming the array at `path`.""" - upgraded, readings = upgrade_array_document(data, V2_ARRAY_UPGRADES) + upgraded, readings = upgrade_array_document(data, 2) # a new dict, because we are modifying it _data: dict[str, Any] = dict(upgraded) # Check that the zarr_format attribute is correct. @@ -187,7 +188,7 @@ def from_dict(cls, data: dict[str, Any], *, path: str | None = None) -> ArrayV2M # zarr v2 allowed arbitrary keys here. # We don't want the ArrayV2Metadata constructor to fail just because someone put an # extra key in the metadata. - expected = {x.name for x in fields(cls) if x.init} + expected = {x.name for x in fields(cls)} expected |= {"dtype", "chunks"} # check if `filters` is an empty sequence; if so use None instead and raise a warning @@ -207,10 +208,7 @@ def from_dict(cls, data: dict[str, Any], *, path: str | None = None) -> ArrayV2M _data = {k: v for k, v in _data.items() if k in expected} - metadata = cls(**_data) - warn_readings(readings, path) - object.__setattr__(metadata, "_stored_document_upgraded", bool(readings)) - return metadata + return mark_upgraded(cls(**_data), readings, path) def to_dict(self) -> dict[str, JSON]: zarray_dict = super().to_dict() @@ -331,7 +329,7 @@ def parse_compressor(data: object) -> Numcodec | None: raise ValueError(msg) -def parse_chunks(chunks: Iterable[int], shape: tuple[int, ...]) -> tuple[int, ...]: +def parse_chunks(chunks: object, shape: tuple[int, ...]) -> tuple[int, ...]: """Check a chunk shape: one chunk edge length (an `int` >= 1) per array axis.""" chunks_parsed = parse_chunk_shape(chunks) if len(chunks_parsed) != len(shape): diff --git a/src/zarr/core/metadata/v3.py b/src/zarr/core/metadata/v3.py index 4eb8dd352c..7231d9c51c 100644 --- a/src/zarr/core/metadata/v3.py +++ b/src/zarr/core/metadata/v3.py @@ -3,7 +3,7 @@ import json from collections.abc import Iterable, Mapping, Sequence from dataclasses import dataclass, field, replace -from typing import TYPE_CHECKING, Any, Final, Literal, NotRequired, TypeGuard, cast +from typing import TYPE_CHECKING, Any, ClassVar, Final, Literal, NotRequired, TypeGuard, cast from typing_extensions import TypedDict @@ -27,6 +27,7 @@ compress_rle, expand_rle, parse_chunk_edge, + parse_chunk_shape, parse_named_configuration, parse_shapelike, validate_rectilinear_edges, @@ -38,9 +39,8 @@ from zarr.core.json_parse import parse_field from zarr.core.metadata.common import parse_attributes from zarr.core.metadata.upgrades import ( - V3_ARRAY_UPGRADES, + mark_upgraded, upgrade_array_document, - warn_readings, ) from zarr.errors import MetadataValidationError, NodeTypeValidationError from zarr.registry import get_codec_class @@ -218,17 +218,6 @@ class RectilinearChunkGridMetadataConfig(TypedDict): ] -def _parse_chunk_shape(chunk_shape: Iterable[int]) -> tuple[int, ...]: - """Validate and normalize a regular chunk shape. - - Delegates to ``_validate_chunk_shapes`` — a regular chunk shape is just - a sequence of bare ints (one per dimension), each of which must be >= 1. - """ - result = _validate_chunk_shapes(tuple(chunk_shape)) - # Regular grids only have bare ints — cast is safe after validation - return cast(tuple[int, ...], result) - - def _validate_chunk_shapes( chunk_shapes: Sequence[int | Sequence[int]], ) -> tuple[int | tuple[int, ...], ...]: @@ -261,7 +250,7 @@ class RegularChunkGridMetadata(Metadata): chunk_shape: tuple[int, ...] def __post_init__(self) -> None: - chunk_shape_parsed = _parse_chunk_shape(self.chunk_shape) + chunk_shape_parsed = parse_chunk_shape(self.chunk_shape) object.__setattr__(self, "chunk_shape", chunk_shape_parsed) @property @@ -278,7 +267,7 @@ def to_dict(self) -> RegularChunkGridMetadataJSON: # type: ignore[override] def from_dict(cls, data: RegularChunkGridMetadataJSON) -> Self: # type: ignore[override] parse_named_configuration(data, "regular") # validate name configuration = data["configuration"] - return cls(chunk_shape=_parse_chunk_shape(configuration["chunk_shape"])) + return cls(chunk_shape=parse_chunk_shape(configuration["chunk_shape"])) @dataclass(frozen=True, kw_only=True) @@ -479,9 +468,10 @@ class ArrayV3Metadata(Metadata): node_type: Literal["array"] = field(default="array", init=False) storage_transformers: tuple[dict[str, JSON], ...] extra_fields: dict[str, AllowedExtraField] - _stored_document_upgraded: bool = field(default=False, init=False, compare=False, repr=False) - """Whether `from_dict` read this metadata from a stored document it had to upgrade, - so the store holds an invalid document until this metadata is stored.""" + _stored_document_upgraded: ClassVar[bool] = False + """Whether `from_dict` read this metadata from a stored document it had to upgrade + (set on the instance by `mark_upgraded`), so the store may still hold that invalid + document.""" def __init__( self, @@ -625,7 +615,7 @@ def to_buffer_dict(self, prototype: BufferPrototype) -> dict[str, Buffer]: def from_dict(cls, data: dict[str, JSON], *, path: str | None = None) -> Self: """Read a stored `zarr.json` array document. An invalid document that `zarr.core.metadata.upgrades` can read warns, naming the array at `path`.""" - upgraded, readings = upgrade_array_document(data, V3_ARRAY_UPGRADES) + upgraded, readings = upgrade_array_document(data, 3) # a new dict, because we are modifying it _data = dict(upgraded) @@ -681,9 +671,7 @@ def from_dict(cls, data: dict[str, JSON], *, path: str | None = None) -> Self: extra_fields=allowed_extra_fields, storage_transformers=_data_typed.get("storage_transformers", ()), # type: ignore[arg-type] ) - warn_readings(readings, path) - object.__setattr__(metadata, "_stored_document_upgraded", bool(readings)) - return metadata + return mark_upgraded(metadata, readings, path) def to_dict(self) -> dict[str, JSON]: out_dict = super().to_dict() diff --git a/tests/test_metadata/test_io.py b/tests/test_metadata/test_io.py new file mode 100644 index 0000000000..9accc0f79f --- /dev/null +++ b/tests/test_metadata/test_io.py @@ -0,0 +1,199 @@ +"""Tests for comparing stored metadata documents with the documents metadata would store, +and for storing only the documents that differ.""" + +from __future__ import annotations + +import json +from typing import TYPE_CHECKING, Any, Literal + +import pytest + +import zarr +from zarr.core.buffer import cpu +from zarr.core.metadata import ArrayV2Metadata, ArrayV3Metadata +from zarr.core.metadata.io import ABSENT, DocumentChange, diff_documents, upsert_metadata +from zarr.core.sync import sync +from zarr.storage import MemoryStore, StorePath + +if TYPE_CHECKING: + from zarr.abc.store import Store + from zarr.core.buffer import Buffer + from zarr.core.common import JSON + +V3_DOC: dict[str, JSON] = { + "zarr_format": 3, + "node_type": "array", + "shape": [3], + "chunk_grid": {"name": "regular", "configuration": {"chunk_shape": [3]}}, + "fill_value": 0, +} + + +def _with(doc: dict[str, JSON], **changes: JSON) -> dict[str, JSON]: + return {**doc, **changes} + + +@pytest.mark.parametrize( + ("stored", "new", "expected"), + [ + ({"zarr.json": V3_DOC}, {"zarr.json": V3_DOC}, ()), + ( + {"zarr.json": V3_DOC}, + {"zarr.json": _with(V3_DOC, fill_value=1)}, + ((("zarr.json", "fill_value"), 0, 1),), + ), + ( + { + "zarr.json": _with( + V3_DOC, chunk_grid={"name": "regular", "configuration": {"chunk_shape": [0]}} + ) + }, + {"zarr.json": V3_DOC}, + ((("zarr.json", "chunk_grid", "configuration", "chunk_shape", 0), 0, 3),), + ), + ( + {"zarr.json": V3_DOC}, + {"zarr.json": _with(V3_DOC, shape=[3, 4])}, + ((("zarr.json", "shape", 1), ABSENT, 4),), + ), + ( + {"zarr.json": _with(V3_DOC, attributes={"a": 1})}, + {"zarr.json": _with(V3_DOC, dimension_names=["x"])}, + ( + (("zarr.json", "attributes"), {"a": 1}, ABSENT), + (("zarr.json", "dimension_names"), ABSENT, ["x"]), + ), + ), + ( + {"zarr.json": _with(V3_DOC, fill_value=True)}, + {"zarr.json": _with(V3_DOC, fill_value=1)}, + ((("zarr.json", "fill_value"), True, 1),), + ), + ( + {"zarr.json": _with(V3_DOC, fill_value=1.0)}, + {"zarr.json": _with(V3_DOC, fill_value=1)}, + ((("zarr.json", "fill_value"), 1.0, 1),), + ), + ( + {"zarr.json": _with(V3_DOC, fill_value=float("nan"))}, + {"zarr.json": _with(V3_DOC, fill_value=float("nan"))}, + (), + ), + ( + {"zarr.json": _with(V3_DOC, shape={"0": 3})}, + {"zarr.json": V3_DOC}, + ((("zarr.json", "shape"), {"0": 3}, [3]),), + ), + ( + {".zarray": {"shape": [3], "chunks": [0]}}, + {".zarray": {"shape": [3], "chunks": [3]}, ".zattrs": {}}, + (((".zarray", "chunks", 0), 0, 3), ((".zattrs",), ABSENT, {})), + ), + ], + ids=[ + "identical", + "changed-scalar", + "changed-nested-list", + "longer-list", + "removed-and-added-key", + "bool-is-not-int", + "float-is-not-int", + "nan-is-nan", + "object-is-not-array", + "v2-two-documents", + ], +) +def test_diff_documents( + stored: dict[str, JSON], new: dict[str, JSON], expected: tuple[Any, ...] +) -> None: + """Documents are compared value by value, each change named by its JSON path under + the document's key; values other than objects and arrays are identical only if + their JSON encodings are. `diff_documents` is total over JSON: it raises no error.""" + assert diff_documents(stored, new) == tuple(DocumentChange(*change) for change in expected) + + +class _CountingStore(MemoryStore): + """A memory store that counts the values set in it.""" + + sets = 0 + + async def set(self, key: str, value: Buffer, byte_range: tuple[int, int] | None = None) -> None: + self.sets += 1 + await super().set(key, value, byte_range) + + +def _documents(store: Store) -> dict[str, Any]: + assert isinstance(store, MemoryStore) + return {key: json.loads(value.to_bytes()) for key, value in store._store_dict.items()} + + +def _legacy(zarr_format: Literal[2, 3]) -> tuple[StorePath, ArrayV2Metadata | ArrayV3Metadata]: + """An array stored with chunk shape `[0]`, and the metadata its upgrade reads.""" + store = _CountingStore() + array = zarr.create_array( + store, shape=(3,), chunks=(3,), dtype="int16", zarr_format=zarr_format + ) + key = ".zarray" if zarr_format == 2 else "zarr.json" + doc = _documents(store)[key] + if zarr_format == 2: + doc["chunks"] = [0] + else: + doc["chunk_grid"]["configuration"]["chunk_shape"] = [0] + sync(store.set(key, cpu.Buffer.from_bytes(json.dumps(doc).encode()))) + return StorePath(store), array.metadata + + +@pytest.mark.parametrize("zarr_format", [2, 3]) +def test_upsert_metadata_stores_documents_that_differ(zarr_format: Literal[2, 3]) -> None: + """The documents that differ from the stored ones are stored, and the changes are + returned.""" + store_path, metadata = _legacy(zarr_format) + assert isinstance(store_path.store, _CountingStore) + store_path.store.sets = 0 + key = ".zarray" if zarr_format == 2 else "zarr.json" + chunk_path = ( + ("chunks", 0) if zarr_format == 2 else ("chunk_grid", "configuration", "chunk_shape", 0) + ) + + changes = sync(upsert_metadata(store_path, metadata)) + + assert changes == (DocumentChange((key, *chunk_path), 0, 3),) + assert store_path.store.sets == 1 + stored = _documents(store_path.store) + assert stored[key] == json.loads(metadata.to_buffer_dict(cpu.buffer_prototype)[key].to_bytes()) + + +@pytest.mark.parametrize("zarr_format", [2, 3]) +def test_upsert_metadata_identical_stores_nothing(zarr_format: Literal[2, 3]) -> None: + """Metadata identical to what is stored stores nothing.""" + store = _CountingStore() + array = zarr.create_array( + store, shape=(3,), chunks=(3,), dtype="int16", zarr_format=zarr_format + ) + store.sets = 0 + + assert sync(upsert_metadata(StorePath(store), array.metadata)) == () + assert store.sets == 0 + + +def test_upsert_metadata_unstorable_leaves_store_untouched(monkeypatch: pytest.MonkeyPatch) -> None: + """Metadata that cannot be encoded fails before the store is read or written.""" + store_path, metadata = _legacy(3) + before = _documents(store_path.store) + + def refuse(*args: object) -> None: + raise ValueError("cannot be stored") + + monkeypatch.setattr(ArrayV3Metadata, "to_buffer_dict", refuse) + with pytest.raises(ValueError, match="cannot be stored"): + sync(upsert_metadata(store_path, metadata)) + assert _documents(store_path.store) == before + + +def test_upsert_metadata_stored_document_not_an_object() -> None: + """A stored document that is not a JSON object is not overwritten.""" + store_path, metadata = _legacy(3) + sync(store_path.store.set("zarr.json", cpu.Buffer.from_bytes(b"[]"))) + with pytest.raises(TypeError, match="Expected a JSON object, got list"): + sync(upsert_metadata(store_path, metadata)) + assert _documents(store_path.store) == {"zarr.json": []} diff --git a/tests/test_metadata/test_upgrades.py b/tests/test_metadata/test_upgrades.py index b976660eb2..6e1126fadc 100644 --- a/tests/test_metadata/test_upgrades.py +++ b/tests/test_metadata/test_upgrades.py @@ -6,30 +6,31 @@ import json import re import warnings -from typing import TYPE_CHECKING, Any, Literal +from typing import TYPE_CHECKING, Any, Literal, cast import numpy as np import pytest import zarr from zarr.codecs import ShardingCodec +from zarr.core.array import AsyncArray from zarr.core.metadata import ArrayV2Metadata, ArrayV3Metadata from zarr.core.metadata.upgrades import ( RESAVE_HINT, - V2_ARRAY_UPGRADES, - V3_ARRAY_UPGRADES, upgrade_array_document, ) from zarr.core.metadata.v3 import RectilinearChunkGridMetadata, RegularChunkGridMetadata from zarr.core.sync import sync from zarr.dtype import Int16 from zarr.errors import ZarrUserWarning +from zarr.storage._common import make_store_path if TYPE_CHECKING: from collections.abc import Callable from pathlib import Path - from zarr.core.common import JSON + from zarr.core.common import JSON, ZarrFormat + from zarr.types import AnyArray def _v2_doc(shape: list[int], chunks: list[Any]) -> dict[str, JSON]: @@ -168,8 +169,7 @@ def test_upgrade_array_document( `from_dict` warns once for the document, naming the array, saying how each part was read (and that the array holds only its fill value where a chunk size of 0 was stored for a non-empty axis) and how to re-save.""" - upgrades = V2_ARRAY_UPGRADES if doc["zarr_format"] == 2 else V3_ARRAY_UPGRADES - upgraded, readings = upgrade_array_document(doc, upgrades) + upgraded, readings = upgrade_array_document(doc, cast("ZarrFormat", doc["zarr_format"])) assert {k: v for k, v in upgraded.items() if k not in ("chunks", "chunk_grid", "codecs")} == { k: v for k, v in doc.items() if k not in ("chunks", "chunk_grid", "codecs") } @@ -204,7 +204,7 @@ def test_invalid_upgraded_document_raises_without_warning(doc: dict[str, JSON], """A document the upgrades read that the metadata constructor then rejects raises that error, without first warning how it was read.""" metadata_cls = ArrayV2Metadata if doc["zarr_format"] == 2 else ArrayV3Metadata - assert upgrade_array_document(doc, V2_ARRAY_UPGRADES + V3_ARRAY_UPGRADES)[1] + assert upgrade_array_document(doc, cast("ZarrFormat", doc["zarr_format"]))[1] with warnings.catch_warnings(): warnings.simplefilter("error", ZarrUserWarning) with pytest.raises(ValueError, match=error): @@ -306,10 +306,25 @@ def test_metadata_rejects_chunk_edge_below_one(site: str, size: int) -> None: CHUNK_EDGE_SITES[site](size) -@pytest.mark.parametrize("chunks", [4, np.int64(4)]) -def test_v2_constructor_rejects_scalar_chunks(chunks: object) -> None: - with pytest.raises(TypeError, match="A chunk shape must be a sequence of chunk edge lengths"): - _v2_metadata(chunks) +CHUNK_SHAPE_SITES: dict[str, Callable[[Any], object]] = { + "regular": lambda chunk_shape: RegularChunkGridMetadata(chunk_shape=chunk_shape), + "v2": _v2_metadata, + "sharding-inner": lambda chunk_shape: ShardingCodec(chunk_shape=chunk_shape), +} + + +@pytest.mark.parametrize("site", CHUNK_SHAPE_SITES) +@pytest.mark.parametrize("chunk_shape", [4, np.int64(4), "10", {"a": 1}, range(1, 2)]) +def test_metadata_rejects_chunk_shape_not_list_or_tuple(site: str, chunk_shape: object) -> None: + """A regular chunk shape is a list or tuple; anything else is rejected as a whole, + not iterated as if its elements were chunk edge lengths.""" + with pytest.raises( + TypeError, + match=re.escape( + f"A chunk shape must be a list or tuple of chunk edge lengths, got {chunk_shape!r}" + ), + ): + CHUNK_SHAPE_SITES[site](chunk_shape) def _rewrite_doc(path: Path, zarr_format: Literal[2, 3], edit: Any) -> None: @@ -424,6 +439,7 @@ async def write_twice() -> None: sync(write_twice()) + assert not arr.metadata._stored_document_upgraded with warnings.catch_warnings(): warnings.simplefilter("error", ZarrUserWarning) reopened = zarr.open_array(store=path, mode="r") @@ -431,6 +447,99 @@ async def write_twice() -> None: np.testing.assert_array_equal(reopened[...], data) +def _legacy_array(path: Path, zarr_format: Literal[2, 3]) -> None: + """Store an array of shape (3,) whose stored chunk shape is `[0]`.""" + zarr.create_array(store=path, shape=(3,), chunks=(3,), dtype="int16", zarr_format=zarr_format) + _rewrite_doc( + path, + zarr_format, + lambda doc: ( + doc.update(chunks=[0]) + if zarr_format == 2 + else doc["chunk_grid"]["configuration"].update(chunk_shape=[0]) + ), + ) + + +def _open_strictly(path: Path) -> AnyArray: + """Open the array at `path`, failing on any warning that its document was upgraded.""" + with warnings.catch_warnings(): + warnings.simplefilter("error", ZarrUserWarning) + array = zarr.open_array(store=path, mode="r") + assert isinstance(array, zarr.Array) + return array + + +@pytest.mark.parametrize("zarr_format", [2, 3]) +def test_stale_handle_write_keeps_newer_metadata( + tmp_path: Path, zarr_format: Literal[2, 3] +) -> None: + """A handle read from an upgraded document stores the upgrade of what the store + holds when it first writes chunks; if another handle stored valid metadata since, + it stores no metadata and writes only its chunks.""" + path = tmp_path / "legacy.zarr" + _legacy_array(path, zarr_format) + with pytest.warns(ZarrUserWarning, match="is read as"): + stale = zarr.open_array(store=path, mode="r+") + with pytest.warns(ZarrUserWarning, match="is read as"): + other = zarr.open_array(store=path, mode="r+") + other.append(np.arange(1, 7, dtype="int16")) + other.attrs["x"] = 1 + + stale[0] = 9 + + assert not stale.metadata._stored_document_upgraded + reopened = _open_strictly(path) + assert reopened.shape == (9,) + assert reopened.attrs.asdict() == {"x": 1} + np.testing.assert_array_equal(reopened[...], [9, 0, 0, 1, 2, 3, 4, 5, 6]) + + +@pytest.mark.parametrize("zarr_format", [2, 3]) +def test_empty_write_stores_no_metadata(tmp_path: Path, zarr_format: Literal[2, 3]) -> None: + """A write of an empty selection stores no chunks, so it stores no metadata either.""" + path = tmp_path / "legacy.zarr" + _legacy_array(path, zarr_format) + documents = {p.name: p.read_bytes() for p in path.iterdir()} + with pytest.warns(ZarrUserWarning, match="is read as"): + array = zarr.open_array(store=path, mode="r+") + + array[0:0] = np.empty(0, dtype="int16") + + assert array.metadata._stored_document_upgraded + assert {p.name: p.read_bytes() for p in path.iterdir()} == documents + + +@pytest.mark.parametrize("zarr_format", [2, 3]) +def test_async_array_from_dict_names_array(tmp_path: Path, zarr_format: Literal[2, 3]) -> None: + """`AsyncArray.from_dict` names the array at its store path in the upgrade warning.""" + store_path = sync(make_store_path(tmp_path / "legacy.zarr")) + doc = _v2_doc([4], [0]) if zarr_format == 2 else _v3_doc([4], [0]) + with pytest.warns(ZarrUserWarning, match=f"^Array {re.escape(repr(str(store_path)))}: "): + AsyncArray.from_dict(store_path, doc) + + +@pytest.mark.filterwarnings("ignore:Consolidated metadata is currently not part:UserWarning") +@pytest.mark.parametrize("zarr_format", [2, 3]) +def test_consolidate_stores_upgraded_members(tmp_path: Path, zarr_format: Literal[2, 3]) -> None: + """Consolidating a group stores the upgrade of every member document that needed + one before the consolidated document, so chunks written through the consolidated + metadata are stored under a member document that agrees with it.""" + path = tmp_path / "group.zarr" + zarr.open_group(path, mode="w", zarr_format=zarr_format) + _legacy_array(path / "a", zarr_format) + with pytest.warns(ZarrUserWarning, match="is read as"): + zarr.consolidate_metadata(path) + + member = _open_strictly(path / "a") + assert member.chunks == (3,) + group = zarr.open_group(path, mode="r+", use_consolidated=True) + array = group["a"] + assert isinstance(array, zarr.Array) + array[:] = [7, 8, 9] + np.testing.assert_array_equal(_open_strictly(path / "a")[...], [7, 8, 9]) + + @pytest.mark.filterwarnings("ignore:Consolidated metadata is currently not part:UserWarning") @pytest.mark.parametrize("zarr_format", [2, 3]) def test_legacy_chunk_size_consolidated(tmp_path: Path, zarr_format: Literal[2, 3]) -> None: diff --git a/tests/test_unified_chunk_grid.py b/tests/test_unified_chunk_grid.py index 4a8d6f791e..8f06863a05 100644 --- a/tests/test_unified_chunk_grid.py +++ b/tests/test_unified_chunk_grid.py @@ -562,6 +562,22 @@ def test_rle_expand_rejects_non_int(rle_input: list[Any], match: str) -> None: expand_rle(rle_input) +@pytest.mark.parametrize( + ("rle_input", "match"), + [ + ([0], "chunk edge length must be >= 1"), + ([10.0], "chunk edge length must be an int"), + ([[5, 0]], "RLE repeat count must be >= 1"), + ([[5, 2, 1]], r"RLE entries must be an integer or \[size, count\]"), + ], + ids=["zero-edge", "float-edge", "zero-rle-count", "rle-entry-of-three"], +) +def test_rle_expand_names_dimension(rle_input: list[Any], match: str) -> None: + """Given the dimension `axis` the edges belong to, every error of expand_rle names it.""" + with pytest.raises((TypeError, ValueError), match=f"^Dimension 2: {match}"): + expand_rle(rle_input, axis=2) + + # --------------------------------------------------------------------------- # _is_rectilinear_chunks tests # --------------------------------------------------------------------------- From 9e2933352d50c22220d9105df3d0f8e74f8d9657 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Sat, 26 Sep 2026 10:12:58 +0200 Subject: [PATCH 25/41] test: prefer zero-length axes for stored chunk size 0 in the lifecycle machine An empty write stores no metadata, so it no longer ends the legacy state. Assisted-by: ClaudeCode:claude-opus-5-5 Co-Authored-By: Claude Opus 5.5 --- tests/test_array_stateful.py | 17 ++++++++++------- 1 file changed, 10 insertions(+), 7 deletions(-) diff --git a/tests/test_array_stateful.py b/tests/test_array_stateful.py index 7ce4780e5d..6dc03768b3 100644 --- a/tests/test_array_stateful.py +++ b/tests/test_array_stateful.py @@ -134,11 +134,12 @@ def create(self, data: st.DataObject) -> None: if spelling != "rectilinear" and data.draw(st.booleans(), label="legacy zero"): # A stored chunk size of 0, as zarr-python wrote for arrays created with a # zero-length axis; older releases could then grow the axis without storing - # a chunk, so any extent is possible. Sharded arrays store it in the outer grid. - self.legacy_axes = data.draw( - st.lists(st.integers(0, len(shape) - 1), min_size=1, unique=True), - label="axes stored with chunk size 0", - ) + # a chunk, so any extent is possible, but zero-length axes come first. + # Sharded arrays store it in the outer grid. + axes = st.lists(st.integers(0, len(shape) - 1), min_size=1, unique=True) + if zero_axes := [axis for axis, extent in enumerate(shape) if extent == 0]: + axes = st.lists(st.sampled_from(zero_axes), min_size=1, unique=True) | axes + self.legacy_axes = data.draw(axes, label="axes stored with chunk size 0") stored_zero = data.draw(st.sampled_from([0, False]), label="stored zero") self._rewrite_stored_chunks(zarr_format, self.legacy_axes, stored_zero) event("legacy zero chunk size") @@ -269,8 +270,10 @@ def write(self, data: st.DataObject) -> None: note(f"write {region}") arr[region] = values self._model_write(arr, region, values) - # Writing chunks first stores the metadata they are written under. - self.legacy_axes = [] + if all(shape): + # Writing chunks first stores the metadata they are written under; a write + # of nothing stores nothing. + self.legacy_axes = [] @precondition(lambda self: self.legacy_axes) @rule() From 632ac2b539ee52bde947d7d79cdf5c790926607d Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Sat, 26 Sep 2026 11:13:45 +0200 Subject: [PATCH 26/41] fix(metadata): keep accepting the chunk sizes zarr 3.4.0 accepted in metadata constructors (patch release) Nothing in a patch release may reject a chunk size that zarr 3.4.0 accepted. The one chunk edge rule (`parse_chunk_edge`, which also reads RLE repeat counts) now reads any integral number as the `int` it equals: an `int`, a `bool`, a NumPy integer or an integral float, so rectilinear grids written with float edges (`[[4.0, 2]]`) read again and the grid constructors accept NumPy integers; a fractional float is still rejected. A regular chunk shape may again be any iterable. `ArrayV2Metadata(chunks=...)` and `ShardingCodec(chunk_shape=...)` read their chunk shape with `parse_shapelike`, as in 3.4.0: a scalar, NumPy integers, bools and 0 are accepted, and a 0 is written back as given (the kerchunk writer in VirtualiZarr relies on `chunks=(0,)` for empty inlined arrays). Stored documents with a chunk size of 0 are still read by the upgrades. The strict rule returns in the next minor release. Assisted-by: ClaudeCode:claude-opus-5-5 Co-Authored-By: Claude Opus 5.5 --- src/zarr/codecs/sharding.py | 9 +- src/zarr/core/common.py | 25 +++-- src/zarr/core/metadata/upgrades.py | 2 +- src/zarr/core/metadata/v2.py | 9 +- tests/test_metadata/test_upgrades.py | 140 +++++++++++++++++++-------- tests/test_unified_chunk_grid.py | 16 +-- 6 files changed, 136 insertions(+), 65 deletions(-) diff --git a/src/zarr/codecs/sharding.py b/src/zarr/codecs/sharding.py index 52d100bb91..7a053867d6 100644 --- a/src/zarr/codecs/sharding.py +++ b/src/zarr/codecs/sharding.py @@ -45,8 +45,9 @@ merge_and_encode_chunk, ) from zarr.core.common import ( - parse_chunk_shape, + ShapeLike, parse_named_configuration, + parse_shapelike, product, ) from zarr.core.config import config as zarr_config @@ -458,13 +459,13 @@ class ShardingCodec( def __init__( self, *, - chunk_shape: tuple[int, ...] | list[int], + chunk_shape: ShapeLike, codecs: Iterable[Codec | dict[str, JSON]] = (BytesCodec(),), index_codecs: Iterable[Codec | dict[str, JSON]] = (BytesCodec(), Crc32cCodec()), index_location: ShardingCodecIndexLocation | IndexLocation = "end", subchunk_write_order: SubchunkWriteOrder = "morton", ) -> None: - chunk_shape_parsed = parse_chunk_shape(chunk_shape) + chunk_shape_parsed = parse_shapelike(chunk_shape) codecs_parsed = parse_codecs(codecs) index_codecs_parsed = parse_codecs(index_codecs) _check_index_codecs_fixed_size(index_codecs_parsed) @@ -503,7 +504,7 @@ def __getstate__(self) -> dict[str, Any]: def __setstate__(self, state: dict[str, Any]) -> None: config = state["configuration"] - object.__setattr__(self, "chunk_shape", parse_chunk_shape(config["chunk_shape"])) + object.__setattr__(self, "chunk_shape", parse_shapelike(config["chunk_shape"])) object.__setattr__(self, "codecs", parse_codecs(config["codecs"])) object.__setattr__(self, "index_codecs", parse_codecs(config["index_codecs"])) object.__setattr__(self, "index_location", _parse_index_location(config["index_location"])) diff --git a/src/zarr/core/common.py b/src/zarr/core/common.py index 3967c8ca34..cfeb6d3371 100644 --- a/src/zarr/core/common.py +++ b/src/zarr/core/common.py @@ -2,6 +2,7 @@ import asyncio import math +import numbers import warnings from collections.abc import Iterable, Mapping, Sequence from enum import Enum @@ -282,16 +283,24 @@ def _subject(name: str, axis: int | None) -> str: def _parse_positive_int(value: object, name: str, axis: int | None) -> int: + """`value` as an `int` of at least 1. Any integral number is read as the `int` it + equals: an `int`, a `bool`, a NumPy integer or an integral float.""" subject = _subject(name, axis) - if isinstance(value, bool) or not isinstance(value, int): - raise TypeError(f"{subject} must be an int, got {value!r}") - if value < 1: + match value: + case numbers.Integral(): + parsed = int(value) + case float() if value.is_integer(): + parsed = int(value) + case _: + raise TypeError(f"{subject} must be an integer, got {value!r}") + if parsed < 1: raise ValueError(f"{subject} must be >= 1, got {value!r}") - return value + return parsed def parse_chunk_edge(size: object, axis: int | None = None) -> int: - """Check that `size` is a chunk edge length: an `int` (not a `bool`) of at least 1. + """Check that `size` is a chunk edge length: an integral number of at least 1, read + as an `int`. This is the one rule for chunk edge lengths in metadata: bare chunk sizes, explicit edges and run-length encoded sizes. `axis`, when given, is named in the error. @@ -300,12 +309,12 @@ def parse_chunk_edge(size: object, axis: int | None = None) -> int: def parse_chunk_shape(data: object) -> tuple[int, ...]: - """Check a regular chunk shape: a list or tuple of one chunk edge length per axis + """Check a regular chunk shape: an iterable of one chunk edge length per axis (see `parse_chunk_edge`).""" match data: - case list() | tuple(): + case Iterable(): return tuple(parse_chunk_edge(size, axis) for axis, size in enumerate(data)) - raise TypeError(f"A chunk shape must be a list or tuple of chunk edge lengths, got {data!r}") + raise TypeError(f"A chunk shape must be an iterable of chunk edge lengths, got {data!r}") def expand_rle(data: Sequence[object], axis: int | None = None) -> list[int]: diff --git a/src/zarr/core/metadata/upgrades.py b/src/zarr/core/metadata/upgrades.py index 3d7d554f12..8589a0f0c4 100644 --- a/src/zarr/core/metadata/upgrades.py +++ b/src/zarr/core/metadata/upgrades.py @@ -1,7 +1,7 @@ """Upgrades that read invalid stored array metadata documents written by older software. This is the only place invalid metadata is read leniently; the metadata constructors -are strict. An upgrade maps a stored array metadata document (parsed JSON) to a valid +never reinterpret a chunk size. An upgrade maps a stored array metadata document (parsed JSON) to a valid one and says how it read the document. `ArrayV2Metadata.from_dict` and `ArrayV3Metadata.from_dict` apply the upgrades for their Zarr format, so every path that parses a stored document, including consolidated metadata, goes through them, and diff --git a/src/zarr/core/metadata/v2.py b/src/zarr/core/metadata/v2.py index 58b6837b98..4adae2834e 100644 --- a/src/zarr/core/metadata/v2.py +++ b/src/zarr/core/metadata/v2.py @@ -38,7 +38,7 @@ ZARRAY_JSON, ZATTRS_JSON, MemoryOrder, - parse_chunk_shape, + ShapeLike, parse_shapelike, ) from zarr.core.config import config, parse_indexing_order @@ -329,9 +329,10 @@ def parse_compressor(data: object) -> Numcodec | None: raise ValueError(msg) -def parse_chunks(chunks: object, shape: tuple[int, ...]) -> tuple[int, ...]: - """Check a chunk shape: one chunk edge length (an `int` >= 1) per array axis.""" - chunks_parsed = parse_chunk_shape(chunks) +def parse_chunks(chunks: ShapeLike, shape: tuple[int, ...]) -> tuple[int, ...]: + """Check a chunk shape: one non-negative integer per array axis (see + `parse_shapelike`). Stored chunk sizes of 0 are read by `zarr.core.metadata.upgrades`.""" + chunks_parsed = parse_shapelike(chunks) if len(chunks_parsed) != len(shape): raise ValueError( f"The `shape` and `chunks` attributes must have the same length. " diff --git a/tests/test_metadata/test_upgrades.py b/tests/test_metadata/test_upgrades.py index 6e1126fadc..62ae417584 100644 --- a/tests/test_metadata/test_upgrades.py +++ b/tests/test_metadata/test_upgrades.py @@ -221,7 +221,7 @@ def _read_strictly(doc: dict[str, JSON]) -> ArrayV2Metadata | ArrayV3Metadata: def test_stored_negative_chunk_size_rejected() -> None: """No known writer stored a negative chunk size: it is rejected, not upgraded.""" - with pytest.raises(ValueError, match="^Dimension 0: chunk edge length must be >= 1, got -1$"): + with pytest.raises(ValueError, match="^Expected all values to be non-negative"): _read_strictly(_v2_doc([4], [-1])) @@ -232,19 +232,12 @@ def test_stored_chunk_shape_ndim_mismatch_rejected() -> None: _read_strictly(_v3_doc([4, 4], [0])) -def test_stored_float_chunk_size_rejected() -> None: - """No known writer stored a float chunk size: it is rejected, not upgraded.""" +def test_stored_fractional_chunk_size_rejected() -> None: + """A stored chunk size that is not an integral number is rejected, not upgraded.""" with pytest.raises( - TypeError, match=r"^Dimension 0: chunk edge length must be an int, got 4\.0$" + TypeError, match=r"^Dimension 0: chunk edge length must be an integer, got 4\.5$" ): - _read_strictly(_v3_doc([4], [4.0])) - - -def test_stored_zero_inner_chunk_size_rejected() -> None: - """No known writer stored an inner chunk size of 0, and no span defines one: it is - rejected, not upgraded.""" - with pytest.raises(ValueError, match="^Dimension 0: chunk edge length must be >= 1, got 0$"): - _read_strictly(_v3_doc([4], [0], inner=[0])) + _read_strictly(_v3_doc([4], [4.5])) def test_v2_constructor_rejects_chunks_of_wrong_length() -> None: @@ -271,33 +264,52 @@ def _rectilinear_from_dict(chunk_shapes: list[Any]) -> RectilinearChunkGridMetad ) +def _second_edge(grid: RectilinearChunkGridMetadata) -> int: + edges = grid.chunk_shapes[0] + assert isinstance(edges, tuple) + return edges[1] + + CHUNK_EDGE_SITES: dict[str, Callable[[Any], object]] = { - "regular": lambda size: RegularChunkGridMetadata(chunk_shape=(size,)), - "v2": lambda size: _v2_metadata((size,)), - "rectilinear-bare": lambda size: _rectilinear((size,)), - "rectilinear-edge": lambda size: _rectilinear(((4, size),)), - "rectilinear-bare-json": lambda size: _rectilinear_from_dict([size]), - "rectilinear-edge-json": lambda size: _rectilinear_from_dict([[4, size]]), - "rectilinear-rle-json": lambda size: _rectilinear_from_dict([[[size, 2]]]), - "sharding-inner": lambda size: ShardingCodec(chunk_shape=(size,)), + "regular": lambda size: RegularChunkGridMetadata(chunk_shape=(size,)).chunk_shape[0], + "rectilinear-bare": lambda size: _rectilinear((size,)).chunk_shapes[0], + "rectilinear-edge": lambda size: _second_edge(_rectilinear(((4, size),))), + "rectilinear-bare-json": lambda size: _rectilinear_from_dict([size]).chunk_shapes[0], + "rectilinear-edge-json": lambda size: _second_edge(_rectilinear_from_dict([[4, size]])), + "rectilinear-rle-json": lambda size: _second_edge(_rectilinear_from_dict([[[size, 2]]])), } +"""Each place chunk grid metadata reads a chunk edge length, returning the edge it read.""" @pytest.mark.parametrize("site", CHUNK_EDGE_SITES) -@pytest.mark.parametrize("size", [True, False, 4.0, np.int64(4), "4"]) -def test_metadata_rejects_non_int_chunk_edge(site: str, size: object) -> None: - """Metadata built in code takes chunk edge lengths as `int`s only, everywhere.""" +@pytest.mark.parametrize( + ("size", "expected"), + [(4, 4), (True, 1), (np.int64(4), 4), (4.0, 4), (np.float64(4.0), 4)], + ids=["int", "bool", "numpy-int", "float", "numpy-float"], +) +def test_metadata_reads_integral_chunk_edge(site: str, size: object, expected: int) -> None: + """Chunk grid metadata reads any integral number as the `int` chunk edge length it + equals.""" + edge = CHUNK_EDGE_SITES[site](size) + assert type(edge) is int + assert edge == expected + + +@pytest.mark.parametrize("site", CHUNK_EDGE_SITES) +@pytest.mark.parametrize("size", [4.5, float("inf"), "4", None]) +def test_metadata_rejects_non_integral_chunk_edge(site: str, size: object) -> None: with pytest.raises( - TypeError, match=re.escape(f"Dimension 0: chunk edge length must be an int, got {size!r}") + TypeError, + match=re.escape(f"Dimension 0: chunk edge length must be an integer, got {size!r}"), ): CHUNK_EDGE_SITES[site](size) @pytest.mark.parametrize("site", CHUNK_EDGE_SITES) -@pytest.mark.parametrize("size", [0, -1]) +@pytest.mark.parametrize("size", [0, False, -1]) def test_metadata_rejects_chunk_edge_below_one(site: str, size: int) -> None: - """Metadata built in code is strict: a chunk edge length below 1 is rejected, - without a warning.""" + """Chunk grid metadata built in code is strict: a chunk edge length below 1 is + rejected, without a warning.""" with warnings.catch_warnings(): warnings.simplefilter("error", ZarrUserWarning) with pytest.raises( @@ -306,25 +318,71 @@ def test_metadata_rejects_chunk_edge_below_one(site: str, size: int) -> None: CHUNK_EDGE_SITES[site](size) -CHUNK_SHAPE_SITES: dict[str, Callable[[Any], object]] = { - "regular": lambda chunk_shape: RegularChunkGridMetadata(chunk_shape=chunk_shape), - "v2": _v2_metadata, - "sharding-inner": lambda chunk_shape: ShardingCodec(chunk_shape=chunk_shape), -} - - -@pytest.mark.parametrize("site", CHUNK_SHAPE_SITES) -@pytest.mark.parametrize("chunk_shape", [4, np.int64(4), "10", {"a": 1}, range(1, 2)]) -def test_metadata_rejects_chunk_shape_not_list_or_tuple(site: str, chunk_shape: object) -> None: - """A regular chunk shape is a list or tuple; anything else is rejected as a whole, - not iterated as if its elements were chunk edge lengths.""" +@pytest.mark.parametrize("chunk_shape", [4, np.int64(4), None]) +def test_regular_chunk_grid_rejects_chunk_shape_not_iterable(chunk_shape: Any) -> None: with pytest.raises( TypeError, match=re.escape( - f"A chunk shape must be a list or tuple of chunk edge lengths, got {chunk_shape!r}" + f"A chunk shape must be an iterable of chunk edge lengths, got {chunk_shape!r}" ), ): - CHUNK_SHAPE_SITES[site](chunk_shape) + RegularChunkGridMetadata(chunk_shape=chunk_shape) + + +def _sharding_chunk_shape(chunks: Any) -> tuple[tuple[int, ...], object]: + codec = ShardingCodec(chunk_shape=chunks) + configuration = cast("dict[str, JSON]", codec.to_dict()["configuration"]) + return codec.chunk_shape, configuration["chunk_shape"] + + +CHUNK_SHAPE_SITES: dict[str, Callable[[Any], tuple[tuple[int, ...], object]]] = { + "v2": lambda chunks: ((md := _v2_metadata(chunks)).chunks, md.to_dict()["chunks"]), + "sharding-inner": _sharding_chunk_shape, +} +"""`ArrayV2Metadata` and `ShardingCodec` read a chunk shape as an array shape, returning +the chunk shape and the value `to_dict` writes for it.""" + + +@pytest.mark.parametrize("site", CHUNK_SHAPE_SITES) +@pytest.mark.parametrize( + ("chunks", "expected"), + [ + ((4,), (4,)), + ([4], (4,)), + (4, (4,)), + (np.int64(4), (4,)), + ((np.int64(4),), (4,)), + (np.array([4]), (4,)), + ((True,), (1,)), + ((0,), (0,)), + ((False,), (0,)), + (range(4, 5), (4,)), + ], +) +def test_chunk_shape_read_as_array_shape( + site: str, chunks: object, expected: tuple[int, ...] +) -> None: + """`ArrayV2Metadata` and `ShardingCodec` read their chunk shape as `parse_shapelike` + reads an array shape: an integer or an iterable of non-negative integers, including + NumPy integers and bools. A chunk size of 0 is written back as given; reading a + stored 0 is `zarr.core.metadata.upgrades`' business.""" + parsed, written = CHUNK_SHAPE_SITES[site](chunks) + assert parsed == expected + assert all(type(size) is int for size in parsed) + assert written == expected + + +@pytest.mark.parametrize("site", CHUNK_SHAPE_SITES) +def test_chunk_shape_read_as_array_shape_rejects_negative(site: str) -> None: + with pytest.raises(ValueError, match="Expected all values to be non-negative"): + CHUNK_SHAPE_SITES[site]((-1,)) + + +@pytest.mark.parametrize("site", CHUNK_SHAPE_SITES) +@pytest.mark.parametrize("chunks", [(4.0,), "4", None]) +def test_chunk_shape_read_as_array_shape_rejects_non_integer(site: str, chunks: object) -> None: + with pytest.raises(TypeError, match="Expected an"): + CHUNK_SHAPE_SITES[site](chunks) def _rewrite_doc(path: Path, zarr_format: Literal[2, 3], edit: Any) -> None: diff --git a/tests/test_unified_chunk_grid.py b/tests/test_unified_chunk_grid.py index 8f06863a05..ae3feacd3d 100644 --- a/tests/test_unified_chunk_grid.py +++ b/tests/test_unified_chunk_grid.py @@ -494,6 +494,8 @@ def test_chunk_grid_iter() -> None: [ ([[10, 3]], [10, 10, 10]), ([[10, 2], [20, 1]], [10, 10, 20]), + ([4.0, [4.0, 2], [5, 2.0]], [4, 4, 4, 5, 5]), + ([[True, 2], [np.int64(3), np.int64(1)]], [1, 1, 3]), ], ) def test_rle_expand(compressed: list[Any], expected: list[int]) -> None: @@ -550,14 +552,14 @@ def test_rle_expand_rejects_invalid(rle_input: list[Any], match: str) -> None: @pytest.mark.parametrize( ("rle_input", "match"), [ - ([10.0], "Chunk edge length must be an int, got 10.0"), - ([True], "Chunk edge length must be an int, got True"), - ([[10, 3.0]], "RLE repeat count must be an int, got 3.0"), + ([10.5], "Chunk edge length must be an integer, got 10.5"), + (["10"], "Chunk edge length must be an integer, got '10'"), + ([[10, 3.5]], "RLE repeat count must be an integer, got 3.5"), ], - ids=["float-edge", "bool-edge", "float-count"], + ids=["fractional-edge", "string-edge", "fractional-count"], ) def test_rle_expand_rejects_non_int(rle_input: list[Any], match: str) -> None: - """expand_rle takes JSON integers only: no stored document holds integral floats.""" + """expand_rle reads integral numbers only.""" with pytest.raises(TypeError, match=match): expand_rle(rle_input) @@ -566,11 +568,11 @@ def test_rle_expand_rejects_non_int(rle_input: list[Any], match: str) -> None: ("rle_input", "match"), [ ([0], "chunk edge length must be >= 1"), - ([10.0], "chunk edge length must be an int"), + ([10.5], "chunk edge length must be an integer"), ([[5, 0]], "RLE repeat count must be >= 1"), ([[5, 2, 1]], r"RLE entries must be an integer or \[size, count\]"), ], - ids=["zero-edge", "float-edge", "zero-rle-count", "rle-entry-of-three"], + ids=["zero-edge", "fractional-edge", "zero-rle-count", "rle-entry-of-three"], ) def test_rle_expand_names_dimension(rle_input: list[Any], match: str) -> None: """Given the dimension `axis` the edges belong to, every error of expand_rle names it.""" From 245acaed57544ec9801c1678181a4898cf331f34 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Sat, 26 Sep 2026 11:14:18 +0200 Subject: [PATCH 27/41] fix(array): leave a valid stored document as written on a stale handle's first write A handle read from an upgraded document re-reads the stored document before its first chunk write. If that document no longer needs an upgrade, store nothing: re-encoding it could differ from how another implementation wrote it, and nothing about it needs fixing. Assisted-by: ClaudeCode:claude-opus-5-5 Co-Authored-By: Claude Opus 5.5 --- src/zarr/core/array.py | 13 +++++++------ tests/test_metadata/test_upgrades.py | 27 +++++++++++++++++++++++++++ 2 files changed, 34 insertions(+), 6 deletions(-) diff --git a/src/zarr/core/array.py b/src/zarr/core/array.py index 64d6080838..da5b0595a8 100644 --- a/src/zarr/core/array.py +++ b/src/zarr/core/array.py @@ -1617,21 +1617,22 @@ async def _save_metadata(self, metadata: ArrayMetadata, ensure_parents: bool = F await save_metadata(self.store_path, metadata, ensure_parents=ensure_parents) async def _store_upgraded_document(self) -> None: - """Store the upgrade of this array's current stored document, if it differs from - what the store holds. + """Store the upgrade of this array's current stored document, if it needs one. Only for metadata read from a document that had to be upgraded (see `zarr.core.metadata.upgrades`). The document is read again, because the store may hold a newer one than this handle's metadata: if that one needs no upgrade (the - array was re-saved or resized since), nothing is stored. Storing the same upgrade - twice is harmless, so concurrent callers need no coordination. + array was re-saved or resized since, possibly by another implementation), it is + left as written. Storing the same upgrade twice is harmless, so concurrent + callers need no coordination. """ if not self.metadata._stored_document_upgraded: return zarr_format = self.metadata.zarr_format stored = await get_array_metadata(self.store_path, zarr_format=zarr_format) - upgraded, _ = upgrade_array_document(stored, zarr_format) - await upsert_metadata(self.store_path, parse_array_metadata(upgraded)) + upgraded, readings = upgrade_array_document(stored, zarr_format) + if readings: + await upsert_metadata(self.store_path, parse_array_metadata(upgraded)) object.__setattr__(self.metadata, "_stored_document_upgraded", False) async def _set_selection( diff --git a/tests/test_metadata/test_upgrades.py b/tests/test_metadata/test_upgrades.py index 62ae417584..5f86ed233c 100644 --- a/tests/test_metadata/test_upgrades.py +++ b/tests/test_metadata/test_upgrades.py @@ -553,6 +553,33 @@ def test_stale_handle_write_keeps_newer_metadata( np.testing.assert_array_equal(reopened[...], [9, 0, 0, 1, 2, 3, 4, 5, 6]) +@pytest.mark.parametrize("zarr_format", [2, 3]) +def test_stale_handle_write_keeps_valid_document_as_written( + tmp_path: Path, zarr_format: Literal[2, 3] +) -> None: + """If the document the store holds when a handle read from an upgraded document + first writes chunks needs no upgrade, it is left as written, even where zarr would + encode the same metadata differently (as another implementation may have written it).""" + path = tmp_path / "legacy.zarr" + _legacy_array(path, zarr_format) + with pytest.warns(ZarrUserWarning, match="is read as"): + stale = zarr.open_array(store=path, mode="r+") + # Valid metadata for the same array, as another writer might store it: chunk size + # 3, without the optional members zarr writes, as compact JSON. + doc_path = path / (".zarray" if zarr_format == 2 else "zarr.json") + doc = json.loads(doc_path.read_text()) + for optional in ("dimension_separator", "attributes", "storage_transformers"): + doc.pop(optional, None) + _stored_chunks(doc)[0] = 3 + doc_path.write_text(json.dumps(doc, separators=(",", ":"))) + written = doc_path.read_bytes() + + stale[0] = 9 + + assert doc_path.read_bytes() == written + np.testing.assert_array_equal(_open_strictly(path)[...], [9, 0, 0]) + + @pytest.mark.parametrize("zarr_format", [2, 3]) def test_empty_write_stores_no_metadata(tmp_path: Path, zarr_format: Literal[2, 3]) -> None: """A write of an empty selection stores no chunks, so it stores no metadata either.""" From 2d0121db7ea51eb0b4c318c51809bba0824b94ad Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Sat, 26 Sep 2026 11:15:35 +0200 Subject: [PATCH 28/41] docs: describe the 4334 changes to metadata constructors for a patch release Assisted-by: ClaudeCode:claude-opus-5-5 Co-Authored-By: Claude Opus 5.5 --- changes/4334.bugfix.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/changes/4334.bugfix.md b/changes/4334.bugfix.md index 460ba6ece8..220b04e805 100644 --- a/changes/4334.bugfix.md +++ b/changes/4334.bugfix.md @@ -1,5 +1,5 @@ A chunk edge length is now always at least 1, while an array extent may be 0. Every spelling of one chunk spanning an axis (`chunks=-1`, `chunks=False`, `chunks="auto"`, `shards=-1`, `shards=False`) gives chunk size 1 on a zero-length axis, and a shard spanning an axis is a multiple of the inner chunk size, so `shards=-1` no longer fails when the axis length is not a multiple of it. Rectilinear chunk grids can be created on a zero-length dimension with a non-empty list of positive chunk sizes, which are kept for later growth. -Metadata built in code is strict about chunk edge lengths: `ArrayV2Metadata`, `RegularChunkGridMetadata`, `RectilinearChunkGridMetadata` and `ShardingCodec` take them as Python `int`s of at least 1, so a size of 0 now raises a `ValueError`, and a `bool`, a float or a NumPy integer (including a scalar `chunks=np.int64(5)` for `ArrayV2Metadata`) raises a `TypeError`, as does a chunk shape that is not a list or tuple; `FixedDimension(size=0, ...)` raises a `ValueError`. Stored rectilinear chunk grids with integral floats (`10.0`) are rejected too. The array creation functions still accept NumPy integers in `chunks=` and `shards=`. +`FixedDimension(size=0, ...)` now raises a `ValueError`, so an array can no longer be built directly from metadata with a chunk size of 0; `ArrayV2Metadata(chunks=(0,))` itself is still accepted and written as given. The chunk grid metadata classes read a chunk edge length given as a NumPy integer, a `bool` or an integral float as the `int` it equals; `RectilinearChunkGridMetadata` used to keep such a value as given, so that it could not be stored (a NumPy integer) or read back (a `bool`). Stored metadata with a regular chunk size of 0 or JSON `false` now opens with a `ZarrUserWarning` and is read as one chunk spanning the axis (a multiple of the inner chunk size for sharded arrays, which previously failed to open). zarr-python wrote such sizes for arrays created with a zero-length axis until 3.4, and, from 2.18.7 to 3.2.1, for an explicit chunk size of 0 or `False` on an axis of any length, as in `chunks=(0,)` (those arrays could store no data). JSON `true`, as zarr-python 3.0 and 3.2 wrote for `chunks=(True,)`, is read as 1, in the chunk grid and in the inner chunk shape of a sharding codec. The warning names the array and says how to store valid metadata: `array.update_attributes({})`, then `zarr.consolidate_metadata` if the metadata is consolidated. Writing data to such an array (other than an empty selection) first stores the upgrade of the metadata the store then holds, unless another writer has stored valid metadata since, so that other readers find the chunks written; `zarr.consolidate_metadata` likewise stores the upgraded metadata of each such array before the consolidated metadata. From 4c4fe219e124f3e71182a36148a1e1144d77f244 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Sat, 26 Sep 2026 12:05:59 +0200 Subject: [PATCH 29/41] fix(metadata): read float edges only in stored rectilinear documents The patch release widened the one chunk edge rule to accept integral floats so that stored rectilinear grids written with float edges (`[[4.0, 2]]`, as zarr 3.2 wrote for float edges) keep opening. That also made a stored regular chunk shape `[10.0]` open, which zarr 3.4.0 rejected and the next minor release would reject again. The constructor rule now accepts integers of any integer type (`int`, `bool`, NumPy integers) and rejects floats. A new document upgrade reads an integral JSON float of at least 1 as an `int` where zarr 3.2 stored one: the explicit edges and run-length encoded sizes of a rectilinear chunk grid, with the standard warning. Bare sizes, repeat counts, regular chunk shapes and inner chunk shapes stay for the constructors to reject. Assisted-by: ClaudeCode:claude-opus-5-5 Co-Authored-By: Claude Opus 5.5 --- changes/4334.bugfix.md | 4 +- src/zarr/core/common.py | 20 ++-- src/zarr/core/metadata/upgrades.py | 60 +++++++++++- tests/test_metadata/test_upgrades.py | 132 +++++++++++++++++++++++++-- tests/test_unified_chunk_grid.py | 15 ++- 5 files changed, 206 insertions(+), 25 deletions(-) diff --git a/changes/4334.bugfix.md b/changes/4334.bugfix.md index 220b04e805..4e719b78c6 100644 --- a/changes/4334.bugfix.md +++ b/changes/4334.bugfix.md @@ -1,5 +1,5 @@ A chunk edge length is now always at least 1, while an array extent may be 0. Every spelling of one chunk spanning an axis (`chunks=-1`, `chunks=False`, `chunks="auto"`, `shards=-1`, `shards=False`) gives chunk size 1 on a zero-length axis, and a shard spanning an axis is a multiple of the inner chunk size, so `shards=-1` no longer fails when the axis length is not a multiple of it. Rectilinear chunk grids can be created on a zero-length dimension with a non-empty list of positive chunk sizes, which are kept for later growth. -`FixedDimension(size=0, ...)` now raises a `ValueError`, so an array can no longer be built directly from metadata with a chunk size of 0; `ArrayV2Metadata(chunks=(0,))` itself is still accepted and written as given. The chunk grid metadata classes read a chunk edge length given as a NumPy integer, a `bool` or an integral float as the `int` it equals; `RectilinearChunkGridMetadata` used to keep such a value as given, so that it could not be stored (a NumPy integer) or read back (a `bool`). +`FixedDimension(size=0, ...)` now raises a `ValueError`, so an array can no longer be built directly from metadata with a chunk size of 0; `ArrayV2Metadata(chunks=(0,))` itself is still accepted and written as given. The chunk grid metadata classes read a chunk edge length given as a NumPy integer or a `bool` as the `int` it equals; `RectilinearChunkGridMetadata` used to keep such a value as given, so that it could not be stored (a NumPy integer) or read back (a `bool`). `RectilinearChunkGridMetadata` now rejects a float edge length such as `4.0` with a `TypeError`, as the other metadata classes do; it used to keep it and store it as a JSON float, which the Zarr specification does not allow. -Stored metadata with a regular chunk size of 0 or JSON `false` now opens with a `ZarrUserWarning` and is read as one chunk spanning the axis (a multiple of the inner chunk size for sharded arrays, which previously failed to open). zarr-python wrote such sizes for arrays created with a zero-length axis until 3.4, and, from 2.18.7 to 3.2.1, for an explicit chunk size of 0 or `False` on an axis of any length, as in `chunks=(0,)` (those arrays could store no data). JSON `true`, as zarr-python 3.0 and 3.2 wrote for `chunks=(True,)`, is read as 1, in the chunk grid and in the inner chunk shape of a sharding codec. The warning names the array and says how to store valid metadata: `array.update_attributes({})`, then `zarr.consolidate_metadata` if the metadata is consolidated. Writing data to such an array (other than an empty selection) first stores the upgrade of the metadata the store then holds, unless another writer has stored valid metadata since, so that other readers find the chunks written; `zarr.consolidate_metadata` likewise stores the upgraded metadata of each such array before the consolidated metadata. +Stored metadata with a regular chunk size of 0 or JSON `false` now opens with a `ZarrUserWarning` and is read as one chunk spanning the axis (a multiple of the inner chunk size for sharded arrays, which previously failed to open). zarr-python wrote such sizes for arrays created with a zero-length axis until 3.4, and, from 2.18.7 to 3.2.1, for an explicit chunk size of 0 or `False` on an axis of any length, as in `chunks=(0,)` (those arrays could store no data). JSON `true`, as zarr-python 3.0 and 3.2 wrote for `chunks=(True,)`, is read as 1, in the chunk grid and in the inner chunk shape of a sharding codec. A stored rectilinear chunk grid whose edge lengths are integral JSON floats (`[[4.0, 2]]`), as zarr-python 3.2 wrote for float edges, is read with those edges as integers, now with a warning; a float anywhere no release wrote one (a regular chunk shape, the inner chunk shape of a sharding codec, a run-length repeat count) is rejected. The warning names the array and says how to store valid metadata: `array.update_attributes({})`, then `zarr.consolidate_metadata` if the metadata is consolidated. Writing data to such an array (other than an empty selection) first stores the upgrade of the metadata the store then holds, unless another writer has stored valid metadata since, so that other readers find the chunks written; `zarr.consolidate_metadata` likewise stores the upgraded metadata of each such array before the consolidated metadata. diff --git a/src/zarr/core/common.py b/src/zarr/core/common.py index cfeb6d3371..dc85737289 100644 --- a/src/zarr/core/common.py +++ b/src/zarr/core/common.py @@ -283,24 +283,22 @@ def _subject(name: str, axis: int | None) -> str: def _parse_positive_int(value: object, name: str, axis: int | None) -> int: - """`value` as an `int` of at least 1. Any integral number is read as the `int` it - equals: an `int`, a `bool`, a NumPy integer or an integral float.""" + """`value` as an `int` of at least 1. An integer of any integer type (an `int`, a + `bool` or a NumPy integer) is read as the `int` it equals; a float is rejected, even + an integral one (stored documents with integral floats are read by + `zarr.core.metadata.upgrades`).""" subject = _subject(name, axis) - match value: - case numbers.Integral(): - parsed = int(value) - case float() if value.is_integer(): - parsed = int(value) - case _: - raise TypeError(f"{subject} must be an integer, got {value!r}") + if not isinstance(value, numbers.Integral): + raise TypeError(f"{subject} must be an integer, got {value!r}") + parsed = int(value) if parsed < 1: raise ValueError(f"{subject} must be >= 1, got {value!r}") return parsed def parse_chunk_edge(size: object, axis: int | None = None) -> int: - """Check that `size` is a chunk edge length: an integral number of at least 1, read - as an `int`. + """Check that `size` is a chunk edge length: an integer of at least 1, read as an + `int`. This is the one rule for chunk edge lengths in metadata: bare chunk sizes, explicit edges and run-length encoded sizes. `axis`, when given, is named in the error. diff --git a/src/zarr/core/metadata/upgrades.py b/src/zarr/core/metadata/upgrades.py index 8589a0f0c4..4d978d36c9 100644 --- a/src/zarr/core/metadata/upgrades.py +++ b/src/zarr/core/metadata/upgrades.py @@ -57,6 +57,12 @@ def _is_int_list(value: object) -> TypeGuard[list[int]]: return isinstance(value, list) and all(isinstance(v, int) for v in value) +def _abbreviate(value: JSON, limit: int = 60) -> str: + """`value` as JSON, cut to at most `limit` characters.""" + text = json.dumps(value) + return text if len(text) <= limit else f"{text[: limit - 3]}..." + + def _read_chunk_size(size: JSON, span: int | None, unit: int) -> tuple[int, str | None] | None: """Read one entry of a stored regular chunk shape as a chunk edge length. @@ -180,9 +186,61 @@ def _invalid_chunk_sizes_v3(doc: ArrayDocument) -> tuple[ArrayDocument, str] | N return None +def _read_edge_length(edge: JSON) -> JSON: + """Read one stored chunk edge length of a rectilinear chunk grid: an integral JSON + float of at least 1 (`4.0`) is read as the `int` it equals. Anything else is kept, + for the metadata constructors to check.""" + match edge: + case float() if edge.is_integer() and edge >= 1: + return int(edge) + return edge + + +def _read_rectilinear_axis(spec: JSON) -> JSON: + """Read the stored chunk edge lengths of one axis of a rectilinear chunk grid (see + `_read_edge_length`): its explicit edges and the sizes of its run-length encoded + `[size, count]` pairs. A bare chunk size and a repeat count are kept as stored: no + writer stored them as floats.""" + match spec: + case list(): + return [_read_rectilinear_entry(entry) for entry in spec] + return spec + + +def _read_rectilinear_entry(entry: JSON) -> JSON: + match entry: + case [size, count]: + return [_read_edge_length(size), count] + return _read_edge_length(entry) + + +def _float_edge_lengths_v3(doc: ArrayDocument) -> tuple[ArrayDocument, str] | None: + grid = doc.get("chunk_grid") + if not (isinstance(grid, Mapping) and grid.get("name") == "rectilinear"): + return None + configuration = grid.get("configuration") + if not isinstance(configuration, Mapping): + return None + stored = configuration.get("chunk_shapes") + if not isinstance(stored, list): + return None + read = [_read_rectilinear_axis(axis) for axis in stored] + # An integral float and the `int` it equals compare equal, but not as JSON text. + axes = [axis for axis, spec in enumerate(stored) if json.dumps(spec) != json.dumps(read[axis])] + if not axes: + return None + reading = ( + f"The stored chunk edge lengths {_abbreviate(stored)} are invalid: chunk edge " + f"lengths must be integers. They are read as {_abbreviate(read)}, reading each " + f"edge length stored as a float, in dimensions {axes}, as the integer it equals." + ) + upgraded = {**configuration, "chunk_shapes": read} + return {**doc, "chunk_grid": {**grid, "configuration": upgraded}}, reading + + ARRAY_UPGRADES: Final[Mapping[ZarrFormat, tuple[Upgrade, ...]]] = { 2: (_invalid_chunk_sizes_v2,), - 3: (_invalid_inner_chunk_sizes_v3, _invalid_chunk_sizes_v3), + 3: (_invalid_inner_chunk_sizes_v3, _invalid_chunk_sizes_v3, _float_edge_lengths_v3), } """The upgrades of an array document of each Zarr format, applied in order.""" diff --git a/tests/test_metadata/test_upgrades.py b/tests/test_metadata/test_upgrades.py index 5f86ed233c..f2da059b16 100644 --- a/tests/test_metadata/test_upgrades.py +++ b/tests/test_metadata/test_upgrades.py @@ -47,7 +47,7 @@ def _v2_doc(shape: list[int], chunks: list[Any]) -> dict[str, JSON]: def _v3_doc( - shape: list[int], chunk_shape: list[Any], inner: list[int] | None = None + shape: list[int], chunk_shape: list[Any], inner: list[Any] | None = None ) -> dict[str, JSON]: bytes_codec: dict[str, JSON] = {"name": "bytes", "configuration": {"endian": "little"}} codecs: list[JSON] = [bytes_codec] @@ -240,6 +240,116 @@ def test_stored_fractional_chunk_size_rejected() -> None: _read_strictly(_v3_doc([4], [4.5])) +def _rectilinear_doc(shape: list[int], chunk_shapes: list[Any]) -> dict[str, JSON]: + return _v3_doc(shape, [1] * len(shape)) | { + "chunk_grid": { + "name": "rectilinear", + "configuration": {"kind": "inline", "chunk_shapes": chunk_shapes}, + } + } + + +@pytest.mark.parametrize( + ("chunk_shapes", "expected", "warning"), + [ + ([[[4, 2]], [[5, 2]]], ((4, 4), (5, 5)), None), + ( + [[[4.0, 2]], [[5, 2]]], + ((4, 4), (5, 5)), + ( + r"^The stored chunk edge lengths \[\[\[4\.0, 2\]\], \[\[5, 2\]\]\] are " + r"invalid: .* read as \[\[\[4, 2\]\], \[\[5, 2\]\]\], .* in dimensions \[0\]," + ), + ), + ([[3.0, 5.0], [[5, 2]]], ((3, 5), (5, 5)), r"read as \[\[3, 5\], "), + ([[[4.0, 2], 2.0], [[5, 2]]], ((4, 4, 2), (5, 5)), r"read as \[\[\[4, 2\], 2\], "), + ([[[4.0, 2]], [10]], ((4, 4), (10,)), r"read as \[\[\[4, 2\]\], \[10\]\]"), + ([[[4.0, 2]], [5.0, 5]], ((4, 4), (5, 5)), r"in dimensions \[0, 1\],"), + ], + ids=["valid", "rle-size", "edges", "rle-size-and-edge", "sharded-outer", "2d"], +) +def test_read_float_edges_in_rectilinear_grid( + chunk_shapes: list[Any], + expected: tuple[tuple[int, ...], ...], + warning: str | None, +) -> None: + """A stored rectilinear chunk grid whose explicit edges or run-length encoded sizes + are integral floats, as zarr-python wrote them when given float edges, is read with + those edges as `int`s; `from_dict` warns once, naming the array and the dimensions, + and how to re-save.""" + shape = [sum(edges) for edges in expected] + doc = _rectilinear_doc(shape, chunk_shapes) + with ( + zarr.config.set({"array.rectilinear_chunks": True}), + warnings.catch_warnings(record=True) as record, + ): + warnings.simplefilter("always") + metadata = ArrayV3Metadata.from_dict(doc, path="group/array") + assert metadata.chunk_grid == RectilinearChunkGridMetadata(chunk_shapes=expected) + messages = [str(w.message) for w in record] + if warning is None: + assert messages == [] + else: + [message] = messages + assert message.startswith("Array 'group/array': ") + assert re.search(warning, message.removeprefix("Array 'group/array': ")) + assert message.endswith(RESAVE_HINT) + + +@pytest.mark.parametrize( + ("doc", "error"), + [ + (_v3_doc([20], [10.0]), "Dimension 0: chunk edge length must be an integer, got 10.0"), + ( + _v3_doc([8], [4], inner=[2.0]), + "Expected an iterable of integers. Got [2.0] instead.", + ), + (_v2_doc([20], [10.0]), "Expected an iterable of integers. Got [10.0] instead."), + ( + _rectilinear_doc([8], [[[4, 2.0]]]), + "Dimension 0: RLE repeat count must be an integer, got 2.0", + ), + ( + _rectilinear_doc([8], [4.0]), + "Dimension 0: chunk edge length must be an integer, got 4.0", + ), + ], + ids=["regular", "sharding-inner", "v2", "rle-count", "rectilinear-bare"], +) +def test_stored_float_chunk_size_rejected(doc: dict[str, JSON], error: str) -> None: + """A float chunk size is read only where zarr-python stored one, as the edges of a + rectilinear chunk grid. Anywhere else it is rejected, as zarr 3.4.0 rejected a + stored regular chunk size of `10.0`.""" + with zarr.config.set({"array.rectilinear_chunks": True}), pytest.raises(TypeError) as info: + _read_strictly(doc) + assert info.match(re.escape(error)) + + +def test_float_edges_round_trip(tmp_path: Path) -> None: + """A store whose rectilinear chunk grid holds the float edges zarr-python wrote opens + with a warning, reads its data and re-saves the edges as `int`s.""" + path = tmp_path / "float.zarr" + data = np.arange(80, dtype="int16").reshape(8, 10) + with zarr.config.set({"array.rectilinear_chunks": True}): + zarr.create_array(path, shape=data.shape, chunks=[[4, 4], [5, 5]], dtype="int16")[...] = ( + data + ) + + def store_floats(doc: dict[str, Any]) -> None: + doc["chunk_grid"]["configuration"]["chunk_shapes"] = [[[4.0, 2]], [[5, 2]]] + + _rewrite_doc(path, 3, store_floats) + with pytest.warns(ZarrUserWarning, match=r"read as \[\[\[4, 2\]\], \[\[5, 2\]\]\]"): + arr = zarr.open_array(path, mode="a") + np.testing.assert_array_equal(arr[...], data) + arr.update_attributes({}) + stored = json.loads((path / "zarr.json").read_text())["chunk_grid"]["configuration"] + assert json.dumps(stored["chunk_shapes"]) == "[[[4, 2]], [[5, 2]]]" + with warnings.catch_warnings(): + warnings.simplefilter("error", ZarrUserWarning) + np.testing.assert_array_equal(zarr.open_array(path)[...], data) + + def test_v2_constructor_rejects_chunks_of_wrong_length() -> None: with pytest.raises(ValueError, match="`chunks` has length 1, but `shape` has length 2"): ArrayV2Metadata(shape=(4, 4), chunks=(2,), dtype=Int16(), fill_value=0, order="C") @@ -284,20 +394,26 @@ def _second_edge(grid: RectilinearChunkGridMetadata) -> int: @pytest.mark.parametrize("site", CHUNK_EDGE_SITES) @pytest.mark.parametrize( ("size", "expected"), - [(4, 4), (True, 1), (np.int64(4), 4), (4.0, 4), (np.float64(4.0), 4)], - ids=["int", "bool", "numpy-int", "float", "numpy-float"], + [(4, 4), (True, 1), (np.int64(4), 4)], + ids=["int", "bool", "numpy-int"], ) -def test_metadata_reads_integral_chunk_edge(site: str, size: object, expected: int) -> None: - """Chunk grid metadata reads any integral number as the `int` chunk edge length it - equals.""" +def test_metadata_reads_integer_chunk_edge(site: str, size: object, expected: int) -> None: + """Chunk grid metadata reads an integer of any integer type as the `int` chunk edge + length it equals.""" edge = CHUNK_EDGE_SITES[site](size) assert type(edge) is int assert edge == expected @pytest.mark.parametrize("site", CHUNK_EDGE_SITES) -@pytest.mark.parametrize("size", [4.5, float("inf"), "4", None]) -def test_metadata_rejects_non_integral_chunk_edge(site: str, size: object) -> None: +@pytest.mark.parametrize( + "size", + [4.0, np.float64(4.0), 4.5, float("inf"), "4", None], + ids=["float", "numpy-float", "fractional", "inf", "str", "none"], +) +def test_metadata_rejects_non_integer_chunk_edge(site: str, size: object) -> None: + """A float is not a chunk edge length in metadata built in code, even an integral + one; stored documents with integral floats are read by the upgrades.""" with pytest.raises( TypeError, match=re.escape(f"Dimension 0: chunk edge length must be an integer, got {size!r}"), diff --git a/tests/test_unified_chunk_grid.py b/tests/test_unified_chunk_grid.py index ae3feacd3d..8df53df08b 100644 --- a/tests/test_unified_chunk_grid.py +++ b/tests/test_unified_chunk_grid.py @@ -494,7 +494,6 @@ def test_chunk_grid_iter() -> None: [ ([[10, 3]], [10, 10, 10]), ([[10, 2], [20, 1]], [10, 10, 20]), - ([4.0, [4.0, 2], [5, 2.0]], [4, 4, 4, 5, 5]), ([[True, 2], [np.int64(3), np.int64(1)]], [1, 1, 3]), ], ) @@ -553,13 +552,23 @@ def test_rle_expand_rejects_invalid(rle_input: list[Any], match: str) -> None: ("rle_input", "match"), [ ([10.5], "Chunk edge length must be an integer, got 10.5"), + ([10.0], "Chunk edge length must be an integer, got 10.0"), + ([[10.0, 3]], "Chunk edge length must be an integer, got 10.0"), (["10"], "Chunk edge length must be an integer, got '10'"), ([[10, 3.5]], "RLE repeat count must be an integer, got 3.5"), + ([[10, 3.0]], "RLE repeat count must be an integer, got 3.0"), + ], + ids=[ + "fractional-edge", + "float-edge", + "float-rle-size", + "string-edge", + "fractional-count", + "float-count", ], - ids=["fractional-edge", "string-edge", "fractional-count"], ) def test_rle_expand_rejects_non_int(rle_input: list[Any], match: str) -> None: - """expand_rle reads integral numbers only.""" + """expand_rle reads integers only, not floats.""" with pytest.raises(TypeError, match=match): expand_rle(rle_input) From 6fde402d18b48e4721d11bed6b4849f727786b89 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Sat, 26 Sep 2026 14:31:27 +0200 Subject: [PATCH 30/41] test(codecs): pickle a ShardingCodec with an inner chunk size of 0 Assisted-by: ClaudeCode:claude-opus-5-5 Co-Authored-By: Claude Opus 5.5 --- tests/test_codecs/test_sharding.py | 25 ++++++++++++++++--------- 1 file changed, 16 insertions(+), 9 deletions(-) diff --git a/tests/test_codecs/test_sharding.py b/tests/test_codecs/test_sharding.py index e2619a6ce2..831285ae9f 100644 --- a/tests/test_codecs/test_sharding.py +++ b/tests/test_codecs/test_sharding.py @@ -760,17 +760,24 @@ def test_structured_dtype_fill_value() -> None: assert np.array_equal(arr[:], expected) -def test_pickle() -> None: +@pytest.mark.parametrize( + "codec", + [ + ShardingCodec(chunk_shape=(8, 8)), + ShardingCodec(chunk_shape=(8, 8), subchunk_write_order="lexicographic"), + ShardingCodec(chunk_shape=(0,)), + ], + ids=["default", "lexicographic", "zero-chunk-size"], +) +def test_pickle(codec: ShardingCodec) -> None: """ShardingCodec round-trips through pickle, including the non-serialized ``subchunk_write_order`` (which ``to_dict`` omits and which must not silently - revert to the ``morton`` default).""" - codec = ShardingCodec(chunk_shape=(8, 8)) - assert pickle.loads(pickle.dumps(codec)) == codec - - ordered = ShardingCodec(chunk_shape=(8, 8), subchunk_write_order="lexicographic") - restored = pickle.loads(pickle.dumps(ordered)) - assert restored == ordered - assert restored.subchunk_write_order == "lexicographic" + revert to the ``morton`` default), and an inner chunk size of 0, which the + constructor accepts.""" + restored = pickle.loads(pickle.dumps(codec)) + assert restored == codec + assert restored.chunk_shape == codec.chunk_shape + assert restored.subchunk_write_order == codec.subchunk_write_order @pytest.mark.parametrize("store", ["local", "memory"], indirect=["store"]) From b0ea125cb22e7feaeaa48c22dfaad5b589498d33 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Sat, 26 Sep 2026 14:31:31 +0200 Subject: [PATCH 31/41] fix(group): build every node before create_hierarchy deletes or stores anything A node whose metadata no array or group can be built from now fails with the store untouched, even with overwrite=True. Assisted-by: ClaudeCode:claude-opus-5-5 Co-Authored-By: Claude Opus 5.5 --- src/zarr/core/group.py | 38 ++++++++++++++++++++++++++++---------- tests/test_group.py | 24 ++++++++++++++++++++++++ 2 files changed, 52 insertions(+), 10 deletions(-) diff --git a/src/zarr/core/group.py b/src/zarr/core/group.py index d6cc572465..38eede55ec 100644 --- a/src/zarr/core/group.py +++ b/src/zarr/core/group.py @@ -3085,6 +3085,7 @@ async def create_hierarchy( # ensure that all nodes have the same zarr_format, and add implicit groups as needed nodes_parsed = _parse_hierarchy_dict(data=nodes_normed_keys) redundant_implicit_groups = [] + to_delete_keys: list[str] = [] # empty hierarchies should be a no-op if len(nodes_parsed) > 0: @@ -3117,13 +3118,7 @@ async def create_hierarchy( if overwrite: # we will remove any nodes that collide with arrays and non-implicit groups defined in # nodes - - # track the keys of nodes we need to delete - to_delete_keys = [] - to_delete_keys.extend( - [k for k, v in nodes_parsed.items() if k not in implicit_group_keys] - ) - await asyncio.gather(*(store.delete_dir(key) for key in to_delete_keys)) + to_delete_keys = [k for k in nodes_parsed if k not in implicit_group_keys] else: # This type is long. coros: ( @@ -3183,7 +3178,11 @@ async def create_hierarchy( else: nodes_explicit[k] = v - async for key, node in create_nodes(store=store, nodes=nodes_explicit): + # Build every node before deleting or storing anything: metadata that no array or + # group can be built from then fails with the store untouched. + built = _build_nodes(store, nodes_explicit) + await asyncio.gather(*(store.delete_dir(key) for key in to_delete_keys)) + async for key, node in _store_nodes(store, nodes_explicit, built): yield key, node @@ -3213,7 +3212,26 @@ async def create_nodes( AsyncGroup | AsyncArray The created nodes in the order they are created. """ + async for key, node in _store_nodes(store, nodes, _build_nodes(store, nodes)): + yield key, node + +def _build_nodes( + store: Store, nodes: Mapping[str, GroupMetadata | ArrayV2Metadata | ArrayV3Metadata] +) -> dict[str, AsyncGroup | AnyAsyncArray]: + """The array or group each of `nodes` describes, at its path in `store`.""" + return { + path: _build_node(store=store, path=path, metadata=meta) for path, meta in nodes.items() + } + + +async def _store_nodes( + store: Store, + nodes: Mapping[str, GroupMetadata | ArrayV2Metadata | ArrayV3Metadata], + built: Mapping[str, AsyncGroup | AnyAsyncArray], +) -> AsyncIterator[tuple[str, AsyncGroup | AnyAsyncArray]]: + """Store the metadata of `nodes` and yield the nodes `_build_nodes` built from them + (see `create_nodes`).""" # Note: the only way to alter this value is via the config. If that's undesirable for some reason, # then we should consider adding a keyword argument to this function semaphore = asyncio.Semaphore(config.get("async.concurrency")) @@ -3241,7 +3259,7 @@ async def create_nodes( node_name = created_key[: created_key.rfind("/")] meta_out = nodes[node_name] if meta_out.zarr_format == 3: - yield node_name, _build_node(store=store, path=node_name, metadata=meta_out) + yield node_name, built[node_name] else: # For zarr v2 # we only want to yield when both the metadata and attributes are created @@ -3256,7 +3274,7 @@ async def create_nodes( meta_done = _join_paths([node_name, ZARRAY_JSON]) in created_object_keys if meta_done and attrs_done: - yield node_name, _build_node(store=store, path=node_name, metadata=meta_out) + yield node_name, built[node_name] continue diff --git a/tests/test_group.py b/tests/test_group.py index 31fbd138cd..b6da55fa2a 100644 --- a/tests/test_group.py +++ b/tests/test_group.py @@ -1919,6 +1919,30 @@ async def test_create_hierarchy( assert expected_meta == {k: v.metadata for k, v in created.items()} +@pytest.mark.parametrize("zarr_format", [2, 3]) +def test_create_hierarchy_unbuildable_node_leaves_store_untouched( + monkeypatch: pytest.MonkeyPatch, zarr_format: ZarrFormat +) -> None: + """`create_hierarchy` builds every node before it deletes or stores anything, so a + node that cannot be built fails with the store untouched, even when overwriting.""" + store = MemoryStore() + group = zarr.create_group(store, zarr_format=zarr_format) + group.create_array("a", shape=(2,), chunks=(1,), dtype="int8")[:] = [1, 2] + before = dict(store._store_dict) + + def unbuildable(**kwargs: object) -> None: + raise RuntimeError("cannot build this node") + + monkeypatch.setattr(zarr.core.group, "_build_node", unbuildable) + with pytest.raises(RuntimeError, match="cannot build this node"): + dict( + zarr.create_hierarchy( + store=store, nodes={"a": GroupMetadata(zarr_format=zarr_format)}, overwrite=True + ) + ) + assert store._store_dict == before + + @pytest.mark.parametrize("store", ["memory"], indirect=True) @pytest.mark.parametrize("extant_node", ["array", "group"]) @pytest.mark.parametrize("impl", ["async", "sync"]) From a903d66a96c5bc15c46ca3ae8653b8916b1ea25e Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Sat, 26 Sep 2026 14:31:33 +0200 Subject: [PATCH 32/41] fix(metadata): warn only where the user must act; guard and refresh upgraded copies - Readings that give what zarr 3.4.0 read (0/false on an empty axis, JSON true, float rectilinear edges) are silent; the metadata is still marked upgraded, so the first write re-saves it. - JSON true is read as 1 in stored rectilinear edges and RLE sizes and in nested sharding codecs' inner chunk shapes. - A write through a handle whose stored document now lays out chunks differently raises and stores nothing; with no stored document it writes as before. - Storing group metadata refreshes each upgraded consolidated member from its own stored document (the one place: save_metadata); consolidate_metadata no longer needs its own pass. - Metadata built in code with chunk size 0 is read through the upgrades, so an array can be built from it, as in 3.4.0. - Chunk edges in metadata follow the 3.4.0 integer rule (int; bool read as its value); a string or mapping is rejected as a chunk shape as a whole. Assisted-by: ClaudeCode:claude-opus-5-5 Co-Authored-By: Claude Opus 5.5 --- changes/4334.bugfix.md | 6 +- src/zarr/api/asynchronous.py | 11 +- src/zarr/core/_json.py | 6 + src/zarr/core/array.py | 65 +++- src/zarr/core/common.py | 27 +- src/zarr/core/group.py | 6 +- src/zarr/core/metadata/io.py | 57 +++- src/zarr/core/metadata/upgrades.py | 211 ++++++------ src/zarr/core/metadata/v2.py | 3 +- src/zarr/core/metadata/v3.py | 3 +- tests/test_metadata/test_upgrades.py | 473 ++++++++++++++++++--------- tests/test_unified_chunk_grid.py | 25 +- 12 files changed, 572 insertions(+), 321 deletions(-) diff --git a/changes/4334.bugfix.md b/changes/4334.bugfix.md index 4e719b78c6..3828d097eb 100644 --- a/changes/4334.bugfix.md +++ b/changes/4334.bugfix.md @@ -1,5 +1,7 @@ A chunk edge length is now always at least 1, while an array extent may be 0. Every spelling of one chunk spanning an axis (`chunks=-1`, `chunks=False`, `chunks="auto"`, `shards=-1`, `shards=False`) gives chunk size 1 on a zero-length axis, and a shard spanning an axis is a multiple of the inner chunk size, so `shards=-1` no longer fails when the axis length is not a multiple of it. Rectilinear chunk grids can be created on a zero-length dimension with a non-empty list of positive chunk sizes, which are kept for later growth. -`FixedDimension(size=0, ...)` now raises a `ValueError`, so an array can no longer be built directly from metadata with a chunk size of 0; `ArrayV2Metadata(chunks=(0,))` itself is still accepted and written as given. The chunk grid metadata classes read a chunk edge length given as a NumPy integer or a `bool` as the `int` it equals; `RectilinearChunkGridMetadata` used to keep such a value as given, so that it could not be stored (a NumPy integer) or read back (a `bool`). `RectilinearChunkGridMetadata` now rejects a float edge length such as `4.0` with a `TypeError`, as the other metadata classes do; it used to keep it and store it as a JSON float, which the Zarr specification does not allow. +`FixedDimension(size=0, ...)` now raises a `ValueError`. `ArrayV2Metadata(chunks=(0,))` is still accepted and written as given; an array built from such metadata (with `create_hierarchy`, for example) reads the chunk size as a stored chunk size of 0 is read (below). `create_hierarchy` now builds every array and group before it deletes or stores anything, so a node that cannot be built fails with the store untouched. The regular chunk grid metadata class reads a `bool` chunk edge length as the `int` it equals, and rejects a string or a mapping as a chunk shape as a whole. `RectilinearChunkGridMetadata`, which is experimental, now reads a `bool` edge length as the `int` it equals and rejects a NumPy integer or a float edge length such as `4.0` with a `TypeError`; it used to keep them as given, so that it could not store a NumPy integer, could not read back a `bool`, and stored a float as a JSON float, which the Zarr specification does not allow. -Stored metadata with a regular chunk size of 0 or JSON `false` now opens with a `ZarrUserWarning` and is read as one chunk spanning the axis (a multiple of the inner chunk size for sharded arrays, which previously failed to open). zarr-python wrote such sizes for arrays created with a zero-length axis until 3.4, and, from 2.18.7 to 3.2.1, for an explicit chunk size of 0 or `False` on an axis of any length, as in `chunks=(0,)` (those arrays could store no data). JSON `true`, as zarr-python 3.0 and 3.2 wrote for `chunks=(True,)`, is read as 1, in the chunk grid and in the inner chunk shape of a sharding codec. A stored rectilinear chunk grid whose edge lengths are integral JSON floats (`[[4.0, 2]]`), as zarr-python 3.2 wrote for float edges, is read with those edges as integers, now with a warning; a float anywhere no release wrote one (a regular chunk shape, the inner chunk shape of a sharding codec, a run-length repeat count) is rejected. The warning names the array and says how to store valid metadata: `array.update_attributes({})`, then `zarr.consolidate_metadata` if the metadata is consolidated. Writing data to such an array (other than an empty selection) first stores the upgrade of the metadata the store then holds, unless another writer has stored valid metadata since, so that other readers find the chunks written; `zarr.consolidate_metadata` likewise stores the upgraded metadata of each such array before the consolidated metadata. +Stored metadata with a regular chunk size of 0 or JSON `false` (Zarr format 2 `chunks`, Zarr format 3 `regular` `chunk_shape`) is now read as one chunk spanning the axis (a multiple of the inner chunk size for sharded arrays, which previously failed to open). zarr-python wrote such sizes for arrays created with a zero-length axis until 3.4, and, from 2.18.7 to 3.2.1, for an explicit chunk size of 0 or `False` on an axis of any length, as in `chunks=(0,)` (those arrays could store no data). On a zero-length axis this opens silently, as before; appending to such an axis then stores chunks of size 1. On an axis of positive length it opens with a `ZarrUserWarning`, which says that the axis holds only the fill value and how to store valid metadata: `array.update_attributes({})`, then `zarr.consolidate_metadata` if the metadata is consolidated. JSON `true`, as zarr-python 3.0 and 3.2 wrote for a chunk size of `True`, is read as 1, silently, in the chunk grid (regular, and the explicit edges and run-length encoded sizes of a rectilinear one) and in the inner chunk shape of every sharding codec, nested or not. A stored rectilinear chunk grid whose edge lengths are integral JSON floats (`[[4.0, 2]]`), as zarr-python 3.2 wrote for float edges, is read with those edges as integers, silently; a float anywhere no release wrote one (a regular chunk shape, the inner chunk shape of a sharding codec, a run-length repeat count, an edge below 1) is rejected. + +Writing data to an array read from such metadata (other than an empty selection) first stores the upgrade of the metadata the store then holds, unless another writer has stored valid metadata since, so that other readers find the chunks written; in a `ZipStore` this adds a second entry for the metadata document, as every metadata update does. If the metadata the store then holds lays out chunks differently (another writer resized the array keeping the chunk size of 0, say), the write raises a `ValueError` asking to reopen the array, and stores nothing. Storing a group's metadata, which `zarr.consolidate_metadata` and every change to a group with consolidated metadata do, likewise first stores the upgrade of each such member's metadata and consolidates the member's metadata as the store then holds it. diff --git a/src/zarr/api/asynchronous.py b/src/zarr/api/asynchronous.py index 4832bafaba..1fc10cdd1e 100644 --- a/src/zarr/api/asynchronous.py +++ b/src/zarr/api/asynchronous.py @@ -231,15 +231,10 @@ async def consolidate_metadata( group = await AsyncGroup.open(store_path, zarr_format=zarr_format, use_consolidated=False) group.store_path.store._check_writable() - members = { - k: v async for k, v in group.members(max_depth=None, use_consolidated_for_children=False) + members_metadata = { + k: v.metadata + async for k, v in group.members(max_depth=None, use_consolidated_for_children=False) } - # Store the upgrade of every member document that had to be upgraded before the - # consolidated document that describes the upgraded metadata. - for member in members.values(): - if isinstance(member, AsyncArray): - await member._store_upgraded_document() - members_metadata = {k: v.metadata for k, v in members.items()} # While consolidating, we want to be explicit about when child groups # are empty by inserting an empty dict for consolidated_metadata.metadata for k, v in members_metadata.items(): diff --git a/src/zarr/core/_json.py b/src/zarr/core/_json.py index efe8152a4f..4fd5ffab95 100644 --- a/src/zarr/core/_json.py +++ b/src/zarr/core/_json.py @@ -41,6 +41,12 @@ def buffer_to_json(buffer: Buffer) -> JSON: return cast("JSON", json.loads(buffer.to_bytes())) +def json_equal(a: object, b: object) -> bool: + """Whether two JSON values have the same JSON encoding. Python compares `True` and + `1`, or `1.0` and `1`, as equal; JSON does not.""" + return json.dumps(a) == json.dumps(b) + + def buffer_to_json_object(buffer: Buffer) -> dict[str, JSON]: """Parse the contents of a `Buffer` as a JSON object (a `dict`). diff --git a/src/zarr/core/array.py b/src/zarr/core/array.py index da5b0595a8..a1d6a73033 100644 --- a/src/zarr/core/array.py +++ b/src/zarr/core/array.py @@ -118,8 +118,8 @@ ArrayV2MetadataDict, ArrayV3Metadata, ) -from zarr.core.metadata.io import save_metadata, upsert_metadata -from zarr.core.metadata.upgrades import upgrade_array_document +from zarr.core.metadata.io import read_stored_array, save_metadata, upsert_metadata +from zarr.core.metadata.upgrades import mark_upgraded, upgrade_array_document from zarr.core.metadata.v2 import ( CompressorLikev2, get_object_codec_id, @@ -202,10 +202,30 @@ def _chunk_sizes_from_shape( return tuple(result) +def _as_json(value: Any) -> Any: + """`value`, as `to_dict` returns it, with its tuples as JSON arrays.""" + match value: + case tuple() | list(): + return [_as_json(item) for item in value] + case dict(): + return {key: _as_json(item) for key, item in value.items()} + return value + + def parse_array_metadata(data: Any, path: str | None = None) -> ArrayMetadata: + """Array metadata from a metadata object or a metadata document, naming the array at + `path` in warnings about how an invalid document was read. + + The metadata constructors accept chunk sizes that only an invalid document holds + (such as 0), as they always have; such metadata is read as its document is (see + `zarr.core.metadata.upgrades`), so an array can be built from it. No data was read + or written under those chunk sizes, so none of its readings is a warning.""" if isinstance(data, ArrayMetadata): - return data - elif isinstance(data, dict): + document, readings = upgrade_array_document(_as_json(data.to_dict()), data.zarr_format) + if not readings: + return data + return mark_upgraded(parse_array_metadata(dict(document)), [None for _ in readings], path) + if isinstance(data, dict): zarr_format = data.get("zarr_format") if zarr_format == 3: meta_out = ArrayV3Metadata.from_dict(data, path=path) @@ -1617,22 +1637,31 @@ async def _save_metadata(self, metadata: ArrayMetadata, ensure_parents: bool = F await save_metadata(self.store_path, metadata, ensure_parents=ensure_parents) async def _store_upgraded_document(self) -> None: - """Store the upgrade of this array's current stored document, if it needs one. + """Store the upgrade of this array's current stored document, if it needs one, + before chunks are written under this handle's metadata. Only for metadata read from a document that had to be upgraded (see `zarr.core.metadata.upgrades`). The document is read again, because the store may - hold a newer one than this handle's metadata: if that one needs no upgrade (the - array was re-saved or resized since, possibly by another implementation), it is - left as written. Storing the same upgrade twice is harmless, so concurrent - callers need no coordination. + hold a newer one than this handle's metadata. If that one lays out chunks + differently (the array was resized since by software that kept the invalid chunk + size), this handle would write chunks no reader finds, so it raises and stores + nothing. If it needs no upgrade (the array was re-saved since, possibly by another + implementation), it is left as written; if there is none, there is nothing to + upgrade. Storing the same upgrade twice is harmless, so concurrent callers need + no coordination. """ if not self.metadata._stored_document_upgraded: return - zarr_format = self.metadata.zarr_format - stored = await get_array_metadata(self.store_path, zarr_format=zarr_format) - upgraded, readings = upgrade_array_document(stored, zarr_format) - if readings: - await upsert_metadata(self.store_path, parse_array_metadata(upgraded)) + read = await read_stored_array(self.store_path, self.metadata.zarr_format) + if read is not None: + current, upgraded = read + if _chunk_layout(current) != _chunk_layout(self.metadata): + raise ValueError( + f"The metadata stored for the array at {str(self.store_path)!r} has " + "changed since this array was opened: reopen the array to write to it." + ) + if upgraded: + await upsert_metadata(self.store_path, current) object.__setattr__(self.metadata, "_stored_document_upgraded", False) async def _set_selection( @@ -4864,6 +4893,14 @@ async def create_array( ) +def _chunk_layout(metadata: ArrayMetadata) -> tuple[object, tuple[int, ...] | None]: + """How an array's chunks are laid out: its chunk grid and, if it is sharded, the + inner chunk shape.""" + grid = metadata.chunks if isinstance(metadata, ArrayV2Metadata) else metadata.chunk_grid + sharding = _sharding_codec(metadata) + return grid, None if sharding is None else sharding.chunk_shape + + def _sharding_codec(metadata: ArrayMetadata) -> ShardingCodec | None: """The array's sharding codec, or None if the array is not sharded. diff --git a/src/zarr/core/common.py b/src/zarr/core/common.py index dc85737289..7ec4e2d04b 100644 --- a/src/zarr/core/common.py +++ b/src/zarr/core/common.py @@ -2,7 +2,6 @@ import asyncio import math -import numbers import warnings from collections.abc import Iterable, Mapping, Sequence from enum import Enum @@ -283,22 +282,20 @@ def _subject(name: str, axis: int | None) -> str: def _parse_positive_int(value: object, name: str, axis: int | None) -> int: - """`value` as an `int` of at least 1. An integer of any integer type (an `int`, a - `bool` or a NumPy integer) is read as the `int` it equals; a float is rejected, even - an integral one (stored documents with integral floats are read by - `zarr.core.metadata.upgrades`).""" + """`value` as an `int` of at least 1. A `bool` is read as the `int` it equals; any + other type, a NumPy integer or a float (even an integral one: stored documents with + integral floats are read by `zarr.core.metadata.upgrades`), is rejected.""" subject = _subject(name, axis) - if not isinstance(value, numbers.Integral): - raise TypeError(f"{subject} must be an integer, got {value!r}") - parsed = int(value) - if parsed < 1: + if not isinstance(value, int): + raise TypeError(f"{subject} must be an int, got {value!r}") + if value < 1: raise ValueError(f"{subject} must be >= 1, got {value!r}") - return parsed + return int(value) def parse_chunk_edge(size: object, axis: int | None = None) -> int: - """Check that `size` is a chunk edge length: an integer of at least 1, read as an - `int`. + """Check that `size` is a chunk edge length: an `int` of at least 1 (a `bool` is + read as the `int` it equals). This is the one rule for chunk edge lengths in metadata: bare chunk sizes, explicit edges and run-length encoded sizes. `axis`, when given, is named in the error. @@ -307,9 +304,11 @@ def parse_chunk_edge(size: object, axis: int | None = None) -> int: def parse_chunk_shape(data: object) -> tuple[int, ...]: - """Check a regular chunk shape: an iterable of one chunk edge length per axis - (see `parse_chunk_edge`).""" + """Check a regular chunk shape: an iterable, other than a string or a mapping, of one + chunk edge length per axis (see `parse_chunk_edge`).""" match data: + case str() | Mapping(): + pass case Iterable(): return tuple(parse_chunk_edge(size, axis) for axis, size in enumerate(data)) raise TypeError(f"A chunk shape must be an iterable of chunk edge lengths, got {data!r}") diff --git a/src/zarr/core/group.py b/src/zarr/core/group.py index 38eede55ec..5a52e3bbf3 100644 --- a/src/zarr/core/group.py +++ b/src/zarr/core/group.py @@ -13,7 +13,6 @@ import zarr.api.asynchronous as async_api from zarr.abc.metadata import Metadata -from zarr.abc.store import Store, set_or_delete from zarr.core._info import GroupInfo from zarr.core._json import buffer_to_json_object, json_to_buffer from zarr.core.array import ( @@ -75,6 +74,7 @@ ) from typing import Any + from zarr.abc.store import Store from zarr.core.array_spec import ArrayConfigLike from zarr.core.buffer import Buffer, BufferPrototype from zarr.core.chunk_key_encodings import ChunkKeyEncodingLike @@ -2115,9 +2115,7 @@ async def update_attributes_async(self, new_attributes: dict[str, Any]) -> Group new_metadata = replace(self.metadata, attributes=new_attributes) # Write new metadata - to_save = new_metadata.to_buffer_dict(default_buffer_prototype()) - awaitables = [set_or_delete(self.store_path / key, value) for key, value in to_save.items()] - await asyncio.gather(*awaitables) + await save_metadata(self.store_path, new_metadata) async_group = replace(self._async_group, metadata=new_metadata) return replace(self, _async_group=async_group) diff --git a/src/zarr/core/metadata/io.py b/src/zarr/core/metadata/io.py index 7b5e0ced0d..1503736e04 100644 --- a/src/zarr/core/metadata/io.py +++ b/src/zarr/core/metadata/io.py @@ -1,16 +1,17 @@ from __future__ import annotations import asyncio -import json +from dataclasses import replace from enum import Enum from itertools import zip_longest from typing import TYPE_CHECKING, Final, NamedTuple from zarr.abc.store import set_or_delete -from zarr.core._json import buffer_to_json_object +from zarr.core._json import buffer_to_json_object, json_equal from zarr.core.buffer.core import default_buffer_prototype from zarr.core.buffer.cpu import buffer_prototype as cpu_buffer_prototype -from zarr.errors import ContainsArrayError +from zarr.core.metadata.upgrades import upgrade_array_document +from zarr.errors import ArrayNotFoundError, ContainsArrayError from zarr.storage._common import StorePath, ensure_no_existing_node if TYPE_CHECKING: @@ -65,7 +66,7 @@ def _diff( case list(), list(): for index, pair in enumerate(zip_longest(stored, new, fillvalue=ABSENT)): yield from _diff((*path, index), *pair) - case _ if ABSENT in (stored, new) or json.dumps(stored) != json.dumps(new): + case _ if ABSENT in (stored, new) or not json_equal(stored, new): yield DocumentChange(path, stored, new) @@ -103,6 +104,49 @@ async def upsert_metadata( return changes +async def read_stored_array( + store_path: StorePath, zarr_format: ZarrFormat +) -> tuple[ArrayMetadata, bool] | None: + """The metadata of the array document stored at `store_path` as it now is, read with + the upgrades but without their warnings (the handle that asks has warned), and + whether the document had to be upgraded; `None` if no array document is stored + there.""" + from zarr.core.array import get_array_metadata, parse_array_metadata + + try: + stored = await get_array_metadata(store_path, zarr_format=zarr_format) + except ArrayNotFoundError: + return None + upgraded, readings = upgrade_array_document(stored, zarr_format) + return parse_array_metadata(upgraded, str(store_path)), bool(readings) + + +async def _refresh_consolidated(store_path: StorePath, metadata: GroupMetadata) -> GroupMetadata: + """`metadata` with each member of its consolidated metadata that was read from a + document that had to be upgraded replaced by the metadata of the member's own + document as it now is, after storing that document's upgrade if it needs one: the + member document may have changed since, so the consolidated copy is never stored + as if it were valid.""" + from zarr.core.group import GroupMetadata + + consolidated = metadata.consolidated_metadata + if consolidated is None: + return metadata + members = dict(consolidated.metadata) + for name, member in consolidated.metadata.items(): + if isinstance(member, GroupMetadata): + members[name] = await _refresh_consolidated(store_path / name, member) + elif member._stored_document_upgraded: + read = await read_stored_array(store_path / name, member.zarr_format) + if read is not None: + members[name], upgraded = read + if upgraded: + await upsert_metadata(store_path / name, members[name]) + if all(members[name] is member for name, member in consolidated.metadata.items()): + return metadata + return replace(metadata, consolidated_metadata=replace(consolidated, metadata=members)) + + def _build_parents(store_path: StorePath, zarr_format: ZarrFormat) -> dict[str, GroupMetadata]: from zarr.core.group import GroupMetadata @@ -140,6 +184,11 @@ async def save_metadata( ------ ValueError """ + from zarr.core.group import GroupMetadata + + if isinstance(metadata, GroupMetadata): + # The one place group metadata is stored, and with it consolidated metadata. + metadata = await _refresh_consolidated(store_path, metadata) to_save = metadata.to_buffer_dict(default_buffer_prototype()) set_awaitables = [store_documents(store_path, to_save)] diff --git a/src/zarr/core/metadata/upgrades.py b/src/zarr/core/metadata/upgrades.py index 4d978d36c9..52812e334c 100644 --- a/src/zarr/core/metadata/upgrades.py +++ b/src/zarr/core/metadata/upgrades.py @@ -1,13 +1,17 @@ """Upgrades that read invalid stored array metadata documents written by older software. -This is the only place invalid metadata is read leniently; the metadata constructors -never reinterpret a chunk size. An upgrade maps a stored array metadata document (parsed JSON) to a valid -one and says how it read the document. `ArrayV2Metadata.from_dict` and +This is the only place invalid metadata is read leniently. An upgrade maps a stored +array metadata document (parsed JSON) to a valid one. `ArrayV2Metadata.from_dict` and `ArrayV3Metadata.from_dict` apply the upgrades for their Zarr format, so every path -that parses a stored document, including consolidated metadata, goes through them, and -warn once, with every reading, after the upgraded document has passed the metadata -constructor. An invalid document therefore raises its own error, not a warning about -how it was read. +that parses a stored document, including consolidated metadata, goes through them. + +A reading warns only where the user must act on it; a reading that gives what zarr +read from the same document before these upgrades existed is silent, so a document +that opened without a warning still does. The warnings are given once per document, +after the upgraded document has passed the metadata constructor, so an invalid +document raises its own error, not a warning about how it was read. Silent or not, +metadata read from an upgraded document is marked (see `mark_upgraded`), so the array +stores the upgrade before it writes chunks under it. To read another kind of invalid document, add an upgrade to `ARRAY_UPGRADES`. """ @@ -18,8 +22,9 @@ import warnings from collections.abc import Callable, Iterable, Mapping, Sequence from itertools import chain, repeat -from typing import TYPE_CHECKING, Final, TypeGuard, cast +from typing import TYPE_CHECKING, Final, TypeGuard +from zarr.core._json import json_equal from zarr.core.chunk_grids import full_span_chunk_size from zarr.errors import ZarrUserWarning @@ -27,9 +32,9 @@ from zarr.core.common import JSON, ZarrFormat type ArrayDocument = Mapping[str, JSON] -type Upgrade = Callable[[ArrayDocument], tuple[ArrayDocument, str] | None] -"""Returns `None` if the document needs no upgrade, else the upgraded document and a -sentence saying how it was read.""" +type Upgrade = Callable[[ArrayDocument], tuple[ArrayDocument, str | None] | None] +"""Returns `None` if the document needs no upgrade, else the upgraded document and, +if the user must act on how it was read, a warning saying so (else `None`).""" RESAVE_HINT: Final = ( "To store valid metadata, open the array writable and call `array.update_attributes({})`; " @@ -38,17 +43,19 @@ ) -def mark_upgraded[M](metadata: M, readings: Sequence[str], path: str | None) -> M: +def mark_upgraded[M](metadata: M, readings: Sequence[str | None], path: str | None) -> M: """Record that `metadata` was read from a stored document that needed the upgrades - whose `readings` `upgrade_array_document` returned, if any: warn once, naming the - array at `path` when the caller knows it, and set `_stored_document_upgraded` on - `metadata`, so the array stores the upgrade before it writes chunks under it.""" + whose `readings` `upgrade_array_document` returned, if any: set + `_stored_document_upgraded` on `metadata`, so the array stores the upgrade before + it writes chunks under it, and warn once with the readings that are warnings, + naming the array at `path` when the caller knows it.""" if readings: + object.__setattr__(metadata, "_stored_document_upgraded", True) + if messages := [reading for reading in readings if reading is not None]: subject = "" if path is None else f"Array {path!r}: " # The synchronous API parses metadata on zarr's IO thread, whose stack holds no # user code, so the warning points at the `from_dict` that read the document. - warnings.warn(f"{subject}{' '.join(readings)} {RESAVE_HINT}", ZarrUserWarning, stacklevel=2) - object.__setattr__(metadata, "_stored_document_upgraded", True) + warnings.warn(f"{subject}{' '.join(messages)} {RESAVE_HINT}", ZarrUserWarning, stacklevel=2) return metadata @@ -57,47 +64,42 @@ def _is_int_list(value: object) -> TypeGuard[list[int]]: return isinstance(value, list) and all(isinstance(v, int) for v in value) -def _abbreviate(value: JSON, limit: int = 60) -> str: - """`value` as JSON, cut to at most `limit` characters.""" - text = json.dumps(value) - return text if len(text) <= limit else f"{text[: limit - 3]}..." - - def _read_chunk_size(size: JSON, span: int | None, unit: int) -> tuple[int, str | None] | None: """Read one entry of a stored regular chunk shape as a chunk edge length. - Returns the edge length and, for an invalid entry, how it was read; `None` if the - entry cannot be read, which leaves it for the metadata constructors to reject. A - JSON int >= 1 is kept, JSON `true` is read as 1, and 0 or JSON `false` is read as - one chunk spanning the axis of length `span`, a multiple of `unit` (the inner chunk - size of a shard). `span` is `None` where no stored 0 is known, as in the inner - chunk shape of a sharding codec: 0 is then left for the constructors to reject. + Returns the edge length and, where the user must act on how it was read, how it was + read; `None` if the entry cannot be read, which leaves it for the metadata + constructors to check. A JSON int >= 1 is kept, JSON `true` is read as 1, and 0 or + JSON `false` is read as one chunk spanning the axis of length `span`, a multiple of + `unit` (the inner chunk size of a shard): on an axis of positive length no chunk + can have been stored under it, so the array holds only its fill value. `span` is + `None` where no stored 0 is known, as in the inner chunk shape of a sharding codec: + 0 is then left as stored. """ match size: case True: - return 1, "1" + return 1, None case int() if size >= 1: return size, None case int() if size == 0 and span is not None: edge = full_span_chunk_size(span, unit) - how = f"one chunk spanning the dimension ({edge})" - if span > 0: - how += ( - ", and as no chunk can be stored under a chunk size of 0, the array " - "holds only its fill value" - ) - return edge, how + if span == 0: + return edge, None + return edge, ( + f"one chunk spanning the dimension ({edge}), and as no chunk can be stored " + "under a chunk size of 0, the array holds only its fill value" + ) return None def _read_chunk_shape( - stored: JSON, spans: Sequence[int | None], units: Iterable[int], name: str + stored: JSON, spans: Sequence[int | None], units: Iterable[int] = () ) -> tuple[list[int], str | None] | None: """Read a stored regular chunk shape, entry by entry (see `_read_chunk_size`), for axes of lengths `spans` whose chunks are multiples of `units` (1 where not given). - Returns the chunk shape and, if an entry is invalid, a sentence saying how the - `name` was read; `None` if it cannot be read. + Returns the chunk shape and, where the user must act on how an entry was read, a + sentence saying how the chunk shape was read; `None` if it cannot be read. """ if not (isinstance(stored, list) and len(stored) == len(spans)): return None @@ -115,61 +117,68 @@ def _read_chunk_shape( if not readings: return edges, None return edges, ( - f"The stored {name} {json.dumps(list(stored))} is invalid: chunk sizes must be " + f"The stored chunk shape {json.dumps(stored)} is invalid: chunk sizes must be " f"integers of at least 1. It is read as {edges}, reading {'; '.join(readings)}." ) -def _invalid_chunk_sizes_v2(doc: ArrayDocument) -> tuple[ArrayDocument, str] | None: +def _invalid_chunk_sizes_v2(doc: ArrayDocument) -> tuple[ArrayDocument, str | None] | None: shape = doc.get("shape") if not _is_int_list(shape): return None - match _read_chunk_shape(doc.get("chunks"), shape, (), "chunk shape"): - case chunks, str(reading): + stored = doc.get("chunks") + match _read_chunk_shape(stored, shape): + case chunks, reading if not json_equal(chunks, stored): return {**doc, "chunks": chunks}, reading return None -def _sharding_codec(doc: ArrayDocument) -> tuple[Sequence[JSON], int, Mapping[str, JSON]] | None: - """The codec list of a Zarr format 3 array document, with the position and the - configuration of its sharding codec, if it has one.""" - codecs = doc.get("codecs") - if isinstance(codecs, list): - for index, codec in enumerate(codecs): - if isinstance(codec, Mapping) and codec.get("name") == "sharding_indexed": - configuration = codec.get("configuration") - if isinstance(configuration, Mapping): - return codecs, index, configuration - return None - - -def _read_inner_chunk_shape(doc: ArrayDocument) -> tuple[list[int], str | None] | None: - """Read the inner chunk shape of a sharded array. No stored inner chunk size of 0 - or `false` is known, so the spans of its axes are not given.""" - shape = doc.get("shape") - sharding = _sharding_codec(doc) - if sharding is None or not _is_int_list(shape): +def _read_codec(codec: JSON) -> JSON: + """Read a stored codec: the inner chunk shape of a sharding codec is read as a chunk + shape with no known axis lengths (see `_read_chunk_shape`), and so are those of the + sharding codecs nested in its codecs.""" + if not (isinstance(codec, Mapping) and codec.get("name") == "sharding_indexed"): + return codec + configuration = codec.get("configuration") + if not isinstance(configuration, Mapping): + return codec + upgraded = dict(configuration) + stored = configuration.get("chunk_shape") + if isinstance(stored, list): + match _read_chunk_shape(stored, [None] * len(stored)): + case chunk_shape, _: + upgraded["chunk_shape"] = list(chunk_shape) + if isinstance(codecs := configuration.get("codecs"), list): + upgraded["codecs"] = [_read_codec(inner) for inner in codecs] + return {**codec, "configuration": upgraded} + + +def _invalid_inner_chunk_sizes_v3(doc: ArrayDocument) -> tuple[ArrayDocument, str | None] | None: + stored = doc.get("codecs") + if not isinstance(stored, list): return None - _, _, configuration = sharding - return _read_chunk_shape( - configuration.get("chunk_shape"), - [None] * len(shape), - (), - "inner chunk shape of the sharding codec", - ) + codecs = [_read_codec(codec) for codec in stored] + if json_equal(codecs, stored): + return None + return {**doc, "codecs": codecs}, None -def _invalid_inner_chunk_sizes_v3(doc: ArrayDocument) -> tuple[ArrayDocument, str] | None: - match _read_inner_chunk_shape(doc), _sharding_codec(doc): - case (inner, str(reading)), (codecs, index, configuration): - # `_sharding_codec` found a mapping at `index`. - codec = cast("Mapping[str, JSON]", codecs[index]) - upgraded = {**codec, "configuration": {**configuration, "chunk_shape": inner}} - return {**doc, "codecs": [*codecs[:index], upgraded, *codecs[index + 1 :]]}, reading - return None +def _inner_chunk_shape(doc: ArrayDocument) -> list[int]: + """The inner chunk shape of the sharding codec of a Zarr format 3 array document, if + it has one and that is a list of integers; else `[]`.""" + codecs = doc.get("codecs") + if isinstance(codecs, list): + for codec in codecs: + match codec: + case { + "name": "sharding_indexed", + "configuration": {"chunk_shape": list() as inner}, + }: + return inner if _is_int_list(inner) else [] + return [] -def _invalid_chunk_sizes_v3(doc: ArrayDocument) -> tuple[ArrayDocument, str] | None: +def _invalid_chunk_sizes_v3(doc: ArrayDocument) -> tuple[ArrayDocument, str | None] | None: grid = doc.get("chunk_grid") shape = doc.get("shape") if not (isinstance(grid, Mapping) and grid.get("name") == "regular" and _is_int_list(shape)): @@ -177,20 +186,21 @@ def _invalid_chunk_sizes_v3(doc: ArrayDocument) -> tuple[ArrayDocument, str] | N configuration = grid.get("configuration") if not isinstance(configuration, Mapping): return None - inner = _read_inner_chunk_shape(doc) - units = () if inner is None else inner[0] - match _read_chunk_shape(configuration.get("chunk_shape"), shape, units, "chunk shape"): - case chunk_shape, str(reading): + stored = configuration.get("chunk_shape") + match _read_chunk_shape(stored, shape, _inner_chunk_shape(doc)): + case chunk_shape, reading if not json_equal(chunk_shape, stored): upgraded = {**configuration, "chunk_shape": chunk_shape} return {**doc, "chunk_grid": {**grid, "configuration": upgraded}}, reading return None def _read_edge_length(edge: JSON) -> JSON: - """Read one stored chunk edge length of a rectilinear chunk grid: an integral JSON - float of at least 1 (`4.0`) is read as the `int` it equals. Anything else is kept, - for the metadata constructors to check.""" + """Read one stored chunk edge length of a rectilinear chunk grid: JSON `true` is + read as 1 and an integral JSON float of at least 1 (`4.0`) as the `int` it equals. + Anything else is kept, for the metadata constructors to check.""" match edge: + case True: + return 1 case float() if edge.is_integer() and edge >= 1: return int(edge) return edge @@ -200,7 +210,7 @@ def _read_rectilinear_axis(spec: JSON) -> JSON: """Read the stored chunk edge lengths of one axis of a rectilinear chunk grid (see `_read_edge_length`): its explicit edges and the sizes of its run-length encoded `[size, count]` pairs. A bare chunk size and a repeat count are kept as stored: no - writer stored them as floats.""" + writer stored them as `true` or as floats.""" match spec: case list(): return [_read_rectilinear_entry(entry) for entry in spec] @@ -214,7 +224,7 @@ def _read_rectilinear_entry(entry: JSON) -> JSON: return _read_edge_length(entry) -def _float_edge_lengths_v3(doc: ArrayDocument) -> tuple[ArrayDocument, str] | None: +def _invalid_edge_lengths_v3(doc: ArrayDocument) -> tuple[ArrayDocument, str | None] | None: grid = doc.get("chunk_grid") if not (isinstance(grid, Mapping) and grid.get("name") == "rectilinear"): return None @@ -225,35 +235,32 @@ def _float_edge_lengths_v3(doc: ArrayDocument) -> tuple[ArrayDocument, str] | No if not isinstance(stored, list): return None read = [_read_rectilinear_axis(axis) for axis in stored] - # An integral float and the `int` it equals compare equal, but not as JSON text. - axes = [axis for axis, spec in enumerate(stored) if json.dumps(spec) != json.dumps(read[axis])] - if not axes: + if json_equal(read, stored): return None - reading = ( - f"The stored chunk edge lengths {_abbreviate(stored)} are invalid: chunk edge " - f"lengths must be integers. They are read as {_abbreviate(read)}, reading each " - f"edge length stored as a float, in dimensions {axes}, as the integer it equals." - ) upgraded = {**configuration, "chunk_shapes": read} - return {**doc, "chunk_grid": {**grid, "configuration": upgraded}}, reading + return {**doc, "chunk_grid": {**grid, "configuration": upgraded}}, None ARRAY_UPGRADES: Final[Mapping[ZarrFormat, tuple[Upgrade, ...]]] = { 2: (_invalid_chunk_sizes_v2,), - 3: (_invalid_inner_chunk_sizes_v3, _invalid_chunk_sizes_v3, _float_edge_lengths_v3), + # The inner chunk shape is read first: it gives the unit of the outer chunk shape. + # The rectilinear edge lengths are read last, after any upgrade that yields a + # rectilinear chunk grid. + 3: (_invalid_inner_chunk_sizes_v3, _invalid_chunk_sizes_v3, _invalid_edge_lengths_v3), } """The upgrades of an array document of each Zarr format, applied in order.""" def upgrade_array_document( doc: ArrayDocument, zarr_format: ZarrFormat -) -> tuple[ArrayDocument, list[str]]: +) -> tuple[ArrayDocument, list[str | None]]: """Apply the upgrades for `zarr_format` to a stored array metadata document. - Returns the upgraded document and the readings of the upgrades that changed it, - for `mark_upgraded` once the document has been validated. + Returns the upgraded document and the reading of each upgrade that changed it (a + warning, or `None` for a silent one), for `mark_upgraded` once the document has + been validated. """ - readings: list[str] = [] + readings: list[str | None] = [] for upgrade in ARRAY_UPGRADES[zarr_format]: upgraded = upgrade(doc) if upgraded is not None: diff --git a/src/zarr/core/metadata/v2.py b/src/zarr/core/metadata/v2.py index 4adae2834e..7df14a513f 100644 --- a/src/zarr/core/metadata/v2.py +++ b/src/zarr/core/metadata/v2.py @@ -154,7 +154,8 @@ def to_buffer_dict(self, prototype: BufferPrototype) -> dict[str, Buffer]: @classmethod def from_dict(cls, data: dict[str, Any], *, path: str | None = None) -> ArrayV2Metadata: """Read a stored `.zarray` document (with its attributes). An invalid document - that `zarr.core.metadata.upgrades` can read warns, naming the array at `path`.""" + that `zarr.core.metadata.upgrades` can read is read as upgraded; a reading the user + must act on warns, naming the array at `path`.""" upgraded, readings = upgrade_array_document(data, 2) # a new dict, because we are modifying it _data: dict[str, Any] = dict(upgraded) diff --git a/src/zarr/core/metadata/v3.py b/src/zarr/core/metadata/v3.py index 7231d9c51c..ad16afc3e1 100644 --- a/src/zarr/core/metadata/v3.py +++ b/src/zarr/core/metadata/v3.py @@ -614,7 +614,8 @@ def to_buffer_dict(self, prototype: BufferPrototype) -> dict[str, Buffer]: @classmethod def from_dict(cls, data: dict[str, JSON], *, path: str | None = None) -> Self: """Read a stored `zarr.json` array document. An invalid document that - `zarr.core.metadata.upgrades` can read warns, naming the array at `path`.""" + `zarr.core.metadata.upgrades` can read is read as upgraded; a reading the user + must act on warns, naming the array at `path`.""" upgraded, readings = upgrade_array_document(data, 3) # a new dict, because we are modifying it _data = dict(upgraded) diff --git a/tests/test_metadata/test_upgrades.py b/tests/test_metadata/test_upgrades.py index f2da059b16..89dd1d236c 100644 --- a/tests/test_metadata/test_upgrades.py +++ b/tests/test_metadata/test_upgrades.py @@ -23,6 +23,7 @@ from zarr.core.sync import sync from zarr.dtype import Int16 from zarr.errors import ZarrUserWarning +from zarr.storage import MemoryStore, StorePath from zarr.storage._common import make_store_path if TYPE_CHECKING: @@ -83,66 +84,73 @@ def _stored_chunks(doc: dict[str, Any]) -> Any: ) -def _chunk_shapes(metadata: ArrayV2Metadata | ArrayV3Metadata) -> tuple[Any, Any]: - """The chunk shape of `metadata` and, if it is sharded, its inner chunk shape.""" +def _chunk_shapes(metadata: ArrayV2Metadata | ArrayV3Metadata) -> tuple[Any, ...]: + """The chunk shape of `metadata`, then the inner chunk shape of each sharding codec, + from the outermost in.""" if isinstance(metadata, ArrayV2Metadata): - return metadata.chunks, None + return (metadata.chunks,) assert isinstance(metadata.chunk_grid, RegularChunkGridMetadata) - inner = next((c.chunk_shape for c in metadata.codecs if isinstance(c, ShardingCodec)), None) - return metadata.chunk_grid.chunk_shape, inner + shapes: list[Any] = [metadata.chunk_grid.chunk_shape] + codecs: tuple[Any, ...] = metadata.codecs + while sharding := next((c for c in codecs if isinstance(c, ShardingCodec)), None): + shapes.append(sharding.chunk_shape) + codecs = sharding.codecs + return tuple(shapes) + + +def _nested_sharded_doc(inner: list[Any], nested: list[Any]) -> dict[str, JSON]: + """A Zarr format 3 document of shape `[8]` in one shard of chunk shape `inner`, whose + codecs shard each chunk again, in chunks of shape `nested`.""" + doc = _v3_doc([8], [8], inner=inner) + outer = cast("dict[str, Any]", cast("list[JSON]", doc["codecs"])[0]) + configuration = outer["configuration"] + configuration["codecs"] = [{**outer, "configuration": {**configuration, "chunk_shape": nested}}] + return doc @pytest.mark.parametrize( - ("doc", "expected", "warning"), + ("doc", "expected", "upgraded", "warning"), [ - (_v2_doc([10, 10], [4, 5]), ((4, 5), None), None), - (_v3_doc([0, 0], [1, 1]), ((1, 1), None), None), - (_v3_doc([10], [4], inner=[2]), ((4,), (2,)), None), - ( - _v2_doc([0, 4], [0, 4]), - ((1, 4), None), - r"0 in dimension 0 as one chunk spanning the dimension \(1\)\.", - ), - ( - _v3_doc([0], [False]), - ((1,), None), - r"false in dimension 0 as one chunk spanning the dimension \(1\)\.", - ), - (_v2_doc([5], [True]), ((1,), None), "true in dimension 0 as 1"), - ( - _v3_doc([5, 4], [True, 4]), - ((1, 4), None), - r"\[true, 4\] is invalid.*true in dimension 0 as 1", - ), + (_v2_doc([10, 10], [4, 5]), ((4, 5),), False, None), + (_v3_doc([0, 0], [1, 1]), ((1, 1),), False, None), + (_v3_doc([10], [4], inner=[2]), ((4,), (2,)), False, None), + (_v2_doc([0, 4], [0, 4]), ((1, 4),), True, None), + (_v3_doc([0], [False]), ((1,),), True, None), + (_v2_doc([5], [True]), ((1,),), True, None), + (_v3_doc([5, 4], [True, 4]), ((1, 4),), True, None), ( _v2_doc([3], [0]), - ((3,), None), - r"spanning the dimension \(3\), and .* holds only its fill value", + ((3,),), + True, + ( + r"^The stored chunk shape \[0\] is invalid: .* read as \[3\], reading 0 in " + r"dimension 0 as one chunk spanning the dimension \(3\), and .* holds only " + r"its fill value\.$" + ), ), ( _v3_doc([4, 3], [4, 0]), - ((4, 3), None), - "0 in dimension 1 as .* holds only its fill value", + ((4, 3),), + True, + r"reading 0 in dimension 1 as .* holds only its fill value\.$", ), - (_v3_doc([0], [0], inner=[4]), ((4,), (4,)), r"spanning the dimension \(4\)\."), + ( + _v2_doc([0, 3], [0, 0]), + ((1, 3),), + True, + r"read as \[1, 3\], reading 0 in dimension 1 as .* holds only its fill value\.$", + ), + (_v3_doc([0], [0], inner=[4]), ((4,), (4,)), True, None), ( _v3_doc([10], [0], inner=[4]), ((12,), (4,)), + True, r"spanning the dimension \(12\), and .* holds only its fill value", ), - ( - _v3_doc([0, 3], [0, 3], inner=[2, 3]), - ((2, 3), (2, 3)), - r"spanning the dimension \(2\)\.", - ), - ( - _v3_doc([5], [True], inner=[True]), - ((1,), (1,)), - ( - r"^The stored inner chunk shape of the sharding codec \[true\] is invalid: .* " - r"The stored chunk shape \[true\] is invalid: " - ), - ), + (_v3_doc([0, 3], [0, 3], inner=[2, 3]), ((2, 3), (2, 3)), True, None), + (_v3_doc([5], [True], inner=[True]), ((1,), (1,)), True, None), + (_nested_sharded_doc([4], [2]), ((8,), (4,), (2,)), False, None), + (_nested_sharded_doc([4], [True]), ((8,), (4,), (1,)), True, None), ], ids=[ "v2-valid", @@ -154,42 +162,48 @@ def _chunk_shapes(metadata: ArrayV2Metadata | ArrayV3Metadata) -> tuple[Any, Any "v3-true", "v2-zero-grown-axis", "v3-zero-grown-axis", + "v2-zero-empty-and-grown-axes", "v3-sharded-zero-empty-axis", "v3-sharded-zero-grown-axis", "v3-sharded-zero-2d", "v3-sharded-true-inner-and-outer", + "v3-nested-sharded-valid", + "v3-nested-sharded-true", ], ) def test_upgrade_array_document( - doc: dict[str, JSON], expected: tuple[Any, Any], warning: str | None + doc: dict[str, JSON], expected: tuple[Any, ...], upgraded: bool, warning: str | None ) -> None: - """Valid documents pass unchanged and silently. A stored chunk size of 0 or `false` - is read as one chunk spanning the axis (a multiple of the inner chunk when sharded) - and `true` as 1, in the chunk shape and in a sharding codec's inner chunk shape; - `from_dict` warns once for the document, naming the array, saying how each part was - read (and that the array holds only its fill value where a chunk size of 0 was - stored for a non-empty axis) and how to re-save.""" - upgraded, readings = upgrade_array_document(doc, cast("ZarrFormat", doc["zarr_format"])) - assert {k: v for k, v in upgraded.items() if k not in ("chunks", "chunk_grid", "codecs")} == { - k: v for k, v in doc.items() if k not in ("chunks", "chunk_grid", "codecs") - } + """Valid documents pass unchanged. A stored chunk size of 0 or `false` is read as one + chunk spanning the axis (a multiple of the inner chunk when sharded) and `true` as 1, + in the chunk shape and in the inner chunk shape of every sharding codec, nested or + not. `from_dict` marks the metadata of an upgraded document; it warns once, naming + the array, only where a chunk size of 0 was stored for a non-empty axis (which then + holds only its fill value), saying how that part was read and how to re-save. The + other readings give what zarr read before, so they are silent.""" + upgraded_doc, readings = upgrade_array_document(doc, cast("ZarrFormat", doc["zarr_format"])) + assert { + k: v for k, v in upgraded_doc.items() if k not in ("chunks", "chunk_grid", "codecs") + } == {k: v for k, v in doc.items() if k not in ("chunks", "chunk_grid", "codecs")} + assert bool(readings) is upgraded + if not upgraded: + assert upgraded_doc is doc metadata_cls = ArrayV2Metadata if doc["zarr_format"] == 2 else ArrayV3Metadata with warnings.catch_warnings(record=True) as record: warnings.simplefilter("always") metadata = metadata_cls.from_dict(dict(doc), path="group/array") assert _chunk_shapes(metadata) == expected + assert metadata._stored_document_upgraded is upgraded messages = [str(w.message) for w in record] if warning is None: - assert upgraded is doc - assert readings == [] assert messages == [] else: - assert len(messages) == 1 - message = messages[0] + [message] = messages assert message.startswith("Array 'group/array': ") - assert re.search(warning, message.removeprefix("Array 'group/array': ")) - assert ("holds only its fill value" in message) == ("fill value" in warning) assert message.endswith(RESAVE_HINT) + assert re.search( + warning, message.removeprefix("Array 'group/array': ").removesuffix(f" {RESAVE_HINT}") + ) @pytest.mark.parametrize( @@ -219,6 +233,15 @@ def _read_strictly(doc: dict[str, JSON]) -> ArrayV2Metadata | ArrayV3Metadata: return metadata_cls.from_dict(doc) +def _open_strictly(path: Path, mode: Literal["r", "a", "r+"] = "r") -> AnyArray: + """Open the array at `path`, failing on any warning that its document was upgraded.""" + with warnings.catch_warnings(): + warnings.simplefilter("error", ZarrUserWarning) + array = zarr.open_array(store=path, mode=mode) + assert isinstance(array, zarr.Array) + return array + + def test_stored_negative_chunk_size_rejected() -> None: """No known writer stored a negative chunk size: it is rejected, not upgraded.""" with pytest.raises(ValueError, match="^Expected all values to be non-negative"): @@ -232,14 +255,6 @@ def test_stored_chunk_shape_ndim_mismatch_rejected() -> None: _read_strictly(_v3_doc([4, 4], [0])) -def test_stored_fractional_chunk_size_rejected() -> None: - """A stored chunk size that is not an integral number is rejected, not upgraded.""" - with pytest.raises( - TypeError, match=r"^Dimension 0: chunk edge length must be an integer, got 4\.5$" - ): - _read_strictly(_v3_doc([4], [4.5])) - - def _rectilinear_doc(shape: list[int], chunk_shapes: list[Any]) -> dict[str, JSON]: return _v3_doc(shape, [1] * len(shape)) | { "chunk_grid": { @@ -250,56 +265,48 @@ def _rectilinear_doc(shape: list[int], chunk_shapes: list[Any]) -> dict[str, JSO @pytest.mark.parametrize( - ("chunk_shapes", "expected", "warning"), + ("chunk_shapes", "expected", "upgraded"), [ - ([[[4, 2]], [[5, 2]]], ((4, 4), (5, 5)), None), - ( - [[[4.0, 2]], [[5, 2]]], - ((4, 4), (5, 5)), - ( - r"^The stored chunk edge lengths \[\[\[4\.0, 2\]\], \[\[5, 2\]\]\] are " - r"invalid: .* read as \[\[\[4, 2\]\], \[\[5, 2\]\]\], .* in dimensions \[0\]," - ), - ), - ([[3.0, 5.0], [[5, 2]]], ((3, 5), (5, 5)), r"read as \[\[3, 5\], "), - ([[[4.0, 2], 2.0], [[5, 2]]], ((4, 4, 2), (5, 5)), r"read as \[\[\[4, 2\], 2\], "), - ([[[4.0, 2]], [10]], ((4, 4), (10,)), r"read as \[\[\[4, 2\]\], \[10\]\]"), - ([[[4.0, 2]], [5.0, 5]], ((4, 4), (5, 5)), r"in dimensions \[0, 1\],"), + ([[[4, 2]], [[5, 2]]], ((4, 4), (5, 5)), False), + ([[[4.0, 2]], [[5, 2]]], ((4, 4), (5, 5)), True), + ([[3.0, 5.0], [[5, 2]]], ((3, 5), (5, 5)), True), + ([[[4.0, 2], 2.0], [[5, 2]]], ((4, 4, 2), (5, 5)), True), + ([[[4.0, 2]], [10]], ((4, 4), (10,)), True), + ([[[4.0, 2]], [5.0, 5]], ((4, 4), (5, 5)), True), + ([[True, 4], [[5, 2]]], ((1, 4), (5, 5)), True), + ([[[True, 2], 3], [[5, 2]]], ((1, 1, 3), (5, 5)), True), + ], + ids=[ + "valid", + "float-rle-size", + "float-edges", + "float-rle-size-and-edge", + "float-sharded-outer", + "float-2d", + "true-edge", + "true-rle-size", ], - ids=["valid", "rle-size", "edges", "rle-size-and-edge", "sharded-outer", "2d"], ) -def test_read_float_edges_in_rectilinear_grid( - chunk_shapes: list[Any], - expected: tuple[tuple[int, ...], ...], - warning: str | None, +def test_read_invalid_edges_in_rectilinear_grid( + chunk_shapes: list[Any], expected: tuple[tuple[int, ...], ...], upgraded: bool ) -> None: """A stored rectilinear chunk grid whose explicit edges or run-length encoded sizes - are integral floats, as zarr-python wrote them when given float edges, is read with - those edges as `int`s; `from_dict` warns once, naming the array and the dimensions, - and how to re-save.""" + are integral floats or JSON `true`, as zarr-python wrote them when given float or + `True` edges, is read with those edges as the `int`s they equal. `from_dict` marks + the metadata as upgraded, silently: zarr read these edges so before.""" shape = [sum(edges) for edges in expected] doc = _rectilinear_doc(shape, chunk_shapes) - with ( - zarr.config.set({"array.rectilinear_chunks": True}), - warnings.catch_warnings(record=True) as record, - ): - warnings.simplefilter("always") - metadata = ArrayV3Metadata.from_dict(doc, path="group/array") + with zarr.config.set({"array.rectilinear_chunks": True}): + metadata = _read_strictly(doc) assert metadata.chunk_grid == RectilinearChunkGridMetadata(chunk_shapes=expected) - messages = [str(w.message) for w in record] - if warning is None: - assert messages == [] - else: - [message] = messages - assert message.startswith("Array 'group/array': ") - assert re.search(warning, message.removeprefix("Array 'group/array': ")) - assert message.endswith(RESAVE_HINT) + assert metadata._stored_document_upgraded is upgraded @pytest.mark.parametrize( ("doc", "error"), [ - (_v3_doc([20], [10.0]), "Dimension 0: chunk edge length must be an integer, got 10.0"), + (_v3_doc([20], [10.0]), "Dimension 0: chunk edge length must be an int, got 10.0"), + (_v3_doc([4], [4.5]), "Dimension 0: chunk edge length must be an int, got 4.5"), ( _v3_doc([8], [4], inner=[2.0]), "Expected an iterable of integers. Got [2.0] instead.", @@ -307,47 +314,64 @@ def test_read_float_edges_in_rectilinear_grid( (_v2_doc([20], [10.0]), "Expected an iterable of integers. Got [10.0] instead."), ( _rectilinear_doc([8], [[[4, 2.0]]]), - "Dimension 0: RLE repeat count must be an integer, got 2.0", + "Dimension 0: RLE repeat count must be an int, got 2.0", ), ( _rectilinear_doc([8], [4.0]), - "Dimension 0: chunk edge length must be an integer, got 4.0", + "Dimension 0: chunk edge length must be an int, got 4.0", + ), + ( + _rectilinear_doc([8], [[0.0, 8]]), + "Dimension 0: chunk edge length must be an int, got 0.0", ), ], - ids=["regular", "sharding-inner", "v2", "rle-count", "rectilinear-bare"], + ids=[ + "regular", + "regular-fractional", + "sharding-inner", + "v2", + "rle-count", + "rectilinear-bare", + "rectilinear-zero", + ], ) def test_stored_float_chunk_size_rejected(doc: dict[str, JSON], error: str) -> None: - """A float chunk size is read only where zarr-python stored one, as the edges of a - rectilinear chunk grid. Anywhere else it is rejected, as zarr 3.4.0 rejected a - stored regular chunk size of `10.0`.""" + """A float chunk size is read only where zarr-python stored one, as an edge of at + least 1 of a rectilinear chunk grid. Anywhere else it is rejected, as zarr 3.4.0 + rejected a stored regular chunk size of `10.0`.""" with zarr.config.set({"array.rectilinear_chunks": True}), pytest.raises(TypeError) as info: _read_strictly(doc) assert info.match(re.escape(error)) -def test_float_edges_round_trip(tmp_path: Path) -> None: - """A store whose rectilinear chunk grid holds the float edges zarr-python wrote opens - with a warning, reads its data and re-saves the edges as `int`s.""" - path = tmp_path / "float.zarr" +@pytest.mark.parametrize( + ("chunks", "stored", "resaved"), + [ + ([[4, 4], [5, 5]], [[[4.0, 2]], [[5, 2]]], "[[[4, 2]], [[5, 2]]]"), + ([[1, 3, 4], [5, 5]], [[True, 3, 4], [[5, 2]]], "[[1, 3, 4], [[5, 2]]]"), + ], + ids=["float", "true"], +) +def test_invalid_edges_round_trip( + tmp_path: Path, chunks: list[list[int]], stored: list[Any], resaved: str +) -> None: + """A store whose rectilinear chunk grid holds the float or `true` edges zarr-python + wrote opens silently, reads its data, and stores its edges as `int`s before the + first write.""" + path = tmp_path / "rectilinear.zarr" data = np.arange(80, dtype="int16").reshape(8, 10) with zarr.config.set({"array.rectilinear_chunks": True}): - zarr.create_array(path, shape=data.shape, chunks=[[4, 4], [5, 5]], dtype="int16")[...] = ( - data + zarr.create_array(path, shape=data.shape, chunks=chunks, dtype="int16")[...] = data + _rewrite_doc( + path, 3, lambda doc: doc["chunk_grid"]["configuration"].update(chunk_shapes=stored) ) - - def store_floats(doc: dict[str, Any]) -> None: - doc["chunk_grid"]["configuration"]["chunk_shapes"] = [[[4.0, 2]], [[5, 2]]] - - _rewrite_doc(path, 3, store_floats) - with pytest.warns(ZarrUserWarning, match=r"read as \[\[\[4, 2\]\], \[\[5, 2\]\]\]"): - arr = zarr.open_array(path, mode="a") + arr = _open_strictly(path, mode="a") np.testing.assert_array_equal(arr[...], data) - arr.update_attributes({}) - stored = json.loads((path / "zarr.json").read_text())["chunk_grid"]["configuration"] - assert json.dumps(stored["chunk_shapes"]) == "[[[4, 2]], [[5, 2]]]" - with warnings.catch_warnings(): - warnings.simplefilter("error", ZarrUserWarning) - np.testing.assert_array_equal(zarr.open_array(path)[...], data) + arr[0, 0] = -1 + written = json.loads((path / "zarr.json").read_text())["chunk_grid"]["configuration"] + assert json.dumps(written["chunk_shapes"]) == resaved + data[0, 0] = -1 + np.testing.assert_array_equal(_open_strictly(path)[...], data) def test_v2_constructor_rejects_chunks_of_wrong_length() -> None: @@ -394,12 +418,12 @@ def _second_edge(grid: RectilinearChunkGridMetadata) -> int: @pytest.mark.parametrize("site", CHUNK_EDGE_SITES) @pytest.mark.parametrize( ("size", "expected"), - [(4, 4), (True, 1), (np.int64(4), 4)], - ids=["int", "bool", "numpy-int"], + [(4, 4), (True, 1)], + ids=["int", "bool"], ) def test_metadata_reads_integer_chunk_edge(site: str, size: object, expected: int) -> None: - """Chunk grid metadata reads an integer of any integer type as the `int` chunk edge - length it equals.""" + """Chunk grid metadata reads an `int` or a `bool` as the `int` chunk edge length it + equals.""" edge = CHUNK_EDGE_SITES[site](size) assert type(edge) is int assert edge == expected @@ -408,15 +432,16 @@ def test_metadata_reads_integer_chunk_edge(site: str, size: object, expected: in @pytest.mark.parametrize("site", CHUNK_EDGE_SITES) @pytest.mark.parametrize( "size", - [4.0, np.float64(4.0), 4.5, float("inf"), "4", None], - ids=["float", "numpy-float", "fractional", "inf", "str", "none"], + [4.0, np.float64(4.0), 4.5, float("inf"), "4", None, np.int64(4)], + ids=["float", "numpy-float", "fractional", "inf", "str", "none", "numpy-int"], ) def test_metadata_rejects_non_integer_chunk_edge(site: str, size: object) -> None: - """A float is not a chunk edge length in metadata built in code, even an integral - one; stored documents with integral floats are read by the upgrades.""" + """A chunk edge length in metadata built in code is an `int`: a float is rejected, + even an integral one (stored documents with integral floats are read by the + upgrades), and so is a NumPy integer, as zarr 3.4.0 rejected one.""" with pytest.raises( TypeError, - match=re.escape(f"Dimension 0: chunk edge length must be an integer, got {size!r}"), + match=re.escape(f"Dimension 0: chunk edge length must be an int, got {size!r}"), ): CHUNK_EDGE_SITES[site](size) @@ -434,8 +459,10 @@ def test_metadata_rejects_chunk_edge_below_one(site: str, size: int) -> None: CHUNK_EDGE_SITES[site](size) -@pytest.mark.parametrize("chunk_shape", [4, np.int64(4), None]) -def test_regular_chunk_grid_rejects_chunk_shape_not_iterable(chunk_shape: Any) -> None: +@pytest.mark.parametrize("chunk_shape", [4, np.int64(4), None, "44", {"4": 4}]) +def test_regular_chunk_grid_rejects_chunk_shape_not_a_sequence(chunk_shape: Any) -> None: + """A regular chunk shape is an iterable of chunk edge lengths, but not a string or a + mapping, which is rejected as a whole, not entry by entry.""" with pytest.raises( TypeError, match=re.escape( @@ -509,11 +536,11 @@ def _rewrite_doc(path: Path, zarr_format: Literal[2, 3], edit: Any) -> None: @pytest.mark.parametrize( - ("zarr_format", "shape", "stored", "inner", "expected"), + ("zarr_format", "shape", "stored", "inner", "expected", "warns"), [ - (2, (0, 4), [0, 4], None, (1, 4)), - (3, (5,), [True], None, (1,)), - (3, (10,), [0], (4,), (12,)), + (2, (0, 4), [0, 4], None, (1, 4), False), + (3, (5,), [True], None, (1,), False), + (3, (10,), [0], (4,), (12,), True), ], ids=["v2-empty-2d", "v3-true", "v3-sharded-grown"], ) @@ -524,9 +551,11 @@ def test_legacy_chunk_size_round_trip( stored: list[Any], inner: tuple[int, ...] | None, expected: tuple[int, ...], + warns: bool, ) -> None: - """A store whose metadata holds a chunk size written by older software opens with a - warning, reads and appends under the upgraded grid, and re-saves valid metadata.""" + """A store whose metadata holds a chunk size written by older software opens (with a + warning where a non-empty axis was stored with chunk size 0), reads and appends under + the upgraded grid, and re-saves valid metadata.""" path = tmp_path / "legacy.zarr" arr = zarr.create_array( store=path, @@ -548,8 +577,13 @@ def store_legacy(doc: dict[str, Any]) -> None: _rewrite_doc(path, zarr_format, store_legacy) - with pytest.warns(ZarrUserWarning, match=r"^Array '.*legacy\.zarr': .* is read as"): + with warnings.catch_warnings(record=True) as record: + warnings.simplefilter("always", ZarrUserWarning) arr = zarr.open_array(store=path, mode="a") + assert [ + re.match(r"^Array '.*legacy\.zarr': .* is read as", str(w.message)) is not None + for w in record + ] == [True] * warns assert (arr.shards or arr.chunks) == expected np.testing.assert_array_equal(arr[...], data) @@ -635,15 +669,6 @@ def _legacy_array(path: Path, zarr_format: Literal[2, 3]) -> None: ) -def _open_strictly(path: Path) -> AnyArray: - """Open the array at `path`, failing on any warning that its document was upgraded.""" - with warnings.catch_warnings(): - warnings.simplefilter("error", ZarrUserWarning) - array = zarr.open_array(store=path, mode="r") - assert isinstance(array, zarr.Array) - return array - - @pytest.mark.parametrize("zarr_format", [2, 3]) def test_stale_handle_write_keeps_newer_metadata( tmp_path: Path, zarr_format: Literal[2, 3] @@ -696,6 +721,132 @@ def test_stale_handle_write_keeps_valid_document_as_written( np.testing.assert_array_equal(_open_strictly(path)[...], [9, 0, 0]) +@pytest.mark.parametrize("zarr_format", [2, 3]) +def test_stale_handle_write_after_chunk_grid_change_raises( + tmp_path: Path, zarr_format: Literal[2, 3] +) -> None: + """If the document the store holds when a handle read from an upgraded document + first writes chunks lays out chunks differently from the handle's metadata (here the + array was resized by software that kept the stored chunk size of 0, which now reads + as a larger chunk), the handle's chunks would not be found under it: the write + raises and stores nothing.""" + path = tmp_path / "legacy.zarr" + _legacy_array(path, zarr_format) + with pytest.warns(ZarrUserWarning, match="is read as"): + stale = zarr.open_array(store=path, mode="r+") + _rewrite_doc(path, zarr_format, lambda doc: doc.update(shape=[10])) + documents = {p.name: p.read_bytes() for p in path.iterdir()} + + with pytest.raises(ValueError, match="has changed since this array was opened: reopen"): + stale[0:3] = [7, 8, 9] + + assert stale.metadata._stored_document_upgraded + assert {p.name: p.read_bytes() for p in path.iterdir()} == documents + + +@pytest.mark.parametrize("zarr_format", [2, 3]) +def test_write_without_stored_document(zarr_format: Literal[2, 3]) -> None: + """An array read from an upgraded document that no store holds (as + `AsyncArray.from_dict` builds one) writes its chunks as any array does: there is no + stored document to upgrade.""" + store = MemoryStore() + doc = _v2_doc([3], [True]) if zarr_format == 2 else _v3_doc([3], [True]) + array = zarr.Array(AsyncArray.from_dict(StorePath(store), doc)) + upgraded = array.metadata._stored_document_upgraded + + array[:] = [1, 2, 3] + + assert (upgraded, array.metadata._stored_document_upgraded) == (True, False) + np.testing.assert_array_equal(array[:], [1, 2, 3]) + assert not [key for key in store._store_dict if key.endswith((".zarray", "zarr.json"))] + + +@pytest.mark.parametrize(("shape", "expected"), [((0,), (1,)), ((3,), (3,))]) +def test_array_from_metadata_with_chunk_size_zero(shape: tuple[int], expected: tuple[int]) -> None: + """`ArrayV2Metadata` accepts a chunk size of 0, as a stored document may hold it. An + array built from such metadata reads it as the upgrades read that document, silently + (no data was read or written under it): `create_hierarchy` stores the metadata as + given and yields such an array, which stores the upgrade before its first write.""" + metadata = ArrayV2Metadata(shape=shape, chunks=(0,), dtype=Int16(), fill_value=0, order="C") + store = MemoryStore() + with warnings.catch_warnings(): + warnings.simplefilter("error", ZarrUserWarning) + nodes = dict(zarr.create_hierarchy(store=store, nodes={"a": metadata})) + array = nodes["a"] + assert isinstance(array, zarr.Array) + assert array.chunks == expected + assert json.loads(store._store_dict["a/.zarray"].to_bytes())["chunks"] == [0] + + array[...] = 1 + + # Writing an empty selection stores no chunks, so it stores no metadata either. + resaved = list(expected) if array.size else [0] + assert json.loads(store._store_dict["a/.zarray"].to_bytes())["chunks"] == resaved + np.testing.assert_array_equal(zarr.open_array(store, path="a")[...], np.ones(shape)) + + +def _store_zero(doc: dict[str, Any]) -> None: + _stored_chunks(doc)[0] = 0 + + +def _rewrite_consolidated( + path: Path, zarr_format: Literal[2, 3], name: str, edit: Callable[[dict[str, Any]], None] +) -> None: + """Edit the document of the member `name` in the consolidated metadata at `path`.""" + if zarr_format == 2: + document = json.loads((path / ".zmetadata").read_text()) + edit(document["metadata"][f"{name}/.zarray"]) + (path / ".zmetadata").write_text(json.dumps(document)) + else: + _rewrite_doc(path, 3, lambda doc: edit(doc["consolidated_metadata"]["metadata"][name])) + + +def _consolidated_member(path: Path, zarr_format: Literal[2, 3], name: str) -> Any: + if zarr_format == 2: + return json.loads((path / ".zmetadata").read_text())["metadata"][f"{name}/.zarray"] + return json.loads((path / "zarr.json").read_text())["consolidated_metadata"]["metadata"][name] + + +@pytest.mark.filterwarnings("ignore:Consolidated metadata is currently not part:UserWarning") +@pytest.mark.parametrize("zarr_format", [2, 3]) +@pytest.mark.parametrize("operation", ["attrs", "update_attributes_async", "delete-member"]) +def test_group_write_refreshes_upgraded_consolidated_member( + tmp_path: Path, zarr_format: Literal[2, 3], operation: str +) -> None: + """Storing a group's metadata stores its consolidated metadata. Each member of it read + from a document that had to be upgraded is first read again from the member's own + document as it now is (here resized by software that kept the stored chunk size of + 0), whose upgrade is stored, so no group write stores a stale copy as if valid.""" + path = tmp_path / "group.zarr" + group = zarr.open_group(path, mode="w", zarr_format=zarr_format) + group.create_array("a", shape=(3,), chunks=(3,), dtype="int16") + group.create_array("b", shape=(1,), chunks=(1,), dtype="int16") + zarr.consolidate_metadata(path) + _rewrite_consolidated(path, zarr_format, "a", _store_zero) + + def resize_keeping_zero(doc: dict[str, Any]) -> None: + _store_zero(doc) + doc["shape"] = [10] + + _rewrite_doc(path / "a", zarr_format, resize_keeping_zero) + with pytest.warns(ZarrUserWarning, match="is read as"): + group = zarr.open_group(path, mode="r+", use_consolidated=True) + + if operation == "attrs": + group.attrs["x"] = 1 + elif operation == "update_attributes_async": + sync(group.update_attributes_async({"x": 1})) + else: + del group["b"] + + assert _stored_chunks(_consolidated_member(path, zarr_format, "a")) == [10] + reopened = zarr.open_group(path, mode="r", use_consolidated=True) + member = reopened["a"] + assert isinstance(member, zarr.Array) + assert (member.shape, member.chunks) == ((10,), (10,)) + assert _open_strictly(path / "a").chunks == (10,) + + @pytest.mark.parametrize("zarr_format", [2, 3]) def test_empty_write_stores_no_metadata(tmp_path: Path, zarr_format: Literal[2, 3]) -> None: """A write of an empty selection stores no chunks, so it stores no metadata either.""" @@ -751,10 +902,10 @@ def test_legacy_chunk_size_consolidated(tmp_path: Path, zarr_format: Literal[2, group = zarr.open_group(path, mode="w", zarr_format=zarr_format) names = ("a", "b") for name in names: - group.create_array(name, shape=(0,), chunks=(1,), dtype="int32") + group.create_array(name, shape=(2,), chunks=(2,), dtype="int32") zarr.consolidate_metadata(path) - # What zarr-python wrote for `chunks=(0,)` on an empty array, in both copies. + # What zarr-python wrote for `chunks=(0,)`, in both copies. for name in names: if zarr_format == 2: _rewrite_doc(path / name, 2, lambda doc: doc.update(chunks=[0])) @@ -786,7 +937,7 @@ def test_legacy_chunk_size_consolidated(tmp_path: Path, zarr_format: Literal[2, ] for array in arrays: assert isinstance(array, zarr.Array) - assert array.chunks == (1,) + assert array.chunks == (2,) array.update_attributes({}) zarr.consolidate_metadata(path) with warnings.catch_warnings(): @@ -797,4 +948,4 @@ def test_legacy_chunk_size_consolidated(tmp_path: Path, zarr_format: Literal[2, for name in names: array = reopened[name] assert isinstance(array, zarr.Array) - assert array.chunks == (1,) + assert array.chunks == (2,) diff --git a/tests/test_unified_chunk_grid.py b/tests/test_unified_chunk_grid.py index 8df53df08b..6a43cb4f3b 100644 --- a/tests/test_unified_chunk_grid.py +++ b/tests/test_unified_chunk_grid.py @@ -8,6 +8,7 @@ from __future__ import annotations +import re from typing import TYPE_CHECKING, Any import numpy as np @@ -494,7 +495,7 @@ def test_chunk_grid_iter() -> None: [ ([[10, 3]], [10, 10, 10]), ([[10, 2], [20, 1]], [10, 10, 20]), - ([[True, 2], [np.int64(3), np.int64(1)]], [1, 1, 3]), + ([[True, 2], [3, 1]], [1, 1, 3]), ], ) def test_rle_expand(compressed: list[Any], expected: list[int]) -> None: @@ -551,12 +552,14 @@ def test_rle_expand_rejects_invalid(rle_input: list[Any], match: str) -> None: @pytest.mark.parametrize( ("rle_input", "match"), [ - ([10.5], "Chunk edge length must be an integer, got 10.5"), - ([10.0], "Chunk edge length must be an integer, got 10.0"), - ([[10.0, 3]], "Chunk edge length must be an integer, got 10.0"), - (["10"], "Chunk edge length must be an integer, got '10'"), - ([[10, 3.5]], "RLE repeat count must be an integer, got 3.5"), - ([[10, 3.0]], "RLE repeat count must be an integer, got 3.0"), + ([10.5], "Chunk edge length must be an int, got 10.5"), + ([10.0], "Chunk edge length must be an int, got 10.0"), + ([[10.0, 3]], "Chunk edge length must be an int, got 10.0"), + (["10"], "Chunk edge length must be an int, got '10'"), + ([[10, 3.5]], "RLE repeat count must be an int, got 3.5"), + ([[10, 3.0]], "RLE repeat count must be an int, got 3.0"), + ([np.int64(10)], "Chunk edge length must be an int, got np.int64(10)"), + ([[10, np.int64(3)]], "RLE repeat count must be an int, got np.int64(3)"), ], ids=[ "fractional-edge", @@ -565,11 +568,13 @@ def test_rle_expand_rejects_invalid(rle_input: list[Any], match: str) -> None: "string-edge", "fractional-count", "float-count", + "numpy-int-edge", + "numpy-int-count", ], ) def test_rle_expand_rejects_non_int(rle_input: list[Any], match: str) -> None: - """expand_rle reads integers only, not floats.""" - with pytest.raises(TypeError, match=match): + """expand_rle reads `int`s (and `bool`s) only, not floats or NumPy integers.""" + with pytest.raises(TypeError, match=re.escape(match)): expand_rle(rle_input) @@ -577,7 +582,7 @@ def test_rle_expand_rejects_non_int(rle_input: list[Any], match: str) -> None: ("rle_input", "match"), [ ([0], "chunk edge length must be >= 1"), - ([10.5], "chunk edge length must be an integer"), + ([10.5], "chunk edge length must be an int,"), ([[5, 0]], "RLE repeat count must be >= 1"), ([[5, 2, 1]], r"RLE entries must be an integer or \[size, count\]"), ], From f78e1152614582c9e56fd52be45cd2f15c2396c3 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Sat, 26 Sep 2026 14:35:32 +0200 Subject: [PATCH 33/41] test: expect the lifecycle machine's upgrade warning only on non-empty axes Assisted-by: ClaudeCode:claude-opus-5-5 Co-Authored-By: Claude Opus 5.5 --- tests/test_array_stateful.py | 6 +++++- 1 file changed, 5 insertions(+), 1 deletion(-) diff --git a/tests/test_array_stateful.py b/tests/test_array_stateful.py index 6dc03768b3..109d65eec1 100644 --- a/tests/test_array_stateful.py +++ b/tests/test_array_stateful.py @@ -167,8 +167,12 @@ def _open(self) -> zarr.Array[Any]: with warnings.catch_warnings(record=True) as record: warnings.simplefilter("always", ZarrUserWarning) arr = zarr.open_array(self.store, path=self.path, mode="r+") + # A stored chunk size of 0 is read silently on an empty axis; on a non-empty one + # it warns that the axis holds only the fill value. Either way it is upgraded. warned = any(issubclass(w.category, ZarrUserWarning) for w in record) - assert warned is bool(self.legacy_axes), [str(w.message) for w in record] + must_warn = any(self.shape[axis] > 0 for axis in self.legacy_axes) + assert warned is must_warn, [str(w.message) for w in record] + assert arr.metadata._stored_document_upgraded is bool(self.legacy_axes) return arr # ----------------------------------------------------------------- model From bc306aa5f97890241e13fab483d33ac25fc04fc6 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Sat, 26 Sep 2026 16:42:48 +0200 Subject: [PATCH 34/41] fix(metadata): leave a sharded chunk size of 0 unread when the inner shape is invalid A stored outer chunk size of 0 of a sharded array is read in multiples of the inner chunk size. When that inner size is itself 0 or `false`, the unit is unknown: the upgrade now leaves the 0 for the constructor, which rejects it with the same `ValueError` zarr 3.4.0 raised, instead of a ZeroDivisionError. `_read_codec` now matches the sharding codec like `_inner_chunk_shape` does. Assisted-by: ClaudeCode:claude-opus-5-5 Co-Authored-By: Claude Opus 5.5 --- src/zarr/core/metadata/upgrades.py | 55 ++++++++++++++-------------- tests/test_metadata/test_upgrades.py | 11 ++++++ 2 files changed, 38 insertions(+), 28 deletions(-) diff --git a/src/zarr/core/metadata/upgrades.py b/src/zarr/core/metadata/upgrades.py index 52812e334c..3b09dab496 100644 --- a/src/zarr/core/metadata/upgrades.py +++ b/src/zarr/core/metadata/upgrades.py @@ -22,7 +22,7 @@ import warnings from collections.abc import Callable, Iterable, Mapping, Sequence from itertools import chain, repeat -from typing import TYPE_CHECKING, Final, TypeGuard +from typing import TYPE_CHECKING, Final, TypeGuard, cast from zarr.core._json import json_equal from zarr.core.chunk_grids import full_span_chunk_size @@ -137,20 +137,19 @@ def _read_codec(codec: JSON) -> JSON: """Read a stored codec: the inner chunk shape of a sharding codec is read as a chunk shape with no known axis lengths (see `_read_chunk_shape`), and so are those of the sharding codecs nested in its codecs.""" - if not (isinstance(codec, Mapping) and codec.get("name") == "sharding_indexed"): - return codec - configuration = codec.get("configuration") - if not isinstance(configuration, Mapping): - return codec - upgraded = dict(configuration) - stored = configuration.get("chunk_shape") - if isinstance(stored, list): - match _read_chunk_shape(stored, [None] * len(stored)): - case chunk_shape, _: - upgraded["chunk_shape"] = list(chunk_shape) - if isinstance(codecs := configuration.get("codecs"), list): - upgraded["codecs"] = [_read_codec(inner) for inner in codecs] - return {**codec, "configuration": upgraded} + match codec: + case {"name": "sharding_indexed", "configuration": Mapping() as configuration}: + upgraded = dict(configuration) + stored = configuration.get("chunk_shape") + if isinstance(stored, list) and ( + read := _read_chunk_shape(stored, [None] * len(stored)) + ): + upgraded["chunk_shape"] = read[0] + if isinstance(codecs := configuration.get("codecs"), list): + upgraded["codecs"] = [_read_codec(inner) for inner in codecs] + # The mapping pattern does not narrow `codec` for mypy. + return {**cast("Mapping[str, JSON]", codec), "configuration": upgraded} + return codec def _invalid_inner_chunk_sizes_v3(doc: ArrayDocument) -> tuple[ArrayDocument, str | None] | None: @@ -163,18 +162,16 @@ def _invalid_inner_chunk_sizes_v3(doc: ArrayDocument) -> tuple[ArrayDocument, st return {**doc, "codecs": codecs}, None -def _inner_chunk_shape(doc: ArrayDocument) -> list[int]: - """The inner chunk shape of the sharding codec of a Zarr format 3 array document, if - it has one and that is a list of integers; else `[]`.""" - codecs = doc.get("codecs") - if isinstance(codecs, list): - for codec in codecs: - match codec: - case { - "name": "sharding_indexed", - "configuration": {"chunk_shape": list() as inner}, - }: - return inner if _is_int_list(inner) else [] +def _inner_chunk_shape(doc: ArrayDocument) -> list[int] | None: + """The inner chunk shape of the sharding codec of a Zarr format 3 array document: + `[]` if it has none, `None` if its inner chunk sizes are not all integers of at + least 1 (the unit of its outer chunk shape is then unknown).""" + match doc.get("codecs"): + case list() as codecs: + for codec in codecs: + match codec: + case {"name": "sharding_indexed", "configuration": {"chunk_shape": inner}}: + return inner if _is_int_list(inner) and min(inner, default=1) >= 1 else None return [] @@ -183,11 +180,13 @@ def _invalid_chunk_sizes_v3(doc: ArrayDocument) -> tuple[ArrayDocument, str | No shape = doc.get("shape") if not (isinstance(grid, Mapping) and grid.get("name") == "regular" and _is_int_list(shape)): return None + if (units := _inner_chunk_shape(doc)) is None: + return None configuration = grid.get("configuration") if not isinstance(configuration, Mapping): return None stored = configuration.get("chunk_shape") - match _read_chunk_shape(stored, shape, _inner_chunk_shape(doc)): + match _read_chunk_shape(stored, shape, units): case chunk_shape, reading if not json_equal(chunk_shape, stored): upgraded = {**configuration, "chunk_shape": chunk_shape} return {**doc, "chunk_grid": {**grid, "configuration": upgraded}}, reading diff --git a/tests/test_metadata/test_upgrades.py b/tests/test_metadata/test_upgrades.py index 89dd1d236c..5e31bc4f96 100644 --- a/tests/test_metadata/test_upgrades.py +++ b/tests/test_metadata/test_upgrades.py @@ -255,6 +255,17 @@ def test_stored_chunk_shape_ndim_mismatch_rejected() -> None: _read_strictly(_v3_doc([4, 4], [0])) +@pytest.mark.parametrize("inner", [[0], [False]]) +def test_stored_zero_chunk_size_of_shard_with_invalid_inner_chunk_shape_rejected( + inner: list[Any], +) -> None: + """A stored chunk size of 0 of a sharded array is read in multiples of the inner + chunk size; if that is not an integer of at least 1, the 0 is not upgraded, so it + is rejected.""" + with pytest.raises(ValueError, match="^Dimension 0: chunk edge length must be >= 1, got 0$"): + _read_strictly(_v3_doc([4], [0], inner=inner)) + + def _rectilinear_doc(shape: list[int], chunk_shapes: list[Any]) -> dict[str, JSON]: return _v3_doc(shape, [1] * len(shape)) | { "chunk_grid": { From 3bc802fb0e0e88437f1f24ce24d5f18e9b8b5b2f Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Sat, 26 Sep 2026 16:43:13 +0200 Subject: [PATCH 35/41] perf(metadata): read each upgraded member once, concurrently, and adopt it A group handle kept the consolidated copies of upgraded members flagged, so every later group write re-read every such member, one after another, and read each twice on the first write (once to parse, once to diff). - `save_metadata` returns the metadata it stored; `AsyncGroup._save_metadata` (attrs, `update_attributes`, `delitem`, `consolidate_metadata`) and `Group.update_attributes_async` adopt it, so later writes read no member. - The refresh visits only nested groups and flagged members, and reads them with one `asyncio.gather`. The documents it reads are the ones the upsert diffs against (`read_documents` + `upsert_metadata(..., stored)`), and the array write path does the same. - `parse_stored_array` (documents -> silently marked metadata) replaces `read_stored_array`'s `(metadata, bool)` tuple, and is the one "read, mark, don't warn" path, also for code-built metadata. - A flagged member whose own document is gone or cannot be read (invalid, replaced by a group) keeps its consolidated copy instead of failing every group write. - `parse_array_metadata` of a metadata object reads it as a stored document only when `ArrayV2Metadata.chunks` holds a 0, the one chunk size that constructors still accept and the upgrades change. This removes the per-construction `to_dict` + upgrade (AsyncArray() back to 3.4.0 speed) and building arrays from codec configurations holding NumPy scalars works again, as in 3.4.0. - `create_hierarchy` documents that it stores a group's consolidated metadata as given. - Tests: second group write reads no member; nested member refresh; member without a readable document; NumPy-scalar codec configuration; stale handle whose inner chunk shape alone changed (kills `_chunk_layout` returning only the outer grid). Assisted-by: ClaudeCode:claude-opus-5-5 Co-Authored-By: Claude Opus 5.5 --- src/zarr/core/_json.py | 2 +- src/zarr/core/array.py | 52 +++++----- src/zarr/core/group.py | 8 +- src/zarr/core/metadata/io.py | 134 ++++++++++++++++--------- tests/test_metadata/test_io.py | 27 ++++-- tests/test_metadata/test_upgrades.py | 140 ++++++++++++++++++++++----- 6 files changed, 254 insertions(+), 109 deletions(-) diff --git a/src/zarr/core/_json.py b/src/zarr/core/_json.py index 4fd5ffab95..66f6236f44 100644 --- a/src/zarr/core/_json.py +++ b/src/zarr/core/_json.py @@ -41,7 +41,7 @@ def buffer_to_json(buffer: Buffer) -> JSON: return cast("JSON", json.loads(buffer.to_bytes())) -def json_equal(a: object, b: object) -> bool: +def json_equal(a: JSON, b: JSON) -> bool: """Whether two JSON values have the same JSON encoding. Python compares `True` and `1`, or `1.0` and `1`, as equal; JSON does not.""" return json.dumps(a) == json.dumps(b) diff --git a/src/zarr/core/array.py b/src/zarr/core/array.py index a1d6a73033..481aec46fb 100644 --- a/src/zarr/core/array.py +++ b/src/zarr/core/array.py @@ -118,8 +118,13 @@ ArrayV2MetadataDict, ArrayV3Metadata, ) -from zarr.core.metadata.io import read_stored_array, save_metadata, upsert_metadata -from zarr.core.metadata.upgrades import mark_upgraded, upgrade_array_document +from zarr.core.metadata.io import ( + ARRAY_DOCUMENTS, + parse_stored_array, + read_documents, + save_metadata, + upsert_metadata, +) from zarr.core.metadata.v2 import ( CompressorLikev2, get_object_codec_id, @@ -202,29 +207,18 @@ def _chunk_sizes_from_shape( return tuple(result) -def _as_json(value: Any) -> Any: - """`value`, as `to_dict` returns it, with its tuples as JSON arrays.""" - match value: - case tuple() | list(): - return [_as_json(item) for item in value] - case dict(): - return {key: _as_json(item) for key, item in value.items()} - return value - - def parse_array_metadata(data: Any, path: str | None = None) -> ArrayMetadata: """Array metadata from a metadata object or a metadata document, naming the array at `path` in warnings about how an invalid document was read. - The metadata constructors accept chunk sizes that only an invalid document holds - (such as 0), as they always have; such metadata is read as its document is (see - `zarr.core.metadata.upgrades`), so an array can be built from it. No data was read - or written under those chunk sizes, so none of its readings is a warning.""" + `ArrayV2Metadata` accepts a chunk size of 0, as it always has, though only an + invalid document holds one: such metadata is read as the documents it would store + are (see `zarr.core.metadata.upgrades`), so an array can be built from it. No data + was read or written under that chunk size, so the reading is silent.""" + if isinstance(data, ArrayV2Metadata) and 0 in data.chunks: + return parse_stored_array(data.to_buffer_dict(default_buffer_prototype()), 2) if isinstance(data, ArrayMetadata): - document, readings = upgrade_array_document(_as_json(data.to_dict()), data.zarr_format) - if not readings: - return data - return mark_upgraded(parse_array_metadata(dict(document)), [None for _ in readings], path) + return data if isinstance(data, dict): zarr_format = data.get("zarr_format") if zarr_format == 3: @@ -1652,16 +1646,20 @@ async def _store_upgraded_document(self) -> None: """ if not self.metadata._stored_document_upgraded: return - read = await read_stored_array(self.store_path, self.metadata.zarr_format) - if read is not None: - current, upgraded = read + zarr_format = self.metadata.zarr_format + documents = await read_documents(self.store_path, ARRAY_DOCUMENTS[zarr_format]) + try: + current = parse_stored_array(documents, zarr_format) + except ArrayNotFoundError: + pass + else: if _chunk_layout(current) != _chunk_layout(self.metadata): raise ValueError( f"The metadata stored for the array at {str(self.store_path)!r} has " "changed since this array was opened: reopen the array to write to it." ) - if upgraded: - await upsert_metadata(self.store_path, current) + if current._stored_document_upgraded: + await upsert_metadata(self.store_path, current, documents) object.__setattr__(self.metadata, "_stored_document_upgraded", False) async def _set_selection( @@ -4893,7 +4891,9 @@ async def create_array( ) -def _chunk_layout(metadata: ArrayMetadata) -> tuple[object, tuple[int, ...] | None]: +def _chunk_layout( + metadata: ArrayMetadata, +) -> tuple[tuple[int, ...] | ChunkGridMetadata, tuple[int, ...] | None]: """How an array's chunks are laid out: its chunk grid and, if it is sharded, the inner chunk shape.""" grid = metadata.chunks if isinstance(metadata, ArrayV2Metadata) else metadata.chunk_grid diff --git a/src/zarr/core/group.py b/src/zarr/core/group.py index 5a52e3bbf3..f0a9a98d7a 100644 --- a/src/zarr/core/group.py +++ b/src/zarr/core/group.py @@ -899,7 +899,9 @@ async def get_group(self, path: str) -> AsyncGroup: return node async def _save_metadata(self, ensure_parents: bool = False) -> None: - await save_metadata(self.store_path, self.metadata, ensure_parents=ensure_parents) + # Adopt the metadata stored: its consolidated members may be newer. + stored = await save_metadata(self.store_path, self.metadata, ensure_parents=ensure_parents) + object.__setattr__(self, "metadata", stored) @property def path(self) -> str: @@ -2115,7 +2117,7 @@ async def update_attributes_async(self, new_attributes: dict[str, Any]) -> Group new_metadata = replace(self.metadata, attributes=new_attributes) # Write new metadata - await save_metadata(self.store_path, new_metadata) + new_metadata = await save_metadata(self.store_path, new_metadata) async_group = replace(self._async_group, metadata=new_metadata) return replace(self, _async_group=async_group) @@ -3032,6 +3034,8 @@ async def create_hierarchy( ```{'': GroupMetadata, 'a': GroupMetadata, 'b': Groupmetadata}``` After input parsing, this function then creates all the nodes in the hierarchy concurrently. + The metadata of each node is stored as given: the consolidated metadata of a group is + stored as it is, without reading the documents of its members. Arrays and Groups are yielded in the order they are created. This order is not stable and should not be relied on. diff --git a/src/zarr/core/metadata/io.py b/src/zarr/core/metadata/io.py index 1503736e04..82b089ba0d 100644 --- a/src/zarr/core/metadata/io.py +++ b/src/zarr/core/metadata/io.py @@ -10,12 +10,13 @@ from zarr.core._json import buffer_to_json_object, json_equal from zarr.core.buffer.core import default_buffer_prototype from zarr.core.buffer.cpu import buffer_prototype as cpu_buffer_prototype -from zarr.core.metadata.upgrades import upgrade_array_document +from zarr.core.common import ZARR_JSON, ZARRAY_JSON, ZATTRS_JSON +from zarr.core.metadata.upgrades import mark_upgraded, upgrade_array_document from zarr.errors import ArrayNotFoundError, ContainsArrayError from zarr.storage._common import StorePath, ensure_no_existing_node if TYPE_CHECKING: - from collections.abc import Iterator, Mapping + from collections.abc import Iterable, Iterator, Mapping from zarr.core.buffer import Buffer from zarr.core.common import JSON, ZarrFormat @@ -66,7 +67,7 @@ def _diff( case list(), list(): for index, pair in enumerate(zip_longest(stored, new, fillvalue=ABSENT)): yield from _diff((*path, index), *pair) - case _ if ABSENT in (stored, new) or not json_equal(stored, new): + case _ if stored is ABSENT or new is ABSENT or not json_equal(stored, new): yield DocumentChange(path, stored, new) @@ -77,26 +78,29 @@ async def store_documents(store_path: StorePath, documents: Mapping[str, Buffer] ) +async def read_documents(store_path: StorePath, keys: Iterable[str]) -> dict[str, Buffer]: + """The documents stored under `store_path` at `keys`, read concurrently, by key; a + key with no document is left out.""" + keys = tuple(keys) + buffers = await asyncio.gather( + *((store_path / key).get(prototype=cpu_buffer_prototype) for key in keys) + ) + return {key: buf for key, buf in zip(keys, buffers, strict=True) if buf is not None} + + async def upsert_metadata( - store_path: StorePath, metadata: ArrayMetadata | GroupMetadata + store_path: StorePath, metadata: ArrayMetadata | GroupMetadata, stored: Mapping[str, Buffer] ) -> tuple[DocumentChange, ...]: - """Store the documents of `metadata` under `store_path` that differ from the stored - ones, and return how they differed (see `diff_documents`): empty if nothing was - stored. + """Store the documents of `metadata` under `store_path` that differ from `stored`, + the documents `read_documents` read there, and return how they differed (see + `diff_documents`): empty if nothing was stored. - The documents are encoded before the store is read, so metadata that cannot be - stored fails with the store untouched. + The documents are encoded before any is stored, so metadata that cannot be stored + fails with the store untouched. """ documents = metadata.to_buffer_dict(default_buffer_prototype()) - stored = await asyncio.gather( - *((store_path / key).get(prototype=cpu_buffer_prototype) for key in documents) - ) changes = diff_documents( - { - key: buffer_to_json_object(buf) - for key, buf in zip(documents, stored, strict=True) - if buf is not None - }, + {key: buffer_to_json_object(buf) for key, buf in stored.items() if key in documents}, {key: buffer_to_json_object(buf) for key, buf in documents.items()}, ) changed = {change.path[0] for change in changes} @@ -104,46 +108,77 @@ async def upsert_metadata( return changes -async def read_stored_array( - store_path: StorePath, zarr_format: ZarrFormat -) -> tuple[ArrayMetadata, bool] | None: - """The metadata of the array document stored at `store_path` as it now is, read with - the upgrades but without their warnings (the handle that asks has warned), and - whether the document had to be upgraded; `None` if no array document is stored - there.""" - from zarr.core.array import get_array_metadata, parse_array_metadata +ARRAY_DOCUMENTS: Final[Mapping[ZarrFormat, tuple[str, ...]]] = { + 2: (ZARRAY_JSON, ZATTRS_JSON), + 3: (ZARR_JSON,), +} +"""The store keys of the metadata documents of an array of each Zarr format.""" - try: - stored = await get_array_metadata(store_path, zarr_format=zarr_format) - except ArrayNotFoundError: - return None + +def parse_stored_array(documents: Mapping[str, Buffer], zarr_format: ZarrFormat) -> ArrayMetadata: + """The metadata of an array from its documents (by store key, see `ARRAY_DOCUMENTS`), + read with the upgrades but without their warnings (whoever asks has warned, or reads + metadata built in code), and marked (see `mark_upgraded`) if they had to be + upgraded. Raises `ArrayNotFoundError` if there is no array document among them.""" + from zarr.core.array import ( + _array_metadata_dict_v2, + _array_metadata_dict_v3, + parse_array_metadata, + ) + + if zarr_format == 2 and ZARRAY_JSON in documents: + stored = _array_metadata_dict_v2(documents[ZARRAY_JSON], documents.get(ZATTRS_JSON)) + elif zarr_format == 3 and ZARR_JSON in documents: + stored = _array_metadata_dict_v3(documents[ZARR_JSON]) + else: + raise ArrayNotFoundError(f"No Zarr format {zarr_format} array metadata document.") upgraded, readings = upgrade_array_document(stored, zarr_format) - return parse_array_metadata(upgraded, str(store_path)), bool(readings) + return mark_upgraded(parse_array_metadata(dict(upgraded)), [None] * len(readings), None) + + +async def _refresh_array(store_path: StorePath, member: ArrayMetadata) -> ArrayMetadata: + """The metadata of the array document stored at `store_path` as it now is, after + storing its upgrade if it needs one, for `member`, a member of consolidated metadata + that was read from a document that had to be upgraded; `member` itself if no array + document that can be read is stored there (it was deleted, or replaced by a group).""" + documents = await read_documents(store_path, ARRAY_DOCUMENTS[member.zarr_format]) + try: + current = parse_stored_array(documents, member.zarr_format) + except (KeyError, TypeError, ValueError): + return member + if current._stored_document_upgraded: + await upsert_metadata(store_path, current, documents) + object.__setattr__(current, "_stored_document_upgraded", False) + return current async def _refresh_consolidated(store_path: StorePath, metadata: GroupMetadata) -> GroupMetadata: """`metadata` with each member of its consolidated metadata that was read from a document that had to be upgraded replaced by the metadata of the member's own - document as it now is, after storing that document's upgrade if it needs one: the - member document may have changed since, so the consolidated copy is never stored - as if it were valid.""" + document as it now is (see `_refresh_array`), read concurrently: the member document + may have changed since, so the consolidated copy is never stored as if it were + valid.""" from zarr.core.group import GroupMetadata consolidated = metadata.consolidated_metadata if consolidated is None: return metadata - members = dict(consolidated.metadata) - for name, member in consolidated.metadata.items(): - if isinstance(member, GroupMetadata): - members[name] = await _refresh_consolidated(store_path / name, member) - elif member._stored_document_upgraded: - read = await read_stored_array(store_path / name, member.zarr_format) - if read is not None: - members[name], upgraded = read - if upgraded: - await upsert_metadata(store_path / name, members[name]) - if all(members[name] is member for name, member in consolidated.metadata.items()): + stale = { + name: member + for name, member in consolidated.metadata.items() + if isinstance(member, GroupMetadata) or member._stored_document_upgraded + } + refreshed = await asyncio.gather( + *( + _refresh_consolidated(store_path / name, member) + if isinstance(member, GroupMetadata) + else _refresh_array(store_path / name, member) + for name, member in stale.items() + ) + ) + if all(new is old for new, old in zip(refreshed, stale.values(), strict=True)): return metadata + members = {**consolidated.metadata, **dict(zip(stale, refreshed, strict=True))} return replace(metadata, consolidated_metadata=replace(consolidated, metadata=members)) @@ -166,10 +201,12 @@ def _build_parents(store_path: StorePath, zarr_format: ZarrFormat) -> dict[str, return parents -async def save_metadata( - store_path: StorePath, metadata: ArrayMetadata | GroupMetadata, ensure_parents: bool = False -) -> None: - """Asynchronously save the array or group metadata. +async def save_metadata[M: (ArrayMetadata, GroupMetadata)]( + store_path: StorePath, metadata: M, ensure_parents: bool = False +) -> M: + """Asynchronously save the array or group metadata, and return the metadata saved: + `metadata`, but for a group, with the consolidated members it stored (see + `_refresh_consolidated`), for the group to adopt. Parameters ---------- @@ -228,3 +265,4 @@ async def save_metadata( ) from e await asyncio.gather(*set_awaitables) + return metadata diff --git a/tests/test_metadata/test_io.py b/tests/test_metadata/test_io.py index 9accc0f79f..7fb4a06914 100644 --- a/tests/test_metadata/test_io.py +++ b/tests/test_metadata/test_io.py @@ -11,7 +11,14 @@ import zarr from zarr.core.buffer import cpu from zarr.core.metadata import ArrayV2Metadata, ArrayV3Metadata -from zarr.core.metadata.io import ABSENT, DocumentChange, diff_documents, upsert_metadata +from zarr.core.metadata.io import ( + ABSENT, + ARRAY_DOCUMENTS, + DocumentChange, + diff_documents, + read_documents, + upsert_metadata, +) from zarr.core.sync import sync from zarr.storage import MemoryStore, StorePath @@ -143,6 +150,14 @@ def _legacy(zarr_format: Literal[2, 3]) -> tuple[StorePath, ArrayV2Metadata | Ar return StorePath(store), array.metadata +def _upsert( + store_path: StorePath, metadata: ArrayV2Metadata | ArrayV3Metadata +) -> tuple[DocumentChange, ...]: + """Upsert `metadata` against the documents stored at `store_path`.""" + stored = sync(read_documents(store_path, ARRAY_DOCUMENTS[metadata.zarr_format])) + return sync(upsert_metadata(store_path, metadata, stored)) + + @pytest.mark.parametrize("zarr_format", [2, 3]) def test_upsert_metadata_stores_documents_that_differ(zarr_format: Literal[2, 3]) -> None: """The documents that differ from the stored ones are stored, and the changes are @@ -155,7 +170,7 @@ def test_upsert_metadata_stores_documents_that_differ(zarr_format: Literal[2, 3] ("chunks", 0) if zarr_format == 2 else ("chunk_grid", "configuration", "chunk_shape", 0) ) - changes = sync(upsert_metadata(store_path, metadata)) + changes = _upsert(store_path, metadata) assert changes == (DocumentChange((key, *chunk_path), 0, 3),) assert store_path.store.sets == 1 @@ -172,12 +187,12 @@ def test_upsert_metadata_identical_stores_nothing(zarr_format: Literal[2, 3]) -> ) store.sets = 0 - assert sync(upsert_metadata(StorePath(store), array.metadata)) == () + assert _upsert(StorePath(store), array.metadata) == () assert store.sets == 0 def test_upsert_metadata_unstorable_leaves_store_untouched(monkeypatch: pytest.MonkeyPatch) -> None: - """Metadata that cannot be encoded fails before the store is read or written.""" + """Metadata that cannot be encoded fails before the store is written.""" store_path, metadata = _legacy(3) before = _documents(store_path.store) @@ -186,7 +201,7 @@ def refuse(*args: object) -> None: monkeypatch.setattr(ArrayV3Metadata, "to_buffer_dict", refuse) with pytest.raises(ValueError, match="cannot be stored"): - sync(upsert_metadata(store_path, metadata)) + _upsert(store_path, metadata) assert _documents(store_path.store) == before @@ -195,5 +210,5 @@ def test_upsert_metadata_stored_document_not_an_object() -> None: store_path, metadata = _legacy(3) sync(store_path.store.set("zarr.json", cpu.Buffer.from_bytes(b"[]"))) with pytest.raises(TypeError, match="Expected a JSON object, got list"): - sync(upsert_metadata(store_path, metadata)) + _upsert(store_path, metadata) assert _documents(store_path.store) == {"zarr.json": []} diff --git a/tests/test_metadata/test_upgrades.py b/tests/test_metadata/test_upgrades.py index 5e31bc4f96..e299fe72ed 100644 --- a/tests/test_metadata/test_upgrades.py +++ b/tests/test_metadata/test_upgrades.py @@ -3,6 +3,7 @@ from __future__ import annotations import asyncio +import dataclasses import json import re import warnings @@ -13,6 +14,7 @@ import zarr from zarr.codecs import ShardingCodec +from zarr.codecs.numcodecs import Quantize from zarr.core.array import AsyncArray from zarr.core.metadata import ArrayV2Metadata, ArrayV3Metadata from zarr.core.metadata.upgrades import ( @@ -23,7 +25,7 @@ from zarr.core.sync import sync from zarr.dtype import Int16 from zarr.errors import ZarrUserWarning -from zarr.storage import MemoryStore, StorePath +from zarr.storage import LocalStore, MemoryStore, StorePath from zarr.storage._common import make_store_path if TYPE_CHECKING: @@ -732,20 +734,43 @@ def test_stale_handle_write_keeps_valid_document_as_written( np.testing.assert_array_equal(_open_strictly(path)[...], [9, 0, 0]) -@pytest.mark.parametrize("zarr_format", [2, 3]) +def _store_zero(doc: dict[str, Any]) -> None: + _stored_chunks(doc)[0] = 0 + + +def _resize_to_10(doc: dict[str, Any]) -> None: + doc["shape"] = [10] + + +def _halve_inner_chunk_shape(doc: dict[str, Any]) -> None: + doc["codecs"][0]["configuration"]["chunk_shape"] = [2] + + +@pytest.mark.parametrize( + ("zarr_format", "sharded", "change"), + [(2, False, _resize_to_10), (3, False, _resize_to_10), (3, True, _halve_inner_chunk_shape)], + ids=["v2-resized", "v3-resized", "v3-sharded-inner-chunk-shape"], +) def test_stale_handle_write_after_chunk_grid_change_raises( - tmp_path: Path, zarr_format: Literal[2, 3] + tmp_path: Path, + zarr_format: Literal[2, 3], + sharded: bool, + change: Callable[[dict[str, Any]], None], ) -> None: """If the document the store holds when a handle read from an upgraded document - first writes chunks lays out chunks differently from the handle's metadata (here the + first writes chunks lays out chunks differently from the handle's metadata (the array was resized by software that kept the stored chunk size of 0, which now reads - as a larger chunk), the handle's chunks would not be found under it: the write - raises and stores nothing.""" + as a larger chunk, or its inner chunk shape changed), the handle's chunks would not + be found under it: the write raises and stores nothing.""" path = tmp_path / "legacy.zarr" - _legacy_array(path, zarr_format) + if sharded: + zarr.create_array(store=path, shape=(3,), chunks=(4,), shards=(4,), dtype="int16") + _rewrite_doc(path, 3, _store_zero) + else: + _legacy_array(path, zarr_format) with pytest.warns(ZarrUserWarning, match="is read as"): stale = zarr.open_array(store=path, mode="r+") - _rewrite_doc(path, zarr_format, lambda doc: doc.update(shape=[10])) + _rewrite_doc(path, zarr_format, change) documents = {p.name: p.read_bytes() for p in path.iterdir()} with pytest.raises(ValueError, match="has changed since this array was opened: reopen"): @@ -796,8 +821,19 @@ def test_array_from_metadata_with_chunk_size_zero(shape: tuple[int], expected: t np.testing.assert_array_equal(zarr.open_array(store, path="a")[...], np.ones(shape)) -def _store_zero(doc: dict[str, Any]) -> None: - _stored_chunks(doc)[0] = 0 +def test_array_from_metadata_with_numpy_scalar_codec_configuration() -> None: + """An array is built from metadata whose codec configuration holds NumPy scalars + (which are not JSON values), as zarr always built one.""" + array = zarr.create_array( + MemoryStore(), shape=(4,), chunks=(2,), dtype="f8", filters=[Quantize(digits=3, dtype="f8")] + ) + assert isinstance(array.metadata, ArrayV3Metadata) + # The codec keeps its configuration as given, though it is typed as JSON. + digits = cast("JSON", np.int64(3)) + codecs = (Quantize(digits=digits, dtype="f8"), *array.metadata.codecs[1:]) + metadata = dataclasses.replace(array.metadata, codecs=codecs) + + assert AsyncArray(metadata, StorePath(MemoryStore())).metadata is metadata def _rewrite_consolidated( @@ -818,44 +854,96 @@ def _consolidated_member(path: Path, zarr_format: Literal[2, 3], name: str) -> A return json.loads((path / "zarr.json").read_text())["consolidated_metadata"]["metadata"][name] +def _flagged_consolidated_group(path: Path, zarr_format: Literal[2, 3], member: str) -> None: + """A group whose consolidated copy of the array `member` holds the stored chunk size + 0, and whose array `b` is valid.""" + group = zarr.open_group(path, mode="w", zarr_format=zarr_format) + parent, _, name = member.rpartition("/") + (group.require_group(parent) if parent else group).create_array( + name, shape=(3,), chunks=(3,), dtype="int16" + ) + group.create_array("b", shape=(1,), chunks=(1,), dtype="int16") + zarr.consolidate_metadata(path) + _rewrite_consolidated(path, zarr_format, member, _store_zero) + + @pytest.mark.filterwarnings("ignore:Consolidated metadata is currently not part:UserWarning") @pytest.mark.parametrize("zarr_format", [2, 3]) +@pytest.mark.parametrize("member", ["a", "g/a"]) @pytest.mark.parametrize("operation", ["attrs", "update_attributes_async", "delete-member"]) def test_group_write_refreshes_upgraded_consolidated_member( - tmp_path: Path, zarr_format: Literal[2, 3], operation: str + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, + zarr_format: Literal[2, 3], + member: str, + operation: str, ) -> None: """Storing a group's metadata stores its consolidated metadata. Each member of it read - from a document that had to be upgraded is first read again from the member's own - document as it now is (here resized by software that kept the stored chunk size of - 0), whose upgrade is stored, so no group write stores a stale copy as if valid.""" + from a document that had to be upgraded, at any depth, is first read again from the + member's own document as it now is (here resized by software that kept the stored + chunk size of 0), whose upgrade is stored, so no group write stores a stale copy as + if valid. The group adopts the members it stored, so later writes read no member.""" path = tmp_path / "group.zarr" - group = zarr.open_group(path, mode="w", zarr_format=zarr_format) - group.create_array("a", shape=(3,), chunks=(3,), dtype="int16") - group.create_array("b", shape=(1,), chunks=(1,), dtype="int16") - zarr.consolidate_metadata(path) - _rewrite_consolidated(path, zarr_format, "a", _store_zero) + _flagged_consolidated_group(path, zarr_format, member) def resize_keeping_zero(doc: dict[str, Any]) -> None: _store_zero(doc) doc["shape"] = [10] - _rewrite_doc(path / "a", zarr_format, resize_keeping_zero) + _rewrite_doc(path / member, zarr_format, resize_keeping_zero) with pytest.warns(ZarrUserWarning, match="is read as"): group = zarr.open_group(path, mode="r+", use_consolidated=True) if operation == "attrs": group.attrs["x"] = 1 elif operation == "update_attributes_async": - sync(group.update_attributes_async({"x": 1})) + group = sync(group.update_attributes_async({"x": 1})) else: del group["b"] - assert _stored_chunks(_consolidated_member(path, zarr_format, "a")) == [10] + assert _stored_chunks(_consolidated_member(path, zarr_format, member)) == [10] reopened = zarr.open_group(path, mode="r", use_consolidated=True) - member = reopened["a"] - assert isinstance(member, zarr.Array) - assert (member.shape, member.chunks) == ((10,), (10,)) - assert _open_strictly(path / "a").chunks == (10,) + array = reopened[member] + assert isinstance(array, zarr.Array) + assert (array.shape, array.chunks) == ((10,), (10,)) + assert _open_strictly(path / member).chunks == (10,) + + reads: list[str] = [] + get = LocalStore.get + + async def recording_get(self: LocalStore, key: str, *args: Any, **kwargs: Any) -> Any: + reads.append(key) + return await get(self, key, *args, **kwargs) + + monkeypatch.setattr(LocalStore, "get", recording_get) + group.attrs["y"] = 2 + assert reads == [] + + +@pytest.mark.filterwarnings("ignore:Consolidated metadata is currently not part:UserWarning") +@pytest.mark.parametrize("zarr_format", [2, 3]) +@pytest.mark.parametrize("replacement", ["group", "invalid", "none"]) +def test_group_write_keeps_upgraded_member_without_readable_document( + tmp_path: Path, zarr_format: Literal[2, 3], replacement: str +) -> None: + """A member of consolidated metadata read from a document that had to be upgraded, + whose own document is no longer an array document that can be read, has no document + to upgrade: a group write stores the consolidated copy as it is.""" + path = tmp_path / "group.zarr" + _flagged_consolidated_group(path, zarr_format, "a") + with pytest.warns(ZarrUserWarning, match="is read as"): + group = zarr.open_group(path, mode="r+", use_consolidated=True) + del zarr.open_group(path, mode="r+", use_consolidated=False)["a"] + if replacement == "group": + zarr.create_group(path / "a", zarr_format=zarr_format) + elif replacement == "invalid": + (path / "a").mkdir() + document = path / "a" / (".zarray" if zarr_format == 2 else "zarr.json") + document.write_text(json.dumps({"zarr_format": zarr_format, "node_type": "array"})) + + group.attrs["x"] = 1 + + assert _stored_chunks(_consolidated_member(path, zarr_format, "a")) == [3] @pytest.mark.parametrize("zarr_format", [2, 3]) From dc50ec61d8ebd0facee9cf99b4433f338c2debf6 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Sat, 26 Sep 2026 18:02:28 +0200 Subject: [PATCH 36/41] fix(group): adopt refreshed consolidated members in place A group write replaced the handle's metadata with a copy holding the consolidated members it read again, so the handle no longer shared its consolidated dicts with subgroup handles taken earlier: an array deleted through such a subgroup was stored again by the parent's next write, and reappeared on reopen. `_refresh_consolidated` now returns, along with the copy it encodes, the members it replaced (with the consolidated dict that holds each), and `save_metadata` writes them back into those dicts once the store succeeds, unless the member was deleted or replaced meanwhile. `save_metadata` returns nothing again, and the group callers keep their metadata objects. A failed store leaves the members flagged, so a retry reads them again. Assisted-by: ClaudeCode:claude-opus-5-5 Co-Authored-By: Claude Opus 5.5 --- src/zarr/core/group.py | 6 +- src/zarr/core/metadata/io.py | 89 +++++++++++++++++++--------- tests/test_metadata/test_upgrades.py | 27 +++++++++ 3 files changed, 91 insertions(+), 31 deletions(-) diff --git a/src/zarr/core/group.py b/src/zarr/core/group.py index f0a9a98d7a..cd91582315 100644 --- a/src/zarr/core/group.py +++ b/src/zarr/core/group.py @@ -899,9 +899,7 @@ async def get_group(self, path: str) -> AsyncGroup: return node async def _save_metadata(self, ensure_parents: bool = False) -> None: - # Adopt the metadata stored: its consolidated members may be newer. - stored = await save_metadata(self.store_path, self.metadata, ensure_parents=ensure_parents) - object.__setattr__(self, "metadata", stored) + await save_metadata(self.store_path, self.metadata, ensure_parents=ensure_parents) @property def path(self) -> str: @@ -2117,7 +2115,7 @@ async def update_attributes_async(self, new_attributes: dict[str, Any]) -> Group new_metadata = replace(self.metadata, attributes=new_attributes) # Write new metadata - new_metadata = await save_metadata(self.store_path, new_metadata) + await save_metadata(self.store_path, new_metadata) async_group = replace(self._async_group, metadata=new_metadata) return replace(self, _async_group=async_group) diff --git a/src/zarr/core/metadata/io.py b/src/zarr/core/metadata/io.py index 82b089ba0d..281aee04ff 100644 --- a/src/zarr/core/metadata/io.py +++ b/src/zarr/core/metadata/io.py @@ -152,34 +152,66 @@ async def _refresh_array(store_path: StorePath, member: ArrayMetadata) -> ArrayM return current -async def _refresh_consolidated(store_path: StorePath, metadata: GroupMetadata) -> GroupMetadata: +class _Refreshed(NamedTuple): + """A member of consolidated metadata that `_refresh_consolidated` read again from the + member's own document.""" + + members: dict[str, ArrayMetadata | GroupMetadata] + """The consolidated metadata that holds the member, by member name.""" + name: str + stale: ArrayMetadata + current: ArrayMetadata + + def adopt(self) -> None: + """Replace the stale member with the current one in place, so every handle that + shares this consolidated metadata sees it; unless the member was deleted or + replaced since it was read.""" + if self.members.get(self.name) is self.stale: + self.members[self.name] = self.current + + +async def _refresh_consolidated( + store_path: StorePath, metadata: GroupMetadata +) -> tuple[GroupMetadata, list[_Refreshed]]: """`metadata` with each member of its consolidated metadata that was read from a - document that had to be upgraded replaced by the metadata of the member's own - document as it now is (see `_refresh_array`), read concurrently: the member document - may have changed since, so the consolidated copy is never stored as if it were - valid.""" + document that had to be upgraded, at any depth, replaced by the metadata of the + member's own document as it now is (see `_refresh_array`), read concurrently, and + the members it replaced: the member document may have changed since, so the + consolidated copy is never stored as if it were valid. `metadata` is left as it is, + for the group to adopt the members (see `_Refreshed.adopt`) once they are stored.""" from zarr.core.group import GroupMetadata consolidated = metadata.consolidated_metadata if consolidated is None: - return metadata - stale = { + return metadata, [] + groups = { name: member for name, member in consolidated.metadata.items() - if isinstance(member, GroupMetadata) or member._stored_document_upgraded + if isinstance(member, GroupMetadata) } - refreshed = await asyncio.gather( - *( - _refresh_consolidated(store_path / name, member) - if isinstance(member, GroupMetadata) - else _refresh_array(store_path / name, member) - for name, member in stale.items() - ) + arrays = { + name: member + for name, member in consolidated.metadata.items() + if not isinstance(member, GroupMetadata) and member._stored_document_upgraded + } + nested, current = await asyncio.gather( + asyncio.gather(*(_refresh_consolidated(store_path / n, m) for n, m in groups.items())), + asyncio.gather(*(_refresh_array(store_path / n, m) for n, m in arrays.items())), ) - if all(new is old for new, old in zip(refreshed, stale.values(), strict=True)): - return metadata - members = {**consolidated.metadata, **dict(zip(stale, refreshed, strict=True))} - return replace(metadata, consolidated_metadata=replace(consolidated, metadata=members)) + refreshed = [ + _Refreshed(consolidated.metadata, name, stale, new) + for (name, stale), new in zip(arrays.items(), current, strict=True) + if new is not stale + ] + members = dict(consolidated.metadata) + members.update({member.name: member.current for member in refreshed}) + for name, (group, inner) in zip(groups, nested, strict=True): + members[name] = group + refreshed.extend(inner) + if not refreshed: + return metadata, [] + consolidated = replace(consolidated, metadata=members) + return replace(metadata, consolidated_metadata=consolidated), refreshed def _build_parents(store_path: StorePath, zarr_format: ZarrFormat) -> dict[str, GroupMetadata]: @@ -201,12 +233,12 @@ def _build_parents(store_path: StorePath, zarr_format: ZarrFormat) -> dict[str, return parents -async def save_metadata[M: (ArrayMetadata, GroupMetadata)]( - store_path: StorePath, metadata: M, ensure_parents: bool = False -) -> M: - """Asynchronously save the array or group metadata, and return the metadata saved: - `metadata`, but for a group, with the consolidated members it stored (see - `_refresh_consolidated`), for the group to adopt. +async def save_metadata( + store_path: StorePath, metadata: ArrayMetadata | GroupMetadata, ensure_parents: bool = False +) -> None: + """Asynchronously save the array or group metadata. A group adopts, in place, the + members of its consolidated metadata that had to be read again to store it (see + `_refresh_consolidated`), once they are stored. Parameters ---------- @@ -223,9 +255,10 @@ async def save_metadata[M: (ArrayMetadata, GroupMetadata)]( """ from zarr.core.group import GroupMetadata + refreshed: list[_Refreshed] = [] if isinstance(metadata, GroupMetadata): # The one place group metadata is stored, and with it consolidated metadata. - metadata = await _refresh_consolidated(store_path, metadata) + metadata, refreshed = await _refresh_consolidated(store_path, metadata) to_save = metadata.to_buffer_dict(default_buffer_prototype()) set_awaitables = [store_documents(store_path, to_save)] @@ -265,4 +298,6 @@ async def save_metadata[M: (ArrayMetadata, GroupMetadata)]( ) from e await asyncio.gather(*set_awaitables) - return metadata + # Only once stored: a group whose store failed keeps the members flagged, to read again. + for member in refreshed: + member.adopt() diff --git a/tests/test_metadata/test_upgrades.py b/tests/test_metadata/test_upgrades.py index e299fe72ed..8da2b24a5e 100644 --- a/tests/test_metadata/test_upgrades.py +++ b/tests/test_metadata/test_upgrades.py @@ -920,6 +920,33 @@ async def recording_get(self: LocalStore, key: str, *args: Any, **kwargs: Any) - assert reads == [] +@pytest.mark.filterwarnings("ignore:Consolidated metadata is currently not part:UserWarning") +@pytest.mark.parametrize("zarr_format", [2, 3]) +def test_group_write_keeps_consolidated_metadata_shared_with_subgroups( + tmp_path: Path, zarr_format: Literal[2, 3] +) -> None: + """A subgroup handle taken from a group shares the group's consolidated metadata, also + once the group has adopted the upgraded members its first write stored: an array + deleted through the subgroup is gone from what the group stores next.""" + path = tmp_path / "group.zarr" + created = zarr.open_group(path, mode="w", zarr_format=zarr_format).create_group("g") + for name in ("a", "c"): + created.create_array(name, shape=(3,), chunks=(3,), dtype="int16") + zarr.consolidate_metadata(path) + _rewrite_consolidated(path, zarr_format, "g/a", _store_zero) + with pytest.warns(ZarrUserWarning, match="is read as"): + group = zarr.open_group(path, mode="r+", use_consolidated=True) + subgroup = group["g"] + assert isinstance(subgroup, zarr.Group) + + group.attrs["x"] = 1 + del subgroup["c"] + group.attrs["y"] = 2 + + reopened = zarr.open_group(path, mode="r", use_consolidated=True) + assert sorted(dict(reopened.members(max_depth=None))) == ["g", "g/a"] + + @pytest.mark.filterwarnings("ignore:Consolidated metadata is currently not part:UserWarning") @pytest.mark.parametrize("zarr_format", [2, 3]) @pytest.mark.parametrize("replacement", ["group", "invalid", "none"]) From cddb877261bc6c8b42961bc2c0382680e1386824 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Sat, 26 Sep 2026 18:02:38 +0200 Subject: [PATCH 37/41] fix(metadata): raise the rectilinear flag error when refreshing a consolidated member A group write reads each upgraded consolidated member again from its own document, and kept the consolidated copy when that document could not be read. That also swallowed the rectilinear chunks flag error, so a member whose document now declares a rectilinear grid had its stale copy stored as valid. The flag error is now `RectilinearChunksDisabledError`, a `ValueError`, and the refresh lets it propagate: the group write raises and stores nothing. Unreadable documents keep their consolidated copy as before. Assisted-by: ClaudeCode:claude-opus-5-5 Co-Authored-By: Claude Opus 5.5 --- src/zarr/core/metadata/io.py | 3 +++ src/zarr/core/metadata/v3.py | 6 +++++- tests/test_metadata/test_upgrades.py | 18 ++++++++++++++++++ 3 files changed, 26 insertions(+), 1 deletion(-) diff --git a/src/zarr/core/metadata/io.py b/src/zarr/core/metadata/io.py index 281aee04ff..c477517b54 100644 --- a/src/zarr/core/metadata/io.py +++ b/src/zarr/core/metadata/io.py @@ -12,6 +12,7 @@ from zarr.core.buffer.cpu import buffer_prototype as cpu_buffer_prototype from zarr.core.common import ZARR_JSON, ZARRAY_JSON, ZATTRS_JSON from zarr.core.metadata.upgrades import mark_upgraded, upgrade_array_document +from zarr.core.metadata.v3 import RectilinearChunksDisabledError from zarr.errors import ArrayNotFoundError, ContainsArrayError from zarr.storage._common import StorePath, ensure_no_existing_node @@ -144,6 +145,8 @@ async def _refresh_array(store_path: StorePath, member: ArrayMetadata) -> ArrayM documents = await read_documents(store_path, ARRAY_DOCUMENTS[member.zarr_format]) try: current = parse_stored_array(documents, member.zarr_format) + except RectilinearChunksDisabledError: + raise # A document that can be read, but only with the flag. except (KeyError, TypeError, ValueError): return member if current._stored_document_upgraded: diff --git a/src/zarr/core/metadata/v3.py b/src/zarr/core/metadata/v3.py index ad16afc3e1..17e88d17e4 100644 --- a/src/zarr/core/metadata/v3.py +++ b/src/zarr/core/metadata/v3.py @@ -270,6 +270,10 @@ def from_dict(cls, data: RegularChunkGridMetadataJSON) -> Self: # type: ignore[ return cls(chunk_shape=parse_chunk_shape(configuration["chunk_shape"])) +class RectilinearChunksDisabledError(ValueError): + """Rectilinear chunk grids are used while the `array.rectilinear_chunks` flag is off.""" + + @dataclass(frozen=True, kw_only=True) class RectilinearChunkGridMetadata(Metadata): """Metadata-only description of a rectilinear chunk grid. @@ -290,7 +294,7 @@ class RectilinearChunkGridMetadata(Metadata): def __post_init__(self) -> None: if not config.get("array.rectilinear_chunks"): - raise ValueError( + raise RectilinearChunksDisabledError( "Rectilinear chunk grids are experimental and disabled by default. " "Enable them with: zarr.config.set({'array.rectilinear_chunks': True}) " "or set the environment variable ZARR_ARRAY__RECTILINEAR_CHUNKS=True" diff --git a/tests/test_metadata/test_upgrades.py b/tests/test_metadata/test_upgrades.py index 8da2b24a5e..c2b76b12aa 100644 --- a/tests/test_metadata/test_upgrades.py +++ b/tests/test_metadata/test_upgrades.py @@ -947,6 +947,24 @@ def test_group_write_keeps_consolidated_metadata_shared_with_subgroups( assert sorted(dict(reopened.members(max_depth=None))) == ["g", "g/a"] +@pytest.mark.filterwarnings("ignore:Consolidated metadata is currently not part:UserWarning") +def test_group_write_refuses_upgraded_member_read_as_rectilinear(tmp_path: Path) -> None: + """A member of consolidated metadata read from a document that had to be upgraded, + whose own document now declares a rectilinear chunk grid, is read again only with the + rectilinear chunks flag: without it, a group write raises and stores nothing.""" + path = tmp_path / "group.zarr" + _flagged_consolidated_group(path, 3, "a") + with pytest.warns(ZarrUserWarning, match="is read as"): + group = zarr.open_group(path, mode="r+", use_consolidated=True) + (path / "a" / "zarr.json").write_text(json.dumps(_rectilinear_doc([3], [[1, 2]]))) + documents = {p: p.read_bytes() for p in path.rglob("*") if p.is_file()} + + with pytest.raises(ValueError, match="Rectilinear chunk grids are experimental"): + group.attrs["x"] = 1 + + assert {p: p.read_bytes() for p in path.rglob("*") if p.is_file()} == documents + + @pytest.mark.filterwarnings("ignore:Consolidated metadata is currently not part:UserWarning") @pytest.mark.parametrize("zarr_format", [2, 3]) @pytest.mark.parametrize("replacement", ["group", "invalid", "none"]) From 4b9f82c22dee55d88a623b1090145014dfda46e0 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Sat, 26 Sep 2026 19:20:54 +0200 Subject: [PATCH 38/41] fix(group): store a group's documents only; keep upgraded members as stored Group writes (attributes, `update_attributes`, deleting a member, `consolidate_metadata`, `create_hierarchy`) read each upgraded member of the consolidated metadata again from its own document and stored its upgrade. Those member writes raced concurrent deletions: two concurrent `delitem`s on a consolidated group recreated a deleted array's metadata document in most runs, which zarr 3.4.0 never did. A group write now stores only the group's own documents and reads none. Array metadata read from a document that had to be upgraded keeps that document (`_stored_document`, set by `mark_upgraded`, replacing the `_stored_document_upgraded` flag), and consolidated metadata stores such a member as it was stored. Every reader of the consolidated metadata then reads the member as upgraded again, and the array's own first chunk write stores the upgrade of its current document (or refuses a changed chunk grid) as before: the only write that upgrades a member document. Removed: the refresh and adoption of consolidated members in `save_metadata` (`_refresh_consolidated`, `_refresh_array`, `_Refreshed`), and `Group.update_attributes_async`'s detour through `save_metadata`. Tests pinning the refresh are replaced by tests that a group write reads nothing and writes only the group's documents, stores the member's consolidated copy byte for byte as stored, that concurrent deletions leave no member, and that the first write through consolidated metadata stores the member's upgrade or refuses a changed grid. Assisted-by: ClaudeCode:claude-opus-5-5 Co-Authored-By: Claude Opus 5.5 --- changes/4334.bugfix.md | 2 +- src/zarr/core/array.py | 6 +- src/zarr/core/group.py | 13 +- src/zarr/core/metadata/io.py | 99 +------------ src/zarr/core/metadata/upgrades.py | 18 ++- src/zarr/core/metadata/v2.py | 10 +- src/zarr/core/metadata/v3.py | 10 +- tests/test_array_stateful.py | 2 +- tests/test_metadata/test_upgrades.py | 203 +++++++++++++-------------- 9 files changed, 137 insertions(+), 226 deletions(-) diff --git a/changes/4334.bugfix.md b/changes/4334.bugfix.md index 3828d097eb..6279623347 100644 --- a/changes/4334.bugfix.md +++ b/changes/4334.bugfix.md @@ -4,4 +4,4 @@ A chunk edge length is now always at least 1, while an array extent may be 0. Ev Stored metadata with a regular chunk size of 0 or JSON `false` (Zarr format 2 `chunks`, Zarr format 3 `regular` `chunk_shape`) is now read as one chunk spanning the axis (a multiple of the inner chunk size for sharded arrays, which previously failed to open). zarr-python wrote such sizes for arrays created with a zero-length axis until 3.4, and, from 2.18.7 to 3.2.1, for an explicit chunk size of 0 or `False` on an axis of any length, as in `chunks=(0,)` (those arrays could store no data). On a zero-length axis this opens silently, as before; appending to such an axis then stores chunks of size 1. On an axis of positive length it opens with a `ZarrUserWarning`, which says that the axis holds only the fill value and how to store valid metadata: `array.update_attributes({})`, then `zarr.consolidate_metadata` if the metadata is consolidated. JSON `true`, as zarr-python 3.0 and 3.2 wrote for a chunk size of `True`, is read as 1, silently, in the chunk grid (regular, and the explicit edges and run-length encoded sizes of a rectilinear one) and in the inner chunk shape of every sharding codec, nested or not. A stored rectilinear chunk grid whose edge lengths are integral JSON floats (`[[4.0, 2]]`), as zarr-python 3.2 wrote for float edges, is read with those edges as integers, silently; a float anywhere no release wrote one (a regular chunk shape, the inner chunk shape of a sharding codec, a run-length repeat count, an edge below 1) is rejected. -Writing data to an array read from such metadata (other than an empty selection) first stores the upgrade of the metadata the store then holds, unless another writer has stored valid metadata since, so that other readers find the chunks written; in a `ZipStore` this adds a second entry for the metadata document, as every metadata update does. If the metadata the store then holds lays out chunks differently (another writer resized the array keeping the chunk size of 0, say), the write raises a `ValueError` asking to reopen the array, and stores nothing. Storing a group's metadata, which `zarr.consolidate_metadata` and every change to a group with consolidated metadata do, likewise first stores the upgrade of each such member's metadata and consolidates the member's metadata as the store then holds it. +Writing data to an array read from such metadata (other than an empty selection) first stores the upgrade of the metadata the store then holds, unless another writer has stored valid metadata since, so that other readers find the chunks written; in a `ZipStore` this adds a second entry for the metadata document, as every metadata update does. If the metadata the store then holds lays out chunks differently (another writer resized the array keeping the chunk size of 0, say), the write raises a `ValueError` asking to reopen the array, and stores nothing. This is the only operation that stores the upgrade: a group's consolidated metadata keeps such an array's metadata as it was stored (`zarr.consolidate_metadata` copies it as the array's own document holds it), and changing a group reads and writes no metadata of its members. diff --git a/src/zarr/core/array.py b/src/zarr/core/array.py index 481aec46fb..7b23f3b917 100644 --- a/src/zarr/core/array.py +++ b/src/zarr/core/array.py @@ -1644,7 +1644,7 @@ async def _store_upgraded_document(self) -> None: upgrade. Storing the same upgrade twice is harmless, so concurrent callers need no coordination. """ - if not self.metadata._stored_document_upgraded: + if self.metadata._stored_document is None: return zarr_format = self.metadata.zarr_format documents = await read_documents(self.store_path, ARRAY_DOCUMENTS[zarr_format]) @@ -1658,9 +1658,9 @@ async def _store_upgraded_document(self) -> None: f"The metadata stored for the array at {str(self.store_path)!r} has " "changed since this array was opened: reopen the array to write to it." ) - if current._stored_document_upgraded: + if current._stored_document is not None: await upsert_metadata(self.store_path, current, documents) - object.__setattr__(self.metadata, "_stored_document_upgraded", False) + object.__setattr__(self.metadata, "_stored_document", None) async def _set_selection( self, diff --git a/src/zarr/core/group.py b/src/zarr/core/group.py index cd91582315..4f60856425 100644 --- a/src/zarr/core/group.py +++ b/src/zarr/core/group.py @@ -13,6 +13,7 @@ import zarr.api.asynchronous as async_api from zarr.abc.metadata import Metadata +from zarr.abc.store import Store, set_or_delete from zarr.core._info import GroupInfo from zarr.core._json import buffer_to_json_object, json_to_buffer from zarr.core.array import ( @@ -74,7 +75,6 @@ ) from typing import Any - from zarr.abc.store import Store from zarr.core.array_spec import ArrayConfigLike from zarr.core.buffer import Buffer, BufferPrototype from zarr.core.chunk_key_encodings import ChunkKeyEncodingLike @@ -147,11 +147,16 @@ class ConsolidatedMetadata: must_understand: Literal[False] = False def to_dict(self) -> dict[str, JSON]: + """The consolidated metadata document. An array read from a stored document that + had to be upgraded is written as that document was stored: only the array's + own first chunk write stores its upgrade.""" return { "kind": self.kind, "must_understand": self.must_understand, "metadata": { k: v.to_dict() + if isinstance(v, GroupMetadata) or v._stored_document is None + else dict(v._stored_document) for k, v in sorted( self.flattened_metadata.items(), key=lambda item: ( @@ -2115,7 +2120,9 @@ async def update_attributes_async(self, new_attributes: dict[str, Any]) -> Group new_metadata = replace(self.metadata, attributes=new_attributes) # Write new metadata - await save_metadata(self.store_path, new_metadata) + to_save = new_metadata.to_buffer_dict(default_buffer_prototype()) + awaitables = [set_or_delete(self.store_path / key, value) for key, value in to_save.items()] + await asyncio.gather(*awaitables) async_group = replace(self._async_group, metadata=new_metadata) return replace(self, _async_group=async_group) @@ -3032,8 +3039,6 @@ async def create_hierarchy( ```{'': GroupMetadata, 'a': GroupMetadata, 'b': Groupmetadata}``` After input parsing, this function then creates all the nodes in the hierarchy concurrently. - The metadata of each node is stored as given: the consolidated metadata of a group is - stored as it is, without reading the documents of its members. Arrays and Groups are yielded in the order they are created. This order is not stable and should not be relied on. diff --git a/src/zarr/core/metadata/io.py b/src/zarr/core/metadata/io.py index c477517b54..c5c1c24da3 100644 --- a/src/zarr/core/metadata/io.py +++ b/src/zarr/core/metadata/io.py @@ -1,7 +1,6 @@ from __future__ import annotations import asyncio -from dataclasses import replace from enum import Enum from itertools import zip_longest from typing import TYPE_CHECKING, Final, NamedTuple @@ -12,7 +11,6 @@ from zarr.core.buffer.cpu import buffer_prototype as cpu_buffer_prototype from zarr.core.common import ZARR_JSON, ZARRAY_JSON, ZATTRS_JSON from zarr.core.metadata.upgrades import mark_upgraded, upgrade_array_document -from zarr.core.metadata.v3 import RectilinearChunksDisabledError from zarr.errors import ArrayNotFoundError, ContainsArrayError from zarr.storage._common import StorePath, ensure_no_existing_node @@ -134,87 +132,7 @@ def parse_stored_array(documents: Mapping[str, Buffer], zarr_format: ZarrFormat) else: raise ArrayNotFoundError(f"No Zarr format {zarr_format} array metadata document.") upgraded, readings = upgrade_array_document(stored, zarr_format) - return mark_upgraded(parse_array_metadata(dict(upgraded)), [None] * len(readings), None) - - -async def _refresh_array(store_path: StorePath, member: ArrayMetadata) -> ArrayMetadata: - """The metadata of the array document stored at `store_path` as it now is, after - storing its upgrade if it needs one, for `member`, a member of consolidated metadata - that was read from a document that had to be upgraded; `member` itself if no array - document that can be read is stored there (it was deleted, or replaced by a group).""" - documents = await read_documents(store_path, ARRAY_DOCUMENTS[member.zarr_format]) - try: - current = parse_stored_array(documents, member.zarr_format) - except RectilinearChunksDisabledError: - raise # A document that can be read, but only with the flag. - except (KeyError, TypeError, ValueError): - return member - if current._stored_document_upgraded: - await upsert_metadata(store_path, current, documents) - object.__setattr__(current, "_stored_document_upgraded", False) - return current - - -class _Refreshed(NamedTuple): - """A member of consolidated metadata that `_refresh_consolidated` read again from the - member's own document.""" - - members: dict[str, ArrayMetadata | GroupMetadata] - """The consolidated metadata that holds the member, by member name.""" - name: str - stale: ArrayMetadata - current: ArrayMetadata - - def adopt(self) -> None: - """Replace the stale member with the current one in place, so every handle that - shares this consolidated metadata sees it; unless the member was deleted or - replaced since it was read.""" - if self.members.get(self.name) is self.stale: - self.members[self.name] = self.current - - -async def _refresh_consolidated( - store_path: StorePath, metadata: GroupMetadata -) -> tuple[GroupMetadata, list[_Refreshed]]: - """`metadata` with each member of its consolidated metadata that was read from a - document that had to be upgraded, at any depth, replaced by the metadata of the - member's own document as it now is (see `_refresh_array`), read concurrently, and - the members it replaced: the member document may have changed since, so the - consolidated copy is never stored as if it were valid. `metadata` is left as it is, - for the group to adopt the members (see `_Refreshed.adopt`) once they are stored.""" - from zarr.core.group import GroupMetadata - - consolidated = metadata.consolidated_metadata - if consolidated is None: - return metadata, [] - groups = { - name: member - for name, member in consolidated.metadata.items() - if isinstance(member, GroupMetadata) - } - arrays = { - name: member - for name, member in consolidated.metadata.items() - if not isinstance(member, GroupMetadata) and member._stored_document_upgraded - } - nested, current = await asyncio.gather( - asyncio.gather(*(_refresh_consolidated(store_path / n, m) for n, m in groups.items())), - asyncio.gather(*(_refresh_array(store_path / n, m) for n, m in arrays.items())), - ) - refreshed = [ - _Refreshed(consolidated.metadata, name, stale, new) - for (name, stale), new in zip(arrays.items(), current, strict=True) - if new is not stale - ] - members = dict(consolidated.metadata) - members.update({member.name: member.current for member in refreshed}) - for name, (group, inner) in zip(groups, nested, strict=True): - members[name] = group - refreshed.extend(inner) - if not refreshed: - return metadata, [] - consolidated = replace(consolidated, metadata=members) - return replace(metadata, consolidated_metadata=consolidated), refreshed + return mark_upgraded(parse_array_metadata(dict(upgraded)), stored, [None] * len(readings), None) def _build_parents(store_path: StorePath, zarr_format: ZarrFormat) -> dict[str, GroupMetadata]: @@ -239,9 +157,7 @@ def _build_parents(store_path: StorePath, zarr_format: ZarrFormat) -> dict[str, async def save_metadata( store_path: StorePath, metadata: ArrayMetadata | GroupMetadata, ensure_parents: bool = False ) -> None: - """Asynchronously save the array or group metadata. A group adopts, in place, the - members of its consolidated metadata that had to be read again to store it (see - `_refresh_consolidated`), once they are stored. + """Asynchronously save the array or group metadata. Parameters ---------- @@ -256,14 +172,8 @@ async def save_metadata( ------ ValueError """ - from zarr.core.group import GroupMetadata - - refreshed: list[_Refreshed] = [] - if isinstance(metadata, GroupMetadata): - # The one place group metadata is stored, and with it consolidated metadata. - metadata, refreshed = await _refresh_consolidated(store_path, metadata) to_save = metadata.to_buffer_dict(default_buffer_prototype()) - set_awaitables = [store_documents(store_path, to_save)] + set_awaitables = [set_or_delete(store_path / key, value) for key, value in to_save.items()] if ensure_parents: # To enable zarr.create(store, path="a/b/c"), we need to create all the intermediate groups. @@ -301,6 +211,3 @@ async def save_metadata( ) from e await asyncio.gather(*set_awaitables) - # Only once stored: a group whose store failed keeps the members flagged, to read again. - for member in refreshed: - member.adopt() diff --git a/src/zarr/core/metadata/upgrades.py b/src/zarr/core/metadata/upgrades.py index 3b09dab496..728a0e53f5 100644 --- a/src/zarr/core/metadata/upgrades.py +++ b/src/zarr/core/metadata/upgrades.py @@ -18,6 +18,7 @@ from __future__ import annotations +import copy import json import warnings from collections.abc import Callable, Iterable, Mapping, Sequence @@ -43,14 +44,17 @@ ) -def mark_upgraded[M](metadata: M, readings: Sequence[str | None], path: str | None) -> M: - """Record that `metadata` was read from a stored document that needed the upgrades - whose `readings` `upgrade_array_document` returned, if any: set - `_stored_document_upgraded` on `metadata`, so the array stores the upgrade before - it writes chunks under it, and warn once with the readings that are warnings, - naming the array at `path` when the caller knows it.""" +def mark_upgraded[M]( + metadata: M, stored: ArrayDocument, readings: Sequence[str | None], path: str | None +) -> M: + """Record that `metadata` was read from the document `stored`, which needed the + upgrades whose `readings` `upgrade_array_document` returned, if any: keep a copy of + `stored` as `_stored_document` on `metadata`, so the array stores the upgrade before + it writes chunks under it and consolidated metadata stores it as it was stored, and + warn once with the readings that are warnings, naming the array at `path` when the + caller knows it.""" if readings: - object.__setattr__(metadata, "_stored_document_upgraded", True) + object.__setattr__(metadata, "_stored_document", copy.deepcopy(stored)) if messages := [reading for reading in readings if reading is not None]: subject = "" if path is None else f"Array {path!r}: " # The synchronous API parses metadata on zarr's IO thread, whose stack holds no diff --git a/src/zarr/core/metadata/v2.py b/src/zarr/core/metadata/v2.py index 7df14a513f..7f6a33e93b 100644 --- a/src/zarr/core/metadata/v2.py +++ b/src/zarr/core/metadata/v2.py @@ -25,6 +25,7 @@ TBaseScalar, ZDType, ) + from zarr.core.metadata.upgrades import ArrayDocument from dataclasses import dataclass, field, fields, replace @@ -72,10 +73,9 @@ class ArrayV2Metadata(Metadata): compressor: Numcodec | None attributes: dict[str, JSON] = field(default_factory=dict) zarr_format: Literal[2] = field(init=False, default=2) - _stored_document_upgraded: ClassVar[bool] = False - """Whether `from_dict` read this metadata from a stored document it had to upgrade - (set on the instance by `mark_upgraded`), so the store may still hold that invalid - document.""" + _stored_document: ClassVar[ArrayDocument | None] = None + """The stored document `from_dict` read this metadata from, if it had to upgrade it + (set on the instance by `mark_upgraded`): the store may still hold it.""" def __init__( self, @@ -209,7 +209,7 @@ def from_dict(cls, data: dict[str, Any], *, path: str | None = None) -> ArrayV2M _data = {k: v for k, v in _data.items() if k in expected} - return mark_upgraded(cls(**_data), readings, path) + return mark_upgraded(cls(**_data), data, readings, path) def to_dict(self) -> dict[str, JSON]: zarray_dict = super().to_dict() diff --git a/src/zarr/core/metadata/v3.py b/src/zarr/core/metadata/v3.py index 17e88d17e4..b654c36978 100644 --- a/src/zarr/core/metadata/v3.py +++ b/src/zarr/core/metadata/v3.py @@ -52,6 +52,7 @@ from zarr.core.buffer import Buffer, BufferPrototype from zarr.core.chunk_grids import ChunkGrid from zarr.core.dtype.wrapper import TBaseDType, TBaseScalar + from zarr.core.metadata.upgrades import ArrayDocument def parse_zarr_format(data: object) -> Literal[3]: @@ -472,10 +473,9 @@ class ArrayV3Metadata(Metadata): node_type: Literal["array"] = field(default="array", init=False) storage_transformers: tuple[dict[str, JSON], ...] extra_fields: dict[str, AllowedExtraField] - _stored_document_upgraded: ClassVar[bool] = False - """Whether `from_dict` read this metadata from a stored document it had to upgrade - (set on the instance by `mark_upgraded`), so the store may still hold that invalid - document.""" + _stored_document: ClassVar[ArrayDocument | None] = None + """The stored document `from_dict` read this metadata from, if it had to upgrade it + (set on the instance by `mark_upgraded`): the store may still hold it.""" def __init__( self, @@ -676,7 +676,7 @@ def from_dict(cls, data: dict[str, JSON], *, path: str | None = None) -> Self: extra_fields=allowed_extra_fields, storage_transformers=_data_typed.get("storage_transformers", ()), # type: ignore[arg-type] ) - return mark_upgraded(metadata, readings, path) + return mark_upgraded(metadata, data, readings, path) def to_dict(self) -> dict[str, JSON]: out_dict = super().to_dict() diff --git a/tests/test_array_stateful.py b/tests/test_array_stateful.py index 109d65eec1..5319dd6f89 100644 --- a/tests/test_array_stateful.py +++ b/tests/test_array_stateful.py @@ -172,7 +172,7 @@ def _open(self) -> zarr.Array[Any]: warned = any(issubclass(w.category, ZarrUserWarning) for w in record) must_warn = any(self.shape[axis] > 0 for axis in self.legacy_axes) assert warned is must_warn, [str(w.message) for w in record] - assert arr.metadata._stored_document_upgraded is bool(self.legacy_axes) + assert (arr.metadata._stored_document is not None) is bool(self.legacy_axes) return arr # ----------------------------------------------------------------- model diff --git a/tests/test_metadata/test_upgrades.py b/tests/test_metadata/test_upgrades.py index c2b76b12aa..3274351de5 100644 --- a/tests/test_metadata/test_upgrades.py +++ b/tests/test_metadata/test_upgrades.py @@ -195,7 +195,7 @@ def test_upgrade_array_document( warnings.simplefilter("always") metadata = metadata_cls.from_dict(dict(doc), path="group/array") assert _chunk_shapes(metadata) == expected - assert metadata._stored_document_upgraded is upgraded + assert metadata._stored_document == (doc if upgraded else None) messages = [str(w.message) for w in record] if warning is None: assert messages == [] @@ -312,7 +312,7 @@ def test_read_invalid_edges_in_rectilinear_grid( with zarr.config.set({"array.rectilinear_chunks": True}): metadata = _read_strictly(doc) assert metadata.chunk_grid == RectilinearChunkGridMetadata(chunk_shapes=expected) - assert metadata._stored_document_upgraded is upgraded + assert metadata._stored_document == (doc if upgraded else None) @pytest.mark.parametrize( @@ -660,7 +660,7 @@ async def write_twice() -> None: sync(write_twice()) - assert not arr.metadata._stored_document_upgraded + assert arr.metadata._stored_document is None with warnings.catch_warnings(): warnings.simplefilter("error", ZarrUserWarning) reopened = zarr.open_array(store=path, mode="r") @@ -700,7 +700,7 @@ def test_stale_handle_write_keeps_newer_metadata( stale[0] = 9 - assert not stale.metadata._stored_document_upgraded + assert stale.metadata._stored_document is None reopened = _open_strictly(path) assert reopened.shape == (9,) assert reopened.attrs.asdict() == {"x": 1} @@ -776,7 +776,7 @@ def test_stale_handle_write_after_chunk_grid_change_raises( with pytest.raises(ValueError, match="has changed since this array was opened: reopen"): stale[0:3] = [7, 8, 9] - assert stale.metadata._stored_document_upgraded + assert stale.metadata._stored_document is not None assert {p.name: p.read_bytes() for p in path.iterdir()} == documents @@ -788,11 +788,11 @@ def test_write_without_stored_document(zarr_format: Literal[2, 3]) -> None: store = MemoryStore() doc = _v2_doc([3], [True]) if zarr_format == 2 else _v3_doc([3], [True]) array = zarr.Array(AsyncArray.from_dict(StorePath(store), doc)) - upgraded = array.metadata._stored_document_upgraded + upgraded = array.metadata._stored_document is not None array[:] = [1, 2, 3] - assert (upgraded, array.metadata._stored_document_upgraded) == (True, False) + assert (upgraded, array.metadata._stored_document) == (True, None) np.testing.assert_array_equal(array[:], [1, 2, 3]) assert not [key for key in store._store_dict if key.endswith((".zarray", "zarr.json"))] @@ -867,128 +867,144 @@ def _flagged_consolidated_group(path: Path, zarr_format: Literal[2, 3], member: _rewrite_consolidated(path, zarr_format, member, _store_zero) +def _record_store_access(monkeypatch: pytest.MonkeyPatch) -> list[tuple[str, str]]: + """Record every `get` and `set` of a `LocalStore`, as `(method, key)`.""" + accesses: list[tuple[str, str]] = [] + for method in ("get", "set"): + original = getattr(LocalStore, method) + + async def recording( + self: LocalStore, + key: str, + *args: Any, + _method: str = method, + _original: Any = original, + **kwargs: Any, + ) -> Any: + accesses.append((_method, key)) + return await _original(self, key, *args, **kwargs) + + monkeypatch.setattr(LocalStore, method, recording) + return accesses + + @pytest.mark.filterwarnings("ignore:Consolidated metadata is currently not part:UserWarning") @pytest.mark.parametrize("zarr_format", [2, 3]) @pytest.mark.parametrize("member", ["a", "g/a"]) @pytest.mark.parametrize("operation", ["attrs", "update_attributes_async", "delete-member"]) -def test_group_write_refreshes_upgraded_consolidated_member( +def test_group_write_stores_upgraded_consolidated_member_as_stored( tmp_path: Path, monkeypatch: pytest.MonkeyPatch, zarr_format: Literal[2, 3], member: str, operation: str, ) -> None: - """Storing a group's metadata stores its consolidated metadata. Each member of it read - from a document that had to be upgraded, at any depth, is first read again from the - member's own document as it now is (here resized by software that kept the stored - chunk size of 0), whose upgrade is stored, so no group write stores a stale copy as - if valid. The group adopts the members it stored, so later writes read no member.""" + """A group write stores only the group's own documents, reading none: it reads and + writes no member document, and stores the consolidated copy of a member read from a + document that had to be upgraded exactly as it was stored, so every reader of the + consolidated metadata reads that member as upgraded again.""" path = tmp_path / "group.zarr" _flagged_consolidated_group(path, zarr_format, member) - - def resize_keeping_zero(doc: dict[str, Any]) -> None: - _store_zero(doc) - doc["shape"] = [10] - - _rewrite_doc(path / member, zarr_format, resize_keeping_zero) + stored = json.dumps(_consolidated_member(path, zarr_format, member)) with pytest.warns(ZarrUserWarning, match="is read as"): group = zarr.open_group(path, mode="r+", use_consolidated=True) + accesses = _record_store_access(monkeypatch) if operation == "attrs": group.attrs["x"] = 1 elif operation == "update_attributes_async": - group = sync(group.update_attributes_async({"x": 1})) + sync(group.update_attributes_async({"x": 1})) else: del group["b"] - assert _stored_chunks(_consolidated_member(path, zarr_format, member)) == [10] - reopened = zarr.open_group(path, mode="r", use_consolidated=True) - array = reopened[member] - assert isinstance(array, zarr.Array) - assert (array.shape, array.chunks) == ((10,), (10,)) - assert _open_strictly(path / member).chunks == (10,) - - reads: list[str] = [] - get = LocalStore.get - - async def recording_get(self: LocalStore, key: str, *args: Any, **kwargs: Any) -> Any: - reads.append(key) - return await get(self, key, *args, **kwargs) - - monkeypatch.setattr(LocalStore, "get", recording_get) - group.attrs["y"] = 2 - assert reads == [] + own = {"zarr.json"} if zarr_format == 3 else {".zgroup", ".zattrs", ".zmetadata"} + assert {method for method, _ in accesses} == {"set"} + assert {key for _, key in accesses} == own + assert json.dumps(_consolidated_member(path, zarr_format, member)) == stored + with pytest.warns(ZarrUserWarning, match="is read as"): + zarr.open_group(path, mode="r", use_consolidated=True) @pytest.mark.filterwarnings("ignore:Consolidated metadata is currently not part:UserWarning") @pytest.mark.parametrize("zarr_format", [2, 3]) -def test_group_write_keeps_consolidated_metadata_shared_with_subgroups( +def test_concurrent_deletions_leave_no_upgraded_member( tmp_path: Path, zarr_format: Literal[2, 3] ) -> None: - """A subgroup handle taken from a group shares the group's consolidated metadata, also - once the group has adopted the upgraded members its first write stored: an array - deleted through the subgroup is gone from what the group stores next.""" + """Members deleted concurrently through one consolidated group handle stay deleted, + also those read from documents that had to be upgraded: no group write stores a + member document.""" path = tmp_path / "group.zarr" - created = zarr.open_group(path, mode="w", zarr_format=zarr_format).create_group("g") - for name in ("a", "c"): - created.create_array(name, shape=(3,), chunks=(3,), dtype="int16") - zarr.consolidate_metadata(path) - _rewrite_consolidated(path, zarr_format, "g/a", _store_zero) + zarr.open_group(path, mode="w", zarr_format=zarr_format) + for name in ("a", "b"): + _legacy_array(path / name, zarr_format) + with pytest.warns(ZarrUserWarning, match="is read as"): + zarr.consolidate_metadata(path) with pytest.warns(ZarrUserWarning, match="is read as"): group = zarr.open_group(path, mode="r+", use_consolidated=True) - subgroup = group["g"] - assert isinstance(subgroup, zarr.Group) - group.attrs["x"] = 1 - del subgroup["c"] - group.attrs["y"] = 2 + async def delete_both() -> None: + await asyncio.gather(*(group._async_group.delitem(name) for name in ("a", "b"))) + + sync(delete_both()) + + assert not (path / "a").exists() + assert not (path / "b").exists() + - reopened = zarr.open_group(path, mode="r", use_consolidated=True) - assert sorted(dict(reopened.members(max_depth=None))) == ["g", "g/a"] +def _consolidated_legacy_member(path: Path, zarr_format: Literal[2, 3]) -> Path: + """A group whose array `a` is stored with chunk shape `[0]`, consolidated; the path + of the array's own document.""" + zarr.open_group(path, mode="w", zarr_format=zarr_format) + _legacy_array(path / "a", zarr_format) + with pytest.warns(ZarrUserWarning, match="is read as"): + zarr.consolidate_metadata(path) + return path / "a" / (".zarray" if zarr_format == 2 else "zarr.json") @pytest.mark.filterwarnings("ignore:Consolidated metadata is currently not part:UserWarning") -def test_group_write_refuses_upgraded_member_read_as_rectilinear(tmp_path: Path) -> None: - """A member of consolidated metadata read from a document that had to be upgraded, - whose own document now declares a rectilinear chunk grid, is read again only with the - rectilinear chunks flag: without it, a group write raises and stores nothing.""" +@pytest.mark.parametrize("zarr_format", [2, 3]) +def test_consolidated_upgraded_member_stored_by_its_first_write( + tmp_path: Path, zarr_format: Literal[2, 3] +) -> None: + """Consolidating a group copies the document of a member that had to be upgraded as + it is stored, and leaves that document as it is. The first chunk write through the + consolidated metadata stores the member's upgrade before its chunks, the one write + that stores it; consolidating again then copies the upgrade.""" path = tmp_path / "group.zarr" - _flagged_consolidated_group(path, 3, "a") - with pytest.warns(ZarrUserWarning, match="is read as"): - group = zarr.open_group(path, mode="r+", use_consolidated=True) - (path / "a" / "zarr.json").write_text(json.dumps(_rectilinear_doc([3], [[1, 2]]))) - documents = {p: p.read_bytes() for p in path.rglob("*") if p.is_file()} + document = _consolidated_legacy_member(path, zarr_format) + legacy = document.read_bytes() + assert json.dumps(_consolidated_member(path, zarr_format, "a")).encode() == legacy - with pytest.raises(ValueError, match="Rectilinear chunk grids are experimental"): - group.attrs["x"] = 1 + with pytest.warns(ZarrUserWarning, match="is read as"): + array = zarr.open_group(path, mode="r+", use_consolidated=True)["a"] + assert isinstance(array, zarr.Array) + array[:] = [7, 8, 9] - assert {p: p.read_bytes() for p in path.rglob("*") if p.is_file()} == documents + np.testing.assert_array_equal(_open_strictly(path / "a")[...], [7, 8, 9]) + zarr.consolidate_metadata(path) + assert _stored_chunks(_consolidated_member(path, zarr_format, "a")) == [3] @pytest.mark.filterwarnings("ignore:Consolidated metadata is currently not part:UserWarning") @pytest.mark.parametrize("zarr_format", [2, 3]) -@pytest.mark.parametrize("replacement", ["group", "invalid", "none"]) -def test_group_write_keeps_upgraded_member_without_readable_document( - tmp_path: Path, zarr_format: Literal[2, 3], replacement: str +def test_consolidated_upgraded_member_write_after_chunk_grid_change_raises( + tmp_path: Path, zarr_format: Literal[2, 3] ) -> None: - """A member of consolidated metadata read from a document that had to be upgraded, - whose own document is no longer an array document that can be read, has no document - to upgrade: a group write stores the consolidated copy as it is.""" + """A member read from its consolidated copy, whose own document has since been + resized by software that kept the stored chunk size of 0, would write chunks no + reader finds: its first write raises and stores nothing.""" path = tmp_path / "group.zarr" - _flagged_consolidated_group(path, zarr_format, "a") + _consolidated_legacy_member(path, zarr_format) with pytest.warns(ZarrUserWarning, match="is read as"): - group = zarr.open_group(path, mode="r+", use_consolidated=True) - del zarr.open_group(path, mode="r+", use_consolidated=False)["a"] - if replacement == "group": - zarr.create_group(path / "a", zarr_format=zarr_format) - elif replacement == "invalid": - (path / "a").mkdir() - document = path / "a" / (".zarray" if zarr_format == 2 else "zarr.json") - document.write_text(json.dumps({"zarr_format": zarr_format, "node_type": "array"})) + array = zarr.open_group(path, mode="r+", use_consolidated=True)["a"] + assert isinstance(array, zarr.Array) + _rewrite_doc(path / "a", zarr_format, _resize_to_10) + documents = {p: p.read_bytes() for p in path.rglob("*") if p.is_file()} - group.attrs["x"] = 1 + with pytest.raises(ValueError, match="has changed since this array was opened: reopen"): + array[0:3] = [7, 8, 9] - assert _stored_chunks(_consolidated_member(path, zarr_format, "a")) == [3] + assert {p: p.read_bytes() for p in path.rglob("*") if p.is_file()} == documents @pytest.mark.parametrize("zarr_format", [2, 3]) @@ -1002,7 +1018,7 @@ def test_empty_write_stores_no_metadata(tmp_path: Path, zarr_format: Literal[2, array[0:0] = np.empty(0, dtype="int16") - assert array.metadata._stored_document_upgraded + assert array.metadata._stored_document is not None assert {p.name: p.read_bytes() for p in path.iterdir()} == documents @@ -1015,27 +1031,6 @@ def test_async_array_from_dict_names_array(tmp_path: Path, zarr_format: Literal[ AsyncArray.from_dict(store_path, doc) -@pytest.mark.filterwarnings("ignore:Consolidated metadata is currently not part:UserWarning") -@pytest.mark.parametrize("zarr_format", [2, 3]) -def test_consolidate_stores_upgraded_members(tmp_path: Path, zarr_format: Literal[2, 3]) -> None: - """Consolidating a group stores the upgrade of every member document that needed - one before the consolidated document, so chunks written through the consolidated - metadata are stored under a member document that agrees with it.""" - path = tmp_path / "group.zarr" - zarr.open_group(path, mode="w", zarr_format=zarr_format) - _legacy_array(path / "a", zarr_format) - with pytest.warns(ZarrUserWarning, match="is read as"): - zarr.consolidate_metadata(path) - - member = _open_strictly(path / "a") - assert member.chunks == (3,) - group = zarr.open_group(path, mode="r+", use_consolidated=True) - array = group["a"] - assert isinstance(array, zarr.Array) - array[:] = [7, 8, 9] - np.testing.assert_array_equal(_open_strictly(path / "a")[...], [7, 8, 9]) - - @pytest.mark.filterwarnings("ignore:Consolidated metadata is currently not part:UserWarning") @pytest.mark.parametrize("zarr_format", [2, 3]) def test_legacy_chunk_size_consolidated(tmp_path: Path, zarr_format: Literal[2, 3]) -> None: From 65f3f8ee7abeaa104bc60b02eeac0d79992a651f Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Sat, 26 Sep 2026 19:34:20 +0200 Subject: [PATCH 39/41] docs(group): say what consolidated metadata stores for an upgraded array Assisted-by: ClaudeCode:claude-opus-5-5 Co-Authored-By: Claude Opus 5.5 --- src/zarr/core/group.py | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/src/zarr/core/group.py b/src/zarr/core/group.py index 4f60856425..d76106b1cd 100644 --- a/src/zarr/core/group.py +++ b/src/zarr/core/group.py @@ -148,8 +148,8 @@ class ConsolidatedMetadata: def to_dict(self) -> dict[str, JSON]: """The consolidated metadata document. An array read from a stored document that - had to be upgraded is written as that document was stored: only the array's - own first chunk write stores its upgrade.""" + had to be upgraded is written as that document was stored, so every reader of + the consolidated metadata reads it as upgraded again (see `mark_upgraded`).""" return { "kind": self.kind, "must_understand": self.must_understand, From faa6efe0af1165a57ca3764a7e9710ea52571360 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Sat, 26 Sep 2026 20:23:02 +0200 Subject: [PATCH 40/41] fix(array): clear the stored document whenever an array stores its metadata Setting attributes of an array read from a document that had to be upgraded stored the upgraded document with the new attributes, but left the metadata marked with the document it was read from. A consolidated group handle shares that metadata, so its next write stored the old document in the consolidated metadata and the new attributes were lost from it (zarr 3.4.0 kept them). Every write of an array's own documents (creating, resizing, setting attributes) now goes through `AsyncArray._save_metadata`, which clears the mark afterwards, as the first chunk write does after storing the upgrade. Both clear it through `_stored_document_replaced`, the one place that declares it: the first chunk write clears it also when it stores nothing (the store already holds a valid document, or none), so it cannot be folded into the save. The deep copy in `mark_upgraded` stays: the metadata's attributes share objects with the document it was read from, which the caller also holds. A test pins it. Assisted-by: ClaudeCode:claude-opus-5-5 Co-Authored-By: Claude Opus 5.5 --- changes/4334.bugfix.md | 2 +- src/zarr/core/array.py | 20 +++++++++---- tests/test_metadata/test_upgrades.py | 43 ++++++++++++++++++++++++++++ 3 files changed, 58 insertions(+), 7 deletions(-) diff --git a/changes/4334.bugfix.md b/changes/4334.bugfix.md index 6279623347..ea7ec480ce 100644 --- a/changes/4334.bugfix.md +++ b/changes/4334.bugfix.md @@ -4,4 +4,4 @@ A chunk edge length is now always at least 1, while an array extent may be 0. Ev Stored metadata with a regular chunk size of 0 or JSON `false` (Zarr format 2 `chunks`, Zarr format 3 `regular` `chunk_shape`) is now read as one chunk spanning the axis (a multiple of the inner chunk size for sharded arrays, which previously failed to open). zarr-python wrote such sizes for arrays created with a zero-length axis until 3.4, and, from 2.18.7 to 3.2.1, for an explicit chunk size of 0 or `False` on an axis of any length, as in `chunks=(0,)` (those arrays could store no data). On a zero-length axis this opens silently, as before; appending to such an axis then stores chunks of size 1. On an axis of positive length it opens with a `ZarrUserWarning`, which says that the axis holds only the fill value and how to store valid metadata: `array.update_attributes({})`, then `zarr.consolidate_metadata` if the metadata is consolidated. JSON `true`, as zarr-python 3.0 and 3.2 wrote for a chunk size of `True`, is read as 1, silently, in the chunk grid (regular, and the explicit edges and run-length encoded sizes of a rectilinear one) and in the inner chunk shape of every sharding codec, nested or not. A stored rectilinear chunk grid whose edge lengths are integral JSON floats (`[[4.0, 2]]`), as zarr-python 3.2 wrote for float edges, is read with those edges as integers, silently; a float anywhere no release wrote one (a regular chunk shape, the inner chunk shape of a sharding codec, a run-length repeat count, an edge below 1) is rejected. -Writing data to an array read from such metadata (other than an empty selection) first stores the upgrade of the metadata the store then holds, unless another writer has stored valid metadata since, so that other readers find the chunks written; in a `ZipStore` this adds a second entry for the metadata document, as every metadata update does. If the metadata the store then holds lays out chunks differently (another writer resized the array keeping the chunk size of 0, say), the write raises a `ValueError` asking to reopen the array, and stores nothing. This is the only operation that stores the upgrade: a group's consolidated metadata keeps such an array's metadata as it was stored (`zarr.consolidate_metadata` copies it as the array's own document holds it), and changing a group reads and writes no metadata of its members. +Writing data to an array read from such metadata (other than an empty selection) first stores the upgrade of the metadata the store then holds, unless another writer has stored valid metadata since, so that other readers find the chunks written; in a `ZipStore` this adds a second entry for the metadata document, as every metadata update does. If the metadata the store then holds lays out chunks differently (another writer resized the array keeping the chunk size of 0, say), the write raises a `ValueError` asking to reopen the array, and stores nothing. Apart from the array's own metadata writes (`update_attributes`, `resize`), which store the upgrade as before, this is the only operation that stores it: a group's consolidated metadata keeps such an array's metadata as it was stored (`zarr.consolidate_metadata` copies it as the array's own document holds it) until the array stores its upgrade through the group's handle, and changing a group reads and writes no metadata of its members. diff --git a/src/zarr/core/array.py b/src/zarr/core/array.py index 7b23f3b917..cf6420e5b7 100644 --- a/src/zarr/core/array.py +++ b/src/zarr/core/array.py @@ -1625,10 +1625,18 @@ async def get_coordinate_selection( return out_array async def _save_metadata(self, metadata: ArrayMetadata, ensure_parents: bool = False) -> None: - """ - Asynchronously save the array metadata. - """ + """Store `metadata` as this array's own documents: every write of them + (creating, resizing, setting attributes) goes through here.""" await save_metadata(self.store_path, metadata, ensure_parents=ensure_parents) + self._stored_document_replaced() + + def _stored_document_replaced(self) -> None: + """Record that the store no longer holds a document of this array that needs an + upgrade: it holds the upgrade, a valid document, or none. The metadata this handle + holds, which a consolidated group handle may share, then stops standing for the + document it was read from (see `mark_upgraded`), so no later write through either + handle stores that document again.""" + object.__setattr__(self.metadata, "_stored_document", None) async def _store_upgraded_document(self) -> None: """Store the upgrade of this array's current stored document, if it needs one, @@ -1660,7 +1668,7 @@ async def _store_upgraded_document(self) -> None: ) if current._stored_document is not None: await upsert_metadata(self.store_path, current, documents) - object.__setattr__(self.metadata, "_stored_document", None) + self._stored_document_replaced() async def _set_selection( self, @@ -5931,7 +5939,7 @@ async def _delete_key(key: str) -> None: ) # Write new metadata - await save_metadata(array.store_path, new_metadata) + await array._save_metadata(new_metadata) # Update metadata and chunk_grid (in place) object.__setattr__(array, "metadata", new_metadata) @@ -6023,7 +6031,7 @@ async def _update_attributes( array.metadata.attributes.update(new_attributes) # Write new metadata - await save_metadata(array.store_path, array.metadata) + await array._save_metadata(array.metadata) return array diff --git a/tests/test_metadata/test_upgrades.py b/tests/test_metadata/test_upgrades.py index 3274351de5..2698a5f198 100644 --- a/tests/test_metadata/test_upgrades.py +++ b/tests/test_metadata/test_upgrades.py @@ -16,6 +16,7 @@ from zarr.codecs import ShardingCodec from zarr.codecs.numcodecs import Quantize from zarr.core.array import AsyncArray +from zarr.core.group import ConsolidatedMetadata from zarr.core.metadata import ArrayV2Metadata, ArrayV3Metadata from zarr.core.metadata.upgrades import ( RESAVE_HINT, @@ -1007,6 +1008,48 @@ def test_consolidated_upgraded_member_write_after_chunk_grid_change_raises( assert {p: p.read_bytes() for p in path.rglob("*") if p.is_file()} == documents +@pytest.mark.filterwarnings("ignore:Consolidated metadata is currently not part:UserWarning") +@pytest.mark.parametrize("zarr_format", [2, 3]) +def test_consolidated_upgraded_member_attributes_kept_by_group_write( + tmp_path: Path, zarr_format: Literal[2, 3] +) -> None: + """Setting attributes of a member read from consolidated metadata stores the member's + upgraded document with them. A later write of the group, whose consolidated metadata + shares the member's metadata, then stores that upgrade: the new attributes, not the + legacy document as it was stored before.""" + path = tmp_path / "group.zarr" + _consolidated_legacy_member(path, zarr_format) + with pytest.warns(ZarrUserWarning, match="is read as"): + group = zarr.open_group(path, mode="r+", use_consolidated=True) + + group["a"].attrs["x"] = 1 + group.attrs["y"] = 2 + + with warnings.catch_warnings(record=True) as record: + warnings.simplefilter("always") + attributes = dict(zarr.open_group(path, mode="r", use_consolidated=True)["a"].attrs) + assert attributes == {"x": 1} + assert _stored_chunks(_consolidated_member(path, zarr_format, "a")) == [3] + assert not [w for w in record if "is read as" in str(w.message)] + + +@pytest.mark.parametrize("zarr_format", [2, 3]) +def test_upgraded_metadata_keeps_the_document_it_read(zarr_format: Literal[2, 3]) -> None: + """Metadata read from a document that had to be upgraded keeps that document as it + was read, whatever becomes of the caller's dict (or the objects in it, which the + metadata's attributes may hold): consolidated metadata stores it as read.""" + doc = _v2_doc([0], [0]) if zarr_format == 2 else _v3_doc([0], [0]) + doc["attributes"] = {"k": [1]} + read = json.loads(json.dumps(doc)) + metadata_cls = ArrayV2Metadata if zarr_format == 2 else ArrayV3Metadata + metadata = metadata_cls.from_dict(doc) + + cast("list[int]", metadata.attributes["k"]).append(2) + doc["shape"] = [4] + + assert ConsolidatedMetadata(metadata={"a": metadata}).to_dict()["metadata"] == {"a": read} + + @pytest.mark.parametrize("zarr_format", [2, 3]) def test_empty_write_stores_no_metadata(tmp_path: Path, zarr_format: Literal[2, 3]) -> None: """A write of an empty selection stores no chunks, so it stores no metadata either.""" From 30d7ca6282e1f2aede86ecc7d5553f85ad5edef2 Mon Sep 17 00:00:00 2001 From: Davis Vann Bennett Date: Sat, 26 Sep 2026 20:49:47 +0200 Subject: [PATCH 41/41] test(metadata): pin that a failed metadata save keeps the stored document mark Setting attributes on or resizing an array read from a document that needed an upgrade clears the `_stored_document` mark only after the store accepts the new metadata. A test now injects a store failure for the array's own document and checks the mark and the stored bytes survive, for Zarr formats 2 and 3. The `_save_metadata` docstring now says only what the method does. Assisted-by: ClaudeCode:claude-opus-5-5 Co-Authored-By: Claude Opus 5.5 --- src/zarr/core/array.py | 4 +-- tests/test_metadata/test_upgrades.py | 40 ++++++++++++++++++++++++++++ 2 files changed, 42 insertions(+), 2 deletions(-) diff --git a/src/zarr/core/array.py b/src/zarr/core/array.py index cf6420e5b7..e402382a33 100644 --- a/src/zarr/core/array.py +++ b/src/zarr/core/array.py @@ -1625,8 +1625,8 @@ async def get_coordinate_selection( return out_array async def _save_metadata(self, metadata: ArrayMetadata, ensure_parents: bool = False) -> None: - """Store `metadata` as this array's own documents: every write of them - (creating, resizing, setting attributes) goes through here.""" + """Store `metadata` as this array's own documents, then clear the + `_stored_document` mark (see `_stored_document_replaced`).""" await save_metadata(self.store_path, metadata, ensure_parents=ensure_parents) self._stored_document_replaced() diff --git a/tests/test_metadata/test_upgrades.py b/tests/test_metadata/test_upgrades.py index 2698a5f198..cf33aa2f15 100644 --- a/tests/test_metadata/test_upgrades.py +++ b/tests/test_metadata/test_upgrades.py @@ -781,6 +781,46 @@ def test_stale_handle_write_after_chunk_grid_change_raises( assert {p.name: p.read_bytes() for p in path.iterdir()} == documents +def _set_attribute(array: AnyArray) -> None: + array.attrs["x"] = 1 + + +def _grow(array: AnyArray) -> None: + array.resize((9,)) + + +@pytest.mark.parametrize("operation", [_set_attribute, _grow], ids=["attrs", "resize"]) +@pytest.mark.parametrize("zarr_format", [2, 3]) +def test_failed_metadata_save_keeps_stored_document( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, + zarr_format: Literal[2, 3], + operation: Callable[[AnyArray], None], +) -> None: + """If storing an array's metadata fails, the array still stands for the document + the store holds, which still needs its upgrade.""" + path = tmp_path / "legacy.zarr" + _legacy_array(path, zarr_format) + doc_name = ".zarray" if zarr_format == 2 else "zarr.json" + with pytest.warns(ZarrUserWarning, match="is read as"): + arr = zarr.open_array(store=path, mode="r+") + stored = (path / doc_name).read_bytes() + original_set = LocalStore.set + + async def failing_set(self: LocalStore, key: str, *args: Any, **kwargs: Any) -> None: + if key == doc_name: + raise OSError(f"cannot store {key}") + await original_set(self, key, *args, **kwargs) + + monkeypatch.setattr(LocalStore, "set", failing_set) + + with pytest.raises(OSError, match=f"cannot store {re.escape(doc_name)}"): + operation(arr) + + assert arr.metadata._stored_document is not None + assert (path / doc_name).read_bytes() == stored + + @pytest.mark.parametrize("zarr_format", [2, 3]) def test_write_without_stored_document(zarr_format: Literal[2, 3]) -> None: """An array read from an upgraded document that no store holds (as