Skip to content

Commit 4930ed5

Browse files
committed
feat(codecs): deprecate the host-byte-order default of BytesCodec on big-endian hosts
`BytesCodec()` without an `endian` argument means `sys.byteorder`, so the same code writes big-endian chunks on s390x and little-endian chunks elsewhere. The default will become "little" on every host, which is what the codec originally defaulted to before #1660 and what the default serializer uses. On big-endian hosts, omitting `endian` now raises a ZarrFutureWarning; nothing changes on little-endian hosts. Callers inside zarr (`zarr.testing` strategies, docstrings) and the tests now pass `endian` explicitly, so they do not depend on the default. Refs #3438 Assisted-by: ClaudeCode:claude-opus-5-5
1 parent d2b32d4 commit 4930ed5

17 files changed

Lines changed: 154 additions & 76 deletions

‎changes/PR_NUMBER.removal.md‎

Lines changed: 1 addition & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1 @@
1+
On big-endian hosts, constructing `BytesCodec()` without an `endian` argument is deprecated and raises a `ZarrFutureWarning`. The omitted `endian` currently means the host byte order. A future release will change it to `"little"` on every host, so that stored bytes do not depend on the machine that wrote them. To keep the current behavior, pass `endian="big"`; to adopt the new default now, pass `endian="little"`. Nothing changes on little-endian hosts, where the default is already `"little"`. `BytesCodec.from_dict` and zarr's default serializer are unaffected.

‎src/zarr/api/asynchronous.py‎

Lines changed: 1 addition & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -1024,7 +1024,7 @@ async def create(
10241024
Zarr format 3 only. Zarr format 2 arrays should use `filters` and `compressor` instead.
10251025
10261026
If no codecs are provided, default codecs will be used based on the data type of the array.
1027-
For most data types, the default codecs are the tuple `(BytesCodec(), ZstdCodec())`;
1027+
For most data types, the default codecs are the tuple `(BytesCodec(endian="little"), ZstdCodec())`;
10281028
data types that require a special [`zarr.abc.codec.ArrayBytesCodec`][], like variable-length strings or bytes,
10291029
will use the [`zarr.abc.codec.ArrayBytesCodec`][] required for the data type instead of [`zarr.codecs.BytesCodec`][].
10301030
dimension_names : Iterable[str | None] | None = None

‎src/zarr/api/synchronous.py‎

Lines changed: 1 addition & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -762,7 +762,7 @@ def create(
762762
Zarr format 3 only. Zarr format 2 arrays should use `filters` and `compressor` instead.
763763
764764
If no codecs are provided, default codecs will be used based on the data type of the array.
765-
For most data types, the default codecs are the tuple `(BytesCodec(), ZstdCodec())`;
765+
For most data types, the default codecs are the tuple `(BytesCodec(endian="little"), ZstdCodec())`;
766766
data types that require a special [`zarr.abc.codec.ArrayBytesCodec`][], like variable-length strings or bytes,
767767
will use the [`zarr.abc.codec.ArrayBytesCodec`][] required for the data type instead of [`zarr.codecs.BytesCodec`][].
768768
dimension_names : Iterable[str | None] | None = None

‎src/zarr/codecs/bytes.py‎

Lines changed: 39 additions & 2 deletions
Original file line numberDiff line numberDiff line change
@@ -3,13 +3,15 @@
33
import sys
44
import warnings
55
from dataclasses import dataclass, replace
6+
from enum import Enum
67
from typing import TYPE_CHECKING, ClassVar, Final, Literal
78

89
from zarr.abc.codec import ArrayBytesCodec
910
from zarr.codecs._deprecated_enum import _coerce_enum_input, _DeprecatedStrEnumMeta
1011
from zarr.core.common import JSON, parse_named_configuration
1112
from zarr.core.dtype.common import HasEndianness
1213
from zarr.core.dtype.npy.structured import Struct
14+
from zarr.errors import ZarrFutureWarning
1315

1416
if TYPE_CHECKING:
1517
from typing import Self
@@ -33,6 +35,30 @@ class Endian(metaclass=_DeprecatedStrEnumMeta):
3335
_members: ClassVar[dict[str, str]] = {"little": "little", "big": "big"}
3436

3537

38+
class _HostEndian(Enum):
39+
"""Marks an omitted `endian` argument, which currently means the host byte order."""
40+
41+
token = 0
42+
43+
44+
def _resolve_host_endian() -> EndianLiteral:
45+
"""The byte order an omitted `endian` argument stands for, warning where it will change.
46+
47+
The default will become `"little"` on every host, so only big-endian hosts are affected.
48+
"""
49+
if sys.byteorder == "big":
50+
warnings.warn(
51+
"BytesCodec() without an `endian` argument stores chunks in the byte order of the "
52+
"host, which is big-endian here. A future version of Zarr Python will default to "
53+
"endian='little' on every host, so that stored bytes do not depend on the machine "
54+
"that wrote them. Pass endian='big' to keep the current behavior, or "
55+
"endian='little' to adopt the new default now.",
56+
ZarrFutureWarning,
57+
stacklevel=3,
58+
)
59+
return sys.byteorder
60+
61+
3662
def _parse_endian(data: object) -> EndianLiteral:
3763
if isinstance(data, str) and data in ENDIAN:
3864
return data # type: ignore[return-value]
@@ -41,13 +67,24 @@ def _parse_endian(data: object) -> EndianLiteral:
4167

4268
@dataclass(frozen=True)
4369
class BytesCodec(ArrayBytesCodec):
44-
"""bytes codec"""
70+
"""bytes codec
71+
72+
When `endian` is omitted it is the byte order of the host. That default is deprecated on
73+
big-endian hosts: it will become `"little"` on every host, so that stored bytes do not depend
74+
on the machine that wrote them.
75+
"""
4576

4677
is_fixed_size = True
4778

4879
endian: EndianLiteral | None
4980

50-
def __init__(self, *, endian: Endian | EndianLiteral | None = sys.byteorder) -> None:
81+
def __init__(
82+
self,
83+
*,
84+
endian: Endian | EndianLiteral | Literal[_HostEndian.token] | None = _HostEndian.token,
85+
) -> None:
86+
if endian is _HostEndian.token:
87+
endian = _resolve_host_endian()
5188
if endian is None:
5289
endian_parsed: EndianLiteral | None = None
5390
else:

‎src/zarr/testing/stateful.py‎

Lines changed: 1 addition & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -145,7 +145,7 @@ def add_array(self, data: DataObject, name: str) -> None:
145145
paths=st.just(parent),
146146
array_names=st.just(name),
147147
zarr_formats=st.just(3),
148-
compressors=st.just(BytesCodec()),
148+
compressors=st.just(BytesCodec(endian="little")),
149149
open_mode="a",
150150
),
151151
label="generated array",

‎src/zarr/testing/strategies.py‎

Lines changed: 4 additions & 4 deletions
Original file line numberDiff line numberDiff line change
@@ -259,8 +259,8 @@ def dimension_names(draw: st.DrawFn, *, ndim: int | None = None) -> list[str | N
259259
# silently disables the fast path under every property test.
260260
sharding_inner_codecs: st.SearchStrategy[list[BytesCodec | ZstdCodec]] = st.sampled_from(
261261
[
262-
[BytesCodec()],
263-
[BytesCodec(), ZstdCodec()],
262+
[BytesCodec(endian="little")],
263+
[BytesCodec(endian="little"), ZstdCodec()],
264264
]
265265
)
266266

@@ -303,7 +303,7 @@ def array_metadata(
303303
attributes=draw(attributes), # type: ignore[arg-type]
304304
dimension_names=draw(dimension_names(ndim=ndim)),
305305
chunk_key_encoding=DefaultChunkKeyEncoding(separator="/"), # FIXME
306-
codecs=[BytesCodec()],
306+
codecs=[BytesCodec(endian="little")],
307307
storage_transformers=(),
308308
)
309309

@@ -386,7 +386,7 @@ def _sharding_codecs(
386386
return ShardingCodec(
387387
subchunk_write_order=subchunk_write_order,
388388
codecs=inner_codecs,
389-
index_codecs=[BytesCodec(), Crc32cCodec()],
389+
index_codecs=[BytesCodec(endian="little"), Crc32cCodec()],
390390
chunk_shape=chunk_shape,
391391
)
392392

‎tests/test_array.py‎

Lines changed: 2 additions & 2 deletions
Original file line numberDiff line numberDiff line change
@@ -342,7 +342,7 @@ def test_storage_transformers(store: MemoryStore, zarr_format: ZarrFormat | str)
342342
"chunk_grid": {"name": "regular", "configuration": {"chunk_shape": (1,)}},
343343
"data_type": "uint8",
344344
"chunk_key_encoding": {"name": "v2", "configuration": {"separator": "/"}},
345-
"codecs": (BytesCodec().to_dict(),),
345+
"codecs": (BytesCodec(endian="little").to_dict(),),
346346
"fill_value": 0,
347347
"storage_transformers": ({"test": "should_raise"}),
348348
}
@@ -353,7 +353,7 @@ def test_storage_transformers(store: MemoryStore, zarr_format: ZarrFormat | str)
353353
"chunks": (1,),
354354
"dtype": "|u1",
355355
"dimension_separator": ".",
356-
"codecs": (BytesCodec().to_dict(),),
356+
"codecs": (BytesCodec(endian="little").to_dict(),),
357357
"fill_value": 0,
358358
"order": "C",
359359
"storage_transformers": ({"test": "should_raise"}),

‎tests/test_chunk_transform.py‎

Lines changed: 12 additions & 12 deletions
Original file line numberDiff line numberDiff line change
@@ -60,9 +60,9 @@ def _make_nd_buffer(arr: np.ndarray[Any, np.dtype[Any]]) -> NDBuffer:
6060
@pytest.mark.parametrize(
6161
("shape", "codecs"),
6262
[
63-
((100,), (BytesCodec(),)),
64-
((100,), (BytesCodec(), GzipCodec())),
65-
((3, 4), (TransposeCodec(order=(1, 0)), BytesCodec(), ZstdCodec())),
63+
((100,), (BytesCodec(endian="little"),)),
64+
((100,), (BytesCodec(endian="little"), GzipCodec())),
65+
((3, 4), (TransposeCodec(order=(1, 0)), BytesCodec(endian="little"), ZstdCodec())),
6666
],
6767
ids=["bytes-only", "with-compression", "full-chain"],
6868
)
@@ -90,14 +90,14 @@ def test_construction_rejects_non_sync(shape: tuple[int, ...], codecs: tuple[Cod
9090
@pytest.mark.parametrize(
9191
("arr", "codecs"),
9292
[
93-
(np.arange(100, dtype="float64"), (BytesCodec(),)),
94-
(np.arange(100, dtype="float64"), (BytesCodec(), GzipCodec(level=1))),
93+
(np.arange(100, dtype="float64"), (BytesCodec(endian="little"),)),
94+
(np.arange(100, dtype="float64"), (BytesCodec(endian="little"), GzipCodec(level=1))),
9595
(
9696
np.arange(12, dtype="float64").reshape(3, 4),
97-
(TransposeCodec(order=(1, 0)), BytesCodec(), ZstdCodec(level=1)),
97+
(TransposeCodec(order=(1, 0)), BytesCodec(endian="little"), ZstdCodec(level=1)),
9898
),
99-
(np.arange(100, dtype="float64"), (BytesCodec(), Crc32cCodec())),
100-
(np.arange(50, dtype="int32"), (BytesCodec(), ZstdCodec(level=1))),
99+
(np.arange(100, dtype="float64"), (BytesCodec(endian="little"), Crc32cCodec())),
100+
(np.arange(50, dtype="int32"), (BytesCodec(endian="little"), ZstdCodec(level=1))),
101101
],
102102
ids=["bytes-only", "gzip", "transpose+zstd", "crc32c", "int32"],
103103
)
@@ -118,9 +118,9 @@ def test_encode_decode_roundtrip(
118118
@pytest.mark.parametrize(
119119
("shape", "codecs", "input_size", "expected_size"),
120120
[
121-
((100,), (BytesCodec(),), 800, 800),
122-
((100,), (BytesCodec(), Crc32cCodec()), 800, 804),
123-
((3, 4), (TransposeCodec(order=(1, 0)), BytesCodec()), 96, 96),
121+
((100,), (BytesCodec(endian="little"),), 800, 800),
122+
((100,), (BytesCodec(endian="little"), Crc32cCodec()), 800, 804),
123+
((3, 4), (TransposeCodec(order=(1, 0)), BytesCodec(endian="little")), 96, 96),
124124
],
125125
ids=["bytes-only", "crc32c", "transpose"],
126126
)
@@ -147,7 +147,7 @@ def _encode_sync(self, chunk_array: NDBuffer, chunk_spec: ArraySpec) -> NDBuffer
147147

148148
spec = _make_array_spec((3, 4), np.dtype("float64"))
149149
chain = ChunkTransform(
150-
codecs=(NoneReturningAACodec(order=(1, 0)), BytesCodec()),
150+
codecs=(NoneReturningAACodec(order=(1, 0)), BytesCodec(endian="little")),
151151
)
152152
arr = np.arange(12, dtype="float64").reshape(3, 4)
153153
nd_buf = _make_nd_buffer(arr)

‎tests/test_codec_pipeline.py‎

Lines changed: 1 addition & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -275,7 +275,7 @@ async def _encode_single(self, chunk_array: Any, chunk_spec: ArraySpec) -> Any:
275275

276276
_CODEC_FACTORY: dict[str, Callable[[], Codec]] = {
277277
_AA: lambda: TransposeCodec(order=(0, 1)),
278-
_AB: BytesCodec,
278+
_AB: lambda: BytesCodec(endian="little"),
279279
_BB: GzipCodec,
280280
}
281281

‎tests/test_codec_pipeline_suite.py‎

Lines changed: 5 additions & 3 deletions
Original file line numberDiff line numberDiff line change
@@ -285,7 +285,7 @@ def _val(n: int, dtype: str, offset: int = 1) -> np.ndarray:
285285
"chunks": (2, 4),
286286
"shards": None,
287287
"filters": [TransposeCodec(order=(1, 0))],
288-
"serializer": BytesCodec(),
288+
"serializer": BytesCodec(endian="little"),
289289
**_I32,
290290
},
291291
writes=((slice(None), np.arange(96, dtype="int32").reshape(8, 12)),),
@@ -298,7 +298,7 @@ def _val(n: int, dtype: str, offset: int = 1) -> np.ndarray:
298298
"chunks": (2, 4),
299299
"shards": None,
300300
"filters": [TransposeCodec(order=(1, 0))],
301-
"serializer": BytesCodec(),
301+
"serializer": BytesCodec(endian="little"),
302302
"compressors": GzipCodec(level=1),
303303
**_I32,
304304
},
@@ -517,7 +517,9 @@ def test_partial_write_after_reopen_is_correct(
517517
compressors=None,
518518
config={"write_empty_chunks": True},
519519
serializer=ShardingCodec(
520-
chunk_shape=inner, codecs=[BytesCodec()], subchunk_write_order=subchunk_write_order
520+
chunk_shape=inner,
521+
codecs=[BytesCodec(endian="little")],
522+
subchunk_write_order=subchunk_write_order,
521523
),
522524
)
523525
ref = np.arange(24, dtype="int32").reshape(shape)

0 commit comments

Comments
 (0)