Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
7 changes: 7 additions & 0 deletions CHANGELOG.md
Original file line number Diff line number Diff line change
Expand Up @@ -2,6 +2,13 @@

## Unreleased
- Move all packages to sibling top-level directories and make the root project workspace-only.
- Support irregular (rectilinear) chunk grids for Zarr v3 arrays. `zcreate`,
`zzeros`, and `ZArray(data; chunks=...)` accept `DiskArrays.GridChunks`, and
rectilinear metadata and chunks interoperate with zarr-python. `resize!` and
`append!` work on rectilinear arrays, except for shrinking an irregular axis
[#332](https://github.com/JuliaIO/Zarr.jl/pull/332), [#326](https://github.com/JuliaIO/Zarr.jl/pull/326).
- `resize!` and `append!` now throw on a read-only `ZArray` instead of rewriting its metadata and deleting chunks.
- `resize!` through a `ConsolidatedStore` no longer deletes chunks or changes the in-memory shape before failing, and Zarr v3 consolidated stores now reject it like v2 ones instead of leaving the consolidated metadata stale.
- Move code to ZarrCore.jl with low dependencies
- Move HTTP, GCS, S3, and ZIP backends into `ZarrHTTP`, `ZarrGCS`, `ZarrS3`, and `ZarrZip`.
- Move Blosc, zlib, and Zstandard support into `ZarrBlosc`, `ZarrZlib`, and `ZarrZstd`; each provides its v2 compressor and v3 codec.
Expand Down
1 change: 1 addition & 0 deletions CLAUDE.md
Original file line number Diff line number Diff line change
Expand Up @@ -139,6 +139,7 @@ Core paths below are relative to `ZarrCore/`. Backend implementations are in

- **Format version dispatch**: `ZarrFormat{V}` selects version-specific behavior
- **Shape is mutable**: `metadata.shape` is `Base.RefValue{NTuple{N,Int}}` to allow `resize!` without replacing the metadata struct
- **V3 chunks are mutable**: `MetadataV3.chunks` is a `Base.RefValue` holding an `NTuple` (regular grid) or a `DiskArrays.GridChunks` (rectilinear grid, which `resize!` replaces); read it through `chunkspec(md)`
- **Column-major ↔ row-major**: reverse dimensions at metadata boundaries
- **Compressor registry**: `compressortypes` maps v2 names to compressor types; `v2_to_v3_codecs` maps compressors to v3 codecs by dispatch
- **DiskArrays integration**: `ZArray <: AbstractDiskArray`, chunk-aware I/O via `readblock!`/`writeblock!`, `eachchunk`, `haschunks`
Expand Down
1 change: 1 addition & 0 deletions Zarr/test/Project.toml
Original file line number Diff line number Diff line change
Expand Up @@ -3,6 +3,7 @@ AWSS3 = "1c724243-ef5b-51ab-93f4-b0a88ac62a95"
CondaPkg = "992eb4ea-22a4-4c89-a5bb-47a3300528ab"
DateTimes64 = "b342263e-b350-472a-b1a9-8dfd21b51589"
Dates = "ade2ca70-3891-5945-98fb-dc099432e06a"
DiskArrays = "3c3547ce-8d99-4f5e-a174-61eb10b00ae3"
JSON = "682c06a0-de6a-54ab-a142-c8b1cf79cde6"
Minio = "4281f0d9-7ae0-406e-9172-b7277c1efa20"
Mmap = "a63ad114-7e13-5084-954f-fe012c677804"
Expand Down
30 changes: 29 additions & 1 deletion Zarr/test/consolidated.jl
Original file line number Diff line number Diff line change
Expand Up @@ -85,7 +85,7 @@ path_v3_julia = joinpath(@__DIR__, "v3_julia", "data.zarr")
# getmetadata v3 reads from cons["metadata"][key]
meta = ZarrCore.getmetadata(ZarrCore.ZarrFormat(3), cs.storage, "1d.chunked.i2", false)
@test eltype(meta) == Int16
@test meta.chunks == (2,)
@test meta.chunks[] == (2,)
end

@testset "is_zarray / is_zgroup v3 on ConsolidatedStore" begin
Expand Down Expand Up @@ -409,4 +409,32 @@ path_v3_julia = joinpath(@__DIR__, "v3_julia", "data.zarr")
@test g2["a1"][:,:] == reshape(1:200, 10, 20)
end

@testset "resize! through a ConsolidatedStore changes nothing" begin
# v2
dir = joinpath(mktempdir(), "v2.zarr")
g = zgroup(dir)
a = zcreate(Int, g, "a", 4, 6, chunks=(2, 3), fill_value=0)
a[:, :] = reshape(1:24, 4, 6)
Zarr.consolidate_metadata(g)
chunkfiles = sort(readdir(joinpath(dir, "a")))
z = zopen(dir, "w", consolidated=true)["a"]
@test_throws ArgumentError resize!(z, 2, 6)
@test_throws ArgumentError resize!(z, 6, 6)
@test size(z) == (4, 6)
@test sort(readdir(joinpath(dir, "a"))) == chunkfiles
@test zopen(dir)["a"][:, :] == reshape(1:24, 4, 6)

# v3: the array's own zarr.json must not drift from the consolidated copy
dir = joinpath(mktempdir(), "v3.zarr")
cp(joinpath(path_v3_julia, "consolidated"), dir)
name = "1d.chunked.i2"
before = zopen(dir)[name][:]
z = zopen(dir, "w", consolidated=true)[name]
@test_throws ArgumentError resize!(z, length(before) - 2)
@test_throws ArgumentError resize!(z, length(before) + 4)
@test size(z) == size(before)
@test zopen(dir)[name][:] == before
@test zopen(dir, consolidated=true)[name][:] == before
end

end
8 changes: 4 additions & 4 deletions Zarr/test/http_sharded.jl
Original file line number Diff line number Diff line change
Expand Up @@ -45,7 +45,7 @@ const TESSERA_BASE = "https://dl2.geotessera.org/zarr/v1/2024.zarr"
# Julia reverses to column-major: (128, 66560, 1355776)
@test size(z) == (128, 66560, 1355776)
# Outer shard shape [256, 256, 128] → Julia (128, 256, 256)
@test z.metadata.chunks == (128, 256, 256)
@test z.metadata.chunks[] == (128, 256, 256)

# Verify the codec is sharding_indexed with the expected inner chunk shape
pipeline = z.metadata.pipeline
Expand Down Expand Up @@ -144,7 +144,7 @@ const FLAMINGO_BASE = "https://radosgw.public.os.wwu.de/n4bi-goe"
@test eltype(z) == UInt16
@test ndims(z) == 3
@test size(z) == (1024, 1024, 192)
@test z.metadata.chunks == (512, 512, 512)
@test z.metadata.chunks[] == (512, 512, 512)

sharding = z.metadata.pipeline.array_bytes
@test sharding isa Zarr.Codecs.V3Codecs.ShardingCodec
Expand Down Expand Up @@ -173,7 +173,7 @@ const MUENSTER_BASE = "https://radosgw.public.os2.wwu.de/ngff"
@test eltype(z) == UInt8
@test ndims(z) == 3
@test size(z) == (6000, 6000, 6000)
@test z.metadata.chunks == (8192, 8192, 1)
@test z.metadata.chunks[] == (8192, 8192, 1)

sharding = z.metadata.pipeline.array_bytes
@test sharding isa Zarr.Codecs.V3Codecs.ShardingCodec
Expand Down Expand Up @@ -234,7 +234,7 @@ end # @testset "Remote HTTP sharded arrays (SSBD — RIKEN)"
@test eltype(z) == UInt16
@test ndims(z) == 5
@test size(z) == (522693, 244215, 1, 3, 1)
@test z.metadata.chunks == (2048, 2048, 1, 1, 1)
@test z.metadata.chunks[] == (2048, 2048, 1, 1, 1)

sharding = z.metadata.pipeline.array_bytes
@test sharding isa Zarr.Codecs.V3Codecs.ShardingCodec
Expand Down
161 changes: 161 additions & 0 deletions Zarr/test/irregular_chunks.jl
Original file line number Diff line number Diff line change
@@ -0,0 +1,161 @@
using DiskArrays: DiskArrays, GridChunks, IrregularChunks, RegularChunks

@testset "Zarr v3 rectilinear chunk grids" begin
chunks = GridChunks(
RegularChunks(2, 0, 5),
IrregularChunks(; chunksizes=[3, 4, 5, 6, 2]),
)

@testset "read, write, and exact stored extents" begin
z = zcreate(Int32, 5, 20;
zarr_format=3,
chunks,
compressor=Zarr.NoCompressor(),
)
data = reshape(Int32.(1:100), 5, 20)
z[:, :] = data

@test DiskArrays.eachchunk(z) == chunks
@test z[:, :] == data
@test z[2:4, 2:10] == data[2:4, 2:10]
@test z[1:5, 6:18] == data[1:5, 6:18]

z[2:5, 6:15] .= Int32(-7)
data[2:5, 6:15] .= Int32(-7)
@test z[:, :] == data

stored_lengths = sort([
length(bytes) for (key, bytes) in z.storage.a if startswith(key, "c/")
])
expected_lengths = sort(vec([
sizeof(Int32) * rows * columns
for rows in (2, 2, 2), columns in (3, 4, 5, 6, 2)
]))
@test stored_lengths == expected_lengths
end

@testset "missing chunks and zero initialization" begin
z = zcreate(Int16, 5, 20; zarr_format=3, chunks, fill_value=Int16(0))
z[2:4, 3:9] .= Int16(12)
expected = zeros(Int16, 5, 20)
expected[2:4, 3:9] .= Int16(12)
@test z[:, :] == expected

zz = zzeros(Int16, 5, 20; zarr_format=3, chunks)
@test zz[:, :] == zeros(Int16, 5, 20)

with_missing = zcreate(Int16, 5, 20;
zarr_format=3,
chunks,
fill_value=Int16(-1),
fill_as_missing=true,
)
expected_missing = Matrix{Union{Missing,Int16}}(missing, 5, 20)
expected_missing[2:4, 3:9] .= Int16(12)
expected_missing[3, 5] = missing
with_missing[2:4, 3:9] = expected_missing[2:4, 3:9]
@test all(isequal.(with_missing[:, :], expected_missing))
end

@testset "metadata round-trip and run-length encoding" begin
mktempdir() do dir
repeated = GridChunks(
IrregularChunks(; chunksizes=[2, 2, 3, 3]),
RegularChunks(4, 0, 8),
)
z = zcreate(Int16, 10, 8;
zarr_format=3,
chunks=repeated,
compressor=Zarr.NoCompressor(),
path=dir,
)
data = reshape(Int16.(1:80), 10, 8)
z[:, :] = data

metadata = JSON.parsefile(joinpath(dir, "zarr.json"))
chunk_grid = metadata["chunk_grid"]
@test chunk_grid["name"] == "rectilinear"
@test chunk_grid["configuration"]["kind"] == "inline"
@test chunk_grid["configuration"]["chunk_shapes"] ==
Any[4, Any[Any[2, 2], Any[3, 2]]]

reopened = zopen(dir)
@test DiskArrays.eachchunk(reopened) == repeated
@test reopened[:, :] == data
end
end

@testset "metadata validation" begin
z = zcreate(Int32, 5, 20;
zarr_format=3,
chunks,
compressor=Zarr.NoCompressor(),
)
metadata = JSON.parse(JSON.json(z.metadata); dicttype=Dict{String,Any})
@test ZarrCore.Metadata(metadata, false) == z.metadata

# Legal per the spec, but chunks extending past the array are not supported.
overflowing = deepcopy(metadata)
overflowing["chunk_grid"]["configuration"]["chunk_shapes"][1] = Any[3, 4, 5, 10]
@test_throws ArgumentError ZarrCore.Metadata(overflowing, false)

too_short = deepcopy(metadata)
too_short["chunk_grid"]["configuration"]["chunk_shapes"][1] = Any[3, 4]
@test_throws DimensionMismatch ZarrCore.Metadata(too_short, false)

bad_rank = GridChunks(IrregularChunks(; chunksizes=[2, 3]))
@test_throws DimensionMismatch zcreate(Int, 5, 20; zarr_format=3, chunks=bad_rank)

bad_shape = GridChunks(
RegularChunks(2, 0, 6),
IrregularChunks(; chunksizes=[3, 4, 5, 6, 2]),
)
@test_throws DimensionMismatch zcreate(Int, 5, 20; zarr_format=3, chunks=bad_shape)

offset_grid = GridChunks(
RegularChunks(2, 1, 5),
IrregularChunks(; chunksizes=[3, 4, 5, 6, 2]),
)
@test_throws ArgumentError zcreate(Int, 5, 20; zarr_format=3, chunks=offset_grid)
@test_throws ArgumentError zcreate(Int, 5, 20; zarr_format=2, chunks=chunks)
end

@testset "resize and append" begin
mktempdir() do dir
z = zcreate(Int, 5, 20; zarr_format=3, chunks, path=dir, fill_value=0)
data = reshape(1:100, 5, 20)
z[:, :] = data

# Irregular axes cannot shrink, and the rejection leaves the array untouched.
@test_throws ArgumentError resize!(z, 5, 18)
@test_throws ArgumentError resize!(z, 8, 18)
@test size(z) == (5, 20)
@test DiskArrays.eachchunk(z) == chunks

append!(z, fill(7, 3, 20); dims=1)
@test size(z) == (8, 20)
@test DiskArrays.eachchunk(z) == GridChunks(RegularChunks(2, 0, 8), chunks.chunks[2])
@test z[:, :] == vcat(data, fill(7, 3, 20))

reopened = zopen(dir, "w")
@test DiskArrays.eachchunk(reopened) == DiskArrays.eachchunk(z)
@test reopened[:, :] == vcat(data, fill(7, 3, 20))

resize!(reopened, 3, 20)
@test reopened[:, :] == data[1:3, :]
@test !isfile(joinpath(dir, "c", "0", "2"))
@test isfile(joinpath(dir, "c", "0", "1"))
resize!(reopened, 5, 20)
@test reopened[1:3, :] == data[1:3, :]
@test all(iszero, reopened[5, :])

# Growing an irregular axis appends one chunk, as zarr-python does.
append!(reopened, fill(9, 5, 4); dims=2)
@test DiskArrays.eachchunk(reopened).chunks[2] ==
IrregularChunks(; chunksizes=[3, 4, 5, 6, 2, 4])
@test reopened.metadata == zopen(dir).metadata
@test zopen(dir)[:, 21:24] == fill(9, 5, 4)
@test zopen(dir)[1:3, 1:20] == data[1:3, :]
end
end
end
86 changes: 86 additions & 0 deletions Zarr/test/python.jl
Original file line number Diff line number Diff line change
Expand Up @@ -90,6 +90,92 @@ for (filterstr, filter) in filters
a[] = testzerodim
end

@testset "Python zarr v3 rectilinear chunks" begin
using DiskArrays: GridChunks, IrregularChunks, RegularChunks

zarr = pyimport("zarr")
numpy = pyimport("numpy")
pybuiltins = pyimport("builtins")
zarr.config.set(pydict(Dict("array.rectilinear_chunks" => true)))

julia_path = tempname()
python_path = tempname()

# Julia writes exact-size rectilinear chunks that zarr-python can read.
chunks = GridChunks(
RegularChunks(4, 0, 20),
IrregularChunks(; chunksizes=[3, 4, 5, 6, 2]),
)
z = zcreate(Int32, 20, 20;
zarr_format=3,
path=julia_path,
chunks,
compressor=Zarr.NoCompressor(),
)
julia_data = reshape(Int32.(1:400), 20, 20)
z[:, :] = julia_data

py_array = zarr.open_array(julia_path, mode="r")
py_grid = pyconvert(Any, py_array.metadata.chunk_grid.to_dict())
@test py_grid["name"] == "rectilinear"
@test py_grid["configuration"]["kind"] == "inline"
py_chunk_shapes = py_grid["configuration"]["chunk_shapes"]
@test collect(py_chunk_shapes[1]) == [3, 4, 5, 6, 2]
@test py_chunk_shapes[2] == 4
@test pyconvert(Array, py_array[pybuiltins.Ellipsis]) == permutedims(julia_data, (2, 1))

# zarr-python writes exact-size rectilinear chunks that Julia can read.
python_data = reshape(Int32.(0:399), 20, 20)
py_written = zarr.create_array(
python_path;
zarr_format=3,
shape=(20, 20),
chunks=pylist([
pylist([3, 4, 5, 6, 2]),
pylist([4, 4, 4, 4, 4]),
]),
dtype="int32",
)
py_written.__setitem__(pybuiltins.Ellipsis, numpy.array(python_data))

reopened = zopen(python_path)
@test any(c -> c isa IrregularChunks, DiskArrays.eachchunk(reopened).chunks)
@test permutedims(reopened[:, :], (2, 1)) == python_data

# A regular axis that does not divide the extent stores a full-size final
# chunk, in both directions.
edge_path = tempname()
edge_chunks = GridChunks(
RegularChunks(4, 0, 18),
IrregularChunks(; chunksizes=[3, 4, 5, 6, 2]),
)
edge = zcreate(Int32, 18, 20;
zarr_format=3,
path=edge_path,
chunks=edge_chunks,
compressor=Zarr.NoCompressor(),
)
edge_data = reshape(Int32.(1:360), 18, 20)
edge[:, :] = edge_data
py_edge = zarr.open_array(edge_path, mode="r+")
@test pyconvert(Array, py_edge[pybuiltins.Ellipsis]) == permutedims(edge_data, (2, 1))

py_edge.__setitem__(pybuiltins.Ellipsis, numpy.array(permutedims(edge_data .+ Int32(1), (2, 1))))
@test zopen(edge_path)[:, :] == edge_data .+ Int32(1)

# Both sides grow an irregular axis by appending one chunk.
append!(zopen(edge_path, "w"), fill(Int32(-3), 18, 5); dims=2)
py_grown = zarr.open_array(edge_path, mode="r+")
@test pyconvert(Tuple, py_grown.shape) == (25, 18)
@test pyconvert(Array, py_grown[pybuiltins.Ellipsis]) ==
permutedims(hcat(edge_data .+ Int32(1), fill(Int32(-3), 18, 5)), (2, 1))
py_grown.resize((28, 18))
jl_grown = zopen(edge_path)
@test size(jl_grown) == (18, 28)
@test diff(DiskArrays.eachchunk(jl_grown).chunks[2].offsets) == [3, 4, 5, 6, 2, 5, 3]
@test all(iszero, jl_grown[:, 26:28])
end

#Also save as zip file.
open(pjulia*".zip";write=true) do io
Zarr.writezip(io, g)
Expand Down
14 changes: 14 additions & 0 deletions Zarr/test/runtests.jl
Original file line number Diff line number Diff line change
Expand Up @@ -406,6 +406,18 @@ end
@test size(a)==(13,31)
@test a[12:13,:]==vcat(singlerow', singlerow')
@test_throws ArgumentError resize!(a,(-1,2))

# A read-only array must not touch the stored metadata or chunks.
mktempdir() do dir
w = zzeros(Int64, 10, 10, path=dir, chunks=(5,2))
chunkfiles = sort(readdir(dir))
readonly = zopen(dir)
@test_throws ErrorException resize!(readonly, 5, 4)
@test_throws ErrorException append!(readonly, ones(Int64, 10, 2))
@test size(readonly) == (10, 10)
@test size(zopen(dir)) == (10, 10)
@test sort(readdir(dir)) == chunkfiles
end
end

@testset "zcreate does not allocate dense storage" begin
Expand Down Expand Up @@ -505,6 +517,8 @@ include("storage.jl")

include("Filters.jl")

include("irregular_chunks.jl")

include("python.jl")

include("v3_codecs.jl")
Expand Down
Loading
Loading