diff --git a/CHANGELOG.md b/CHANGELOG.md index 1ee433d3..1419cd4e 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -2,6 +2,13 @@ ## Unreleased - Move all packages to sibling top-level directories and make the root project workspace-only. +- Support irregular (rectilinear) chunk grids for Zarr v3 arrays. `zcreate`, + `zzeros`, and `ZArray(data; chunks=...)` accept `DiskArrays.GridChunks`, and + rectilinear metadata and chunks interoperate with zarr-python. `resize!` and + `append!` work on rectilinear arrays, except for shrinking an irregular axis + [#332](https://github.com/JuliaIO/Zarr.jl/pull/332), [#326](https://github.com/JuliaIO/Zarr.jl/pull/326). +- `resize!` and `append!` now throw on a read-only `ZArray` instead of rewriting its metadata and deleting chunks. +- `resize!` through a `ConsolidatedStore` no longer deletes chunks or changes the in-memory shape before failing, and Zarr v3 consolidated stores now reject it like v2 ones instead of leaving the consolidated metadata stale. - Move code to ZarrCore.jl with low dependencies - Move HTTP, GCS, S3, and ZIP backends into `ZarrHTTP`, `ZarrGCS`, `ZarrS3`, and `ZarrZip`. - Move Blosc, zlib, and Zstandard support into `ZarrBlosc`, `ZarrZlib`, and `ZarrZstd`; each provides its v2 compressor and v3 codec. diff --git a/CLAUDE.md b/CLAUDE.md index 9d08025d..74e54a2a 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -139,6 +139,7 @@ Core paths below are relative to `ZarrCore/`. Backend implementations are in - **Format version dispatch**: `ZarrFormat{V}` selects version-specific behavior - **Shape is mutable**: `metadata.shape` is `Base.RefValue{NTuple{N,Int}}` to allow `resize!` without replacing the metadata struct +- **V3 chunks are mutable**: `MetadataV3.chunks` is a `Base.RefValue` holding an `NTuple` (regular grid) or a `DiskArrays.GridChunks` (rectilinear grid, which `resize!` replaces); read it through `chunkspec(md)` - **Column-major ↔ row-major**: reverse dimensions at metadata boundaries - **Compressor registry**: `compressortypes` maps v2 names to compressor types; `v2_to_v3_codecs` maps compressors to v3 codecs by dispatch - **DiskArrays integration**: `ZArray <: AbstractDiskArray`, chunk-aware I/O via `readblock!`/`writeblock!`, `eachchunk`, `haschunks` diff --git a/Zarr/test/Project.toml b/Zarr/test/Project.toml index af4c2336..5ca39ecd 100644 --- a/Zarr/test/Project.toml +++ b/Zarr/test/Project.toml @@ -3,6 +3,7 @@ AWSS3 = "1c724243-ef5b-51ab-93f4-b0a88ac62a95" CondaPkg = "992eb4ea-22a4-4c89-a5bb-47a3300528ab" DateTimes64 = "b342263e-b350-472a-b1a9-8dfd21b51589" Dates = "ade2ca70-3891-5945-98fb-dc099432e06a" +DiskArrays = "3c3547ce-8d99-4f5e-a174-61eb10b00ae3" JSON = "682c06a0-de6a-54ab-a142-c8b1cf79cde6" Minio = "4281f0d9-7ae0-406e-9172-b7277c1efa20" Mmap = "a63ad114-7e13-5084-954f-fe012c677804" diff --git a/Zarr/test/consolidated.jl b/Zarr/test/consolidated.jl index 65c8fc73..d63b8e87 100644 --- a/Zarr/test/consolidated.jl +++ b/Zarr/test/consolidated.jl @@ -85,7 +85,7 @@ path_v3_julia = joinpath(@__DIR__, "v3_julia", "data.zarr") # getmetadata v3 reads from cons["metadata"][key] meta = ZarrCore.getmetadata(ZarrCore.ZarrFormat(3), cs.storage, "1d.chunked.i2", false) @test eltype(meta) == Int16 - @test meta.chunks == (2,) + @test meta.chunks[] == (2,) end @testset "is_zarray / is_zgroup v3 on ConsolidatedStore" begin @@ -409,4 +409,32 @@ path_v3_julia = joinpath(@__DIR__, "v3_julia", "data.zarr") @test g2["a1"][:,:] == reshape(1:200, 10, 20) end + @testset "resize! through a ConsolidatedStore changes nothing" begin + # v2 + dir = joinpath(mktempdir(), "v2.zarr") + g = zgroup(dir) + a = zcreate(Int, g, "a", 4, 6, chunks=(2, 3), fill_value=0) + a[:, :] = reshape(1:24, 4, 6) + Zarr.consolidate_metadata(g) + chunkfiles = sort(readdir(joinpath(dir, "a"))) + z = zopen(dir, "w", consolidated=true)["a"] + @test_throws ArgumentError resize!(z, 2, 6) + @test_throws ArgumentError resize!(z, 6, 6) + @test size(z) == (4, 6) + @test sort(readdir(joinpath(dir, "a"))) == chunkfiles + @test zopen(dir)["a"][:, :] == reshape(1:24, 4, 6) + + # v3: the array's own zarr.json must not drift from the consolidated copy + dir = joinpath(mktempdir(), "v3.zarr") + cp(joinpath(path_v3_julia, "consolidated"), dir) + name = "1d.chunked.i2" + before = zopen(dir)[name][:] + z = zopen(dir, "w", consolidated=true)[name] + @test_throws ArgumentError resize!(z, length(before) - 2) + @test_throws ArgumentError resize!(z, length(before) + 4) + @test size(z) == size(before) + @test zopen(dir)[name][:] == before + @test zopen(dir, consolidated=true)[name][:] == before + end + end \ No newline at end of file diff --git a/Zarr/test/http_sharded.jl b/Zarr/test/http_sharded.jl index de85a64d..e318566d 100644 --- a/Zarr/test/http_sharded.jl +++ b/Zarr/test/http_sharded.jl @@ -45,7 +45,7 @@ const TESSERA_BASE = "https://dl2.geotessera.org/zarr/v1/2024.zarr" # Julia reverses to column-major: (128, 66560, 1355776) @test size(z) == (128, 66560, 1355776) # Outer shard shape [256, 256, 128] → Julia (128, 256, 256) - @test z.metadata.chunks == (128, 256, 256) + @test z.metadata.chunks[] == (128, 256, 256) # Verify the codec is sharding_indexed with the expected inner chunk shape pipeline = z.metadata.pipeline @@ -144,7 +144,7 @@ const FLAMINGO_BASE = "https://radosgw.public.os.wwu.de/n4bi-goe" @test eltype(z) == UInt16 @test ndims(z) == 3 @test size(z) == (1024, 1024, 192) - @test z.metadata.chunks == (512, 512, 512) + @test z.metadata.chunks[] == (512, 512, 512) sharding = z.metadata.pipeline.array_bytes @test sharding isa Zarr.Codecs.V3Codecs.ShardingCodec @@ -173,7 +173,7 @@ const MUENSTER_BASE = "https://radosgw.public.os2.wwu.de/ngff" @test eltype(z) == UInt8 @test ndims(z) == 3 @test size(z) == (6000, 6000, 6000) - @test z.metadata.chunks == (8192, 8192, 1) + @test z.metadata.chunks[] == (8192, 8192, 1) sharding = z.metadata.pipeline.array_bytes @test sharding isa Zarr.Codecs.V3Codecs.ShardingCodec @@ -234,7 +234,7 @@ end # @testset "Remote HTTP sharded arrays (SSBD — RIKEN)" @test eltype(z) == UInt16 @test ndims(z) == 5 @test size(z) == (522693, 244215, 1, 3, 1) - @test z.metadata.chunks == (2048, 2048, 1, 1, 1) + @test z.metadata.chunks[] == (2048, 2048, 1, 1, 1) sharding = z.metadata.pipeline.array_bytes @test sharding isa Zarr.Codecs.V3Codecs.ShardingCodec diff --git a/Zarr/test/irregular_chunks.jl b/Zarr/test/irregular_chunks.jl new file mode 100644 index 00000000..828bb58e --- /dev/null +++ b/Zarr/test/irregular_chunks.jl @@ -0,0 +1,161 @@ +using DiskArrays: DiskArrays, GridChunks, IrregularChunks, RegularChunks + +@testset "Zarr v3 rectilinear chunk grids" begin + chunks = GridChunks( + RegularChunks(2, 0, 5), + IrregularChunks(; chunksizes=[3, 4, 5, 6, 2]), + ) + + @testset "read, write, and exact stored extents" begin + z = zcreate(Int32, 5, 20; + zarr_format=3, + chunks, + compressor=Zarr.NoCompressor(), + ) + data = reshape(Int32.(1:100), 5, 20) + z[:, :] = data + + @test DiskArrays.eachchunk(z) == chunks + @test z[:, :] == data + @test z[2:4, 2:10] == data[2:4, 2:10] + @test z[1:5, 6:18] == data[1:5, 6:18] + + z[2:5, 6:15] .= Int32(-7) + data[2:5, 6:15] .= Int32(-7) + @test z[:, :] == data + + stored_lengths = sort([ + length(bytes) for (key, bytes) in z.storage.a if startswith(key, "c/") + ]) + expected_lengths = sort(vec([ + sizeof(Int32) * rows * columns + for rows in (2, 2, 2), columns in (3, 4, 5, 6, 2) + ])) + @test stored_lengths == expected_lengths + end + + @testset "missing chunks and zero initialization" begin + z = zcreate(Int16, 5, 20; zarr_format=3, chunks, fill_value=Int16(0)) + z[2:4, 3:9] .= Int16(12) + expected = zeros(Int16, 5, 20) + expected[2:4, 3:9] .= Int16(12) + @test z[:, :] == expected + + zz = zzeros(Int16, 5, 20; zarr_format=3, chunks) + @test zz[:, :] == zeros(Int16, 5, 20) + + with_missing = zcreate(Int16, 5, 20; + zarr_format=3, + chunks, + fill_value=Int16(-1), + fill_as_missing=true, + ) + expected_missing = Matrix{Union{Missing,Int16}}(missing, 5, 20) + expected_missing[2:4, 3:9] .= Int16(12) + expected_missing[3, 5] = missing + with_missing[2:4, 3:9] = expected_missing[2:4, 3:9] + @test all(isequal.(with_missing[:, :], expected_missing)) + end + + @testset "metadata round-trip and run-length encoding" begin + mktempdir() do dir + repeated = GridChunks( + IrregularChunks(; chunksizes=[2, 2, 3, 3]), + RegularChunks(4, 0, 8), + ) + z = zcreate(Int16, 10, 8; + zarr_format=3, + chunks=repeated, + compressor=Zarr.NoCompressor(), + path=dir, + ) + data = reshape(Int16.(1:80), 10, 8) + z[:, :] = data + + metadata = JSON.parsefile(joinpath(dir, "zarr.json")) + chunk_grid = metadata["chunk_grid"] + @test chunk_grid["name"] == "rectilinear" + @test chunk_grid["configuration"]["kind"] == "inline" + @test chunk_grid["configuration"]["chunk_shapes"] == + Any[4, Any[Any[2, 2], Any[3, 2]]] + + reopened = zopen(dir) + @test DiskArrays.eachchunk(reopened) == repeated + @test reopened[:, :] == data + end + end + + @testset "metadata validation" begin + z = zcreate(Int32, 5, 20; + zarr_format=3, + chunks, + compressor=Zarr.NoCompressor(), + ) + metadata = JSON.parse(JSON.json(z.metadata); dicttype=Dict{String,Any}) + @test ZarrCore.Metadata(metadata, false) == z.metadata + + # Legal per the spec, but chunks extending past the array are not supported. + overflowing = deepcopy(metadata) + overflowing["chunk_grid"]["configuration"]["chunk_shapes"][1] = Any[3, 4, 5, 10] + @test_throws ArgumentError ZarrCore.Metadata(overflowing, false) + + too_short = deepcopy(metadata) + too_short["chunk_grid"]["configuration"]["chunk_shapes"][1] = Any[3, 4] + @test_throws DimensionMismatch ZarrCore.Metadata(too_short, false) + + bad_rank = GridChunks(IrregularChunks(; chunksizes=[2, 3])) + @test_throws DimensionMismatch zcreate(Int, 5, 20; zarr_format=3, chunks=bad_rank) + + bad_shape = GridChunks( + RegularChunks(2, 0, 6), + IrregularChunks(; chunksizes=[3, 4, 5, 6, 2]), + ) + @test_throws DimensionMismatch zcreate(Int, 5, 20; zarr_format=3, chunks=bad_shape) + + offset_grid = GridChunks( + RegularChunks(2, 1, 5), + IrregularChunks(; chunksizes=[3, 4, 5, 6, 2]), + ) + @test_throws ArgumentError zcreate(Int, 5, 20; zarr_format=3, chunks=offset_grid) + @test_throws ArgumentError zcreate(Int, 5, 20; zarr_format=2, chunks=chunks) + end + + @testset "resize and append" begin + mktempdir() do dir + z = zcreate(Int, 5, 20; zarr_format=3, chunks, path=dir, fill_value=0) + data = reshape(1:100, 5, 20) + z[:, :] = data + + # Irregular axes cannot shrink, and the rejection leaves the array untouched. + @test_throws ArgumentError resize!(z, 5, 18) + @test_throws ArgumentError resize!(z, 8, 18) + @test size(z) == (5, 20) + @test DiskArrays.eachchunk(z) == chunks + + append!(z, fill(7, 3, 20); dims=1) + @test size(z) == (8, 20) + @test DiskArrays.eachchunk(z) == GridChunks(RegularChunks(2, 0, 8), chunks.chunks[2]) + @test z[:, :] == vcat(data, fill(7, 3, 20)) + + reopened = zopen(dir, "w") + @test DiskArrays.eachchunk(reopened) == DiskArrays.eachchunk(z) + @test reopened[:, :] == vcat(data, fill(7, 3, 20)) + + resize!(reopened, 3, 20) + @test reopened[:, :] == data[1:3, :] + @test !isfile(joinpath(dir, "c", "0", "2")) + @test isfile(joinpath(dir, "c", "0", "1")) + resize!(reopened, 5, 20) + @test reopened[1:3, :] == data[1:3, :] + @test all(iszero, reopened[5, :]) + + # Growing an irregular axis appends one chunk, as zarr-python does. + append!(reopened, fill(9, 5, 4); dims=2) + @test DiskArrays.eachchunk(reopened).chunks[2] == + IrregularChunks(; chunksizes=[3, 4, 5, 6, 2, 4]) + @test reopened.metadata == zopen(dir).metadata + @test zopen(dir)[:, 21:24] == fill(9, 5, 4) + @test zopen(dir)[1:3, 1:20] == data[1:3, :] + end + end +end diff --git a/Zarr/test/python.jl b/Zarr/test/python.jl index de80d585..dd787f42 100644 --- a/Zarr/test/python.jl +++ b/Zarr/test/python.jl @@ -90,6 +90,92 @@ for (filterstr, filter) in filters a[] = testzerodim end +@testset "Python zarr v3 rectilinear chunks" begin + using DiskArrays: GridChunks, IrregularChunks, RegularChunks + + zarr = pyimport("zarr") + numpy = pyimport("numpy") + pybuiltins = pyimport("builtins") + zarr.config.set(pydict(Dict("array.rectilinear_chunks" => true))) + + julia_path = tempname() + python_path = tempname() + + # Julia writes exact-size rectilinear chunks that zarr-python can read. + chunks = GridChunks( + RegularChunks(4, 0, 20), + IrregularChunks(; chunksizes=[3, 4, 5, 6, 2]), + ) + z = zcreate(Int32, 20, 20; + zarr_format=3, + path=julia_path, + chunks, + compressor=Zarr.NoCompressor(), + ) + julia_data = reshape(Int32.(1:400), 20, 20) + z[:, :] = julia_data + + py_array = zarr.open_array(julia_path, mode="r") + py_grid = pyconvert(Any, py_array.metadata.chunk_grid.to_dict()) + @test py_grid["name"] == "rectilinear" + @test py_grid["configuration"]["kind"] == "inline" + py_chunk_shapes = py_grid["configuration"]["chunk_shapes"] + @test collect(py_chunk_shapes[1]) == [3, 4, 5, 6, 2] + @test py_chunk_shapes[2] == 4 + @test pyconvert(Array, py_array[pybuiltins.Ellipsis]) == permutedims(julia_data, (2, 1)) + + # zarr-python writes exact-size rectilinear chunks that Julia can read. + python_data = reshape(Int32.(0:399), 20, 20) + py_written = zarr.create_array( + python_path; + zarr_format=3, + shape=(20, 20), + chunks=pylist([ + pylist([3, 4, 5, 6, 2]), + pylist([4, 4, 4, 4, 4]), + ]), + dtype="int32", + ) + py_written.__setitem__(pybuiltins.Ellipsis, numpy.array(python_data)) + + reopened = zopen(python_path) + @test any(c -> c isa IrregularChunks, DiskArrays.eachchunk(reopened).chunks) + @test permutedims(reopened[:, :], (2, 1)) == python_data + + # A regular axis that does not divide the extent stores a full-size final + # chunk, in both directions. + edge_path = tempname() + edge_chunks = GridChunks( + RegularChunks(4, 0, 18), + IrregularChunks(; chunksizes=[3, 4, 5, 6, 2]), + ) + edge = zcreate(Int32, 18, 20; + zarr_format=3, + path=edge_path, + chunks=edge_chunks, + compressor=Zarr.NoCompressor(), + ) + edge_data = reshape(Int32.(1:360), 18, 20) + edge[:, :] = edge_data + py_edge = zarr.open_array(edge_path, mode="r+") + @test pyconvert(Array, py_edge[pybuiltins.Ellipsis]) == permutedims(edge_data, (2, 1)) + + py_edge.__setitem__(pybuiltins.Ellipsis, numpy.array(permutedims(edge_data .+ Int32(1), (2, 1)))) + @test zopen(edge_path)[:, :] == edge_data .+ Int32(1) + + # Both sides grow an irregular axis by appending one chunk. + append!(zopen(edge_path, "w"), fill(Int32(-3), 18, 5); dims=2) + py_grown = zarr.open_array(edge_path, mode="r+") + @test pyconvert(Tuple, py_grown.shape) == (25, 18) + @test pyconvert(Array, py_grown[pybuiltins.Ellipsis]) == + permutedims(hcat(edge_data .+ Int32(1), fill(Int32(-3), 18, 5)), (2, 1)) + py_grown.resize((28, 18)) + jl_grown = zopen(edge_path) + @test size(jl_grown) == (18, 28) + @test diff(DiskArrays.eachchunk(jl_grown).chunks[2].offsets) == [3, 4, 5, 6, 2, 5, 3] + @test all(iszero, jl_grown[:, 26:28]) +end + #Also save as zip file. open(pjulia*".zip";write=true) do io Zarr.writezip(io, g) diff --git a/Zarr/test/runtests.jl b/Zarr/test/runtests.jl index dd70ce7c..43140c91 100644 --- a/Zarr/test/runtests.jl +++ b/Zarr/test/runtests.jl @@ -406,6 +406,18 @@ end @test size(a)==(13,31) @test a[12:13,:]==vcat(singlerow', singlerow') @test_throws ArgumentError resize!(a,(-1,2)) + + # A read-only array must not touch the stored metadata or chunks. + mktempdir() do dir + w = zzeros(Int64, 10, 10, path=dir, chunks=(5,2)) + chunkfiles = sort(readdir(dir)) + readonly = zopen(dir) + @test_throws ErrorException resize!(readonly, 5, 4) + @test_throws ErrorException append!(readonly, ones(Int64, 10, 2)) + @test size(readonly) == (10, 10) + @test size(zopen(dir)) == (10, 10) + @test sort(readdir(dir)) == chunkfiles + end end @testset "zcreate does not allocate dense storage" begin @@ -505,6 +517,8 @@ include("storage.jl") include("Filters.jl") +include("irregular_chunks.jl") + include("python.jl") include("v3_codecs.jl") diff --git a/Zarr/test/v3_codecs.jl b/Zarr/test/v3_codecs.jl index d08a32fc..eda82507 100644 --- a/Zarr/test/v3_codecs.jl +++ b/Zarr/test/v3_codecs.jl @@ -435,7 +435,7 @@ end md = ZarrCore.Metadata(json_str, false) @test md isa ZarrCore.MetadataV3 @test md.shape[] == (4,) - @test md.chunks == (4,) + @test md.chunks[] == (4,) @test md.fill_value == Int32(0) pipeline = ZarrCore.get_pipeline(md) diff --git a/ZarrCore/src/Storage/consolidated.jl b/ZarrCore/src/Storage/consolidated.jl index 890facf4..b749b30c 100644 --- a/ZarrCore/src/Storage/consolidated.jl +++ b/ZarrCore/src/Storage/consolidated.jl @@ -93,7 +93,7 @@ function is_zgroup(::ZarrFormat{3}, d::ConsolidatedStore, p) end ZarrFormat(d::ConsolidatedStore, path) = ZarrFormat(d.parent, path) # detect format from parent, not cons -check_consolidated_write(i::String) = split(i, '/')[end] in (".zattrs", ".zarray", ".zgroup") && +check_consolidated_write(i::String) = split(i, '/')[end] in (".zattrs", ".zarray", ".zgroup", "zarr.json") && throw(ArgumentError("Can not modify consolidated metadata, please re-open the dataset with `consolidated=false`")) function _pdict(d::ConsolidatedStore, p) zv = ZarrFormat(d.parent, d.path) diff --git a/ZarrCore/src/ZArray.jl b/ZarrCore/src/ZArray.jl index ede00460..d4db83d5 100644 --- a/ZarrCore/src/ZArray.jl +++ b/ZarrCore/src/ZArray.jl @@ -97,7 +97,7 @@ function zinfo(io::IO,z::ZArray) "Type" => "ZArray", "Data type" => eltype(z), "Shape" => size(z), - "Chunk Shape" => z.metadata.chunks, + "Chunk Shape" => chunk_edges(chunkspec(z.metadata)), "Order" => try get_order(z.metadata) catch e "unknown ($(e.msg))" end, "Read-Only" => !z.writeable, "Compressor" => z.metadata isa MetadataV2 ? z.metadata.compressor : get_pipeline(z.metadata), @@ -130,14 +130,44 @@ zarr_format(z::ZArray) = zarr_format(z.metadata) dimension_separator(z::ZArray) = dimension_separator(z.metadata) -""" - trans_ind(r, bs) +# Chunk sizes as a tuple for a regular grid, or a `GridChunks` for a rectilinear one. +chunkspec(md::MetadataV2) = md.chunks +chunkspec(md::MetadataV3) = md.chunks[] -For a given index and blocksize determines which chunks of the Zarray will have to -be accessed. -""" -trans_ind(r::AbstractUnitRange, bs) = fld1(first(r),bs):fld1(last(r),bs) -trans_ind(r::Integer, bs) = fld1(r,bs) +chunkgrid(md::AbstractMetadata) = chunkgrid(md.shape[], chunkspec(md)) +chunkgrid(shape, chunks::NTuple) = DiskArrays.GridChunks(shape, chunks) +chunkgrid(::Any, chunks::DiskArrays.GridChunks) = chunks + +# Per-axis chunk sizes: an `Int` for a regular axis, the list of sizes otherwise. +chunk_edges(chunks::NTuple) = chunks +chunk_edges(chunks::DiskArrays.GridChunks) = map(chunk_edges, chunks.chunks) +chunk_edges(chunks::DiskArrays.RegularChunks) = chunks.chunksize +chunk_edges(chunks::DiskArrays.IrregularChunks) = diff(chunks.offsets) + +# Shape of the largest stored chunk, which sizes the scratch buffer. +chunk_buffer_shape(z::ZArray) = chunk_buffer_shape(chunkspec(z.metadata)) +chunk_buffer_shape(chunks::NTuple) = chunks +chunk_buffer_shape(chunks::DiskArrays.GridChunks) = DiskArrays.max_chunksize(chunks) + +function stored_chunk_ranges(md::AbstractMetadata, index::CartesianIndex{N}) where {N} + return stored_chunk_ranges(chunkspec(md), index) +end + +function stored_chunk_ranges(chunks::NTuple{N,Int}, index::CartesianIndex{N}) where {N} + return ntuple(N) do axis + chunk_size = chunks[axis] + chunk_index = index[axis] + ((chunk_index - 1) * chunk_size + 1):(chunk_index * chunk_size) + end +end + +function stored_chunk_ranges(chunks::DiskArrays.GridChunks{N}, index::CartesianIndex{N}) where {N} + return ntuple(axis -> stored_axis_range(chunks.chunks[axis], index[axis]), N) +end + +# As in a regular grid, the final chunk of a regular axis is stored at full size. +stored_axis_range(chunks::DiskArrays.RegularChunks, i::Int) = ((i - 1) * chunks.chunksize + 1):(i * chunks.chunksize) +stored_axis_range(chunks::DiskArrays.IrregularChunks, i::Int) = chunks[i] function boundint(r1, s2, o2) r2 = range(o2+1,length=s2) @@ -148,7 +178,7 @@ end function getchunkarray(z::ZArray{>:Missing}) # temporary workaround to use strings as data values - inner = fill(z.metadata.fill_value, z.metadata.chunks) + inner = fill(z.metadata.fill_value, chunk_buffer_shape(z)) a = SenMissArray(inner,z.metadata.fill_value) end _zero(T) = zero(T) @@ -156,7 +186,7 @@ _zero(T::Type{<:MaxLengthString}) = zero(T) _zero(T::Type{ASCIIChar}) = ASCIIChar(0) _zero(::Type{<:Vector{T}}) where T = T[] _zero(::Type{Char}) = Char(0) -getchunkarray(z::ZArray) = fill(_zero(eltype(z)), z.metadata.chunks) +getchunkarray(z::ZArray) = fill(_zero(eltype(z)), chunk_buffer_shape(z)) # Same as `getchunkarray` but skips the zero/fill_value-fill. Use only when # the caller guarantees the buffer will be fully overwritten before any read @@ -168,23 +198,33 @@ getchunkarray(z::ZArray) = fill(_zero(eltype(z)), z.metadata.chunks) # friends reject non-isbits eltypes). function getchunkarray_undef(z::ZArray{T}) where {T} Missing <: T && return getchunkarray(z) - return Array{T}(undef, z.metadata.chunks) + return Array{T}(undef, chunk_buffer_shape(z)) end -maybeinner(a::Array) = a +maybeinner(a::AbstractArray) = a maybeinner(a::SenMissArray) = a.x -resetbuffer!(fv,a::Array) = fv === nothing || fill!(a,fv) +function maybeinner(a::SubArray{<:Any,<:Any,<:SenMissArray}) + return view(parent(a).x, parentindices(a)...) +end +resetbuffer!(::Nothing,a::Array) = fill!(a,_zero(eltype(a))) +resetbuffer!(fv,a::Array) = fill!(a,fv) resetbuffer!(_,a::SenMissArray) = fill!(a,missing) +function chunk_buffer_view(a, chunk_shape) + size(a) == chunk_shape && return a + return view(a, Base.OneTo.(chunk_shape)...) +end + # Returns the chunk index when the call qualifies for the single-chunk fast # path: a plain `Array{T,N}` matching the chunk shape, `Missing <: T` false, # and only one chunk touched. `size(arr) == chunks` with `length(blockr) == 1` # implies the slice aligns with that chunk's full range, so no extra # coverage check is needed. function singlechunk_fastpath(arr, z::ZArray{T,N}, blockr::CartesianIndices{N}) where {T,N} - arr isa Array{T,N} && !(Missing <: T) && - size(arr) == z.metadata.chunks && length(blockr) == 1 || return nothing - return first(blockr) + arr isa Array{T,N} && !(Missing <: T) && length(blockr) == 1 || return nothing + chunk_index = first(blockr) + size(arr) == length.(stored_chunk_ranges(z.metadata, chunk_index)) || return nothing + return chunk_index end # Single-chunk full-overwrite write. Encodes `ain` and pushes the chunk to @@ -234,7 +274,8 @@ function readblock!(aout::AbstractArray{<:Any,N}, z::ZArray{<:Any, N}, r::Cartes output_base_offsets = map(i->first(i)-1,r.indices) # Determines which chunks are affected - blockr = CartesianIndices(map(trans_ind, r.indices, z.metadata.chunks)) + chunks = DiskArrays.eachchunk(z) + blockr = CartesianIndices(map(DiskArrays.findchunk, chunks.chunks, r.indices)) # Fast path: single-chunk full-read decodes directly into `aout`, skipping the readtask channel and scratch buffer. bI = singlechunk_fastpath(aout, z, blockr) if bI !== nothing @@ -260,11 +301,13 @@ function readblock!(aout::AbstractArray{<:Any,N}, z::ZArray{<:Any, N}, r::Cartes bI,chunk_compressed = take!(c) - current_chunk_offsets = map((s,i)->s*(i-1),size(a),Tuple(bI)) + chunk_ranges = stored_chunk_ranges(z.metadata, bI) + current_chunk_shape = length.(chunk_ranges) + current_chunk_offsets = first.(chunk_ranges) .- 1 + indranges = map(boundint, r.indices, current_chunk_shape, current_chunk_offsets) + active_buffer = chunk_buffer_view(a, current_chunk_shape) - indranges = map(boundint,r.indices,size(a),current_chunk_offsets) - - uncompress_to_output!(aout,output_base_offsets,z,chunk_compressed,current_chunk_offsets,a,indranges) + uncompress_to_output!(aout,output_base_offsets,z,chunk_compressed,current_chunk_offsets,active_buffer,indranges) nothing end finally @@ -279,7 +322,8 @@ function writeblock!(ain::AbstractArray{<:Any,N}, z::ZArray{<:Any, N}, r::Cartes z.writeable || error("Can not write to read-only ZArray") input_base_offsets = map(i->first(i)-1,r.indices) # Determines which chunks are affected - blockr = CartesianIndices(map(trans_ind, r.indices, z.metadata.chunks)) + chunks = DiskArrays.eachchunk(z) + blockr = CartesianIndices(map(DiskArrays.findchunk, chunks.chunks, r.indices)) # Fast path: single-chunk full-overwrite skips the readtask/writetask channels and the scratch buffer. bI = singlechunk_fastpath(ain, z, blockr) if bI !== nothing @@ -311,27 +355,26 @@ function writeblock!(ain::AbstractArray{<:Any,N}, z::ZArray{<:Any, N}, r::Cartes bI,chunk_compressed = take!(readchannel) - current_chunk_offsets = map((s,i)->s*(i-1),size(a),Tuple(bI)) + chunk_ranges = stored_chunk_ranges(z.metadata, bI) + current_chunk_shape = length.(chunk_ranges) + current_chunk_offsets = first.(chunk_ranges) .- 1 + indranges = map(boundint, r.indices, current_chunk_shape, current_chunk_offsets) + full_chunk = length.(indranges) == current_chunk_shape - indranges = map(boundint,r.indices,size(a),current_chunk_offsets) - - if isnothing(chunk_compressed) || (length.(indranges) != size(a)) + if isnothing(chunk_compressed) resetbuffer!(z.metadata.fill_value,a) end - curchunk = if length.(indranges) != size(a) - view(a,dotminus.(indranges,current_chunk_offsets)...) - else - a - end - - if chunk_compressed !== nothing - uncompress_raw!(a,z,chunk_compressed) + active_buffer = chunk_buffer_view(a, current_chunk_shape) + if chunk_compressed !== nothing && !full_chunk + uncompress_raw!(active_buffer,z,chunk_compressed) end + curchunk = view(active_buffer,dotminus.(indranges,current_chunk_offsets)...) curchunk .= view(ain,dotminus.(indranges,input_base_offsets)...) - put!(writechannel,bI=>compress_raw(maybeinner(a),z)) + active_inner = chunk_buffer_view(maybeinner(a), current_chunk_shape) + put!(writechannel,bI=>compress_raw(active_inner,z)) nothing end finally @@ -345,7 +388,7 @@ end DiskArrays.readblock!(a::ZArray,aout,i::AbstractUnitRange...) = readblock!(aout,a,CartesianIndices(i)) DiskArrays.writeblock!(a::ZArray,v,i::AbstractUnitRange...) = writeblock!(v,a,CartesianIndices(i)) DiskArrays.haschunks(::ZArray) = DiskArrays.Chunked() -DiskArrays.eachchunk(a::ZArray) = DiskArrays.GridChunks(a,a.metadata.chunks) +DiskArrays.eachchunk(a::ZArray) = chunkgrid(a.metadata) """ uncompress_raw!(a::DenseArray{T},z::ZArray{T,N},i::CartesianIndex{N}) @@ -359,7 +402,7 @@ function uncompress_raw!(a,z::ZArray{<:Any,N},curchunk) where N end fill!(a, z.metadata.fill_value) else - pipeline_decode!(get_pipeline(z.metadata), a, curchunk; fill_value=z.metadata.fill_value) + pipeline_decode!(get_pipeline(z.metadata), maybeinner(a), curchunk; fill_value=z.metadata.fill_value) end a end @@ -379,7 +422,6 @@ function uncompress_to_output!(aout,output_base_offsets,z,chunk_compressed,curre end function compress_raw(a,z) - length(a) == prod(z.metadata.chunks) || throw(DimensionMismatch("Array size does not equal chunk size")) pipeline_encode(get_pipeline(z.metadata), a, z.metadata.fill_value) end @@ -394,7 +436,7 @@ Creates a new empty zarr array with element type `T` and array dimensions `dims` * `name=""` name of the zarr array, defaults to the directory name * `zarr_format`=$(DV) Zarr format version (2 or 3) * `storagetype` determines the storage to use, current options are `DirectoryStore` or `DictStore` -* `chunks=dims` size of the individual array chunks, must be a tuple of length `length(dims)` +* `chunks=dims` chunk layout. Pass a tuple of sizes for regular chunks or a `DiskArrays.GridChunks` for a Zarr v3 rectilinear grid. * `fill_value=nothing` value to represent missing values * `fill_as_missing=false` set to `true` shall fillvalue s be converted to `missing`s * `filters`=filters to be applied @@ -452,9 +494,16 @@ function zcreate(::Type{T},storage::AbstractStore, end chunk_key_encoding = ChunkKeyEncoding(dimension_separator, default_prefix(v)) - length(dims) == length(chunks) || throw(DimensionMismatch("Dims must have the same length as chunks")) N = length(dims) - C = typeof(compressor) + if chunks isa Tuple + length(dims) == length(chunks) || throw(DimensionMismatch("Dims must have the same length as chunks")) + all(c -> c isa Integer, chunks) || throw(ArgumentError("Chunk sizes must be integers")) + chunks = map(Int, chunks) + elseif chunks isa DiskArrays.GridChunks + length(dims) == ndims(chunks) || throw(DimensionMismatch("Dims must have the same length as chunks")) + else + throw(ArgumentError("chunks must be a tuple of integers or a DiskArrays.GridChunks object")) + end # Create a dummy array to use with Metadata constructor # This allows us to leverage the multiple dispatch in Metadata constructors @@ -506,7 +555,7 @@ end Returns the Cartesian Indices of the chunks of a given ZArray """ -chunkindices(z::ZArray) = CartesianIndices(map((s, c) -> 1:ceil(Int, s/c), z.metadata.shape[], z.metadata.chunks)) +chunkindices(z::ZArray) = CartesianIndices(chunkcounts(chunkspec(z.metadata), size(z))) """ zzeros(T, dims...; kwargs... ) @@ -515,11 +564,21 @@ Creates a zarr array and initializes all values with zero. Accepts the same keyw """ function zzeros(T,dims...;kwargs...) z = zcreate(T,dims...;kwargs...) - as = zeros(T, z.metadata.chunks...) - data_encoded = compress_raw(as,z) p = z.path - if data_encoded !== nothing + chunks = DiskArrays.eachchunk(z) + if all(c -> c isa DiskArrays.RegularChunks, chunks.chunks) + as = zeros(T, chunk_buffer_shape(z)...) + data_encoded = compress_raw(as,z) + if data_encoded !== nothing + for i in chunkindices(z) + store_writechunk(z.storage, data_encoded, p, i, z.metadata.chunk_key_encoding) + end + end + else for i in chunkindices(z) + as = zeros(T, length.(stored_chunk_ranges(z.metadata, i))...) + data_encoded = compress_raw(as, z) + data_encoded === nothing || store_writechunk(z.storage, data_encoded, p, i, z.metadata.chunk_key_encoding) end end @@ -534,14 +593,27 @@ Resizes a `ZArray` to the new specified size. If the size along any of the axes is decreased, unused chunks will be deleted from the store. """ function Base.resize!(z::ZArray{T,N}, newsize::NTuple{N}) where {T,N} + z.writeable || error("Can not resize read-only ZArray") any(<(0), newsize) && throw(ArgumentError("Size must be positive")) oldsize = z.metadata.shape[] + oldchunks = chunkspec(z.metadata) + newchunks = resized_chunks(oldchunks, oldsize, newsize) z.metadata.shape[] = newsize + # Only a rectilinear grid depends on the shape. + oldchunks === newchunks || (z.metadata.chunks[] = newchunks) + # Write the metadata before deleting chunks, so a store that rejects the + # write (e.g. consolidated metadata) loses no data. + try + writemetadata(zarr_format(z), z.storage, z.path, z.metadata) + catch + z.metadata.shape[] = oldsize + oldchunks === newchunks || (z.metadata.chunks[] = oldchunks) + rethrow() + end #Check if array was shrunk if any(map(<,newsize, oldsize)) - prune_oob_chunks(z.storage, z.path, oldsize, newsize, z.metadata.chunks, z.metadata.chunk_key_encoding) + prune_oob_chunks(z.storage, z.path, chunkcounts(oldchunks, oldsize), chunkcounts(newchunks, newsize), z.metadata.chunk_key_encoding) end - writemetadata(zarr_format(z), z.storage, z.path, z.metadata) nothing end Base.resize!(z::ZArray, newsize::Integer...) = resize!(z,newsize) @@ -582,11 +654,38 @@ function Base.append!(z::ZArray{<:Any, N},a;dims = N) where N nothing end -function prune_oob_chunks(s::AbstractStore, path, oldsize, newsize, chunks, chunk_key_encoding) - dimstoshorten = findall(map(<,newsize, oldsize)) +# `length(RegularChunks)` is 1 for an empty axis, so count chunks from the sizes. +chunkcounts(chunks::NTuple, shape) = map(cld, shape, chunks) +function chunkcounts(chunks::DiskArrays.GridChunks, shape) + return map(chunks.chunks, shape) do c, axis_length + c isa DiskArrays.RegularChunks ? cld(axis_length, c.chunksize) : length(c) + end +end + +resized_chunks(chunks::NTuple, oldsize, newsize) = chunks +function resized_chunks(chunks::DiskArrays.GridChunks{N}, oldsize, newsize) where {N} + return DiskArrays.GridChunks(ntuple(N) do axis + resized_axis(chunks.chunks[axis], oldsize[axis], newsize[axis], axis) + end) +end + +resized_axis(c::DiskArrays.RegularChunks, oldlength, newlength, axis) = + DiskArrays.RegularChunks(c.chunksize, 0, newlength) +function resized_axis(c::DiskArrays.IrregularChunks, oldlength, newlength, axis) + newlength == oldlength && return c + # zarr-python shrinks by letting the edge lengths overflow the array, which + # `IrregularChunks` cannot describe. + newlength < oldlength && throw(ArgumentError( + "Cannot shrink axis $axis: shrinking along irregularly chunked axes is not supported")) + # As in zarr-python, one new chunk covers the added extent. + return DiskArrays.IrregularChunks(vcat(c.offsets, newlength)) +end + +function prune_oob_chunks(s::AbstractStore, path, oldcounts, newcounts, chunk_key_encoding) + dimstoshorten = findall(map(<,newcounts, oldcounts)) for idim in dimstoshorten - delrange = (fld1(newsize[idim],chunks[idim])+1):(fld1(oldsize[idim],chunks[idim])) - allchunkranges = map(i->1:fld1(oldsize[i],chunks[i]),1:length(oldsize)) + delrange = (newcounts[idim]+1):oldcounts[idim] + allchunkranges = map(n->1:n, oldcounts) r = (allchunkranges[1:idim-1]..., delrange, allchunkranges[idim+1:end]...) for cI in CartesianIndices(r) store_deletechunk(s, path, cI, chunk_key_encoding) diff --git a/ZarrCore/src/metadata.jl b/ZarrCore/src/metadata.jl index 9d7356a9..0b26ca1f 100644 --- a/ZarrCore/src/metadata.jl +++ b/ZarrCore/src/metadata.jl @@ -1,4 +1,5 @@ import Dates: Date, DateTime +import DiskArrays using DateTimes64: DateTime64, pydatetime_string, datetime_from_pystring """NumPy array protocol type string (typestr) format @@ -148,7 +149,7 @@ end "Construct Metadata based on your data" -function Metadata(A::AbstractArray{T,N}, chunks::NTuple{N,Int}, zarr_format=DV; +function Metadata(A::AbstractArray{T,N}, chunks, zarr_format=DV; node_type::String="array", compressor::C=default_compressor(), fill_value::Union{T, Nothing}=nothing, @@ -168,6 +169,27 @@ function Metadata(A::AbstractArray{T,N}, chunks::NTuple{N,Int}, zarr_format=DV; ) end +function validate_chunk_grid(shape::NTuple{N,Int}, chunks::DiskArrays.GridChunks{N}) where {N} + for (axis, (chunk_axis, axis_length)) in enumerate(zip(chunks.chunks, shape)) + DiskArrays.arraysize_from_chunksize(chunk_axis) == axis_length || + throw(DimensionMismatch( + "Chunk grid axis $axis describes length " * + "$(DiskArrays.arraysize_from_chunksize(chunk_axis)), expected $axis_length" + )) + if chunk_axis isa DiskArrays.RegularChunks && chunk_axis.offset != 0 + throw(ArgumentError("Zarr chunk grids cannot have a non-zero offset (axis $axis has offset $(chunk_axis.offset))")) + end + end + return chunks +end + +function regular_chunk_shape(shape::NTuple{N,Int}, chunks::DiskArrays.GridChunks{N}) where {N} + validate_chunk_grid(shape, chunks) + all(c -> c isa DiskArrays.RegularChunks, chunks.chunks) || + throw(ArgumentError("Irregular chunk grids require Zarr format 3")) + return map(c -> c.chunksize, chunks.chunks) +end + # V2 constructor function Metadata(A::AbstractArray{T,N}, chunks::NTuple{N,Int}, ::ZarrFormat{2}; node_type::String="array", @@ -193,6 +215,10 @@ function Metadata(A::AbstractArray{T,N}, chunks::NTuple{N,Int}, ::ZarrFormat{2}; ) end +function Metadata(A::AbstractArray{T,N}, chunks::DiskArrays.GridChunks{N}, v::ZarrFormat{2}; kwargs...) where {T,N} + return Metadata(A, regular_chunk_shape(size(A), chunks), v; kwargs...) +end + Metadata(s::Union{AbstractString, IO}, fill_as_missing) = Metadata(JSON.parse(s; dicttype=Dict{String,Any}), fill_as_missing) "Construct Metadata from Dict" diff --git a/ZarrCore/src/metadata3.jl b/ZarrCore/src/metadata3.jl index ffd6e860..01548bf7 100644 --- a/ZarrCore/src/metadata3.jl +++ b/ZarrCore/src/metadata3.jl @@ -63,38 +63,60 @@ function check_keys(d::AbstractDict, keys) end """Metadata for Zarr version 3 arrays""" -struct MetadataV3{T,N,P<:AbstractCodecPipeline,E<:AbstractChunkKeyEncoding} <: AbstractMetadata{T,N,E} +struct MetadataV3{T,N,P<:AbstractCodecPipeline,E<:AbstractChunkKeyEncoding,CT} <: AbstractMetadata{T,N,E} zarr_format::Int node_type::String shape::Base.RefValue{NTuple{N, Int}} - chunks::NTuple{N, Int} + chunks::Base.RefValue{CT} # replaced by `resize!` when the chunk grid is rectilinear dtype::Union{String, Dict{String, Any}} # data_type in v3 pipeline::P fill_value::Union{T, Nothing} chunk_key_encoding::E - function MetadataV3{T2,N,P,E}(zarr_format, node_type, shape, chunks, dtype, pipeline, fill_value, chunk_key_encoding) where {T2,N,P,E} + function MetadataV3{T2,N,P,E,CT}(zarr_format, node_type, shape, chunks::CT, dtype, pipeline, fill_value, chunk_key_encoding) where {T2,N,P,E,CT} zarr_format == 3 || throw(ArgumentError("MetadataV3 only functions if zarr_format == 3")) #Do some sanity checks to make sure we have a sane array any(<(0), shape) && throw(ArgumentError("Size must be positive")) - any(<(1), chunks) && throw(ArgumentError("Chunk size must be >= 1 along each dimension")) - new{T2,N,P,E}(zarr_format, node_type, Base.RefValue{NTuple{N,Int}}(shape), chunks, dtype, pipeline, fill_value, chunk_key_encoding) + validate_v3_chunks(shape, chunks) + new{T2,N,P,E,CT}(zarr_format, node_type, Base.RefValue{NTuple{N,Int}}(shape), Base.RefValue{CT}(chunks), dtype, pipeline, fill_value, chunk_key_encoding) end end +MetadataV3{T2,N,P,E}(zarr_format, node_type, shape, chunks, dtype, pipeline, fill_value, chunk_key_encoding) where {T2,N,P,E} = + MetadataV3{T2,N,P,E,typeof(chunks)}(zarr_format, node_type, shape, chunks, dtype, pipeline, fill_value, chunk_key_encoding) MetadataV3{T2,N,P}(args...) where {T2,N,P} = MetadataV3{T2,N,P,ChunkKeyEncoding}(args...) zarr_format(::MetadataV3) = ZarrFormat(Val(3)) +function validate_v3_chunks(shape::NTuple{N,Int}, chunks::NTuple{N,Int}) where {N} + any(<(1), chunks) && throw(ArgumentError("Chunk size must be >= 1 along each dimension")) + return chunks +end + +validate_v3_chunks(shape::NTuple{N,Int}, chunks::DiskArrays.GridChunks{N}) where {N} = + validate_chunk_grid(shape, chunks) + +canonical_v3_chunks(shape::NTuple{N,Int}, chunks::NTuple{N,Int}) where {N} = validate_v3_chunks(shape, chunks) + +# A grid of only regular axes is a `regular` chunk grid, stored as its chunk sizes. +function canonical_v3_chunks(shape::NTuple{N,Int}, chunks::DiskArrays.GridChunks{N}) where {N} + validate_v3_chunks(shape, chunks) + if all(c -> c isa DiskArrays.RegularChunks, chunks.chunks) + return map(c -> c.chunksize, chunks.chunks) + end + return chunks +end + """ Convenience constructor for MetadataV3 that builds the codec pipeline from `order` (translated to a TransposeCodec), `endian` (translated to a BytesCodec), and `compressor` (translated to bytes->bytes codecs). """ -function MetadataV3{T2,N}(zarr_format, node_type, shape::NTuple{N,Int}, chunks::NTuple{N,Int}, +function MetadataV3{T2,N}(zarr_format, node_type, shape::NTuple{N,Int}, chunks, dtype, fill_value; order::Char='C', endian::Symbol=:little, compressor=default_compressor(), chunk_key_encoding::E=ChunkKeyEncoding('/', true) ) where {T2, N, E} + chunks = canonical_v3_chunks(shape, chunks) T_base = Base.nonmissingtype(T2) array_array_codecs = if order == 'F' (Codecs.V3Codecs.TransposeCodec(ntuple(i -> N - i + 1, N)),) @@ -117,7 +139,7 @@ function Base.:(==)(m1::MetadataV3, m2::MetadataV3) m1.zarr_format == m2.zarr_format && m1.node_type == m2.node_type && m1.shape[] == m2.shape[] && - m1.chunks == m2.chunks && + m1.chunks[] == m2.chunks[] && m1.dtype == m2.dtype && m1.fill_value == m2.fill_value && m1.pipeline == m2.pipeline && @@ -170,6 +192,63 @@ get_order(md::MetadataV2) = md.order _sizeof(x) = sizeof(x) _sizeof(x::Type{String}) = 1 +function decode_rectilinear_axis(spec, axis_length::Int, axis::Int) + if spec isa Integer + spec >= 1 || throw(ArgumentError("Rectilinear chunk size on axis $axis must be >= 1")) + return DiskArrays.RegularChunks(Int(spec), 0, axis_length) + end + spec isa AbstractVector || + throw(ArgumentError("Rectilinear chunk_shapes entry for axis $axis must be an integer or a list")) + + chunk_sizes = Int[] + for entry in spec + if entry isa Integer + entry >= 1 || throw(ArgumentError("Rectilinear chunk sizes must be >= 1 (axis $axis)")) + push!(chunk_sizes, entry) + elseif entry isa AbstractVector && length(entry) == 2 && + entry[1] isa Integer && entry[2] isa Integer + value, count = entry + value >= 1 || throw(ArgumentError("Rectilinear chunk sizes must be >= 1 (axis $axis)")) + count >= 1 || throw(ArgumentError("Rectilinear run lengths must be >= 1 (axis $axis)")) + append!(chunk_sizes, fill(Int(value), Int(count))) + else + throw(ArgumentError( + "Invalid rectilinear chunk descriptor $entry on axis $axis; " * + "expected an integer or [chunk_size, count]" + )) + end + end + + covered = sum(chunk_sizes) + covered < axis_length && throw(DimensionMismatch( + "Rectilinear chunks on axis $axis cover $covered elements, expected $axis_length" + )) + # The spec allows this, but `IrregularChunks` cannot describe chunks that extend past the array. + covered > axis_length && throw(ArgumentError( + "Rectilinear chunks on axis $axis cover $covered elements, which overflows the axis length " * + "$axis_length; overflowing chunk edges are not supported" + )) + return DiskArrays.IrregularChunks(; chunksizes=chunk_sizes) +end + +encode_rectilinear_axis(chunks::DiskArrays.RegularChunks) = chunks.chunksize + +function encode_rectilinear_axis(chunks::DiskArrays.IrregularChunks) + sizes = diff(chunks.offsets) + encoded = Any[] + i = firstindex(sizes) + while i <= lastindex(sizes) + j = i + while j < lastindex(sizes) && sizes[j + 1] == sizes[i] + j += 1 + end + count = j - i + 1 + push!(encoded, count == 1 ? sizes[i] : Any[sizes[i], count]) + i = j + 1 + end + return encoded +end + function Metadata3(d::AbstractDict, fill_as_missing) check_keys(d, ("zarr_format", "node_type")) @@ -230,10 +309,23 @@ function Metadata3(d::AbstractDict, fill_as_missing) # Chunk Grid chunk_grid = d["chunk_grid"] if chunk_grid["name"] == "regular" - chunks = Int.(chunk_grid["configuration"]["chunk_shape"]) - if length(shape) != length(chunks) - throw(ArgumentError("Shape has rank $(length(shape)) which does not match the chunk_shape rank of $(length(chunks))")) + chunk_shape = Int.(chunk_grid["configuration"]["chunk_shape"]) + if length(shape) != length(chunk_shape) + throw(ArgumentError("Shape has rank $(length(shape)) which does not match the chunk_shape rank of $(length(chunk_shape))")) end + chunks = NTuple{length(chunk_shape),Int}(reverse(chunk_shape)) + elseif chunk_grid["name"] == "rectilinear" + configuration = chunk_grid["configuration"] + get(configuration, "kind", nothing) == "inline" || + throw(ArgumentError("Only rectilinear chunk grids of kind \"inline\" are supported")) + chunk_shapes = get(configuration, "chunk_shapes", nothing) + chunk_shapes isa AbstractVector || + throw(ArgumentError("Rectilinear chunk grid configuration must contain chunk_shapes")) + length(chunk_shapes) == length(shape) || throw(DimensionMismatch( + "Shape has rank $(length(shape)) which does not match the chunk_shapes rank of $(length(chunk_shapes))" + )) + axes_c = map(decode_rectilinear_axis, chunk_shapes, shape, eachindex(shape)) + chunks = DiskArrays.GridChunks(reverse(axes_c)...) else throw(ArgumentError("Unknown chunk_grid of name, $(chunk_grid["name"])")) end @@ -259,7 +351,7 @@ function Metadata3(d::AbstractDict, fill_as_missing) zarr_format, node_type, NTuple{N, Int}(shape) |> reverse, - NTuple{N, Int}(chunks) |> reverse, + chunks, typestr3(T), pipeline, fv, @@ -268,7 +360,7 @@ function Metadata3(d::AbstractDict, fill_as_missing) end "Construct MetadataV3 based on your data" -function Metadata3(A::AbstractArray{T, N}, chunks::NTuple{N, Int}; +function Metadata3(A::AbstractArray{T, N}, chunks; node_type::String="array", compressor=default_compressor(), fill_value::Union{T, Nothing}=nothing, @@ -282,6 +374,7 @@ function Metadata3(A::AbstractArray{T, N}, chunks::NTuple{N, Int}; if fill_value === nothing fill_value = zero(T) end + chunks = canonical_v3_chunks(size(A), chunks) return MetadataV3{T2, N}( 3, node_type, @@ -297,12 +390,23 @@ function Metadata3(A::AbstractArray{T, N}, chunks::NTuple{N, Int}; end function lower3(md::MetadataV3{T}) where T - chunk_grid = Dict{String,Any}( - "name" => "regular", - "configuration" => Dict{String,Any}( - "chunk_shape" => md.chunks |> reverse + chunks = md.chunks[] + chunk_grid = if chunks isa DiskArrays.GridChunks + Dict{String,Any}( + "name" => "rectilinear", + "configuration" => Dict{String,Any}( + "kind" => "inline", + "chunk_shapes" => Any[encode_rectilinear_axis(c) for c in reverse(chunks.chunks)] + ) ) - ) + else + Dict{String,Any}( + "name" => "regular", + "configuration" => Dict{String,Any}( + "chunk_shape" => reverse(chunks) + ) + ) + end # chunk_key_encoding chunk_key_encoding = lower_chunk_key_encoding(md.chunk_key_encoding) @@ -321,7 +425,7 @@ function lower3(md::MetadataV3{T}) where T ) end -function Metadata(A::AbstractArray{T,N}, chunks::NTuple{N,Int}, ::ZarrFormat{3}; +function Metadata(A::AbstractArray{T,N}, chunks::Union{NTuple{N,Int},DiskArrays.GridChunks{N}}, ::ZarrFormat{3}; node_type::String="array", compressor::C=default_compressor(), fill_value::Union{T, Nothing}=nothing, diff --git a/docs/src/.vitepress/config.mts b/docs/src/.vitepress/config.mts index 89761cb2..b2087c2e 100644 --- a/docs/src/.vitepress/config.mts +++ b/docs/src/.vitepress/config.mts @@ -24,6 +24,7 @@ const userGuideItems = [ // { text: 'Data Types', link: '/UserGuide/data_types' }, // { text: 'Codecs & Performance', link: '/UserGuide/performance' }, { text: 'Operations', link: '/UserGuide/operations'}, + { text: 'Chunking', link: '/UserGuide/chunking' }, // { text: 'Sharding', link: '/UserGuide/sharding' }, { text: 'Missing Values', link: '/UserGuide/missings' }, ] @@ -169,4 +170,4 @@ export default defineConfig({ 'Made with DocumenterVitepress.jl
Released under the MIT License. Powered by the Julia Programming Language.
', }, } -}) \ No newline at end of file +}) diff --git a/docs/src/UserGuide/chunking.md b/docs/src/UserGuide/chunking.md new file mode 100644 index 00000000..4280121a --- /dev/null +++ b/docs/src/UserGuide/chunking.md @@ -0,0 +1,109 @@ +# Regular and Rectilinear Chunking + +Zarr arrays divide their data into independently encoded chunks. With regular +chunking, every chunk has the same nominal size along an axis. This is the +layout selected by a tuple-valued `chunks` keyword: + +````jldoctest regular-chunks +julia> using Zarr + +julia> z = zcreate(Int, 100, 80; chunks=(10, 20)) +ZArray{Int64} of size 100 x 80 +```` + +## Rectilinear grids + +Zarr v3 rectilinear grids allow chunk sizes to vary independently along each +axis. Zarr.jl represents these grids with the chunk types from DiskArrays.jl: + +````jldoctest rectilinear-chunks +julia> using Zarr + +julia> using DiskArrays: GridChunks, IrregularChunks, RegularChunks + +julia> import DiskArrays + +julia> chunks = GridChunks( + RegularChunks(2, 0, 5), + IrregularChunks(chunksizes=[3, 4, 5, 6, 2]), + ); + +julia> z = zcreate(Int, 5, 20; zarr_format=3, chunks) +ZArray{Int64} of size 5 x 20 + +julia> z[:, :] = reshape(1:100, 5, 20); + +julia> z[:, :] == reshape(1:100, 5, 20) +true +```` + +The first axis uses regular chunks of two elements (with a one-element final +chunk). The second uses chunks of 3, 4, 5, 6, and 2 elements. Reads and writes +may cross any number of these boundaries without changing the indexing API. + +`zcreate`, `zzeros`, and the `ZArray(data; chunks=...)` constructor accept a +`GridChunks` value. The grid must: + +- have one chunk specification per array axis; +- start at offset zero; and +- describe the complete array extent on every axis. + +Irregular grids require `zarr_format=3`; Zarr v2 metadata cannot represent +them. + +## Metadata and interoperability + +Zarr.jl serializes an irregular grid using Zarr v3's `rectilinear` chunk-grid +extension with an inline `chunk_shapes` configuration. A regular axis is +written as its chunk size, an irregular axis as its list of edge lengths with +repeated sizes run-length encoded. This layout can be exchanged with +zarr-python. + +Each storage chunk is encoded at its own size rather than padded to a global +maximum. As in a regular grid, the final chunk of a regular axis is stored at +the full chunk size even when the axis length is not a multiple of it. + +The specification also allows a list of edge lengths to overflow the array +extent. Zarr.jl does not support this and throws an `ArgumentError` when +opening such an array. + +The grid survives reopening: + +````jldoctest rectilinear-chunks +julia> dir = joinpath(mktempdir(), "rectilinear.zarr"); + +julia> z = zcreate(Int, 5, 20; zarr_format=3, chunks, path=dir); + +julia> reopened = zopen(dir); + +julia> DiskArrays.eachchunk(reopened) == chunks +true +```` + +## Resizing + +`resize!` and `append!` work on rectilinear arrays. A regular axis keeps its +chunk size. Growing an irregular axis appends one chunk covering the added +extent, as zarr-python does: + +````jldoctest rectilinear-chunks +julia> append!(z, fill(7, 3, 20); dims=1) + +julia> append!(z, fill(9, 8, 4); dims=2) + +julia> size(z) +(8, 24) + +julia> length.(DiskArrays.eachchunk(z).chunks[2]) +6-element Vector{Int64}: + 3 + 4 + 5 + 6 + 2 + 4 +```` + +Shrinking an irregular axis is not supported: zarr-python does it by letting +the edge lengths overflow the array, which Zarr.jl cannot represent (see +above). Zarr.jl throws an `ArgumentError` before modifying the array. diff --git a/docs/src/get_started.md b/docs/src/get_started.md index f9211381..9034a91a 100644 --- a/docs/src/get_started.md +++ b/docs/src/get_started.md @@ -80,6 +80,23 @@ z = ZArray(rand(Float64, 100, 100)) zinfo(z) ``` +For irregular chunking—where chunk sizes vary along an axis—pass a +`DiskArrays.GridChunks` object and create a Zarr v3 array: + +```@example irregular-chunks +using Zarr +using DiskArrays: GridChunks, IrregularChunks, RegularChunks + +chunks = GridChunks( + RegularChunks(2, 0, 5), + IrregularChunks(chunksizes=[3, 4, 5, 6, 2]), +) +z = zcreate(Int, 5, 20; zarr_format=3, chunks) +``` + +See [Chunking](UserGuide/chunking.md) for metadata, interoperability, and +resizing details. + ## Reading and Writing ```@example rw using Zarr