- DataLayout::Chunked::chunk_dimensions is Vec<u64> (was Vec<u32>), and the chunk index readers, writers and serializers take &[u64]: layout messages of version 4/5 store each dimension in up to 8 bytes, and libhdf5 2.x writes dimensions of 2^32 or more (layout version 5). Such dimensions were refused on read (InvalidChunkDimensions) and write. A version-3 layout (4-byte dimensions) is never written for them; a chunk whose size overflows 64 bits is refused when opened. - The file writer lays chunked datasets out as pieces referring to the chunks instead of copying them into one buffer per pass, and an unfiltered chunk that is a contiguous run of the dataset's data (a dataset stored as one chunk of its shape, row blocks) borrows it; a filtered one is compressed straight from it. Contiguous datasets are not copied either. FileWriter::finish_with streams the file to a callback; FileBuilder::write uses it, so the file is never assembled in memory. DatasetBuilder::with_u8_data_owned takes the data without a copy. Peak RSS writing one unfiltered 1 GiB chunk (with_u8_data_owned + write): 5.0 GiB before, 1.0 GiB after; with_u8_data: 6.0 -> 2.0 GiB. - extract_chunk no longer panics on data shorter than the shape. - Tests: huge_chunk_dims.h5 fixture (libhdf5 2.0.0 via h5py 3.16, u8 chunks of 2^32 + 7), and opt-in end-to-end tests of chunk dims >= 2^32 (filtered and unfiltered, read and written, h5py and h5dump 2.2.0), of an unfiltered 4 GiB+ chunk written by clawhdf5, and of LZ4/Zstd chunks of that size; example write_one_chunk for memory measurements. Co-Authored-By: Claude Opus 5.5 (1M context) <[email protected]>
166 lines
6.8 KiB
Python
166 lines
6.8 KiB
Python
"""Write HDF5 files whose chunks are 4 GiB or more, one dataset per chunk index.
|
|
|
|
python gen_huge_chunks.py filtered OUT.h5 # huge_chunks_filtered.h5
|
|
python gen_huge_chunks.py unfiltered OUT.h5 # generated at test time
|
|
python gen_huge_chunks.py dims OUT.h5 # huge_chunk_dims.h5
|
|
python gen_huge_chunks.py dims-unfiltered OUT.h5 # generated at test time
|
|
|
|
`dims` and `dims-unfiltered` (see `make_dims`) hold `u1` datasets whose
|
|
chunk dimension is N8 = 2**32 + 7: a chunk dimension of 2**32 or more,
|
|
which a layout message of version 5 stores in 5 bytes.
|
|
|
|
libhdf5 2.x writes a chunk of more than 0xFFFFFFFF bytes with layout message
|
|
version 5 (`H5D__chunk_construct`: "chunk size > 4GB requires
|
|
H5F_LIBVER_V200"), so never with a version-1 B-tree; a filtered chunk index
|
|
element of a version-5 layout stores the chunk's size in "size of lengths"
|
|
bytes (8). Every dataset is `<f8` with chunks of N = 2**29 + 1 elements
|
|
(4 GiB + 8 bytes):
|
|
|
|
single shape (N,), chunks (N,) Single Chunk
|
|
implicit shape (2N,), early allocation (unfiltered) Implicit
|
|
farray shape (N+10,) Fixed Array
|
|
earray shape (N+10,), maxshape (None,) Extensible Array
|
|
btree2 shape (2, N+10), chunks (1, N),
|
|
maxshape (None, None) v2 B-tree
|
|
|
|
Only a few elements are written: `d[0:10] = 0..9` and, where there is a
|
|
second chunk along the axis, `d[N:N+10] = 100..109` (btree2: row 0 as that,
|
|
row 1 `200..209` and `300..309`; implicit: `d[N:N+10]` and
|
|
`d[2N-10:2N] = 500..509`). Everything else reads as the fill value.
|
|
|
|
`filtered` (the committed fixture, no `implicit`: that index is never
|
|
filtered): deflate level 9 twice in the pipeline, fill value -1.0. One
|
|
deflate leaves a 4 GiB chunk of a repeated 8-byte pattern at about 6 MiB;
|
|
the second pass takes that to about 14 KiB, so the file is small while each
|
|
chunk still inflates to 4 GiB + 8 bytes.
|
|
|
|
`unfiltered`: fill time "never" and the default fill value, so libhdf5
|
|
writes only the elements written (an unfiltered chunk larger than the chunk
|
|
cache is written in place) and the file is sparse: tens of GiB long but a few
|
|
blocks on disk. Unwritten elements read as whatever the file holds there:
|
|
zeros. Do not put it on tmpfs, which is memory. Its Single Chunk is
|
|
allocated early (see `make`).
|
|
|
|
libhdf5 holds a whole filtered chunk in memory while it writes it, so the
|
|
`filtered` run needs about 4 GiB of memory and 100 s (tank).
|
|
|
|
huge_chunks_filtered.h5 was generated 2026-09-28 and huge_chunk_dims.h5
|
|
2026-09-29, both with h5py 3.16.0 (libhdf5 2.0.0).
|
|
"""
|
|
import sys
|
|
|
|
import h5py
|
|
import numpy as np
|
|
|
|
N = 2**29 + 1
|
|
UNLIM = h5py.h5s.UNLIMITED
|
|
|
|
|
|
def make(fid, name, index, filtered):
|
|
dcpl = h5py.h5p.create(h5py.h5p.DATASET_CREATE)
|
|
if index == "single":
|
|
shape, maxshape, chunk = (N,), (N,), (N,)
|
|
elif index == "implicit":
|
|
shape, maxshape, chunk = (2 * N,), (2 * N,), (N,)
|
|
dcpl.set_alloc_time(h5py.h5d.ALLOC_TIME_EARLY)
|
|
elif index == "farray":
|
|
shape, maxshape, chunk = (N + 10,), (N + 10,), (N,)
|
|
elif index == "earray":
|
|
shape, maxshape, chunk = (N + 10,), (UNLIM,), (N,)
|
|
elif index == "btree2":
|
|
shape, maxshape, chunk = (2, N + 10), (UNLIM, UNLIM), (1, N)
|
|
dcpl.set_chunk(chunk)
|
|
if filtered:
|
|
dcpl.set_deflate(9)
|
|
dcpl.set_deflate(9)
|
|
dcpl.set_fill_value(np.array(-1.0, dtype="<f8"))
|
|
else:
|
|
dcpl.set_fill_time(h5py.h5d.FILL_TIME_NEVER)
|
|
if index == "single":
|
|
# libhdf5 2.0.0 drops a write to an unallocated unfiltered
|
|
# Single Chunk this large when the fill time is "never" (the
|
|
# dataset stays unallocated and reads zeros); allocated at
|
|
# creation, the write lands in place.
|
|
dcpl.set_alloc_time(h5py.h5d.ALLOC_TIME_EARLY)
|
|
space = h5py.h5s.create_simple(shape, maxshape)
|
|
dsid = h5py.h5d.create(fid, name.encode(), h5py.h5t.IEEE_F64LE, space, dcpl=dcpl)
|
|
ds = h5py.Dataset(dsid)
|
|
if len(shape) == 2:
|
|
ds[0, 0:10] = np.arange(10.0)
|
|
ds[0, N : N + 10] = np.arange(100.0, 110.0)
|
|
ds[1, 0:10] = np.arange(200.0, 210.0)
|
|
ds[1, N : N + 10] = np.arange(300.0, 310.0)
|
|
else:
|
|
ds[0:10] = np.arange(10.0)
|
|
if shape[0] > N:
|
|
ds[N : N + 10] = np.arange(100.0, 110.0)
|
|
if index == "implicit":
|
|
ds[2 * N - 10 : 2 * N] = np.arange(500.0, 510.0)
|
|
dsid.close()
|
|
|
|
|
|
N8 = 2**32 + 7
|
|
|
|
|
|
def make_dims(fid, name, index, filtered):
|
|
"""A `u1` dataset whose chunks are N8 elements (a chunk dimension of
|
|
2**32 or more). `d[0:10] = 0..9` and `d[N8-10:N8] = 100..109`; where
|
|
there is a second chunk (`farray`, `earray`: shape N8 + 10),
|
|
`d[N8:N8+10] = 200..209`.
|
|
|
|
filtered (`dims`, committed): deflate level 9 twice, fill value 7;
|
|
`single` (Single Chunk) and `earray` (Extensible Array, maxshape None).
|
|
Writing it holds a 4 GiB chunk: 4.1 GiB peak resident memory and about
|
|
a minute (tank, 2026-09-29).
|
|
unfiltered (`dims-unfiltered`): fill time "never", sparse; `single`
|
|
(allocated early, as in `make`) and `farray` (Fixed Array).
|
|
"""
|
|
dcpl = h5py.h5p.create(h5py.h5p.DATASET_CREATE)
|
|
if index == "single":
|
|
shape, maxshape = (N8,), (N8,)
|
|
elif index == "farray":
|
|
shape, maxshape = (N8 + 10,), (N8 + 10,)
|
|
elif index == "earray":
|
|
shape, maxshape = (N8 + 10,), (UNLIM,)
|
|
dcpl.set_chunk((N8,))
|
|
if filtered:
|
|
dcpl.set_deflate(9)
|
|
dcpl.set_deflate(9)
|
|
dcpl.set_fill_value(np.array(7, dtype="u1"))
|
|
else:
|
|
dcpl.set_fill_time(h5py.h5d.FILL_TIME_NEVER)
|
|
if index == "single":
|
|
dcpl.set_alloc_time(h5py.h5d.ALLOC_TIME_EARLY)
|
|
space = h5py.h5s.create_simple(shape, maxshape)
|
|
dsid = h5py.h5d.create(fid, name.encode(), h5py.h5t.STD_U8LE, space, dcpl=dcpl)
|
|
ds = h5py.Dataset(dsid)
|
|
ds[0:10] = np.arange(10, dtype="u1")
|
|
ds[N8 - 10 : N8] = np.arange(100, 110, dtype="u1")
|
|
if shape[0] > N8:
|
|
ds[N8 : N8 + 10] = np.arange(200, 210, dtype="u1")
|
|
dsid.close()
|
|
|
|
|
|
def main():
|
|
mode, out = sys.argv[1], sys.argv[2]
|
|
fapl = h5py.h5p.create(h5py.h5p.FILE_ACCESS)
|
|
fapl.set_libver_bounds(h5py.h5f.LIBVER_EARLIEST, h5py.h5f.LIBVER_V200)
|
|
fid = h5py.h5f.create(out.encode(), h5py.h5f.ACC_TRUNC, fapl=fapl)
|
|
if mode in ("dims", "dims-unfiltered"):
|
|
filtered = mode == "dims"
|
|
for index in ["single", "earray"] if filtered else ["single", "farray"]:
|
|
make_dims(fid, index, index, filtered)
|
|
fid.close()
|
|
return
|
|
filtered = {"filtered": True, "unfiltered": False}[mode]
|
|
indexes = ["single", "farray", "earray", "btree2"]
|
|
if not filtered:
|
|
indexes.insert(1, "implicit")
|
|
for index in indexes:
|
|
make(fid, index, index, filtered)
|
|
fid.close()
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main()
|