Three fixed entries (reading, writing, LZ4 over 256 MiB) and the open limits entry; the selection-read entry loses its fill-value case; the FileEditor limits gain the refused rewrites; the README tables name what is supported and what is not. Co-Authored-By: Claude Opus 5.5 (1M context) <[email protected]>
111 lines
4.5 KiB
Python
111 lines
4.5 KiB
Python
"""Write HDF5 files whose chunks are 4 GiB or more, one dataset per chunk index.
|
|
|
|
python gen_huge_chunks.py filtered OUT.h5 # huge_chunks_filtered.h5
|
|
python gen_huge_chunks.py unfiltered OUT.h5 # generated at test time
|
|
|
|
libhdf5 2.x writes a chunk of more than 0xFFFFFFFF bytes with layout message
|
|
version 5 (`H5D__chunk_construct`: "chunk size > 4GB requires
|
|
H5F_LIBVER_V200"), so never with a version-1 B-tree; a filtered chunk index
|
|
element of a version-5 layout stores the chunk's size in "size of lengths"
|
|
bytes (8). Every dataset is `<f8` with chunks of N = 2**29 + 1 elements
|
|
(4 GiB + 8 bytes):
|
|
|
|
single shape (N,), chunks (N,) Single Chunk
|
|
implicit shape (2N,), early allocation (unfiltered) Implicit
|
|
farray shape (N+10,) Fixed Array
|
|
earray shape (N+10,), maxshape (None,) Extensible Array
|
|
btree2 shape (2, N+10), chunks (1, N),
|
|
maxshape (None, None) v2 B-tree
|
|
|
|
Only a few elements are written: `d[0:10] = 0..9` and, where there is a
|
|
second chunk along the axis, `d[N:N+10] = 100..109` (btree2: row 0 as that,
|
|
row 1 `200..209` and `300..309`; implicit: `d[N:N+10]` and
|
|
`d[2N-10:2N] = 500..509`). Everything else reads as the fill value.
|
|
|
|
`filtered` (the committed fixture, no `implicit`: that index is never
|
|
filtered): deflate level 9 twice in the pipeline, fill value -1.0. One
|
|
deflate leaves a 4 GiB chunk of a repeated 8-byte pattern at about 6 MiB;
|
|
the second pass takes that to about 14 KiB, so the file is small while each
|
|
chunk still inflates to 4 GiB + 8 bytes.
|
|
|
|
`unfiltered`: fill time "never" and the default fill value, so libhdf5
|
|
writes only the elements written (an unfiltered chunk larger than the chunk
|
|
cache is written in place) and the file is sparse: tens of GiB long but a few
|
|
blocks on disk. Unwritten elements read as whatever the file holds there:
|
|
zeros. Do not put it on tmpfs, which is memory. Its Single Chunk is
|
|
allocated early (see `make`).
|
|
|
|
libhdf5 holds a whole filtered chunk in memory while it writes it, so the
|
|
`filtered` run needs about 4 GiB of memory and 100 s (tank).
|
|
|
|
Generated 2026-09-28 with h5py 3.16.0 (libhdf5 2.0.0).
|
|
"""
|
|
import sys
|
|
|
|
import h5py
|
|
import numpy as np
|
|
|
|
N = 2**29 + 1
|
|
UNLIM = h5py.h5s.UNLIMITED
|
|
|
|
|
|
def make(fid, name, index, filtered):
|
|
dcpl = h5py.h5p.create(h5py.h5p.DATASET_CREATE)
|
|
if index == "single":
|
|
shape, maxshape, chunk = (N,), (N,), (N,)
|
|
elif index == "implicit":
|
|
shape, maxshape, chunk = (2 * N,), (2 * N,), (N,)
|
|
dcpl.set_alloc_time(h5py.h5d.ALLOC_TIME_EARLY)
|
|
elif index == "farray":
|
|
shape, maxshape, chunk = (N + 10,), (N + 10,), (N,)
|
|
elif index == "earray":
|
|
shape, maxshape, chunk = (N + 10,), (UNLIM,), (N,)
|
|
elif index == "btree2":
|
|
shape, maxshape, chunk = (2, N + 10), (UNLIM, UNLIM), (1, N)
|
|
dcpl.set_chunk(chunk)
|
|
if filtered:
|
|
dcpl.set_deflate(9)
|
|
dcpl.set_deflate(9)
|
|
dcpl.set_fill_value(np.array(-1.0, dtype="<f8"))
|
|
else:
|
|
dcpl.set_fill_time(h5py.h5d.FILL_TIME_NEVER)
|
|
if index == "single":
|
|
# libhdf5 2.0.0 drops a write to an unallocated unfiltered
|
|
# Single Chunk this large when the fill time is "never" (the
|
|
# dataset stays unallocated and reads zeros); allocated at
|
|
# creation, the write lands in place.
|
|
dcpl.set_alloc_time(h5py.h5d.ALLOC_TIME_EARLY)
|
|
space = h5py.h5s.create_simple(shape, maxshape)
|
|
dsid = h5py.h5d.create(fid, name.encode(), h5py.h5t.IEEE_F64LE, space, dcpl=dcpl)
|
|
ds = h5py.Dataset(dsid)
|
|
if len(shape) == 2:
|
|
ds[0, 0:10] = np.arange(10.0)
|
|
ds[0, N : N + 10] = np.arange(100.0, 110.0)
|
|
ds[1, 0:10] = np.arange(200.0, 210.0)
|
|
ds[1, N : N + 10] = np.arange(300.0, 310.0)
|
|
else:
|
|
ds[0:10] = np.arange(10.0)
|
|
if shape[0] > N:
|
|
ds[N : N + 10] = np.arange(100.0, 110.0)
|
|
if index == "implicit":
|
|
ds[2 * N - 10 : 2 * N] = np.arange(500.0, 510.0)
|
|
dsid.close()
|
|
|
|
|
|
def main():
|
|
mode, out = sys.argv[1], sys.argv[2]
|
|
filtered = {"filtered": True, "unfiltered": False}[mode]
|
|
fapl = h5py.h5p.create(h5py.h5p.FILE_ACCESS)
|
|
fapl.set_libver_bounds(h5py.h5f.LIBVER_EARLIEST, h5py.h5f.LIBVER_V200)
|
|
fid = h5py.h5f.create(out.encode(), h5py.h5f.ACC_TRUNC, fapl=fapl)
|
|
indexes = ["single", "farray", "earray", "btree2"]
|
|
if not filtered:
|
|
indexes.insert(1, "implicit")
|
|
for index in indexes:
|
|
make(fid, index, index, filtered)
|
|
fid.close()
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main()
|