"""Write HDF5 files whose chunks are 4 GiB or more, one dataset per chunk index. python gen_huge_chunks.py filtered OUT.h5 # huge_chunks_filtered.h5 python gen_huge_chunks.py unfiltered OUT.h5 # generated at test time python gen_huge_chunks.py dims OUT.h5 # huge_chunk_dims.h5 python gen_huge_chunks.py dims-unfiltered OUT.h5 # generated at test time `dims` and `dims-unfiltered` (see `make_dims`) hold `u1` datasets whose chunk dimension is N8 = 2**32 + 7: a chunk dimension of 2**32 or more, which a layout message of version 5 stores in 5 bytes. libhdf5 2.x writes a chunk of more than 0xFFFFFFFF bytes with layout message version 5 (`H5D__chunk_construct`: "chunk size > 4GB requires H5F_LIBVER_V200"), so never with a version-1 B-tree; a filtered chunk index element of a version-5 layout stores the chunk's size in "size of lengths" bytes (8). Every dataset is ` N: ds[N : N + 10] = np.arange(100.0, 110.0) if index == "implicit": ds[2 * N - 10 : 2 * N] = np.arange(500.0, 510.0) dsid.close() N8 = 2**32 + 7 def make_dims(fid, name, index, filtered): """A `u1` dataset whose chunks are N8 elements (a chunk dimension of 2**32 or more). `d[0:10] = 0..9` and `d[N8-10:N8] = 100..109`; where there is a second chunk (`farray`, `earray`: shape N8 + 10), `d[N8:N8+10] = 200..209`. filtered (`dims`, committed): deflate level 9 twice, fill value 7; `single` (Single Chunk) and `earray` (Extensible Array, maxshape None). Writing it holds a 4 GiB chunk: 4.1 GiB peak resident memory and about a minute (tank, 2026-09-29). unfiltered (`dims-unfiltered`): fill time "never", sparse; `single` (allocated early, as in `make`) and `farray` (Fixed Array). """ dcpl = h5py.h5p.create(h5py.h5p.DATASET_CREATE) if index == "single": shape, maxshape = (N8,), (N8,) elif index == "farray": shape, maxshape = (N8 + 10,), (N8 + 10,) elif index == "earray": shape, maxshape = (N8 + 10,), (UNLIM,) dcpl.set_chunk((N8,)) if filtered: dcpl.set_deflate(9) dcpl.set_deflate(9) dcpl.set_fill_value(np.array(7, dtype="u1")) else: dcpl.set_fill_time(h5py.h5d.FILL_TIME_NEVER) if index == "single": dcpl.set_alloc_time(h5py.h5d.ALLOC_TIME_EARLY) space = h5py.h5s.create_simple(shape, maxshape) dsid = h5py.h5d.create(fid, name.encode(), h5py.h5t.STD_U8LE, space, dcpl=dcpl) ds = h5py.Dataset(dsid) ds[0:10] = np.arange(10, dtype="u1") ds[N8 - 10 : N8] = np.arange(100, 110, dtype="u1") if shape[0] > N8: ds[N8 : N8 + 10] = np.arange(200, 210, dtype="u1") dsid.close() def main(): mode, out = sys.argv[1], sys.argv[2] fapl = h5py.h5p.create(h5py.h5p.FILE_ACCESS) fapl.set_libver_bounds(h5py.h5f.LIBVER_EARLIEST, h5py.h5f.LIBVER_V200) fid = h5py.h5f.create(out.encode(), h5py.h5f.ACC_TRUNC, fapl=fapl) if mode in ("dims", "dims-unfiltered"): filtered = mode == "dims" for index in ["single", "earray"] if filtered else ["single", "farray"]: make_dims(fid, index, index, filtered) fid.close() return filtered = {"filtered": True, "unfiltered": False}[mode] indexes = ["single", "farray", "earray", "btree2"] if not filtered: indexes.insert(1, "implicit") for index in indexes: make(fid, index, index, filtered) fid.close() if __name__ == "__main__": main()