"""Write HDF5 and NetCDF-4 test files with h5py/netCDF4, and what libhdf5 reads back from them, for the clawhdf5-wasm tests. python make_fixture.py OUT_DIR writes OUT_DIR/fixture.h5, OUT_DIR/fixture.nc and OUT_DIR/expected.json, and the limit-test files OUT_DIR/limits.h5 and OUT_DIR/hostile_vl.h5 (see write_limits). Both the Rust test (crates/clawhdf5-wasm/tests/h5py_interop.rs, native) and the Node test (test.mjs, the built wasm package) compare against the same expected.json, so the two check the same values. Every expected value comes from h5py reading the file back (numpy slicing for hyperslabs), never from the arrays that were written. Integers are encoded as strings so JSON.parse keeps 64-bit values exact. """ import json import os import sys import warnings from pathlib import Path import h5py import netCDF4 import numpy as np try: # registers the LZ4/Zstd filters with libhdf5; optional import hdf5plugin except ImportError: hdf5plugin = None # netCDF4 1.7 trips numpy 2.5's shape-setting deprecation on assignment. warnings.filterwarnings("ignore", category=DeprecationWarning) out = Path(sys.argv[1]) out.mkdir(parents=True, exist_ok=True) h5 = out / "fixture.h5" nc = out / "fixture.nc" rng = np.random.default_rng(7) with h5py.File(h5, "w") as f: f.attrs["title"] = "wasm fixture" f.attrs["version"] = np.int64(3) f.attrs["scale"] = np.array([0.5, 2.0]) f.attrs["big"] = np.uint64(2**63 + 5) f.attrs.create("vlen_note", "héllo", dtype=h5py.string_dtype()) # No plain JavaScript form: listed with value null and its type. f.attrs["origin"] = np.array((1.5, 2), dtype=[("x", "f4")) f.create_dataset("f16", data=np.array([0.5, -1.25, 65504], dtype=" 2 else 1] + [2] * (obj.ndim - 1) count = [min(c, (n - s - 1) // st + 1) for c, s, st, n in zip(count, start, stride, obj.shape)] return (start, count, stride) json.dump({"fixture.h5": describe(h5), "fixture.nc": describe(nc)}, open(out / "expected.json", "w"), indent=1, ensure_ascii=False) HUGE_U8 = 2**28 + 1024 # The collection size hostile_vl.h5 claims: past 2 GiB, which a wasm32 # buffer cannot hold. HOSTILE_GCOL_SIZE = 2**31 + 4096 # The file length a server claims for hostile_vl.h5 (the tests' mock fetch # answers every range with zeros past the real bytes): 3 GiB, within what # wasm32 opens, and room for the collection. HOSTILE_LENGTH = 3 << 30 def write_limits(out): """Files for the size limits (the tests must get errors, not aborts): - limits.h5: /huge_u8, 2^28 + 1024 bytes of u8 in compressed chunks (a small file): read whole it would take over 2 GiB while decoding; its last value is 7. - hostile_vl.h5: a variable-length string dataset /a whose global heap collection claims HOSTILE_GCOL_SIZE bytes, with the superblock's end of file set to HOSTILE_LENGTH (libhdf5 cannot read it; it is only served by a mock that claims that length). - far.h5 and far.json: /x, 16 float64 values, whose contiguous data address is moved FAR_SHIFT bytes on (past 2 GiB, the sign bit of a wasm32 isize) in a file whose end of file is moved as far; the tests' mock serves the data there, to show offsets up to 4 GiB work on wasm32. """ with h5py.File(out / "limits.h5", "w") as f: d = f.create_dataset("huge_u8", shape=(HUGE_U8,), dtype="u1", chunks=(1 << 20,), compression="gzip") d[-1] = 7 path = out / "hostile_vl.h5" with h5py.File(path, "w", libver="earliest") as f: f.create_dataset("a", data=["x", "yy"], dtype=h5py.string_dtype()) b = bytearray(path.read_bytes()) assert b[8] == 0, "a version 0 superblock" b[40:48] = HOSTILE_LENGTH.to_bytes(8, "little") # end of file address at = b.index(b"GCOL") b[at + 8:at + 16] = HOSTILE_GCOL_SIZE.to_bytes(8, "little") path.write_bytes(bytes(b)) path = out / "far.h5" values = np.arange(16, dtype=" 0: json.dump(write_big(out / "big.h5", big_mb), open(out / "big.json", "w"), indent=1)