Files
clawhdf5/examples/wasm-viewer/test/make_fixture.py
T
osobhandClaude Opus 5.5 e4ba09946f
CI / test-arm64 (pull_request) Successful in 1m38s
CI / test (pull_request) Successful in 21m11s
HDF5 2.0 native complex as a first-class type on read; Python libver=
- Datatype::parse returns Datatype::Complex for class 11 (also inside
  compounds, arrays and VL types) instead of the {r, i} compound view.
- Facade: DType::Complex(Box<DType>); read_complex_f32/f64 accept it.
- h5rs dump/ls/diff print native complex as h5dump/h5ls/h5diff 2.2.0 do
  (checked against a fixture written by h5py 3.16 / libhdf5 2.0.0);
  dump --json keeps the {r, i} compound (hdf5-json has no complex class).
- clawhdf5-wasm reads native complex datasets as [re, im] pairs.
- Python: clawhdf5.File(path, 'w', libver=...) with h5py's values,
  mapped to FileBuilder::libver_bounds; 'v108' output opens in HDF5 1.8.23.
- Docs: known-issues entry moved to Fixed (history), CHANGELOG, READMEs.

Co-Authored-By: Claude Opus 5.5 (1M context) <[email protected]>
2026-09-29 20:47:33 -05:00

328 lines
14 KiB
Python

"""Write HDF5 and NetCDF-4 test files with h5py/netCDF4, and what libhdf5
reads back from them, for the clawhdf5-wasm tests.
python make_fixture.py OUT_DIR
writes OUT_DIR/fixture.h5, OUT_DIR/fixture.nc and OUT_DIR/expected.json,
and the limit-test files OUT_DIR/limits.h5 and OUT_DIR/hostile_vl.h5 (see
write_limits).
Both the Rust test (crates/clawhdf5-wasm/tests/h5py_interop.rs, native) and
the Node test (test.mjs, the built wasm package) compare against the same
expected.json, so the two check the same values.
Every expected value comes from h5py reading the file back (numpy slicing
for hyperslabs), never from the arrays that were written. Integers are
encoded as strings so JSON.parse keeps 64-bit values exact.
"""
import json
import os
import sys
import warnings
from pathlib import Path
import h5py
import netCDF4
import numpy as np
try: # registers the LZ4/Zstd filters with libhdf5; optional
import hdf5plugin
except ImportError:
hdf5plugin = None
# netCDF4 1.7 trips numpy 2.5's shape-setting deprecation on assignment.
warnings.filterwarnings("ignore", category=DeprecationWarning)
out = Path(sys.argv[1])
out.mkdir(parents=True, exist_ok=True)
h5 = out / "fixture.h5"
nc = out / "fixture.nc"
rng = np.random.default_rng(7)
with h5py.File(h5, "w") as f:
f.attrs["title"] = "wasm fixture"
f.attrs["version"] = np.int64(3)
f.attrs["scale"] = np.array([0.5, 2.0])
f.attrs["big"] = np.uint64(2**63 + 5)
f.attrs.create("vlen_note", "héllo", dtype=h5py.string_dtype())
# No plain JavaScript form: listed with value null and its type.
f.attrs["origin"] = np.array((1.5, 2), dtype=[("x", "<f8"), ("n", "<i4")])
f.create_dataset(
"grid", data=np.arange(60, dtype="<f8").reshape(6, 10) / 4,
chunks=(4, 3), compression="gzip", shuffle=True,
)
f.create_dataset("f32_be", data=rng.standard_normal(7).astype(">f4"))
f.create_dataset("f16", data=np.array([0.5, -1.25, 65504], dtype="<f2"))
f.create_dataset("i8", data=np.array([-128, -1, 0, 127], dtype="i1"))
f.create_dataset("i16_be", data=np.array([-32768, 5, 32767], dtype=">i2"))
f.create_dataset("u16", data=np.array([0, 40000, 65535], dtype="<u2"))
f.create_dataset("u32", data=np.array([0, 4_000_000_000], dtype="<u4"))
f.create_dataset("i64", data=np.array([-(2**63), 2**53 + 1, 7], dtype="<i8"))
f.create_dataset("u64", data=np.array([2**64 - 1, 1], dtype="<u8"))
f.create_dataset("scalar", data=np.float64(2.5))
f.create_dataset("fixed_str", data=np.array([b"alpha", b"be", b""], dtype="S5"))
f.create_dataset(
"vlen_str", data=["one", "двa", ""], dtype=h5py.string_dtype()
)
f.create_dataset("flags", data=np.array([True, False, True]))
# An array datatype: four elements, each an i4[2].
pairs = f.create_dataset("pairs", shape=(4,), dtype=np.dtype(("<i4", (2,))))
pairs[...] = np.arange(8, dtype="<i4").reshape(4, 2)
f.create_dataset(
"cube", data=np.arange(2 * 5 * 6, dtype="<i4").reshape(2, 5, 6),
chunks=(1, 2, 3), compression="gzip",
)
if hdf5plugin is not None:
# LZ4 is built into clawhdf5-wasm; Zstd links C and is not.
f.create_dataset("lz4", data=np.arange(40, dtype="<i4"), chunks=(10,),
**hdf5plugin.LZ4())
f.create_dataset("zstd", data=np.arange(40, dtype="<i4"), chunks=(10,),
**hdf5plugin.Zstd())
comp = np.zeros(2, dtype=[("x", "<f8"), ("n", "<i4")])
f.create_dataset("table", data=comp)
if getattr(h5py.get_config(), "has_native_complex", False):
# HDF5 2.0 native complex (class 11): read as [re, im] pairs.
from h5py import h5d, h5s, h5t
for name, t, dt in [(b"native_c128", h5t.COMPLEX_IEEE_F64LE, "<c16"),
(b"native_c64_be", h5t.COMPLEX_IEEE_F32BE, ">c8")]:
z = (np.arange(6) - 1.5j * np.arange(6)).reshape(3, 2).astype(dt)
d = h5d.create(f.id, name, t, h5s.create_simple(z.shape))
d.write(h5s.ALL, h5s.ALL, z, mtype=t)
g = f.create_group("sensors")
g.attrs["location"] = "lab"
g.create_dataset("temp", data=np.array([21.5, 22.0, 22.25], dtype="<f4"))
g.create_group("empty")
f["alias"] = h5py.SoftLink("/sensors/temp")
with netCDF4.Dataset(nc, "w") as d:
d.title = "nc fixture"
d.createDimension("time", None)
d.createDimension("x", 4)
t = d.createVariable("time", "f8", ("time",))
t.units = "days since 2000-01-01"
v = d.createVariable("temp", "f4", ("time", "x"), zlib=True)
t[:] = np.arange(3)
v[:] = np.arange(12, dtype="f4").reshape(3, 4) + 0.5
def kind(dt):
"""The typed-array kind clawhdf5-wasm returns for a numpy dtype."""
if dt.kind == "b" or h5py.check_enum_dtype(dt) is not None:
return "strings"
if h5py.check_string_dtype(dt) is not None or dt.kind == "S":
return "strings"
if dt.subdtype is not None:
return kind(dt.subdtype[0])
if dt.kind == "c":
return "f64" if dt.itemsize == 16 else "f32"
if dt.kind == "f":
return "f64" if dt.itemsize == 8 else "f32"
if dt.kind in "iu":
return f"{dt.kind}{dt.itemsize * 8}"
raise ValueError(dt)
def flat(a, k):
a = np.asarray(a)
if k == "strings":
if a.dtype.kind == "b":
return ["TRUE" if x else "FALSE" for x in a.ravel()]
return [x.decode() if isinstance(x, bytes) else str(x) for x in a.ravel()]
if k.startswith(("i", "u")):
return [str(int(x)) for x in a.ravel()]
if a.dtype.kind == "c":
# [re, im] pairs, in order.
a = np.stack([a.real, a.imag], axis=-1)
return [float(x) for x in a.ravel()]
def entry(ds, slab=None):
k = kind(ds.dtype)
data = ds[()]
# A complex element reads as its two parts: one more dimension.
pair = [2] if ds.dtype.kind == "c" else []
e = {"kind": k, "shape": list(np.shape(data)) + pair, "values": flat(data, k)}
if slab:
start, count, stride = slab
idx = tuple(slice(s, s + (c - 1) * st + 1, st)
for s, c, st in zip(start, count, stride))
part = ds[idx]
e["slab"] = {"start": start, "count": count, "stride": stride,
"shape": list(part.shape) + pair, "values": flat(part, k)}
return e
def attr(v):
v = np.asarray(v) if not isinstance(v, (str, bytes)) else v
if isinstance(v, np.ndarray) and v.dtype.names:
return {"raw": "compound"}
if isinstance(v, bytes):
return {"string": v.decode()}
if isinstance(v, str):
return {"string": v}
if v.dtype.kind in "iu":
return {"int": [str(int(x)) for x in v.ravel()], "scalar": v.ndim == 0}
if v.dtype.kind == "f":
return {"float": [float(x) for x in v.ravel()], "scalar": v.ndim == 0}
if v.dtype.kind in "OSU":
items = [x.decode() if isinstance(x, bytes) else str(x) for x in v.ravel()]
return {"string": items[0]} if v.ndim == 0 else {"strings": items}
raise ValueError(v.dtype)
# Attributes the reader returns but the comparison leaves out: netCDF-4's
# internal ones (a leading underscore), and dimension-scale bookkeeping.
SKIP_ATTRS = ["DIMENSION_LIST", "REFERENCE_LIST", "CLASS", "NAME"]
def describe(path):
expected = {"datasets": {}, "errors": {}, "attrs": {}, "lists": {},
"skip_attrs": SKIP_ATTRS}
with h5py.File(path, "r") as f:
def walk(key, obj):
expected["attrs"][key] = {
k: attr(obj.attrs[k]) for k in obj.attrs
if not k.startswith("_") and k not in SKIP_ATTRS}
if isinstance(obj, h5py.Group):
members = {n: obj.get(n) for n in obj}
groups = [n for n, o in members.items() if isinstance(o, h5py.Group)]
sets = [n for n, o in members.items() if isinstance(o, h5py.Dataset)]
expected["lists"][key] = {"groups": sorted(groups),
"datasets": sorted(sets)}
for n, o in members.items():
walk(key.rstrip("/") + "/" + n, o)
elif obj.dtype.names or (
obj.dtype.kind == "c"
and obj.id.get_type().get_class() != getattr(h5py.h5t, "COMPLEX", None)):
# h5py's own complex numbers are a compound {r, i}: refused.
expected["errors"][key] = "compound"
else:
expected["datasets"][key] = entry(obj, slab_for(obj))
if key == "/zstd":
# Refused by the wasm build (no zstd); a native build
# that unifies in clawhdf5-format/zstd reads it.
expected["datasets"][key]["unavailable"] = "unsupported filter: 32015"
walk("/", f)
return expected
def slab_for(obj):
"""A strided hyperslab inside the dataset's extent, or None."""
if obj.ndim == 0 or obj.shape[0] < 2 or 0 in obj.shape:
return None
start = [1] + [0] * (obj.ndim - 1)
count = [max(1, (obj.shape[0] - 1) // 2)] + [
max(1, (n + 1) // 2) for n in obj.shape[1:]]
stride = [2 if obj.shape[0] > 2 else 1] + [2] * (obj.ndim - 1)
count = [min(c, (n - s - 1) // st + 1)
for c, s, st, n in zip(count, start, stride, obj.shape)]
return (start, count, stride)
json.dump({"fixture.h5": describe(h5), "fixture.nc": describe(nc)},
open(out / "expected.json", "w"), indent=1, ensure_ascii=False)
HUGE_U8 = 2**28 + 1024
# The collection size hostile_vl.h5 claims: past 2 GiB, which a wasm32
# buffer cannot hold.
HOSTILE_GCOL_SIZE = 2**31 + 4096
# The file length a server claims for hostile_vl.h5 (the tests' mock fetch
# answers every range with zeros past the real bytes): 3 GiB, within what
# wasm32 opens, and room for the collection.
HOSTILE_LENGTH = 3 << 30
def write_limits(out):
"""Files for the size limits (the tests must get errors, not aborts):
- limits.h5: /huge_u8, 2^28 + 1024 bytes of u8 in compressed chunks
(a small file): read whole it would take over 2 GiB while decoding;
its last value is 7.
- hostile_vl.h5: a variable-length string dataset /a whose global heap
collection claims HOSTILE_GCOL_SIZE bytes, with the superblock's end
of file set to HOSTILE_LENGTH (libhdf5 cannot read it; it is only
served by a mock that claims that length).
- far.h5 and far.json: /x, 16 float64 values, whose contiguous data
address is moved FAR_SHIFT bytes on (past 2 GiB, the sign bit of a
wasm32 isize) in a file whose end of file is moved as far; the tests'
mock serves the data there, to show offsets up to 4 GiB work on
wasm32.
"""
with h5py.File(out / "limits.h5", "w") as f:
d = f.create_dataset("huge_u8", shape=(HUGE_U8,), dtype="u1",
chunks=(1 << 20,), compression="gzip")
d[-1] = 7
path = out / "hostile_vl.h5"
with h5py.File(path, "w", libver="earliest") as f:
f.create_dataset("a", data=["x", "yy"], dtype=h5py.string_dtype())
b = bytearray(path.read_bytes())
assert b[8] == 0, "a version 0 superblock"
b[40:48] = HOSTILE_LENGTH.to_bytes(8, "little") # end of file address
at = b.index(b"GCOL")
b[at + 8:at + 16] = HOSTILE_GCOL_SIZE.to_bytes(8, "little")
path.write_bytes(bytes(b))
path = out / "far.h5"
values = np.arange(16, dtype="<f8") * 1.5
with h5py.File(path, "w", libver="earliest") as f:
f.create_dataset("x", data=values)
data_at = f["x"].id.get_offset()
b = bytearray(path.read_bytes())
# The layout message: the data's address, then its size.
old = data_at.to_bytes(8, "little") + (values.nbytes).to_bytes(8, "little")
assert b.count(old) == 1
at = b.index(old)
b[at:at + 8] = (data_at + FAR_SHIFT).to_bytes(8, "little")
length = len(b) + FAR_SHIFT
b[40:48] = length.to_bytes(8, "little")
path.write_bytes(bytes(b))
json.dump({"data_at": data_at, "far_at": data_at + FAR_SHIFT,
"nbytes": values.nbytes, "length": length,
"values": [float(x) for x in values]},
open(out / "far.json", "w"))
# far.h5's data moves this far: past 2 GiB, below 4 GiB.
FAR_SHIFT = 3 << 30
write_limits(out)
def write_big(path, megabytes):
"""A large file for the range-request tests (`openUrl`): `/big`, about
`megabytes` MB of float64 in 1 MiB chunks, written after a small
dataset and a group, so listing and reading `/small` touch a few blocks
of the file and a window of `/big` one chunk. Returns what h5py reads
back."""
n = megabytes * 1_000_000 // 8
chunk = 1 << 17
with h5py.File(path, "w") as f:
f.attrs["note"] = "large file for range reads"
f.create_dataset("small", data=np.array([1.5, -2.0, 3.25]))
g = f.create_group("meta")
g.attrs["units"] = "m"
g.create_dataset("ids", data=np.arange(10, dtype="<i4"))
big = f.create_dataset("big", shape=(n,), dtype="<f8", chunks=(chunk,))
for s in range(0, n, 1 << 22):
e = min(n, s + (1 << 22))
big[s:e] = np.arange(s, e, dtype="<f8") * 0.5
start = n // 2 + 12_345
with h5py.File(path, "r") as f:
return {
"size": path.stat().st_size,
"list": {"groups": ["meta"], "datasets": ["big", "small"]},
"small": [float(x) for x in f["small"][()]],
"ids": [str(int(x)) for x in f["meta/ids"][()]],
"big_shape": list(f["big"].shape),
"window": {"start": start, "count": 10,
"values": [float(x) for x in f["big"][start:start + 10]]},
}
big_mb = int(os.environ.get("WASM_BIG_MB", "0"))
if big_mb > 0:
json.dump(write_big(out / "big.h5", big_mb), open(out / "big.json", "w"), indent=1)