Files
clawhdf5/examples/wasm-viewer/test/make_fixture.py
T
osobhandClaude Opus 5.5 dbafa952ac wasm: sizes a server or a dataset names are errors, not aborts
A read longer than isize::MAX (2 GiB on wasm32) aborted the module in
LazyStorage::assemble (capacity_overflow), taking every open file on the
page with it, and a hostile server only had to claim a large length and
serve a heap collection of 2 GiB + 4 KiB to get there (after fetching
2 GiB). Reading a large u8 dataset whole aborted the same way when its
values were widened to 64 bits.

- LazyConfig::max_fetch (openUrl option maxFetch, default 512 MiB, at
  most 1 GiB): a read longer than it fails at once, before anything is
  fetched, and an operation whose passes would fetch more than it fails
  before fetching (Operation::charge). assemble reserves fallibly.
- Reader::read refuses a read that would use more than 1 GiB while
  decoding (core::MAX_READ_BYTES: stored bytes + 64-bit values + result)
  with an error naming readHyperslab, before reading.
- openUrl refuses a file of 4 GiB or more at open on wasm32: the format
  code turns offsets into usize, so nothing past 4 GiB can be read there
  (shown by a new test: data at 3 GiB reads, a 4 GiB file is refused).
  maxDownload is bounded to 1 GiB.

Tests: make_fixture.py writes limits.h5 (a sparse 2^28 + 1024 byte u8
dataset), hostile_vl.h5 (the reviewer's collection) and far.h5 (data at
3 GiB); test.mjs (wasm32) and tests/lazy.rs (native) check each is an
error or reads, and that the module survives. Before: RuntimeError:
unreachable in Node; the native test read the huge dataset and fetched
2 GiB.

Co-Authored-By: Claude Opus 5.5 (1M context) <[email protected]>
2026-09-27 07:36:21 -05:00

309 lines
12 KiB
Python

"""Write HDF5 and NetCDF-4 test files with h5py/netCDF4, and what libhdf5
reads back from them, for the clawhdf5-wasm tests.
python make_fixture.py OUT_DIR
writes OUT_DIR/fixture.h5, OUT_DIR/fixture.nc and OUT_DIR/expected.json,
and the limit-test files OUT_DIR/limits.h5 and OUT_DIR/hostile_vl.h5 (see
write_limits).
Both the Rust test (crates/clawhdf5-wasm/tests/h5py_interop.rs, native) and
the Node test (test.mjs, the built wasm package) compare against the same
expected.json, so the two check the same values.
Every expected value comes from h5py reading the file back (numpy slicing
for hyperslabs), never from the arrays that were written. Integers are
encoded as strings so JSON.parse keeps 64-bit values exact.
"""
import json
import os
import sys
import warnings
from pathlib import Path
import h5py
import netCDF4
import numpy as np
try: # registers the LZ4/Zstd filters with libhdf5; optional
import hdf5plugin
except ImportError:
hdf5plugin = None
# netCDF4 1.7 trips numpy 2.5's shape-setting deprecation on assignment.
warnings.filterwarnings("ignore", category=DeprecationWarning)
out = Path(sys.argv[1])
out.mkdir(parents=True, exist_ok=True)
h5 = out / "fixture.h5"
nc = out / "fixture.nc"
rng = np.random.default_rng(7)
with h5py.File(h5, "w") as f:
f.attrs["title"] = "wasm fixture"
f.attrs["version"] = np.int64(3)
f.attrs["scale"] = np.array([0.5, 2.0])
f.attrs["big"] = np.uint64(2**63 + 5)
f.attrs.create("vlen_note", "héllo", dtype=h5py.string_dtype())
# No plain JavaScript form: listed with value null and its type.
f.attrs["origin"] = np.array((1.5, 2), dtype=[("x", "<f8"), ("n", "<i4")])
f.create_dataset(
"grid", data=np.arange(60, dtype="<f8").reshape(6, 10) / 4,
chunks=(4, 3), compression="gzip", shuffle=True,
)
f.create_dataset("f32_be", data=rng.standard_normal(7).astype(">f4"))
f.create_dataset("f16", data=np.array([0.5, -1.25, 65504], dtype="<f2"))
f.create_dataset("i8", data=np.array([-128, -1, 0, 127], dtype="i1"))
f.create_dataset("i16_be", data=np.array([-32768, 5, 32767], dtype=">i2"))
f.create_dataset("u16", data=np.array([0, 40000, 65535], dtype="<u2"))
f.create_dataset("u32", data=np.array([0, 4_000_000_000], dtype="<u4"))
f.create_dataset("i64", data=np.array([-(2**63), 2**53 + 1, 7], dtype="<i8"))
f.create_dataset("u64", data=np.array([2**64 - 1, 1], dtype="<u8"))
f.create_dataset("scalar", data=np.float64(2.5))
f.create_dataset("fixed_str", data=np.array([b"alpha", b"be", b""], dtype="S5"))
f.create_dataset(
"vlen_str", data=["one", "двa", ""], dtype=h5py.string_dtype()
)
f.create_dataset("flags", data=np.array([True, False, True]))
# An array datatype: four elements, each an i4[2].
pairs = f.create_dataset("pairs", shape=(4,), dtype=np.dtype(("<i4", (2,))))
pairs[...] = np.arange(8, dtype="<i4").reshape(4, 2)
f.create_dataset(
"cube", data=np.arange(2 * 5 * 6, dtype="<i4").reshape(2, 5, 6),
chunks=(1, 2, 3), compression="gzip",
)
if hdf5plugin is not None:
# LZ4 is built into clawhdf5-wasm; Zstd links C and is not.
f.create_dataset("lz4", data=np.arange(40, dtype="<i4"), chunks=(10,),
**hdf5plugin.LZ4())
f.create_dataset("zstd", data=np.arange(40, dtype="<i4"), chunks=(10,),
**hdf5plugin.Zstd())
comp = np.zeros(2, dtype=[("x", "<f8"), ("n", "<i4")])
f.create_dataset("table", data=comp)
g = f.create_group("sensors")
g.attrs["location"] = "lab"
g.create_dataset("temp", data=np.array([21.5, 22.0, 22.25], dtype="<f4"))
g.create_group("empty")
f["alias"] = h5py.SoftLink("/sensors/temp")
with netCDF4.Dataset(nc, "w") as d:
d.title = "nc fixture"
d.createDimension("time", None)
d.createDimension("x", 4)
t = d.createVariable("time", "f8", ("time",))
t.units = "days since 2000-01-01"
v = d.createVariable("temp", "f4", ("time", "x"), zlib=True)
t[:] = np.arange(3)
v[:] = np.arange(12, dtype="f4").reshape(3, 4) + 0.5
def kind(dt):
"""The typed-array kind clawhdf5-wasm returns for a numpy dtype."""
if dt.kind == "b" or h5py.check_enum_dtype(dt) is not None:
return "strings"
if h5py.check_string_dtype(dt) is not None or dt.kind == "S":
return "strings"
if dt.subdtype is not None:
return kind(dt.subdtype[0])
if dt.kind == "f":
return "f64" if dt.itemsize == 8 else "f32"
if dt.kind in "iu":
return f"{dt.kind}{dt.itemsize * 8}"
raise ValueError(dt)
def flat(a, k):
a = np.asarray(a)
if k == "strings":
if a.dtype.kind == "b":
return ["TRUE" if x else "FALSE" for x in a.ravel()]
return [x.decode() if isinstance(x, bytes) else str(x) for x in a.ravel()]
if k.startswith(("i", "u")):
return [str(int(x)) for x in a.ravel()]
return [float(x) for x in a.ravel()]
def entry(ds, slab=None):
k = kind(ds.dtype)
data = ds[()]
e = {"kind": k, "shape": list(np.shape(data)), "values": flat(data, k)}
if slab:
start, count, stride = slab
idx = tuple(slice(s, s + (c - 1) * st + 1, st)
for s, c, st in zip(start, count, stride))
part = ds[idx]
e["slab"] = {"start": start, "count": count, "stride": stride,
"shape": list(part.shape), "values": flat(part, k)}
return e
def attr(v):
v = np.asarray(v) if not isinstance(v, (str, bytes)) else v
if isinstance(v, np.ndarray) and v.dtype.names:
return {"raw": "compound"}
if isinstance(v, bytes):
return {"string": v.decode()}
if isinstance(v, str):
return {"string": v}
if v.dtype.kind in "iu":
return {"int": [str(int(x)) for x in v.ravel()], "scalar": v.ndim == 0}
if v.dtype.kind == "f":
return {"float": [float(x) for x in v.ravel()], "scalar": v.ndim == 0}
if v.dtype.kind in "OSU":
items = [x.decode() if isinstance(x, bytes) else str(x) for x in v.ravel()]
return {"string": items[0]} if v.ndim == 0 else {"strings": items}
raise ValueError(v.dtype)
# Attributes the reader returns but the comparison leaves out: netCDF-4's
# internal ones (a leading underscore), and dimension-scale bookkeeping.
SKIP_ATTRS = ["DIMENSION_LIST", "REFERENCE_LIST", "CLASS", "NAME"]
def describe(path):
expected = {"datasets": {}, "errors": {}, "attrs": {}, "lists": {},
"skip_attrs": SKIP_ATTRS}
with h5py.File(path, "r") as f:
def walk(key, obj):
expected["attrs"][key] = {
k: attr(obj.attrs[k]) for k in obj.attrs
if not k.startswith("_") and k not in SKIP_ATTRS}
if isinstance(obj, h5py.Group):
members = {n: obj.get(n) for n in obj}
groups = [n for n, o in members.items() if isinstance(o, h5py.Group)]
sets = [n for n, o in members.items() if isinstance(o, h5py.Dataset)]
expected["lists"][key] = {"groups": sorted(groups),
"datasets": sorted(sets)}
for n, o in members.items():
walk(key.rstrip("/") + "/" + n, o)
elif obj.dtype.names:
expected["errors"][key] = "compound"
else:
expected["datasets"][key] = entry(obj, slab_for(obj))
if key == "/zstd":
# Refused by the wasm build (no zstd); a native build
# that unifies in clawhdf5-format/zstd reads it.
expected["datasets"][key]["unavailable"] = "unsupported filter: 32015"
walk("/", f)
return expected
def slab_for(obj):
"""A strided hyperslab inside the dataset's extent, or None."""
if obj.ndim == 0 or obj.shape[0] < 2 or 0 in obj.shape:
return None
start = [1] + [0] * (obj.ndim - 1)
count = [max(1, (obj.shape[0] - 1) // 2)] + [
max(1, (n + 1) // 2) for n in obj.shape[1:]]
stride = [2 if obj.shape[0] > 2 else 1] + [2] * (obj.ndim - 1)
count = [min(c, (n - s - 1) // st + 1)
for c, s, st, n in zip(count, start, stride, obj.shape)]
return (start, count, stride)
json.dump({"fixture.h5": describe(h5), "fixture.nc": describe(nc)},
open(out / "expected.json", "w"), indent=1, ensure_ascii=False)
HUGE_U8 = 2**28 + 1024
# The collection size hostile_vl.h5 claims: past 2 GiB, which a wasm32
# buffer cannot hold.
HOSTILE_GCOL_SIZE = 2**31 + 4096
# The file length a server claims for hostile_vl.h5 (the tests' mock fetch
# answers every range with zeros past the real bytes): 3 GiB, within what
# wasm32 opens, and room for the collection.
HOSTILE_LENGTH = 3 << 30
def write_limits(out):
"""Files for the size limits (the tests must get errors, not aborts):
- limits.h5: /huge_u8, 2^28 + 1024 bytes of u8 in compressed chunks
(a small file): read whole it would take over 2 GiB while decoding;
its last value is 7.
- hostile_vl.h5: a variable-length string dataset /a whose global heap
collection claims HOSTILE_GCOL_SIZE bytes, with the superblock's end
of file set to HOSTILE_LENGTH (libhdf5 cannot read it; it is only
served by a mock that claims that length).
- far.h5 and far.json: /x, 16 float64 values, whose contiguous data
address is moved FAR_SHIFT bytes on (past 2 GiB, the sign bit of a
wasm32 isize) in a file whose end of file is moved as far; the tests'
mock serves the data there, to show offsets up to 4 GiB work on
wasm32.
"""
with h5py.File(out / "limits.h5", "w") as f:
d = f.create_dataset("huge_u8", shape=(HUGE_U8,), dtype="u1",
chunks=(1 << 20,), compression="gzip")
d[-1] = 7
path = out / "hostile_vl.h5"
with h5py.File(path, "w", libver="earliest") as f:
f.create_dataset("a", data=["x", "yy"], dtype=h5py.string_dtype())
b = bytearray(path.read_bytes())
assert b[8] == 0, "a version 0 superblock"
b[40:48] = HOSTILE_LENGTH.to_bytes(8, "little") # end of file address
at = b.index(b"GCOL")
b[at + 8:at + 16] = HOSTILE_GCOL_SIZE.to_bytes(8, "little")
path.write_bytes(bytes(b))
path = out / "far.h5"
values = np.arange(16, dtype="<f8") * 1.5
with h5py.File(path, "w", libver="earliest") as f:
f.create_dataset("x", data=values)
data_at = f["x"].id.get_offset()
b = bytearray(path.read_bytes())
# The layout message: the data's address, then its size.
old = data_at.to_bytes(8, "little") + (values.nbytes).to_bytes(8, "little")
assert b.count(old) == 1
at = b.index(old)
b[at:at + 8] = (data_at + FAR_SHIFT).to_bytes(8, "little")
length = len(b) + FAR_SHIFT
b[40:48] = length.to_bytes(8, "little")
path.write_bytes(bytes(b))
json.dump({"data_at": data_at, "far_at": data_at + FAR_SHIFT,
"nbytes": values.nbytes, "length": length,
"values": [float(x) for x in values]},
open(out / "far.json", "w"))
# far.h5's data moves this far: past 2 GiB, below 4 GiB.
FAR_SHIFT = 3 << 30
write_limits(out)
def write_big(path, megabytes):
"""A large file for the range-request tests (`openUrl`): `/big`, about
`megabytes` MB of float64 in 1 MiB chunks, written after a small
dataset and a group, so listing and reading `/small` touch a few blocks
of the file and a window of `/big` one chunk. Returns what h5py reads
back."""
n = megabytes * 1_000_000 // 8
chunk = 1 << 17
with h5py.File(path, "w") as f:
f.attrs["note"] = "large file for range reads"
f.create_dataset("small", data=np.array([1.5, -2.0, 3.25]))
g = f.create_group("meta")
g.attrs["units"] = "m"
g.create_dataset("ids", data=np.arange(10, dtype="<i4"))
big = f.create_dataset("big", shape=(n,), dtype="<f8", chunks=(chunk,))
for s in range(0, n, 1 << 22):
e = min(n, s + (1 << 22))
big[s:e] = np.arange(s, e, dtype="<f8") * 0.5
start = n // 2 + 12_345
with h5py.File(path, "r") as f:
return {
"size": path.stat().st_size,
"list": {"groups": ["meta"], "datasets": ["big", "small"]},
"small": [float(x) for x in f["small"][()]],
"ids": [str(int(x)) for x in f["meta/ids"][()]],
"big_shape": list(f["big"].shape),
"window": {"start": start, "count": 10,
"values": [float(x) for x in f["big"][start:start + 10]]},
}
big_mb = int(os.environ.get("WASM_BIG_MB", "0"))
if big_mb > 0:
json.dump(write_big(out / "big.h5", big_mb), open(out / "big.json", "w"), indent=1)