conformance/run.sh fetches eight public HDF5 corpora pinned by commit (conformance/corpus.txt) into a gitignored cache, reads every file with clawhdf5 (conformance/probe, a crate outside the workspace) and with h5py/libhdf5 (ref.py), and the CVE files with h5dump, each under a timeout and an address-space limit; compare.py classifies the files, report.py writes CONFORMANCE.md and check.py gates on panics/hangs/crashes/OOM and on regressions against conformance/baseline.json. ~25 s once cached. Changes from the ad-hoc audit harness: - the probe compares non-IEEE-layout floats (N-Bit) and integers with a bit offset or reduced precision as the values libhdf5 converts them to, not raw file bytes: 8 files that showed as mismatches now read identically; - ref.py exits without tearing down h5py objects: libhdf5 2.0 aborts while freeing them for two files about half the time, which flipped them between ok and h5py-cannot-read from run to run; - the file list is defined (list_files.py): netCDF classic files are left out, 11 HDF5 files the ad-hoc sweep missed are in. Co-Authored-By: Claude Opus 5.5 (1M context) <[email protected]>
260 lines
8.1 KiB
Python
Executable File
260 lines
8.1 KiB
Python
Executable File
#!/usr/bin/env python3
|
|
"""Reference probe: same JSON as the Rust `conformance-probe`, produced with h5py.
|
|
|
|
Walk: iterative DFS from '/', children in sorted (UTF-8 byte) name order, hard
|
|
links only, each object once (first path wins, deduplicated by object identity).
|
|
Canonical value encoding: see harness/src/main.rs.
|
|
"""
|
|
import hashlib
|
|
import json
|
|
import os
|
|
import struct
|
|
import sys
|
|
|
|
import numpy as np
|
|
import h5py
|
|
|
|
try:
|
|
import hdf5plugin # noqa: F401 registers blosc/lz4/zstd/bzip2/... filters
|
|
except Exception: # pragma: no cover
|
|
pass
|
|
|
|
MAX_BYTES = 200 * 1024 * 1024
|
|
MAX_OBJECTS = 200_000
|
|
|
|
|
|
def canon_str(b, out):
|
|
if isinstance(b, str):
|
|
b = b.encode("utf-8", "surrogateescape")
|
|
b = bytes(b)
|
|
cut = b.find(b"\x00")
|
|
if cut >= 0:
|
|
b = b[:cut]
|
|
b = b.rstrip(b" ")
|
|
out += b"S" + struct.pack("<I", len(b)) + b
|
|
|
|
|
|
def simple(dt):
|
|
if dt.fields:
|
|
return all(simple(dt.fields[n][0]) for n in dt.names)
|
|
if dt.subdtype:
|
|
return simple(dt.subdtype[0])
|
|
return dt.kind in "iufcbV"
|
|
|
|
|
|
def packed(dt):
|
|
if dt.fields:
|
|
return np.dtype([(n, packed(dt.fields[n][0])) for n in dt.names])
|
|
if dt.subdtype:
|
|
base, shape = dt.subdtype
|
|
return np.dtype((packed(base), shape))
|
|
if dt.kind in "iufcb":
|
|
return dt.newbyteorder("<")
|
|
return dt
|
|
|
|
|
|
def canon_el(dt, val, out):
|
|
if dt.fields:
|
|
for n in dt.names:
|
|
canon_el(dt.fields[n][0], val[n], out)
|
|
return
|
|
if dt.subdtype:
|
|
base, _ = dt.subdtype
|
|
for x in np.asarray(val).reshape(-1):
|
|
canon_el(base, x, out)
|
|
return
|
|
k = dt.kind
|
|
if k in "iufcb":
|
|
out += np.asarray(val, dtype=dt).astype(dt.newbyteorder("<")).tobytes()
|
|
elif k == "V":
|
|
out += np.asarray(val, dtype=dt).tobytes()
|
|
elif k == "S":
|
|
canon_str(val, out)
|
|
elif k == "O":
|
|
if h5py.check_string_dtype(dt) is not None:
|
|
canon_str(val if val is not None else b"", out)
|
|
elif h5py.check_ref_dtype(dt) is not None:
|
|
out += b"R"
|
|
else:
|
|
base = h5py.check_vlen_dtype(dt)
|
|
if base is None:
|
|
raise TypeError(f"unhandled object dtype {dt!r}")
|
|
arr = np.asarray(val if val is not None else [], dtype=base).reshape(-1)
|
|
out += b"V" + struct.pack("<I", arr.shape[0])
|
|
if simple(base):
|
|
out += arr.astype(packed(base)).tobytes()
|
|
else:
|
|
for x in arr:
|
|
canon_el(base, x, out)
|
|
elif k == "U":
|
|
canon_str(str(val), out)
|
|
else:
|
|
raise TypeError(f"unhandled dtype kind {k} ({dt!r})")
|
|
|
|
|
|
def has_obj(dt):
|
|
if dt.fields:
|
|
return any(has_obj(dt.fields[n][0]) for n in dt.names)
|
|
if dt.subdtype:
|
|
return has_obj(dt.subdtype[0])
|
|
return dt.kind == "O"
|
|
|
|
|
|
def note_conversion(tid, dt, rec):
|
|
"""h5py converts some file types (FP8, bfloat16, x87 long double, ...) to a
|
|
different-sized numpy type; then value bytes are not comparable."""
|
|
try:
|
|
if not has_obj(dt) and tid.get_size() != dt.itemsize:
|
|
rec["converted"] = f"file type size {tid.get_size()} -> numpy {dt} ({dt.itemsize})"
|
|
except Exception: # noqa: BLE001
|
|
pass
|
|
|
|
|
|
def hash_values(arr, dt, rec):
|
|
if dt.subdtype is not None:
|
|
# h5py expands an HDF5 array element type into trailing array dims
|
|
dt = dt.subdtype[0]
|
|
arr = np.asarray(arr, dtype=dt)
|
|
if simple(dt):
|
|
c = np.ascontiguousarray(arr).astype(packed(dt)).tobytes()
|
|
else:
|
|
out = bytearray()
|
|
for x in arr.reshape(-1):
|
|
canon_el(dt, x, out)
|
|
c = bytes(out)
|
|
rec["hash"] = hashlib.sha256(c).hexdigest()
|
|
rec["head"] = c[:48].hex()
|
|
|
|
|
|
def err(e):
|
|
s = f"{type(e).__name__}: {e}"
|
|
return s.splitlines()[0][:400] if s else type(e).__name__
|
|
|
|
|
|
def shape_of(s):
|
|
return "null" if s is None else list(s)
|
|
|
|
|
|
def n_bytes(shape, tid):
|
|
n = 1
|
|
for d in shape or ():
|
|
n *= d
|
|
return n * tid.get_size()
|
|
|
|
|
|
def read_attrs(obj):
|
|
out = {}
|
|
names = sorted(obj.attrs.keys(), key=lambda s: s.encode("utf-8", "surrogateescape"))
|
|
for name in names:
|
|
rec = {}
|
|
try:
|
|
aid = obj.attrs.get_id(name)
|
|
rec["dtype"] = str(aid.dtype)
|
|
rec["shape"] = shape_of(aid.shape)
|
|
note_conversion(aid.get_type(), aid.dtype, rec)
|
|
if aid.shape is None:
|
|
hash_values(np.empty((0,), dtype=aid.dtype), aid.dtype, rec)
|
|
else:
|
|
val = obj.attrs[name]
|
|
hash_values(val, aid.dtype, rec)
|
|
except Exception as e: # noqa: BLE001
|
|
rec = {"error": err(e)}
|
|
out[name] = rec
|
|
return out
|
|
|
|
|
|
def main(path):
|
|
top = {"file": path}
|
|
try:
|
|
f = h5py.File(path, "r")
|
|
except Exception as e: # noqa: BLE001
|
|
top["open_error"] = err(e)
|
|
print(json.dumps(top))
|
|
return
|
|
objects = []
|
|
seen = set()
|
|
stack = [("/", None)]
|
|
while stack:
|
|
p, obj = stack.pop()
|
|
if len(objects) >= MAX_OBJECTS:
|
|
top["truncated"] = True
|
|
break
|
|
rec = {"path": p}
|
|
try:
|
|
if obj is None:
|
|
obj = f[p]
|
|
key = hash(obj.id) # h5py ObjectID hash = (fileno, object address/token)
|
|
except Exception as e: # noqa: BLE001
|
|
rec["kind"] = "unknown"
|
|
rec["error"] = err(e)
|
|
objects.append(rec)
|
|
continue
|
|
if key in seen:
|
|
continue
|
|
seen.add(key)
|
|
if isinstance(obj, h5py.Dataset):
|
|
kind = "dataset"
|
|
elif isinstance(obj, h5py.Group):
|
|
kind = "group"
|
|
elif isinstance(obj, h5py.Datatype):
|
|
kind = "datatype"
|
|
else:
|
|
kind = "unknown"
|
|
rec["kind"] = kind
|
|
if kind == "dataset":
|
|
try:
|
|
dt = obj.dtype
|
|
rec["dtype"] = str(dt)
|
|
rec["shape"] = shape_of(obj.shape)
|
|
note_conversion(obj.id.get_type(), dt, rec)
|
|
if obj.shape is None:
|
|
hash_values(np.empty((0,), dtype=dt), dt, rec)
|
|
elif n_bytes(obj.shape, obj.id.get_type()) > MAX_BYTES:
|
|
rec["skipped"] = "too large"
|
|
else:
|
|
arr = np.empty(obj.shape, dtype=dt)
|
|
if arr.size:
|
|
try:
|
|
obj.read_direct(arr)
|
|
except Exception: # noqa: BLE001
|
|
arr = obj[()]
|
|
hash_values(arr, dt, rec)
|
|
except Exception as e: # noqa: BLE001
|
|
rec["error"] = err(e)
|
|
if kind != "datatype":
|
|
try:
|
|
rec["attrs"] = read_attrs(obj)
|
|
except Exception as e: # noqa: BLE001
|
|
rec["attrs_error"] = err(e)
|
|
if kind == "group":
|
|
try:
|
|
names = sorted(obj.keys(), key=lambda s: s.encode("utf-8", "surrogateescape"))
|
|
base = "" if p == "/" else p
|
|
kids = []
|
|
for n in names:
|
|
try:
|
|
link = obj.get(n, getlink=True)
|
|
except Exception: # noqa: BLE001
|
|
link = None
|
|
if link is not None and not isinstance(link, h5py.HardLink):
|
|
continue
|
|
kids.append(f"{base}/{n}")
|
|
for k in reversed(kids):
|
|
stack.append((k, None))
|
|
except Exception as e: # noqa: BLE001
|
|
rec["list_error"] = err(e)
|
|
objects.append(rec)
|
|
top["objects"] = objects
|
|
print(json.dumps(top), flush=True)
|
|
# Exit without tearing down the h5py objects: freeing them for some files
|
|
# that hold references (hdf5's h5repack_attr_refs.h5, cve-2024-32623.h5)
|
|
# makes libhdf5 2.0 abort with "free(): chunks in smallbin corrupted"
|
|
# about half the time. That happens after the reading is done, so it says
|
|
# nothing about what h5py read, but it flipped those files between ok and
|
|
# h5py-cannot-read from one run to the next.
|
|
os._exit(0)
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main(sys.argv[1])
|