#!/usr/bin/env python3 """ref_bugs.py : re-check the objects h5py reads only through a libhdf5 bug. For each object of READ_BUGS (below), h5py reads it in several fresh processes whose heaps differ: h5py imported before numpy (three runs, plus two with glibc's MALLOC_PERTURB_, which fills newly allocated and freed heap blocks with a byte pattern) and numpy imported first. Values the file determines come out the same every time. An object whose values differ between those runs is read from memory the file does not determine — an over-read or an uninitialised buffer in libhdf5 — so the values h5py reports for it are not the file's, and clawhdf5 refusing the object is not a clawhdf5 error. compare.py classifies a file as `ref-bug` only on objects confirmed that way in the same run (`$OUT/ref_bugs.json`); an object whose reading turns out stable stays an our-error. Run by conformance/run.sh; on its own it is the reproducer (JSON on stdout). """ import concurrent.futures import json import os import subprocess import sys # (file, object) -> what goes wrong. Checked 2026-09-27 against HDF5 2.0.0 # (h5py 3.16), h5dump 1.14.6 and the HDFGroup/hdf5 sources (tag hdf5_1_14_6 # and develop); see docs/known-issues.md, "Conformance: the last non-ok files". READ_BUGS = { ("cve_hdf5/cvefiles/cve-2025-2308.h5", "/Scale_offset_long_long_data_le"): "the first chunk records minbits 11: its 12 values need 17 bytes of codes, and the " "26-byte chunk holds 5 after its 21-byte header; libhdf5's scale-offset decoder reads " "past its buffer, and develop refuses the chunk (\"Buffer too short\")", ("cve_hdf5/cvefiles/cve-2025-44904.h5", "/Scale_offset_float_data_le"): "unfiltered chunks stored as 38 and 37 bytes for 48-byte chunks: 1.14/2.0 read the " "stored bytes into a buffer of that size and use it as the whole chunk " "(H5D__chunk_lock), so the rest is heap memory; develop refuses them (\"incorrect chunk " "size returned from index for unfiltered chunk\")", ("hdf5/test/testfiles/bad_nbit_parms_walk.h5", "/Nbit_int_data_le"): "the N-Bit parameter list holds 7 values (cd_values[0] = 7) where an integer needs 8: " "the decoder takes the bit offset from cd_values[7], past the list; libhdf5's own test " "(`test_filter_bad_params`, test/dsets.c on develop) requires the read to fail", } # (which module is imported first, MALLOC_PERTURB_) RUNS = [("h5py", None), ("h5py", None), ("h5py", None), ("h5py", "170"), ("h5py", "255"), ("numpy", None)] READ = r""" import hashlib, sys if sys.argv[3] == "h5py": import h5py, numpy as np else: import numpy as np, h5py try: import hdf5plugin # noqa: F401 except Exception: pass try: with h5py.File(sys.argv[1], "r") as f: a = np.ascontiguousarray(f[sys.argv[2]][()]) print("values " + hashlib.sha256(a.tobytes()).hexdigest()[:16]) except Exception as e: print("error " + (str(e).splitlines() or [type(e).__name__])[0][:120]) """ def read_once(path, obj, first, perturb): env = dict(os.environ) env.pop("MALLOC_PERTURB_", None) if perturb: env["MALLOC_PERTURB_"] = perturb try: p = subprocess.run([sys.executable, "-c", READ, path, obj, first], env=env, capture_output=True, text=True, timeout=60) out = p.stdout.strip().splitlines() return out[-1] if out else f"exit {p.returncode}" except subprocess.TimeoutExpired: return "timeout" def check(corpus, key): f, obj = key path = os.path.join(corpus, f) rec = {"file": f, "object": obj, "why": READ_BUGS[key]} if not os.path.exists(path): return rec | {"missing": True, "confirmed": False} runs = [{"first": a, "malloc_perturb": p, "outcome": read_once(path, obj, a, p)} for a, p in RUNS] distinct = sorted({r["outcome"] for r in runs}) return rec | { "runs": runs, "distinct": len(distinct), "confirmed": len(distinct) > 1 and any(o.startswith("values ") for o in distinct), } def main(): corpus = sys.argv[1] keys = list(READ_BUGS) with concurrent.futures.ThreadPoolExecutor(max_workers=len(keys)) as ex: out = list(ex.map(lambda k: check(corpus, k), keys)) print(json.dumps({"read_bugs": out}, indent=1)) if __name__ == "__main__": main()