The last 3 our-errors and 2 mismatches were documented as not ours but
still counted against us, on a heuristic (any big-endian VL mismatch) and
a fixed list.
- ref.py checks that the installed h5py returns big-endian VL elements
with the file's bytes under a little-endian dtype (writing and reading
a vlen('>f4') in memory) and, if so, relabels them with the file's byte
order before hashing, marking the object `ref_fix`. The values are now
compared: attr_datatypes.hdf5 /@vlen_uint64 and tcomplex_be.h5
/VariableLengthDatasetFloatComplex are identical to ours (h5dump 1.14.6
prints the same (1, 2), (3, 4, 5), (42)).
- ref_bugs.py re-reads each object h5py reads only through a libhdf5 bug
in six processes with different heaps (import order, MALLOC_PERTURB_).
Values the file determines are the same every time; these three change
(6, 6 and 3 distinct results), so they are over-read memory, not data
clawhdf5 could match. compare.py classifies a file `ref-bug` only when
every difference is such an object confirmed in the same run.
- report.py: the ref-bug class, the evidence table, the corrected
objects; test_ref.py covers both (run in the nightly job).
Co-Authored-By: Claude Opus 5.5 (1M context) <[email protected]>
106 lines
4.3 KiB
Python
106 lines
4.3 KiB
Python
#!/usr/bin/env python3
|
|
"""ref_bugs.py <corpus_dir>: re-check the objects h5py reads only through a
|
|
libhdf5 bug.
|
|
|
|
For each object of READ_BUGS (below), h5py reads it in several fresh
|
|
processes whose heaps differ: h5py imported before numpy (three runs, plus
|
|
two with glibc's MALLOC_PERTURB_, which fills newly allocated and freed heap
|
|
blocks with a byte pattern) and numpy imported first. Values the file
|
|
determines come out the same every time. An object whose values differ
|
|
between those runs is read from memory the file does not determine — an
|
|
over-read or an uninitialised buffer in libhdf5 — so the values h5py reports
|
|
for it are not the file's, and clawhdf5 refusing the object is not a
|
|
clawhdf5 error. compare.py classifies a file as `ref-bug` only on objects
|
|
confirmed that way in the same run (`$OUT/ref_bugs.json`); an object whose
|
|
reading turns out stable stays an our-error.
|
|
|
|
Run by conformance/run.sh; on its own it is the reproducer (JSON on stdout).
|
|
"""
|
|
import concurrent.futures
|
|
import json
|
|
import os
|
|
import subprocess
|
|
import sys
|
|
|
|
# (file, object) -> what goes wrong. Checked 2026-09-27 against HDF5 2.0.0
|
|
# (h5py 3.16), h5dump 1.14.6 and the HDFGroup/hdf5 sources (tag hdf5_1_14_6
|
|
# and develop); see docs/known-issues.md, "Conformance: the last non-ok files".
|
|
READ_BUGS = {
|
|
("cve_hdf5/cvefiles/cve-2025-2308.h5", "/Scale_offset_long_long_data_le"):
|
|
"the first chunk records minbits 11: its 12 values need 17 bytes of codes, and the "
|
|
"26-byte chunk holds 5 after its 21-byte header; libhdf5's scale-offset decoder reads "
|
|
"past its buffer, and develop refuses the chunk (\"Buffer too short\")",
|
|
("cve_hdf5/cvefiles/cve-2025-44904.h5", "/Scale_offset_float_data_le"):
|
|
"unfiltered chunks stored as 38 and 37 bytes for 48-byte chunks: 1.14/2.0 read the "
|
|
"stored bytes into a buffer of that size and use it as the whole chunk "
|
|
"(H5D__chunk_lock), so the rest is heap memory; develop refuses them (\"incorrect chunk "
|
|
"size returned from index for unfiltered chunk\")",
|
|
("hdf5/test/testfiles/bad_nbit_parms_walk.h5", "/Nbit_int_data_le"):
|
|
"the N-Bit parameter list holds 7 values (cd_values[0] = 7) where an integer needs 8: "
|
|
"the decoder takes the bit offset from cd_values[7], past the list; libhdf5's own test "
|
|
"(`test_filter_bad_params`, test/dsets.c on develop) requires the read to fail",
|
|
}
|
|
|
|
# (which module is imported first, MALLOC_PERTURB_)
|
|
RUNS = [("h5py", None), ("h5py", None), ("h5py", None), ("h5py", "170"), ("h5py", "255"),
|
|
("numpy", None)]
|
|
|
|
READ = r"""
|
|
import hashlib, sys
|
|
if sys.argv[3] == "h5py":
|
|
import h5py, numpy as np
|
|
else:
|
|
import numpy as np, h5py
|
|
try:
|
|
import hdf5plugin # noqa: F401
|
|
except Exception:
|
|
pass
|
|
try:
|
|
with h5py.File(sys.argv[1], "r") as f:
|
|
a = np.ascontiguousarray(f[sys.argv[2]][()])
|
|
print("values " + hashlib.sha256(a.tobytes()).hexdigest()[:16])
|
|
except Exception as e:
|
|
print("error " + (str(e).splitlines() or [type(e).__name__])[0][:120])
|
|
"""
|
|
|
|
|
|
def read_once(path, obj, first, perturb):
|
|
env = dict(os.environ)
|
|
env.pop("MALLOC_PERTURB_", None)
|
|
if perturb:
|
|
env["MALLOC_PERTURB_"] = perturb
|
|
try:
|
|
p = subprocess.run([sys.executable, "-c", READ, path, obj, first], env=env,
|
|
capture_output=True, text=True, timeout=60)
|
|
out = p.stdout.strip().splitlines()
|
|
return out[-1] if out else f"exit {p.returncode}"
|
|
except subprocess.TimeoutExpired:
|
|
return "timeout"
|
|
|
|
|
|
def check(corpus, key):
|
|
f, obj = key
|
|
path = os.path.join(corpus, f)
|
|
rec = {"file": f, "object": obj, "why": READ_BUGS[key]}
|
|
if not os.path.exists(path):
|
|
return rec | {"missing": True, "confirmed": False}
|
|
runs = [{"first": a, "malloc_perturb": p, "outcome": read_once(path, obj, a, p)} for a, p in RUNS]
|
|
distinct = sorted({r["outcome"] for r in runs})
|
|
return rec | {
|
|
"runs": runs,
|
|
"distinct": len(distinct),
|
|
"confirmed": len(distinct) > 1 and any(o.startswith("values ") for o in distinct),
|
|
}
|
|
|
|
|
|
def main():
|
|
corpus = sys.argv[1]
|
|
keys = list(READ_BUGS)
|
|
with concurrent.futures.ThreadPoolExecutor(max_workers=len(keys)) as ex:
|
|
out = list(ex.map(lambda k: check(corpus, k), keys))
|
|
print(json.dumps({"read_bugs": out}, indent=1))
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main()
|